@iowarp/clio-coder 0.3.8 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (552) hide show
  1. package/CHANGELOG.md +102 -0
  2. package/NOTICE +33 -0
  3. package/README.md +12 -3
  4. package/dist/{acp-U67UHUK2.js → acp-G5WJBNCT.js} +13 -12
  5. package/dist/{agents-YU6SGALZ.js → agents-FMV2Q5G4.js} +41 -36
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5ZPJOIVG.js → auth-3IDSJEIK.js} +18 -18
  8. package/dist/{builtins-C6JMZVV6.js → builtins-XCZWXSC7.js} +5 -5
  9. package/dist/{chunk-IFBNV6H6.js → chunk-2ANTL7MR.js} +3 -3
  10. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  11. package/dist/{chunk-XWSF374K.js → chunk-2OQE55CK.js} +3 -3
  12. package/dist/{chunk-5H3GB5BO.js → chunk-32KWKNSF.js} +8 -384
  13. package/dist/{chunk-KTYTFRMB.js → chunk-36EJLSQQ.js} +34 -36
  14. package/dist/{chunk-FHJEP5SW.js → chunk-3BT2XMV4.js} +19 -13
  15. package/dist/chunk-3DPEIQKN.js +113 -0
  16. package/dist/{chunk-GU2UIAFZ.js → chunk-3URVFKWK.js} +7 -7
  17. package/dist/{chunk-WEH5XRJQ.js → chunk-3XML7CDN.js} +3 -3
  18. package/dist/chunk-42FMPA75.js +101 -0
  19. package/dist/{chunk-A2NJGIB3.js → chunk-5LXZXPKX.js} +2 -2
  20. package/dist/{chunk-VCBR6CU7.js → chunk-5PSMVOLM.js} +2 -2
  21. package/dist/{chunk-NMPKI6XL.js → chunk-6DB53AJS.js} +203 -25
  22. package/dist/{chunk-3BINW3FP.js → chunk-76ONBSIA.js} +2 -2
  23. package/dist/{chunk-26LEYJZH.js → chunk-7MCTRUCE.js} +2 -2
  24. package/dist/{chunk-TYPGUK6W.js → chunk-AB44T6BB.js} +111 -5
  25. package/dist/{chunk-K4XHGFR5.js → chunk-B7OBL7PK.js} +317 -721
  26. package/dist/{chunk-B5CSFE7B.js → chunk-BBVJUZHB.js} +2 -2
  27. package/dist/{chunk-EQ63NRB7.js → chunk-BBVYXMFO.js} +2 -2
  28. package/dist/{chunk-ZVJ5BLO2.js → chunk-BKFJHQCA.js} +154 -16
  29. package/dist/chunk-BKFM6EJV.js +462 -0
  30. package/dist/chunk-BMS5RKQY.js +27 -0
  31. package/dist/{chunk-ZNLWCMVZ.js → chunk-BPKCPIL7.js} +2 -2
  32. package/dist/{chunk-TLQJPP24.js → chunk-BUMFYQFY.js} +1394 -1349
  33. package/dist/{chunk-BNAZZHFG.js → chunk-BYP5D4HI.js} +1 -1
  34. package/dist/chunk-C2LTL2W6.js +2447 -0
  35. package/dist/{chunk-ME6CCNFO.js → chunk-CODPRO7Q.js} +8 -8
  36. package/dist/{chunk-E77JEWSD.js → chunk-CTJ4RNAA.js} +7 -37
  37. package/dist/{chunk-ODFEOB4F.js → chunk-CY6FY24N.js} +26 -8
  38. package/dist/{chunk-7RFXX52T.js → chunk-DQITNCXG.js} +642 -172
  39. package/dist/{chunk-TTHACPOM.js → chunk-DYJP44XW.js} +578 -117
  40. package/dist/{chunk-2HFZQUHL.js → chunk-F4CKPOEQ.js} +18 -8
  41. package/dist/{chunk-DGSYXYMX.js → chunk-FEFIFZTL.js} +3 -3
  42. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  43. package/dist/{chunk-MV3K5QF2.js → chunk-GCSMB2KY.js} +2 -2
  44. package/dist/chunk-GKF55TAZ.js +391 -0
  45. package/dist/{chunk-VWZOAB7K.js → chunk-GR5G2PVF.js} +9 -8
  46. package/dist/{chunk-4DGYLA73.js → chunk-GXNLGKAB.js} +80 -9
  47. package/dist/chunk-HHV2GANA.js +88 -0
  48. package/dist/{chunk-IIZWH4XA.js → chunk-HI63TFOG.js} +5 -4
  49. package/dist/{chunk-WLFILSD5.js → chunk-HJJTYUHX.js} +113 -83
  50. package/dist/{chunk-TT36MB5S.js → chunk-HLW2MRKE.js} +3 -1
  51. package/dist/{chunk-PMDBGQSJ.js → chunk-HWHKMHUA.js} +7 -7
  52. package/dist/chunk-HZHHCK24.js +1631 -0
  53. package/dist/{chunk-WSB3FPX7.js → chunk-I5VEOC6I.js} +39 -143
  54. package/dist/chunk-IBEBSCYA.js +564 -0
  55. package/dist/chunk-IQ7KR472.js +362 -0
  56. package/dist/{chunk-A3WNZD3P.js → chunk-J4W7KFM7.js} +949 -972
  57. package/dist/{chunk-TB5666IT.js → chunk-JDG2WCRO.js} +5 -5
  58. package/dist/{chunk-N22QMJKY.js → chunk-K5C3NCBD.js} +4 -4
  59. package/dist/{chunk-CGKSTWHD.js → chunk-K6BSR66V.js} +2 -1
  60. package/dist/{chunk-5C3AQNDW.js → chunk-KFZI4NIL.js} +216 -38
  61. package/dist/chunk-KMVISBZR.js +132 -0
  62. package/dist/{chunk-WXY7KU3G.js → chunk-LDQ2ZF2M.js} +2 -2
  63. package/dist/{chunk-XN3L4EYL.js → chunk-LQ3DZAMX.js} +3 -3
  64. package/dist/{chunk-U6MBIEMB.js → chunk-LY4S7GJC.js} +173 -144
  65. package/dist/{chunk-4SPRNWDE.js → chunk-MLKNTWH2.js} +19 -19
  66. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  67. package/dist/chunk-NEKRRTYW.js +56 -0
  68. package/dist/{chunk-SPULKLCF.js → chunk-NHCZP4K7.js} +3 -3
  69. package/dist/chunk-NHLBIGRH.js +1506 -0
  70. package/dist/chunk-NQQH3YT7.js +302 -0
  71. package/dist/chunk-NYS75XW5.js +15 -0
  72. package/dist/{chunk-GN57SG4G.js → chunk-O4XIVISU.js} +10 -8
  73. package/dist/{chunk-EMYUUSFG.js → chunk-O6TL7WWY.js} +6 -6
  74. package/dist/chunk-OQBA45DZ.js +97 -0
  75. package/dist/{chunk-LU7P4LHA.js → chunk-P3FOHJT4.js} +2 -2
  76. package/dist/chunk-PMZCIOCJ.js +25 -0
  77. package/dist/{chunk-I4HZDVNP.js → chunk-PQEFIJ36.js} +2 -2
  78. package/dist/{chunk-J3YUBZWY.js → chunk-QBJA7R7N.js} +62 -6
  79. package/dist/chunk-QDC3K2U3.js +262 -0
  80. package/dist/{chunk-2HEJ2F35.js → chunk-QLFS5GO2.js} +22 -10
  81. package/dist/{chunk-7RGZWPB6.js → chunk-QLL7ILRG.js} +95 -32
  82. package/dist/{chunk-YS5VLNH5.js → chunk-QREDIESB.js} +6 -6
  83. package/dist/chunk-QSNYB6ZV.js +195 -0
  84. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  85. package/dist/chunk-RKKLTLYB.js +45 -0
  86. package/dist/{chunk-P43ETTHK.js → chunk-SJ5ZKQ4S.js} +2 -2
  87. package/dist/{chunk-GPIEI3LY.js → chunk-SP2RXXYO.js} +6 -54
  88. package/dist/chunk-SUCTJL45.js +45 -0
  89. package/dist/{chunk-DYIM5TJT.js → chunk-SUW5DORT.js} +263 -7
  90. package/dist/chunk-T56WDKA5.js +183 -0
  91. package/dist/chunk-TVHHYFHE.js +255 -0
  92. package/dist/{chunk-FJ3H4MN5.js → chunk-TZ3SGWZZ.js} +3 -3
  93. package/dist/{chunk-MXKJU4JB.js → chunk-U77AMWDL.js} +91 -10
  94. package/dist/{chunk-5DHKRSMQ.js → chunk-ULC6OTWO.js} +11 -7
  95. package/dist/{chunk-RWSI4YD7.js → chunk-UM7N4G5A.js} +33 -12
  96. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  97. package/dist/{chunk-IGWKHNIQ.js → chunk-UXMFQ54G.js} +44 -37
  98. package/dist/{chunk-HFSBBKSQ.js → chunk-V5DHCITQ.js} +171 -3
  99. package/dist/{chunk-5WIGXA4T.js → chunk-VAZSBTKF.js} +111 -4
  100. package/dist/{chunk-VHN4MY6O.js → chunk-VEO4AP2K.js} +2 -2
  101. package/dist/{chunk-IJ7RPIYJ.js → chunk-VFA6GDY5.js} +65 -4
  102. package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
  103. package/dist/chunk-VO67MWHC.js +75 -0
  104. package/dist/{chunk-XK56QHLX.js → chunk-VPKWYKEY.js} +19 -5
  105. package/dist/{chunk-TANS5ZJS.js → chunk-VYMXRQI6.js} +36 -22
  106. package/dist/{chunk-WWCZ5F23.js → chunk-W5VSYASO.js} +77 -16
  107. package/dist/{chunk-WNIJTQQK.js → chunk-WZR7K7ZX.js} +72 -116
  108. package/dist/{chunk-5Q2VVUKB.js → chunk-X3YGUTOB.js} +4 -4
  109. package/dist/chunk-X75E3D2N.js +686 -0
  110. package/dist/{chunk-VAWNZU7Z.js → chunk-YDFRH54B.js} +4 -4
  111. package/dist/chunk-YJX4SHTD.js +40 -0
  112. package/dist/{chunk-ZI647VB5.js → chunk-YPI3QQCF.js} +2 -2
  113. package/dist/chunk-Z2RR6MAK.js +127 -0
  114. package/dist/{chunk-FBVTI2TJ.js → chunk-Z4TXYIEG.js} +12 -131
  115. package/dist/cli/index.js +47 -36
  116. package/dist/{clio-QVTYJ57A.js → clio-2JXHBBY5.js} +7 -7
  117. package/dist/{code-nav-FGGFIE7L.js → code-nav-3YYRMYNF.js} +8 -8
  118. package/dist/{compile-cache-CVJMMODC.js → compile-cache-7FPE6PS3.js} +3 -3
  119. package/dist/{components-ZFA3SAER.js → components-RYZV4JGP.js} +5 -5
  120. package/dist/{config-LW5IJFQN.js → config-QZPCMYSO.js} +99 -63
  121. package/dist/{configure-7XIZCOU4.js → configure-TEGEBYCA.js} +23 -22
  122. package/dist/{context-Y6Y7QPR6.js → context-AV7OEZ4D.js} +12 -12
  123. package/dist/{context-L3WL3X7K.js → context-E6H5RNMC.js} +56 -47
  124. package/dist/{context-N52ZA626.js → context-GSXUE4CT.js} +29 -27
  125. package/dist/{context-clear-MBQRLSDQ.js → context-clear-SHIBYK6T.js} +55 -46
  126. package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
  127. package/dist/{context-working-set-GS6DSO7F.js → context-working-set-5ZGKPGZQ.js} +13 -13
  128. package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-EFMJT4LD.js} +93 -64
  129. package/dist/{docs-7LQ23DLM.js → docs-23KQS3XK.js} +5 -5
  130. package/dist/doctor-QOA5FNY5.js +313 -0
  131. package/dist/{eval-BEC2WHDA.js → eval-TFBYQH4H.js} +2032 -156
  132. package/dist/eval-inventory-SXH7PDKX.js +316 -0
  133. package/dist/{evidence-REJUMSKM.js → evidence-ERGESKGN.js} +203 -48
  134. package/dist/{evolve-PY5ZBA5K.js → evolve-VDXTSYCJ.js} +52 -43
  135. package/dist/{extensions-HVKU65YU.js → extensions-7BGBHN57.js} +13 -7
  136. package/dist/{fleet-7WZEWRFA.js → fleet-2RRVDF2V.js} +228 -108
  137. package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-VJ726XIA.js} +11 -11
  138. package/dist/fleet-decisions-EPAPM3XJ.js +157 -0
  139. package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-JF5QOATM.js} +18 -15
  140. package/dist/fleet-inspect-VLY4S7QM.js +442 -0
  141. package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-AIZUEJOY.js} +5 -5
  142. package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-AJRPDMDV.js} +22 -19
  143. package/dist/fleet-verify-JFEL2L3H.js +175 -0
  144. package/dist/fleet-view-ZCON35AG.js +102 -0
  145. package/dist/{init-OG3TPGQG.js → init-DN2WWLFE.js} +72 -62
  146. package/dist/install-XGLBQY5E.js +13 -0
  147. package/dist/interop-OZBKXAYL.js +114 -0
  148. package/dist/{library-CNTMPLRF.js → library-YWZG7IMW.js} +21 -18
  149. package/dist/{memory-6IS7F275.js → memory-I4C4HMLW.js} +54 -45
  150. package/dist/{models-ENRJDA5W.js → models-CEYXJBO6.js} +33 -30
  151. package/dist/{monitor-XLDVO7TN.js → monitor-NZ6GCI3P.js} +59 -52
  152. package/dist/{orchestrator-6KSPYRHA.js → orchestrator-GCGQ4N5I.js} +7827 -7328
  153. package/dist/panes-HMABYVO4.js +58 -0
  154. package/dist/panes-KY6W3V2E.js +103 -0
  155. package/dist/{paths-DBXMZMDU.js → paths-II4K7DNR.js} +5 -5
  156. package/dist/{reset-RZ4ER727.js → reset-DQ6FGCSH.js} +13 -11
  157. package/dist/resources-BB3MVJMD.js +111 -0
  158. package/dist/{run-Y2CNK5RU.js → run-H2GQDUER.js} +119 -87
  159. package/dist/{share-A55GYP6Z.js → share-GTJN6A5O.js} +20 -17
  160. package/dist/{skills-ALC5J6AT.js → skills-L55TEW6R.js} +33 -24
  161. package/dist/{skills-eval-JPBEBYQU.js → skills-eval-XVXPH2JI.js} +67 -56
  162. package/dist/skills-inventory-S4MXPJFV.js +126 -0
  163. package/dist/slash-commands-ZSGASKJC.js +77 -0
  164. package/dist/{steer-GGWFUJUD.js → steer-RZGSCY4R.js} +4 -4
  165. package/dist/{support-MIETYA5E.js → support-PKEUNNQL.js} +6 -6
  166. package/dist/{targets-VGNXIR3S.js → targets-NCPZ644J.js} +68 -40
  167. package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-44SV3YCN.js} +6 -4
  168. package/dist/tools-DAF3DI3C.js +27 -0
  169. package/dist/{trace-PNCASAXC.js → trace-FYVW2MQA.js} +207 -10
  170. package/dist/tui-primitives-2AKXQNZK.js +13 -0
  171. package/dist/{uninstall-ZJF5H5ZN.js → uninstall-DW2PNOIC.js} +5 -5
  172. package/dist/{upgrade-FUSUAGHR.js → upgrade-3XPP6OQL.js} +27 -24
  173. package/dist/{usage-N4MKVHKD.js → usage-3NLHGTU2.js} +114 -62
  174. package/dist/{verifiers-YAWOJ3H2.js → verifiers-SSQONKRT.js} +172 -13
  175. package/dist/{verify-LTDHYBGY.js → verify-3U6J7FZI.js} +10 -10
  176. package/dist/{web-fetch-2YHJ3KTG.js → web-fetch-S7RR6GZ7.js} +3 -3
  177. package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-CEHYGPGQ.js} +75 -65
  178. package/dist/with-panes-MKB46MPQ.js +782 -0
  179. package/dist/worker/entry.js +106 -75
  180. package/docs/README.md +3 -2
  181. package/docs/acp.md +24 -3
  182. package/docs/alcf-provider.md +1 -1
  183. package/docs/architecture.md +2 -2
  184. package/docs/artifact-versions.md +9 -1
  185. package/docs/built-in-agents.md +1 -1
  186. package/docs/capacity-and-scheduling.md +62 -3
  187. package/docs/commands-and-modes.md +31 -2
  188. package/docs/configuration-and-targets.md +69 -12
  189. package/docs/context-engine.md +63 -4
  190. package/docs/development-pipeline.md +19 -0
  191. package/docs/dispatch-typed-intent.md +385 -0
  192. package/docs/documentation-coverage.md +5 -5
  193. package/docs/documentation-guide.md +1 -1
  194. package/docs/environment-variables.md +3 -0
  195. package/docs/eval-runner.md +262 -11
  196. package/docs/evals-internal.md +72 -2
  197. package/docs/evidence-and-memory.md +12 -11
  198. package/docs/evolution.md +1 -1
  199. package/docs/exit-codes-and-output.md +1 -1
  200. package/docs/extensions-and-sharing.md +27 -1
  201. package/docs/fleet-dispatch.md +25 -4
  202. package/docs/installation-and-lifecycle.md +15 -2
  203. package/docs/middleware-and-components.md +1 -1
  204. package/docs/model-catalog.md +10 -1
  205. package/docs/observability.md +55 -4
  206. package/docs/proactive-memory.md +127 -14
  207. package/docs/prompt-envelope-and-tools.md +19 -1
  208. package/docs/provider-adapter-cookbook.md +1 -1
  209. package/docs/release-cut-checklist.md +19 -3
  210. package/docs/safety-model.md +2 -2
  211. package/docs/scientific-validation.md +3 -3
  212. package/docs/session-lifecycle.md +1 -1
  213. package/docs/skills-marketplace.md +1 -1
  214. package/docs/tool-usage.md +18 -8
  215. package/docs/trace-store.md +1 -1
  216. package/docs/troubleshooting.md +88 -1
  217. package/docs/tui-design.md +1 -1
  218. package/docs/worker-dispatch-mechanics.md +1 -1
  219. package/package.json +5 -2
  220. package/src/cli/acp.ts +6 -2
  221. package/src/cli/agents.ts +1 -1
  222. package/src/cli/argv.ts +25 -0
  223. package/src/cli/config-inspect.ts +33 -6
  224. package/src/cli/config.ts +1 -1
  225. package/src/cli/configure.ts +23 -21
  226. package/src/cli/doctor-panes.ts +124 -0
  227. package/src/cli/doctor-state-size.ts +82 -0
  228. package/src/cli/doctor-toolchain.ts +57 -0
  229. package/src/cli/doctor.ts +22 -1
  230. package/src/cli/eval-inventory.ts +436 -0
  231. package/src/cli/eval.ts +93 -16
  232. package/src/cli/evidence-detail.ts +88 -0
  233. package/src/cli/evidence-inventory.ts +183 -0
  234. package/src/cli/evidence.ts +30 -5
  235. package/src/cli/extensions.ts +5 -1
  236. package/src/cli/fleet-decisions.ts +69 -0
  237. package/src/cli/fleet-inspect.ts +334 -0
  238. package/src/cli/fleet-verify.ts +133 -0
  239. package/src/cli/fleet-view.ts +810 -0
  240. package/src/cli/fleet.ts +179 -39
  241. package/src/cli/index.ts +14 -2
  242. package/src/cli/interop-inspect.ts +128 -0
  243. package/src/cli/interop.ts +34 -0
  244. package/src/cli/panes.ts +35 -0
  245. package/src/cli/reset.ts +5 -2
  246. package/src/cli/run.ts +58 -0
  247. package/src/cli/skills-inventory.ts +185 -0
  248. package/src/cli/skills.ts +16 -13
  249. package/src/cli/targets.ts +44 -13
  250. package/src/cli/tools.ts +321 -0
  251. package/src/cli/trace-inspect.ts +252 -0
  252. package/src/cli/trace.ts +85 -5
  253. package/src/cli/usage.ts +63 -14
  254. package/src/cli/verifiers-inspect.ts +347 -0
  255. package/src/cli/verifiers.ts +9 -0
  256. package/src/core/bus-events.ts +33 -1
  257. package/src/core/cache-telemetry.ts +42 -0
  258. package/src/core/config.ts +67 -0
  259. package/src/core/defaults.ts +124 -8
  260. package/src/core/endpoint-key.ts +27 -0
  261. package/src/core/residency-target-key.ts +25 -0
  262. package/src/core/response-schema.ts +80 -6
  263. package/src/core/theme-token-hex.ts +43 -0
  264. package/src/core/tool-names.ts +2 -1
  265. package/src/core/xdg.ts +1 -1
  266. package/src/domains/agents/fleets/build-review.md +0 -3
  267. package/src/domains/agents/fleets/build-test.md +0 -3
  268. package/src/domains/agents/result-contract-filesystem.ts +32 -0
  269. package/src/domains/agents/result-contract.ts +164 -35
  270. package/src/domains/config/classify.ts +6 -0
  271. package/src/domains/context/codewiki/coordinator.ts +12 -4
  272. package/src/domains/dispatch/admission-error.ts +9 -0
  273. package/src/domains/dispatch/admission.ts +52 -14
  274. package/src/domains/dispatch/capacity-lease.ts +118 -9
  275. package/src/domains/dispatch/contract.ts +11 -0
  276. package/src/domains/dispatch/council-topology.ts +398 -0
  277. package/src/domains/dispatch/execution-plan.ts +44 -4
  278. package/src/domains/dispatch/extension.ts +378 -82
  279. package/src/domains/dispatch/fleet-node-prompt.ts +62 -0
  280. package/src/domains/dispatch/fleet-plan.ts +7 -2
  281. package/src/domains/dispatch/fleet-run.ts +87 -4
  282. package/src/domains/dispatch/gate-decisions.ts +11 -1
  283. package/src/domains/dispatch/gate-role-prompts.ts +9 -0
  284. package/src/domains/dispatch/gate-topology.ts +289 -0
  285. package/src/domains/dispatch/heartbeat.ts +32 -8
  286. package/src/domains/dispatch/index.ts +22 -0
  287. package/src/domains/dispatch/intent-compatibility.ts +330 -0
  288. package/src/domains/dispatch/intent.ts +85 -1
  289. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  290. package/src/domains/dispatch/reservation-store.ts +139 -11
  291. package/src/domains/dispatch/run-event-journal-bridge.ts +149 -0
  292. package/src/domains/dispatch/run-event-journal.ts +598 -0
  293. package/src/domains/dispatch/state.ts +49 -1
  294. package/src/domains/dispatch/types.ts +13 -0
  295. package/src/domains/dispatch/validation.ts +33 -8
  296. package/src/domains/dispatch/worker-spawn.ts +25 -11
  297. package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
  298. package/src/domains/dispatch/write-boundary.ts +62 -1
  299. package/src/domains/eval/artifacts/store.ts +62 -0
  300. package/src/domains/eval/compare/behavioral.ts +224 -0
  301. package/src/domains/eval/compare/compare.ts +342 -2
  302. package/src/domains/eval/compare/envelope.ts +128 -0
  303. package/src/domains/eval/compare/gates.ts +24 -6
  304. package/src/domains/eval/compare/thresholds.ts +30 -3
  305. package/src/domains/eval/execution-provenance.ts +240 -0
  306. package/src/domains/eval/inventory.ts +113 -0
  307. package/src/domains/eval/metrics/aggregate.ts +136 -0
  308. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  309. package/src/domains/eval/metrics/tracked.ts +413 -0
  310. package/src/domains/eval/provenance.ts +117 -0
  311. package/src/domains/eval/reports/comparison.ts +128 -0
  312. package/src/domains/eval/reports/junit.ts +17 -3
  313. package/src/domains/eval/reports/markdown.ts +3 -3
  314. package/src/domains/eval/reports/text.ts +14 -0
  315. package/src/domains/eval/run-compare.ts +20 -0
  316. package/src/domains/eval/runners/clio-run.ts +127 -0
  317. package/src/domains/eval/runners/external-command.ts +28 -3
  318. package/src/domains/eval/schema/adapter.ts +111 -0
  319. package/src/domains/eval/schema/artifact.ts +20 -0
  320. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  321. package/src/domains/eval/schema/behavioral.ts +520 -0
  322. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  323. package/src/domains/eval/schema/serving.ts +105 -0
  324. package/src/domains/eval/schema/suite.ts +38 -8
  325. package/src/domains/eval/schema/validate.ts +58 -3
  326. package/src/domains/eval/schema/verdict.ts +237 -0
  327. package/src/domains/eval/suites/resolve.ts +2 -0
  328. package/src/domains/eval/suites/run.ts +264 -33
  329. package/src/domains/eval/verifiers/command.ts +2 -1
  330. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  331. package/src/domains/evidence/build.ts +2 -13
  332. package/src/domains/evidence/eval.ts +2 -12
  333. package/src/domains/evidence/findings-markdown.ts +33 -0
  334. package/src/domains/evidence/run-trust.ts +7 -113
  335. package/src/domains/evidence/store.ts +6 -0
  336. package/src/domains/evidence/trust-projection.ts +2 -2
  337. package/src/domains/extensions/compatibility.ts +285 -0
  338. package/src/domains/extensions/discovery.ts +38 -3
  339. package/src/domains/extensions/resources.ts +1 -1
  340. package/src/domains/extensions/state.ts +12 -3
  341. package/src/domains/extensions/types.ts +2 -0
  342. package/src/domains/lifecycle/doctor.ts +69 -1
  343. package/src/domains/memory/index.ts +14 -1
  344. package/src/domains/memory/task-bank-promotion.ts +64 -0
  345. package/src/domains/memory/task-memory-policy.ts +82 -17
  346. package/src/domains/memory/task-memory-spend.ts +131 -0
  347. package/src/domains/memory/task-memory-status.ts +7 -0
  348. package/src/domains/memory/task-memory-telemetry.ts +3 -0
  349. package/src/domains/middleware/index.ts +1 -0
  350. package/src/domains/middleware/memory-intervention.ts +97 -21
  351. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  352. package/src/domains/mux/contract.ts +434 -0
  353. package/src/domains/mux/detect.ts +158 -0
  354. package/src/domains/mux/extension.ts +47 -0
  355. package/src/domains/mux/index.ts +96 -0
  356. package/src/domains/mux/manifest.ts +6 -0
  357. package/src/domains/mux/operations.ts +164 -0
  358. package/src/domains/mux/pane-registry.ts +90 -0
  359. package/src/domains/mux/protocol.ts +49 -0
  360. package/src/domains/mux/socket-client.ts +816 -0
  361. package/src/domains/mux/types.ts +222 -0
  362. package/src/domains/mux/viewer-command.ts +59 -0
  363. package/src/domains/mux/yazi/assets/init.lua +2 -0
  364. package/src/domains/mux/yazi/assets/plugins/git.yazi/LICENSE +21 -0
  365. package/src/domains/mux/yazi/assets/plugins/git.yazi/README.md +78 -0
  366. package/src/domains/mux/yazi/assets/plugins/git.yazi/main.lua +255 -0
  367. package/src/domains/mux/yazi/assets/plugins/git.yazi/types.lua +12 -0
  368. package/src/domains/mux/yazi/assets/yazi.toml +17 -0
  369. package/src/domains/mux/yazi/event-stream.ts +180 -0
  370. package/src/domains/mux/yazi/profile.ts +299 -0
  371. package/src/domains/mux/yazi/session.ts +228 -0
  372. package/src/domains/mux/yazi/theme.ts +30 -0
  373. package/src/domains/observability/background-memory-usage.ts +140 -0
  374. package/src/domains/observability/cost.ts +22 -1
  375. package/src/domains/observability/index.ts +9 -0
  376. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  377. package/src/domains/observability/trace-store.ts +234 -2
  378. package/src/domains/prompts/compiler.ts +100 -13
  379. package/src/domains/providers/endpoint-capacity.ts +228 -0
  380. package/src/domains/providers/endpoint-slots-store.ts +189 -0
  381. package/src/domains/providers/extension.ts +20 -3
  382. package/src/domains/providers/index.ts +32 -0
  383. package/src/domains/providers/model-runtime-capabilities.ts +32 -0
  384. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  385. package/src/domains/providers/runtime-resolution.ts +8 -1
  386. package/src/domains/providers/runtimes/boot-manifest.ts +1 -0
  387. package/src/domains/providers/runtimes/builtins.ts +2 -0
  388. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  389. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  390. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  391. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  392. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  393. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  394. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  395. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  396. package/src/domains/providers/runtimes/protocol/litellm.ts +375 -0
  397. package/src/domains/providers/support.ts +1 -0
  398. package/src/domains/providers/target-model-cache.ts +124 -0
  399. package/src/domains/providers/types/capability-flags.ts +2 -0
  400. package/src/domains/providers/types/target-descriptor.ts +2 -0
  401. package/src/domains/resources/index.ts +3 -0
  402. package/src/domains/resources/prompts/loader.ts +95 -33
  403. package/src/domains/resources/skills/loader.ts +33 -0
  404. package/src/domains/safety/action-classifier.ts +6 -0
  405. package/src/domains/safety/call-target.ts +52 -0
  406. package/src/domains/safety/run-effects.ts +35 -4
  407. package/src/domains/session/context-accounting.ts +52 -1
  408. package/src/domains/session/context-ledger.ts +37 -13
  409. package/src/domains/session/index.ts +6 -0
  410. package/src/domains/session/prompt-cache.ts +140 -0
  411. package/src/domains/session/prompt-manifest.ts +42 -0
  412. package/src/domains/toolchain/archive.ts +175 -0
  413. package/src/domains/toolchain/contract.ts +28 -0
  414. package/src/domains/toolchain/extension.ts +47 -0
  415. package/src/domains/toolchain/index.ts +39 -0
  416. package/src/domains/toolchain/install.ts +327 -0
  417. package/src/domains/toolchain/manifest.ts +8 -0
  418. package/src/domains/toolchain/paths.ts +34 -0
  419. package/src/domains/toolchain/registry.ts +265 -0
  420. package/src/domains/toolchain/remove.ts +218 -0
  421. package/src/domains/toolchain/resolve.ts +182 -0
  422. package/src/domains/toolchain/types.ts +113 -0
  423. package/src/domains/toolchain/version.ts +88 -0
  424. package/src/engine/acp/adapter.ts +18 -3
  425. package/src/engine/acp/server.ts +413 -70
  426. package/src/engine/acp/types.ts +19 -1
  427. package/src/engine/ai.ts +35 -0
  428. package/src/engine/apis/llamacpp-residency.ts +55 -3
  429. package/src/engine/apis/lmstudio.ts +25 -5
  430. package/src/engine/apis/ollama-native.ts +2 -1
  431. package/src/engine/apis/openai-completions.ts +80 -17
  432. package/src/engine/apis/residency-lock.ts +3 -1
  433. package/src/engine/apis/residency.ts +34 -1
  434. package/src/engine/claude/sdk-module.ts +98 -0
  435. package/src/engine/claude/sdk-runtime.ts +19 -11
  436. package/src/engine/provider-payload.ts +29 -1
  437. package/src/engine/tui-primitives.ts +21 -0
  438. package/src/engine/tui.ts +1 -0
  439. package/src/engine/worker-runtime.ts +2 -13
  440. package/src/entry/boot-options.ts +2 -0
  441. package/src/entry/orchestrator.ts +279 -35
  442. package/src/entry/panes-activation.ts +31 -0
  443. package/src/entry/with-panes.ts +20 -0
  444. package/src/interactive/chat-loop-messages.ts +26 -7
  445. package/src/interactive/chat-loop.ts +318 -41
  446. package/src/interactive/chat-panel.ts +62 -8
  447. package/src/interactive/clio-editor.ts +45 -8
  448. package/src/interactive/context-activity.ts +5 -1
  449. package/src/interactive/context-meter.ts +1 -1
  450. package/src/interactive/context-overlay.ts +41 -10
  451. package/src/interactive/cost-overlay.ts +66 -6
  452. package/src/interactive/council-grid.ts +1 -3
  453. package/src/interactive/council.ts +11 -0
  454. package/src/interactive/dispatch-board.ts +102 -20
  455. package/src/interactive/fleet-run-preview.ts +41 -15
  456. package/src/interactive/handoff-round.ts +41 -2
  457. package/src/interactive/interactive-application.ts +177 -7
  458. package/src/interactive/interactive-input-runtime.ts +15 -0
  459. package/src/interactive/interactive-presentation.ts +4 -0
  460. package/src/interactive/interactive-shell.ts +20 -17
  461. package/src/interactive/interactive-slash-runtime.ts +63 -10
  462. package/src/interactive/memory-overlay.ts +9 -0
  463. package/src/interactive/modal-marker.ts +170 -0
  464. package/src/interactive/mutation-preview.ts +295 -0
  465. package/src/interactive/mux-bridge.ts +214 -0
  466. package/src/interactive/overlay-frame.ts +58 -2
  467. package/src/interactive/overlay-general-openers.ts +17 -0
  468. package/src/interactive/overlay-key-routing.ts +52 -3
  469. package/src/interactive/overlay-lifecycle.ts +56 -7
  470. package/src/interactive/overlay-model-selectors.ts +40 -3
  471. package/src/interactive/overlay-permission-lifecycle.ts +112 -24
  472. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  473. package/src/interactive/overlay-transitions.ts +18 -4
  474. package/src/interactive/overlays/agents.ts +1 -0
  475. package/src/interactive/overlays/ask-user.ts +227 -49
  476. package/src/interactive/overlays/auth-dialog.ts +1 -0
  477. package/src/interactive/overlays/context-reset.ts +1 -0
  478. package/src/interactive/overlays/cwd-fallback.ts +1 -0
  479. package/src/interactive/overlays/decisions.ts +11 -11
  480. package/src/interactive/overlays/extensions.ts +1 -0
  481. package/src/interactive/overlays/fleet-run-approval.ts +1 -0
  482. package/src/interactive/overlays/handoff-review.ts +1 -0
  483. package/src/interactive/overlays/help-reference.ts +6 -0
  484. package/src/interactive/overlays/interop.ts +1 -0
  485. package/src/interactive/overlays/library-install-confirm.ts +1 -0
  486. package/src/interactive/overlays/library-tabs.ts +28 -0
  487. package/src/interactive/overlays/list-overlay.ts +10 -1
  488. package/src/interactive/overlays/message-picker.ts +1 -0
  489. package/src/interactive/overlays/model-scope.ts +86 -0
  490. package/src/interactive/overlays/model-selector.ts +1 -0
  491. package/src/interactive/overlays/prompts.ts +12 -1
  492. package/src/interactive/overlays/session-selector.ts +1 -0
  493. package/src/interactive/overlays/settings-sections.ts +30 -0
  494. package/src/interactive/overlays/settings.ts +614 -45
  495. package/src/interactive/overlays/side-question.ts +1 -0
  496. package/src/interactive/overlays/skills-hub.ts +3 -11
  497. package/src/interactive/overlays/tree-selector.ts +1 -0
  498. package/src/interactive/pane-policy.ts +46 -0
  499. package/src/interactive/panes-runtime.ts +292 -0
  500. package/src/interactive/permission-hint.ts +34 -2
  501. package/src/interactive/permission-overlay.ts +159 -9
  502. package/src/interactive/prewarm.ts +197 -0
  503. package/src/interactive/render-trace.ts +162 -15
  504. package/src/interactive/renderers/compaction-summary.ts +29 -0
  505. package/src/interactive/renderers/tool-execution.ts +4 -0
  506. package/src/interactive/renderers/worker-entry.ts +122 -14
  507. package/src/interactive/side-question.ts +58 -1
  508. package/src/interactive/slash-commands.ts +251 -15
  509. package/src/interactive/status/controller.ts +11 -0
  510. package/src/interactive/status/state-machine.ts +54 -2
  511. package/src/interactive/status/types.ts +7 -0
  512. package/src/interactive/tasks-overlay.ts +1 -0
  513. package/src/interactive/terminal-lease.ts +2 -0
  514. package/src/interactive/theme/tokens.ts +3 -14
  515. package/src/interactive/turn-context.ts +346 -33
  516. package/src/interactive/turn-persistence.ts +14 -4
  517. package/src/interactive/turn-prewarm.ts +364 -0
  518. package/src/interactive/turn-queues.ts +7 -4
  519. package/src/interactive/turn-runtime.ts +8 -1
  520. package/src/interactive/turn-state.ts +23 -0
  521. package/src/interactive/view/artifacts.ts +109 -1
  522. package/src/interactive/view/view-overlay.ts +29 -3
  523. package/src/interactive/watch-pane.ts +152 -0
  524. package/src/interactive/worker-progress.ts +7 -1
  525. package/src/interactive/worker-receipts.ts +19 -1
  526. package/src/interactive/worker-stream.ts +5 -0
  527. package/src/interactive/yazi-bridge.ts +444 -0
  528. package/src/tools/ask-user.ts +43 -2
  529. package/src/tools/bootstrap.ts +26 -2
  530. package/src/tools/builtin-tool-catalog.ts +15 -0
  531. package/src/tools/compete-worktrees.ts +83 -2
  532. package/src/tools/core-bootstrap.ts +2 -1
  533. package/src/tools/dispatch-admission.ts +14 -3
  534. package/src/tools/dispatch-arguments.ts +20 -20
  535. package/src/tools/dispatch-plan.ts +17 -9
  536. package/src/tools/dispatch-run-events.ts +134 -19
  537. package/src/tools/dispatch-runner.ts +29 -7
  538. package/src/tools/dispatch-scout.ts +1 -1
  539. package/src/tools/dispatch-types.ts +15 -3
  540. package/src/tools/dispatch.ts +1 -1
  541. package/src/tools/executables.ts +17 -14
  542. package/src/tools/observation.ts +54 -4
  543. package/src/tools/panes-surface.ts +38 -0
  544. package/src/tools/panes.ts +112 -0
  545. package/src/tools/policy.ts +10 -1
  546. package/src/tools/presentation.ts +1 -0
  547. package/src/tools/registry.ts +16 -0
  548. package/dist/chunk-AOCYTWAV.js +0 -449
  549. package/dist/chunk-HLE42MG7.js +0 -37
  550. package/dist/chunk-HWUFFB6L.js +0 -83
  551. package/dist/chunk-JOZYP4GM.js +0 -279
  552. package/dist/doctor-M7YEDGAE.js +0 -91
@@ -1,32 +1,79 @@
1
1
  import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
2
2
  import {
3
+ EVAL_SUITE_V2_VERSION
4
+ } from "./chunk-KMVISBZR.js";
5
+ import {
6
+ loadFragments,
3
7
  renderCodewikiDigest
4
- } from "./chunk-5WIGXA4T.js";
8
+ } from "./chunk-VAZSBTKF.js";
5
9
  import {
6
10
  EvalTaskFileError,
7
11
  TRUST_STATUS_AXES,
8
12
  adaptRunReceiptTrustStatus,
9
- createEvalId,
13
+ assertComparableTrackedMetricSources,
10
14
  evalClioProvenance,
11
15
  evalEnvironmentProvenance,
16
+ evalServingConfiguration,
17
+ evalServingObservationFrom,
12
18
  formatTrustSummary,
13
19
  inspectRunReceiptTrustStatus,
14
- loadEvalArtifactV4,
15
20
  loadEvalTaskFile,
16
21
  summarizeTrustStatus,
17
- verifyReceiptIntegrity,
22
+ verifyReceiptIntegrity
23
+ } from "./chunk-B7OBL7PK.js";
24
+ import {
25
+ listSessionLedgerRefs,
26
+ parseSessionEntries
27
+ } from "./chunk-QREDIESB.js";
28
+ import "./chunk-3DPEIQKN.js";
29
+ import {
30
+ EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
31
+ EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
32
+ EVAL_TRACKED_METRIC_NAMES,
33
+ EVAL_VERDICT_SCHEMA_V1,
34
+ assertEvalBehaviorReferencesVerdictV1,
35
+ buildEvalBehaviorMetricsV1,
36
+ createEvalId,
37
+ evalHarnessMetricsFromReceipt,
38
+ evalServingConfigurationOf,
39
+ judgeEvalBehaviorV1,
40
+ loadEvalArtifactV4,
41
+ parseEvalBehaviorScenarioV1,
42
+ parseEvalExecutionMatrixDimensionsV1,
43
+ parseEvalVerdictEnvelopeV1,
44
+ renderEvalServingConfiguration,
45
+ sameEvalServingConfiguration,
18
46
  writeEvalArtifactV4
19
- } from "./chunk-K4XHGFR5.js";
47
+ } from "./chunk-NHLBIGRH.js";
48
+ import "./chunk-W6GROXXM.js";
20
49
  import {
21
50
  shellQuote
22
51
  } from "./chunk-TXOTCRLG.js";
23
- import "./chunk-P43ETTHK.js";
52
+ import "./chunk-HVDIIIQW.js";
53
+ import {
54
+ discoverAgentRecipes
55
+ } from "./chunk-Z4TXYIEG.js";
56
+ import "./chunk-VYMXRQI6.js";
57
+ import "./chunk-FEFIFZTL.js";
58
+ import "./chunk-RKKLTLYB.js";
59
+ import "./chunk-SUW5DORT.js";
60
+ import "./chunk-SJ5ZKQ4S.js";
61
+ import "./chunk-2JDWVJND.js";
62
+ import {
63
+ createSafetyPolicyEngine
64
+ } from "./chunk-32KWKNSF.js";
24
65
  import "./chunk-H7IXIC72.js";
25
- import "./chunk-AOCYTWAV.js";
26
- import "./chunk-MV3K5QF2.js";
66
+ import "./chunk-5LXZXPKX.js";
67
+ import "./chunk-RAPCMZL4.js";
68
+ import "./chunk-HHV2GANA.js";
69
+ import {
70
+ agentSpecFingerprint,
71
+ normalizeAgentSpec
72
+ } from "./chunk-DYJP44XW.js";
73
+ import "./chunk-GCSMB2KY.js";
27
74
  import "./chunk-UL3WSD3F.js";
28
75
  import "./chunk-ECH6PKUQ.js";
29
- import "./chunk-CGKSTWHD.js";
76
+ import "./chunk-K6BSR66V.js";
30
77
  import {
31
78
  readCodewiki,
32
79
  structuralCodewikiHash
@@ -38,16 +85,31 @@ import {
38
85
  enumerateWorkspaceFiles
39
86
  } from "./chunk-33YXPOE3.js";
40
87
  import "./chunk-7CR24IG7.js";
41
- import "./chunk-IFBNV6H6.js";
88
+ import "./chunk-XPLRXC72.js";
89
+ import "./chunk-2ANTL7MR.js";
42
90
  import {
43
91
  printError
44
- } from "./chunk-XK56QHLX.js";
92
+ } from "./chunk-VPKWYKEY.js";
45
93
  import "./chunk-5TSRNF4G.js";
94
+ import "./chunk-CFGTUFWB.js";
95
+ import {
96
+ extractReasoningTokens
97
+ } from "./chunk-UM7N4G5A.js";
98
+ import "./chunk-HLW2MRKE.js";
99
+ import "./chunk-76ONBSIA.js";
46
100
  import {
47
101
  InvalidIdError
48
102
  } from "./chunk-R346GLFC.js";
103
+ import "./chunk-IHXBNWMM.js";
104
+ import "./chunk-VFA6GDY5.js";
105
+ import "./chunk-BBVJUZHB.js";
106
+ import "./chunk-FQ4SKYE4.js";
107
+ import "./chunk-6EJMN2Y3.js";
49
108
  import "./chunk-IWHMRKLL.js";
50
- import "./chunk-EQ63NRB7.js";
109
+ import "./chunk-GXNLGKAB.js";
110
+ import "./chunk-LL4KHSZI.js";
111
+ import "./chunk-4ZG3XFUR.js";
112
+ import "./chunk-BBVYXMFO.js";
51
113
  import "./chunk-SST6Z5JA.js";
52
114
  import "./chunk-IKCO5N3L.js";
53
115
  import "./chunk-3I7MS7N2.js";
@@ -57,7 +119,7 @@ import {
57
119
  import {
58
120
  clioDataDir,
59
121
  clioStateDir
60
- } from "./chunk-BNAZZHFG.js";
122
+ } from "./chunk-BYP5D4HI.js";
61
123
  import "./chunk-WEPFGWHJ.js";
62
124
  import "./chunk-YXLYO42X.js";
63
125
  import {
@@ -67,31 +129,574 @@ import {
67
129
 
68
130
  // src/cli/eval.ts
69
131
  init_esm_shims();
70
- import { resolve as resolve7 } from "node:path";
132
+ import { resolve as resolve8 } from "node:path";
71
133
 
72
134
  // src/domains/eval/compare/compare.ts
73
135
  init_esm_shims();
74
- function compareEvalArtifactsV4(baseline, candidate) {
136
+
137
+ // src/domains/eval/metrics/aggregate.ts
138
+ init_esm_shims();
139
+ function aggregateEvalVerdicts(verdicts) {
140
+ const byScenario = /* @__PURE__ */ new Map();
141
+ for (const verdict of verdicts) {
142
+ const group = byScenario.get(verdict.scenarioId) ?? [];
143
+ group.push(verdict);
144
+ byScenario.set(verdict.scenarioId, group);
145
+ }
146
+ return [...byScenario.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([scenarioId, group]) => aggregateScenario(scenarioId, group));
147
+ }
148
+ function aggregateScenario(scenarioId, verdicts) {
149
+ const ordered = [...verdicts].sort((left, right) => left.trialIndex - right.trialIndex);
150
+ const passed = ordered.filter((verdict) => verdict.outcome === "pass").length;
151
+ const failed = ordered.filter((verdict) => verdict.outcome === "fail").length;
152
+ const unmeasured = ordered.filter((verdict) => verdict.outcome === "unmeasured").length;
153
+ const machineryFailures = ordered.filter((verdict) => verdict.machinery === "infrastructure_failure").length;
154
+ const fixed = Object.fromEntries(
155
+ EVAL_TRACKED_METRIC_NAMES.map((name) => [name, distribution(ordered.map((verdict) => verdict.trackedMetrics[name]))])
156
+ );
157
+ const reasons = new Set(ordered.flatMap((verdict) => Object.keys(verdict.trackedMetrics.expectedColdReasons)));
158
+ const expectedColdReasons = Object.fromEntries(
159
+ [...reasons].sort((left, right) => left.localeCompare(right)).map((reason) => [
160
+ reason,
161
+ distribution(
162
+ ordered.map(
163
+ (verdict) => verdict.trackedMetrics.expectedColdReasons[reason] ?? { value: 0, source: "ledger" }
164
+ )
165
+ )
166
+ ])
167
+ );
168
+ const k = ordered.length;
169
+ return {
170
+ scenarioId,
171
+ trials: k,
172
+ k,
173
+ passed,
174
+ failed,
175
+ unmeasured,
176
+ machineryFailures,
177
+ passAtK: k > 0 && passed > 0 ? 1 : 0,
178
+ passPowK: k > 0 && passed === k ? 1 : 0,
179
+ trackedMetrics: { ...fixed, expectedColdReasons }
180
+ };
181
+ }
182
+ function distribution(metrics) {
183
+ const values = metrics.flatMap((metric) => metric.value === null ? [] : [metric.value]);
184
+ const sources = [...new Set(metrics.map((metric) => metric.source))].sort(compareSources);
185
+ if (values.length === 0) {
186
+ return {
187
+ observations: metrics.length,
188
+ measured: 0,
189
+ unmeasured: metrics.length,
190
+ mean: null,
191
+ min: null,
192
+ max: null,
193
+ p90: null,
194
+ variance: null,
195
+ standardDeviation: null,
196
+ sources
197
+ };
198
+ }
199
+ const ordered = [...values].sort((left, right) => left - right);
200
+ const p90Index = Math.max(0, Math.ceil(ordered.length * 0.9) - 1);
201
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
202
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
203
+ return {
204
+ observations: metrics.length,
205
+ measured: values.length,
206
+ unmeasured: metrics.length - values.length,
207
+ mean,
208
+ min: ordered[0] ?? null,
209
+ max: ordered.at(-1) ?? null,
210
+ p90: ordered[p90Index] ?? null,
211
+ variance,
212
+ standardDeviation: Math.sqrt(variance),
213
+ sources
214
+ };
215
+ }
216
+ function compareSources(left, right) {
217
+ return sourceOrder(left) - sourceOrder(right);
218
+ }
219
+ function sourceOrder(source) {
220
+ if (source === "ledger") return 0;
221
+ if (source === "receipt") return 1;
222
+ return 2;
223
+ }
224
+
225
+ // src/domains/eval/compare/behavioral.ts
226
+ init_esm_shims();
227
+
228
+ // src/domains/eval/compare/envelope.ts
229
+ init_esm_shims();
230
+ function compareEvalExecutionEnvelopesV1(identity, baseline, candidate, baselineDimensions, candidateDimensions) {
231
+ const leftDimensions = [...baselineDimensions].sort();
232
+ const rightDimensions = [...candidateDimensions].sort();
233
+ if (stableJson(leftDimensions) !== stableJson(rightDimensions)) {
234
+ return { ...identity, fields: ["matrix.dimensions"] };
235
+ }
236
+ if (baseline.some((result) => result.executionEnvelope !== void 0) && baseline.some((result) => result.executionEnvelope === void 0) || candidate.some((result) => result.executionEnvelope !== void 0) && candidate.some((result) => result.executionEnvelope === void 0)) {
237
+ return { ...identity, fields: ["executionEnvelope.missingTrial"] };
238
+ }
239
+ const ignored = new Set(leftDimensions);
240
+ const baselineEnvelopes = uniqueEnvelopes(baseline, ignored);
241
+ const candidateEnvelopes = uniqueEnvelopes(candidate, ignored);
242
+ if (baselineEnvelopes.length === 0 && candidateEnvelopes.length === 0) return null;
243
+ if (baselineEnvelopes.length === 0 || candidateEnvelopes.length === 0) {
244
+ return { ...identity, fields: ["executionEnvelope"] };
245
+ }
246
+ if (baselineEnvelopes.length > 1 || candidateEnvelopes.length > 1) {
247
+ return { ...identity, fields: ["executionEnvelope.withinRunVariance"] };
248
+ }
249
+ const left = baselineEnvelopes[0];
250
+ const right = candidateEnvelopes[0];
251
+ if (left === void 0 || right === void 0 || stableJson(left) === stableJson(right)) return null;
252
+ return { ...identity, fields: differingFields(left, right, ignored) };
253
+ }
254
+ function uniqueEnvelopes(results, ignored) {
255
+ const byIdentity = /* @__PURE__ */ new Map();
256
+ for (const result of results) {
257
+ if (result.executionEnvelope === void 0) continue;
258
+ const normalized = normalizedEnvelope(result.executionEnvelope, ignored);
259
+ byIdentity.set(stableJson(normalized), normalized);
260
+ }
261
+ return [...byIdentity.values()];
262
+ }
263
+ function normalizedEnvelope(envelope, ignored) {
264
+ return {
265
+ ...envelope,
266
+ prompt: ignored.has("prompt") ? { fragments: [], compositionHash: null } : envelope.prompt,
267
+ recipe: ignored.has("recipe") ? null : envelope.recipe,
268
+ target: ignored.has("target") ? "<matrix>" : envelope.target,
269
+ wireModel: ignored.has("wireModel") ? null : envelope.wireModel,
270
+ runtime: ignored.has("runtime") ? null : envelope.runtime,
271
+ thinkingLevel: ignored.has("thinkingLevel") ? null : envelope.thinkingLevel,
272
+ toolSignature: ignored.has("toolSignature") ? null : envelope.toolSignature,
273
+ autonomy: ignored.has("autonomy") ? null : envelope.autonomy,
274
+ policyHashes: ignored.has("policy") ? { rulePack: null, project: null } : envelope.policyHashes,
275
+ projectContext: ignored.has("projectContext") ? {
276
+ kind: "none",
277
+ tier: null,
278
+ contentHash: null,
279
+ chars: null,
280
+ sections: [],
281
+ rulesApplied: [],
282
+ operatorProfileApplied: null
283
+ } : envelope.projectContext,
284
+ corpus: ignored.has("corpus") ? { id: "<matrix>", version: "<matrix>" } : envelope.corpus
285
+ };
286
+ }
287
+ function differingFields(left, right, ignored) {
288
+ const fields = [
289
+ ["prompt", "prompt", left.prompt, right.prompt],
290
+ ["recipe", "recipe", left.recipe, right.recipe],
291
+ ["target", "target", left.target, right.target],
292
+ ["wireModel", "wireModel", left.wireModel, right.wireModel],
293
+ ["runtime", "runtime", left.runtime, right.runtime],
294
+ ["thinkingLevel", "thinkingLevel", left.thinkingLevel, right.thinkingLevel],
295
+ ["toolSignature", "toolSignature", left.toolSignature, right.toolSignature],
296
+ ["autonomy", "autonomy", left.autonomy, right.autonomy],
297
+ ["policy", "policyHashes", left.policyHashes, right.policyHashes],
298
+ ["projectContext", "projectContext", left.projectContext, right.projectContext],
299
+ ["corpus", "corpus", left.corpus, right.corpus]
300
+ ];
301
+ return fields.flatMap(
302
+ ([dimension, field, baseline, candidate]) => ignored.has(dimension) || stableJson(baseline) === stableJson(candidate) ? [] : [field]
303
+ );
304
+ }
305
+ function stableJson(value) {
306
+ if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
307
+ if (typeof value === "object" && value !== null) {
308
+ return `{${Object.entries(value).filter(([, entry]) => entry !== void 0).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
309
+ }
310
+ return JSON.stringify(value);
311
+ }
312
+
313
+ // src/domains/eval/compare/behavioral.ts
314
+ function compareEvalBehaviorMetricsV1(baseline, candidate) {
315
+ const baselineGroups = behaviorGroups(baseline);
316
+ const candidateGroups = behaviorGroups(candidate);
317
+ const keys = /* @__PURE__ */ new Set([...baselineGroups.keys(), ...candidateGroups.keys()]);
318
+ const comparisons = [];
319
+ const envelopeMismatches = [];
320
+ const baselineDimensions = baseline.matrix.dimensions ?? [];
321
+ const candidateDimensions = candidate.matrix.dimensions ?? [];
322
+ for (const key of [...keys].sort((left, right) => left.localeCompare(right))) {
323
+ const baselineGroup = baselineGroups.get(key);
324
+ const candidateGroup = candidateGroups.get(key);
325
+ const identity = baselineGroup ?? candidateGroup;
326
+ if (identity === void 0) continue;
327
+ const envelopeMismatch = compareEvalExecutionEnvelopesV1(
328
+ identity,
329
+ baselineGroup?.results ?? [],
330
+ candidateGroup?.results ?? [],
331
+ baselineDimensions,
332
+ candidateDimensions
333
+ );
334
+ if (envelopeMismatch !== null) envelopeMismatches.push(envelopeMismatch);
335
+ const comparability = {
336
+ comparable: envelopeMismatch === null,
337
+ mismatchedFields: envelopeMismatch?.fields ?? []
338
+ };
339
+ for (const definition of EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1) {
340
+ const baselineDistribution = behaviorDistribution(baselineGroup?.results ?? [], definition);
341
+ const candidateDistribution = behaviorDistribution(candidateGroup?.results ?? [], definition);
342
+ comparisons.push({
343
+ scenarioId: identity.scenarioId,
344
+ role: identity.role,
345
+ target: identity.target,
346
+ metric: definition.name,
347
+ family: definition.family,
348
+ direction: definition.direction,
349
+ hardGate: definition.hardGate,
350
+ baseline: baselineDistribution,
351
+ candidate: candidateDistribution,
352
+ change: envelopeMismatch === null ? classifyChange(baselineDistribution.mean, candidateDistribution.mean, definition.direction) : "incomparable",
353
+ meanDelta: subtractNullable(candidateDistribution.mean, baselineDistribution.mean),
354
+ varianceChange: envelopeMismatch === null ? classifyChange(baselineDistribution.variance, candidateDistribution.variance, "lower") : "incomparable",
355
+ varianceDelta: subtractNullable(candidateDistribution.variance, baselineDistribution.variance),
356
+ comparability
357
+ });
358
+ }
359
+ }
360
+ const failures = comparisons.flatMap(
361
+ (comparison) => comparison.hardGate && (comparison.change === "regressed" || comparison.change === "incomparable" && comparison.baseline.mean !== null && comparison.candidate.mean === null) ? [
362
+ {
363
+ scenarioId: comparison.scenarioId,
364
+ role: comparison.role,
365
+ target: comparison.target,
366
+ metric: comparison.metric,
367
+ change: comparison.change
368
+ }
369
+ ] : []
370
+ );
371
+ return {
372
+ comparisons,
373
+ hardGate: {
374
+ pass: failures.length === 0 && envelopeMismatches.length === 0,
375
+ failures,
376
+ envelopeFailures: envelopeMismatches
377
+ },
378
+ envelopeMismatches
379
+ };
380
+ }
381
+ function classifyChange(baseline, candidate, direction) {
382
+ if (baseline === null || candidate === null) return "incomparable";
383
+ if (baseline === candidate) return "unchanged";
384
+ if (direction === "higher") return candidate > baseline ? "improved" : "regressed";
385
+ return candidate < baseline ? "improved" : "regressed";
386
+ }
387
+ function behaviorGroups(artifact) {
388
+ const groups = /* @__PURE__ */ new Map();
389
+ for (const result of artifact.results) {
390
+ const behavioral = result.behavioralMetrics;
391
+ if (behavioral === void 0) continue;
392
+ const key = groupKey(behavioral.scenarioId, behavioral.role, behavioral.target);
393
+ const group = groups.get(key) ?? {
394
+ scenarioId: behavioral.scenarioId,
395
+ role: behavioral.role,
396
+ target: behavioral.target,
397
+ results: []
398
+ };
399
+ group.results.push(result);
400
+ groups.set(key, group);
401
+ }
402
+ return groups;
403
+ }
404
+ function behaviorDistribution(results, definition) {
405
+ const observations = results.map((result) => result.behavioralMetrics?.metrics[definition.name].value ?? null);
406
+ const values = observations.flatMap((value) => value === null ? [] : [value]);
407
+ if (values.length === 0) {
408
+ return {
409
+ observations: observations.length,
410
+ measured: 0,
411
+ unmeasured: observations.length,
412
+ mean: null,
413
+ min: null,
414
+ max: null,
415
+ p90: null,
416
+ variance: null,
417
+ standardDeviation: null,
418
+ source: definition.source
419
+ };
420
+ }
421
+ const ordered = [...values].sort((left, right) => left - right);
422
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
423
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
424
+ return {
425
+ observations: observations.length,
426
+ measured: values.length,
427
+ unmeasured: observations.length - values.length,
428
+ mean,
429
+ min: ordered[0] ?? null,
430
+ max: ordered.at(-1) ?? null,
431
+ p90: ordered[Math.max(0, Math.ceil(ordered.length * 0.9) - 1)] ?? null,
432
+ variance,
433
+ standardDeviation: Math.sqrt(variance),
434
+ source: definition.source
435
+ };
436
+ }
437
+ function groupKey(scenarioId, role, target) {
438
+ return JSON.stringify([scenarioId, role, target.id, target.model]);
439
+ }
440
+ function subtractNullable(left, right) {
441
+ return left === null || right === null ? null : left - right;
442
+ }
443
+
444
+ // src/domains/eval/compare/compare.ts
445
+ var EvalServingConfigurationDriftError = class extends Error {
446
+ baseline;
447
+ candidate;
448
+ constructor(baseline, candidate) {
449
+ super(
450
+ [
451
+ "serving configuration drift; pass --allow-config-drift to compare these runs",
452
+ `baseline serving: ${renderEvalServingConfiguration(baseline)}`,
453
+ `candidate serving: ${renderEvalServingConfiguration(candidate)}`
454
+ ].join("\n")
455
+ );
456
+ this.name = "EvalServingConfigurationDriftError";
457
+ this.baseline = baseline;
458
+ this.candidate = candidate;
459
+ }
460
+ };
461
+ function compareEvalArtifactsV4(baseline, candidate, options = {}) {
75
462
  const baselineTokens = baseline.summary.tokens;
76
463
  const candidateTokens = candidate.summary.tokens;
464
+ const baselineServing = evalServingConfigurationOf(baseline);
465
+ const candidateServing = evalServingConfigurationOf(candidate);
466
+ const configDrift = !sameEvalServingConfiguration(baselineServing, candidateServing);
467
+ if (configDrift && options.allowConfigDrift !== true) {
468
+ throw new EvalServingConfigurationDriftError(baselineServing, candidateServing);
469
+ }
470
+ const behavioral = compareEvalBehaviorMetricsV1(baseline, candidate);
471
+ const trackedMetrics = compareTrackedMetrics(baseline, candidate, options.metric);
472
+ const normalizedFilter = normalizeMetricFilter(options.metric);
473
+ const behavioralMetrics = normalizedFilter === void 0 ? behavioral.comparisons : behavioral.comparisons.filter((row) => row.metric === normalizedFilter || row.family === normalizedFilter);
474
+ if (options.metric !== void 0 && trackedMetrics.length === 0 && behavioralMetrics.length === 0) {
475
+ throw new Error(`eval metric not found: ${options.metric}`);
476
+ }
77
477
  return {
78
478
  baselineEvalId: baseline.evalId,
79
479
  candidateEvalId: candidate.evalId,
480
+ baselineServingConfiguration: baselineServing,
481
+ candidateServingConfiguration: candidateServing,
482
+ configDrift,
80
483
  passRateDelta: candidate.summary.passRate - baseline.summary.passRate,
81
484
  tokenDelta: baselineTokens.measured && candidateTokens.measured ? candidateTokens.total - baselineTokens.total : null,
82
- wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs
485
+ wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs,
486
+ trackedMetrics,
487
+ behavioralMetrics,
488
+ hardGate: behavioral.hardGate,
489
+ envelopeMismatches: behavioral.envelopeMismatches,
490
+ scenarioReports: behaviorRollups(behavioralMetrics, (row) => row.scenarioId),
491
+ roleReports: behaviorRollups(behavioralMetrics, (row) => row.role),
492
+ affectedCorpusResults: behavioral.envelopeMismatches.flatMap((mismatch) => {
493
+ const changedFields = mismatch.fields.filter((field) => field === "prompt" || field === "recipe");
494
+ return changedFields.length === 0 ? [] : [{ scenarioId: mismatch.scenarioId, role: mismatch.role, changedFields }];
495
+ })
83
496
  };
84
497
  }
85
498
  function renderEvalComparisonV4(summary) {
499
+ const envelopeFailures = summary.envelopeMismatches.map(
500
+ (mismatch) => ` incomparable envelope: ${mismatch.scenarioId} ${mismatch.role} ${mismatch.target.id}/${mismatch.target.model ?? "none"} fields=${mismatch.fields.join(",")}`
501
+ );
502
+ const affected = summary.affectedCorpusResults.map(
503
+ (result) => ` affected corpus result: ${result.scenarioId} role=${result.role} changed=${result.changedFields.join(",")}`
504
+ );
505
+ const scenarioReports = renderRollups("per-scenario baseline/candidate report", summary.scenarioReports);
506
+ const roleReports = renderRollups("per-role baseline/candidate report", summary.roleReports);
507
+ const hardFailures = summary.hardGate.failures.map(
508
+ (failure) => ` hard failure: ${failure.scenarioId} ${failure.role} ${failure.target.id}/${failure.target.model ?? "none"} ${failure.metric} ${failure.change}`
509
+ );
510
+ const tracked = summary.trackedMetrics.flatMap((row, index) => [
511
+ ...index === 0 ? [
512
+ "tracked metrics:",
513
+ "scenario metric baseline_mean baseline_p90 baseline_variance candidate_mean candidate_p90 candidate_variance mean_delta p90_delta variance_delta change variance_change sources"
514
+ ] : [],
515
+ [
516
+ row.scenarioId,
517
+ row.metric,
518
+ formatMetric(row.baseline.mean),
519
+ formatMetric(row.baseline.p90),
520
+ formatMetric(row.baseline.variance ?? null),
521
+ formatMetric(row.candidate.mean),
522
+ formatMetric(row.candidate.p90),
523
+ formatMetric(row.candidate.variance ?? null),
524
+ formatSignedMetric(row.meanDelta),
525
+ formatSignedMetric(row.p90Delta),
526
+ formatSignedMetric(row.varianceDelta),
527
+ row.change,
528
+ row.varianceChange,
529
+ `${row.baseline.sources.join("+") || "none"}->${row.candidate.sources.join("+") || "none"}`
530
+ ].join(" ")
531
+ ]);
532
+ const behavioral = summary.behavioralMetrics.flatMap((row, index) => [
533
+ ...index === 0 ? [
534
+ "behavioral metrics:",
535
+ "scenario role target model family metric baseline_mean baseline_variance baseline_coverage candidate_mean candidate_variance candidate_coverage mean_delta variance_delta change variance_change comparability gate source"
536
+ ] : [],
537
+ [
538
+ row.scenarioId,
539
+ row.role,
540
+ row.target.id,
541
+ row.target.model ?? "none",
542
+ row.family,
543
+ row.metric,
544
+ formatMetric(row.baseline.mean),
545
+ formatMetric(row.baseline.variance),
546
+ `${row.baseline.measured}/${row.baseline.observations}`,
547
+ formatMetric(row.candidate.mean),
548
+ formatMetric(row.candidate.variance),
549
+ `${row.candidate.measured}/${row.candidate.observations}`,
550
+ formatSignedMetric(row.meanDelta),
551
+ formatSignedMetric(row.varianceDelta),
552
+ row.change,
553
+ row.varianceChange,
554
+ row.comparability.comparable ? "comparable" : `incomparable:${row.comparability.mismatchedFields.join(",")}`,
555
+ row.hardGate ? "hard" : "informational",
556
+ row.baseline.source
557
+ ].join(" ")
558
+ ]);
86
559
  return [
87
560
  `baseline eval: ${summary.baselineEvalId}`,
88
561
  `candidate eval: ${summary.candidateEvalId}`,
562
+ `baseline serving: ${renderEvalServingConfiguration(summary.baselineServingConfiguration)}`,
563
+ `candidate serving: ${renderEvalServingConfiguration(summary.candidateServingConfiguration)}`,
564
+ `config drift: ${summary.configDrift ? "allowed" : "none"}`,
89
565
  `pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
90
566
  `token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
91
567
  `wall-time delta ms: ${summary.wallTimeDelta}`,
568
+ `behavioral hard gate: ${summary.hardGate.pass ? "pass" : `fail (${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length})`}`,
569
+ ...hardFailures,
570
+ ...envelopeFailures,
571
+ ...affected,
572
+ ...scenarioReports,
573
+ ...roleReports,
574
+ ...tracked,
575
+ ...behavioral,
92
576
  ""
93
577
  ].join("\n");
94
578
  }
579
+ function behaviorRollups(rows, keyOf) {
580
+ const groups = /* @__PURE__ */ new Map();
581
+ for (const row of rows) groups.set(keyOf(row), [...groups.get(keyOf(row)) ?? [], row]);
582
+ return [...groups.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([id, grouped]) => ({
583
+ id,
584
+ metrics: changeCounts(grouped.map((row) => row.change)),
585
+ variance: changeCounts(grouped.map((row) => row.varianceChange))
586
+ }));
587
+ }
588
+ function changeCounts(changes) {
589
+ return {
590
+ improved: changes.filter((change) => change === "improved").length,
591
+ regressed: changes.filter((change) => change === "regressed").length,
592
+ unchanged: changes.filter((change) => change === "unchanged").length,
593
+ incomparable: changes.filter((change) => change === "incomparable").length
594
+ };
595
+ }
596
+ function renderRollups(title, reports) {
597
+ if (reports.length === 0) return [];
598
+ return [
599
+ `${title}:`,
600
+ ...reports.map(
601
+ (report) => ` ${report.id}: metrics ${renderChangeCounts(report.metrics)}; variance ${renderChangeCounts(report.variance)}`
602
+ )
603
+ ];
604
+ }
605
+ function renderChangeCounts(counts) {
606
+ return `improved=${counts.improved} regressed=${counts.regressed} unchanged=${counts.unchanged} incomparable=${counts.incomparable}`;
607
+ }
608
+ function compareTrackedMetrics(baseline, candidate, metricFilter) {
609
+ const baselineAggregates = baseline.aggregates ?? aggregateEvalVerdicts(baseline.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
610
+ const candidateAggregates = candidate.aggregates ?? aggregateEvalVerdicts(candidate.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
611
+ const baselineByScenario = new Map(baselineAggregates.map((entry) => [entry.scenarioId, entry]));
612
+ const candidateByScenario = new Map(candidateAggregates.map((entry) => [entry.scenarioId, entry]));
613
+ const scenarioIds = [...baselineByScenario.keys()].filter((scenarioId) => candidateByScenario.has(scenarioId)).sort((left, right) => left.localeCompare(right));
614
+ const filter = normalizeMetricFilter(metricFilter);
615
+ const rows = [];
616
+ for (const scenarioId of scenarioIds) {
617
+ const baselineAggregate = baselineByScenario.get(scenarioId);
618
+ const candidateAggregate = candidateByScenario.get(scenarioId);
619
+ if (baselineAggregate === void 0 || candidateAggregate === void 0) continue;
620
+ for (const metric of EVAL_TRACKED_METRIC_NAMES) {
621
+ if (filter !== void 0 && filter !== metric) continue;
622
+ rows.push(
623
+ metricComparison(
624
+ scenarioId,
625
+ metric,
626
+ baselineAggregate.trackedMetrics[metric],
627
+ candidateAggregate.trackedMetrics[metric]
628
+ )
629
+ );
630
+ }
631
+ const reasons = /* @__PURE__ */ new Set([
632
+ ...Object.keys(baselineAggregate.trackedMetrics.expectedColdReasons),
633
+ ...Object.keys(candidateAggregate.trackedMetrics.expectedColdReasons)
634
+ ]);
635
+ for (const reason of [...reasons].sort((left, right) => left.localeCompare(right))) {
636
+ const metric = `expectedColdReasons.${reason}`;
637
+ if (filter !== void 0 && filter !== metric && filter !== "expectedColdReasons") continue;
638
+ rows.push(
639
+ metricComparison(
640
+ scenarioId,
641
+ metric,
642
+ baselineAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(baselineAggregate.k),
643
+ candidateAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(candidateAggregate.k)
644
+ )
645
+ );
646
+ }
647
+ }
648
+ return rows;
649
+ }
650
+ function metricComparison(scenarioId, metric, baseline, candidate) {
651
+ assertComparableTrackedMetricSources(`${scenarioId}.${metric}`, baseline.sources, candidate.sources);
652
+ const direction = trackedMetricDirection(metric);
653
+ return {
654
+ scenarioId,
655
+ metric,
656
+ baseline,
657
+ candidate,
658
+ meanDelta: subtractNullable2(candidate.mean, baseline.mean),
659
+ p90Delta: subtractNullable2(candidate.p90, baseline.p90),
660
+ varianceDelta: subtractNullable2(candidate.variance ?? null, baseline.variance ?? null),
661
+ change: classifyChange(baseline.mean, candidate.mean, direction),
662
+ varianceChange: classifyChange(baseline.variance ?? null, candidate.variance ?? null, "lower")
663
+ };
664
+ }
665
+ function normalizeMetricFilter(metric) {
666
+ if (metric === void 0) return void 0;
667
+ const trimmed = metric.trim();
668
+ if (trimmed.startsWith("trackedMetrics.")) return trimmed.slice("trackedMetrics.".length);
669
+ if (trimmed.startsWith("behavioralMetrics.")) return trimmed.slice("behavioralMetrics.".length);
670
+ return trimmed;
671
+ }
672
+ function trackedMetricDirection(metric) {
673
+ return metric === "cacheReadTokens" ? "higher" : "lower";
674
+ }
675
+ function zeroDistribution(observations) {
676
+ return {
677
+ observations,
678
+ measured: observations,
679
+ unmeasured: 0,
680
+ mean: 0,
681
+ min: 0,
682
+ max: 0,
683
+ p90: 0,
684
+ variance: 0,
685
+ standardDeviation: 0,
686
+ sources: ["ledger"]
687
+ };
688
+ }
689
+ function subtractNullable2(left, right) {
690
+ return left === null || right === null ? null : left - right;
691
+ }
692
+ function formatMetric(value) {
693
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(2);
694
+ }
695
+ function formatSignedMetric(value) {
696
+ if (value === null) return "null";
697
+ const formatted = formatMetric(value);
698
+ return value > 0 ? `+${formatted}` : formatted;
699
+ }
95
700
 
96
701
  // src/domains/eval/compare/gates.ts
97
702
  init_esm_shims();
@@ -102,12 +707,34 @@ var import_yaml = __toESM(require_dist(), 1);
102
707
  import { readFileSync } from "node:fs";
103
708
  function loadThresholds(path) {
104
709
  const parsed = (0, import_yaml.parse)(readFileSync(path, "utf8"));
105
- if (isRecord(parsed) && Array.isArray(parsed.fail)) return { fail: parsed.fail };
106
- if (isRecord(parsed) && isRecord(parsed.thresholds) && Array.isArray(parsed.thresholds.fail)) {
107
- return { fail: parsed.thresholds.fail };
710
+ const root = isRecord(parsed) && isRecord(parsed.thresholds) ? parsed.thresholds : parsed;
711
+ if (isRecord(root) && (Array.isArray(root.fail) || Array.isArray(root.informational))) {
712
+ return {
713
+ fail: parseAssertions(root.fail, `${path}.fail`),
714
+ informational: parseAssertions(root.informational, `${path}.informational`)
715
+ };
108
716
  }
109
717
  throw new Error(`invalid thresholds file: ${path}`);
110
718
  }
719
+ function parseAssertions(value, source) {
720
+ if (value === void 0) return [];
721
+ if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
722
+ return value.map((entry, index) => {
723
+ if (!isRecord(entry)) throw new Error(`${source}[${index}]: expected object`);
724
+ if (typeof entry.metric !== "string" || entry.metric.length === 0) {
725
+ throw new Error(`${source}[${index}].metric: expected non-empty string`);
726
+ }
727
+ if (!isOp(entry.op)) throw new Error(`${source}[${index}].op: expected lt, lte, gt, gte, eq, or neq`);
728
+ if (!isScalar(entry.value)) throw new Error(`${source}[${index}].value: expected scalar`);
729
+ return { metric: entry.metric, op: entry.op, value: entry.value };
730
+ });
731
+ }
732
+ function isOp(value) {
733
+ return value === "lt" || value === "lte" || value === "gt" || value === "gte" || value === "eq" || value === "neq";
734
+ }
735
+ function isScalar(value) {
736
+ return typeof value === "number" && Number.isFinite(value) || typeof value === "string" || typeof value === "boolean";
737
+ }
111
738
  function resolveMetricAssertion(assertion, metrics, artifact) {
112
739
  const actual = metricValue(assertion.metric, metrics, artifact);
113
740
  return { actual, unresolved: actual === null, holds: comparisonHolds(assertion, actual) };
@@ -152,21 +779,26 @@ function isRecord(value) {
152
779
 
153
780
  // src/domains/eval/compare/gates.ts
154
781
  function evaluateGate(artifact, thresholds) {
155
- const failures = [];
156
- for (const assertion of thresholds.fail) {
782
+ const failures = evaluateAssertions(artifact, thresholds.fail);
783
+ const informational = evaluateAssertions(artifact, thresholds.informational ?? []);
784
+ return { pass: failures.length === 0, failures, informational };
785
+ }
786
+ function evaluateAssertions(artifact, assertions) {
787
+ const findings = [];
788
+ for (const assertion of assertions) {
157
789
  const whole = resolveMetricAssertion(assertion, {}, artifact);
158
790
  if (!whole.unresolved) {
159
- if (whole.holds) failures.push({ assertion, actual: whole.actual, unresolved: false });
791
+ if (whole.holds) findings.push({ assertion, actual: whole.actual, unresolved: false });
160
792
  continue;
161
793
  }
162
794
  if (artifact.results.length === 0) {
163
- failures.push({ assertion, actual: null, unresolved: true });
795
+ findings.push({ assertion, actual: null, unresolved: true });
164
796
  continue;
165
797
  }
166
798
  for (const result of artifact.results) {
167
799
  const perRun = resolveMetricAssertion(assertion, result.metrics);
168
800
  if (!perRun.unresolved && !perRun.holds) continue;
169
- failures.push({
801
+ findings.push({
170
802
  assertion,
171
803
  actual: perRun.actual,
172
804
  unresolved: perRun.unresolved,
@@ -175,7 +807,7 @@ function evaluateGate(artifact, thresholds) {
175
807
  });
176
808
  }
177
809
  }
178
- return { pass: failures.length === 0, failures };
810
+ return findings;
179
811
  }
180
812
  function renderGateFailure(failure) {
181
813
  const run = failure.taskId === void 0 ? "" : ` [${failure.taskId}#${failure.repeatIndex ?? 0}]`;
@@ -184,6 +816,121 @@ function renderGateFailure(failure) {
184
816
  ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(failure.actual)}
185
817
  `;
186
818
  }
819
+ function renderInformationalBudget(finding) {
820
+ const run = finding.taskId === void 0 ? "" : ` [${finding.taskId}#${finding.repeatIndex ?? 0}]`;
821
+ const { metric, op, value } = finding.assertion;
822
+ return finding.unresolved ? ` ${metric}${run}: unmeasured informational budget
823
+ ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(finding.actual)}
824
+ `;
825
+ }
826
+
827
+ // src/domains/eval/reports/comparison.ts
828
+ init_esm_shims();
829
+ function renderEvalComparisonReportV1(summary, format2) {
830
+ if (format2 === "json") return `${JSON.stringify(summary, null, 2)}
831
+ `;
832
+ if (format2 === "md") return renderMarkdown(summary);
833
+ if (format2 === "junit") return renderJunit(summary);
834
+ return renderEvalComparisonV4(summary);
835
+ }
836
+ function renderMarkdown(summary) {
837
+ const rows = summary.behavioralMetrics.map(
838
+ (row) => `| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`
839
+ );
840
+ const scenarioRows = summary.scenarioReports.map(
841
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
842
+ );
843
+ const roleRows = summary.roleReports.map(
844
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
845
+ );
846
+ return [
847
+ `# Eval comparison ${summary.baselineEvalId} \u2192 ${summary.candidateEvalId}`,
848
+ "",
849
+ `Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
850
+ ...summary.hardGate.failures.map(
851
+ (failure) => `- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`
852
+ ),
853
+ ...summary.envelopeMismatches.map(
854
+ (mismatch) => `- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`
855
+ ),
856
+ ...summary.affectedCorpusResults.map(
857
+ (result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`
858
+ ),
859
+ `Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
860
+ `Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
861
+ `Wall-time delta ms: ${summary.wallTimeDelta}`,
862
+ "",
863
+ "| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
864
+ "|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
865
+ ...rows,
866
+ "",
867
+ "## Per-scenario baseline/candidate report",
868
+ "",
869
+ "| Scenario | Metric changes | Variance changes |",
870
+ "|---|---|---|",
871
+ ...scenarioRows,
872
+ "",
873
+ "## Per-role baseline/candidate report",
874
+ "",
875
+ "| Role | Metric changes | Variance changes |",
876
+ "|---|---|---|",
877
+ ...roleRows,
878
+ ""
879
+ ].join("\n");
880
+ }
881
+ function renderJunit(summary) {
882
+ const failures = new Set(
883
+ summary.hardGate.failures.map(
884
+ (failure) => JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric])
885
+ )
886
+ );
887
+ const represented = /* @__PURE__ */ new Set();
888
+ const cases = summary.behavioralMetrics.map((row) => {
889
+ const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
890
+ const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
891
+ represented.add(key);
892
+ const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
893
+ return failures.has(key) ? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>` : ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
894
+ });
895
+ for (const failure of summary.hardGate.failures) {
896
+ const key = JSON.stringify([
897
+ failure.scenarioId,
898
+ failure.role,
899
+ failure.target.id,
900
+ failure.target.model,
901
+ failure.metric
902
+ ]);
903
+ if (represented.has(key)) continue;
904
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
905
+ cases.push(
906
+ ` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`
907
+ );
908
+ }
909
+ for (const failure of summary.hardGate.envelopeFailures) {
910
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
911
+ cases.push(
912
+ ` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`
913
+ );
914
+ }
915
+ return [
916
+ `<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
917
+ ...cases,
918
+ "</testsuite>",
919
+ ""
920
+ ].join("\n");
921
+ }
922
+ function changeCounts2(counts) {
923
+ return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
924
+ }
925
+ function format(value) {
926
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
927
+ }
928
+ function cell(value) {
929
+ return value.replaceAll("|", "\\|");
930
+ }
931
+ function escapeXml(value) {
932
+ return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
933
+ }
187
934
 
188
935
  // src/domains/eval/reports/json.ts
189
936
  init_esm_shims();
@@ -195,21 +942,35 @@ function renderEvalJsonReportV4(artifact) {
195
942
  // src/domains/eval/reports/junit.ts
196
943
  init_esm_shims();
197
944
  function renderEvalJunitReportV4(artifact) {
945
+ let failures = 0;
946
+ let skipped = 0;
198
947
  const cases = artifact.results.map((result) => {
199
- const name = escapeXml(
948
+ const name = escapeXml2(
200
949
  `${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`
201
950
  );
202
- if (result.pass) return ` <testcase name="${name}" />`;
203
- return ` <testcase name="${name}"><failure message="${escapeXml(result.failureClass ?? "failed")}" /></testcase>`;
951
+ if (!result.pass) {
952
+ failures += 1;
953
+ return ` <testcase name="${name}"><failure message="${escapeXml2(result.failureClass ?? "failed")}" /></testcase>`;
954
+ }
955
+ const outcome = result.behavioral?.outcome;
956
+ if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
957
+ failures += 1;
958
+ return ` <testcase name="${name}"><failure message="${escapeXml2(outcome)}" /></testcase>`;
959
+ }
960
+ if (outcome === "unknown" || outcome === "unmeasured") {
961
+ skipped += 1;
962
+ return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml2(outcome)}" /></testcase>`;
963
+ }
964
+ return ` <testcase name="${name}" />`;
204
965
  }).join("\n");
205
966
  return [
206
- `<testsuite name="${escapeXml(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${artifact.summary.failed}">`,
967
+ `<testsuite name="${escapeXml2(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
207
968
  cases,
208
969
  "</testsuite>",
209
970
  ""
210
971
  ].join("\n");
211
972
  }
212
- function escapeXml(value) {
973
+ function escapeXml2(value) {
213
974
  return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
214
975
  }
215
976
 
@@ -223,10 +984,10 @@ function renderEvalMarkdownReportV4(artifact) {
223
984
  `Target: ${artifact.matrix.target}`,
224
985
  `Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
225
986
  "",
226
- "| Task | Target | Model | Repeat | Pass | Failure |",
227
- "|---|---|---|---:|---|---|",
987
+ "| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
988
+ "|---|---|---|---|---:|---|---|---|",
228
989
  ...artifact.results.map(
229
- (result) => `| ${result.taskId} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.failureClass ?? ""} |`
990
+ (result) => `| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`
230
991
  ),
231
992
  ""
232
993
  ];
@@ -251,6 +1012,12 @@ function renderEvalSweJsonlReportV4(artifact) {
251
1012
  init_esm_shims();
252
1013
  function renderEvalTextReportV4(artifact) {
253
1014
  const tokens = artifact.summary.tokens;
1015
+ const behavioral = artifact.results.flatMap(
1016
+ (result) => result.behavioral === void 0 ? [] : [result.behavioral.outcome]
1017
+ );
1018
+ const behavioralSummary = behavioral.length === 0 ? [] : [
1019
+ `behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`
1020
+ ];
254
1021
  return [
255
1022
  `eval: ${artifact.evalId}`,
256
1023
  `suite: ${artifact.suite.id}`,
@@ -265,9 +1032,13 @@ function renderEvalTextReportV4(artifact) {
265
1032
  // reported next to how many runs it actually covers.
266
1033
  !tokens.measured ? `tokens total: unmeasured (0 of ${tokens.runs} runs reported usage)` : tokens.measuredRuns === tokens.runs ? `tokens total: ${tokens.total}` : `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
267
1034
  `wall time ms: ${artifact.summary.wallTimeMs}`,
1035
+ ...behavioralSummary,
268
1036
  ""
269
1037
  ].join("\n");
270
1038
  }
1039
+ function count(values, wanted) {
1040
+ return values.filter((value) => value === wanted).length;
1041
+ }
271
1042
 
272
1043
  // src/domains/eval/suites/load.ts
273
1044
  init_esm_shims();
@@ -278,12 +1049,6 @@ import { dirname, resolve } from "node:path";
278
1049
 
279
1050
  // src/domains/eval/schema/validate.ts
280
1051
  init_esm_shims();
281
-
282
- // src/domains/eval/schema/suite.ts
283
- init_esm_shims();
284
- var EVAL_SUITE_V2_VERSION = 2;
285
-
286
- // src/domains/eval/schema/validate.ts
287
1052
  var RUNNER_KINDS = /* @__PURE__ */ new Set(["clio-run", "context-index", "context-init", "external-command"]);
288
1053
  var WORKSPACE_KINDS = /* @__PURE__ */ new Set(["local", "git", "temp-copy"]);
289
1054
  var OPS = /* @__PURE__ */ new Set(["lt", "lte", "gt", "gte", "eq", "neq"]);
@@ -351,12 +1116,26 @@ function readMatrix(value, path, issues) {
351
1116
  ];
352
1117
  });
353
1118
  if (repeats === null || targets.length === 0) return null;
1119
+ let dimensions;
1120
+ if (value.dimensions !== void 0) {
1121
+ try {
1122
+ dimensions = parseEvalExecutionMatrixDimensionsV1(value.dimensions, `${path}.dimensions`);
1123
+ } catch (error) {
1124
+ issues.push({ path: `${path}.dimensions`, message: error instanceof Error ? error.message : String(error) });
1125
+ return null;
1126
+ }
1127
+ }
354
1128
  const maxCostUsd = value.maxCostUsd;
355
1129
  if (maxCostUsd !== void 0 && (typeof maxCostUsd !== "number" || !Number.isFinite(maxCostUsd) || maxCostUsd < 0)) {
356
1130
  issues.push({ path: `${path}.maxCostUsd`, message: "expected non-negative number" });
357
1131
  return null;
358
1132
  }
359
- return { targets, repeats, ...maxCostUsd === void 0 ? {} : { maxCostUsd } };
1133
+ return {
1134
+ targets,
1135
+ repeats,
1136
+ ...dimensions === void 0 ? {} : { dimensions },
1137
+ ...maxCostUsd === void 0 ? {} : { maxCostUsd }
1138
+ };
360
1139
  }
361
1140
  function readTasks(value, path, issues) {
362
1141
  if (!Array.isArray(value) || value.length === 0) {
@@ -383,10 +1162,12 @@ function readTask(value, path, issues) {
383
1162
  const workspace = readWorkspace(value.workspace, `${path}.workspace`, issues);
384
1163
  const runner = readRunner(value.runner, `${path}.runner`, issues);
385
1164
  const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
1165
+ const behavioral = readBehavioral(value.behavioral, `${path}.behavioral`, issues);
386
1166
  if (id === null || workspace === null || runner === null || timeoutMs === null) return null;
387
1167
  return {
388
1168
  id,
389
1169
  tags: readOptionalStringArray(value, "tags", `${path}.tags`, issues),
1170
+ ...behavioral === void 0 ? {} : { behavioral },
390
1171
  workspace,
391
1172
  runner,
392
1173
  verify: readVerify(value.verify, `${path}.verify`, issues),
@@ -394,6 +1175,15 @@ function readTask(value, path, issues) {
394
1175
  timeoutMs
395
1176
  };
396
1177
  }
1178
+ function readBehavioral(value, path, issues) {
1179
+ if (value === void 0) return void 0;
1180
+ try {
1181
+ return parseEvalBehaviorScenarioV1(value, path);
1182
+ } catch (error) {
1183
+ issues.push({ path, message: error instanceof Error ? error.message : String(error) });
1184
+ return void 0;
1185
+ }
1186
+ }
397
1187
  function readWorkspace(value, path, issues) {
398
1188
  if (!isRecord2(value)) {
399
1189
  issues.push({ path, message: "expected object" });
@@ -430,6 +1220,10 @@ function readRunner(value, path, issues) {
430
1220
  }
431
1221
  const prompt = optionalString(value, "prompt");
432
1222
  const agent = optionalString(value, "agent");
1223
+ const autonomy = optionalString(value, "autonomy");
1224
+ if (autonomy !== void 0 && !["read-only", "suggest", "auto-edit", "full-auto"].includes(autonomy)) {
1225
+ issues.push({ path: `${path}.autonomy`, message: "expected read-only, suggest, auto-edit, or full-auto" });
1226
+ }
433
1227
  if (agent !== void 0 && kind !== "clio-run") {
434
1228
  issues.push({ path: `${path}.agent`, message: "agent is only valid on the clio-run runner" });
435
1229
  }
@@ -437,6 +1231,7 @@ function readRunner(value, path, issues) {
437
1231
  return {
438
1232
  kind,
439
1233
  ...prompt === void 0 ? {} : { prompt },
1234
+ ...autonomy === void 0 ? {} : { autonomy },
440
1235
  ...agent === void 0 ? {} : { agent },
441
1236
  ...command === void 0 ? {} : { command },
442
1237
  commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
@@ -463,7 +1258,22 @@ function readMetrics(value, path, issues) {
463
1258
  issues.push({ path, message: "expected object" });
464
1259
  return { collect: [] };
465
1260
  }
466
- return { collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues) };
1261
+ const observation = value.readObservation;
1262
+ let readObservation;
1263
+ if (observation !== void 0) {
1264
+ if (!isRecord2(observation)) {
1265
+ issues.push({ path: `${path}.readObservation`, message: "expected object" });
1266
+ } else {
1267
+ readObservation = {
1268
+ allowedPaths: readOptionalStringArray(observation, "allowedPaths", `${path}.readObservation.allowedPaths`, issues),
1269
+ decoyPaths: readOptionalStringArray(observation, "decoyPaths", `${path}.readObservation.decoyPaths`, issues)
1270
+ };
1271
+ }
1272
+ }
1273
+ return {
1274
+ collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues),
1275
+ ...readObservation === void 0 ? {} : { readObservation }
1276
+ };
467
1277
  }
468
1278
  function readThresholds(value, path, issues) {
469
1279
  if (value === void 0) return void 0;
@@ -471,7 +1281,10 @@ function readThresholds(value, path, issues) {
471
1281
  issues.push({ path, message: "expected object" });
472
1282
  return void 0;
473
1283
  }
474
- return { fail: readAssertions(value.fail, `${path}.fail`, issues) };
1284
+ return {
1285
+ fail: readAssertions(value.fail, `${path}.fail`, issues),
1286
+ informational: readAssertions(value.informational, `${path}.informational`, issues)
1287
+ };
475
1288
  }
476
1289
  function readAssertions(value, path, issues) {
477
1290
  if (value === void 0) return [];
@@ -620,7 +1433,8 @@ function resolveSuiteForRun(suite, options) {
620
1433
  ...suite,
621
1434
  matrix: {
622
1435
  ...suite.matrix,
623
- targets
1436
+ targets,
1437
+ ...options.trials === void 0 ? {} : { repeats: options.trials }
624
1438
  }
625
1439
  };
626
1440
  }
@@ -646,9 +1460,166 @@ function resolveTargets(targets, options) {
646
1460
 
647
1461
  // src/domains/eval/suites/run.ts
648
1462
  init_esm_shims();
649
- import { mkdtemp as mkdtemp3, rm as rm3 } from "node:fs/promises";
1463
+ import { mkdtemp as mkdtemp3, rm as rm3, writeFile } from "node:fs/promises";
650
1464
  import { tmpdir as tmpdir3 } from "node:os";
651
- import { resolve as resolve6 } from "node:path";
1465
+ import { resolve as resolve7 } from "node:path";
1466
+
1467
+ // src/domains/eval/execution-provenance.ts
1468
+ init_esm_shims();
1469
+ import { createHash as createHash2 } from "node:crypto";
1470
+ function buildEvalExecutionEnvelopeV1(input) {
1471
+ const scenario = input.task.behavioral;
1472
+ if (scenario === void 0) throw new Error(`behavioral task ${input.task.id} has no behavioral scenario`);
1473
+ const manifest = input.ledger.promptManifests.at(-1) ?? null;
1474
+ const contextSnapshot = input.ledger.contextSnapshots.at(-1) ?? null;
1475
+ const recipe = recipeIdentity(
1476
+ input,
1477
+ scenario.execution.subject.kind === "worker" ? scenario.execution.subject.role : null
1478
+ );
1479
+ const policy = policyIdentity(input.cwd, input.receipt, input.observation);
1480
+ const autonomy = input.receipt?.autonomyEnforcement?.autonomy ?? input.observation?.autonomy ?? input.task.runner.autonomy ?? null;
1481
+ const promptFragments = promptFragmentIdentities(manifest, recipe, autonomy);
1482
+ const compositionHash = input.receipt?.staticCompositionHash ?? input.observation?.compositionHash ?? manifest?.systemPromptHash ?? contextSnapshot?.promptHash ?? null;
1483
+ const projectContext = projectContextIdentity(input, manifest, promptFragments);
1484
+ return {
1485
+ schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
1486
+ prompt: { fragments: promptFragments, compositionHash },
1487
+ recipe: recipe === null ? null : { id: recipe.id, version: recipe.version, contentHash: recipe.contentHash },
1488
+ target: input.receipt?.targetId ?? input.observation?.target ?? input.target.id,
1489
+ wireModel: input.receipt?.wireModelId ?? input.observation?.wireModel ?? contextSnapshot?.modelId ?? input.target.model ?? null,
1490
+ runtime: input.receipt?.runtimeId ?? input.observation?.runtime ?? contextSnapshot?.runtimeId ?? null,
1491
+ thinkingLevel: input.receipt?.runtimeResolution?.effectiveThinkingLevel ?? input.observation?.thinkingLevel ?? manifest?.thinkingLevel ?? input.target.thinking ?? null,
1492
+ toolSignature: input.receipt?.toolSignature ?? input.observation?.toolSignature ?? contextSnapshot?.toolSignature ?? null,
1493
+ autonomy,
1494
+ policyHashes: policy,
1495
+ projectContext,
1496
+ corpus: { ...scenario.corpus }
1497
+ };
1498
+ }
1499
+ function recipeIdentity(input, role) {
1500
+ const id = input.receipt?.agentId ?? input.task.runner.agent ?? role;
1501
+ if (id === null || input.cwd === null) return null;
1502
+ try {
1503
+ const recipe = discoverAgentRecipes(input.cwd).find((entry) => entry.id === id);
1504
+ if (recipe === void 0) return null;
1505
+ return {
1506
+ id: recipe.id,
1507
+ version: recipe.version,
1508
+ contentHash: agentSpecFingerprint(normalizeAgentSpec(recipe)),
1509
+ personaHash: sha256(recipe.body)
1510
+ };
1511
+ } catch {
1512
+ return null;
1513
+ }
1514
+ }
1515
+ function promptFragmentIdentities(manifest, recipe, autonomy) {
1516
+ let versions = /* @__PURE__ */ new Map();
1517
+ try {
1518
+ versions = new Map([...loadFragments().byId.values()].map((fragment) => [fragment.id, fragment.version]));
1519
+ } catch {
1520
+ }
1521
+ if (manifest !== null) {
1522
+ return manifest.fragments.map((fragment) => ({
1523
+ id: fragment.id,
1524
+ version: versions.get(fragment.id) ?? "unversioned",
1525
+ contentHash: fragment.contentHash
1526
+ })).sort((left, right) => left.id.localeCompare(right.id));
1527
+ }
1528
+ if (recipe === null) return [];
1529
+ const selected = ["identity.clio-worker", "operating.contract", "operating.worker"];
1530
+ if (autonomy !== null) selected.push(`safety.${autonomy}`);
1531
+ const fragments = [];
1532
+ try {
1533
+ const table = loadFragments();
1534
+ for (const id of selected) {
1535
+ const fragment = table.byId.get(id);
1536
+ if (fragment !== void 0) {
1537
+ fragments.push({ id, version: fragment.version, contentHash: fragment.contentHash });
1538
+ }
1539
+ }
1540
+ } catch {
1541
+ }
1542
+ fragments.push({ id: `persona.${recipe.id}`, version: recipe.version, contentHash: recipe.personaHash });
1543
+ return fragments.sort((left, right) => left.id.localeCompare(right.id));
1544
+ }
1545
+ function policyIdentity(cwd, receipt, observation) {
1546
+ const sealed = receipt?.reproducibility?.safetyPolicy;
1547
+ if (sealed !== void 0) return { rulePack: sealed.rulePackHash, project: sealed.projectPolicyHash };
1548
+ if (observation !== void 0) return { ...observation.policyHashes };
1549
+ if (cwd === null) return { rulePack: null, project: null };
1550
+ try {
1551
+ const metadata = createSafetyPolicyEngine({ cwd }).metadata();
1552
+ return { rulePack: metadata.rulePackHash, project: metadata.projectPolicyHash };
1553
+ } catch {
1554
+ return { rulePack: null, project: null };
1555
+ }
1556
+ }
1557
+ function projectContextIdentity(input, manifest, fragments) {
1558
+ const receipt = input.receipt;
1559
+ if (receipt?.projectContext !== void 0) {
1560
+ const sections = [...receipt.projectContext.sections ?? []].sort();
1561
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1562
+ return {
1563
+ kind: "worker",
1564
+ tier: receipt.projectContext.tier,
1565
+ contentHash: hasContentBearingContext ? receipt.projectContext.contentHash ?? null : null,
1566
+ chars: hasContentBearingContext ? receipt.projectContext.chars ?? null : null,
1567
+ sections,
1568
+ rulesApplied: [...receipt.rulesApplied ?? []].sort(),
1569
+ operatorProfileApplied: receipt.operatorProfileApplied ?? null
1570
+ };
1571
+ }
1572
+ const observed = input.observation?.projectContext;
1573
+ if (observed !== void 0 && observed !== null) {
1574
+ const sections = [...observed.sections].sort();
1575
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1576
+ return {
1577
+ kind: "worker",
1578
+ tier: observed.tier,
1579
+ contentHash: hasContentBearingContext ? observed.contentHash : null,
1580
+ chars: hasContentBearingContext ? observed.chars : null,
1581
+ sections,
1582
+ rulesApplied: [...observed.rulesApplied].sort(),
1583
+ operatorProfileApplied: observed.operatorProfileApplied
1584
+ };
1585
+ }
1586
+ if (manifest !== null) {
1587
+ const contextFragments = fragments.filter((fragment) => fragment.id.startsWith("context."));
1588
+ const preload = manifest.projectPreload;
1589
+ const identity = {
1590
+ preload,
1591
+ fragments: contextFragments.map((fragment) => [fragment.id, fragment.contentHash])
1592
+ };
1593
+ return {
1594
+ kind: "session",
1595
+ tier: preload?.mode ?? null,
1596
+ contentHash: sha256(stableJson2(identity)),
1597
+ chars: preload?.chars ?? null,
1598
+ sections: contextFragments.map((fragment) => fragment.id).sort(),
1599
+ rulesApplied: [],
1600
+ operatorProfileApplied: contextFragments.some((fragment) => fragment.id === "context.operator-profile")
1601
+ };
1602
+ }
1603
+ return {
1604
+ kind: "none",
1605
+ tier: null,
1606
+ contentHash: null,
1607
+ chars: null,
1608
+ sections: [],
1609
+ rulesApplied: [],
1610
+ operatorProfileApplied: null
1611
+ };
1612
+ }
1613
+ function stableJson2(value) {
1614
+ if (Array.isArray(value)) return `[${value.map(stableJson2).join(",")}]`;
1615
+ if (typeof value === "object" && value !== null) {
1616
+ return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson2(entry)}`).join(",")}}`;
1617
+ }
1618
+ return JSON.stringify(value);
1619
+ }
1620
+ function sha256(value) {
1621
+ return createHash2("sha256").update(value, "utf8").digest("hex");
1622
+ }
652
1623
 
653
1624
  // src/domains/eval/metrics/context.ts
654
1625
  init_esm_shims();
@@ -1295,19 +2266,317 @@ function zeroToolCallMetrics() {
1295
2266
  };
1296
2267
  }
1297
2268
 
2269
+ // src/domains/eval/metrics/tracked.ts
2270
+ init_esm_shims();
2271
+ import { readFile as readFile2 } from "node:fs/promises";
2272
+ import { dirname as dirname2, join as join2 } from "node:path";
2273
+ async function readEvalLedgerSnapshot(stateDir) {
2274
+ const refs = await listSessionLedgerRefs(stateDir);
2275
+ const entries = [];
2276
+ const compiledPromptHashes = [];
2277
+ const promptManifests = [];
2278
+ const contextSnapshots = [];
2279
+ for (const ref of refs) {
2280
+ try {
2281
+ const raw = await readFile2(ref.path, "utf8");
2282
+ entries.push(...parseSessionEntries(raw, ref.path).entries);
2283
+ } catch {
2284
+ }
2285
+ try {
2286
+ const manifest = await readFile2(join2(dirname2(ref.path), "prompt-manifest.jsonl"), "utf8");
2287
+ for (const line of manifest.split(/\r?\n/u)) {
2288
+ const record = parseJsonRecord2(line);
2289
+ if (record === null) continue;
2290
+ const hash = record?.systemPromptHash;
2291
+ if (typeof hash !== "string" || !/^[a-f0-9]{64}$/u.test(hash)) continue;
2292
+ compiledPromptHashes.push(hash);
2293
+ const observation = promptManifestObservation(record);
2294
+ if (observation !== null) promptManifests.push(observation);
2295
+ }
2296
+ } catch {
2297
+ }
2298
+ try {
2299
+ const snapshots = await readFile2(join2(dirname2(ref.path), "context-snapshots.jsonl"), "utf8");
2300
+ for (const line of snapshots.split(/\r?\n/u)) {
2301
+ const record = parseJsonRecord2(line);
2302
+ if (record === null) continue;
2303
+ contextSnapshots.push({
2304
+ runtimeId: nullableString(record.runtimeId),
2305
+ modelId: nullableString(record.modelId),
2306
+ promptHash: nullableDigest(record.promptHash),
2307
+ toolSignature: nullableDigest(record.toolSignature)
2308
+ });
2309
+ }
2310
+ } catch {
2311
+ }
2312
+ }
2313
+ return { entries, compiledPromptHashes: [...new Set(compiledPromptHashes)], promptManifests, contextSnapshots };
2314
+ }
2315
+ function buildEvalTrackedMetrics(input) {
2316
+ const calls = assistantCalls(input.ledgerEntries);
2317
+ const compactionEntries = input.ledgerEntries.filter((entry) => entry.kind === "compactionSummary");
2318
+ const compactionUsage = compactionEntries.flatMap((entry) => {
2319
+ if (entry.kind !== "compactionSummary" || !isRecord5(entry.usage)) return [];
2320
+ return [entry.usage];
2321
+ });
2322
+ const modelCallReadings = calls.map(() => ledgerReading(1));
2323
+ for (const entry of compactionEntries) {
2324
+ if (entry.kind !== "compactionSummary") continue;
2325
+ const apiCalls = isRecord5(entry.usage) ? nonNegativeNumber(entry.usage.apiCalls) : null;
2326
+ modelCallReadings.push(apiCalls === null ? estimatedReading(1) : ledgerReading(apiCalls));
2327
+ }
2328
+ const uncachedReadings = calls.map(uncachedPrefillForCall);
2329
+ const cacheReadings = calls.map(cacheReadForCall);
2330
+ const generatedReadings = calls.map(generatedForCall);
2331
+ for (const usage of compactionUsage) {
2332
+ uncachedReadings.push(readingFromUsage(usage, "input"));
2333
+ cacheReadings.push(readingFromUsage(usage, "cacheRead"));
2334
+ generatedReadings.push(readingFromUsage(usage, "output"));
2335
+ }
2336
+ const reasoning = reasoningMetric(input.receipt, calls, compactionUsage);
2337
+ const receiptToolMetrics = input.receipt === null ? null : evalHarnessMetricsFromReceipt(input.receipt);
2338
+ const ledgerToolCalls = input.ledgerEntries.filter(
2339
+ (entry) => entry.kind === "message" && entry.role === "tool_call"
2340
+ ).length;
2341
+ const ledgerToolErrors = input.ledgerEntries.filter((entry) => {
2342
+ if (entry.kind !== "message" || entry.role !== "tool_result" || !isRecord5(entry.payload)) return false;
2343
+ return entry.payload.isError === true || entry.payload.outcome === "error";
2344
+ }).length;
2345
+ const receiptToolErrors = input.receipt?.toolStats.reduce((sum2, stat) => sum2 + finiteNonNegative(stat.errors), 0);
2346
+ const expectedColdReasons = expectedColdReasonMetrics(calls);
2347
+ return {
2348
+ modelCalls: sumReadings(modelCallReadings, "ledger"),
2349
+ uncachedPrefillTokens: sumReadings(uncachedReadings, "estimated"),
2350
+ cacheReadTokens: sumReadings(cacheReadings, "estimated"),
2351
+ generatedTokens: sumReadings(generatedReadings, "estimated"),
2352
+ reasoningTokens: reasoning,
2353
+ toolCalls: receiptToolMetrics === null ? { value: ledgerToolCalls, source: "ledger" } : { value: receiptToolMetrics.toolCalls, source: "receipt" },
2354
+ toolErrors: receiptToolErrors === void 0 ? { value: ledgerToolErrors, source: "ledger" } : { value: receiptToolErrors, source: "receipt" },
2355
+ ttftMsFirstCall: firstCallTtft(calls),
2356
+ wallClockMs: wallClockMetric(input.receipt, input.fallbackWallClockMs),
2357
+ contextTokensAtEnd: contextTokensAtEnd(calls, compactionEntries),
2358
+ compactions: { value: compactionEntries.length, source: "ledger" },
2359
+ expectedColdReasons
2360
+ };
2361
+ }
2362
+ function emptyEvalTrackedMetrics(source = "estimated") {
2363
+ const zero = () => ({ value: 0, source });
2364
+ return {
2365
+ modelCalls: zero(),
2366
+ uncachedPrefillTokens: zero(),
2367
+ cacheReadTokens: zero(),
2368
+ generatedTokens: zero(),
2369
+ reasoningTokens: { value: null, source },
2370
+ toolCalls: zero(),
2371
+ toolErrors: zero(),
2372
+ ttftMsFirstCall: zero(),
2373
+ wallClockMs: zero(),
2374
+ contextTokensAtEnd: zero(),
2375
+ compactions: zero(),
2376
+ expectedColdReasons: {}
2377
+ };
2378
+ }
2379
+ function assistantCalls(entries) {
2380
+ return entries.flatMap((entry) => {
2381
+ if (entry.kind !== "message" || entry.role !== "assistant" || !isRecord5(entry.payload)) return [];
2382
+ const promptCache = recordField(entry.payload, "promptCache");
2383
+ const timing = recordField(entry.payload, "timing");
2384
+ const usage = recordField(entry.payload, "usage");
2385
+ if (promptCache === null && timing === null && usage === null) return [];
2386
+ return [
2387
+ {
2388
+ payload: entry.payload,
2389
+ promptCache,
2390
+ backend: promptCache === null ? null : recordField(promptCache, "backend"),
2391
+ timing,
2392
+ usage
2393
+ }
2394
+ ];
2395
+ });
2396
+ }
2397
+ function uncachedPrefillForCall(call) {
2398
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2399
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2400
+ if (promptTokens !== null && cachedTokens !== null && cachedTokens <= promptTokens) {
2401
+ return ledgerReading(promptTokens - cachedTokens);
2402
+ }
2403
+ const piInput = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2404
+ if (piInput !== null) return ledgerReading(piInput);
2405
+ const legacyInput = call.usage === null ? null : nonNegativeNumber(call.usage.input);
2406
+ return legacyInput === null ? estimatedReading(0) : estimatedReading(legacyInput);
2407
+ }
2408
+ function cacheReadForCall(call) {
2409
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2410
+ if (cachedTokens !== null) return ledgerReading(cachedTokens);
2411
+ const piCacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2412
+ if (piCacheRead !== null) return ledgerReading(piCacheRead);
2413
+ const legacyCacheRead = call.usage === null ? null : nonNegativeNumber(call.usage.cacheRead);
2414
+ return legacyCacheRead === null ? estimatedReading(0) : estimatedReading(legacyCacheRead);
2415
+ }
2416
+ function generatedForCall(call) {
2417
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2418
+ if (predictedTokens !== null) return ledgerReading(predictedTokens);
2419
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2420
+ return output === null ? estimatedReading(0) : ledgerReading(output);
2421
+ }
2422
+ function readingFromUsage(usage, field) {
2423
+ const value = nonNegativeNumber(usage[field]);
2424
+ return value === null ? estimatedReading(0) : ledgerReading(value);
2425
+ }
2426
+ function reasoningMetric(receipt, calls, compactionUsage) {
2427
+ if (receipt !== null && typeof receipt.reasoningTokenCount === "number") {
2428
+ return { value: finiteNonNegative(receipt.reasoningTokenCount), source: "receipt" };
2429
+ }
2430
+ let total = 0;
2431
+ let measured = false;
2432
+ for (const call of calls) {
2433
+ const value = extractReasoningTokens(call.usage);
2434
+ if (value === null) continue;
2435
+ measured = true;
2436
+ total += finiteNonNegative(value);
2437
+ }
2438
+ for (const usage of compactionUsage) {
2439
+ const value = nonNegativeNumber(usage.reasoning);
2440
+ if (value === null) continue;
2441
+ measured = true;
2442
+ total += value;
2443
+ }
2444
+ return measured ? { value: total, source: "ledger" } : { value: null, source: "estimated" };
2445
+ }
2446
+ function firstCallTtft(calls) {
2447
+ const first = calls[0];
2448
+ const value = first?.timing === null || first?.timing === void 0 ? null : nonNegativeNumber(first.timing.ttftMs);
2449
+ return value === null ? { value: 0, source: "estimated" } : { value, source: "ledger" };
2450
+ }
2451
+ function wallClockMetric(receipt, fallback) {
2452
+ if (receipt !== null) {
2453
+ const started = Date.parse(receipt.startedAt);
2454
+ const ended = Date.parse(receipt.endedAt);
2455
+ if (Number.isFinite(started) && Number.isFinite(ended) && ended >= started) {
2456
+ return { value: ended - started, source: "receipt" };
2457
+ }
2458
+ }
2459
+ return { value: finiteNonNegative(fallback), source: "estimated" };
2460
+ }
2461
+ function contextTokensAtEnd(calls, compactions) {
2462
+ const lastCall = calls.at(-1);
2463
+ if (lastCall !== void 0) {
2464
+ const lastReading = contextForCall(lastCall);
2465
+ if (lastReading !== null && lastReading > 0) return { value: lastReading, source: "ledger" };
2466
+ if (lastReading === 0) {
2467
+ for (const call of [...calls.slice(0, -1)].reverse()) {
2468
+ const reading = contextForCall(call);
2469
+ if (reading !== null && reading > 0) return { value: reading, source: "ledger" };
2470
+ }
2471
+ }
2472
+ }
2473
+ const lastCompaction = compactions.at(-1);
2474
+ if (lastCompaction?.kind === "compactionSummary") {
2475
+ const tokensAfter = nonNegativeNumber(lastCompaction.tokensAfter);
2476
+ if (tokensAfter !== null) return { value: tokensAfter, source: "ledger" };
2477
+ }
2478
+ return { value: 0, source: "estimated" };
2479
+ }
2480
+ function contextForCall(call) {
2481
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2482
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2483
+ if (promptTokens !== null && predictedTokens !== null) return promptTokens + predictedTokens;
2484
+ const input = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2485
+ const cacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2486
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2487
+ return input === null || cacheRead === null || output === null ? null : input + cacheRead + output;
2488
+ }
2489
+ function expectedColdReasonMetrics(calls) {
2490
+ const counts = /* @__PURE__ */ new Map();
2491
+ for (const call of calls) {
2492
+ const reasons = call.promptCache?.expectedColdReasons;
2493
+ if (!Array.isArray(reasons)) continue;
2494
+ const unique = new Set(reasons.filter((reason) => typeof reason === "string" && reason.length > 0));
2495
+ for (const reason of unique) counts.set(reason, (counts.get(reason) ?? 0) + 1);
2496
+ }
2497
+ return Object.fromEntries(
2498
+ [...counts.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([reason, value]) => [reason, { value, source: "ledger" }])
2499
+ );
2500
+ }
2501
+ function sumReadings(readings, emptySource) {
2502
+ if (readings.length === 0) return { value: 0, source: emptySource };
2503
+ return {
2504
+ value: readings.reduce((sum2, reading) => sum2 + reading.value, 0),
2505
+ source: readings.some((reading) => reading.source === "estimated") ? "estimated" : "ledger"
2506
+ };
2507
+ }
2508
+ function ledgerReading(value) {
2509
+ return { value: finiteNonNegative(value), source: "ledger" };
2510
+ }
2511
+ function estimatedReading(value) {
2512
+ return { value: finiteNonNegative(value), source: "estimated" };
2513
+ }
2514
+ function finiteNonNegative(value) {
2515
+ return Number.isFinite(value) && value >= 0 ? value : 0;
2516
+ }
2517
+ function nonNegativeNumber(value) {
2518
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2519
+ }
2520
+ function recordField(record, field) {
2521
+ return isRecord5(record[field]) ? record[field] : null;
2522
+ }
2523
+ function parseJsonRecord2(line) {
2524
+ if (line.trim().length === 0) return null;
2525
+ try {
2526
+ const parsed = JSON.parse(line);
2527
+ return isRecord5(parsed) ? parsed : null;
2528
+ } catch {
2529
+ return null;
2530
+ }
2531
+ }
2532
+ function promptManifestObservation(record) {
2533
+ const systemPromptHash = nullableDigest(record.systemPromptHash);
2534
+ if (systemPromptHash === null || !Array.isArray(record.fragments)) return null;
2535
+ const fragments = record.fragments.flatMap((entry) => {
2536
+ if (!isRecord5(entry) || typeof entry.id !== "string") return [];
2537
+ const contentHash = nullableDigest(entry.contentHash);
2538
+ return contentHash === null ? [] : [{ id: entry.id, contentHash }];
2539
+ });
2540
+ const preload = record.projectPreload;
2541
+ const projectPreload = preload === null ? null : isRecord5(preload) && (preload.mode === "full" || preload.mode === "synopsis" || preload.mode === "none") && typeof preload.chars === "number" && Number.isInteger(preload.chars) && typeof preload.lines === "number" && Number.isInteger(preload.lines) && typeof preload.nearLimit === "boolean" && typeof preload.label === "string" ? {
2542
+ mode: preload.mode,
2543
+ chars: preload.chars,
2544
+ lines: preload.lines,
2545
+ reason: nullableString(preload.reason),
2546
+ nearLimit: preload.nearLimit,
2547
+ label: preload.label
2548
+ } : null;
2549
+ return {
2550
+ systemPromptHash,
2551
+ thinkingLevel: nullableString(record.thinkingLevel),
2552
+ projectPreload,
2553
+ fragments
2554
+ };
2555
+ }
2556
+ function nullableString(value) {
2557
+ return typeof value === "string" && value.length > 0 ? value : null;
2558
+ }
2559
+ function nullableDigest(value) {
2560
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
2561
+ }
2562
+ function isRecord5(value) {
2563
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2564
+ }
2565
+
1298
2566
  // src/domains/eval/runners/clio-run.ts
1299
2567
  init_esm_shims();
2568
+ import { isAbsolute, relative, resolve as resolve2, sep } from "node:path";
1300
2569
 
1301
2570
  // src/domains/eval/metrics/evidence.ts
1302
2571
  init_esm_shims();
1303
2572
  import { readFileSync as readFileSync3 } from "node:fs";
1304
- import { join as join2 } from "node:path";
2573
+ import { join as join3 } from "node:path";
1305
2574
  function dispatchScopeMetrics(receipt) {
1306
2575
  const scope = receipt.pathScope;
1307
2576
  if (scope === void 0) return {};
1308
2577
  const entries = [...scope.workingContextPaths, ...scope.writeBoundaries];
1309
2578
  const evidence = entries.flatMap((entry) => entry.evidence);
1310
- const count = (source) => evidence.filter((entry) => entry.source === source).length;
2579
+ const count2 = (source) => evidence.filter((entry) => entry.source === source).length;
1311
2580
  return {
1312
2581
  "dispatch.scope.mode": scope.mode,
1313
2582
  "dispatch.scope.inferredPathCount": entries.filter(
@@ -1316,9 +2585,9 @@ function dispatchScopeMetrics(receipt) {
1316
2585
  "dispatch.scope.derivedPathCount": entries.filter(
1317
2586
  (entry) => entry.evidence.some((item) => item.provenance === "derived")
1318
2587
  ).length,
1319
- "dispatch.scope.source.task": count("task"),
1320
- "dispatch.scope.source.briefing": count("briefing"),
1321
- "dispatch.scope.source.writeRoots": count("writeRoots")
2588
+ "dispatch.scope.source.task": count2("task"),
2589
+ "dispatch.scope.source.briefing": count2("briefing"),
2590
+ "dispatch.scope.source.writeRoots": count2("writeRoots")
1322
2591
  };
1323
2592
  }
1324
2593
  function receiptFromRunJsonStdout(stdout) {
@@ -1365,7 +2634,7 @@ function evidenceTrustMetrics(receipt, envelope) {
1365
2634
  }
1366
2635
  function readRunEnvelopeForReceipt(receipt, stateDir) {
1367
2636
  try {
1368
- const parsed = JSON.parse(readFileSync3(join2(stateDir, "runs.json"), "utf8"));
2637
+ const parsed = JSON.parse(readFileSync3(join3(stateDir, "runs.json"), "utf8"));
1369
2638
  if (!Array.isArray(parsed)) return null;
1370
2639
  const row = parsed.find(
1371
2640
  (entry) => typeof entry === "object" && entry !== null && entry.id === receipt.runId
@@ -1396,10 +2665,10 @@ function createTokenUsageFold() {
1396
2665
  } catch {
1397
2666
  return;
1398
2667
  }
1399
- if (!isRecord5(event) || event.type !== "message_end") return;
1400
- const message = isRecord5(event.message) ? event.message : void 0;
2668
+ if (!isRecord6(event) || event.type !== "message_end") return;
2669
+ const message = isRecord6(event.message) ? event.message : void 0;
1401
2670
  if (message === void 0 || message.role !== "assistant") return;
1402
- const usage = isRecord5(message.usage) ? message.usage : void 0;
2671
+ const usage = isRecord6(message.usage) ? message.usage : void 0;
1403
2672
  if (usage === void 0) return;
1404
2673
  measured = true;
1405
2674
  const input = numberField2(usage, "input");
@@ -1412,7 +2681,7 @@ function createTokenUsageFold() {
1412
2681
  tokens.cacheRead += cacheRead;
1413
2682
  tokens.cacheWrite += cacheWrite;
1414
2683
  tokens.total += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
1415
- if (isRecord5(usage.cost)) costUsd += numberField2(usage.cost, "total");
2684
+ if (isRecord6(usage.cost)) costUsd += numberField2(usage.cost, "total");
1416
2685
  };
1417
2686
  return {
1418
2687
  push(chunk) {
@@ -1462,14 +2731,115 @@ function numberField2(record, field) {
1462
2731
  const value = record[field];
1463
2732
  return typeof value === "number" && Number.isFinite(value) ? value : 0;
1464
2733
  }
1465
- function isRecord5(value) {
2734
+ function isRecord6(value) {
1466
2735
  return typeof value === "object" && value !== null && !Array.isArray(value);
1467
2736
  }
1468
2737
 
1469
2738
  // src/domains/eval/runners/external-command.ts
1470
2739
  init_esm_shims();
1471
2740
  import { spawn } from "node:child_process";
2741
+ import { performance as performance2 } from "node:perf_hooks";
2742
+
2743
+ // src/domains/eval/metrics/call-ledger-stream.ts
2744
+ init_esm_shims();
1472
2745
  import { performance } from "node:perf_hooks";
2746
+ function createEvalCallLedgerFold(now = () => performance.now()) {
2747
+ const entries = [];
2748
+ let pending = "";
2749
+ let activeStartedAt = null;
2750
+ let activeFirstOutputAt = null;
2751
+ const consume = (line) => {
2752
+ const event = parseRecord(line);
2753
+ if (event === null) return;
2754
+ if (event.type === "message_start" && isAssistantMessage(event.message)) {
2755
+ activeStartedAt = now();
2756
+ activeFirstOutputAt = null;
2757
+ return;
2758
+ }
2759
+ if (event.type === "message_update" && activeStartedAt !== null && activeFirstOutputAt === null) {
2760
+ activeFirstOutputAt = now();
2761
+ return;
2762
+ }
2763
+ if (event.type !== "message_end" || !isAssistantMessage(event.message)) return;
2764
+ const message = event.message;
2765
+ const usage = isRecord7(message.usage) ? message.usage : null;
2766
+ if (usage === null) {
2767
+ activeStartedAt = null;
2768
+ activeFirstOutputAt = null;
2769
+ return;
2770
+ }
2771
+ const endedAt = now();
2772
+ const promptCache = {
2773
+ input: nonNegativeNumber2(usage.input) ?? 0,
2774
+ cacheRead: nonNegativeNumber2(usage.cacheRead) ?? 0,
2775
+ cacheWrite: nonNegativeNumber2(usage.cacheWrite) ?? 0,
2776
+ backendVerdict: "unknown"
2777
+ };
2778
+ if (isRecord7(message.backendTimings)) promptCache.backend = structuredClone(message.backendTimings);
2779
+ const previous = entries.at(-1);
2780
+ entries.push({
2781
+ kind: "message",
2782
+ role: "assistant",
2783
+ turnId: `eval-call-${entries.length + 1}`,
2784
+ parentTurnId: previous?.turnId ?? null,
2785
+ timestamp: messageTimestamp(message.timestamp),
2786
+ payload: {
2787
+ promptCache,
2788
+ timing: {
2789
+ ttftMs: activeStartedAt === null ? null : Math.round(Math.max(0, (activeFirstOutputAt ?? endedAt) - activeStartedAt)),
2790
+ apiMs: activeStartedAt === null ? 0 : Math.round(Math.max(0, endedAt - activeStartedAt))
2791
+ },
2792
+ usage: structuredClone(usage)
2793
+ }
2794
+ });
2795
+ activeStartedAt = null;
2796
+ activeFirstOutputAt = null;
2797
+ };
2798
+ return {
2799
+ push(chunk) {
2800
+ pending += chunk;
2801
+ for (; ; ) {
2802
+ const newline = pending.indexOf("\n");
2803
+ if (newline === -1) break;
2804
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
2805
+ pending = pending.slice(newline + 1);
2806
+ }
2807
+ },
2808
+ entries() {
2809
+ if (pending.length > 0) {
2810
+ consume(pending.replace(/\r$/u, ""));
2811
+ pending = "";
2812
+ }
2813
+ return structuredClone(entries);
2814
+ }
2815
+ };
2816
+ }
2817
+ function isAssistantMessage(value) {
2818
+ return isRecord7(value) && value.role === "assistant";
2819
+ }
2820
+ function messageTimestamp(value) {
2821
+ const milliseconds = nonNegativeNumber2(value);
2822
+ if (milliseconds === null) return (/* @__PURE__ */ new Date(0)).toISOString();
2823
+ const date = new Date(milliseconds);
2824
+ return Number.isFinite(date.getTime()) ? date.toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
2825
+ }
2826
+ function nonNegativeNumber2(value) {
2827
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2828
+ }
2829
+ function parseRecord(line) {
2830
+ if (line.trim().length === 0) return null;
2831
+ try {
2832
+ const value = JSON.parse(line);
2833
+ return isRecord7(value) ? value : null;
2834
+ } catch {
2835
+ return null;
2836
+ }
2837
+ }
2838
+ function isRecord7(value) {
2839
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2840
+ }
2841
+
2842
+ // src/domains/eval/runners/external-command.ts
1473
2843
  var OUTPUT_LIMIT = 2e5;
1474
2844
  var OUTPUT_HEAD_LIMIT = 2e4;
1475
2845
  var OUTPUT_TRUNCATION_MARKER = "\n[output middle truncated; tail preserved]\n";
@@ -1484,6 +2854,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1484
2854
  let usage = UNMEASURED_TOKEN_USAGE;
1485
2855
  let streamInvariants = EMPTY_STREAM_INVARIANTS;
1486
2856
  let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
2857
+ const ledgerEntries = [];
1487
2858
  for (const command of commands) {
1488
2859
  const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
1489
2860
  stdout = appendLimited(stdout, result.stdout);
@@ -1492,6 +2863,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1492
2863
  usage = addTokenStreamUsage(usage, result.usage);
1493
2864
  streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
1494
2865
  fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
2866
+ ledgerEntries.push(...result.ledgerEntries);
1495
2867
  if (result.exitCode !== 0) {
1496
2868
  return {
1497
2869
  assignmentId: null,
@@ -1507,7 +2879,8 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1507
2879
  ...fleetLoopMetricEntries(fleetLoops),
1508
2880
  "verifier.exitCode": result.exitCode
1509
2881
  },
1510
- artifacts: {}
2882
+ artifacts: {},
2883
+ ledgerEntries
1511
2884
  };
1512
2885
  }
1513
2886
  }
@@ -1525,18 +2898,20 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1525
2898
  ...fleetLoopMetricEntries(fleetLoops),
1526
2899
  "verifier.exitCode": 0
1527
2900
  },
1528
- artifacts: {}
2901
+ artifacts: {},
2902
+ ledgerEntries
1529
2903
  };
1530
2904
  }
1531
2905
  function runShellCommand(command, cwd, timeoutMs, env) {
1532
- const started = performance.now();
1533
- return new Promise((resolve8) => {
2906
+ const started = performance2.now();
2907
+ return new Promise((resolve9) => {
1534
2908
  let stdout = "";
1535
2909
  let stderr = "";
1536
2910
  const metricCapture = createJsonlMetricCapture();
1537
2911
  const usageFold = createTokenUsageFold();
1538
2912
  const streamFold = createStreamInvariantFold();
1539
2913
  const fleetLoopFold = createFleetLoopFold();
2914
+ const callLedgerFold = createEvalCallLedgerFold();
1540
2915
  let timedOut = false;
1541
2916
  let settled = false;
1542
2917
  const child = spawn(command, {
@@ -1555,6 +2930,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1555
2930
  usageFold.push(chunk);
1556
2931
  streamFold.push(chunk);
1557
2932
  fleetLoopFold.push(chunk);
2933
+ callLedgerFold.push(chunk);
1558
2934
  });
1559
2935
  child.stderr.on("data", (chunk) => {
1560
2936
  stderr = appendLimited(stderr, chunk);
@@ -1568,7 +2944,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1568
2944
  if (settled) return;
1569
2945
  settled = true;
1570
2946
  clearTimeout(timer);
1571
- resolve8({
2947
+ resolve9({
1572
2948
  command,
1573
2949
  exitCode,
1574
2950
  stdout,
@@ -1576,8 +2952,9 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1576
2952
  usage: usageFold.usage(),
1577
2953
  streamInvariants: streamFold.invariants(),
1578
2954
  fleetLoops: fleetLoopFold.observation(),
2955
+ ledgerEntries: callLedgerFold.entries(),
1579
2956
  stderr,
1580
- wallTimeMs: Math.round(performance.now() - started),
2957
+ wallTimeMs: Math.round(performance2.now() - started),
1581
2958
  timedOut
1582
2959
  });
1583
2960
  };
@@ -1609,7 +2986,7 @@ function createJsonlMetricCapture() {
1609
2986
  } catch {
1610
2987
  return;
1611
2988
  }
1612
- if (!isRecord6(parsed)) return;
2989
+ if (!isRecord8(parsed)) return;
1613
2990
  const compact = compactMetricEvent(parsed);
1614
2991
  if (compact === null) return;
1615
2992
  const encoded = JSON.stringify(compact);
@@ -1663,7 +3040,7 @@ function compactMetricEvent(event) {
1663
3040
  type,
1664
3041
  ...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
1665
3042
  toolName,
1666
- ...toolName === "dispatch" && isRecord6(event.args) ? { args: event.args } : toolName === "code_nav" && isRecord6(event.args) ? { args: { mode: event.args.mode } } : {}
3043
+ ...toolName === "dispatch" && isRecord8(event.args) ? { args: event.args } : toolName === "read" && isRecord8(event.args) ? { args: boundedReadArgs(event.args) } : toolName === "code_nav" && isRecord8(event.args) ? { args: { mode: event.args.mode } } : {}
1667
3044
  };
1668
3045
  }
1669
3046
  if (type === "tool_execution_end") {
@@ -1675,7 +3052,7 @@ function compactMetricEvent(event) {
1675
3052
  ...stringField(event, "outcome") !== void 0 ? { outcome: stringField(event, "outcome") } : {}
1676
3053
  };
1677
3054
  }
1678
- if (type !== "clio_tool_finish" || !isRecord6(event.payload)) return null;
3055
+ if (type !== "clio_tool_finish" || !isRecord8(event.payload)) return null;
1679
3056
  return {
1680
3057
  type,
1681
3058
  payload: {
@@ -1685,7 +3062,14 @@ function compactMetricEvent(event) {
1685
3062
  }
1686
3063
  };
1687
3064
  }
1688
- function isRecord6(value) {
3065
+ function boundedReadArgs(args) {
3066
+ for (const field of ["path", "filePath", "file_path"]) {
3067
+ const value = args[field];
3068
+ if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4096) };
3069
+ }
3070
+ return {};
3071
+ }
3072
+ function isRecord8(value) {
1689
3073
  return typeof value === "object" && value !== null && !Array.isArray(value);
1690
3074
  }
1691
3075
  function stringField(record, field) {
@@ -1694,7 +3078,7 @@ function stringField(record, field) {
1694
3078
  }
1695
3079
 
1696
3080
  // src/domains/eval/runners/clio-run.ts
1697
- async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
3081
+ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env, readObservation) {
1698
3082
  const prompt = runner.prompt ?? "";
1699
3083
  const args = [
1700
3084
  shellQuote(clioEntry),
@@ -1705,12 +3089,14 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1705
3089
  shellQuote(target.id),
1706
3090
  ...target.model === void 0 ? [] : ["--model", shellQuote(target.model)],
1707
3091
  ...target.thinking === void 0 ? [] : ["--thinking", shellQuote(target.thinking)],
3092
+ ...runner.autonomy === void 0 ? [] : ["--autonomy", runner.autonomy],
1708
3093
  shellQuote(prompt)
1709
3094
  ];
1710
3095
  const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
1711
3096
  const tokens = result.usage;
1712
3097
  const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
1713
3098
  const tools = toolCallMetricsFromJsonl(toolMetricStream);
3099
+ const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
1714
3100
  const receipt = receiptFromRunJsonStdout(result.stdout);
1715
3101
  const envelope = receipt === null ? null : readRunEnvelopeForReceipt(receipt, env?.CLIO_CODER_STATE_DIR ?? clioStateDir());
1716
3102
  return {
@@ -1729,6 +3115,7 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1729
3115
  "tools.totalCalls": tools.totalCalls,
1730
3116
  "tools.failed": tools.failed,
1731
3117
  "tools.blocked": tools.blocked,
3118
+ ...behavioralTools,
1732
3119
  "verifier.exitCode": result.exitCode,
1733
3120
  ...receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope }),
1734
3121
  ...receipt === null ? {} : { "evidence.qualityLabel": receipt.quality.typedValidations.length > 0 ? "measured" : "unmeasured" }
@@ -1736,8 +3123,11 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1736
3123
  artifacts: {
1737
3124
  stdout: result.stdout,
1738
3125
  stderr: result.stderr,
3126
+ callLedger: JSON.stringify(result.ledgerEntries),
1739
3127
  ...receipt === null ? {} : { receipt: JSON.stringify(receipt) }
1740
- }
3128
+ },
3129
+ receipt,
3130
+ ledgerEntries: result.ledgerEntries
1741
3131
  };
1742
3132
  }
1743
3133
  function toolCallMetricsFromJsonl(stdout) {
@@ -1750,7 +3140,7 @@ function toolCallMetricsFromJsonl(stdout) {
1750
3140
  let event;
1751
3141
  try {
1752
3142
  const parsed = JSON.parse(line);
1753
- if (!isRecord7(parsed)) continue;
3143
+ if (!isRecord9(parsed)) continue;
1754
3144
  event = parsed;
1755
3145
  } catch {
1756
3146
  continue;
@@ -1764,7 +3154,7 @@ function toolCallMetricsFromJsonl(stdout) {
1764
3154
  recordToolOutcome(executionEnds, toolOutcome(event) ?? (event.isError === true ? "error" : "ok"));
1765
3155
  continue;
1766
3156
  }
1767
- if (event.type !== "clio_tool_finish" || !isRecord7(event.payload)) continue;
3157
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
1768
3158
  const outcome = toolOutcome(event.payload);
1769
3159
  if (outcome === void 0) continue;
1770
3160
  const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId");
@@ -1776,6 +3166,95 @@ function toolCallMetricsFromJsonl(stdout) {
1776
3166
  }
1777
3167
  return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
1778
3168
  }
3169
+ function toolBehaviorMetricEntriesFromJsonl(stdout, cwd, readObservation) {
3170
+ const starts = /* @__PURE__ */ new Map();
3171
+ const readPaths = /* @__PURE__ */ new Set();
3172
+ const executionEnds = [];
3173
+ const canonicalFinishes = [];
3174
+ const seenExecution = /* @__PURE__ */ new Set();
3175
+ const seenCanonical = /* @__PURE__ */ new Set();
3176
+ for (const line of stdout.split(/\r?\n/)) {
3177
+ if (line.trim().length === 0) continue;
3178
+ let event;
3179
+ try {
3180
+ const parsed = JSON.parse(line);
3181
+ if (!isRecord9(parsed)) continue;
3182
+ event = parsed;
3183
+ } catch {
3184
+ continue;
3185
+ }
3186
+ if (event.type === "tool_execution_start") {
3187
+ const callId2 = stringField2(event, "toolCallId");
3188
+ const tool2 = stringField2(event, "toolName");
3189
+ if (callId2 === void 0 || tool2 === void 0) continue;
3190
+ const path = tool2 === "read" && isRecord9(event.args) ? toolPath(event.args) : null;
3191
+ starts.set(callId2, { tool: tool2, path });
3192
+ if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
3193
+ continue;
3194
+ }
3195
+ if (event.type === "tool_execution_end") {
3196
+ const callId2 = stringField2(event, "toolCallId") ?? null;
3197
+ if (callId2 !== null && seenExecution.has(callId2)) continue;
3198
+ if (callId2 !== null) seenExecution.add(callId2);
3199
+ const tool2 = stringField2(event, "toolName") ?? (callId2 === null ? void 0 : starts.get(callId2)?.tool);
3200
+ if (tool2 === void 0) continue;
3201
+ executionEnds.push({ callId: callId2, tool: tool2, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
3202
+ continue;
3203
+ }
3204
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
3205
+ const outcome = toolOutcome(event.payload);
3206
+ const tool = stringField2(event.payload, "tool");
3207
+ if (outcome === void 0 || tool === void 0) continue;
3208
+ const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId") ?? null;
3209
+ if (callId !== null && seenCanonical.has(callId)) continue;
3210
+ if (callId !== null) seenCanonical.add(callId);
3211
+ canonicalFinishes.push({ callId, tool, outcome });
3212
+ }
3213
+ const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
3214
+ const calls = /* @__PURE__ */ new Map();
3215
+ const blocked = /* @__PURE__ */ new Map();
3216
+ for (const terminal of terminals) {
3217
+ const tool = metricToolName(terminal.tool);
3218
+ calls.set(tool, (calls.get(tool) ?? 0) + 1);
3219
+ if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
3220
+ }
3221
+ const namedTools = /* @__PURE__ */ new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
3222
+ const entries = { "tools.read.distinctPaths": readPaths.size };
3223
+ for (const tool of [...namedTools].sort()) {
3224
+ entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
3225
+ entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
3226
+ }
3227
+ if (readObservation !== void 0) {
3228
+ const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
3229
+ const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
3230
+ entries["tools.read.outsideAllowed"] = [...readPaths].filter(
3231
+ (path) => !allowed.some((root) => pathWithin(path, root))
3232
+ ).length;
3233
+ entries["tools.read.decoyHits"] = [...readPaths].filter(
3234
+ (path) => decoys.some((root) => pathWithin(path, root))
3235
+ ).length;
3236
+ }
3237
+ return entries;
3238
+ }
3239
+ function toolPath(args) {
3240
+ for (const field of ["path", "filePath", "file_path"]) {
3241
+ const value = args[field];
3242
+ if (typeof value === "string" && value.length > 0 && value.length <= 4096) return value;
3243
+ }
3244
+ return null;
3245
+ }
3246
+ function normalizeObservedPath(cwd, path) {
3247
+ const absolute = resolve2(cwd, path);
3248
+ const local = relative(cwd, absolute);
3249
+ return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".").split(sep).join("/");
3250
+ }
3251
+ function pathWithin(path, root) {
3252
+ if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
3253
+ return path === root || path.startsWith(`${root}/`);
3254
+ }
3255
+ function metricToolName(tool) {
3256
+ return tool.toLowerCase().replaceAll(/[^a-z0-9_-]/gu, "_").slice(0, 64) || "unknown";
3257
+ }
1779
3258
  function recordToolOutcome(metrics, outcome) {
1780
3259
  metrics.totalCalls += 1;
1781
3260
  if (outcome === "error") metrics.failed += 1;
@@ -1789,7 +3268,7 @@ function stringField2(record, field) {
1789
3268
  const value = record[field];
1790
3269
  return typeof value === "string" && value.length > 0 ? value : void 0;
1791
3270
  }
1792
- function isRecord7(value) {
3271
+ function isRecord9(value) {
1793
3272
  return typeof value === "object" && value !== null && !Array.isArray(value);
1794
3273
  }
1795
3274
 
@@ -1840,7 +3319,7 @@ function parseContextIndexOutput(stdout) {
1840
3319
  // src/domains/eval/runners/context-init.ts
1841
3320
  init_esm_shims();
1842
3321
  import { existsSync as existsSync2, statSync } from "node:fs";
1843
- import { join as join3 } from "node:path";
3322
+ import { join as join4 } from "node:path";
1844
3323
  async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
1845
3324
  const extraArgs = runner.args ?? [];
1846
3325
  const command = [
@@ -1858,22 +3337,22 @@ async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, e
1858
3337
  ].map(shellQuote).join(" ");
1859
3338
  const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
1860
3339
  const payload = parseInitPayload(result.stdout);
1861
- const candidateGeneration = recordField(payload, "generation");
3340
+ const candidateGeneration = recordField2(payload, "generation");
1862
3341
  const generation = isValidGenerationPayload(payload, candidateGeneration) ? candidateGeneration : null;
1863
3342
  const routeError = generation ? generationRouteError(generation, target) : null;
1864
3343
  const payloadError = generation ? routeError : "context-init runner did not receive a valid JSON generation result";
1865
3344
  const exitCode = result.exitCode === 0 && payloadError ? 1 : result.exitCode;
1866
3345
  const stderr = payloadError ? `${result.stderr}${result.stderr.endsWith("\n") || result.stderr.length === 0 ? "" : "\n"}${payloadError}
1867
3346
  ` : result.stderr;
1868
- const run = recordField(generation, "run");
1869
- const tokens = recordField(run, "tokens");
3347
+ const run = recordField2(generation, "run");
3348
+ const tokens = recordField2(run, "tokens");
1870
3349
  const effectiveTarget = stringField3(run, "targetId");
1871
3350
  const effectiveModel = stringField3(run, "wireModelId");
1872
3351
  const effectiveRuntime = stringField3(run, "runtimeId");
1873
3352
  const effectiveRuntimeKind = stringField3(run, "runtimeKind");
1874
3353
  const effectiveThinking = stringField3(run, "thinkingLevel");
1875
3354
  const structuredOutputMode = stringField3(run, "structuredOutputMode");
1876
- const clioMdPath = join3(cwd, "CLIO-CODER.md");
3355
+ const clioMdPath = join4(cwd, "CLIO-CODER.md");
1877
3356
  const clioMdBytes = existsSync2(clioMdPath) ? statSync(clioMdPath).size : 0;
1878
3357
  return {
1879
3358
  assignmentId: null,
@@ -1927,7 +3406,7 @@ function isNonnegativeFiniteNumber(value) {
1927
3406
  return typeof value === "number" && Number.isFinite(value) && value >= 0;
1928
3407
  }
1929
3408
  function isValidRunPayload(value) {
1930
- const run = recordField(value);
3409
+ const run = recordField2(value);
1931
3410
  if (!run) return false;
1932
3411
  for (const key of ["durationMs", "promptBytes", "outputBytes"]) {
1933
3412
  if (!isNonnegativeFiniteNumber(run[key])) return false;
@@ -1936,7 +3415,7 @@ function isValidRunPayload(value) {
1936
3415
  for (const key of ["toolCalls", "toolFailures", "toolBlocked"]) {
1937
3416
  if (run[key] !== void 0 && !isNonnegativeFiniteNumber(run[key])) return false;
1938
3417
  }
1939
- const tokens = recordField(run, "tokens");
3418
+ const tokens = recordField2(run, "tokens");
1940
3419
  if (run.tokens !== void 0 && !tokens) return false;
1941
3420
  if (tokens) {
1942
3421
  for (const key of ["total", "input", "output", "cacheRead", "cacheWrite", "reasoning"]) {
@@ -1962,14 +3441,14 @@ function isValidGenerationPayload(payload, generation) {
1962
3441
  }
1963
3442
  const runPresent = generation.run !== void 0;
1964
3443
  if (runPresent && !isValidRunPayload(generation.run)) return false;
1965
- const run = recordField(generation, "run");
3444
+ const run = recordField2(generation, "run");
1966
3445
  if (mode === "model" && (parserOutcome !== "parsed" || !hasReceiptIdentity(run))) return false;
1967
3446
  if ((parserOutcome === "parsed" || parserOutcome === "rejected") && !runPresent) return false;
1968
3447
  if (parserOutcome === "rejected" && !hasReceiptIdentity(run)) return false;
1969
3448
  return true;
1970
3449
  }
1971
3450
  function generationRouteError(generation, target) {
1972
- const run = recordField(generation, "run");
3451
+ const run = recordField2(generation, "run");
1973
3452
  if (!run) return null;
1974
3453
  const actualTarget = stringField3(run, "targetId");
1975
3454
  const actualModel = stringField3(run, "wireModelId");
@@ -1988,34 +3467,101 @@ function generationRouteError(generation, target) {
1988
3467
  function parseInitPayload(stdout) {
1989
3468
  try {
1990
3469
  const parsed = JSON.parse(stdout);
1991
- return recordField(parsed);
3470
+ return recordField2(parsed);
1992
3471
  } catch {
1993
3472
  return null;
1994
3473
  }
1995
3474
  }
1996
- function recordField(value, field) {
3475
+ function recordField2(value, field) {
1997
3476
  const selected = field && typeof value === "object" && value !== null && !Array.isArray(value) ? value[field] : value;
1998
3477
  return typeof selected === "object" && selected !== null && !Array.isArray(selected) ? selected : null;
1999
3478
  }
2000
3479
  function numberField3(value, field) {
2001
- const record = recordField(value);
3480
+ const record = recordField2(value);
2002
3481
  const selected = record?.[field];
2003
3482
  return typeof selected === "number" && Number.isFinite(selected) ? selected : null;
2004
3483
  }
2005
3484
  function stringField3(value, field) {
2006
- const record = recordField(value);
3485
+ const record = recordField2(value);
2007
3486
  const selected = record?.[field];
2008
3487
  return typeof selected === "string" && selected.length > 0 ? selected : null;
2009
3488
  }
2010
3489
 
3490
+ // src/domains/eval/schema/adapter.ts
3491
+ init_esm_shims();
3492
+ import { createHash as createHash3 } from "node:crypto";
3493
+ function adaptSuiteV2ResultToVerdictV1(result, trackedMetrics) {
3494
+ const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
3495
+ const outcome = result.pass ? "pass" : "fail";
3496
+ const graderExitCode = result.metrics["task.exitCode"];
3497
+ return parseEvalVerdictEnvelopeV1({
3498
+ schema: EVAL_VERDICT_SCHEMA_V1,
3499
+ scenarioId: result.taskId,
3500
+ trialIndex: result.repeatIndex,
3501
+ outcome,
3502
+ machinery,
3503
+ reason: result.pass ? null : result.failureClass ?? "result_failed",
3504
+ trackedMetrics,
3505
+ behavioral: null,
3506
+ evidence: {
3507
+ assignmentId: result.assignmentId,
3508
+ terminalReceiptDigest: result.terminalReceiptDigest,
3509
+ graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null
3510
+ }
3511
+ });
3512
+ }
3513
+ function adaptSuiteV2ResultToBehaviorV1(result, verdict, scenario) {
3514
+ const requestedFacts = new Set(
3515
+ [...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
3516
+ (rule) => `${rule.fact.source}\0${rule.fact.key}`
3517
+ )
3518
+ );
3519
+ const observedSources = /* @__PURE__ */ new Set();
3520
+ const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
3521
+ if (value === null) return [];
3522
+ const source = metricFactSource(key);
3523
+ observedSources.add(source);
3524
+ if (!requestedFacts.has(`${source}\0${key}`)) return [];
3525
+ const serialized = JSON.stringify({ source, key, value });
3526
+ const digest = createHash3("sha256").update(serialized, "utf8").digest("hex");
3527
+ const fact = {
3528
+ id: `metric-${digest.slice(0, 16)}`,
3529
+ source,
3530
+ key,
3531
+ value,
3532
+ evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1e3) }
3533
+ };
3534
+ return [fact];
3535
+ });
3536
+ const allSources = ["transcript", "tool", "receipt", "grader"];
3537
+ const unavailableSources = allSources.filter(
3538
+ (source) => !observedSources.has(source) || source === "tool" && scenario.execution.toolTarget === "none"
3539
+ );
3540
+ const behavior = judgeEvalBehaviorV1(scenario, verdict, {
3541
+ facts,
3542
+ unavailableSources,
3543
+ infrastructureFailure: verdict.machinery === "infrastructure_failure"
3544
+ });
3545
+ assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
3546
+ return behavior;
3547
+ }
3548
+ function metricFactSource(key) {
3549
+ if (key.startsWith("tools.")) return "tool";
3550
+ if (key.startsWith("task.") || key.startsWith("claims.") || key.startsWith("completion.") || key === "result.pass" || key === "verifier.exitCode")
3551
+ return "grader";
3552
+ if (key.startsWith("receipt.") || key.startsWith("evidence.") || key.startsWith("boundary.") || key.startsWith("loop.") || key.startsWith("cost."))
3553
+ return "receipt";
3554
+ return "transcript";
3555
+ }
3556
+
2011
3557
  // src/domains/eval/verifiers/command.ts
2012
3558
  init_esm_shims();
2013
- async function runCommandVerifiers(commands, cwd, timeoutMs) {
3559
+ async function runCommandVerifiers(commands, cwd, timeoutMs, env) {
2014
3560
  let stdout = "";
2015
3561
  let stderr = "";
2016
3562
  let wallTimeMs = 0;
2017
3563
  for (const command of commands) {
2018
- const result = await runShellCommand(command, cwd, timeoutMs);
3564
+ const result = await runShellCommand(command, cwd, timeoutMs, env);
2019
3565
  stdout += result.stdout;
2020
3566
  stderr += result.stderr;
2021
3567
  wallTimeMs += result.wallTimeMs;
@@ -2027,9 +3573,9 @@ async function runCommandVerifiers(commands, cwd, timeoutMs) {
2027
3573
  // src/domains/eval/verifiers/file-exists.ts
2028
3574
  init_esm_shims();
2029
3575
  import { existsSync as existsSync3 } from "node:fs";
2030
- import { resolve as resolve2 } from "node:path";
3576
+ import { resolve as resolve3 } from "node:path";
2031
3577
  function forbiddenPathHits(cwd, paths) {
2032
- return paths.filter((path) => existsSync3(resolve2(cwd, path)));
3578
+ return paths.filter((path) => existsSync3(resolve3(cwd, path)));
2033
3579
  }
2034
3580
 
2035
3581
  // src/domains/eval/verifiers/patch.ts
@@ -2051,10 +3597,10 @@ init_esm_shims();
2051
3597
  import { spawn as spawn2 } from "node:child_process";
2052
3598
  import { mkdtemp, rm } from "node:fs/promises";
2053
3599
  import { tmpdir } from "node:os";
2054
- import { resolve as resolve3 } from "node:path";
3600
+ import { resolve as resolve4 } from "node:path";
2055
3601
  async function prepareGitWorkspace(workspace) {
2056
3602
  if (workspace.url === void 0) throw new Error("git workspace requires url");
2057
- const dest = await mkdtemp(resolve3(tmpdir(), "clio-eval-git-"));
3603
+ const dest = await mkdtemp(resolve4(tmpdir(), "clio-eval-git-"));
2058
3604
  try {
2059
3605
  await runGit(["clone", "--quiet", workspace.url, dest], process.cwd());
2060
3606
  const ref = workspace.checkout ?? workspace.commit;
@@ -2089,9 +3635,9 @@ function runGit(args, cwd) {
2089
3635
  // src/domains/eval/workspaces/local.ts
2090
3636
  init_esm_shims();
2091
3637
  import { access } from "node:fs/promises";
2092
- import { resolve as resolve4 } from "node:path";
3638
+ import { resolve as resolve5 } from "node:path";
2093
3639
  async function prepareLocalWorkspace(baseDir, workspace) {
2094
- const dir = resolve4(baseDir, workspace.path ?? ".");
3640
+ const dir = resolve5(baseDir, workspace.path ?? ".");
2095
3641
  await access(dir);
2096
3642
  return { dir, cleanup: async () => {
2097
3643
  } };
@@ -2099,28 +3645,113 @@ async function prepareLocalWorkspace(baseDir, workspace) {
2099
3645
 
2100
3646
  // src/domains/eval/workspaces/temp-copy.ts
2101
3647
  init_esm_shims();
2102
- import { cp, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
3648
+ import { execFile } from "node:child_process";
3649
+ import { cp, lstat, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
2103
3650
  import { tmpdir as tmpdir2 } from "node:os";
2104
- import { relative, resolve as resolve5 } from "node:path";
2105
- async function prepareTempCopyWorkspace(baseDir, workspace) {
2106
- const source = resolve5(baseDir, workspace.path ?? ".");
2107
- const dest = await mkdtemp2(resolve5(tmpdir2(), "clio-eval-workspace-"));
2108
- const excludes = workspace.excludes ?? [];
2109
- await cp(source, dest, {
2110
- recursive: true,
2111
- filter: (path) => !isExcluded(relative(source, path), excludes)
2112
- });
2113
- return {
2114
- dir: dest,
2115
- cleanup: async () => {
2116
- await rm2(dest, { recursive: true, force: true });
2117
- }
2118
- };
3651
+ import { relative as relative2, resolve as resolve6 } from "node:path";
3652
+ import { promisify } from "node:util";
3653
+ var execFileAsync = promisify(execFile);
3654
+ var GIT_FILE_LIST_LIMIT_BYTES = 128 * 1024 * 1024;
3655
+ async function prepareTempCopyWorkspace(baseDir, workspace, options = {}) {
3656
+ const source = resolve6(baseDir, workspace.path ?? ".");
3657
+ const dest = await mkdtemp2(resolve6(options.tempRoot ?? tmpdir2(), "clio-eval-workspace-"));
3658
+ try {
3659
+ const selection = await gitCopySelection(source);
3660
+ const excludes = workspace.excludes ?? [];
3661
+ const copyWorkspace = options.copy ?? defaultCopy;
3662
+ await copyWorkspace(source, dest, {
3663
+ recursive: true,
3664
+ filter: (path) => shouldCopy(relative2(source, path), excludes, selection)
3665
+ });
3666
+ return {
3667
+ dir: dest,
3668
+ cleanup: async () => {
3669
+ await rm2(dest, { recursive: true, force: true });
3670
+ }
3671
+ };
3672
+ } catch (error) {
3673
+ await rm2(dest, { recursive: true, force: true });
3674
+ throw error;
3675
+ }
2119
3676
  }
2120
3677
  function isExcluded(rel, excludes) {
2121
3678
  const normalized = rel.replaceAll("\\", "/");
2122
3679
  return excludes.some((entry) => normalized === entry || normalized.startsWith(`${entry.replaceAll("\\", "/")}/`));
2123
3680
  }
3681
+ async function defaultCopy(source, destination, options) {
3682
+ await cp(source, destination, options);
3683
+ }
3684
+ function shouldCopy(relativePath, excludes, selection) {
3685
+ const normalized = relativePath.replaceAll("\\", "/");
3686
+ if (normalized.length === 0) return true;
3687
+ if (isExcluded(normalized, excludes)) return false;
3688
+ if (selection === null) return true;
3689
+ return selection.files.has(normalized) || selection.directories.has(normalized);
3690
+ }
3691
+ async function gitCopySelection(source) {
3692
+ let inside;
3693
+ try {
3694
+ inside = await gitOutput(source, ["rev-parse", "--is-inside-work-tree"]);
3695
+ } catch (error) {
3696
+ if (await hasGitMarker(source)) throw error;
3697
+ return null;
3698
+ }
3699
+ if (inside.trim() !== "true") return null;
3700
+ const output = await gitOutput(source, [
3701
+ "--literal-pathspecs",
3702
+ "ls-files",
3703
+ "-z",
3704
+ "--cached",
3705
+ "--others",
3706
+ "--exclude-standard",
3707
+ "--",
3708
+ "."
3709
+ ]);
3710
+ const files = /* @__PURE__ */ new Set();
3711
+ const directories = /* @__PURE__ */ new Set();
3712
+ for (const path of output.split("\0")) {
3713
+ if (path.length === 0) continue;
3714
+ const normalized = normalizeGitPath(path);
3715
+ if (normalized === null) throw new Error("git ls-files returned a path outside the eval workspace");
3716
+ files.add(normalized);
3717
+ let separator = normalized.lastIndexOf("/");
3718
+ while (separator >= 0) {
3719
+ directories.add(normalized.slice(0, separator));
3720
+ separator = normalized.lastIndexOf("/", separator - 1);
3721
+ }
3722
+ }
3723
+ return { files, directories };
3724
+ }
3725
+ async function gitOutput(cwd, args) {
3726
+ const { stdout } = await execFileAsync("git", [...args], {
3727
+ cwd,
3728
+ encoding: "utf8",
3729
+ maxBuffer: GIT_FILE_LIST_LIMIT_BYTES
3730
+ });
3731
+ return stdout;
3732
+ }
3733
+ function normalizeGitPath(path) {
3734
+ const normalized = path.replaceAll("\\", "/").replace(/^\.\//u, "");
3735
+ if (normalized.length === 0 || normalized.startsWith("/") || /^[A-Za-z]:\//u.test(normalized)) return null;
3736
+ const segments = normalized.split("/");
3737
+ if (segments.some((segment) => segment.length === 0 || segment === "." || segment === "..")) return null;
3738
+ return normalized;
3739
+ }
3740
+ async function hasGitMarker(source) {
3741
+ let current = resolve6(source);
3742
+ while (true) {
3743
+ try {
3744
+ await lstat(resolve6(current, ".git"));
3745
+ return true;
3746
+ } catch (error) {
3747
+ const code = typeof error === "object" && error !== null && "code" in error ? error.code : void 0;
3748
+ if (code !== "ENOENT" && code !== "ENOTDIR") throw error;
3749
+ }
3750
+ const parent = resolve6(current, "..");
3751
+ if (parent === current) return false;
3752
+ current = parent;
3753
+ }
3754
+ }
2124
3755
 
2125
3756
  // src/domains/eval/suites/matrix.ts
2126
3757
  init_esm_shims();
@@ -2148,28 +3779,39 @@ async function runEvalSuiteV2(loaded, options) {
2148
3779
  const started = now();
2149
3780
  const evalId = createEvalId(started, loaded.hash);
2150
3781
  const results = [];
3782
+ const servingObservations = [];
2151
3783
  const maxCostUsd = loaded.suite.matrix.maxCostUsd;
2152
3784
  let spentUsd = 0;
2153
3785
  for (const item of expandEvalMatrix(loaded.suite)) {
2154
3786
  if (maxCostUsd !== void 0 && spentUsd > maxCostUsd) {
2155
- results.push(budgetExhaustedResult(item.task.id, item.target, item.repeatIndex, spentUsd, maxCostUsd));
3787
+ results.push(budgetExhaustedResult(loaded, item.task, item.target, item.repeatIndex, spentUsd, maxCostUsd));
2156
3788
  continue;
2157
3789
  }
2158
- const result = await runMatrixItem(loaded, item.task, item.target, item.repeatIndex, options.clioEntry);
2159
- spentUsd += resultCostUsd(result);
2160
- results.push(result);
3790
+ const completed = await runMatrixItem(
3791
+ loaded,
3792
+ item.task,
3793
+ item.target,
3794
+ item.repeatIndex,
3795
+ options.clioEntry,
3796
+ options.freshWorkspaces === true,
3797
+ options.tempCopy
3798
+ );
3799
+ spentUsd += resultCostUsd(completed.result);
3800
+ results.push(completed.result);
3801
+ servingObservations.push(completed.serving);
2161
3802
  }
2162
- return buildArtifact(loaded, evalId, results, options.clioEntry);
3803
+ const serving = await evalServingConfiguration(loaded.suite.matrix.targets, servingObservations);
3804
+ return buildArtifact(loaded, evalId, results, options.clioEntry, serving);
2163
3805
  }
2164
3806
  function resultCostUsd(result) {
2165
3807
  const value = result.metrics["cost.usd"];
2166
3808
  return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0;
2167
3809
  }
2168
- function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd) {
2169
- return {
3810
+ function budgetExhaustedResult(loaded, task, target, repeatIndex, spentUsd, maxCostUsd) {
3811
+ const result = {
2170
3812
  assignmentId: null,
2171
3813
  terminalReceiptDigest: null,
2172
- taskId,
3814
+ taskId: task.id,
2173
3815
  repeatIndex,
2174
3816
  target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
2175
3817
  pass: false,
@@ -2184,21 +3826,36 @@ function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd
2184
3826
  error: `matrix cost budget exhausted: spent $${spentUsd.toFixed(4)} of max $${maxCostUsd.toFixed(4)} before this item`
2185
3827
  }
2186
3828
  };
3829
+ result.verdict = adaptSuiteV2ResultToVerdictV1(result, emptyEvalTrackedMetrics());
3830
+ attachBehavioralResult(result, task);
3831
+ attachExecutionEnvelope(result, task, target, loaded.baseDir, null, emptyLedgerSnapshot());
3832
+ return result;
2187
3833
  }
2188
- async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
3834
+ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry, freshWorkspace, tempCopy) {
2189
3835
  let workspace = null;
2190
- const stateDir = await mkdtemp3(resolve6(tmpdir3(), "clio-eval-state-"));
3836
+ let receipt = null;
3837
+ let runnerWallTimeMs = 0;
3838
+ let executionObservation;
3839
+ const stateDir = await mkdtemp3(resolve7(tempCopy?.tempRoot ?? tmpdir3(), "clio-eval-state-"));
2191
3840
  try {
2192
- workspace = await prepareWorkspace(loaded.baseDir, task);
3841
+ workspace = await prepareWorkspace(loaded.baseDir, task, freshWorkspace, tempCopy);
2193
3842
  const setup = await runCommandVerifiers(task.workspace.setup ?? [], workspace.dir, task.timeoutMs);
2194
3843
  if (!setup.pass) throw new EvalWorkspaceSetupError(setup.exitCode, setup.stderr);
2195
3844
  const runner = await runTaskRunner(task, target, workspace.dir, clioEntry, {
2196
3845
  CLIO_CODER_STATE_DIR: stateDir,
2197
3846
  CLIO_CODER_ENTRY: clioEntry
2198
3847
  });
3848
+ const runnerStdoutFile = resolve7(stateDir, "eval-runner-output.jsonl");
3849
+ await writeFile(runnerStdoutFile, runner.stdout, "utf8");
3850
+ receipt = runner.receipt ?? null;
3851
+ runnerWallTimeMs = runner.wallTimeMs;
2199
3852
  const patch = collectPatchMetrics(workspace.dir);
2200
3853
  const receiptExitCode = runner.exitCode;
2201
3854
  const journalMetrics = invariantMetrics(stateDir, receiptExitCode);
3855
+ const measurement = await measureTaskOutcome(task, workspace.dir, {
3856
+ CLIO_EVAL_RUNNER_STDOUT_FILE: runnerStdoutFile
3857
+ });
3858
+ executionObservation = measurement.executionObservation;
2202
3859
  const metrics = {
2203
3860
  ...zeroToolCallMetrics(),
2204
3861
  ...collectContextMetrics(workspace.dir),
@@ -2214,15 +3871,16 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2214
3871
  "patch.testFilesModified": patch.testFilesModified,
2215
3872
  "result.pass": runner.exitCode === 0,
2216
3873
  "result.failureClass": runner.exitCode === 0 ? null : "runner_failed",
2217
- ...await measureTaskOutcome(task, workspace.dir)
3874
+ ...measurement.metrics
2218
3875
  };
2219
3876
  const verifier = await runVerifiers(task, workspace.dir, metrics);
2220
- const pass = runner.exitCode === 0 && verifier.pass;
2221
- const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : verifier.failureClass;
3877
+ const graderFailed = metrics["task.solved"] === false;
3878
+ const pass = runner.exitCode === 0 && verifier.pass && !graderFailed;
3879
+ const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : !verifier.pass ? verifier.failureClass : "grader_failed";
2222
3880
  metrics["verifier.exitCode"] = verifier.exitCode;
2223
3881
  metrics["result.pass"] = pass;
2224
3882
  metrics["result.failureClass"] = failureClass;
2225
- return {
3883
+ const result = {
2226
3884
  assignmentId: runner.assignmentId,
2227
3885
  terminalReceiptDigest: runner.terminalReceiptDigest,
2228
3886
  taskId: task.id,
@@ -2233,13 +3891,30 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2233
3891
  metrics,
2234
3892
  artifacts: {
2235
3893
  ...runner.artifacts,
3894
+ workspace: workspace.dir,
2236
3895
  ...verifier.stdout.length > 0 ? { verifierStdout: verifier.stdout } : {},
2237
3896
  ...verifier.stderr.length > 0 ? { verifierStderr: verifier.stderr } : {}
2238
3897
  }
2239
3898
  };
3899
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3900
+ const ledgerEntries = [...snapshot.entries, ...runner.ledgerEntries ?? []];
3901
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3902
+ result,
3903
+ buildEvalTrackedMetrics({
3904
+ ledgerEntries,
3905
+ receipt: receipt ?? null,
3906
+ fallbackWallClockMs: runner.wallTimeMs
3907
+ })
3908
+ );
3909
+ attachBehavioralResult(result, task);
3910
+ attachExecutionEnvelope(result, task, target, workspace.dir, receipt ?? null, snapshot, executionObservation);
3911
+ return {
3912
+ result,
3913
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
3914
+ };
2240
3915
  } catch (error) {
2241
3916
  const failureClass = error instanceof EvalWorkspaceSetupError ? "setup_failed" : "command_error";
2242
- return {
3917
+ const result = {
2243
3918
  assignmentId: null,
2244
3919
  terminalReceiptDigest: null,
2245
3920
  taskId: task.id,
@@ -2253,13 +3928,61 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2253
3928
  "verifier.exitCode": 1,
2254
3929
  "latency.wallMs": 0
2255
3930
  },
2256
- artifacts: { error: error instanceof Error ? error.message : String(error) }
3931
+ artifacts: {
3932
+ error: error instanceof Error ? error.message : String(error),
3933
+ ...workspace === null ? {} : { workspace: workspace.dir }
3934
+ }
3935
+ };
3936
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3937
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3938
+ result,
3939
+ buildEvalTrackedMetrics({
3940
+ ledgerEntries: snapshot.entries,
3941
+ receipt: receipt ?? null,
3942
+ fallbackWallClockMs: runnerWallTimeMs
3943
+ })
3944
+ );
3945
+ attachBehavioralResult(result, task);
3946
+ attachExecutionEnvelope(
3947
+ result,
3948
+ task,
3949
+ target,
3950
+ workspace?.dir ?? loaded.baseDir,
3951
+ receipt ?? null,
3952
+ snapshot,
3953
+ executionObservation
3954
+ );
3955
+ return {
3956
+ result,
3957
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
2257
3958
  };
2258
3959
  } finally {
2259
- await workspace?.cleanup();
2260
- await rm3(stateDir, { recursive: true, force: true });
3960
+ try {
3961
+ await workspace?.cleanup();
3962
+ } finally {
3963
+ await rm3(stateDir, { recursive: true, force: true });
3964
+ }
2261
3965
  }
2262
3966
  }
3967
+ function attachBehavioralResult(result, task) {
3968
+ if (task.behavioral === void 0 || result.verdict === void 0) return;
3969
+ result.behavioral = adaptSuiteV2ResultToBehaviorV1(result, result.verdict, task.behavioral);
3970
+ result.behavioralMetrics = buildEvalBehaviorMetricsV1(result, task.behavioral.execution.subject.role);
3971
+ }
3972
+ function attachExecutionEnvelope(result, task, target, cwd, receipt, ledger, observation) {
3973
+ if (task.behavioral === void 0) return;
3974
+ result.executionEnvelope = buildEvalExecutionEnvelopeV1({
3975
+ task,
3976
+ target,
3977
+ cwd,
3978
+ receipt,
3979
+ ledger,
3980
+ ...observation === void 0 ? {} : { observation }
3981
+ });
3982
+ }
3983
+ function emptyLedgerSnapshot() {
3984
+ return { entries: [], compiledPromptHashes: [], promptManifests: [], contextSnapshots: [] };
3985
+ }
2263
3986
  function invariantMetrics(stateDir, runnerExitCode) {
2264
3987
  const journal = readRunJournal(stateDir);
2265
3988
  return {
@@ -2270,23 +3993,80 @@ function invariantMetrics(stateDir, runnerExitCode) {
2270
3993
  ...writeBoundaryInvariantMetrics(stateDir)
2271
3994
  };
2272
3995
  }
2273
- async function prepareWorkspace(baseDir, task) {
3996
+ async function prepareWorkspace(baseDir, task, freshWorkspace, tempCopy) {
3997
+ if (task.workspace.kind === "local" && freshWorkspace) {
3998
+ return prepareTempCopyWorkspace(baseDir, { ...task.workspace, kind: "temp-copy" }, tempCopy);
3999
+ }
2274
4000
  if (task.workspace.kind === "local") return prepareLocalWorkspace(baseDir, task.workspace);
2275
4001
  if (task.workspace.kind === "git") return prepareGitWorkspace(task.workspace);
2276
- return prepareTempCopyWorkspace(baseDir, task.workspace);
4002
+ return prepareTempCopyWorkspace(baseDir, task.workspace, tempCopy);
2277
4003
  }
2278
4004
  async function runTaskRunner(task, target, cwd, clioEntry, env) {
2279
4005
  if (task.runner.kind === "external-command") return runExternalCommandRunner(task.runner, cwd, task.timeoutMs, env);
2280
4006
  if (task.runner.kind === "context-index") return runContextIndexRunner(cwd, clioEntry, task.timeoutMs, target, env);
2281
4007
  if (task.runner.kind === "context-init")
2282
4008
  return runContextInitRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
2283
- return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
4009
+ return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env, task.metrics.readObservation);
2284
4010
  }
2285
- async function measureTaskOutcome(task, cwd) {
4011
+ async function measureTaskOutcome(task, cwd, env) {
2286
4012
  const commands = task.verify.measure ?? [];
2287
- if (commands.length === 0) return {};
2288
- const result = await runCommandVerifiers(commands, cwd, task.timeoutMs);
2289
- return { "task.exitCode": result.exitCode, "task.solved": result.exitCode === 0 };
4013
+ if (commands.length === 0) return { metrics: {} };
4014
+ const result = await runCommandVerifiers(commands, cwd, task.timeoutMs, env);
4015
+ const behavioral = graderBehaviorMeasurement(result.stdout);
4016
+ return {
4017
+ metrics: {
4018
+ "task.exitCode": result.exitCode,
4019
+ "task.solved": result.exitCode === 0,
4020
+ ...behavioral.metrics
4021
+ },
4022
+ ...behavioral.executionObservation === void 0 ? {} : { executionObservation: behavioral.executionObservation }
4023
+ };
4024
+ }
4025
+ function graderBehaviorMeasurement(stdout) {
4026
+ const metrics = {};
4027
+ let executionObservation;
4028
+ for (const line of stdout.split(/\r?\n/u)) {
4029
+ if (line.trim().length === 0) continue;
4030
+ let value;
4031
+ try {
4032
+ value = JSON.parse(line);
4033
+ } catch {
4034
+ continue;
4035
+ }
4036
+ if (!isRecord10(value)) continue;
4037
+ if (value.schema === "clio.eval.measure.v1" && isRecord10(value.metrics)) {
4038
+ for (const [key, metric] of Object.entries(value.metrics)) {
4039
+ if (key !== "claims.unsupported" && key !== "completion.reported") continue;
4040
+ if (typeof metric === "boolean" || typeof metric === "number" && Number.isFinite(metric)) metrics[key] = metric;
4041
+ }
4042
+ }
4043
+ if (value.schema === "clio.eval.execution-observation.v1") {
4044
+ executionObservation = parseExecutionObservation(value);
4045
+ }
4046
+ }
4047
+ return { metrics, ...executionObservation === void 0 ? {} : { executionObservation } };
4048
+ }
4049
+ function parseExecutionObservation(value) {
4050
+ const policies = isRecord10(value.policyHashes) ? value.policyHashes : {};
4051
+ const project = isRecord10(value.projectContext) ? value.projectContext : null;
4052
+ return {
4053
+ compositionHash: nullableDigest2(value.compositionHash),
4054
+ target: nullableString2(value.target),
4055
+ wireModel: nullableString2(value.wireModel),
4056
+ runtime: nullableString2(value.runtime),
4057
+ thinkingLevel: nullableString2(value.thinkingLevel),
4058
+ toolSignature: nullableDigest2(value.toolSignature),
4059
+ autonomy: nullableString2(value.autonomy),
4060
+ policyHashes: { rulePack: nullableDigest2(policies.rulePack), project: nullableDigest2(policies.project) },
4061
+ projectContext: project === null ? null : {
4062
+ tier: nullableString2(project.tier),
4063
+ contentHash: nullableDigest2(project.contentHash),
4064
+ chars: nullableNonNegativeInteger(project.chars),
4065
+ sections: stringArray(project.sections),
4066
+ rulesApplied: stringArray(project.rulesApplied),
4067
+ operatorProfileApplied: typeof project.operatorProfileApplied === "boolean" ? project.operatorProfileApplied : null
4068
+ }
4069
+ };
2290
4070
  }
2291
4071
  async function runVerifiers(task, cwd, metrics) {
2292
4072
  const commandResult = await runCommandVerifiers(task.verify.commands ?? [], cwd, task.timeoutMs);
@@ -2322,7 +4102,7 @@ async function runVerifiers(task, cwd, metrics) {
2322
4102
  }
2323
4103
  return { pass: true, exitCode: 0, failureClass: null, stdout: commandResult.stdout, stderr: commandResult.stderr };
2324
4104
  }
2325
- function buildArtifact(loaded, evalId, results, clioEntry) {
4105
+ function buildArtifact(loaded, evalId, results, clioEntry, servingConfiguration) {
2326
4106
  const passed = results.filter((result) => result.pass).length;
2327
4107
  return {
2328
4108
  version: 4,
@@ -2330,7 +4110,11 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
2330
4110
  suite: { id: loaded.suite.suite.id, hash: loaded.hash },
2331
4111
  clio: evalClioProvenance({ entry: clioEntry }),
2332
4112
  environment: evalEnvironmentProvenance(),
2333
- matrix: artifactMatrixIdentity(loaded.suite.matrix.targets),
4113
+ matrix: {
4114
+ ...artifactMatrixIdentity(loaded.suite.matrix.targets),
4115
+ ...loaded.suite.matrix.dimensions === void 0 ? {} : { dimensions: loaded.suite.matrix.dimensions }
4116
+ },
4117
+ servingConfiguration,
2334
4118
  summary: {
2335
4119
  runs: results.length,
2336
4120
  passed,
@@ -2339,26 +4123,57 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
2339
4123
  tokens: tokenAccountingFrom(results),
2340
4124
  wallTimeMs: results.reduce((sum2, result) => sum2 + wallTimeMetric(result.metrics), 0)
2341
4125
  },
4126
+ aggregates: aggregateEvalVerdicts(
4127
+ results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict])
4128
+ ),
2342
4129
  results
2343
4130
  };
2344
4131
  }
2345
4132
  function assertionMessage(assertion, actual) {
2346
4133
  return `assertion failed: ${assertion.metric} ${assertion.op} ${String(assertion.value)} (actual ${JSON.stringify(actual)})`;
2347
4134
  }
4135
+ function isRecord10(value) {
4136
+ return typeof value === "object" && value !== null && !Array.isArray(value);
4137
+ }
4138
+ function nullableString2(value) {
4139
+ return typeof value === "string" && value.length > 0 ? value : null;
4140
+ }
4141
+ function nullableDigest2(value) {
4142
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
4143
+ }
4144
+ function nullableNonNegativeInteger(value) {
4145
+ return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : null;
4146
+ }
4147
+ function stringArray(value) {
4148
+ return Array.isArray(value) ? value.filter((entry) => typeof entry === "string") : [];
4149
+ }
2348
4150
 
2349
4151
  // src/cli/eval.ts
2350
4152
  var HELP = `clio-coder eval <command>
2351
4153
 
2352
4154
  Commands:
2353
4155
  clio-coder eval validate --suite <suite.yaml>
2354
- clio-coder eval run --suite <suite.yaml> [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
2355
- clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
4156
+ clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
4157
+ clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
2356
4158
  clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
2357
- clio-coder eval compare <baselineEvalId> <candidateEvalId>
4159
+ clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
2358
4160
  clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
4161
+ clio-coder eval inventory --json
4162
+
4163
+ inventory is the fixed machine-readable read a GUI host may run. Unlike report
4164
+ and compare it names no eval id, so the process it starts cannot be steered to a
4165
+ different report or a wider window. It carries each stored report's identity,
4166
+ provenance, serving facts, accounting, and per-scenario outcomes, and none of
4167
+ the runner attachments a report holds.
2359
4168
  `;
2360
4169
  function parseEvalArgs(args) {
2361
- const parsed = { repeat: 1, compareIds: [], format: "text", help: false };
4170
+ const parsed = {
4171
+ repeat: 1,
4172
+ compareIds: [],
4173
+ format: "text",
4174
+ allowConfigDrift: false,
4175
+ help: false
4176
+ };
2362
4177
  for (let index = 0; index < args.length; index += 1) {
2363
4178
  const arg = args[index];
2364
4179
  if (arg === void 0) continue;
@@ -2417,6 +4232,11 @@ function parseEvalArgs(args) {
2417
4232
  index += 1;
2418
4233
  continue;
2419
4234
  }
4235
+ if (arg === "--trials") {
4236
+ parsed.trials = positiveInteger(requiredValue(args, index, "--trials"), "--trials");
4237
+ index += 1;
4238
+ continue;
4239
+ }
2420
4240
  throw new Error(`unknown eval run argument: ${arg}`);
2421
4241
  }
2422
4242
  if (parsed.command === "report") {
@@ -2432,6 +4252,20 @@ function parseEvalArgs(args) {
2432
4252
  throw new Error(`unexpected eval report argument: ${arg}`);
2433
4253
  }
2434
4254
  if (parsed.command === "compare") {
4255
+ if (arg === "--format") {
4256
+ parsed.format = comparisonFormat(requiredValue(args, index, "--format"));
4257
+ index += 1;
4258
+ continue;
4259
+ }
4260
+ if (arg === "--metric") {
4261
+ parsed.metric = requiredValue(args, index, "--metric");
4262
+ index += 1;
4263
+ continue;
4264
+ }
4265
+ if (arg === "--allow-config-drift") {
4266
+ parsed.allowConfigDrift = true;
4267
+ continue;
4268
+ }
2435
4269
  if (!arg.startsWith("-")) {
2436
4270
  parsed.compareIds.push(arg);
2437
4271
  continue;
@@ -2472,6 +4306,10 @@ function parseEvalArgs(args) {
2472
4306
  return parsed;
2473
4307
  }
2474
4308
  async function runEvalCommand(args) {
4309
+ if (args[0] === "inventory") {
4310
+ const { runEvalInventory } = await import("./eval-inventory-SXH7PDKX.js");
4311
+ return runEvalInventory(args.slice(1));
4312
+ }
2475
4313
  let parsed;
2476
4314
  try {
2477
4315
  parsed = parseEvalArgs(args);
@@ -2504,13 +4342,19 @@ async function runEvalValidate(parsed) {
2504
4342
  }
2505
4343
  async function runEvalRun(parsed) {
2506
4344
  try {
2507
- const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.repeat);
4345
+ const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.trials ?? parsed.repeat);
2508
4346
  const resolveOptions = {};
2509
4347
  if (parsed.target !== void 0) resolveOptions.target = parsed.target;
2510
4348
  if (parsed.model !== void 0) resolveOptions.model = parsed.model;
2511
- const suite = resolveSuiteForRun(loaded.suite, resolveOptions);
2512
- const clioEntry = resolve7(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
2513
- const artifact = await runEvalSuiteV2({ ...loaded, suite }, { clioEntry });
4349
+ const suite = resolveSuiteForRun(loaded.suite, {
4350
+ ...resolveOptions,
4351
+ ...parsed.trials ? { trials: parsed.trials } : {}
4352
+ });
4353
+ const clioEntry = resolve8(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
4354
+ const artifact = await runEvalSuiteV2(
4355
+ { ...loaded, suite },
4356
+ { clioEntry, freshWorkspaces: parsed.trials !== void 0 }
4357
+ );
2514
4358
  const artifactPath = await writeEvalArtifactV4(clioDataDir(), artifact, parsed.out);
2515
4359
  process.stdout.write(`${renderEvalTextReportV4(artifact)}artifact: ${artifactPath}
2516
4360
  `);
@@ -2520,6 +4364,11 @@ async function runEvalRun(parsed) {
2520
4364
  `);
2521
4365
  for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
2522
4366
  }
4367
+ if (gate !== null && gate.informational.length > 0) {
4368
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4369
+ `);
4370
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4371
+ }
2523
4372
  return artifact.summary.failed === 0 && (gate === null || gate.pass) ? 0 : 1;
2524
4373
  } catch (error) {
2525
4374
  return handleEvalLoadError(error, 1);
@@ -2543,8 +4392,12 @@ async function runEvalCompareCommand(parsed) {
2543
4392
  const dataDir = clioDataDir();
2544
4393
  const baseline = await loadEvalArtifactV4(dataDir, baselineEvalId);
2545
4394
  const candidate = await loadEvalArtifactV4(dataDir, candidateEvalId);
2546
- process.stdout.write(renderEvalComparisonV4(compareEvalArtifactsV4(baseline, candidate)));
2547
- return 0;
4395
+ const summary = compareEvalArtifactsV4(baseline, candidate, {
4396
+ allowConfigDrift: parsed.allowConfigDrift,
4397
+ ...parsed.metric === void 0 ? {} : { metric: parsed.metric }
4398
+ });
4399
+ process.stdout.write(renderEvalComparisonReportV1(summary, parsed.format));
4400
+ return summary.hardGate.pass ? 0 : 1;
2548
4401
  } catch (error) {
2549
4402
  printError(error instanceof Error ? error.message : String(error));
2550
4403
  return error instanceof InvalidIdError ? 2 : 1;
@@ -2554,27 +4407,46 @@ async function runEvalGateCommand(parsed) {
2554
4407
  try {
2555
4408
  const dataDir = clioDataDir();
2556
4409
  const candidate = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
2557
- await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
2558
- const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }] } : loadThresholds(parsed.thresholds);
4410
+ const baseline = await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
4411
+ const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }], informational: [] } : loadThresholds(parsed.thresholds);
2559
4412
  const gate = evaluateGate(candidate, thresholds);
2560
- if (gate.pass) {
4413
+ const comparison = compareEvalArtifactsV4(baseline, candidate);
4414
+ if (gate.informational.length > 0) {
4415
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4416
+ `);
4417
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4418
+ }
4419
+ if (gate.pass && comparison.hardGate.pass) {
2561
4420
  process.stdout.write("gate: pass\n");
2562
4421
  return 0;
2563
4422
  }
2564
- process.stdout.write(`gate: fail (${gate.failures.length} threshold failure)
4423
+ const failureCount = gate.failures.length + comparison.hardGate.failures.length + comparison.hardGate.envelopeFailures.length;
4424
+ process.stdout.write(`gate: fail (${failureCount} hard failure)
2565
4425
  `);
2566
4426
  for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
4427
+ for (const failure of comparison.hardGate.failures) {
4428
+ process.stdout.write(
4429
+ ` ${failure.metric} [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: ${failure.change} (hard behavioral gate)
4430
+ `
4431
+ );
4432
+ }
4433
+ for (const failure of comparison.hardGate.envelopeFailures) {
4434
+ process.stdout.write(
4435
+ ` execution envelope [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: incomparable fields ${failure.fields.join(", ")}
4436
+ `
4437
+ );
4438
+ }
2567
4439
  return 1;
2568
4440
  } catch (error) {
2569
4441
  printError(error instanceof Error ? error.message : String(error));
2570
4442
  return error instanceof InvalidIdError ? 2 : 1;
2571
4443
  }
2572
4444
  }
2573
- function renderArtifactReport(artifact, format, _dataDir) {
2574
- if (format === "json") return renderEvalJsonReportV4(artifact);
2575
- if (format === "md") return renderEvalMarkdownReportV4(artifact);
2576
- if (format === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
2577
- if (format === "junit") return renderEvalJunitReportV4(artifact);
4445
+ function renderArtifactReport(artifact, format2, _dataDir) {
4446
+ if (format2 === "json") return renderEvalJsonReportV4(artifact);
4447
+ if (format2 === "md") return renderEvalMarkdownReportV4(artifact);
4448
+ if (format2 === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
4449
+ if (format2 === "junit") return renderEvalJunitReportV4(artifact);
2578
4450
  return renderEvalTextReportV4(artifact);
2579
4451
  }
2580
4452
  function handleEvalLoadError(error, fallback = 2) {
@@ -2603,7 +4475,11 @@ function reportFormat(value) {
2603
4475
  if (value === "text" || value === "json" || value === "md" || value === "swe-jsonl" || value === "junit") return value;
2604
4476
  throw new Error("--format must be text, json, md, swe-jsonl, or junit");
2605
4477
  }
4478
+ function comparisonFormat(value) {
4479
+ if (value === "text" || value === "json" || value === "md" || value === "junit") return value;
4480
+ throw new Error("eval compare --format must be text, json, md, or junit");
4481
+ }
2606
4482
  export {
2607
4483
  runEvalCommand
2608
4484
  };
2609
- //# sourceMappingURL=eval-BEC2WHDA.js.map
4485
+ //# sourceMappingURL=eval-TFBYQH4H.js.map