@iowarp/clio-coder 0.3.7 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +13 -4
  3. package/dist/{acp-SK4MD6MM.js → acp-7LOELQFP.js} +13 -13
  4. package/dist/{agents-2FN2K6ME.js → agents-FIBG2SHA.js} +41 -37
  5. package/dist/assets/codewiki.json +1 -1
  6. package/dist/{auth-QIYZWM5I.js → auth-OI4LIH2I.js} +31 -24
  7. package/dist/builtins-AD25UL3C.js +17 -0
  8. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  9. package/dist/chunk-3DPEIQKN.js +113 -0
  10. package/dist/{chunk-EOOQZZDE.js → chunk-3DUR4WUA.js} +19 -19
  11. package/dist/{chunk-WHJYKASB.js → chunk-3MRC2YSQ.js} +2 -2
  12. package/dist/{chunk-EBEFWSGL.js → chunk-3UUY7R3Z.js} +14 -10
  13. package/dist/{chunk-LADCF22A.js → chunk-3V5AYSEQ.js} +113 -54
  14. package/dist/{chunk-BMWK7ZIZ.js → chunk-465CC7FK.js} +16 -13
  15. package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
  16. package/dist/{chunk-CEYBNUGC.js → chunk-4H6ULJ3H.js} +378 -36
  17. package/dist/{chunk-YTYFXUI3.js → chunk-4LJX2PUC.js} +9 -9
  18. package/dist/{chunk-DOOEX22V.js → chunk-56KB5IJP.js} +5 -5
  19. package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
  20. package/dist/{chunk-TSHXZTOQ.js → chunk-5HFBWUMU.js} +23 -11
  21. package/dist/{chunk-5UJ6ECTS.js → chunk-5PVQ4SRS.js} +80 -8
  22. package/dist/{chunk-ZWLZP4ZT.js → chunk-5QKCQQ3E.js} +359 -17
  23. package/dist/{chunk-6M7VS3J3.js → chunk-5T7RBWN2.js} +111 -5
  24. package/dist/chunk-774ILSRL.js +172 -0
  25. package/dist/chunk-7C6RYZGQ.js +391 -0
  26. package/dist/{chunk-GH5622CP.js → chunk-A2NJGIB3.js} +2 -2
  27. package/dist/{chunk-C4JBQ5SR.js → chunk-AD7Y7STJ.js} +6 -6
  28. package/dist/{chunk-GEYXPTRF.js → chunk-AEYBF3TB.js} +33 -12
  29. package/dist/{chunk-2SFS6XQE.js → chunk-AMKHQW3C.js} +3 -2
  30. package/dist/{chunk-D4MDIG46.js → chunk-B5CSFE7B.js} +7 -7
  31. package/dist/{chunk-MXI6J5JF.js → chunk-B5XRQOLB.js} +10 -10
  32. package/dist/{chunk-X2KV5FXT.js → chunk-BVDVID7E.js} +2 -2
  33. package/dist/{chunk-JNXPYBB4.js → chunk-CA42X6KT.js} +3 -3
  34. package/dist/{chunk-VREKEFLL.js → chunk-D73KXYPF.js} +3 -3
  35. package/dist/{chunk-JTSEDYVQ.js → chunk-DG4M6ZUE.js} +7 -7
  36. package/dist/{chunk-DQA7QLMD.js → chunk-EBOC7MT3.js} +10 -25
  37. package/dist/{chunk-KZ2H5X4G.js → chunk-ECUO3KDP.js} +129 -14
  38. package/dist/{chunk-JRIO5UD2.js → chunk-EQ63NRB7.js} +5 -5
  39. package/dist/{chunk-YD734TPH.js → chunk-FALJGAWU.js} +2 -2
  40. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  41. package/dist/{chunk-XEGB6BCN.js → chunk-GAYUJ7LE.js} +68 -14
  42. package/dist/{chunk-UND3GU2L.js → chunk-H7IXIC72.js} +2 -2
  43. package/dist/{chunk-IR4CFBFN.js → chunk-HAY4ZE2P.js} +12 -12
  44. package/dist/{chunk-UVDSQ6LW.js → chunk-HCBCAYZU.js} +74 -147
  45. package/dist/{chunk-4DWFMQDR.js → chunk-HJB5IUKP.js} +89 -145
  46. package/dist/{chunk-M4AKACEO.js → chunk-HKO36JWF.js} +33 -5
  47. package/dist/{chunk-KCMKRQX4.js → chunk-HPCTNZM2.js} +45 -82
  48. package/dist/{chunk-465YSENW.js → chunk-IFBNV6H6.js} +3 -3
  49. package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
  50. package/dist/chunk-JEQQR47K.js +3025 -0
  51. package/dist/{chunk-FO5ZOVUY.js → chunk-KV2AOLDF.js} +27 -7
  52. package/dist/chunk-LU7P4LHA.js +33 -0
  53. package/dist/{chunk-6TUKSZVF.js → chunk-LXPJXFM5.js} +11 -11
  54. package/dist/{chunk-VQNODYQ4.js → chunk-MIX5N5AC.js} +488 -3668
  55. package/dist/chunk-MLOK6ZOS.js +2888 -0
  56. package/dist/{chunk-ZZMN5OM4.js → chunk-MV2VUEJC.js} +2 -2
  57. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  58. package/dist/{chunk-OBMAI2DP.js → chunk-N3PBVRTZ.js} +12 -388
  59. package/dist/{chunk-WJHBC77E.js → chunk-N5XKWMDW.js} +17 -7
  60. package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
  61. package/dist/{chunk-UFQ3F4FW.js → chunk-NQ6UCCOD.js} +4 -4
  62. package/dist/chunk-NUGM5KR6.js +165 -0
  63. package/dist/{chunk-DMD2AGVS.js → chunk-NZU6YDNV.js} +20 -18
  64. package/dist/{chunk-WHGPSPT5.js → chunk-O6I4CIEU.js} +151 -13
  65. package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
  66. package/dist/{chunk-PD3MESLB.js → chunk-P3JGPQFL.js} +4 -4
  67. package/dist/{chunk-UHXRNZ2J.js → chunk-PNY46YEY.js} +23 -6
  68. package/dist/{chunk-THKY7CD7.js → chunk-PZ4I4JE2.js} +134 -29
  69. package/dist/{chunk-SROCI7ZU.js → chunk-QQ7EKM72.js} +5 -5
  70. package/dist/{chunk-QCTRSGHQ.js → chunk-R7LNVMCS.js} +91 -53
  71. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  72. package/dist/chunk-RKKLTLYB.js +45 -0
  73. package/dist/{chunk-OB5HIGJY.js → chunk-RKRLDWD3.js} +4 -1
  74. package/dist/{chunk-DJNLUABN.js → chunk-S4COXYBG.js} +588 -32
  75. package/dist/{chunk-3HAPLH5M.js → chunk-T3Z6VAAF.js} +172 -11
  76. package/dist/{chunk-FOT2FX5J.js → chunk-TD7UE2L5.js} +12 -10
  77. package/dist/{chunk-UUANF5CR.js → chunk-TEO2TLVN.js} +856 -967
  78. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  79. package/dist/{chunk-EELBMBT6.js → chunk-VKBMFOYV.js} +74 -15
  80. package/dist/chunk-VO2LKSTM.js +165 -0
  81. package/dist/{chunk-5C77SEEY.js → chunk-VPTUJU4P.js} +3 -3
  82. package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
  83. package/dist/{chunk-J7PIKKWC.js → chunk-WXCJ7VME.js} +8 -8
  84. package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
  85. package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
  86. package/dist/{chunk-AB4XIIVB.js → chunk-YKOFT37S.js} +6 -6
  87. package/dist/chunk-YSEHGPCT.js +127 -0
  88. package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
  89. package/dist/cli/index.js +32 -32
  90. package/dist/{clio-WBVQEBKO.js → clio-LT5V7SSZ.js} +9 -9
  91. package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +5 -5
  92. package/dist/codewiki/build-worker.js +4 -4
  93. package/dist/{components-F7OEATSO.js → components-ZFA3SAER.js} +8 -8
  94. package/dist/{config-TRBL3RCF.js → config-RXS5T3JT.js} +98 -65
  95. package/dist/{configure-OLCVPHNM.js → configure-2WYWSCSD.js} +26 -22
  96. package/dist/{context-MJIJ6GOX.js → context-I3BTOTCS.js} +12 -12
  97. package/dist/{context-XEWE3MOJ.js → context-MVOORGMF.js} +54 -47
  98. package/dist/{context-WFPKQSM6.js → context-PALKKQYL.js} +28 -28
  99. package/dist/{context-clear-KNOS2JPB.js → context-clear-N2WOYZ2K.js} +53 -46
  100. package/dist/{context-index-SSR5ECNE.js → context-index-HNG3MOME.js} +6 -6
  101. package/dist/{context-working-set-EUXAZI6N.js → context-working-set-MIEVECVZ.js} +17 -18
  102. package/dist/{dispatch-runner-B7MTOVKL.js → dispatch-runner-VVA4SRRH.js} +90 -61
  103. package/dist/{docs-FLJTIDSE.js → docs-7LQ23DLM.js} +8 -8
  104. package/dist/doctor-TWBWFK5V.js +165 -0
  105. package/dist/eval-IJ5VEZDJ.js +4483 -0
  106. package/dist/{evidence-JZNBUOQZ.js → evidence-L5APPXNV.js} +68 -61
  107. package/dist/{evolve-FJVC4KKI.js → evolve-RGNKFJ52.js} +47 -40
  108. package/dist/{extensions-IQL36S7K.js → extensions-7WYWUX5A.js} +13 -7
  109. package/dist/{fleet-BDKYJFCP.js → fleet-6CNVBZZP.js} +113 -76
  110. package/dist/{fleet-commands-ZFIWZSB3.js → fleet-commands-L2SXSYEI.js} +10 -10
  111. package/dist/{fleet-graph-Y6HPXIVF.js → fleet-graph-2J3OOIPO.js} +17 -15
  112. package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-CZRJ4JP5.js} +5 -6
  113. package/dist/{fleet-validate-BIYREGIK.js → fleet-validate-C5RI6DP7.js} +20 -19
  114. package/dist/{init-LQUB5COQ.js → init-VBN2ACVA.js} +70 -63
  115. package/dist/{library-NJAHIGG4.js → library-JHGUMLY2.js} +22 -20
  116. package/dist/{memory-OG6HOYKM.js → memory-K4OQIYWG.js} +49 -42
  117. package/dist/{models-5ZG5XY7J.js → models-2NCZUWDD.js} +35 -29
  118. package/dist/{monitor-TJ7AMTGB.js → monitor-MMVTJABD.js} +64 -45
  119. package/dist/{orchestrator-WZYB54DM.js → orchestrator-ZKBPCHW6.js} +1971 -520
  120. package/dist/{paths-XUC7GS6E.js → paths-DBXMZMDU.js} +5 -5
  121. package/dist/registry-LG64LTF4.js +11 -0
  122. package/dist/{reset-PXQT45IY.js → reset-DD5JGOY3.js} +11 -11
  123. package/dist/{run-FQ74YF62.js → run-QEGNX7FL.js} +89 -83
  124. package/dist/{share-FW7SVCL3.js → share-JKD3BQMW.js} +20 -18
  125. package/dist/{skills-7E7IRB3R.js → skills-LMQIKDOZ.js} +23 -21
  126. package/dist/{skills-eval-LI75W6OK.js → skills-eval-I7X2774U.js} +59 -52
  127. package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
  128. package/dist/support-I7LOJLIF.js +38 -0
  129. package/dist/{targets-4CIFKCTW.js → targets-RUSR6B5Z.js} +77 -42
  130. package/dist/{terminal-lease-WUZY7ZV5.js → terminal-lease-QYVORFR4.js} +6 -4
  131. package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
  132. package/dist/{uninstall-7FV7IP4E.js → uninstall-ZJF5H5ZN.js} +8 -8
  133. package/dist/{upgrade-K2HVIVMQ.js → upgrade-XANW3FXB.js} +29 -26
  134. package/dist/{usage-GTZELZQX.js → usage-4H7ZRXQT.js} +110 -61
  135. package/dist/{verifiers-RLAHT27O.js → verifiers-UZXNBZEB.js} +13 -13
  136. package/dist/{verify-BX3BRKH5.js → verify-BVKWTNDL.js} +9 -9
  137. package/dist/{wiki-generate-ASIFASCN.js → wiki-generate-MY7WV2QI.js} +76 -69
  138. package/dist/worker/entry.js +69 -66
  139. package/docs/alcf-provider.md +1 -1
  140. package/docs/architecture.md +1 -1
  141. package/docs/artifact-versions.md +11 -5
  142. package/docs/built-in-agents.md +1 -1
  143. package/docs/capacity-and-scheduling.md +23 -2
  144. package/docs/commands-and-modes.md +2 -2
  145. package/docs/configuration-and-targets.md +37 -5
  146. package/docs/context-engine.md +63 -4
  147. package/docs/documentation-coverage.md +3 -3
  148. package/docs/documentation-guide.md +1 -1
  149. package/docs/environment-variables.md +2 -0
  150. package/docs/eval-runner.md +262 -11
  151. package/docs/evals-internal.md +72 -2
  152. package/docs/evidence-and-memory.md +77 -12
  153. package/docs/evolution.md +1 -1
  154. package/docs/extensions-and-sharing.md +3 -1
  155. package/docs/fleet-dispatch.md +34 -9
  156. package/docs/glossary.md +21 -1
  157. package/docs/installation-and-lifecycle.md +1 -1
  158. package/docs/middleware-and-components.md +1 -1
  159. package/docs/model-catalog.md +1 -1
  160. package/docs/observability.md +54 -3
  161. package/docs/proactive-memory.md +127 -14
  162. package/docs/prompt-envelope-and-tools.md +22 -2
  163. package/docs/provider-adapter-cookbook.md +1 -1
  164. package/docs/release-cut-checklist.md +60 -41
  165. package/docs/safety-model.md +1 -1
  166. package/docs/scientific-validation.md +1 -1
  167. package/docs/skills-marketplace.md +1 -1
  168. package/docs/tool-usage.md +1 -1
  169. package/docs/trace-store.md +1 -1
  170. package/docs/troubleshooting.md +87 -0
  171. package/docs/tui-design.md +1 -1
  172. package/docs/worker-dispatch-mechanics.md +1 -1
  173. package/package.json +2 -2
  174. package/src/cli/agents.ts +1 -1
  175. package/src/cli/argv.ts +5 -0
  176. package/src/cli/config-inspect.ts +33 -6
  177. package/src/cli/config.ts +1 -1
  178. package/src/cli/configure.ts +107 -23
  179. package/src/cli/doctor-state-size.ts +82 -0
  180. package/src/cli/doctor.ts +7 -1
  181. package/src/cli/eval.ts +80 -16
  182. package/src/cli/evidence.ts +30 -25
  183. package/src/cli/extensions.ts +5 -1
  184. package/src/cli/fleet-preflight.ts +2 -12
  185. package/src/cli/fleet.ts +32 -3
  186. package/src/cli/shared.ts +1 -0
  187. package/src/cli/targets.ts +45 -11
  188. package/src/cli/trace.ts +63 -4
  189. package/src/cli/usage.ts +63 -14
  190. package/src/cli/validate-model.ts +60 -5
  191. package/src/core/bus-events.ts +54 -1
  192. package/src/core/cache-telemetry.ts +42 -0
  193. package/src/core/commit-attribution.ts +4 -4
  194. package/src/core/config.ts +18 -0
  195. package/src/core/defaults.ts +36 -6
  196. package/src/core/endpoint-key.ts +27 -0
  197. package/src/core/path-boundary.ts +100 -0
  198. package/src/core/residency-target-key.ts +25 -0
  199. package/src/core/response-schema.ts +36 -2
  200. package/src/domains/agents/extension.ts +2 -11
  201. package/src/domains/agents/fleet-contract.ts +30 -12
  202. package/src/domains/agents/recipe.ts +7 -1
  203. package/src/domains/agents/registry.ts +73 -5
  204. package/src/domains/agents/result-contract.ts +128 -17
  205. package/src/domains/agents/write-boundary.ts +15 -50
  206. package/src/domains/config/classify.ts +3 -0
  207. package/src/domains/context/codewiki/coordinator.ts +12 -4
  208. package/src/domains/context/project-rules.ts +51 -1
  209. package/src/domains/dispatch/admission.ts +40 -3
  210. package/src/domains/dispatch/assignment-reconcile.ts +22 -5
  211. package/src/domains/dispatch/assignment-store.ts +151 -14
  212. package/src/domains/dispatch/capacity-lease.ts +98 -9
  213. package/src/domains/dispatch/contract.ts +26 -1
  214. package/src/domains/dispatch/delegation-plan.ts +2 -5
  215. package/src/domains/dispatch/execution-plan.ts +44 -4
  216. package/src/domains/dispatch/execution-role.ts +9 -1
  217. package/src/domains/dispatch/extension.ts +309 -96
  218. package/src/domains/dispatch/fleet-run.ts +78 -4
  219. package/src/domains/dispatch/gate-role-prompts.ts +38 -0
  220. package/src/domains/dispatch/heartbeat.ts +32 -8
  221. package/src/domains/dispatch/index.ts +6 -1
  222. package/src/domains/dispatch/intent-requirements.ts +40 -0
  223. package/src/domains/dispatch/intent.ts +84 -8
  224. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  225. package/src/domains/dispatch/path-scope.ts +370 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +2 -1
  227. package/src/domains/dispatch/reservation-store.ts +116 -8
  228. package/src/domains/dispatch/state.ts +4 -0
  229. package/src/domains/dispatch/types.ts +14 -7
  230. package/src/domains/dispatch/validation.ts +6 -3
  231. package/src/domains/dispatch/worker-spawn.ts +25 -11
  232. package/src/domains/dispatch/write-boundary-enforcer.ts +62 -0
  233. package/src/domains/dispatch/write-boundary.ts +262 -22
  234. package/src/domains/eval/artifacts/store.ts +62 -0
  235. package/src/domains/eval/compare/behavioral.ts +224 -0
  236. package/src/domains/eval/compare/compare.ts +355 -2
  237. package/src/domains/eval/compare/envelope.ts +128 -0
  238. package/src/domains/eval/compare/gates.ts +24 -6
  239. package/src/domains/eval/compare/thresholds.ts +30 -3
  240. package/src/domains/eval/execution-provenance.ts +240 -0
  241. package/src/domains/eval/metrics/aggregate.ts +136 -0
  242. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  243. package/src/domains/eval/metrics/evidence.ts +79 -2
  244. package/src/domains/eval/metrics/tracked.ts +413 -0
  245. package/src/domains/eval/provenance.ts +117 -0
  246. package/src/domains/eval/reports/comparison.ts +128 -0
  247. package/src/domains/eval/reports/junit.ts +17 -3
  248. package/src/domains/eval/reports/markdown.ts +3 -3
  249. package/src/domains/eval/reports/text.ts +14 -0
  250. package/src/domains/eval/run-compare.ts +20 -0
  251. package/src/domains/eval/runners/clio-run.ts +139 -2
  252. package/src/domains/eval/runners/external-command.ts +28 -3
  253. package/src/domains/eval/schema/adapter.ts +111 -0
  254. package/src/domains/eval/schema/artifact.ts +20 -0
  255. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  256. package/src/domains/eval/schema/behavioral.ts +520 -0
  257. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  258. package/src/domains/eval/schema/serving.ts +74 -0
  259. package/src/domains/eval/schema/suite.ts +38 -8
  260. package/src/domains/eval/schema/validate.ts +58 -3
  261. package/src/domains/eval/schema/verdict.ts +237 -0
  262. package/src/domains/eval/suites/resolve.ts +2 -0
  263. package/src/domains/eval/suites/run.ts +264 -33
  264. package/src/domains/eval/verifiers/command.ts +2 -1
  265. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  266. package/src/domains/evidence/build.ts +68 -21
  267. package/src/domains/evidence/eval.ts +2 -12
  268. package/src/domains/evidence/findings-markdown.ts +33 -0
  269. package/src/domains/evidence/index.ts +21 -0
  270. package/src/domains/evidence/provenance.ts +46 -11
  271. package/src/domains/evidence/run-trust.ts +7 -113
  272. package/src/domains/evidence/trust-projection.ts +274 -0
  273. package/src/domains/evidence/trust-status.ts +145 -17
  274. package/src/domains/evidence/types.ts +4 -0
  275. package/src/domains/extensions/compatibility.ts +285 -0
  276. package/src/domains/extensions/discovery.ts +126 -4
  277. package/src/domains/extensions/resources.ts +21 -9
  278. package/src/domains/extensions/state.ts +18 -5
  279. package/src/domains/extensions/types.ts +6 -1
  280. package/src/domains/lifecycle/doctor.ts +209 -2
  281. package/src/domains/memory/index.ts +14 -0
  282. package/src/domains/memory/task-bank-promotion.ts +64 -0
  283. package/src/domains/memory/task-memory-policy.ts +77 -8
  284. package/src/domains/memory/task-memory-spend.ts +131 -0
  285. package/src/domains/memory/task-memory-status.ts +7 -0
  286. package/src/domains/memory/task-memory-telemetry.ts +2 -0
  287. package/src/domains/middleware/index.ts +1 -0
  288. package/src/domains/middleware/memory-intervention.ts +69 -5
  289. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  290. package/src/domains/observability/background-memory-usage.ts +140 -0
  291. package/src/domains/observability/cost.ts +1 -1
  292. package/src/domains/observability/index.ts +7 -0
  293. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  294. package/src/domains/observability/trace-store.ts +192 -2
  295. package/src/domains/prompts/compiler.ts +100 -13
  296. package/src/domains/prompts/contract.ts +3 -5
  297. package/src/domains/providers/endpoint-capacity.ts +96 -0
  298. package/src/domains/providers/extension.ts +30 -2
  299. package/src/domains/providers/index.ts +10 -0
  300. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  301. package/src/domains/providers/runtime-resolution.ts +8 -1
  302. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  303. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  304. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  305. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  306. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  307. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  308. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  309. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  310. package/src/domains/providers/types/capability-flags.ts +2 -0
  311. package/src/domains/providers/types/target-descriptor.ts +2 -0
  312. package/src/domains/resources/common-loader.ts +3 -0
  313. package/src/domains/resources/prompts/loader.ts +184 -17
  314. package/src/domains/safety/call-target.ts +52 -0
  315. package/src/domains/safety/policy-engine.ts +5 -5
  316. package/src/domains/safety/run-effects.ts +96 -2
  317. package/src/domains/safety/scope.ts +7 -12
  318. package/src/domains/session/context-accounting.ts +52 -1
  319. package/src/domains/session/context-ledger.ts +37 -13
  320. package/src/domains/session/index.ts +6 -0
  321. package/src/domains/session/prompt-cache.ts +140 -0
  322. package/src/domains/session/prompt-manifest.ts +42 -0
  323. package/src/engine/acp/adapter.ts +18 -3
  324. package/src/engine/acp/server.ts +4 -1
  325. package/src/engine/ai.ts +35 -0
  326. package/src/engine/apis/llamacpp-residency.ts +55 -3
  327. package/src/engine/apis/lmstudio.ts +25 -5
  328. package/src/engine/apis/ollama-native.ts +2 -1
  329. package/src/engine/apis/openai-completions.ts +80 -17
  330. package/src/engine/apis/residency-lock.ts +3 -1
  331. package/src/engine/apis/residency.ts +34 -1
  332. package/src/engine/prompt-templates.ts +18 -1
  333. package/src/engine/provider-payload.ts +29 -1
  334. package/src/engine/worker-runtime.ts +6 -3
  335. package/src/entry/orchestrator.ts +176 -30
  336. package/src/interactive/chat-loop-messages.ts +26 -7
  337. package/src/interactive/chat-loop.ts +318 -41
  338. package/src/interactive/chat-panel.ts +62 -8
  339. package/src/interactive/clio-editor.ts +45 -8
  340. package/src/interactive/context-activity.ts +5 -1
  341. package/src/interactive/context-meter.ts +1 -1
  342. package/src/interactive/context-overlay.ts +40 -10
  343. package/src/interactive/cost-overlay.ts +64 -6
  344. package/src/interactive/dispatch-board.ts +127 -5
  345. package/src/interactive/fleet-run-preview.ts +41 -15
  346. package/src/interactive/handoff-round.ts +41 -2
  347. package/src/interactive/interactive-application.ts +24 -1
  348. package/src/interactive/interactive-event-projection.ts +14 -0
  349. package/src/interactive/interactive-input-runtime.ts +8 -0
  350. package/src/interactive/interactive-presentation.ts +4 -0
  351. package/src/interactive/interactive-shell.ts +20 -17
  352. package/src/interactive/interactive-slash-runtime.ts +27 -4
  353. package/src/interactive/memory-overlay.ts +8 -0
  354. package/src/interactive/mutation-preview.ts +295 -0
  355. package/src/interactive/overlay-general-openers.ts +16 -0
  356. package/src/interactive/overlay-key-routing.ts +38 -0
  357. package/src/interactive/overlay-lifecycle.ts +38 -5
  358. package/src/interactive/overlay-permission-lifecycle.ts +22 -2
  359. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  360. package/src/interactive/overlays/ask-user.ts +91 -19
  361. package/src/interactive/overlays/help-reference.ts +4 -0
  362. package/src/interactive/overlays/prompts.ts +11 -1
  363. package/src/interactive/overlays/settings.ts +176 -48
  364. package/src/interactive/permission-hint.ts +34 -2
  365. package/src/interactive/permission-overlay.ts +159 -9
  366. package/src/interactive/prewarm.ts +197 -0
  367. package/src/interactive/render-trace.ts +162 -15
  368. package/src/interactive/renderers/tool-execution.ts +4 -0
  369. package/src/interactive/side-question.ts +58 -1
  370. package/src/interactive/slash-commands.ts +7 -2
  371. package/src/interactive/status/controller.ts +11 -0
  372. package/src/interactive/status/state-machine.ts +54 -2
  373. package/src/interactive/status/types.ts +7 -0
  374. package/src/interactive/terminal-lease.ts +2 -0
  375. package/src/interactive/turn-context.ts +299 -31
  376. package/src/interactive/turn-persistence.ts +14 -4
  377. package/src/interactive/turn-prewarm.ts +364 -0
  378. package/src/interactive/turn-queues.ts +7 -4
  379. package/src/interactive/turn-runtime.ts +8 -1
  380. package/src/interactive/turn-state.ts +23 -0
  381. package/src/interactive/view/artifacts.ts +42 -9
  382. package/src/interactive/view/view-overlay.ts +43 -6
  383. package/src/interactive/worker-receipts.ts +14 -2
  384. package/src/interactive/worker-stream.ts +8 -0
  385. package/src/tools/ask-user.ts +43 -2
  386. package/src/tools/dispatch-admission.ts +12 -13
  387. package/src/tools/dispatch-arguments.ts +27 -0
  388. package/src/tools/dispatch-plan.ts +46 -9
  389. package/src/tools/dispatch-runner.ts +48 -13
  390. package/src/tools/dispatch-scout.ts +1 -1
  391. package/src/tools/monitor.ts +13 -0
  392. package/src/tools/registry.ts +16 -0
  393. package/src/tools/worker-evidence.ts +19 -13
  394. package/src/worker/spec-contract.ts +2 -1
  395. package/dist/chunk-AOCYTWAV.js +0 -449
  396. package/dist/chunk-HWUFFB6L.js +0 -83
  397. package/dist/chunk-R346GLFC.js +0 -31
  398. package/dist/chunk-ZGH7FGS5.js +0 -1079
  399. package/dist/doctor-RN4YKO2X.js +0 -87
  400. package/dist/eval-RUBJVSNQ.js +0 -2557
@@ -0,0 +1,4483 @@
1
+ import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
2
+ import {
3
+ discoverAgentRecipes
4
+ } from "./chunk-HCBCAYZU.js";
5
+ import "./chunk-AD7Y7STJ.js";
6
+ import "./chunk-AMKHQW3C.js";
7
+ import {
8
+ loadFragments,
9
+ renderCodewikiDigest
10
+ } from "./chunk-47CMYGET.js";
11
+ import {
12
+ EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
13
+ EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
14
+ EVAL_TRACKED_METRIC_NAMES,
15
+ EVAL_VERDICT_SCHEMA_V1,
16
+ EvalTaskFileError,
17
+ TRUST_STATUS_AXES,
18
+ adaptRunReceiptTrustStatus,
19
+ assertComparableTrackedMetricSources,
20
+ assertEvalBehaviorReferencesVerdictV1,
21
+ buildEvalBehaviorMetricsV1,
22
+ createEvalId,
23
+ evalClioProvenance,
24
+ evalEnvironmentProvenance,
25
+ evalHarnessMetricsFromReceipt,
26
+ evalServingConfiguration,
27
+ evalServingObservationFrom,
28
+ formatTrustSummary,
29
+ inspectRunReceiptTrustStatus,
30
+ judgeEvalBehaviorV1,
31
+ loadEvalArtifactV4,
32
+ loadEvalTaskFile,
33
+ parseEvalBehaviorScenarioV1,
34
+ parseEvalExecutionMatrixDimensionsV1,
35
+ parseEvalVerdictEnvelopeV1,
36
+ renderEvalServingConfiguration,
37
+ sameEvalServingConfiguration,
38
+ summarizeTrustStatus,
39
+ verifyReceiptIntegrity,
40
+ writeEvalArtifactV4
41
+ } from "./chunk-MLOK6ZOS.js";
42
+ import {
43
+ listSessionLedgerRefs,
44
+ parseSessionEntries
45
+ } from "./chunk-LXPJXFM5.js";
46
+ import "./chunk-3DPEIQKN.js";
47
+ import "./chunk-W6GROXXM.js";
48
+ import "./chunk-RKKLTLYB.js";
49
+ import "./chunk-5QKCQQ3E.js";
50
+ import {
51
+ shellQuote
52
+ } from "./chunk-TXOTCRLG.js";
53
+ import "./chunk-HVDIIIQW.js";
54
+ import "./chunk-VPTUJU4P.js";
55
+ import "./chunk-2JDWVJND.js";
56
+ import "./chunk-H7IXIC72.js";
57
+ import {
58
+ createSafetyPolicyEngine
59
+ } from "./chunk-N3PBVRTZ.js";
60
+ import "./chunk-A2NJGIB3.js";
61
+ import {
62
+ agentSpecFingerprint,
63
+ normalizeAgentSpec
64
+ } from "./chunk-S4COXYBG.js";
65
+ import "./chunk-MV3K5QF2.js";
66
+ import "./chunk-RAPCMZL4.js";
67
+ import "./chunk-UL3WSD3F.js";
68
+ import "./chunk-ECH6PKUQ.js";
69
+ import "./chunk-CGKSTWHD.js";
70
+ import {
71
+ readCodewiki,
72
+ structuralCodewikiHash
73
+ } from "./chunk-LBMZMYH2.js";
74
+ import {
75
+ detectProjectProfile
76
+ } from "./chunk-XAKHZX5N.js";
77
+ import {
78
+ enumerateWorkspaceFiles
79
+ } from "./chunk-33YXPOE3.js";
80
+ import "./chunk-7CR24IG7.js";
81
+ import "./chunk-XPLRXC72.js";
82
+ import "./chunk-IFBNV6H6.js";
83
+ import {
84
+ printError
85
+ } from "./chunk-XK56QHLX.js";
86
+ import "./chunk-5TSRNF4G.js";
87
+ import "./chunk-CFGTUFWB.js";
88
+ import {
89
+ extractReasoningTokens
90
+ } from "./chunk-AEYBF3TB.js";
91
+ import "./chunk-IHXBNWMM.js";
92
+ import "./chunk-B5CSFE7B.js";
93
+ import "./chunk-PNY46YEY.js";
94
+ import "./chunk-FQ4SKYE4.js";
95
+ import "./chunk-RKRLDWD3.js";
96
+ import {
97
+ InvalidIdError
98
+ } from "./chunk-KV2AOLDF.js";
99
+ import "./chunk-6EJMN2Y3.js";
100
+ import "./chunk-IWHMRKLL.js";
101
+ import "./chunk-XDOQXGFO.js";
102
+ import "./chunk-LL4KHSZI.js";
103
+ import "./chunk-4ZG3XFUR.js";
104
+ import "./chunk-EQ63NRB7.js";
105
+ import "./chunk-SST6Z5JA.js";
106
+ import "./chunk-IKCO5N3L.js";
107
+ import "./chunk-3I7MS7N2.js";
108
+ import {
109
+ require_dist
110
+ } from "./chunk-APJ265NV.js";
111
+ import {
112
+ clioDataDir,
113
+ clioStateDir
114
+ } from "./chunk-BNAZZHFG.js";
115
+ import "./chunk-WEPFGWHJ.js";
116
+ import "./chunk-YXLYO42X.js";
117
+ import {
118
+ __toESM,
119
+ init_esm_shims
120
+ } from "./chunk-3R73A4XB.js";
121
+
122
+ // src/cli/eval.ts
123
+ init_esm_shims();
124
+ import { resolve as resolve8 } from "node:path";
125
+
126
+ // src/domains/eval/compare/compare.ts
127
+ init_esm_shims();
128
+
129
+ // src/domains/eval/metrics/aggregate.ts
130
+ init_esm_shims();
131
+ function aggregateEvalVerdicts(verdicts) {
132
+ const byScenario = /* @__PURE__ */ new Map();
133
+ for (const verdict of verdicts) {
134
+ const group = byScenario.get(verdict.scenarioId) ?? [];
135
+ group.push(verdict);
136
+ byScenario.set(verdict.scenarioId, group);
137
+ }
138
+ return [...byScenario.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([scenarioId, group]) => aggregateScenario(scenarioId, group));
139
+ }
140
+ function aggregateScenario(scenarioId, verdicts) {
141
+ const ordered = [...verdicts].sort((left, right) => left.trialIndex - right.trialIndex);
142
+ const passed = ordered.filter((verdict) => verdict.outcome === "pass").length;
143
+ const failed = ordered.filter((verdict) => verdict.outcome === "fail").length;
144
+ const unmeasured = ordered.filter((verdict) => verdict.outcome === "unmeasured").length;
145
+ const machineryFailures = ordered.filter((verdict) => verdict.machinery === "infrastructure_failure").length;
146
+ const fixed = Object.fromEntries(
147
+ EVAL_TRACKED_METRIC_NAMES.map((name) => [name, distribution(ordered.map((verdict) => verdict.trackedMetrics[name]))])
148
+ );
149
+ const reasons = new Set(ordered.flatMap((verdict) => Object.keys(verdict.trackedMetrics.expectedColdReasons)));
150
+ const expectedColdReasons = Object.fromEntries(
151
+ [...reasons].sort((left, right) => left.localeCompare(right)).map((reason) => [
152
+ reason,
153
+ distribution(
154
+ ordered.map(
155
+ (verdict) => verdict.trackedMetrics.expectedColdReasons[reason] ?? { value: 0, source: "ledger" }
156
+ )
157
+ )
158
+ ])
159
+ );
160
+ const k = ordered.length;
161
+ return {
162
+ scenarioId,
163
+ trials: k,
164
+ k,
165
+ passed,
166
+ failed,
167
+ unmeasured,
168
+ machineryFailures,
169
+ passAtK: k > 0 && passed > 0 ? 1 : 0,
170
+ passPowK: k > 0 && passed === k ? 1 : 0,
171
+ trackedMetrics: { ...fixed, expectedColdReasons }
172
+ };
173
+ }
174
+ function distribution(metrics) {
175
+ const values = metrics.flatMap((metric) => metric.value === null ? [] : [metric.value]);
176
+ const sources = [...new Set(metrics.map((metric) => metric.source))].sort(compareSources);
177
+ if (values.length === 0) {
178
+ return {
179
+ observations: metrics.length,
180
+ measured: 0,
181
+ unmeasured: metrics.length,
182
+ mean: null,
183
+ min: null,
184
+ max: null,
185
+ p90: null,
186
+ variance: null,
187
+ standardDeviation: null,
188
+ sources
189
+ };
190
+ }
191
+ const ordered = [...values].sort((left, right) => left - right);
192
+ const p90Index = Math.max(0, Math.ceil(ordered.length * 0.9) - 1);
193
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
194
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
195
+ return {
196
+ observations: metrics.length,
197
+ measured: values.length,
198
+ unmeasured: metrics.length - values.length,
199
+ mean,
200
+ min: ordered[0] ?? null,
201
+ max: ordered.at(-1) ?? null,
202
+ p90: ordered[p90Index] ?? null,
203
+ variance,
204
+ standardDeviation: Math.sqrt(variance),
205
+ sources
206
+ };
207
+ }
208
+ function compareSources(left, right) {
209
+ return sourceOrder(left) - sourceOrder(right);
210
+ }
211
+ function sourceOrder(source) {
212
+ if (source === "ledger") return 0;
213
+ if (source === "receipt") return 1;
214
+ return 2;
215
+ }
216
+
217
+ // src/domains/eval/compare/behavioral.ts
218
+ init_esm_shims();
219
+
220
+ // src/domains/eval/compare/envelope.ts
221
+ init_esm_shims();
222
+ function compareEvalExecutionEnvelopesV1(identity, baseline, candidate, baselineDimensions, candidateDimensions) {
223
+ const leftDimensions = [...baselineDimensions].sort();
224
+ const rightDimensions = [...candidateDimensions].sort();
225
+ if (stableJson(leftDimensions) !== stableJson(rightDimensions)) {
226
+ return { ...identity, fields: ["matrix.dimensions"] };
227
+ }
228
+ if (baseline.some((result) => result.executionEnvelope !== void 0) && baseline.some((result) => result.executionEnvelope === void 0) || candidate.some((result) => result.executionEnvelope !== void 0) && candidate.some((result) => result.executionEnvelope === void 0)) {
229
+ return { ...identity, fields: ["executionEnvelope.missingTrial"] };
230
+ }
231
+ const ignored = new Set(leftDimensions);
232
+ const baselineEnvelopes = uniqueEnvelopes(baseline, ignored);
233
+ const candidateEnvelopes = uniqueEnvelopes(candidate, ignored);
234
+ if (baselineEnvelopes.length === 0 && candidateEnvelopes.length === 0) return null;
235
+ if (baselineEnvelopes.length === 0 || candidateEnvelopes.length === 0) {
236
+ return { ...identity, fields: ["executionEnvelope"] };
237
+ }
238
+ if (baselineEnvelopes.length > 1 || candidateEnvelopes.length > 1) {
239
+ return { ...identity, fields: ["executionEnvelope.withinRunVariance"] };
240
+ }
241
+ const left = baselineEnvelopes[0];
242
+ const right = candidateEnvelopes[0];
243
+ if (left === void 0 || right === void 0 || stableJson(left) === stableJson(right)) return null;
244
+ return { ...identity, fields: differingFields(left, right, ignored) };
245
+ }
246
+ function uniqueEnvelopes(results, ignored) {
247
+ const byIdentity = /* @__PURE__ */ new Map();
248
+ for (const result of results) {
249
+ if (result.executionEnvelope === void 0) continue;
250
+ const normalized = normalizedEnvelope(result.executionEnvelope, ignored);
251
+ byIdentity.set(stableJson(normalized), normalized);
252
+ }
253
+ return [...byIdentity.values()];
254
+ }
255
+ function normalizedEnvelope(envelope, ignored) {
256
+ return {
257
+ ...envelope,
258
+ prompt: ignored.has("prompt") ? { fragments: [], compositionHash: null } : envelope.prompt,
259
+ recipe: ignored.has("recipe") ? null : envelope.recipe,
260
+ target: ignored.has("target") ? "<matrix>" : envelope.target,
261
+ wireModel: ignored.has("wireModel") ? null : envelope.wireModel,
262
+ runtime: ignored.has("runtime") ? null : envelope.runtime,
263
+ thinkingLevel: ignored.has("thinkingLevel") ? null : envelope.thinkingLevel,
264
+ toolSignature: ignored.has("toolSignature") ? null : envelope.toolSignature,
265
+ autonomy: ignored.has("autonomy") ? null : envelope.autonomy,
266
+ policyHashes: ignored.has("policy") ? { rulePack: null, project: null } : envelope.policyHashes,
267
+ projectContext: ignored.has("projectContext") ? {
268
+ kind: "none",
269
+ tier: null,
270
+ contentHash: null,
271
+ chars: null,
272
+ sections: [],
273
+ rulesApplied: [],
274
+ operatorProfileApplied: null
275
+ } : envelope.projectContext,
276
+ corpus: ignored.has("corpus") ? { id: "<matrix>", version: "<matrix>" } : envelope.corpus
277
+ };
278
+ }
279
+ function differingFields(left, right, ignored) {
280
+ const fields = [
281
+ ["prompt", "prompt", left.prompt, right.prompt],
282
+ ["recipe", "recipe", left.recipe, right.recipe],
283
+ ["target", "target", left.target, right.target],
284
+ ["wireModel", "wireModel", left.wireModel, right.wireModel],
285
+ ["runtime", "runtime", left.runtime, right.runtime],
286
+ ["thinkingLevel", "thinkingLevel", left.thinkingLevel, right.thinkingLevel],
287
+ ["toolSignature", "toolSignature", left.toolSignature, right.toolSignature],
288
+ ["autonomy", "autonomy", left.autonomy, right.autonomy],
289
+ ["policy", "policyHashes", left.policyHashes, right.policyHashes],
290
+ ["projectContext", "projectContext", left.projectContext, right.projectContext],
291
+ ["corpus", "corpus", left.corpus, right.corpus]
292
+ ];
293
+ return fields.flatMap(
294
+ ([dimension, field, baseline, candidate]) => ignored.has(dimension) || stableJson(baseline) === stableJson(candidate) ? [] : [field]
295
+ );
296
+ }
297
+ function stableJson(value) {
298
+ if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
299
+ if (typeof value === "object" && value !== null) {
300
+ return `{${Object.entries(value).filter(([, entry]) => entry !== void 0).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
301
+ }
302
+ return JSON.stringify(value);
303
+ }
304
+
305
+ // src/domains/eval/compare/behavioral.ts
306
+ function compareEvalBehaviorMetricsV1(baseline, candidate) {
307
+ const baselineGroups = behaviorGroups(baseline);
308
+ const candidateGroups = behaviorGroups(candidate);
309
+ const keys = /* @__PURE__ */ new Set([...baselineGroups.keys(), ...candidateGroups.keys()]);
310
+ const comparisons = [];
311
+ const envelopeMismatches = [];
312
+ const baselineDimensions = baseline.matrix.dimensions ?? [];
313
+ const candidateDimensions = candidate.matrix.dimensions ?? [];
314
+ for (const key of [...keys].sort((left, right) => left.localeCompare(right))) {
315
+ const baselineGroup = baselineGroups.get(key);
316
+ const candidateGroup = candidateGroups.get(key);
317
+ const identity = baselineGroup ?? candidateGroup;
318
+ if (identity === void 0) continue;
319
+ const envelopeMismatch = compareEvalExecutionEnvelopesV1(
320
+ identity,
321
+ baselineGroup?.results ?? [],
322
+ candidateGroup?.results ?? [],
323
+ baselineDimensions,
324
+ candidateDimensions
325
+ );
326
+ if (envelopeMismatch !== null) envelopeMismatches.push(envelopeMismatch);
327
+ const comparability = {
328
+ comparable: envelopeMismatch === null,
329
+ mismatchedFields: envelopeMismatch?.fields ?? []
330
+ };
331
+ for (const definition of EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1) {
332
+ const baselineDistribution = behaviorDistribution(baselineGroup?.results ?? [], definition);
333
+ const candidateDistribution = behaviorDistribution(candidateGroup?.results ?? [], definition);
334
+ comparisons.push({
335
+ scenarioId: identity.scenarioId,
336
+ role: identity.role,
337
+ target: identity.target,
338
+ metric: definition.name,
339
+ family: definition.family,
340
+ direction: definition.direction,
341
+ hardGate: definition.hardGate,
342
+ baseline: baselineDistribution,
343
+ candidate: candidateDistribution,
344
+ change: envelopeMismatch === null ? classifyChange(baselineDistribution.mean, candidateDistribution.mean, definition.direction) : "incomparable",
345
+ meanDelta: subtractNullable(candidateDistribution.mean, baselineDistribution.mean),
346
+ varianceChange: envelopeMismatch === null ? classifyChange(baselineDistribution.variance, candidateDistribution.variance, "lower") : "incomparable",
347
+ varianceDelta: subtractNullable(candidateDistribution.variance, baselineDistribution.variance),
348
+ comparability
349
+ });
350
+ }
351
+ }
352
+ const failures = comparisons.flatMap(
353
+ (comparison) => comparison.hardGate && (comparison.change === "regressed" || comparison.change === "incomparable" && comparison.baseline.mean !== null && comparison.candidate.mean === null) ? [
354
+ {
355
+ scenarioId: comparison.scenarioId,
356
+ role: comparison.role,
357
+ target: comparison.target,
358
+ metric: comparison.metric,
359
+ change: comparison.change
360
+ }
361
+ ] : []
362
+ );
363
+ return {
364
+ comparisons,
365
+ hardGate: {
366
+ pass: failures.length === 0 && envelopeMismatches.length === 0,
367
+ failures,
368
+ envelopeFailures: envelopeMismatches
369
+ },
370
+ envelopeMismatches
371
+ };
372
+ }
373
+ function classifyChange(baseline, candidate, direction) {
374
+ if (baseline === null || candidate === null) return "incomparable";
375
+ if (baseline === candidate) return "unchanged";
376
+ if (direction === "higher") return candidate > baseline ? "improved" : "regressed";
377
+ return candidate < baseline ? "improved" : "regressed";
378
+ }
379
+ function behaviorGroups(artifact) {
380
+ const groups = /* @__PURE__ */ new Map();
381
+ for (const result of artifact.results) {
382
+ const behavioral = result.behavioralMetrics;
383
+ if (behavioral === void 0) continue;
384
+ const key = groupKey(behavioral.scenarioId, behavioral.role, behavioral.target);
385
+ const group = groups.get(key) ?? {
386
+ scenarioId: behavioral.scenarioId,
387
+ role: behavioral.role,
388
+ target: behavioral.target,
389
+ results: []
390
+ };
391
+ group.results.push(result);
392
+ groups.set(key, group);
393
+ }
394
+ return groups;
395
+ }
396
+ function behaviorDistribution(results, definition) {
397
+ const observations = results.map((result) => result.behavioralMetrics?.metrics[definition.name].value ?? null);
398
+ const values = observations.flatMap((value) => value === null ? [] : [value]);
399
+ if (values.length === 0) {
400
+ return {
401
+ observations: observations.length,
402
+ measured: 0,
403
+ unmeasured: observations.length,
404
+ mean: null,
405
+ min: null,
406
+ max: null,
407
+ p90: null,
408
+ variance: null,
409
+ standardDeviation: null,
410
+ source: definition.source
411
+ };
412
+ }
413
+ const ordered = [...values].sort((left, right) => left - right);
414
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
415
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
416
+ return {
417
+ observations: observations.length,
418
+ measured: values.length,
419
+ unmeasured: observations.length - values.length,
420
+ mean,
421
+ min: ordered[0] ?? null,
422
+ max: ordered.at(-1) ?? null,
423
+ p90: ordered[Math.max(0, Math.ceil(ordered.length * 0.9) - 1)] ?? null,
424
+ variance,
425
+ standardDeviation: Math.sqrt(variance),
426
+ source: definition.source
427
+ };
428
+ }
429
+ function groupKey(scenarioId, role, target) {
430
+ return JSON.stringify([scenarioId, role, target.id, target.model]);
431
+ }
432
+ function subtractNullable(left, right) {
433
+ return left === null || right === null ? null : left - right;
434
+ }
435
+
436
+ // src/domains/eval/compare/compare.ts
437
+ var EvalServingConfigurationDriftError = class extends Error {
438
+ baseline;
439
+ candidate;
440
+ constructor(baseline, candidate) {
441
+ super(
442
+ [
443
+ "serving configuration drift; pass --allow-config-drift to compare these runs",
444
+ `baseline serving: ${renderEvalServingConfiguration(baseline)}`,
445
+ `candidate serving: ${renderEvalServingConfiguration(candidate)}`
446
+ ].join("\n")
447
+ );
448
+ this.name = "EvalServingConfigurationDriftError";
449
+ this.baseline = baseline;
450
+ this.candidate = candidate;
451
+ }
452
+ };
453
+ function compareEvalArtifactsV4(baseline, candidate, options = {}) {
454
+ const baselineTokens = baseline.summary.tokens;
455
+ const candidateTokens = candidate.summary.tokens;
456
+ const baselineServing = servingConfigurationOf(baseline);
457
+ const candidateServing = servingConfigurationOf(candidate);
458
+ const configDrift = !sameEvalServingConfiguration(baselineServing, candidateServing);
459
+ if (configDrift && options.allowConfigDrift !== true) {
460
+ throw new EvalServingConfigurationDriftError(baselineServing, candidateServing);
461
+ }
462
+ const behavioral = compareEvalBehaviorMetricsV1(baseline, candidate);
463
+ const trackedMetrics = compareTrackedMetrics(baseline, candidate, options.metric);
464
+ const normalizedFilter = normalizeMetricFilter(options.metric);
465
+ const behavioralMetrics = normalizedFilter === void 0 ? behavioral.comparisons : behavioral.comparisons.filter((row) => row.metric === normalizedFilter || row.family === normalizedFilter);
466
+ if (options.metric !== void 0 && trackedMetrics.length === 0 && behavioralMetrics.length === 0) {
467
+ throw new Error(`eval metric not found: ${options.metric}`);
468
+ }
469
+ return {
470
+ baselineEvalId: baseline.evalId,
471
+ candidateEvalId: candidate.evalId,
472
+ baselineServingConfiguration: baselineServing,
473
+ candidateServingConfiguration: candidateServing,
474
+ configDrift,
475
+ passRateDelta: candidate.summary.passRate - baseline.summary.passRate,
476
+ tokenDelta: baselineTokens.measured && candidateTokens.measured ? candidateTokens.total - baselineTokens.total : null,
477
+ wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs,
478
+ trackedMetrics,
479
+ behavioralMetrics,
480
+ hardGate: behavioral.hardGate,
481
+ envelopeMismatches: behavioral.envelopeMismatches,
482
+ scenarioReports: behaviorRollups(behavioralMetrics, (row) => row.scenarioId),
483
+ roleReports: behaviorRollups(behavioralMetrics, (row) => row.role),
484
+ affectedCorpusResults: behavioral.envelopeMismatches.flatMap((mismatch) => {
485
+ const changedFields = mismatch.fields.filter((field) => field === "prompt" || field === "recipe");
486
+ return changedFields.length === 0 ? [] : [{ scenarioId: mismatch.scenarioId, role: mismatch.role, changedFields }];
487
+ })
488
+ };
489
+ }
490
+ function renderEvalComparisonV4(summary) {
491
+ const envelopeFailures = summary.envelopeMismatches.map(
492
+ (mismatch) => ` incomparable envelope: ${mismatch.scenarioId} ${mismatch.role} ${mismatch.target.id}/${mismatch.target.model ?? "none"} fields=${mismatch.fields.join(",")}`
493
+ );
494
+ const affected = summary.affectedCorpusResults.map(
495
+ (result) => ` affected corpus result: ${result.scenarioId} role=${result.role} changed=${result.changedFields.join(",")}`
496
+ );
497
+ const scenarioReports = renderRollups("per-scenario baseline/candidate report", summary.scenarioReports);
498
+ const roleReports = renderRollups("per-role baseline/candidate report", summary.roleReports);
499
+ const hardFailures = summary.hardGate.failures.map(
500
+ (failure) => ` hard failure: ${failure.scenarioId} ${failure.role} ${failure.target.id}/${failure.target.model ?? "none"} ${failure.metric} ${failure.change}`
501
+ );
502
+ const tracked = summary.trackedMetrics.flatMap((row, index) => [
503
+ ...index === 0 ? [
504
+ "tracked metrics:",
505
+ "scenario metric baseline_mean baseline_p90 baseline_variance candidate_mean candidate_p90 candidate_variance mean_delta p90_delta variance_delta change variance_change sources"
506
+ ] : [],
507
+ [
508
+ row.scenarioId,
509
+ row.metric,
510
+ formatMetric(row.baseline.mean),
511
+ formatMetric(row.baseline.p90),
512
+ formatMetric(row.baseline.variance ?? null),
513
+ formatMetric(row.candidate.mean),
514
+ formatMetric(row.candidate.p90),
515
+ formatMetric(row.candidate.variance ?? null),
516
+ formatSignedMetric(row.meanDelta),
517
+ formatSignedMetric(row.p90Delta),
518
+ formatSignedMetric(row.varianceDelta),
519
+ row.change,
520
+ row.varianceChange,
521
+ `${row.baseline.sources.join("+") || "none"}->${row.candidate.sources.join("+") || "none"}`
522
+ ].join(" ")
523
+ ]);
524
+ const behavioral = summary.behavioralMetrics.flatMap((row, index) => [
525
+ ...index === 0 ? [
526
+ "behavioral metrics:",
527
+ "scenario role target model family metric baseline_mean baseline_variance baseline_coverage candidate_mean candidate_variance candidate_coverage mean_delta variance_delta change variance_change comparability gate source"
528
+ ] : [],
529
+ [
530
+ row.scenarioId,
531
+ row.role,
532
+ row.target.id,
533
+ row.target.model ?? "none",
534
+ row.family,
535
+ row.metric,
536
+ formatMetric(row.baseline.mean),
537
+ formatMetric(row.baseline.variance),
538
+ `${row.baseline.measured}/${row.baseline.observations}`,
539
+ formatMetric(row.candidate.mean),
540
+ formatMetric(row.candidate.variance),
541
+ `${row.candidate.measured}/${row.candidate.observations}`,
542
+ formatSignedMetric(row.meanDelta),
543
+ formatSignedMetric(row.varianceDelta),
544
+ row.change,
545
+ row.varianceChange,
546
+ row.comparability.comparable ? "comparable" : `incomparable:${row.comparability.mismatchedFields.join(",")}`,
547
+ row.hardGate ? "hard" : "informational",
548
+ row.baseline.source
549
+ ].join(" ")
550
+ ]);
551
+ return [
552
+ `baseline eval: ${summary.baselineEvalId}`,
553
+ `candidate eval: ${summary.candidateEvalId}`,
554
+ `baseline serving: ${renderEvalServingConfiguration(summary.baselineServingConfiguration)}`,
555
+ `candidate serving: ${renderEvalServingConfiguration(summary.candidateServingConfiguration)}`,
556
+ `config drift: ${summary.configDrift ? "allowed" : "none"}`,
557
+ `pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
558
+ `token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
559
+ `wall-time delta ms: ${summary.wallTimeDelta}`,
560
+ `behavioral hard gate: ${summary.hardGate.pass ? "pass" : `fail (${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length})`}`,
561
+ ...hardFailures,
562
+ ...envelopeFailures,
563
+ ...affected,
564
+ ...scenarioReports,
565
+ ...roleReports,
566
+ ...tracked,
567
+ ...behavioral,
568
+ ""
569
+ ].join("\n");
570
+ }
571
+ function behaviorRollups(rows, keyOf) {
572
+ const groups = /* @__PURE__ */ new Map();
573
+ for (const row of rows) groups.set(keyOf(row), [...groups.get(keyOf(row)) ?? [], row]);
574
+ return [...groups.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([id, grouped]) => ({
575
+ id,
576
+ metrics: changeCounts(grouped.map((row) => row.change)),
577
+ variance: changeCounts(grouped.map((row) => row.varianceChange))
578
+ }));
579
+ }
580
+ function changeCounts(changes) {
581
+ return {
582
+ improved: changes.filter((change) => change === "improved").length,
583
+ regressed: changes.filter((change) => change === "regressed").length,
584
+ unchanged: changes.filter((change) => change === "unchanged").length,
585
+ incomparable: changes.filter((change) => change === "incomparable").length
586
+ };
587
+ }
588
+ function renderRollups(title, reports) {
589
+ if (reports.length === 0) return [];
590
+ return [
591
+ `${title}:`,
592
+ ...reports.map(
593
+ (report) => ` ${report.id}: metrics ${renderChangeCounts(report.metrics)}; variance ${renderChangeCounts(report.variance)}`
594
+ )
595
+ ];
596
+ }
597
+ function renderChangeCounts(counts) {
598
+ return `improved=${counts.improved} regressed=${counts.regressed} unchanged=${counts.unchanged} incomparable=${counts.incomparable}`;
599
+ }
600
+ function compareTrackedMetrics(baseline, candidate, metricFilter) {
601
+ const baselineAggregates = baseline.aggregates ?? aggregateEvalVerdicts(baseline.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
602
+ const candidateAggregates = candidate.aggregates ?? aggregateEvalVerdicts(candidate.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
603
+ const baselineByScenario = new Map(baselineAggregates.map((entry) => [entry.scenarioId, entry]));
604
+ const candidateByScenario = new Map(candidateAggregates.map((entry) => [entry.scenarioId, entry]));
605
+ const scenarioIds = [...baselineByScenario.keys()].filter((scenarioId) => candidateByScenario.has(scenarioId)).sort((left, right) => left.localeCompare(right));
606
+ const filter = normalizeMetricFilter(metricFilter);
607
+ const rows = [];
608
+ for (const scenarioId of scenarioIds) {
609
+ const baselineAggregate = baselineByScenario.get(scenarioId);
610
+ const candidateAggregate = candidateByScenario.get(scenarioId);
611
+ if (baselineAggregate === void 0 || candidateAggregate === void 0) continue;
612
+ for (const metric of EVAL_TRACKED_METRIC_NAMES) {
613
+ if (filter !== void 0 && filter !== metric) continue;
614
+ rows.push(
615
+ metricComparison(
616
+ scenarioId,
617
+ metric,
618
+ baselineAggregate.trackedMetrics[metric],
619
+ candidateAggregate.trackedMetrics[metric]
620
+ )
621
+ );
622
+ }
623
+ const reasons = /* @__PURE__ */ new Set([
624
+ ...Object.keys(baselineAggregate.trackedMetrics.expectedColdReasons),
625
+ ...Object.keys(candidateAggregate.trackedMetrics.expectedColdReasons)
626
+ ]);
627
+ for (const reason of [...reasons].sort((left, right) => left.localeCompare(right))) {
628
+ const metric = `expectedColdReasons.${reason}`;
629
+ if (filter !== void 0 && filter !== metric && filter !== "expectedColdReasons") continue;
630
+ rows.push(
631
+ metricComparison(
632
+ scenarioId,
633
+ metric,
634
+ baselineAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(baselineAggregate.k),
635
+ candidateAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(candidateAggregate.k)
636
+ )
637
+ );
638
+ }
639
+ }
640
+ return rows;
641
+ }
642
+ function metricComparison(scenarioId, metric, baseline, candidate) {
643
+ assertComparableTrackedMetricSources(`${scenarioId}.${metric}`, baseline.sources, candidate.sources);
644
+ const direction = trackedMetricDirection(metric);
645
+ return {
646
+ scenarioId,
647
+ metric,
648
+ baseline,
649
+ candidate,
650
+ meanDelta: subtractNullable2(candidate.mean, baseline.mean),
651
+ p90Delta: subtractNullable2(candidate.p90, baseline.p90),
652
+ varianceDelta: subtractNullable2(candidate.variance ?? null, baseline.variance ?? null),
653
+ change: classifyChange(baseline.mean, candidate.mean, direction),
654
+ varianceChange: classifyChange(baseline.variance ?? null, candidate.variance ?? null, "lower")
655
+ };
656
+ }
657
+ function servingConfigurationOf(artifact) {
658
+ return artifact.servingConfiguration ?? {
659
+ targetId: artifact.matrix.target,
660
+ runtimeId: null,
661
+ modelId: artifact.matrix.model,
662
+ serverBuild: null,
663
+ total_slots: null,
664
+ thinkingLevel: artifact.matrix.thinking,
665
+ compiledPromptHash: null
666
+ };
667
+ }
668
+ function normalizeMetricFilter(metric) {
669
+ if (metric === void 0) return void 0;
670
+ const trimmed = metric.trim();
671
+ if (trimmed.startsWith("trackedMetrics.")) return trimmed.slice("trackedMetrics.".length);
672
+ if (trimmed.startsWith("behavioralMetrics.")) return trimmed.slice("behavioralMetrics.".length);
673
+ return trimmed;
674
+ }
675
+ function trackedMetricDirection(metric) {
676
+ return metric === "cacheReadTokens" ? "higher" : "lower";
677
+ }
678
+ function zeroDistribution(observations) {
679
+ return {
680
+ observations,
681
+ measured: observations,
682
+ unmeasured: 0,
683
+ mean: 0,
684
+ min: 0,
685
+ max: 0,
686
+ p90: 0,
687
+ variance: 0,
688
+ standardDeviation: 0,
689
+ sources: ["ledger"]
690
+ };
691
+ }
692
+ function subtractNullable2(left, right) {
693
+ return left === null || right === null ? null : left - right;
694
+ }
695
+ function formatMetric(value) {
696
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(2);
697
+ }
698
+ function formatSignedMetric(value) {
699
+ if (value === null) return "null";
700
+ const formatted = formatMetric(value);
701
+ return value > 0 ? `+${formatted}` : formatted;
702
+ }
703
+
704
+ // src/domains/eval/compare/gates.ts
705
+ init_esm_shims();
706
+
707
+ // src/domains/eval/compare/thresholds.ts
708
+ init_esm_shims();
709
+ var import_yaml = __toESM(require_dist(), 1);
710
+ import { readFileSync } from "node:fs";
711
+ function loadThresholds(path) {
712
+ const parsed = (0, import_yaml.parse)(readFileSync(path, "utf8"));
713
+ const root = isRecord(parsed) && isRecord(parsed.thresholds) ? parsed.thresholds : parsed;
714
+ if (isRecord(root) && (Array.isArray(root.fail) || Array.isArray(root.informational))) {
715
+ return {
716
+ fail: parseAssertions(root.fail, `${path}.fail`),
717
+ informational: parseAssertions(root.informational, `${path}.informational`)
718
+ };
719
+ }
720
+ throw new Error(`invalid thresholds file: ${path}`);
721
+ }
722
+ function parseAssertions(value, source) {
723
+ if (value === void 0) return [];
724
+ if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
725
+ return value.map((entry, index) => {
726
+ if (!isRecord(entry)) throw new Error(`${source}[${index}]: expected object`);
727
+ if (typeof entry.metric !== "string" || entry.metric.length === 0) {
728
+ throw new Error(`${source}[${index}].metric: expected non-empty string`);
729
+ }
730
+ if (!isOp(entry.op)) throw new Error(`${source}[${index}].op: expected lt, lte, gt, gte, eq, or neq`);
731
+ if (!isScalar(entry.value)) throw new Error(`${source}[${index}].value: expected scalar`);
732
+ return { metric: entry.metric, op: entry.op, value: entry.value };
733
+ });
734
+ }
735
+ function isOp(value) {
736
+ return value === "lt" || value === "lte" || value === "gt" || value === "gte" || value === "eq" || value === "neq";
737
+ }
738
+ function isScalar(value) {
739
+ return typeof value === "number" && Number.isFinite(value) || typeof value === "string" || typeof value === "boolean";
740
+ }
741
+ function resolveMetricAssertion(assertion, metrics, artifact) {
742
+ const actual = metricValue(assertion.metric, metrics, artifact);
743
+ return { actual, unresolved: actual === null, holds: comparisonHolds(assertion, actual) };
744
+ }
745
+ function comparisonHolds(assertion, actual) {
746
+ switch (assertion.op) {
747
+ case "lt":
748
+ return typeof actual === "number" && typeof assertion.value === "number" && actual < assertion.value;
749
+ case "lte":
750
+ return typeof actual === "number" && typeof assertion.value === "number" && actual <= assertion.value;
751
+ case "gt":
752
+ return typeof actual === "number" && typeof assertion.value === "number" && actual > assertion.value;
753
+ case "gte":
754
+ return typeof actual === "number" && typeof assertion.value === "number" && actual >= assertion.value;
755
+ case "eq":
756
+ return actual === assertion.value;
757
+ case "neq":
758
+ return actual !== assertion.value;
759
+ }
760
+ }
761
+ function metricValue(metric, metrics, artifact) {
762
+ if (metric in metrics) return metrics[metric];
763
+ if (artifact !== void 0) {
764
+ if (metric === "result.pass") return artifact.summary.failed === 0;
765
+ if (metric === "latency.wallMs") return artifact.summary.wallTimeMs;
766
+ if (metric === "tokens.total") return artifact.summary.tokens.measured ? artifact.summary.tokens.total : null;
767
+ if (metric.startsWith("summary.")) return summaryValue(artifact, metric.slice("summary.".length));
768
+ }
769
+ return null;
770
+ }
771
+ function summaryValue(artifact, path) {
772
+ let current = artifact.summary;
773
+ for (const segment of path.split(".")) {
774
+ if (!isRecord(current) || !(segment in current)) return null;
775
+ current = current[segment];
776
+ }
777
+ return typeof current === "number" || typeof current === "string" || typeof current === "boolean" ? current : null;
778
+ }
779
+ function isRecord(value) {
780
+ return typeof value === "object" && value !== null && !Array.isArray(value);
781
+ }
782
+
783
+ // src/domains/eval/compare/gates.ts
784
+ function evaluateGate(artifact, thresholds) {
785
+ const failures = evaluateAssertions(artifact, thresholds.fail);
786
+ const informational = evaluateAssertions(artifact, thresholds.informational ?? []);
787
+ return { pass: failures.length === 0, failures, informational };
788
+ }
789
+ function evaluateAssertions(artifact, assertions) {
790
+ const findings = [];
791
+ for (const assertion of assertions) {
792
+ const whole = resolveMetricAssertion(assertion, {}, artifact);
793
+ if (!whole.unresolved) {
794
+ if (whole.holds) findings.push({ assertion, actual: whole.actual, unresolved: false });
795
+ continue;
796
+ }
797
+ if (artifact.results.length === 0) {
798
+ findings.push({ assertion, actual: null, unresolved: true });
799
+ continue;
800
+ }
801
+ for (const result of artifact.results) {
802
+ const perRun = resolveMetricAssertion(assertion, result.metrics);
803
+ if (!perRun.unresolved && !perRun.holds) continue;
804
+ findings.push({
805
+ assertion,
806
+ actual: perRun.actual,
807
+ unresolved: perRun.unresolved,
808
+ taskId: result.taskId,
809
+ repeatIndex: result.repeatIndex
810
+ });
811
+ }
812
+ }
813
+ return findings;
814
+ }
815
+ function renderGateFailure(failure) {
816
+ const run = failure.taskId === void 0 ? "" : ` [${failure.taskId}#${failure.repeatIndex ?? 0}]`;
817
+ const { metric, op, value } = failure.assertion;
818
+ return failure.unresolved ? ` ${metric}${run}: unresolved metric (fail closed)
819
+ ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(failure.actual)}
820
+ `;
821
+ }
822
+ function renderInformationalBudget(finding) {
823
+ const run = finding.taskId === void 0 ? "" : ` [${finding.taskId}#${finding.repeatIndex ?? 0}]`;
824
+ const { metric, op, value } = finding.assertion;
825
+ return finding.unresolved ? ` ${metric}${run}: unmeasured informational budget
826
+ ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(finding.actual)}
827
+ `;
828
+ }
829
+
830
+ // src/domains/eval/reports/comparison.ts
831
+ init_esm_shims();
832
+ function renderEvalComparisonReportV1(summary, format2) {
833
+ if (format2 === "json") return `${JSON.stringify(summary, null, 2)}
834
+ `;
835
+ if (format2 === "md") return renderMarkdown(summary);
836
+ if (format2 === "junit") return renderJunit(summary);
837
+ return renderEvalComparisonV4(summary);
838
+ }
839
+ function renderMarkdown(summary) {
840
+ const rows = summary.behavioralMetrics.map(
841
+ (row) => `| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`
842
+ );
843
+ const scenarioRows = summary.scenarioReports.map(
844
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
845
+ );
846
+ const roleRows = summary.roleReports.map(
847
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
848
+ );
849
+ return [
850
+ `# Eval comparison ${summary.baselineEvalId} \u2192 ${summary.candidateEvalId}`,
851
+ "",
852
+ `Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
853
+ ...summary.hardGate.failures.map(
854
+ (failure) => `- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`
855
+ ),
856
+ ...summary.envelopeMismatches.map(
857
+ (mismatch) => `- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`
858
+ ),
859
+ ...summary.affectedCorpusResults.map(
860
+ (result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`
861
+ ),
862
+ `Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
863
+ `Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
864
+ `Wall-time delta ms: ${summary.wallTimeDelta}`,
865
+ "",
866
+ "| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
867
+ "|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
868
+ ...rows,
869
+ "",
870
+ "## Per-scenario baseline/candidate report",
871
+ "",
872
+ "| Scenario | Metric changes | Variance changes |",
873
+ "|---|---|---|",
874
+ ...scenarioRows,
875
+ "",
876
+ "## Per-role baseline/candidate report",
877
+ "",
878
+ "| Role | Metric changes | Variance changes |",
879
+ "|---|---|---|",
880
+ ...roleRows,
881
+ ""
882
+ ].join("\n");
883
+ }
884
+ function renderJunit(summary) {
885
+ const failures = new Set(
886
+ summary.hardGate.failures.map(
887
+ (failure) => JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric])
888
+ )
889
+ );
890
+ const represented = /* @__PURE__ */ new Set();
891
+ const cases = summary.behavioralMetrics.map((row) => {
892
+ const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
893
+ const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
894
+ represented.add(key);
895
+ const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
896
+ return failures.has(key) ? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>` : ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
897
+ });
898
+ for (const failure of summary.hardGate.failures) {
899
+ const key = JSON.stringify([
900
+ failure.scenarioId,
901
+ failure.role,
902
+ failure.target.id,
903
+ failure.target.model,
904
+ failure.metric
905
+ ]);
906
+ if (represented.has(key)) continue;
907
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
908
+ cases.push(
909
+ ` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`
910
+ );
911
+ }
912
+ for (const failure of summary.hardGate.envelopeFailures) {
913
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
914
+ cases.push(
915
+ ` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`
916
+ );
917
+ }
918
+ return [
919
+ `<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
920
+ ...cases,
921
+ "</testsuite>",
922
+ ""
923
+ ].join("\n");
924
+ }
925
+ function changeCounts2(counts) {
926
+ return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
927
+ }
928
+ function format(value) {
929
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
930
+ }
931
+ function cell(value) {
932
+ return value.replaceAll("|", "\\|");
933
+ }
934
+ function escapeXml(value) {
935
+ return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
936
+ }
937
+
938
+ // src/domains/eval/reports/json.ts
939
+ init_esm_shims();
940
+ function renderEvalJsonReportV4(artifact) {
941
+ return `${JSON.stringify(artifact, null, 2)}
942
+ `;
943
+ }
944
+
945
+ // src/domains/eval/reports/junit.ts
946
+ init_esm_shims();
947
+ function renderEvalJunitReportV4(artifact) {
948
+ let failures = 0;
949
+ let skipped = 0;
950
+ const cases = artifact.results.map((result) => {
951
+ const name = escapeXml2(
952
+ `${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`
953
+ );
954
+ if (!result.pass) {
955
+ failures += 1;
956
+ return ` <testcase name="${name}"><failure message="${escapeXml2(result.failureClass ?? "failed")}" /></testcase>`;
957
+ }
958
+ const outcome = result.behavioral?.outcome;
959
+ if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
960
+ failures += 1;
961
+ return ` <testcase name="${name}"><failure message="${escapeXml2(outcome)}" /></testcase>`;
962
+ }
963
+ if (outcome === "unknown" || outcome === "unmeasured") {
964
+ skipped += 1;
965
+ return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml2(outcome)}" /></testcase>`;
966
+ }
967
+ return ` <testcase name="${name}" />`;
968
+ }).join("\n");
969
+ return [
970
+ `<testsuite name="${escapeXml2(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
971
+ cases,
972
+ "</testsuite>",
973
+ ""
974
+ ].join("\n");
975
+ }
976
+ function escapeXml2(value) {
977
+ return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
978
+ }
979
+
980
+ // src/domains/eval/reports/markdown.ts
981
+ init_esm_shims();
982
+ function renderEvalMarkdownReportV4(artifact) {
983
+ const lines = [
984
+ `# Eval ${artifact.evalId}`,
985
+ "",
986
+ `Suite: ${artifact.suite.id}`,
987
+ `Target: ${artifact.matrix.target}`,
988
+ `Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
989
+ "",
990
+ "| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
991
+ "|---|---|---|---|---:|---|---|---|",
992
+ ...artifact.results.map(
993
+ (result) => `| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`
994
+ ),
995
+ ""
996
+ ];
997
+ return `${lines.join("\n")}`;
998
+ }
999
+
1000
+ // src/domains/eval/reports/swe-jsonl.ts
1001
+ init_esm_shims();
1002
+ function renderEvalSweJsonlReportV4(artifact) {
1003
+ return artifact.results.map(
1004
+ (result) => JSON.stringify({
1005
+ instance_id: result.taskId,
1006
+ model_name_or_path: artifact.matrix.model ?? artifact.matrix.target,
1007
+ model_patch: typeof result.artifacts.patch === "string" ? result.artifacts.patch : "",
1008
+ status: result.pass ? "pass" : "fail",
1009
+ pass: result.pass
1010
+ })
1011
+ ).join("\n").concat(artifact.results.length === 0 ? "" : "\n");
1012
+ }
1013
+
1014
+ // src/domains/eval/reports/text.ts
1015
+ init_esm_shims();
1016
+ function renderEvalTextReportV4(artifact) {
1017
+ const tokens = artifact.summary.tokens;
1018
+ const behavioral = artifact.results.flatMap(
1019
+ (result) => result.behavioral === void 0 ? [] : [result.behavioral.outcome]
1020
+ );
1021
+ const behavioralSummary = behavioral.length === 0 ? [] : [
1022
+ `behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`
1023
+ ];
1024
+ return [
1025
+ `eval: ${artifact.evalId}`,
1026
+ `suite: ${artifact.suite.id}`,
1027
+ `target: ${artifact.matrix.target}`,
1028
+ `model: ${artifact.matrix.model ?? "none"}`,
1029
+ `runs: ${artifact.summary.runs}`,
1030
+ `passed: ${artifact.summary.passed}`,
1031
+ `failed: ${artifact.summary.failed}`,
1032
+ `pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
1033
+ // A run whose child Clio work is out of the harness's sight has no token
1034
+ // count. Saying "0" would claim it cost nothing, so the count is
1035
+ // reported next to how many runs it actually covers.
1036
+ !tokens.measured ? `tokens total: unmeasured (0 of ${tokens.runs} runs reported usage)` : tokens.measuredRuns === tokens.runs ? `tokens total: ${tokens.total}` : `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
1037
+ `wall time ms: ${artifact.summary.wallTimeMs}`,
1038
+ ...behavioralSummary,
1039
+ ""
1040
+ ].join("\n");
1041
+ }
1042
+ function count(values, wanted) {
1043
+ return values.filter((value) => value === wanted).length;
1044
+ }
1045
+
1046
+ // src/domains/eval/suites/load.ts
1047
+ init_esm_shims();
1048
+ var import_yaml2 = __toESM(require_dist(), 1);
1049
+ import { createHash } from "node:crypto";
1050
+ import { readFile } from "node:fs/promises";
1051
+ import { dirname, resolve } from "node:path";
1052
+
1053
+ // src/domains/eval/schema/validate.ts
1054
+ init_esm_shims();
1055
+
1056
+ // src/domains/eval/schema/suite.ts
1057
+ init_esm_shims();
1058
+ var EVAL_SUITE_V2_VERSION = 2;
1059
+
1060
+ // src/domains/eval/schema/validate.ts
1061
+ var RUNNER_KINDS = /* @__PURE__ */ new Set(["clio-run", "context-index", "context-init", "external-command"]);
1062
+ var WORKSPACE_KINDS = /* @__PURE__ */ new Set(["local", "git", "temp-copy"]);
1063
+ var OPS = /* @__PURE__ */ new Set(["lt", "lte", "gt", "gte", "eq", "neq"]);
1064
+ function validateEvalSuiteV2(value) {
1065
+ const issues = [];
1066
+ if (!isRecord2(value)) return { valid: false, issues: [{ path: "$", message: "expected object" }] };
1067
+ if (value.version !== EVAL_SUITE_V2_VERSION) issues.push({ path: "$.version", message: "expected version 2" });
1068
+ const suite = readSuiteInfo(value.suite, "$.suite", issues);
1069
+ const matrix = readMatrix(value.matrix, "$.matrix", issues);
1070
+ const tasks = readTasks(value.tasks, "$.tasks", issues);
1071
+ const thresholds = readThresholds(value.thresholds, "$.thresholds", issues);
1072
+ if (issues.length > 0 || suite === null || matrix === null || tasks === null) return { valid: false, issues };
1073
+ return {
1074
+ valid: true,
1075
+ suite: {
1076
+ version: 2,
1077
+ suite,
1078
+ matrix,
1079
+ tasks,
1080
+ ...thresholds === void 0 ? {} : { thresholds }
1081
+ }
1082
+ };
1083
+ }
1084
+ function readSuiteInfo(value, path, issues) {
1085
+ if (!isRecord2(value)) {
1086
+ issues.push({ path, message: "expected object" });
1087
+ return null;
1088
+ }
1089
+ const id = readId(value, path, "id", issues);
1090
+ const title = readNonEmptyString(value, path, "title", issues);
1091
+ const visibility = readNonEmptyString(value, path, "visibility", issues);
1092
+ if (id === null || title === null || visibility === null) return null;
1093
+ const description = optionalString(value, "description");
1094
+ const provenance = isRecord2(value.provenance) ? value.provenance : void 0;
1095
+ return {
1096
+ id,
1097
+ title,
1098
+ visibility,
1099
+ ...description === void 0 ? {} : { description },
1100
+ ...provenance ? { provenance } : {}
1101
+ };
1102
+ }
1103
+ function readMatrix(value, path, issues) {
1104
+ if (!isRecord2(value)) {
1105
+ issues.push({ path, message: "expected object" });
1106
+ return null;
1107
+ }
1108
+ const repeats = readPositiveInteger(value, path, "repeats", issues);
1109
+ if (!Array.isArray(value.targets) || value.targets.length === 0) {
1110
+ issues.push({ path: `${path}.targets`, message: "expected non-empty array" });
1111
+ return null;
1112
+ }
1113
+ const targets = value.targets.flatMap((target, index) => {
1114
+ if (!isRecord2(target)) {
1115
+ issues.push({ path: `${path}.targets[${index}]`, message: "expected object" });
1116
+ return [];
1117
+ }
1118
+ const id = readId(target, `${path}.targets[${index}]`, "id", issues);
1119
+ if (id === null) return [];
1120
+ const suiteTarget = { id };
1121
+ const model = optionalString(target, "model");
1122
+ const thinking = optionalString(target, "thinking");
1123
+ return [
1124
+ { ...suiteTarget, ...model === void 0 ? {} : { model }, ...thinking === void 0 ? {} : { thinking } }
1125
+ ];
1126
+ });
1127
+ if (repeats === null || targets.length === 0) return null;
1128
+ let dimensions;
1129
+ if (value.dimensions !== void 0) {
1130
+ try {
1131
+ dimensions = parseEvalExecutionMatrixDimensionsV1(value.dimensions, `${path}.dimensions`);
1132
+ } catch (error) {
1133
+ issues.push({ path: `${path}.dimensions`, message: error instanceof Error ? error.message : String(error) });
1134
+ return null;
1135
+ }
1136
+ }
1137
+ const maxCostUsd = value.maxCostUsd;
1138
+ if (maxCostUsd !== void 0 && (typeof maxCostUsd !== "number" || !Number.isFinite(maxCostUsd) || maxCostUsd < 0)) {
1139
+ issues.push({ path: `${path}.maxCostUsd`, message: "expected non-negative number" });
1140
+ return null;
1141
+ }
1142
+ return {
1143
+ targets,
1144
+ repeats,
1145
+ ...dimensions === void 0 ? {} : { dimensions },
1146
+ ...maxCostUsd === void 0 ? {} : { maxCostUsd }
1147
+ };
1148
+ }
1149
+ function readTasks(value, path, issues) {
1150
+ if (!Array.isArray(value) || value.length === 0) {
1151
+ issues.push({ path, message: "expected non-empty array" });
1152
+ return null;
1153
+ }
1154
+ const seen = /* @__PURE__ */ new Set();
1155
+ const tasks = [];
1156
+ for (let index = 0; index < value.length; index += 1) {
1157
+ const task = readTask(value[index], `${path}[${index}]`, issues);
1158
+ if (task === null) continue;
1159
+ if (seen.has(task.id)) issues.push({ path: `${path}[${index}].id`, message: `duplicate task id: ${task.id}` });
1160
+ seen.add(task.id);
1161
+ tasks.push(task);
1162
+ }
1163
+ return tasks.length > 0 ? tasks : null;
1164
+ }
1165
+ function readTask(value, path, issues) {
1166
+ if (!isRecord2(value)) {
1167
+ issues.push({ path, message: "expected object" });
1168
+ return null;
1169
+ }
1170
+ const id = readId(value, path, "id", issues);
1171
+ const workspace = readWorkspace(value.workspace, `${path}.workspace`, issues);
1172
+ const runner = readRunner(value.runner, `${path}.runner`, issues);
1173
+ const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
1174
+ const behavioral = readBehavioral(value.behavioral, `${path}.behavioral`, issues);
1175
+ if (id === null || workspace === null || runner === null || timeoutMs === null) return null;
1176
+ return {
1177
+ id,
1178
+ tags: readOptionalStringArray(value, "tags", `${path}.tags`, issues),
1179
+ ...behavioral === void 0 ? {} : { behavioral },
1180
+ workspace,
1181
+ runner,
1182
+ verify: readVerify(value.verify, `${path}.verify`, issues),
1183
+ metrics: readMetrics(value.metrics, `${path}.metrics`, issues),
1184
+ timeoutMs
1185
+ };
1186
+ }
1187
+ function readBehavioral(value, path, issues) {
1188
+ if (value === void 0) return void 0;
1189
+ try {
1190
+ return parseEvalBehaviorScenarioV1(value, path);
1191
+ } catch (error) {
1192
+ issues.push({ path, message: error instanceof Error ? error.message : String(error) });
1193
+ return void 0;
1194
+ }
1195
+ }
1196
+ function readWorkspace(value, path, issues) {
1197
+ if (!isRecord2(value)) {
1198
+ issues.push({ path, message: "expected object" });
1199
+ return null;
1200
+ }
1201
+ const kind = value.kind;
1202
+ if (typeof kind !== "string" || !WORKSPACE_KINDS.has(kind)) {
1203
+ issues.push({ path: `${path}.kind`, message: "expected local, git, or temp-copy" });
1204
+ return null;
1205
+ }
1206
+ const pathValue = optionalString(value, "path");
1207
+ const url = optionalString(value, "url");
1208
+ const commit = optionalString(value, "commit");
1209
+ const checkout = optionalString(value, "checkout");
1210
+ return {
1211
+ kind,
1212
+ ...pathValue === void 0 ? {} : { path: pathValue },
1213
+ ...url === void 0 ? {} : { url },
1214
+ ...commit === void 0 ? {} : { commit },
1215
+ ...checkout === void 0 ? {} : { checkout },
1216
+ excludes: readOptionalStringArray(value, "excludes", `${path}.excludes`, issues),
1217
+ setup: readOptionalStringArray(value, "setup", `${path}.setup`, issues)
1218
+ };
1219
+ }
1220
+ function readRunner(value, path, issues) {
1221
+ if (!isRecord2(value)) {
1222
+ issues.push({ path, message: "expected object" });
1223
+ return null;
1224
+ }
1225
+ const kind = value.kind;
1226
+ if (typeof kind !== "string" || !RUNNER_KINDS.has(kind)) {
1227
+ issues.push({ path: `${path}.kind`, message: "expected clio-run, context-index, context-init, or external-command" });
1228
+ return null;
1229
+ }
1230
+ const prompt = optionalString(value, "prompt");
1231
+ const agent = optionalString(value, "agent");
1232
+ const autonomy = optionalString(value, "autonomy");
1233
+ if (autonomy !== void 0 && !["read-only", "suggest", "auto-edit", "full-auto"].includes(autonomy)) {
1234
+ issues.push({ path: `${path}.autonomy`, message: "expected read-only, suggest, auto-edit, or full-auto" });
1235
+ }
1236
+ if (agent !== void 0 && kind !== "clio-run") {
1237
+ issues.push({ path: `${path}.agent`, message: "agent is only valid on the clio-run runner" });
1238
+ }
1239
+ const command = optionalString(value, "command");
1240
+ return {
1241
+ kind,
1242
+ ...prompt === void 0 ? {} : { prompt },
1243
+ ...autonomy === void 0 ? {} : { autonomy },
1244
+ ...agent === void 0 ? {} : { agent },
1245
+ ...command === void 0 ? {} : { command },
1246
+ commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
1247
+ args: readOptionalStringArray(value, "args", `${path}.args`, issues),
1248
+ ...typeof value.timeoutMs === "number" ? { timeoutMs: value.timeoutMs } : {}
1249
+ };
1250
+ }
1251
+ function readVerify(value, path, issues) {
1252
+ if (value === void 0) return {};
1253
+ if (!isRecord2(value)) {
1254
+ issues.push({ path, message: "expected object" });
1255
+ return {};
1256
+ }
1257
+ return {
1258
+ commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
1259
+ measure: readOptionalStringArray(value, "measure", `${path}.measure`, issues),
1260
+ assertions: readAssertions(value.assertions, `${path}.assertions`, issues),
1261
+ forbidPaths: readOptionalStringArray(value, "forbidPaths", `${path}.forbidPaths`, issues)
1262
+ };
1263
+ }
1264
+ function readMetrics(value, path, issues) {
1265
+ if (value === void 0) return { collect: [] };
1266
+ if (!isRecord2(value)) {
1267
+ issues.push({ path, message: "expected object" });
1268
+ return { collect: [] };
1269
+ }
1270
+ const observation = value.readObservation;
1271
+ let readObservation;
1272
+ if (observation !== void 0) {
1273
+ if (!isRecord2(observation)) {
1274
+ issues.push({ path: `${path}.readObservation`, message: "expected object" });
1275
+ } else {
1276
+ readObservation = {
1277
+ allowedPaths: readOptionalStringArray(observation, "allowedPaths", `${path}.readObservation.allowedPaths`, issues),
1278
+ decoyPaths: readOptionalStringArray(observation, "decoyPaths", `${path}.readObservation.decoyPaths`, issues)
1279
+ };
1280
+ }
1281
+ }
1282
+ return {
1283
+ collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues),
1284
+ ...readObservation === void 0 ? {} : { readObservation }
1285
+ };
1286
+ }
1287
+ function readThresholds(value, path, issues) {
1288
+ if (value === void 0) return void 0;
1289
+ if (!isRecord2(value)) {
1290
+ issues.push({ path, message: "expected object" });
1291
+ return void 0;
1292
+ }
1293
+ return {
1294
+ fail: readAssertions(value.fail, `${path}.fail`, issues),
1295
+ informational: readAssertions(value.informational, `${path}.informational`, issues)
1296
+ };
1297
+ }
1298
+ function readAssertions(value, path, issues) {
1299
+ if (value === void 0) return [];
1300
+ if (!Array.isArray(value)) {
1301
+ issues.push({ path, message: "expected array" });
1302
+ return [];
1303
+ }
1304
+ const out = [];
1305
+ for (let index = 0; index < value.length; index += 1) {
1306
+ const entry = value[index];
1307
+ if (!isRecord2(entry)) {
1308
+ issues.push({ path: `${path}[${index}]`, message: "expected object" });
1309
+ continue;
1310
+ }
1311
+ const metric = readNonEmptyString(entry, `${path}[${index}]`, "metric", issues);
1312
+ const op = entry.op;
1313
+ if (typeof op !== "string" || !OPS.has(op)) {
1314
+ issues.push({ path: `${path}[${index}].op`, message: "expected lt, lte, gt, gte, eq, or neq" });
1315
+ continue;
1316
+ }
1317
+ const valueField = entry.value;
1318
+ if (metric === null || !["number", "string", "boolean"].includes(typeof valueField)) {
1319
+ issues.push({ path: `${path}[${index}].value`, message: "expected number, string, or boolean" });
1320
+ continue;
1321
+ }
1322
+ out.push({ metric, op, value: valueField });
1323
+ }
1324
+ return out;
1325
+ }
1326
+ function readId(record, path, field, issues) {
1327
+ const value = readNonEmptyString(record, path, field, issues);
1328
+ if (value !== null && !/^[A-Za-z0-9._-]+$/.test(value)) {
1329
+ issues.push({
1330
+ path: `${path}.${field}`,
1331
+ message: "expected id with letters, numbers, dots, underscores, or hyphens"
1332
+ });
1333
+ }
1334
+ return value;
1335
+ }
1336
+ function readNonEmptyString(record, path, field, issues) {
1337
+ const value = record[field];
1338
+ if (typeof value !== "string" || value.trim().length === 0) {
1339
+ issues.push({ path: `${path}.${field}`, message: "expected non-empty string" });
1340
+ return null;
1341
+ }
1342
+ return value;
1343
+ }
1344
+ function readPositiveInteger(record, path, field, issues) {
1345
+ const value = record[field];
1346
+ if (typeof value !== "number" || !Number.isInteger(value) || value <= 0) {
1347
+ issues.push({ path: `${path}.${field}`, message: "expected positive integer" });
1348
+ return null;
1349
+ }
1350
+ return value;
1351
+ }
1352
+ function readOptionalStringArray(record, field, path, issues) {
1353
+ const value = record[field];
1354
+ if (value === void 0) return [];
1355
+ if (!Array.isArray(value)) {
1356
+ issues.push({ path, message: "expected string array" });
1357
+ return [];
1358
+ }
1359
+ return value.flatMap((entry, index) => {
1360
+ if (typeof entry === "string" && entry.trim().length > 0) return [entry];
1361
+ issues.push({ path: `${path}[${index}]`, message: "expected non-empty string" });
1362
+ return [];
1363
+ });
1364
+ }
1365
+ function optionalString(record, field) {
1366
+ const value = record[field];
1367
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1368
+ }
1369
+ function isRecord2(value) {
1370
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1371
+ }
1372
+
1373
+ // src/domains/eval/suites/load.ts
1374
+ var EvalSuiteFileError = class extends Error {
1375
+ issues;
1376
+ constructor(issues) {
1377
+ super(`eval suite invalid (${issues.length} ${issues.length === 1 ? "issue" : "issues"})`);
1378
+ this.name = "EvalSuiteFileError";
1379
+ this.issues = [...issues];
1380
+ }
1381
+ };
1382
+ async function loadEvalSuiteFile(path) {
1383
+ const resolved = resolve(path);
1384
+ const raw = await readFile(resolved, "utf8");
1385
+ let parsed;
1386
+ try {
1387
+ parsed = (0, import_yaml2.parse)(raw);
1388
+ } catch (error) {
1389
+ const message = error instanceof Error ? error.message : String(error);
1390
+ throw new EvalSuiteFileError([{ path: "$", message: `invalid YAML: ${message}` }]);
1391
+ }
1392
+ const result = validateEvalSuiteV2(parsed);
1393
+ if (!result.valid) throw new EvalSuiteFileError(result.issues);
1394
+ return {
1395
+ path: resolved,
1396
+ baseDir: dirname(resolved),
1397
+ hash: sha256Hex(raw),
1398
+ suite: result.suite
1399
+ };
1400
+ }
1401
+ async function loadV1TaskFileAsSuite(path, repeatOverride) {
1402
+ const loaded = await loadEvalTaskFile(path);
1403
+ const suite = {
1404
+ version: 2,
1405
+ suite: {
1406
+ id: "v1-task-file",
1407
+ title: "v1 task file",
1408
+ visibility: "local",
1409
+ description: "Compatibility adapter for version 1 eval task files.",
1410
+ provenance: { taskFileVersion: 1 }
1411
+ },
1412
+ matrix: {
1413
+ targets: [{ id: "local" }],
1414
+ repeats: repeatOverride ?? 1
1415
+ },
1416
+ tasks: loaded.taskFile.tasks.map((task) => ({
1417
+ id: task.id,
1418
+ tags: task.tags,
1419
+ workspace: { kind: "local", path: task.cwd },
1420
+ runner: { kind: "external-command", commands: task.setup },
1421
+ verify: { commands: task.verifier },
1422
+ metrics: { collect: ["latency.wallMs", "verifier.exitCode"] },
1423
+ timeoutMs: task.timeoutMs
1424
+ }))
1425
+ };
1426
+ return {
1427
+ path: loaded.path,
1428
+ baseDir: loaded.baseDir,
1429
+ hash: loaded.contentHash,
1430
+ suite
1431
+ };
1432
+ }
1433
+ function sha256Hex(content) {
1434
+ return createHash("sha256").update(content, "utf8").digest("hex");
1435
+ }
1436
+
1437
+ // src/domains/eval/suites/resolve.ts
1438
+ init_esm_shims();
1439
+ function resolveSuiteForRun(suite, options) {
1440
+ const targets = resolveTargets(suite.matrix.targets, options);
1441
+ return {
1442
+ ...suite,
1443
+ matrix: {
1444
+ ...suite.matrix,
1445
+ targets,
1446
+ ...options.trials === void 0 ? {} : { repeats: options.trials }
1447
+ }
1448
+ };
1449
+ }
1450
+ function artifactMatrixIdentity(targets) {
1451
+ if (targets.length === 1) {
1452
+ const target = targets[0];
1453
+ return {
1454
+ target: target?.id ?? "unknown",
1455
+ model: target?.model ?? null,
1456
+ thinking: target?.thinking ?? null
1457
+ };
1458
+ }
1459
+ return { target: "multiple", model: null, thinking: null };
1460
+ }
1461
+ function resolveTargets(targets, options) {
1462
+ const filtered = options.target === void 0 ? [...targets] : targets.filter((target) => target.id === options.target);
1463
+ const selected = filtered.length > 0 ? filtered : options.target === void 0 ? [...targets] : [{ id: options.target }];
1464
+ return selected.map((target) => ({
1465
+ ...target,
1466
+ ...options.model === void 0 ? {} : { model: options.model }
1467
+ }));
1468
+ }
1469
+
1470
+ // src/domains/eval/suites/run.ts
1471
+ init_esm_shims();
1472
+ import { mkdtemp as mkdtemp3, rm as rm3, writeFile } from "node:fs/promises";
1473
+ import { tmpdir as tmpdir3 } from "node:os";
1474
+ import { resolve as resolve7 } from "node:path";
1475
+
1476
+ // src/domains/eval/execution-provenance.ts
1477
+ init_esm_shims();
1478
+ import { createHash as createHash2 } from "node:crypto";
1479
+ function buildEvalExecutionEnvelopeV1(input) {
1480
+ const scenario = input.task.behavioral;
1481
+ if (scenario === void 0) throw new Error(`behavioral task ${input.task.id} has no behavioral scenario`);
1482
+ const manifest = input.ledger.promptManifests.at(-1) ?? null;
1483
+ const contextSnapshot = input.ledger.contextSnapshots.at(-1) ?? null;
1484
+ const recipe = recipeIdentity(
1485
+ input,
1486
+ scenario.execution.subject.kind === "worker" ? scenario.execution.subject.role : null
1487
+ );
1488
+ const policy = policyIdentity(input.cwd, input.receipt, input.observation);
1489
+ const autonomy = input.receipt?.autonomyEnforcement?.autonomy ?? input.observation?.autonomy ?? input.task.runner.autonomy ?? null;
1490
+ const promptFragments = promptFragmentIdentities(manifest, recipe, autonomy);
1491
+ const compositionHash = input.receipt?.staticCompositionHash ?? input.observation?.compositionHash ?? manifest?.systemPromptHash ?? contextSnapshot?.promptHash ?? null;
1492
+ const projectContext = projectContextIdentity(input, manifest, promptFragments);
1493
+ return {
1494
+ schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
1495
+ prompt: { fragments: promptFragments, compositionHash },
1496
+ recipe: recipe === null ? null : { id: recipe.id, version: recipe.version, contentHash: recipe.contentHash },
1497
+ target: input.receipt?.targetId ?? input.observation?.target ?? input.target.id,
1498
+ wireModel: input.receipt?.wireModelId ?? input.observation?.wireModel ?? contextSnapshot?.modelId ?? input.target.model ?? null,
1499
+ runtime: input.receipt?.runtimeId ?? input.observation?.runtime ?? contextSnapshot?.runtimeId ?? null,
1500
+ thinkingLevel: input.receipt?.runtimeResolution?.effectiveThinkingLevel ?? input.observation?.thinkingLevel ?? manifest?.thinkingLevel ?? input.target.thinking ?? null,
1501
+ toolSignature: input.receipt?.toolSignature ?? input.observation?.toolSignature ?? contextSnapshot?.toolSignature ?? null,
1502
+ autonomy,
1503
+ policyHashes: policy,
1504
+ projectContext,
1505
+ corpus: { ...scenario.corpus }
1506
+ };
1507
+ }
1508
+ function recipeIdentity(input, role) {
1509
+ const id = input.receipt?.agentId ?? input.task.runner.agent ?? role;
1510
+ if (id === null || input.cwd === null) return null;
1511
+ try {
1512
+ const recipe = discoverAgentRecipes(input.cwd).find((entry) => entry.id === id);
1513
+ if (recipe === void 0) return null;
1514
+ return {
1515
+ id: recipe.id,
1516
+ version: recipe.version,
1517
+ contentHash: agentSpecFingerprint(normalizeAgentSpec(recipe)),
1518
+ personaHash: sha256(recipe.body)
1519
+ };
1520
+ } catch {
1521
+ return null;
1522
+ }
1523
+ }
1524
+ function promptFragmentIdentities(manifest, recipe, autonomy) {
1525
+ let versions = /* @__PURE__ */ new Map();
1526
+ try {
1527
+ versions = new Map([...loadFragments().byId.values()].map((fragment) => [fragment.id, fragment.version]));
1528
+ } catch {
1529
+ }
1530
+ if (manifest !== null) {
1531
+ return manifest.fragments.map((fragment) => ({
1532
+ id: fragment.id,
1533
+ version: versions.get(fragment.id) ?? "unversioned",
1534
+ contentHash: fragment.contentHash
1535
+ })).sort((left, right) => left.id.localeCompare(right.id));
1536
+ }
1537
+ if (recipe === null) return [];
1538
+ const selected = ["identity.clio-worker", "operating.contract", "operating.worker"];
1539
+ if (autonomy !== null) selected.push(`safety.${autonomy}`);
1540
+ const fragments = [];
1541
+ try {
1542
+ const table = loadFragments();
1543
+ for (const id of selected) {
1544
+ const fragment = table.byId.get(id);
1545
+ if (fragment !== void 0) {
1546
+ fragments.push({ id, version: fragment.version, contentHash: fragment.contentHash });
1547
+ }
1548
+ }
1549
+ } catch {
1550
+ }
1551
+ fragments.push({ id: `persona.${recipe.id}`, version: recipe.version, contentHash: recipe.personaHash });
1552
+ return fragments.sort((left, right) => left.id.localeCompare(right.id));
1553
+ }
1554
+ function policyIdentity(cwd, receipt, observation) {
1555
+ const sealed = receipt?.reproducibility?.safetyPolicy;
1556
+ if (sealed !== void 0) return { rulePack: sealed.rulePackHash, project: sealed.projectPolicyHash };
1557
+ if (observation !== void 0) return { ...observation.policyHashes };
1558
+ if (cwd === null) return { rulePack: null, project: null };
1559
+ try {
1560
+ const metadata = createSafetyPolicyEngine({ cwd }).metadata();
1561
+ return { rulePack: metadata.rulePackHash, project: metadata.projectPolicyHash };
1562
+ } catch {
1563
+ return { rulePack: null, project: null };
1564
+ }
1565
+ }
1566
+ function projectContextIdentity(input, manifest, fragments) {
1567
+ const receipt = input.receipt;
1568
+ if (receipt?.projectContext !== void 0) {
1569
+ const sections = [...receipt.projectContext.sections ?? []].sort();
1570
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1571
+ return {
1572
+ kind: "worker",
1573
+ tier: receipt.projectContext.tier,
1574
+ contentHash: hasContentBearingContext ? receipt.projectContext.contentHash ?? null : null,
1575
+ chars: hasContentBearingContext ? receipt.projectContext.chars ?? null : null,
1576
+ sections,
1577
+ rulesApplied: [...receipt.rulesApplied ?? []].sort(),
1578
+ operatorProfileApplied: receipt.operatorProfileApplied ?? null
1579
+ };
1580
+ }
1581
+ const observed = input.observation?.projectContext;
1582
+ if (observed !== void 0 && observed !== null) {
1583
+ const sections = [...observed.sections].sort();
1584
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1585
+ return {
1586
+ kind: "worker",
1587
+ tier: observed.tier,
1588
+ contentHash: hasContentBearingContext ? observed.contentHash : null,
1589
+ chars: hasContentBearingContext ? observed.chars : null,
1590
+ sections,
1591
+ rulesApplied: [...observed.rulesApplied].sort(),
1592
+ operatorProfileApplied: observed.operatorProfileApplied
1593
+ };
1594
+ }
1595
+ if (manifest !== null) {
1596
+ const contextFragments = fragments.filter((fragment) => fragment.id.startsWith("context."));
1597
+ const preload = manifest.projectPreload;
1598
+ const identity = {
1599
+ preload,
1600
+ fragments: contextFragments.map((fragment) => [fragment.id, fragment.contentHash])
1601
+ };
1602
+ return {
1603
+ kind: "session",
1604
+ tier: preload?.mode ?? null,
1605
+ contentHash: sha256(stableJson2(identity)),
1606
+ chars: preload?.chars ?? null,
1607
+ sections: contextFragments.map((fragment) => fragment.id).sort(),
1608
+ rulesApplied: [],
1609
+ operatorProfileApplied: contextFragments.some((fragment) => fragment.id === "context.operator-profile")
1610
+ };
1611
+ }
1612
+ return {
1613
+ kind: "none",
1614
+ tier: null,
1615
+ contentHash: null,
1616
+ chars: null,
1617
+ sections: [],
1618
+ rulesApplied: [],
1619
+ operatorProfileApplied: null
1620
+ };
1621
+ }
1622
+ function stableJson2(value) {
1623
+ if (Array.isArray(value)) return `[${value.map(stableJson2).join(",")}]`;
1624
+ if (typeof value === "object" && value !== null) {
1625
+ return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson2(entry)}`).join(",")}}`;
1626
+ }
1627
+ return JSON.stringify(value);
1628
+ }
1629
+ function sha256(value) {
1630
+ return createHash2("sha256").update(value, "utf8").digest("hex");
1631
+ }
1632
+
1633
+ // src/domains/eval/metrics/context.ts
1634
+ init_esm_shims();
1635
+ function collectContextMetrics(cwd) {
1636
+ try {
1637
+ const codewiki = readCodewiki(cwd);
1638
+ if (!codewiki) throw new Error("codewiki unavailable");
1639
+ const visiblePaths = new Set(enumerateWorkspaceFiles(cwd));
1640
+ const artifactSourceFiles = codewiki.files.filter((file) => file.lang !== "config");
1641
+ const indexedFiles = artifactSourceFiles.filter((file) => visiblePaths.has(file.path)).length;
1642
+ const staleFiles = artifactSourceFiles.length - indexedFiles;
1643
+ const sourceFiles = detectProjectProfile(cwd).sourceFiles;
1644
+ const digest = renderCodewikiDigest(codewiki);
1645
+ return {
1646
+ "context.indexedFiles": indexedFiles,
1647
+ "context.artifactFiles": artifactSourceFiles.length,
1648
+ "context.staleFiles": staleFiles,
1649
+ "context.coverage": sourceFiles === 0 ? artifactSourceFiles.length === 0 ? 1 : 0 : indexedFiles / sourceFiles,
1650
+ "context.structuralHash": structuralCodewikiHash(codewiki),
1651
+ "context.digestTokens": Math.ceil(digest.length / 4)
1652
+ };
1653
+ } catch {
1654
+ return {
1655
+ "context.indexedFiles": 0,
1656
+ "context.artifactFiles": 0,
1657
+ "context.staleFiles": 0,
1658
+ "context.coverage": 0,
1659
+ "context.structuralHash": null,
1660
+ "context.digestTokens": 0
1661
+ };
1662
+ }
1663
+ }
1664
+
1665
+ // src/domains/eval/metrics/fleet-loop-stream.ts
1666
+ init_esm_shims();
1667
+ var EMPTY_FLEET_LOOP_OBSERVATION = { summaryCount: 0, loops: [] };
1668
+ var DECLARED_LOOP_REASONS = /* @__PURE__ */ new Set([
1669
+ "resolved",
1670
+ "loop_bound_exhausted",
1671
+ "loop_step_failed",
1672
+ "loop_not_reached"
1673
+ ]);
1674
+ function createFleetLoopFold() {
1675
+ let summaryCount = 0;
1676
+ let loops = [];
1677
+ let pending = "";
1678
+ const consume = (line) => {
1679
+ if (line.trim().length === 0) return;
1680
+ let parsed;
1681
+ try {
1682
+ parsed = JSON.parse(line);
1683
+ } catch {
1684
+ return;
1685
+ }
1686
+ if (!isRecord3(parsed) || typeof parsed.fleet !== "string" || !Array.isArray(parsed.loops)) return;
1687
+ summaryCount += 1;
1688
+ const unneeded = Array.isArray(parsed.unneeded) ? parsed.unneeded.length : 0;
1689
+ const skipped = Array.isArray(parsed.skipped) ? parsed.skipped.length : 0;
1690
+ loops = parsed.loops.flatMap((entry) => {
1691
+ const loop = parseLoop(entry, unneeded, skipped);
1692
+ return loop === null ? [] : [loop];
1693
+ });
1694
+ };
1695
+ return {
1696
+ push(chunk) {
1697
+ pending += chunk;
1698
+ for (; ; ) {
1699
+ const newline = pending.indexOf("\n");
1700
+ if (newline === -1) break;
1701
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
1702
+ pending = pending.slice(newline + 1);
1703
+ }
1704
+ },
1705
+ observation() {
1706
+ if (pending.length > 0) {
1707
+ consume(pending.replace(/\r$/u, ""));
1708
+ pending = "";
1709
+ }
1710
+ return { summaryCount, loops: loops.map((loop) => ({ ...loop })) };
1711
+ }
1712
+ };
1713
+ }
1714
+ function addFleetLoopObservations(left, right) {
1715
+ const summaryCount = left.summaryCount + right.summaryCount;
1716
+ return { summaryCount, loops: summaryCount === 1 ? [...left.loops, ...right.loops] : [] };
1717
+ }
1718
+ function fleetLoopMetricEntries(observation) {
1719
+ if (observation.summaryCount !== 1 || observation.loops.length === 0) return {};
1720
+ const loops = observation.loops;
1721
+ return {
1722
+ "loop.count": loops.length,
1723
+ "loop.attemptsSpent": sum(loops.map((loop) => loop.attempts)),
1724
+ "loop.repairsSpent": sum(loops.map((loop) => loop.repairs)),
1725
+ "loop.resolved": loops.every((loop) => loop.resolved),
1726
+ "loop.reasonExhausted": loops.every((loop) => loop.resolved || loop.reason === "loop_bound_exhausted"),
1727
+ "loop.reasonDeclared": loops.every(
1728
+ (loop) => DECLARED_LOOP_REASONS.has(loop.reason) && loop.resolved === (loop.reason === "resolved")
1729
+ ),
1730
+ "loop.unneededNodes": loops[0]?.unneeded ?? 0,
1731
+ "loop.skippedNodes": loops[0]?.skipped ?? 0
1732
+ };
1733
+ }
1734
+ function fleetLoopReceiptAgreement(loopMetrics, journalMetrics) {
1735
+ const repairs = loopMetrics["loop.repairsSpent"];
1736
+ const recoveries = journalMetrics["receipt.recoveryCount"];
1737
+ if (typeof repairs !== "number" || typeof recoveries !== "number") return {};
1738
+ return { "loop.receiptsMatchRepairs": repairs === recoveries };
1739
+ }
1740
+ function parseLoop(entry, unneeded, skipped) {
1741
+ if (!isRecord3(entry)) return null;
1742
+ const { attempts, repairs, resolved, reason } = entry;
1743
+ if (!isFiniteInteger(attempts) || attempts < 0) return null;
1744
+ if (!isFiniteInteger(repairs) || repairs < 0) return null;
1745
+ if (typeof resolved !== "boolean" || typeof reason !== "string" || reason.length === 0) return null;
1746
+ return { attempts, repairs, resolved, reason, unneeded, skipped };
1747
+ }
1748
+ function sum(values) {
1749
+ return values.reduce((total, value) => total + value, 0);
1750
+ }
1751
+ function isFiniteInteger(value) {
1752
+ return typeof value === "number" && Number.isFinite(value) && Number.isInteger(value);
1753
+ }
1754
+ function isRecord3(value) {
1755
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1756
+ }
1757
+
1758
+ // src/domains/eval/metrics/invariants.ts
1759
+ init_esm_shims();
1760
+ import { existsSync, readdirSync, readFileSync as readFileSync2 } from "node:fs";
1761
+ import { join } from "node:path";
1762
+ function readRunJournal(stateDir) {
1763
+ if (!existsSync(stateDir)) return null;
1764
+ return {
1765
+ envelopes: readEnvelopes(join(stateDir, "runs.json")),
1766
+ ...readReceipts(join(stateDir, "receipts"))
1767
+ };
1768
+ }
1769
+ function readEnvelopes(runsPath) {
1770
+ const envelopes = /* @__PURE__ */ new Map();
1771
+ if (!existsSync(runsPath)) return envelopes;
1772
+ let parsed;
1773
+ try {
1774
+ parsed = JSON.parse(readFileSync2(runsPath, "utf8"));
1775
+ } catch {
1776
+ return envelopes;
1777
+ }
1778
+ if (!Array.isArray(parsed)) return envelopes;
1779
+ for (const entry of parsed) {
1780
+ if (isRecord4(entry) && typeof entry.id === "string") envelopes.set(entry.id, entry);
1781
+ }
1782
+ return envelopes;
1783
+ }
1784
+ function readReceipts(receiptsDir) {
1785
+ if (!existsSync(receiptsDir)) return { receipts: [], receiptFiles: 0 };
1786
+ let names;
1787
+ try {
1788
+ names = readdirSync(receiptsDir).filter((name) => name.endsWith(".json"));
1789
+ } catch {
1790
+ return { receipts: [], receiptFiles: 0 };
1791
+ }
1792
+ const receipts = [];
1793
+ for (const name of names) {
1794
+ try {
1795
+ const parsed = JSON.parse(readFileSync2(join(receiptsDir, name), "utf8"));
1796
+ if (isReceiptShaped(parsed)) receipts.push(parsed);
1797
+ } catch {
1798
+ }
1799
+ }
1800
+ return { receipts, receiptFiles: names.length };
1801
+ }
1802
+ function receiptInvariantMetrics(journal, processExitCode) {
1803
+ if (journal === null) return {};
1804
+ const sealed = journal.receiptFiles > 0;
1805
+ const roots = journal.receipts.filter(isRootReceipt);
1806
+ const base = {
1807
+ "receipt.count": journal.receiptFiles,
1808
+ "receipt.sealed": sealed,
1809
+ "receipt.rootCount": roots.length,
1810
+ // Every attempt after the first is `recovery`. A bounded loop's repairs
1811
+ // are exactly those attempts, so this is what a loop's attempt accounting
1812
+ // is checked against.
1813
+ "receipt.recoveryCount": journal.receipts.filter((receipt) => receipt.executionRole === "recovery").length
1814
+ };
1815
+ if (!sealed) return base;
1816
+ return {
1817
+ ...base,
1818
+ "receipt.integrityValid": integrityValid(journal),
1819
+ "receipt.outcomeMatchesExit": outcomeMatchesExit(journal, roots, processExitCode)
1820
+ };
1821
+ }
1822
+ function receiptUsageMetrics(journal) {
1823
+ if (journal === null) return {};
1824
+ if (journal.receiptFiles === 0) return { "receiptUsage.measured": false };
1825
+ if (!integrityValid(journal)) return { "receiptUsage.measured": false };
1826
+ let totalTokens = 0;
1827
+ let costUsd = 0;
1828
+ for (const receipt of journal.receipts) {
1829
+ totalTokens += finiteNumber(receipt.tokenCount);
1830
+ costUsd += finiteNumber(receipt.costUsd);
1831
+ }
1832
+ return {
1833
+ "receiptUsage.measured": true,
1834
+ "receiptUsage.receiptCount": journal.receipts.length,
1835
+ "receiptUsage.totalTokens": totalTokens,
1836
+ "receiptUsage.costUsd": costUsd
1837
+ };
1838
+ }
1839
+ function finiteNumber(value) {
1840
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
1841
+ }
1842
+ function integrityValid(journal) {
1843
+ if (journal.receipts.length !== journal.receiptFiles) return false;
1844
+ return journal.receipts.every((receipt) => {
1845
+ const envelope = journal.envelopes.get(receipt.runId);
1846
+ if (envelope === void 0) return false;
1847
+ return verifyReceiptIntegrity(receipt, envelope).ok;
1848
+ });
1849
+ }
1850
+ function outcomeMatchesExit(journal, roots, processExitCode) {
1851
+ if (journal.receipts.length !== journal.receiptFiles) return false;
1852
+ const selfConsistent = journal.receipts.every(
1853
+ (receipt) => receipt.outcome === "succeeded" === (receipt.exitCode === 0)
1854
+ );
1855
+ if (!selfConsistent) return false;
1856
+ const root = roots.length === 1 ? roots[0] : void 0;
1857
+ if (root === void 0) return true;
1858
+ return root.exitCode === 0 === (processExitCode === 0);
1859
+ }
1860
+ function isRootReceipt(receipt) {
1861
+ return receipt.lineage === void 0 || receipt.lineage.rootRunId === receipt.runId;
1862
+ }
1863
+ function sessionInvariantMetrics(stateDir) {
1864
+ const sessionDirs = findSessionDirs(join(stateDir, "sessions"));
1865
+ if (sessionDirs.length === 0) return {};
1866
+ let formatVersion = Number.POSITIVE_INFINITY;
1867
+ let toolPairsUnmatched = 0;
1868
+ let assistantBetweenCallAndResult = 0;
1869
+ let compactionSummaryPresent = false;
1870
+ let answeredFromPreCompaction = false;
1871
+ let turnsAfterCompaction = 0;
1872
+ for (const sessionDir of sessionDirs) {
1873
+ const transcript = readSessionTranscript(join(sessionDir, "current.jsonl"));
1874
+ formatVersion = Math.min(formatVersion, transcript.formatVersion);
1875
+ toolPairsUnmatched += transcript.toolPairsUnmatched;
1876
+ assistantBetweenCallAndResult += transcript.assistantBetweenCallAndResult;
1877
+ compactionSummaryPresent ||= transcript.compactionSummaryPresent;
1878
+ answeredFromPreCompaction ||= transcript.answeredFromPreCompaction;
1879
+ turnsAfterCompaction += transcript.turnsAfterCompaction;
1880
+ }
1881
+ return {
1882
+ "ledger.formatVersion": formatVersion,
1883
+ "ledger.toolPairsUnmatched": toolPairsUnmatched,
1884
+ "ledger.assistantBetweenCallAndResult": assistantBetweenCallAndResult,
1885
+ "ledger.sessionCount": sessionDirs.length,
1886
+ "continuity.compactionSummaryPresent": compactionSummaryPresent,
1887
+ "continuity.answeredFromPreCompaction": answeredFromPreCompaction,
1888
+ "continuity.turnsAfterCompaction": turnsAfterCompaction
1889
+ };
1890
+ }
1891
+ function findSessionDirs(sessionsRoot) {
1892
+ try {
1893
+ const dirs = [];
1894
+ for (const cwdDir of readdirSync(sessionsRoot, { withFileTypes: true })) {
1895
+ if (!cwdDir.isDirectory()) continue;
1896
+ const cwdPath = join(sessionsRoot, cwdDir.name);
1897
+ try {
1898
+ for (const sessionDir of readdirSync(cwdPath, { withFileTypes: true })) {
1899
+ if (sessionDir.isDirectory()) dirs.push(join(cwdPath, sessionDir.name));
1900
+ }
1901
+ } catch {
1902
+ }
1903
+ }
1904
+ return dirs;
1905
+ } catch {
1906
+ return [];
1907
+ }
1908
+ }
1909
+ function readSessionTranscript(path) {
1910
+ let lines;
1911
+ try {
1912
+ lines = readFileSync2(path, "utf8").split(/\r?\n/u);
1913
+ } catch {
1914
+ return {
1915
+ formatVersion: 0,
1916
+ toolPairsUnmatched: 0,
1917
+ assistantBetweenCallAndResult: 0,
1918
+ compactionSummaryPresent: false,
1919
+ answeredFromPreCompaction: false,
1920
+ turnsAfterCompaction: 0
1921
+ };
1922
+ }
1923
+ const entries = lines.map(parseJsonRecord);
1924
+ const header = entries[0];
1925
+ const formatVersion = header?.type === "session" && Number.isInteger(header.version) && header.version > 0 ? header.version : 0;
1926
+ const pendingCalls = /* @__PURE__ */ new Map();
1927
+ let invalidCalls = 0;
1928
+ let orphanResults = 0;
1929
+ let assistantGeneration = 0;
1930
+ let assistantBetweenCallAndResult = 0;
1931
+ const readCalls = /* @__PURE__ */ new Map();
1932
+ const completedReads = [];
1933
+ let latestCompactionIndex = -1;
1934
+ for (const [entryIndex, entry] of entries.entries()) {
1935
+ if (isCompactionSummary(entry)) latestCompactionIndex = entryIndex;
1936
+ if (entry?.kind !== "message") continue;
1937
+ if (entry.role === "assistant") {
1938
+ assistantGeneration += 1;
1939
+ continue;
1940
+ }
1941
+ if (entry.role !== "tool_call" && entry.role !== "tool_result") continue;
1942
+ const payload = isRecord4(entry.payload) ? entry.payload : void 0;
1943
+ const toolCallId = payload?.toolCallId;
1944
+ if (entry.role === "tool_call") {
1945
+ if (typeof toolCallId !== "string" || toolCallId.length === 0) {
1946
+ invalidCalls += 1;
1947
+ continue;
1948
+ }
1949
+ const calls2 = pendingCalls.get(toolCallId) ?? [];
1950
+ calls2.push(assistantGeneration);
1951
+ pendingCalls.set(toolCallId, calls2);
1952
+ if (payload?.name === "read") {
1953
+ const args = isRecord4(payload.args) ? payload.args : void 0;
1954
+ const readPath2 = args?.path;
1955
+ if (typeof readPath2 === "string" && readPath2.trim().length > 0) {
1956
+ const paths = readCalls.get(toolCallId) ?? [];
1957
+ paths.push(readPath2.trim());
1958
+ readCalls.set(toolCallId, paths);
1959
+ }
1960
+ }
1961
+ continue;
1962
+ }
1963
+ if (typeof toolCallId !== "string" || toolCallId.length === 0) {
1964
+ orphanResults += 1;
1965
+ continue;
1966
+ }
1967
+ const calls = pendingCalls.get(toolCallId);
1968
+ const callGeneration = calls?.shift();
1969
+ if (callGeneration === void 0) {
1970
+ orphanResults += 1;
1971
+ continue;
1972
+ }
1973
+ if (calls?.length === 0) pendingCalls.delete(toolCallId);
1974
+ if (assistantGeneration > callGeneration) assistantBetweenCallAndResult += 1;
1975
+ const readPaths = readCalls.get(toolCallId);
1976
+ const readPath = readPaths?.shift();
1977
+ if (readPaths?.length === 0) readCalls.delete(toolCallId);
1978
+ if (readPath !== void 0 && payload?.isError !== true) completedReads.push({ path: readPath, entryIndex });
1979
+ }
1980
+ let danglingCalls = invalidCalls;
1981
+ for (const calls of pendingCalls.values()) danglingCalls += calls.length;
1982
+ const pathsReadBeforeCompaction = new Set(
1983
+ completedReads.filter((read) => read.entryIndex < latestCompactionIndex).map((read) => read.path)
1984
+ );
1985
+ const answeredFromPreCompaction = completedReads.some(
1986
+ (read) => read.entryIndex > latestCompactionIndex && pathsReadBeforeCompaction.has(read.path)
1987
+ );
1988
+ return {
1989
+ formatVersion,
1990
+ toolPairsUnmatched: danglingCalls + orphanResults,
1991
+ assistantBetweenCallAndResult,
1992
+ compactionSummaryPresent: latestCompactionIndex >= 0,
1993
+ answeredFromPreCompaction: latestCompactionIndex >= 0 && answeredFromPreCompaction,
1994
+ turnsAfterCompaction: latestCompactionIndex < 0 ? 0 : entries.slice(latestCompactionIndex + 1).filter((entry) => entry?.kind === "message").length
1995
+ };
1996
+ }
1997
+ function isCompactionSummary(entry) {
1998
+ return entry?.kind === "compactionSummary" && typeof entry.summary === "string" && typeof entry.tokensBefore === "number" && Number.isFinite(entry.tokensBefore) && typeof entry.firstKeptTurnId === "string";
1999
+ }
2000
+ function parseJsonRecord(line) {
2001
+ if (line.trim().length === 0) return void 0;
2002
+ try {
2003
+ const parsed = JSON.parse(line);
2004
+ return isRecord4(parsed) ? parsed : void 0;
2005
+ } catch {
2006
+ return void 0;
2007
+ }
2008
+ }
2009
+ function processInvariantMetrics(journal) {
2010
+ if (journal === null) return {};
2011
+ const attested = journal.receipts.flatMap(
2012
+ (receipt) => receipt.attestation === void 0 ? [] : [receipt.attestation]
2013
+ );
2014
+ if (attested.length === 0) return {};
2015
+ const orphaned = attested.filter((attestation) => {
2016
+ if (isProcessAlive(attestation.pid)) return true;
2017
+ const group = attestation.processGroupId;
2018
+ return group !== null && group === attestation.pid && isProcessAlive(-group);
2019
+ });
2020
+ return { "process.attestedWorkers": attested.length, "process.orphanedChildren": orphaned.length };
2021
+ }
2022
+ function isProcessAlive(pid) {
2023
+ if (!Number.isInteger(pid) || pid === 0) return false;
2024
+ try {
2025
+ process.kill(pid, 0);
2026
+ return true;
2027
+ } catch (error) {
2028
+ return error.code !== "ESRCH";
2029
+ }
2030
+ }
2031
+ function writeBoundaryInvariantMetrics(stateDir) {
2032
+ const files = findWriteBoundaryVerdicts(join(stateDir, "write-boundaries"));
2033
+ if (files.length === 0) return {};
2034
+ let sealed = true;
2035
+ let violationsDetected = 0;
2036
+ let violationsRolledBack = 0;
2037
+ let rollbackIncomplete = 0;
2038
+ for (const file of files) {
2039
+ const verdict = parseJsonFile(file);
2040
+ if (verdict === void 0) {
2041
+ sealed = false;
2042
+ continue;
2043
+ }
2044
+ if (!isNonEmptyString(verdict.digest) || !isNonEmptyString(verdict.baselineHead)) sealed = false;
2045
+ if (verdict.reason === WRITES_BOUNDARY_VIOLATION) violationsDetected += 1;
2046
+ if (verdict.status === "rolled-back") violationsRolledBack += 1;
2047
+ if (verdict.status === "rollback-incomplete") rollbackIncomplete += 1;
2048
+ }
2049
+ return {
2050
+ "boundary.verdictCount": files.length,
2051
+ "boundary.verdictSealed": sealed,
2052
+ "boundary.violationsDetected": violationsDetected,
2053
+ "boundary.violationsRolledBack": violationsRolledBack,
2054
+ "boundary.rollbackIncomplete": rollbackIncomplete
2055
+ };
2056
+ }
2057
+ var WRITES_BOUNDARY_VIOLATION = "writes_boundary_violation";
2058
+ function findWriteBoundaryVerdicts(root) {
2059
+ const files = [];
2060
+ try {
2061
+ for (const rootDir of readdirSync(root, { withFileTypes: true })) {
2062
+ if (!rootDir.isDirectory()) continue;
2063
+ const dir = join(root, rootDir.name);
2064
+ try {
2065
+ for (const entry of readdirSync(dir, { withFileTypes: true })) {
2066
+ if (entry.isFile() && entry.name.endsWith(".json")) files.push(join(dir, entry.name));
2067
+ }
2068
+ } catch {
2069
+ }
2070
+ }
2071
+ } catch {
2072
+ return [];
2073
+ }
2074
+ return files;
2075
+ }
2076
+ function parseJsonFile(path) {
2077
+ try {
2078
+ const parsed = JSON.parse(readFileSync2(path, "utf8"));
2079
+ return isRecord4(parsed) ? parsed : void 0;
2080
+ } catch {
2081
+ return void 0;
2082
+ }
2083
+ }
2084
+ function isNonEmptyString(value) {
2085
+ return typeof value === "string" && value.length > 0;
2086
+ }
2087
+ var EMPTY_STREAM_INVARIANTS = {
2088
+ messageUpdateCount: 0,
2089
+ cumulativeSnapshots: 0,
2090
+ usageMessages: 0,
2091
+ repeatedResponses: 0,
2092
+ identifiedResponses: 0,
2093
+ messageTokenTotal: 0,
2094
+ segmentTokenTotal: 0,
2095
+ measuredSegments: 0
2096
+ };
2097
+ function addStreamInvariants(left, right) {
2098
+ return {
2099
+ messageUpdateCount: left.messageUpdateCount + right.messageUpdateCount,
2100
+ cumulativeSnapshots: left.cumulativeSnapshots + right.cumulativeSnapshots,
2101
+ usageMessages: left.usageMessages + right.usageMessages,
2102
+ repeatedResponses: left.repeatedResponses + right.repeatedResponses,
2103
+ identifiedResponses: left.identifiedResponses + right.identifiedResponses,
2104
+ messageTokenTotal: left.messageTokenTotal + right.messageTokenTotal,
2105
+ segmentTokenTotal: left.segmentTokenTotal + right.segmentTokenTotal,
2106
+ measuredSegments: left.measuredSegments + right.measuredSegments
2107
+ };
2108
+ }
2109
+ function createStreamInvariantFold() {
2110
+ const state = {
2111
+ messageUpdateCount: 0,
2112
+ cumulativeSnapshots: 0,
2113
+ usageMessages: 0,
2114
+ repeatedResponses: 0,
2115
+ identifiedResponses: 0,
2116
+ messageTokenTotal: 0,
2117
+ segmentTokenTotal: 0,
2118
+ measuredSegments: 0
2119
+ };
2120
+ const seenResponses = /* @__PURE__ */ new Set();
2121
+ let pending = "";
2122
+ const consume = (line) => {
2123
+ if (line.trim().length === 0) return;
2124
+ let event;
2125
+ try {
2126
+ event = JSON.parse(line);
2127
+ } catch {
2128
+ return;
2129
+ }
2130
+ if (!isRecord4(event)) return;
2131
+ if (event.type === "message_update") {
2132
+ state.messageUpdateCount += 1;
2133
+ const assistantEvent = isRecord4(event.assistantMessageEvent) ? event.assistantMessageEvent : void 0;
2134
+ if (event.message !== void 0 || assistantEvent?.partial !== void 0) state.cumulativeSnapshots += 1;
2135
+ return;
2136
+ }
2137
+ if (event.type === "agent_end") {
2138
+ if (Array.isArray(event.messages)) state.cumulativeSnapshots += 1;
2139
+ const usage2 = isRecord4(event.usage) ? event.usage : void 0;
2140
+ if (usage2 === void 0 || usage2.measured !== true) return;
2141
+ state.measuredSegments += 1;
2142
+ state.segmentTokenTotal += numberField(usage2, "totalTokens");
2143
+ return;
2144
+ }
2145
+ if (event.type !== "message_end") return;
2146
+ const message = isRecord4(event.message) ? event.message : void 0;
2147
+ if (message === void 0 || message.role !== "assistant") return;
2148
+ const usage = isRecord4(message.usage) ? message.usage : void 0;
2149
+ if (usage === void 0) return;
2150
+ state.usageMessages += 1;
2151
+ const input = numberField(usage, "input");
2152
+ const output = numberField(usage, "output");
2153
+ const cacheRead = numberField(usage, "cacheRead");
2154
+ const cacheWrite = numberField(usage, "cacheWrite");
2155
+ const totalTokens = numberField(usage, "totalTokens");
2156
+ state.messageTokenTotal += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
2157
+ const responseId = message.responseId;
2158
+ if (typeof responseId !== "string" || responseId.length === 0) return;
2159
+ state.identifiedResponses += 1;
2160
+ if (seenResponses.has(responseId)) state.repeatedResponses += 1;
2161
+ seenResponses.add(responseId);
2162
+ };
2163
+ return {
2164
+ push(chunk) {
2165
+ pending += chunk;
2166
+ for (; ; ) {
2167
+ const newline = pending.indexOf("\n");
2168
+ if (newline === -1) break;
2169
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
2170
+ pending = pending.slice(newline + 1);
2171
+ }
2172
+ },
2173
+ invariants() {
2174
+ if (pending.length > 0) {
2175
+ consume(pending.replace(/\r$/u, ""));
2176
+ pending = "";
2177
+ }
2178
+ return { ...state };
2179
+ }
2180
+ };
2181
+ }
2182
+ function streamInvariantMetrics(stream) {
2183
+ return {
2184
+ "stream.messageUpdateCount": stream.messageUpdateCount,
2185
+ "stream.cumulativeSnapshots": stream.cumulativeSnapshots,
2186
+ ...stream.identifiedResponses === 0 ? {} : { "stream.usageDoubleCounted": stream.repeatedResponses > 0 },
2187
+ ...stream.measuredSegments === 0 ? {} : { "stream.segmentUsageMatchesMessages": stream.segmentTokenTotal === stream.messageTokenTotal }
2188
+ };
2189
+ }
2190
+ function numberField(record, field) {
2191
+ const value = record[field];
2192
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
2193
+ }
2194
+ function isReceiptShaped(value) {
2195
+ if (!isRecord4(value)) return false;
2196
+ return typeof value.runId === "string" && typeof value.agentId === "string" && typeof value.exitCode === "number" && typeof value.outcome === "string" && isRecord4(value.integrity);
2197
+ }
2198
+ function isRecord4(value) {
2199
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2200
+ }
2201
+
2202
+ // src/domains/eval/metrics/latency.ts
2203
+ init_esm_shims();
2204
+ function wallTimeMetric(metrics) {
2205
+ const value = metrics["latency.wallMs"];
2206
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
2207
+ }
2208
+
2209
+ // src/domains/eval/metrics/tokens.ts
2210
+ init_esm_shims();
2211
+
2212
+ // src/domains/eval/schema/artifact.ts
2213
+ init_esm_shims();
2214
+ var ZERO_TOKEN_METRICS_V4 = {
2215
+ input: 0,
2216
+ output: 0,
2217
+ total: 0,
2218
+ cacheRead: 0,
2219
+ cacheWrite: 0
2220
+ };
2221
+
2222
+ // src/domains/eval/metrics/coverage.ts
2223
+ init_esm_shims();
2224
+ function tokenMeasurementCoverage(results) {
2225
+ let measured = 0;
2226
+ for (const result of results) {
2227
+ if (result.metrics["tokens.measured"] === true) measured += 1;
2228
+ }
2229
+ return { total: results.length, measured };
2230
+ }
2231
+
2232
+ // src/domains/eval/metrics/tokens.ts
2233
+ function tokenAccountingFrom(results) {
2234
+ const coverage = tokenMeasurementCoverage(results);
2235
+ if (coverage.measured === 0) return { measured: false, runs: coverage.total, measuredRuns: 0 };
2236
+ const totals = results.reduce(
2237
+ (total, result) => result.metrics["tokens.measured"] === true ? addTokenMetrics(total, tokenMetricsFrom(result.metrics)) : total,
2238
+ zeroTokenMetrics()
2239
+ );
2240
+ return { measured: true, runs: coverage.total, measuredRuns: coverage.measured, ...totals };
2241
+ }
2242
+ function tokenMetricsFrom(metrics) {
2243
+ return {
2244
+ input: numberMetric(metrics, "tokens.input"),
2245
+ output: numberMetric(metrics, "tokens.output"),
2246
+ total: numberMetric(metrics, "tokens.total"),
2247
+ cacheRead: numberMetric(metrics, "tokens.cacheRead"),
2248
+ cacheWrite: numberMetric(metrics, "tokens.cacheWrite")
2249
+ };
2250
+ }
2251
+ function addTokenMetrics(left, right) {
2252
+ return {
2253
+ input: left.input + right.input,
2254
+ output: left.output + right.output,
2255
+ total: left.total + right.total,
2256
+ cacheRead: left.cacheRead + right.cacheRead,
2257
+ cacheWrite: left.cacheWrite + right.cacheWrite
2258
+ };
2259
+ }
2260
+ function zeroTokenMetrics() {
2261
+ return { ...ZERO_TOKEN_METRICS_V4 };
2262
+ }
2263
+ function numberMetric(metrics, key) {
2264
+ const value = metrics[key];
2265
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
2266
+ }
2267
+
2268
+ // src/domains/eval/metrics/tool-calls.ts
2269
+ init_esm_shims();
2270
+ function zeroToolCallMetrics() {
2271
+ return {
2272
+ "tools.totalCalls": 0,
2273
+ "tools.failed": 0,
2274
+ "tools.blocked": 0
2275
+ };
2276
+ }
2277
+
2278
+ // src/domains/eval/metrics/tracked.ts
2279
+ init_esm_shims();
2280
+ import { readFile as readFile2 } from "node:fs/promises";
2281
+ import { dirname as dirname2, join as join2 } from "node:path";
2282
+ async function readEvalLedgerSnapshot(stateDir) {
2283
+ const refs = await listSessionLedgerRefs(stateDir);
2284
+ const entries = [];
2285
+ const compiledPromptHashes = [];
2286
+ const promptManifests = [];
2287
+ const contextSnapshots = [];
2288
+ for (const ref of refs) {
2289
+ try {
2290
+ const raw = await readFile2(ref.path, "utf8");
2291
+ entries.push(...parseSessionEntries(raw, ref.path).entries);
2292
+ } catch {
2293
+ }
2294
+ try {
2295
+ const manifest = await readFile2(join2(dirname2(ref.path), "prompt-manifest.jsonl"), "utf8");
2296
+ for (const line of manifest.split(/\r?\n/u)) {
2297
+ const record = parseJsonRecord2(line);
2298
+ if (record === null) continue;
2299
+ const hash = record?.systemPromptHash;
2300
+ if (typeof hash !== "string" || !/^[a-f0-9]{64}$/u.test(hash)) continue;
2301
+ compiledPromptHashes.push(hash);
2302
+ const observation = promptManifestObservation(record);
2303
+ if (observation !== null) promptManifests.push(observation);
2304
+ }
2305
+ } catch {
2306
+ }
2307
+ try {
2308
+ const snapshots = await readFile2(join2(dirname2(ref.path), "context-snapshots.jsonl"), "utf8");
2309
+ for (const line of snapshots.split(/\r?\n/u)) {
2310
+ const record = parseJsonRecord2(line);
2311
+ if (record === null) continue;
2312
+ contextSnapshots.push({
2313
+ runtimeId: nullableString(record.runtimeId),
2314
+ modelId: nullableString(record.modelId),
2315
+ promptHash: nullableDigest(record.promptHash),
2316
+ toolSignature: nullableDigest(record.toolSignature)
2317
+ });
2318
+ }
2319
+ } catch {
2320
+ }
2321
+ }
2322
+ return { entries, compiledPromptHashes: [...new Set(compiledPromptHashes)], promptManifests, contextSnapshots };
2323
+ }
2324
+ function buildEvalTrackedMetrics(input) {
2325
+ const calls = assistantCalls(input.ledgerEntries);
2326
+ const compactionEntries = input.ledgerEntries.filter((entry) => entry.kind === "compactionSummary");
2327
+ const compactionUsage = compactionEntries.flatMap((entry) => {
2328
+ if (entry.kind !== "compactionSummary" || !isRecord5(entry.usage)) return [];
2329
+ return [entry.usage];
2330
+ });
2331
+ const modelCallReadings = calls.map(() => ledgerReading(1));
2332
+ for (const entry of compactionEntries) {
2333
+ if (entry.kind !== "compactionSummary") continue;
2334
+ const apiCalls = isRecord5(entry.usage) ? nonNegativeNumber(entry.usage.apiCalls) : null;
2335
+ modelCallReadings.push(apiCalls === null ? estimatedReading(1) : ledgerReading(apiCalls));
2336
+ }
2337
+ const uncachedReadings = calls.map(uncachedPrefillForCall);
2338
+ const cacheReadings = calls.map(cacheReadForCall);
2339
+ const generatedReadings = calls.map(generatedForCall);
2340
+ for (const usage of compactionUsage) {
2341
+ uncachedReadings.push(readingFromUsage(usage, "input"));
2342
+ cacheReadings.push(readingFromUsage(usage, "cacheRead"));
2343
+ generatedReadings.push(readingFromUsage(usage, "output"));
2344
+ }
2345
+ const reasoning = reasoningMetric(input.receipt, calls, compactionUsage);
2346
+ const receiptToolMetrics = input.receipt === null ? null : evalHarnessMetricsFromReceipt(input.receipt);
2347
+ const ledgerToolCalls = input.ledgerEntries.filter(
2348
+ (entry) => entry.kind === "message" && entry.role === "tool_call"
2349
+ ).length;
2350
+ const ledgerToolErrors = input.ledgerEntries.filter((entry) => {
2351
+ if (entry.kind !== "message" || entry.role !== "tool_result" || !isRecord5(entry.payload)) return false;
2352
+ return entry.payload.isError === true || entry.payload.outcome === "error";
2353
+ }).length;
2354
+ const receiptToolErrors = input.receipt?.toolStats.reduce((sum2, stat) => sum2 + finiteNonNegative(stat.errors), 0);
2355
+ const expectedColdReasons = expectedColdReasonMetrics(calls);
2356
+ return {
2357
+ modelCalls: sumReadings(modelCallReadings, "ledger"),
2358
+ uncachedPrefillTokens: sumReadings(uncachedReadings, "estimated"),
2359
+ cacheReadTokens: sumReadings(cacheReadings, "estimated"),
2360
+ generatedTokens: sumReadings(generatedReadings, "estimated"),
2361
+ reasoningTokens: reasoning,
2362
+ toolCalls: receiptToolMetrics === null ? { value: ledgerToolCalls, source: "ledger" } : { value: receiptToolMetrics.toolCalls, source: "receipt" },
2363
+ toolErrors: receiptToolErrors === void 0 ? { value: ledgerToolErrors, source: "ledger" } : { value: receiptToolErrors, source: "receipt" },
2364
+ ttftMsFirstCall: firstCallTtft(calls),
2365
+ wallClockMs: wallClockMetric(input.receipt, input.fallbackWallClockMs),
2366
+ contextTokensAtEnd: contextTokensAtEnd(calls, compactionEntries),
2367
+ compactions: { value: compactionEntries.length, source: "ledger" },
2368
+ expectedColdReasons
2369
+ };
2370
+ }
2371
+ function emptyEvalTrackedMetrics(source = "estimated") {
2372
+ const zero = () => ({ value: 0, source });
2373
+ return {
2374
+ modelCalls: zero(),
2375
+ uncachedPrefillTokens: zero(),
2376
+ cacheReadTokens: zero(),
2377
+ generatedTokens: zero(),
2378
+ reasoningTokens: { value: null, source },
2379
+ toolCalls: zero(),
2380
+ toolErrors: zero(),
2381
+ ttftMsFirstCall: zero(),
2382
+ wallClockMs: zero(),
2383
+ contextTokensAtEnd: zero(),
2384
+ compactions: zero(),
2385
+ expectedColdReasons: {}
2386
+ };
2387
+ }
2388
+ function assistantCalls(entries) {
2389
+ return entries.flatMap((entry) => {
2390
+ if (entry.kind !== "message" || entry.role !== "assistant" || !isRecord5(entry.payload)) return [];
2391
+ const promptCache = recordField(entry.payload, "promptCache");
2392
+ const timing = recordField(entry.payload, "timing");
2393
+ const usage = recordField(entry.payload, "usage");
2394
+ if (promptCache === null && timing === null && usage === null) return [];
2395
+ return [
2396
+ {
2397
+ payload: entry.payload,
2398
+ promptCache,
2399
+ backend: promptCache === null ? null : recordField(promptCache, "backend"),
2400
+ timing,
2401
+ usage
2402
+ }
2403
+ ];
2404
+ });
2405
+ }
2406
+ function uncachedPrefillForCall(call) {
2407
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2408
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2409
+ if (promptTokens !== null && cachedTokens !== null && cachedTokens <= promptTokens) {
2410
+ return ledgerReading(promptTokens - cachedTokens);
2411
+ }
2412
+ const piInput = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2413
+ if (piInput !== null) return ledgerReading(piInput);
2414
+ const legacyInput = call.usage === null ? null : nonNegativeNumber(call.usage.input);
2415
+ return legacyInput === null ? estimatedReading(0) : estimatedReading(legacyInput);
2416
+ }
2417
+ function cacheReadForCall(call) {
2418
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2419
+ if (cachedTokens !== null) return ledgerReading(cachedTokens);
2420
+ const piCacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2421
+ if (piCacheRead !== null) return ledgerReading(piCacheRead);
2422
+ const legacyCacheRead = call.usage === null ? null : nonNegativeNumber(call.usage.cacheRead);
2423
+ return legacyCacheRead === null ? estimatedReading(0) : estimatedReading(legacyCacheRead);
2424
+ }
2425
+ function generatedForCall(call) {
2426
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2427
+ if (predictedTokens !== null) return ledgerReading(predictedTokens);
2428
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2429
+ return output === null ? estimatedReading(0) : ledgerReading(output);
2430
+ }
2431
+ function readingFromUsage(usage, field) {
2432
+ const value = nonNegativeNumber(usage[field]);
2433
+ return value === null ? estimatedReading(0) : ledgerReading(value);
2434
+ }
2435
+ function reasoningMetric(receipt, calls, compactionUsage) {
2436
+ if (receipt !== null && typeof receipt.reasoningTokenCount === "number") {
2437
+ return { value: finiteNonNegative(receipt.reasoningTokenCount), source: "receipt" };
2438
+ }
2439
+ let total = 0;
2440
+ let measured = false;
2441
+ for (const call of calls) {
2442
+ const value = extractReasoningTokens(call.usage);
2443
+ if (value === null) continue;
2444
+ measured = true;
2445
+ total += finiteNonNegative(value);
2446
+ }
2447
+ for (const usage of compactionUsage) {
2448
+ const value = nonNegativeNumber(usage.reasoning);
2449
+ if (value === null) continue;
2450
+ measured = true;
2451
+ total += value;
2452
+ }
2453
+ return measured ? { value: total, source: "ledger" } : { value: null, source: "estimated" };
2454
+ }
2455
+ function firstCallTtft(calls) {
2456
+ const first = calls[0];
2457
+ const value = first?.timing === null || first?.timing === void 0 ? null : nonNegativeNumber(first.timing.ttftMs);
2458
+ return value === null ? { value: 0, source: "estimated" } : { value, source: "ledger" };
2459
+ }
2460
+ function wallClockMetric(receipt, fallback) {
2461
+ if (receipt !== null) {
2462
+ const started = Date.parse(receipt.startedAt);
2463
+ const ended = Date.parse(receipt.endedAt);
2464
+ if (Number.isFinite(started) && Number.isFinite(ended) && ended >= started) {
2465
+ return { value: ended - started, source: "receipt" };
2466
+ }
2467
+ }
2468
+ return { value: finiteNonNegative(fallback), source: "estimated" };
2469
+ }
2470
+ function contextTokensAtEnd(calls, compactions) {
2471
+ const lastCall = calls.at(-1);
2472
+ if (lastCall !== void 0) {
2473
+ const lastReading = contextForCall(lastCall);
2474
+ if (lastReading !== null && lastReading > 0) return { value: lastReading, source: "ledger" };
2475
+ if (lastReading === 0) {
2476
+ for (const call of [...calls.slice(0, -1)].reverse()) {
2477
+ const reading = contextForCall(call);
2478
+ if (reading !== null && reading > 0) return { value: reading, source: "ledger" };
2479
+ }
2480
+ }
2481
+ }
2482
+ const lastCompaction = compactions.at(-1);
2483
+ if (lastCompaction?.kind === "compactionSummary") {
2484
+ const tokensAfter = nonNegativeNumber(lastCompaction.tokensAfter);
2485
+ if (tokensAfter !== null) return { value: tokensAfter, source: "ledger" };
2486
+ }
2487
+ return { value: 0, source: "estimated" };
2488
+ }
2489
+ function contextForCall(call) {
2490
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2491
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2492
+ if (promptTokens !== null && predictedTokens !== null) return promptTokens + predictedTokens;
2493
+ const input = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2494
+ const cacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2495
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2496
+ return input === null || cacheRead === null || output === null ? null : input + cacheRead + output;
2497
+ }
2498
+ function expectedColdReasonMetrics(calls) {
2499
+ const counts = /* @__PURE__ */ new Map();
2500
+ for (const call of calls) {
2501
+ const reasons = call.promptCache?.expectedColdReasons;
2502
+ if (!Array.isArray(reasons)) continue;
2503
+ const unique = new Set(reasons.filter((reason) => typeof reason === "string" && reason.length > 0));
2504
+ for (const reason of unique) counts.set(reason, (counts.get(reason) ?? 0) + 1);
2505
+ }
2506
+ return Object.fromEntries(
2507
+ [...counts.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([reason, value]) => [reason, { value, source: "ledger" }])
2508
+ );
2509
+ }
2510
+ function sumReadings(readings, emptySource) {
2511
+ if (readings.length === 0) return { value: 0, source: emptySource };
2512
+ return {
2513
+ value: readings.reduce((sum2, reading) => sum2 + reading.value, 0),
2514
+ source: readings.some((reading) => reading.source === "estimated") ? "estimated" : "ledger"
2515
+ };
2516
+ }
2517
+ function ledgerReading(value) {
2518
+ return { value: finiteNonNegative(value), source: "ledger" };
2519
+ }
2520
+ function estimatedReading(value) {
2521
+ return { value: finiteNonNegative(value), source: "estimated" };
2522
+ }
2523
+ function finiteNonNegative(value) {
2524
+ return Number.isFinite(value) && value >= 0 ? value : 0;
2525
+ }
2526
+ function nonNegativeNumber(value) {
2527
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2528
+ }
2529
+ function recordField(record, field) {
2530
+ return isRecord5(record[field]) ? record[field] : null;
2531
+ }
2532
+ function parseJsonRecord2(line) {
2533
+ if (line.trim().length === 0) return null;
2534
+ try {
2535
+ const parsed = JSON.parse(line);
2536
+ return isRecord5(parsed) ? parsed : null;
2537
+ } catch {
2538
+ return null;
2539
+ }
2540
+ }
2541
+ function promptManifestObservation(record) {
2542
+ const systemPromptHash = nullableDigest(record.systemPromptHash);
2543
+ if (systemPromptHash === null || !Array.isArray(record.fragments)) return null;
2544
+ const fragments = record.fragments.flatMap((entry) => {
2545
+ if (!isRecord5(entry) || typeof entry.id !== "string") return [];
2546
+ const contentHash = nullableDigest(entry.contentHash);
2547
+ return contentHash === null ? [] : [{ id: entry.id, contentHash }];
2548
+ });
2549
+ const preload = record.projectPreload;
2550
+ const projectPreload = preload === null ? null : isRecord5(preload) && (preload.mode === "full" || preload.mode === "synopsis" || preload.mode === "none") && typeof preload.chars === "number" && Number.isInteger(preload.chars) && typeof preload.lines === "number" && Number.isInteger(preload.lines) && typeof preload.nearLimit === "boolean" && typeof preload.label === "string" ? {
2551
+ mode: preload.mode,
2552
+ chars: preload.chars,
2553
+ lines: preload.lines,
2554
+ reason: nullableString(preload.reason),
2555
+ nearLimit: preload.nearLimit,
2556
+ label: preload.label
2557
+ } : null;
2558
+ return {
2559
+ systemPromptHash,
2560
+ thinkingLevel: nullableString(record.thinkingLevel),
2561
+ projectPreload,
2562
+ fragments
2563
+ };
2564
+ }
2565
+ function nullableString(value) {
2566
+ return typeof value === "string" && value.length > 0 ? value : null;
2567
+ }
2568
+ function nullableDigest(value) {
2569
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
2570
+ }
2571
+ function isRecord5(value) {
2572
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2573
+ }
2574
+
2575
+ // src/domains/eval/runners/clio-run.ts
2576
+ init_esm_shims();
2577
+ import { isAbsolute, relative, resolve as resolve2, sep } from "node:path";
2578
+
2579
+ // src/domains/eval/metrics/evidence.ts
2580
+ init_esm_shims();
2581
+ import { readFileSync as readFileSync3 } from "node:fs";
2582
+ import { join as join3 } from "node:path";
2583
+ function dispatchScopeMetrics(receipt) {
2584
+ const scope = receipt.pathScope;
2585
+ if (scope === void 0) return {};
2586
+ const entries = [...scope.workingContextPaths, ...scope.writeBoundaries];
2587
+ const evidence = entries.flatMap((entry) => entry.evidence);
2588
+ const count2 = (source) => evidence.filter((entry) => entry.source === source).length;
2589
+ return {
2590
+ "dispatch.scope.mode": scope.mode,
2591
+ "dispatch.scope.inferredPathCount": entries.filter(
2592
+ (entry) => entry.evidence.some((item) => item.provenance === "inferred")
2593
+ ).length,
2594
+ "dispatch.scope.derivedPathCount": entries.filter(
2595
+ (entry) => entry.evidence.some((item) => item.provenance === "derived")
2596
+ ).length,
2597
+ "dispatch.scope.source.task": count2("task"),
2598
+ "dispatch.scope.source.briefing": count2("briefing"),
2599
+ "dispatch.scope.source.writeRoots": count2("writeRoots")
2600
+ };
2601
+ }
2602
+ function receiptFromRunJsonStdout(stdout) {
2603
+ let searchEnd = stdout.length;
2604
+ while (searchEnd > 0) {
2605
+ const start = stdout.lastIndexOf("\n{", searchEnd - 1);
2606
+ if (start === -1) break;
2607
+ const candidate = stdout.slice(start + 1);
2608
+ try {
2609
+ const parsed = JSON.parse(candidate);
2610
+ if (isReceiptShaped2(parsed)) return parsed;
2611
+ } catch {
2612
+ }
2613
+ if (start === 0) break;
2614
+ searchEnd = start;
2615
+ }
2616
+ return null;
2617
+ }
2618
+ function isReceiptShaped2(value) {
2619
+ if (typeof value !== "object" || value === null || Array.isArray(value)) return false;
2620
+ const record = value;
2621
+ return typeof record.runId === "string" && typeof record.agentId === "string" && typeof record.exitCode === "number" && typeof record.integrity === "object" && record.integrity !== null;
2622
+ }
2623
+ function evidenceMetricsFromReceipt(receipt, options = {}) {
2624
+ return {
2625
+ "evidence.verification": receipt.verification.state,
2626
+ ...dispatchScopeMetrics(receipt),
2627
+ ...evidenceTrustMetrics(receipt, options.envelope ?? null),
2628
+ ...receipt.findingsSummary === void 0 ? {} : { "evidence.firstPassSuccess": receipt.findingsSummary.firstPassSuccess === true },
2629
+ "evidence.quality.typedValidationCount": receipt.quality.typedValidations.length,
2630
+ "evidence.responseSchema.digest": receipt.quality.responseSchema.schemaDigest ?? "none",
2631
+ ...typeof receipt.costUsd === "number" && Number.isFinite(receipt.costUsd) ? { "cost.usd": receipt.costUsd } : {}
2632
+ };
2633
+ }
2634
+ function evidenceTrustMetrics(receipt, envelope) {
2635
+ const status = envelope === null ? adaptRunReceiptTrustStatus(receipt) : inspectRunReceiptTrustStatus(receipt, envelope).status;
2636
+ const summary = summarizeTrustStatus(status);
2637
+ return {
2638
+ "evidence.trust.version": summary.version,
2639
+ "evidence.trust.verdict": summary.verdict,
2640
+ "evidence.trust.summary": formatTrustSummary(status),
2641
+ ...Object.fromEntries(TRUST_STATUS_AXES.map((axis) => [`evidence.trust.${axis}`, status[axis].state]))
2642
+ };
2643
+ }
2644
+ function readRunEnvelopeForReceipt(receipt, stateDir) {
2645
+ try {
2646
+ const parsed = JSON.parse(readFileSync3(join3(stateDir, "runs.json"), "utf8"));
2647
+ if (!Array.isArray(parsed)) return null;
2648
+ const row = parsed.find(
2649
+ (entry) => typeof entry === "object" && entry !== null && entry.id === receipt.runId
2650
+ );
2651
+ return row ?? null;
2652
+ } catch {
2653
+ return null;
2654
+ }
2655
+ }
2656
+
2657
+ // src/domains/eval/metrics/token-stream.ts
2658
+ init_esm_shims();
2659
+ var UNMEASURED_TOKEN_USAGE = {
2660
+ measured: false,
2661
+ tokens: { input: 0, output: 0, total: 0, cacheRead: 0, cacheWrite: 0 },
2662
+ costUsd: 0
2663
+ };
2664
+ function createTokenUsageFold() {
2665
+ const tokens = { input: 0, output: 0, total: 0, cacheRead: 0, cacheWrite: 0 };
2666
+ let costUsd = 0;
2667
+ let measured = false;
2668
+ let pending = "";
2669
+ const consume = (line) => {
2670
+ if (line.trim().length === 0) return;
2671
+ let event;
2672
+ try {
2673
+ event = JSON.parse(line);
2674
+ } catch {
2675
+ return;
2676
+ }
2677
+ if (!isRecord6(event) || event.type !== "message_end") return;
2678
+ const message = isRecord6(event.message) ? event.message : void 0;
2679
+ if (message === void 0 || message.role !== "assistant") return;
2680
+ const usage = isRecord6(message.usage) ? message.usage : void 0;
2681
+ if (usage === void 0) return;
2682
+ measured = true;
2683
+ const input = numberField2(usage, "input");
2684
+ const output = numberField2(usage, "output");
2685
+ const cacheRead = numberField2(usage, "cacheRead");
2686
+ const cacheWrite = numberField2(usage, "cacheWrite");
2687
+ const totalTokens = numberField2(usage, "totalTokens");
2688
+ tokens.input += input;
2689
+ tokens.output += output;
2690
+ tokens.cacheRead += cacheRead;
2691
+ tokens.cacheWrite += cacheWrite;
2692
+ tokens.total += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
2693
+ if (isRecord6(usage.cost)) costUsd += numberField2(usage.cost, "total");
2694
+ };
2695
+ return {
2696
+ push(chunk) {
2697
+ pending += chunk;
2698
+ for (; ; ) {
2699
+ const newline = pending.indexOf("\n");
2700
+ if (newline === -1) break;
2701
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
2702
+ pending = pending.slice(newline + 1);
2703
+ }
2704
+ },
2705
+ usage() {
2706
+ if (pending.length > 0) {
2707
+ consume(pending.replace(/\r$/u, ""));
2708
+ pending = "";
2709
+ }
2710
+ return { measured, tokens: { ...tokens }, costUsd };
2711
+ }
2712
+ };
2713
+ }
2714
+ function addTokenStreamUsage(left, right) {
2715
+ return {
2716
+ measured: left.measured || right.measured,
2717
+ tokens: {
2718
+ input: left.tokens.input + right.tokens.input,
2719
+ output: left.tokens.output + right.tokens.output,
2720
+ total: left.tokens.total + right.tokens.total,
2721
+ cacheRead: left.tokens.cacheRead + right.tokens.cacheRead,
2722
+ cacheWrite: left.tokens.cacheWrite + right.tokens.cacheWrite
2723
+ },
2724
+ costUsd: left.costUsd + right.costUsd
2725
+ };
2726
+ }
2727
+ function tokenMetricEntries(usage) {
2728
+ if (!usage.measured) return { "tokens.measured": false };
2729
+ return {
2730
+ "tokens.measured": true,
2731
+ "tokens.input": usage.tokens.input,
2732
+ "tokens.output": usage.tokens.output,
2733
+ "tokens.total": usage.tokens.total,
2734
+ "tokens.cacheRead": usage.tokens.cacheRead,
2735
+ "tokens.cacheWrite": usage.tokens.cacheWrite,
2736
+ ...usage.costUsd > 0 ? { "cost.usd": usage.costUsd } : {}
2737
+ };
2738
+ }
2739
+ function numberField2(record, field) {
2740
+ const value = record[field];
2741
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
2742
+ }
2743
+ function isRecord6(value) {
2744
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2745
+ }
2746
+
2747
+ // src/domains/eval/runners/external-command.ts
2748
+ init_esm_shims();
2749
+ import { spawn } from "node:child_process";
2750
+ import { performance as performance2 } from "node:perf_hooks";
2751
+
2752
+ // src/domains/eval/metrics/call-ledger-stream.ts
2753
+ init_esm_shims();
2754
+ import { performance } from "node:perf_hooks";
2755
+ function createEvalCallLedgerFold(now = () => performance.now()) {
2756
+ const entries = [];
2757
+ let pending = "";
2758
+ let activeStartedAt = null;
2759
+ let activeFirstOutputAt = null;
2760
+ const consume = (line) => {
2761
+ const event = parseRecord(line);
2762
+ if (event === null) return;
2763
+ if (event.type === "message_start" && isAssistantMessage(event.message)) {
2764
+ activeStartedAt = now();
2765
+ activeFirstOutputAt = null;
2766
+ return;
2767
+ }
2768
+ if (event.type === "message_update" && activeStartedAt !== null && activeFirstOutputAt === null) {
2769
+ activeFirstOutputAt = now();
2770
+ return;
2771
+ }
2772
+ if (event.type !== "message_end" || !isAssistantMessage(event.message)) return;
2773
+ const message = event.message;
2774
+ const usage = isRecord7(message.usage) ? message.usage : null;
2775
+ if (usage === null) {
2776
+ activeStartedAt = null;
2777
+ activeFirstOutputAt = null;
2778
+ return;
2779
+ }
2780
+ const endedAt = now();
2781
+ const promptCache = {
2782
+ input: nonNegativeNumber2(usage.input) ?? 0,
2783
+ cacheRead: nonNegativeNumber2(usage.cacheRead) ?? 0,
2784
+ cacheWrite: nonNegativeNumber2(usage.cacheWrite) ?? 0,
2785
+ backendVerdict: "unknown"
2786
+ };
2787
+ if (isRecord7(message.backendTimings)) promptCache.backend = structuredClone(message.backendTimings);
2788
+ const previous = entries.at(-1);
2789
+ entries.push({
2790
+ kind: "message",
2791
+ role: "assistant",
2792
+ turnId: `eval-call-${entries.length + 1}`,
2793
+ parentTurnId: previous?.turnId ?? null,
2794
+ timestamp: messageTimestamp(message.timestamp),
2795
+ payload: {
2796
+ promptCache,
2797
+ timing: {
2798
+ ttftMs: activeStartedAt === null ? null : Math.round(Math.max(0, (activeFirstOutputAt ?? endedAt) - activeStartedAt)),
2799
+ apiMs: activeStartedAt === null ? 0 : Math.round(Math.max(0, endedAt - activeStartedAt))
2800
+ },
2801
+ usage: structuredClone(usage)
2802
+ }
2803
+ });
2804
+ activeStartedAt = null;
2805
+ activeFirstOutputAt = null;
2806
+ };
2807
+ return {
2808
+ push(chunk) {
2809
+ pending += chunk;
2810
+ for (; ; ) {
2811
+ const newline = pending.indexOf("\n");
2812
+ if (newline === -1) break;
2813
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
2814
+ pending = pending.slice(newline + 1);
2815
+ }
2816
+ },
2817
+ entries() {
2818
+ if (pending.length > 0) {
2819
+ consume(pending.replace(/\r$/u, ""));
2820
+ pending = "";
2821
+ }
2822
+ return structuredClone(entries);
2823
+ }
2824
+ };
2825
+ }
2826
+ function isAssistantMessage(value) {
2827
+ return isRecord7(value) && value.role === "assistant";
2828
+ }
2829
+ function messageTimestamp(value) {
2830
+ const milliseconds = nonNegativeNumber2(value);
2831
+ if (milliseconds === null) return (/* @__PURE__ */ new Date(0)).toISOString();
2832
+ const date = new Date(milliseconds);
2833
+ return Number.isFinite(date.getTime()) ? date.toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
2834
+ }
2835
+ function nonNegativeNumber2(value) {
2836
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2837
+ }
2838
+ function parseRecord(line) {
2839
+ if (line.trim().length === 0) return null;
2840
+ try {
2841
+ const value = JSON.parse(line);
2842
+ return isRecord7(value) ? value : null;
2843
+ } catch {
2844
+ return null;
2845
+ }
2846
+ }
2847
+ function isRecord7(value) {
2848
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2849
+ }
2850
+
2851
+ // src/domains/eval/runners/external-command.ts
2852
+ var OUTPUT_LIMIT = 2e5;
2853
+ var OUTPUT_HEAD_LIMIT = 2e4;
2854
+ var OUTPUT_TRUNCATION_MARKER = "\n[output middle truncated; tail preserved]\n";
2855
+ var METRIC_JSONL_LIMIT = 256e3;
2856
+ var METRIC_JSONL_LINE_LIMIT = 64e3;
2857
+ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
2858
+ const runnerCommands = runner.commands ?? [];
2859
+ const commands = runnerCommands.length > 0 ? runnerCommands : runner.command === void 0 ? [] : [runner.command];
2860
+ let stdout = "";
2861
+ let stderr = "";
2862
+ let wallTimeMs = 0;
2863
+ let usage = UNMEASURED_TOKEN_USAGE;
2864
+ let streamInvariants = EMPTY_STREAM_INVARIANTS;
2865
+ let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
2866
+ const ledgerEntries = [];
2867
+ for (const command of commands) {
2868
+ const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
2869
+ stdout = appendLimited(stdout, result.stdout);
2870
+ stderr = appendLimited(stderr, result.stderr);
2871
+ wallTimeMs += result.wallTimeMs;
2872
+ usage = addTokenStreamUsage(usage, result.usage);
2873
+ streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
2874
+ fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
2875
+ ledgerEntries.push(...result.ledgerEntries);
2876
+ if (result.exitCode !== 0) {
2877
+ return {
2878
+ assignmentId: null,
2879
+ terminalReceiptDigest: null,
2880
+ exitCode: result.exitCode,
2881
+ stdout,
2882
+ stderr,
2883
+ wallTimeMs,
2884
+ metrics: {
2885
+ "latency.wallMs": wallTimeMs,
2886
+ ...tokenMetricEntries(usage),
2887
+ ...streamInvariantMetrics(streamInvariants),
2888
+ ...fleetLoopMetricEntries(fleetLoops),
2889
+ "verifier.exitCode": result.exitCode
2890
+ },
2891
+ artifacts: {},
2892
+ ledgerEntries
2893
+ };
2894
+ }
2895
+ }
2896
+ return {
2897
+ assignmentId: null,
2898
+ terminalReceiptDigest: null,
2899
+ exitCode: 0,
2900
+ stdout,
2901
+ stderr,
2902
+ wallTimeMs,
2903
+ metrics: {
2904
+ "latency.wallMs": wallTimeMs,
2905
+ ...tokenMetricEntries(usage),
2906
+ ...streamInvariantMetrics(streamInvariants),
2907
+ ...fleetLoopMetricEntries(fleetLoops),
2908
+ "verifier.exitCode": 0
2909
+ },
2910
+ artifacts: {},
2911
+ ledgerEntries
2912
+ };
2913
+ }
2914
+ function runShellCommand(command, cwd, timeoutMs, env) {
2915
+ const started = performance2.now();
2916
+ return new Promise((resolve9) => {
2917
+ let stdout = "";
2918
+ let stderr = "";
2919
+ const metricCapture = createJsonlMetricCapture();
2920
+ const usageFold = createTokenUsageFold();
2921
+ const streamFold = createStreamInvariantFold();
2922
+ const fleetLoopFold = createFleetLoopFold();
2923
+ const callLedgerFold = createEvalCallLedgerFold();
2924
+ let timedOut = false;
2925
+ let settled = false;
2926
+ const child = spawn(command, {
2927
+ cwd,
2928
+ shell: true,
2929
+ stdio: ["ignore", "pipe", "pipe"],
2930
+ // The overlay is additive: an eval item pins where Clio writes its
2931
+ // journal, and inherits everything else the operator's shell provides.
2932
+ env: env === void 0 ? process.env : { ...process.env, ...env }
2933
+ });
2934
+ child.stdout.setEncoding("utf8");
2935
+ child.stderr.setEncoding("utf8");
2936
+ child.stdout.on("data", (chunk) => {
2937
+ stdout = appendLimited(stdout, chunk);
2938
+ metricCapture.push(chunk);
2939
+ usageFold.push(chunk);
2940
+ streamFold.push(chunk);
2941
+ fleetLoopFold.push(chunk);
2942
+ callLedgerFold.push(chunk);
2943
+ });
2944
+ child.stderr.on("data", (chunk) => {
2945
+ stderr = appendLimited(stderr, chunk);
2946
+ });
2947
+ const timer = setTimeout(() => {
2948
+ timedOut = true;
2949
+ child.kill("SIGTERM");
2950
+ setTimeout(() => child.kill("SIGKILL"), 1e3);
2951
+ }, timeoutMs);
2952
+ const finish = (exitCode) => {
2953
+ if (settled) return;
2954
+ settled = true;
2955
+ clearTimeout(timer);
2956
+ resolve9({
2957
+ command,
2958
+ exitCode,
2959
+ stdout,
2960
+ metricJsonl: metricCapture.finish(),
2961
+ usage: usageFold.usage(),
2962
+ streamInvariants: streamFold.invariants(),
2963
+ fleetLoops: fleetLoopFold.observation(),
2964
+ ledgerEntries: callLedgerFold.entries(),
2965
+ stderr,
2966
+ wallTimeMs: Math.round(performance2.now() - started),
2967
+ timedOut
2968
+ });
2969
+ };
2970
+ child.on("error", (error) => {
2971
+ stderr = appendLimited(stderr, error.message);
2972
+ finish(1);
2973
+ });
2974
+ child.on("close", (code) => {
2975
+ finish(typeof code === "number" ? code : timedOut ? 124 : 1);
2976
+ });
2977
+ });
2978
+ }
2979
+ function appendLimited(current, chunk) {
2980
+ const next = `${current}${chunk}`;
2981
+ if (next.length <= OUTPUT_LIMIT) return next;
2982
+ const tailLimit = OUTPUT_LIMIT - OUTPUT_HEAD_LIMIT - OUTPUT_TRUNCATION_MARKER.length;
2983
+ return `${next.slice(0, OUTPUT_HEAD_LIMIT)}${OUTPUT_TRUNCATION_MARKER}${next.slice(-tailLimit)}`;
2984
+ }
2985
+ function createJsonlMetricCapture() {
2986
+ const lines = [];
2987
+ let storedBytes = 0;
2988
+ let pending = "";
2989
+ let discardingOversizedLine = false;
2990
+ let finished = false;
2991
+ const store = (line) => {
2992
+ let parsed;
2993
+ try {
2994
+ parsed = JSON.parse(line);
2995
+ } catch {
2996
+ return;
2997
+ }
2998
+ if (!isRecord8(parsed)) return;
2999
+ const compact = compactMetricEvent(parsed);
3000
+ if (compact === null) return;
3001
+ const encoded = JSON.stringify(compact);
3002
+ if (encoded.length > METRIC_JSONL_LINE_LIMIT) return;
3003
+ lines.push(encoded);
3004
+ storedBytes += encoded.length + 1;
3005
+ while (storedBytes > METRIC_JSONL_LIMIT && lines.length > 1) {
3006
+ const removed = lines.shift();
3007
+ if (removed !== void 0) storedBytes -= removed.length + 1;
3008
+ }
3009
+ };
3010
+ const push = (chunk) => {
3011
+ if (finished || chunk.length === 0) return;
3012
+ let remaining = chunk;
3013
+ if (discardingOversizedLine) {
3014
+ const newline = remaining.indexOf("\n");
3015
+ if (newline === -1) return;
3016
+ remaining = remaining.slice(newline + 1);
3017
+ discardingOversizedLine = false;
3018
+ }
3019
+ pending += remaining;
3020
+ for (; ; ) {
3021
+ const newline = pending.indexOf("\n");
3022
+ if (newline === -1) break;
3023
+ store(pending.slice(0, newline).replace(/\r$/u, ""));
3024
+ pending = pending.slice(newline + 1);
3025
+ }
3026
+ if (pending.length > METRIC_JSONL_LINE_LIMIT) {
3027
+ pending = "";
3028
+ discardingOversizedLine = true;
3029
+ }
3030
+ };
3031
+ return {
3032
+ push,
3033
+ finish() {
3034
+ if (!finished && !discardingOversizedLine && pending.length > 0) store(pending.replace(/\r$/u, ""));
3035
+ finished = true;
3036
+ pending = "";
3037
+ return lines.join("\n");
3038
+ }
3039
+ };
3040
+ }
3041
+ function compactMetricEvent(event) {
3042
+ const type = event.type;
3043
+ if (type === "tool_execution_start") {
3044
+ const toolName = stringField(event, "toolName");
3045
+ if (toolName !== "dispatch" && toolName !== "code_nav" && toolName !== "read" && toolName !== "grep") {
3046
+ return null;
3047
+ }
3048
+ return {
3049
+ type,
3050
+ ...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
3051
+ toolName,
3052
+ ...toolName === "dispatch" && isRecord8(event.args) ? { args: event.args } : toolName === "read" && isRecord8(event.args) ? { args: boundedReadArgs(event.args) } : toolName === "code_nav" && isRecord8(event.args) ? { args: { mode: event.args.mode } } : {}
3053
+ };
3054
+ }
3055
+ if (type === "tool_execution_end") {
3056
+ return {
3057
+ type,
3058
+ ...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
3059
+ ...stringField(event, "toolName") !== void 0 ? { toolName: stringField(event, "toolName") } : {},
3060
+ ...event.isError === true ? { isError: true } : {},
3061
+ ...stringField(event, "outcome") !== void 0 ? { outcome: stringField(event, "outcome") } : {}
3062
+ };
3063
+ }
3064
+ if (type !== "clio_tool_finish" || !isRecord8(event.payload)) return null;
3065
+ return {
3066
+ type,
3067
+ payload: {
3068
+ ...stringField(event.payload, "tool") !== void 0 ? { tool: stringField(event.payload, "tool") } : {},
3069
+ ...stringField(event.payload, "toolCallId") !== void 0 ? { toolCallId: stringField(event.payload, "toolCallId") } : stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
3070
+ ...stringField(event.payload, "outcome") !== void 0 ? { outcome: stringField(event.payload, "outcome") } : {}
3071
+ }
3072
+ };
3073
+ }
3074
+ function boundedReadArgs(args) {
3075
+ for (const field of ["path", "filePath", "file_path"]) {
3076
+ const value = args[field];
3077
+ if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4096) };
3078
+ }
3079
+ return {};
3080
+ }
3081
+ function isRecord8(value) {
3082
+ return typeof value === "object" && value !== null && !Array.isArray(value);
3083
+ }
3084
+ function stringField(record, field) {
3085
+ const value = record[field];
3086
+ return typeof value === "string" && value.length > 0 ? value : void 0;
3087
+ }
3088
+
3089
+ // src/domains/eval/runners/clio-run.ts
3090
+ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env, readObservation) {
3091
+ const prompt = runner.prompt ?? "";
3092
+ const args = [
3093
+ shellQuote(clioEntry),
3094
+ "run",
3095
+ "--json",
3096
+ ...runner.agent === void 0 ? [] : ["--agent", shellQuote(runner.agent)],
3097
+ "--target",
3098
+ shellQuote(target.id),
3099
+ ...target.model === void 0 ? [] : ["--model", shellQuote(target.model)],
3100
+ ...target.thinking === void 0 ? [] : ["--thinking", shellQuote(target.thinking)],
3101
+ ...runner.autonomy === void 0 ? [] : ["--autonomy", runner.autonomy],
3102
+ shellQuote(prompt)
3103
+ ];
3104
+ const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
3105
+ const tokens = result.usage;
3106
+ const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
3107
+ const tools = toolCallMetricsFromJsonl(toolMetricStream);
3108
+ const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
3109
+ const receipt = receiptFromRunJsonStdout(result.stdout);
3110
+ const envelope = receipt === null ? null : readRunEnvelopeForReceipt(receipt, env?.CLIO_CODER_STATE_DIR ?? clioStateDir());
3111
+ return {
3112
+ assignmentId: receipt === null ? null : receipt.lineage?.rootRunId ?? receipt.runId,
3113
+ terminalReceiptDigest: receipt?.integrity.digest ?? null,
3114
+ exitCode: result.exitCode,
3115
+ stdout: result.stdout,
3116
+ stderr: result.stderr,
3117
+ wallTimeMs: result.wallTimeMs,
3118
+ metrics: {
3119
+ "latency.wallMs": result.wallTimeMs,
3120
+ ...tokenMetricEntries(tokens),
3121
+ // Folded live for the same reason the usage is: the structural
3122
+ // promise these check is broken by a run whose middle is truncated.
3123
+ ...streamInvariantMetrics(result.streamInvariants),
3124
+ "tools.totalCalls": tools.totalCalls,
3125
+ "tools.failed": tools.failed,
3126
+ "tools.blocked": tools.blocked,
3127
+ ...behavioralTools,
3128
+ "verifier.exitCode": result.exitCode,
3129
+ ...receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope }),
3130
+ ...receipt === null ? {} : { "evidence.qualityLabel": receipt.quality.typedValidations.length > 0 ? "measured" : "unmeasured" }
3131
+ },
3132
+ artifacts: {
3133
+ stdout: result.stdout,
3134
+ stderr: result.stderr,
3135
+ callLedger: JSON.stringify(result.ledgerEntries),
3136
+ ...receipt === null ? {} : { receipt: JSON.stringify(receipt) }
3137
+ },
3138
+ receipt,
3139
+ ledgerEntries: result.ledgerEntries
3140
+ };
3141
+ }
3142
+ function toolCallMetricsFromJsonl(stdout) {
3143
+ const executionEnds = { totalCalls: 0, failed: 0, blocked: 0 };
3144
+ const canonicalFinishes = { totalCalls: 0, failed: 0, blocked: 0 };
3145
+ const seenExecutionEnds = /* @__PURE__ */ new Set();
3146
+ const seenCanonicalFinishes = /* @__PURE__ */ new Set();
3147
+ for (const line of stdout.split(/\r?\n/)) {
3148
+ if (line.trim().length === 0) continue;
3149
+ let event;
3150
+ try {
3151
+ const parsed = JSON.parse(line);
3152
+ if (!isRecord9(parsed)) continue;
3153
+ event = parsed;
3154
+ } catch {
3155
+ continue;
3156
+ }
3157
+ if (event.type === "tool_execution_end") {
3158
+ const callId2 = stringField2(event, "toolCallId");
3159
+ if (callId2 !== void 0) {
3160
+ if (seenExecutionEnds.has(callId2)) continue;
3161
+ seenExecutionEnds.add(callId2);
3162
+ }
3163
+ recordToolOutcome(executionEnds, toolOutcome(event) ?? (event.isError === true ? "error" : "ok"));
3164
+ continue;
3165
+ }
3166
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
3167
+ const outcome = toolOutcome(event.payload);
3168
+ if (outcome === void 0) continue;
3169
+ const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId");
3170
+ if (callId !== void 0) {
3171
+ if (seenCanonicalFinishes.has(callId)) continue;
3172
+ seenCanonicalFinishes.add(callId);
3173
+ }
3174
+ recordToolOutcome(canonicalFinishes, outcome);
3175
+ }
3176
+ return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
3177
+ }
3178
+ function toolBehaviorMetricEntriesFromJsonl(stdout, cwd, readObservation) {
3179
+ const starts = /* @__PURE__ */ new Map();
3180
+ const readPaths = /* @__PURE__ */ new Set();
3181
+ const executionEnds = [];
3182
+ const canonicalFinishes = [];
3183
+ const seenExecution = /* @__PURE__ */ new Set();
3184
+ const seenCanonical = /* @__PURE__ */ new Set();
3185
+ for (const line of stdout.split(/\r?\n/)) {
3186
+ if (line.trim().length === 0) continue;
3187
+ let event;
3188
+ try {
3189
+ const parsed = JSON.parse(line);
3190
+ if (!isRecord9(parsed)) continue;
3191
+ event = parsed;
3192
+ } catch {
3193
+ continue;
3194
+ }
3195
+ if (event.type === "tool_execution_start") {
3196
+ const callId2 = stringField2(event, "toolCallId");
3197
+ const tool2 = stringField2(event, "toolName");
3198
+ if (callId2 === void 0 || tool2 === void 0) continue;
3199
+ const path = tool2 === "read" && isRecord9(event.args) ? toolPath(event.args) : null;
3200
+ starts.set(callId2, { tool: tool2, path });
3201
+ if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
3202
+ continue;
3203
+ }
3204
+ if (event.type === "tool_execution_end") {
3205
+ const callId2 = stringField2(event, "toolCallId") ?? null;
3206
+ if (callId2 !== null && seenExecution.has(callId2)) continue;
3207
+ if (callId2 !== null) seenExecution.add(callId2);
3208
+ const tool2 = stringField2(event, "toolName") ?? (callId2 === null ? void 0 : starts.get(callId2)?.tool);
3209
+ if (tool2 === void 0) continue;
3210
+ executionEnds.push({ callId: callId2, tool: tool2, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
3211
+ continue;
3212
+ }
3213
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
3214
+ const outcome = toolOutcome(event.payload);
3215
+ const tool = stringField2(event.payload, "tool");
3216
+ if (outcome === void 0 || tool === void 0) continue;
3217
+ const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId") ?? null;
3218
+ if (callId !== null && seenCanonical.has(callId)) continue;
3219
+ if (callId !== null) seenCanonical.add(callId);
3220
+ canonicalFinishes.push({ callId, tool, outcome });
3221
+ }
3222
+ const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
3223
+ const calls = /* @__PURE__ */ new Map();
3224
+ const blocked = /* @__PURE__ */ new Map();
3225
+ for (const terminal of terminals) {
3226
+ const tool = metricToolName(terminal.tool);
3227
+ calls.set(tool, (calls.get(tool) ?? 0) + 1);
3228
+ if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
3229
+ }
3230
+ const namedTools = /* @__PURE__ */ new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
3231
+ const entries = { "tools.read.distinctPaths": readPaths.size };
3232
+ for (const tool of [...namedTools].sort()) {
3233
+ entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
3234
+ entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
3235
+ }
3236
+ if (readObservation !== void 0) {
3237
+ const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
3238
+ const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
3239
+ entries["tools.read.outsideAllowed"] = [...readPaths].filter(
3240
+ (path) => !allowed.some((root) => pathWithin(path, root))
3241
+ ).length;
3242
+ entries["tools.read.decoyHits"] = [...readPaths].filter(
3243
+ (path) => decoys.some((root) => pathWithin(path, root))
3244
+ ).length;
3245
+ }
3246
+ return entries;
3247
+ }
3248
+ function toolPath(args) {
3249
+ for (const field of ["path", "filePath", "file_path"]) {
3250
+ const value = args[field];
3251
+ if (typeof value === "string" && value.length > 0 && value.length <= 4096) return value;
3252
+ }
3253
+ return null;
3254
+ }
3255
+ function normalizeObservedPath(cwd, path) {
3256
+ const absolute = resolve2(cwd, path);
3257
+ const local = relative(cwd, absolute);
3258
+ return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".").split(sep).join("/");
3259
+ }
3260
+ function pathWithin(path, root) {
3261
+ if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
3262
+ return path === root || path.startsWith(`${root}/`);
3263
+ }
3264
+ function metricToolName(tool) {
3265
+ return tool.toLowerCase().replaceAll(/[^a-z0-9_-]/gu, "_").slice(0, 64) || "unknown";
3266
+ }
3267
+ function recordToolOutcome(metrics, outcome) {
3268
+ metrics.totalCalls += 1;
3269
+ if (outcome === "error") metrics.failed += 1;
3270
+ else if (outcome === "blocked") metrics.blocked += 1;
3271
+ }
3272
+ function toolOutcome(record) {
3273
+ const outcome = record.outcome;
3274
+ return outcome === "ok" || outcome === "error" || outcome === "blocked" ? outcome : void 0;
3275
+ }
3276
+ function stringField2(record, field) {
3277
+ const value = record[field];
3278
+ return typeof value === "string" && value.length > 0 ? value : void 0;
3279
+ }
3280
+ function isRecord9(value) {
3281
+ return typeof value === "object" && value !== null && !Array.isArray(value);
3282
+ }
3283
+
3284
+ // src/domains/eval/runners/context-index.ts
3285
+ init_esm_shims();
3286
+ async function runContextIndexRunner(cwd, clioEntry, timeoutMs, target, env) {
3287
+ const result = await runShellCommand(
3288
+ `${shellQuote(process.execPath)} ${shellQuote(clioEntry)} context index --json`,
3289
+ cwd,
3290
+ timeoutMs,
3291
+ env
3292
+ );
3293
+ const parsed = parseContextIndexOutput(result.stdout);
3294
+ return {
3295
+ assignmentId: null,
3296
+ terminalReceiptDigest: null,
3297
+ exitCode: result.exitCode,
3298
+ stdout: result.stdout,
3299
+ stderr: result.stderr,
3300
+ wallTimeMs: result.wallTimeMs,
3301
+ metrics: {
3302
+ "latency.wallMs": result.wallTimeMs,
3303
+ "context.indexedFiles": parsed.indexedSourceFiles ?? 0,
3304
+ "context.coverage": parsed.coverage ?? 0,
3305
+ "context.structuralHash": parsed.structuralHash ?? null,
3306
+ "verifier.exitCode": result.exitCode
3307
+ },
3308
+ artifacts: {
3309
+ stdout: result.stdout,
3310
+ stderr: result.stderr,
3311
+ target: target.id
3312
+ }
3313
+ };
3314
+ }
3315
+ function parseContextIndexOutput(stdout) {
3316
+ try {
3317
+ const parsed = JSON.parse(stdout);
3318
+ const metrics = {};
3319
+ if (typeof parsed.indexedSourceFiles === "number") metrics.indexedSourceFiles = parsed.indexedSourceFiles;
3320
+ if (typeof parsed.coverage === "number") metrics.coverage = parsed.coverage;
3321
+ if (typeof parsed.structuralHash === "string") metrics.structuralHash = parsed.structuralHash;
3322
+ return metrics;
3323
+ } catch {
3324
+ return {};
3325
+ }
3326
+ }
3327
+
3328
+ // src/domains/eval/runners/context-init.ts
3329
+ init_esm_shims();
3330
+ import { existsSync as existsSync2, statSync } from "node:fs";
3331
+ import { join as join4 } from "node:path";
3332
+ async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
3333
+ const extraArgs = runner.args ?? [];
3334
+ const command = [
3335
+ process.execPath,
3336
+ clioEntry,
3337
+ "context",
3338
+ "init",
3339
+ "--yes",
3340
+ "--json",
3341
+ "--target",
3342
+ target.id,
3343
+ ...target.model === void 0 ? [] : ["--model", target.model],
3344
+ ...target.thinking === void 0 ? [] : ["--thinking", target.thinking],
3345
+ ...extraArgs
3346
+ ].map(shellQuote).join(" ");
3347
+ const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
3348
+ const payload = parseInitPayload(result.stdout);
3349
+ const candidateGeneration = recordField2(payload, "generation");
3350
+ const generation = isValidGenerationPayload(payload, candidateGeneration) ? candidateGeneration : null;
3351
+ const routeError = generation ? generationRouteError(generation, target) : null;
3352
+ const payloadError = generation ? routeError : "context-init runner did not receive a valid JSON generation result";
3353
+ const exitCode = result.exitCode === 0 && payloadError ? 1 : result.exitCode;
3354
+ const stderr = payloadError ? `${result.stderr}${result.stderr.endsWith("\n") || result.stderr.length === 0 ? "" : "\n"}${payloadError}
3355
+ ` : result.stderr;
3356
+ const run = recordField2(generation, "run");
3357
+ const tokens = recordField2(run, "tokens");
3358
+ const effectiveTarget = stringField3(run, "targetId");
3359
+ const effectiveModel = stringField3(run, "wireModelId");
3360
+ const effectiveRuntime = stringField3(run, "runtimeId");
3361
+ const effectiveRuntimeKind = stringField3(run, "runtimeKind");
3362
+ const effectiveThinking = stringField3(run, "thinkingLevel");
3363
+ const structuredOutputMode = stringField3(run, "structuredOutputMode");
3364
+ const clioMdPath = join4(cwd, "CLIO-CODER.md");
3365
+ const clioMdBytes = existsSync2(clioMdPath) ? statSync(clioMdPath).size : 0;
3366
+ return {
3367
+ assignmentId: null,
3368
+ terminalReceiptDigest: null,
3369
+ exitCode,
3370
+ stdout: result.stdout,
3371
+ stderr,
3372
+ wallTimeMs: result.wallTimeMs,
3373
+ metrics: {
3374
+ "latency.wallMs": result.wallTimeMs,
3375
+ "latency.modelMs": numberField3(run, "durationMs"),
3376
+ "tokens.input": numberField3(tokens, "input"),
3377
+ "tokens.output": numberField3(tokens, "output"),
3378
+ "tokens.total": numberField3(tokens, "total"),
3379
+ "tokens.cacheRead": numberField3(tokens, "cacheRead"),
3380
+ "tokens.cacheWrite": numberField3(tokens, "cacheWrite"),
3381
+ "tools.totalCalls": numberField3(run, "toolCalls"),
3382
+ "tools.failed": numberField3(run, "toolFailures"),
3383
+ "tools.blocked": numberField3(run, "toolBlocked"),
3384
+ "context.clioMdBytes": clioMdBytes,
3385
+ "context.initMode": stringField3(generation, "mode") ?? "unknown",
3386
+ "context.initParserOutcome": stringField3(generation, "parserOutcome") ?? "unknown",
3387
+ "context.initFallback": stringField3(generation, "fallbackReason") !== null,
3388
+ "context.initPromptBytes": numberField3(run, "promptBytes"),
3389
+ "context.initOutputBytes": numberField3(run, "outputBytes"),
3390
+ "context.initTargetId": effectiveTarget,
3391
+ "context.initModelId": effectiveModel,
3392
+ "context.initRuntimeId": effectiveRuntime,
3393
+ "context.initRuntimeKind": effectiveRuntimeKind,
3394
+ "context.initThinkingLevel": effectiveThinking,
3395
+ "context.initStructuredOutputMode": structuredOutputMode,
3396
+ "verifier.exitCode": exitCode
3397
+ },
3398
+ artifacts: {
3399
+ stdout: result.stdout,
3400
+ stderr,
3401
+ requestedTarget: target.id,
3402
+ requestedModel: target.model ?? null,
3403
+ requestedThinking: target.thinking ?? null,
3404
+ effectiveTarget,
3405
+ effectiveModel,
3406
+ effectiveRuntime,
3407
+ effectiveThinking,
3408
+ bootstrapRunId: stringField3(run, "runId")
3409
+ }
3410
+ };
3411
+ }
3412
+ var GENERATION_MODES = /* @__PURE__ */ new Set(["model", "heuristic", "existing"]);
3413
+ var PARSER_OUTCOMES = /* @__PURE__ */ new Set(["parsed", "rejected", "not-run"]);
3414
+ function isNonnegativeFiniteNumber(value) {
3415
+ return typeof value === "number" && Number.isFinite(value) && value >= 0;
3416
+ }
3417
+ function isValidRunPayload(value) {
3418
+ const run = recordField2(value);
3419
+ if (!run) return false;
3420
+ for (const key of ["durationMs", "promptBytes", "outputBytes"]) {
3421
+ if (!isNonnegativeFiniteNumber(run[key])) return false;
3422
+ }
3423
+ if (run.structuredOutputMode !== "native-schema" && run.structuredOutputMode !== "prompt-parser") return false;
3424
+ for (const key of ["toolCalls", "toolFailures", "toolBlocked"]) {
3425
+ if (run[key] !== void 0 && !isNonnegativeFiniteNumber(run[key])) return false;
3426
+ }
3427
+ const tokens = recordField2(run, "tokens");
3428
+ if (run.tokens !== void 0 && !tokens) return false;
3429
+ if (tokens) {
3430
+ for (const key of ["total", "input", "output", "cacheRead", "cacheWrite", "reasoning"]) {
3431
+ if (tokens[key] !== void 0 && !isNonnegativeFiniteNumber(tokens[key])) return false;
3432
+ }
3433
+ }
3434
+ return true;
3435
+ }
3436
+ function hasReceiptIdentity(run) {
3437
+ if (!run) return false;
3438
+ return ["runId", "targetId", "wireModelId", "runtimeId", "runtimeKind", "thinkingLevel"].every(
3439
+ (key) => typeof run[key] === "string" && run[key].trim().length > 0
3440
+ );
3441
+ }
3442
+ function isValidGenerationPayload(payload, generation) {
3443
+ if (payload?.version !== 1 || !generation) return false;
3444
+ const mode = generation.mode;
3445
+ const parserOutcome = generation.parserOutcome;
3446
+ if (typeof mode !== "string" || !GENERATION_MODES.has(mode)) return false;
3447
+ if (typeof parserOutcome !== "string" || !PARSER_OUTCOMES.has(parserOutcome)) return false;
3448
+ if (generation.fallbackReason !== void 0 && (typeof generation.fallbackReason !== "string" || generation.fallbackReason.trim().length === 0)) {
3449
+ return false;
3450
+ }
3451
+ const runPresent = generation.run !== void 0;
3452
+ if (runPresent && !isValidRunPayload(generation.run)) return false;
3453
+ const run = recordField2(generation, "run");
3454
+ if (mode === "model" && (parserOutcome !== "parsed" || !hasReceiptIdentity(run))) return false;
3455
+ if ((parserOutcome === "parsed" || parserOutcome === "rejected") && !runPresent) return false;
3456
+ if (parserOutcome === "rejected" && !hasReceiptIdentity(run)) return false;
3457
+ return true;
3458
+ }
3459
+ function generationRouteError(generation, target) {
3460
+ const run = recordField2(generation, "run");
3461
+ if (!run) return null;
3462
+ const actualTarget = stringField3(run, "targetId");
3463
+ const actualModel = stringField3(run, "wireModelId");
3464
+ const actualThinking = stringField3(run, "thinkingLevel");
3465
+ if (actualTarget !== null && actualTarget !== target.id) {
3466
+ return `context-init runner requested target '${target.id}' but the bootstrap receipt used '${actualTarget}'`;
3467
+ }
3468
+ if (target.model !== void 0 && actualModel !== null && actualModel !== target.model) {
3469
+ return `context-init runner requested model '${target.model}' but the bootstrap receipt used '${actualModel}'`;
3470
+ }
3471
+ if (target.thinking !== void 0 && actualThinking !== null && actualThinking !== target.thinking) {
3472
+ return `context-init runner requested thinking '${target.thinking}' but the bootstrap receipt used '${actualThinking}'`;
3473
+ }
3474
+ return null;
3475
+ }
3476
+ function parseInitPayload(stdout) {
3477
+ try {
3478
+ const parsed = JSON.parse(stdout);
3479
+ return recordField2(parsed);
3480
+ } catch {
3481
+ return null;
3482
+ }
3483
+ }
3484
+ function recordField2(value, field) {
3485
+ const selected = field && typeof value === "object" && value !== null && !Array.isArray(value) ? value[field] : value;
3486
+ return typeof selected === "object" && selected !== null && !Array.isArray(selected) ? selected : null;
3487
+ }
3488
+ function numberField3(value, field) {
3489
+ const record = recordField2(value);
3490
+ const selected = record?.[field];
3491
+ return typeof selected === "number" && Number.isFinite(selected) ? selected : null;
3492
+ }
3493
+ function stringField3(value, field) {
3494
+ const record = recordField2(value);
3495
+ const selected = record?.[field];
3496
+ return typeof selected === "string" && selected.length > 0 ? selected : null;
3497
+ }
3498
+
3499
+ // src/domains/eval/schema/adapter.ts
3500
+ init_esm_shims();
3501
+ import { createHash as createHash3 } from "node:crypto";
3502
+ function adaptSuiteV2ResultToVerdictV1(result, trackedMetrics) {
3503
+ const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
3504
+ const outcome = result.pass ? "pass" : "fail";
3505
+ const graderExitCode = result.metrics["task.exitCode"];
3506
+ return parseEvalVerdictEnvelopeV1({
3507
+ schema: EVAL_VERDICT_SCHEMA_V1,
3508
+ scenarioId: result.taskId,
3509
+ trialIndex: result.repeatIndex,
3510
+ outcome,
3511
+ machinery,
3512
+ reason: result.pass ? null : result.failureClass ?? "result_failed",
3513
+ trackedMetrics,
3514
+ behavioral: null,
3515
+ evidence: {
3516
+ assignmentId: result.assignmentId,
3517
+ terminalReceiptDigest: result.terminalReceiptDigest,
3518
+ graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null
3519
+ }
3520
+ });
3521
+ }
3522
+ function adaptSuiteV2ResultToBehaviorV1(result, verdict, scenario) {
3523
+ const requestedFacts = new Set(
3524
+ [...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
3525
+ (rule) => `${rule.fact.source}\0${rule.fact.key}`
3526
+ )
3527
+ );
3528
+ const observedSources = /* @__PURE__ */ new Set();
3529
+ const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
3530
+ if (value === null) return [];
3531
+ const source = metricFactSource(key);
3532
+ observedSources.add(source);
3533
+ if (!requestedFacts.has(`${source}\0${key}`)) return [];
3534
+ const serialized = JSON.stringify({ source, key, value });
3535
+ const digest = createHash3("sha256").update(serialized, "utf8").digest("hex");
3536
+ const fact = {
3537
+ id: `metric-${digest.slice(0, 16)}`,
3538
+ source,
3539
+ key,
3540
+ value,
3541
+ evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1e3) }
3542
+ };
3543
+ return [fact];
3544
+ });
3545
+ const allSources = ["transcript", "tool", "receipt", "grader"];
3546
+ const unavailableSources = allSources.filter(
3547
+ (source) => !observedSources.has(source) || source === "tool" && scenario.execution.toolTarget === "none"
3548
+ );
3549
+ const behavior = judgeEvalBehaviorV1(scenario, verdict, {
3550
+ facts,
3551
+ unavailableSources,
3552
+ infrastructureFailure: verdict.machinery === "infrastructure_failure"
3553
+ });
3554
+ assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
3555
+ return behavior;
3556
+ }
3557
+ function metricFactSource(key) {
3558
+ if (key.startsWith("tools.")) return "tool";
3559
+ if (key.startsWith("task.") || key.startsWith("claims.") || key.startsWith("completion.") || key === "result.pass" || key === "verifier.exitCode")
3560
+ return "grader";
3561
+ if (key.startsWith("receipt.") || key.startsWith("evidence.") || key.startsWith("boundary.") || key.startsWith("loop.") || key.startsWith("cost."))
3562
+ return "receipt";
3563
+ return "transcript";
3564
+ }
3565
+
3566
+ // src/domains/eval/verifiers/command.ts
3567
+ init_esm_shims();
3568
+ async function runCommandVerifiers(commands, cwd, timeoutMs, env) {
3569
+ let stdout = "";
3570
+ let stderr = "";
3571
+ let wallTimeMs = 0;
3572
+ for (const command of commands) {
3573
+ const result = await runShellCommand(command, cwd, timeoutMs, env);
3574
+ stdout += result.stdout;
3575
+ stderr += result.stderr;
3576
+ wallTimeMs += result.wallTimeMs;
3577
+ if (result.exitCode !== 0) return { pass: false, exitCode: result.exitCode, stdout, stderr, wallTimeMs };
3578
+ }
3579
+ return { pass: true, exitCode: 0, stdout, stderr, wallTimeMs };
3580
+ }
3581
+
3582
+ // src/domains/eval/verifiers/file-exists.ts
3583
+ init_esm_shims();
3584
+ import { existsSync as existsSync3 } from "node:fs";
3585
+ import { resolve as resolve3 } from "node:path";
3586
+ function forbiddenPathHits(cwd, paths) {
3587
+ return paths.filter((path) => existsSync3(resolve3(cwd, path)));
3588
+ }
3589
+
3590
+ // src/domains/eval/verifiers/patch.ts
3591
+ init_esm_shims();
3592
+ import { spawnSync } from "node:child_process";
3593
+ function collectPatchMetrics(cwd) {
3594
+ const diff = spawnSync("git", ["diff", "--", "."], { cwd, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] });
3595
+ const text = diff.status === 0 && typeof diff.stdout === "string" ? diff.stdout : "";
3596
+ const files = text.split(/\r?\n/).filter((line) => line.startsWith("diff --git ")).map((line) => line.split(" b/")[1] ?? "");
3597
+ return {
3598
+ bytes: Buffer.byteLength(text, "utf8"),
3599
+ filesChanged: files.length,
3600
+ testFilesModified: files.filter((file) => /(^|\/)(test|tests|spec|__tests__)(\/|$)|\.(test|spec)\./.test(file)).length
3601
+ };
3602
+ }
3603
+
3604
+ // src/domains/eval/workspaces/git.ts
3605
+ init_esm_shims();
3606
+ import { spawn as spawn2 } from "node:child_process";
3607
+ import { mkdtemp, rm } from "node:fs/promises";
3608
+ import { tmpdir } from "node:os";
3609
+ import { resolve as resolve4 } from "node:path";
3610
+ async function prepareGitWorkspace(workspace) {
3611
+ if (workspace.url === void 0) throw new Error("git workspace requires url");
3612
+ const dest = await mkdtemp(resolve4(tmpdir(), "clio-eval-git-"));
3613
+ try {
3614
+ await runGit(["clone", "--quiet", workspace.url, dest], process.cwd());
3615
+ const ref = workspace.checkout ?? workspace.commit;
3616
+ if (ref !== void 0) await runGit(["checkout", "--quiet", ref], dest);
3617
+ return {
3618
+ dir: dest,
3619
+ cleanup: async () => {
3620
+ await rm(dest, { recursive: true, force: true });
3621
+ }
3622
+ };
3623
+ } catch (error) {
3624
+ await rm(dest, { recursive: true, force: true });
3625
+ throw error;
3626
+ }
3627
+ }
3628
+ function runGit(args, cwd) {
3629
+ return new Promise((resolveRun, rejectRun) => {
3630
+ const child = spawn2("git", [...args], { cwd, stdio: ["ignore", "ignore", "pipe"] });
3631
+ let stderr = "";
3632
+ child.stderr.setEncoding("utf8");
3633
+ child.stderr.on("data", (chunk) => {
3634
+ stderr += chunk;
3635
+ });
3636
+ child.on("error", rejectRun);
3637
+ child.on("close", (code) => {
3638
+ if (code === 0) resolveRun();
3639
+ else rejectRun(new Error(`git ${args.join(" ")} failed: ${stderr.trim()}`));
3640
+ });
3641
+ });
3642
+ }
3643
+
3644
+ // src/domains/eval/workspaces/local.ts
3645
+ init_esm_shims();
3646
+ import { access } from "node:fs/promises";
3647
+ import { resolve as resolve5 } from "node:path";
3648
+ async function prepareLocalWorkspace(baseDir, workspace) {
3649
+ const dir = resolve5(baseDir, workspace.path ?? ".");
3650
+ await access(dir);
3651
+ return { dir, cleanup: async () => {
3652
+ } };
3653
+ }
3654
+
3655
+ // src/domains/eval/workspaces/temp-copy.ts
3656
+ init_esm_shims();
3657
+ import { execFile } from "node:child_process";
3658
+ import { cp, lstat, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
3659
+ import { tmpdir as tmpdir2 } from "node:os";
3660
+ import { relative as relative2, resolve as resolve6 } from "node:path";
3661
+ import { promisify } from "node:util";
3662
+ var execFileAsync = promisify(execFile);
3663
+ var GIT_FILE_LIST_LIMIT_BYTES = 128 * 1024 * 1024;
3664
+ async function prepareTempCopyWorkspace(baseDir, workspace, options = {}) {
3665
+ const source = resolve6(baseDir, workspace.path ?? ".");
3666
+ const dest = await mkdtemp2(resolve6(options.tempRoot ?? tmpdir2(), "clio-eval-workspace-"));
3667
+ try {
3668
+ const selection = await gitCopySelection(source);
3669
+ const excludes = workspace.excludes ?? [];
3670
+ const copyWorkspace = options.copy ?? defaultCopy;
3671
+ await copyWorkspace(source, dest, {
3672
+ recursive: true,
3673
+ filter: (path) => shouldCopy(relative2(source, path), excludes, selection)
3674
+ });
3675
+ return {
3676
+ dir: dest,
3677
+ cleanup: async () => {
3678
+ await rm2(dest, { recursive: true, force: true });
3679
+ }
3680
+ };
3681
+ } catch (error) {
3682
+ await rm2(dest, { recursive: true, force: true });
3683
+ throw error;
3684
+ }
3685
+ }
3686
+ function isExcluded(rel, excludes) {
3687
+ const normalized = rel.replaceAll("\\", "/");
3688
+ return excludes.some((entry) => normalized === entry || normalized.startsWith(`${entry.replaceAll("\\", "/")}/`));
3689
+ }
3690
+ async function defaultCopy(source, destination, options) {
3691
+ await cp(source, destination, options);
3692
+ }
3693
+ function shouldCopy(relativePath, excludes, selection) {
3694
+ const normalized = relativePath.replaceAll("\\", "/");
3695
+ if (normalized.length === 0) return true;
3696
+ if (isExcluded(normalized, excludes)) return false;
3697
+ if (selection === null) return true;
3698
+ return selection.files.has(normalized) || selection.directories.has(normalized);
3699
+ }
3700
+ async function gitCopySelection(source) {
3701
+ let inside;
3702
+ try {
3703
+ inside = await gitOutput(source, ["rev-parse", "--is-inside-work-tree"]);
3704
+ } catch (error) {
3705
+ if (await hasGitMarker(source)) throw error;
3706
+ return null;
3707
+ }
3708
+ if (inside.trim() !== "true") return null;
3709
+ const output = await gitOutput(source, [
3710
+ "--literal-pathspecs",
3711
+ "ls-files",
3712
+ "-z",
3713
+ "--cached",
3714
+ "--others",
3715
+ "--exclude-standard",
3716
+ "--",
3717
+ "."
3718
+ ]);
3719
+ const files = /* @__PURE__ */ new Set();
3720
+ const directories = /* @__PURE__ */ new Set();
3721
+ for (const path of output.split("\0")) {
3722
+ if (path.length === 0) continue;
3723
+ const normalized = normalizeGitPath(path);
3724
+ if (normalized === null) throw new Error("git ls-files returned a path outside the eval workspace");
3725
+ files.add(normalized);
3726
+ let separator = normalized.lastIndexOf("/");
3727
+ while (separator >= 0) {
3728
+ directories.add(normalized.slice(0, separator));
3729
+ separator = normalized.lastIndexOf("/", separator - 1);
3730
+ }
3731
+ }
3732
+ return { files, directories };
3733
+ }
3734
+ async function gitOutput(cwd, args) {
3735
+ const { stdout } = await execFileAsync("git", [...args], {
3736
+ cwd,
3737
+ encoding: "utf8",
3738
+ maxBuffer: GIT_FILE_LIST_LIMIT_BYTES
3739
+ });
3740
+ return stdout;
3741
+ }
3742
+ function normalizeGitPath(path) {
3743
+ const normalized = path.replaceAll("\\", "/").replace(/^\.\//u, "");
3744
+ if (normalized.length === 0 || normalized.startsWith("/") || /^[A-Za-z]:\//u.test(normalized)) return null;
3745
+ const segments = normalized.split("/");
3746
+ if (segments.some((segment) => segment.length === 0 || segment === "." || segment === "..")) return null;
3747
+ return normalized;
3748
+ }
3749
+ async function hasGitMarker(source) {
3750
+ let current = resolve6(source);
3751
+ while (true) {
3752
+ try {
3753
+ await lstat(resolve6(current, ".git"));
3754
+ return true;
3755
+ } catch (error) {
3756
+ const code = typeof error === "object" && error !== null && "code" in error ? error.code : void 0;
3757
+ if (code !== "ENOENT" && code !== "ENOTDIR") throw error;
3758
+ }
3759
+ const parent = resolve6(current, "..");
3760
+ if (parent === current) return false;
3761
+ current = parent;
3762
+ }
3763
+ }
3764
+
3765
+ // src/domains/eval/suites/matrix.ts
3766
+ init_esm_shims();
3767
+ function expandEvalMatrix(suite) {
3768
+ const runs = [];
3769
+ for (let repeatIndex = 0; repeatIndex < suite.matrix.repeats; repeatIndex += 1) {
3770
+ for (const target of suite.matrix.targets) {
3771
+ for (const task of suite.tasks) {
3772
+ runs.push({ task, target, repeatIndex });
3773
+ }
3774
+ }
3775
+ }
3776
+ return runs;
3777
+ }
3778
+
3779
+ // src/domains/eval/suites/run.ts
3780
+ var EvalWorkspaceSetupError = class extends Error {
3781
+ constructor(exitCode, stderr) {
3782
+ super(`workspace setup failed (exit ${exitCode}): ${stderr.trim()}`);
3783
+ this.name = "EvalWorkspaceSetupError";
3784
+ }
3785
+ };
3786
+ async function runEvalSuiteV2(loaded, options) {
3787
+ const now = options.now ?? (() => /* @__PURE__ */ new Date());
3788
+ const started = now();
3789
+ const evalId = createEvalId(started, loaded.hash);
3790
+ const results = [];
3791
+ const servingObservations = [];
3792
+ const maxCostUsd = loaded.suite.matrix.maxCostUsd;
3793
+ let spentUsd = 0;
3794
+ for (const item of expandEvalMatrix(loaded.suite)) {
3795
+ if (maxCostUsd !== void 0 && spentUsd > maxCostUsd) {
3796
+ results.push(budgetExhaustedResult(loaded, item.task, item.target, item.repeatIndex, spentUsd, maxCostUsd));
3797
+ continue;
3798
+ }
3799
+ const completed = await runMatrixItem(
3800
+ loaded,
3801
+ item.task,
3802
+ item.target,
3803
+ item.repeatIndex,
3804
+ options.clioEntry,
3805
+ options.freshWorkspaces === true,
3806
+ options.tempCopy
3807
+ );
3808
+ spentUsd += resultCostUsd(completed.result);
3809
+ results.push(completed.result);
3810
+ servingObservations.push(completed.serving);
3811
+ }
3812
+ const serving = await evalServingConfiguration(loaded.suite.matrix.targets, servingObservations);
3813
+ return buildArtifact(loaded, evalId, results, options.clioEntry, serving);
3814
+ }
3815
+ function resultCostUsd(result) {
3816
+ const value = result.metrics["cost.usd"];
3817
+ return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0;
3818
+ }
3819
+ function budgetExhaustedResult(loaded, task, target, repeatIndex, spentUsd, maxCostUsd) {
3820
+ const result = {
3821
+ assignmentId: null,
3822
+ terminalReceiptDigest: null,
3823
+ taskId: task.id,
3824
+ repeatIndex,
3825
+ target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
3826
+ pass: false,
3827
+ failureClass: "budget_exhausted",
3828
+ metrics: {
3829
+ "result.pass": false,
3830
+ "result.failureClass": "budget_exhausted",
3831
+ "verifier.exitCode": 1,
3832
+ "latency.wallMs": 0
3833
+ },
3834
+ artifacts: {
3835
+ error: `matrix cost budget exhausted: spent $${spentUsd.toFixed(4)} of max $${maxCostUsd.toFixed(4)} before this item`
3836
+ }
3837
+ };
3838
+ result.verdict = adaptSuiteV2ResultToVerdictV1(result, emptyEvalTrackedMetrics());
3839
+ attachBehavioralResult(result, task);
3840
+ attachExecutionEnvelope(result, task, target, loaded.baseDir, null, emptyLedgerSnapshot());
3841
+ return result;
3842
+ }
3843
+ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry, freshWorkspace, tempCopy) {
3844
+ let workspace = null;
3845
+ let receipt = null;
3846
+ let runnerWallTimeMs = 0;
3847
+ let executionObservation;
3848
+ const stateDir = await mkdtemp3(resolve7(tempCopy?.tempRoot ?? tmpdir3(), "clio-eval-state-"));
3849
+ try {
3850
+ workspace = await prepareWorkspace(loaded.baseDir, task, freshWorkspace, tempCopy);
3851
+ const setup = await runCommandVerifiers(task.workspace.setup ?? [], workspace.dir, task.timeoutMs);
3852
+ if (!setup.pass) throw new EvalWorkspaceSetupError(setup.exitCode, setup.stderr);
3853
+ const runner = await runTaskRunner(task, target, workspace.dir, clioEntry, {
3854
+ CLIO_CODER_STATE_DIR: stateDir,
3855
+ CLIO_CODER_ENTRY: clioEntry
3856
+ });
3857
+ const runnerStdoutFile = resolve7(stateDir, "eval-runner-output.jsonl");
3858
+ await writeFile(runnerStdoutFile, runner.stdout, "utf8");
3859
+ receipt = runner.receipt ?? null;
3860
+ runnerWallTimeMs = runner.wallTimeMs;
3861
+ const patch = collectPatchMetrics(workspace.dir);
3862
+ const receiptExitCode = runner.exitCode;
3863
+ const journalMetrics = invariantMetrics(stateDir, receiptExitCode);
3864
+ const measurement = await measureTaskOutcome(task, workspace.dir, {
3865
+ CLIO_EVAL_RUNNER_STDOUT_FILE: runnerStdoutFile
3866
+ });
3867
+ executionObservation = measurement.executionObservation;
3868
+ const metrics = {
3869
+ ...zeroToolCallMetrics(),
3870
+ ...collectContextMetrics(workspace.dir),
3871
+ // A runner may have exact measurements from command output. Those win
3872
+ // over the generic post-run artifact collector.
3873
+ ...runner.metrics,
3874
+ ...journalMetrics,
3875
+ // The one reading that needs both sides: what the loop reported it
3876
+ // spent, and what the journal shows it sealed.
3877
+ ...fleetLoopReceiptAgreement(runner.metrics, journalMetrics),
3878
+ "patch.bytes": patch.bytes,
3879
+ "patch.filesChanged": patch.filesChanged,
3880
+ "patch.testFilesModified": patch.testFilesModified,
3881
+ "result.pass": runner.exitCode === 0,
3882
+ "result.failureClass": runner.exitCode === 0 ? null : "runner_failed",
3883
+ ...measurement.metrics
3884
+ };
3885
+ const verifier = await runVerifiers(task, workspace.dir, metrics);
3886
+ const graderFailed = metrics["task.solved"] === false;
3887
+ const pass = runner.exitCode === 0 && verifier.pass && !graderFailed;
3888
+ const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : !verifier.pass ? verifier.failureClass : "grader_failed";
3889
+ metrics["verifier.exitCode"] = verifier.exitCode;
3890
+ metrics["result.pass"] = pass;
3891
+ metrics["result.failureClass"] = failureClass;
3892
+ const result = {
3893
+ assignmentId: runner.assignmentId,
3894
+ terminalReceiptDigest: runner.terminalReceiptDigest,
3895
+ taskId: task.id,
3896
+ repeatIndex,
3897
+ target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
3898
+ pass,
3899
+ failureClass,
3900
+ metrics,
3901
+ artifacts: {
3902
+ ...runner.artifacts,
3903
+ workspace: workspace.dir,
3904
+ ...verifier.stdout.length > 0 ? { verifierStdout: verifier.stdout } : {},
3905
+ ...verifier.stderr.length > 0 ? { verifierStderr: verifier.stderr } : {}
3906
+ }
3907
+ };
3908
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3909
+ const ledgerEntries = [...snapshot.entries, ...runner.ledgerEntries ?? []];
3910
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3911
+ result,
3912
+ buildEvalTrackedMetrics({
3913
+ ledgerEntries,
3914
+ receipt: receipt ?? null,
3915
+ fallbackWallClockMs: runner.wallTimeMs
3916
+ })
3917
+ );
3918
+ attachBehavioralResult(result, task);
3919
+ attachExecutionEnvelope(result, task, target, workspace.dir, receipt ?? null, snapshot, executionObservation);
3920
+ return {
3921
+ result,
3922
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
3923
+ };
3924
+ } catch (error) {
3925
+ const failureClass = error instanceof EvalWorkspaceSetupError ? "setup_failed" : "command_error";
3926
+ const result = {
3927
+ assignmentId: null,
3928
+ terminalReceiptDigest: null,
3929
+ taskId: task.id,
3930
+ repeatIndex,
3931
+ target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
3932
+ pass: false,
3933
+ failureClass,
3934
+ metrics: {
3935
+ "result.pass": false,
3936
+ "result.failureClass": failureClass,
3937
+ "verifier.exitCode": 1,
3938
+ "latency.wallMs": 0
3939
+ },
3940
+ artifacts: {
3941
+ error: error instanceof Error ? error.message : String(error),
3942
+ ...workspace === null ? {} : { workspace: workspace.dir }
3943
+ }
3944
+ };
3945
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3946
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3947
+ result,
3948
+ buildEvalTrackedMetrics({
3949
+ ledgerEntries: snapshot.entries,
3950
+ receipt: receipt ?? null,
3951
+ fallbackWallClockMs: runnerWallTimeMs
3952
+ })
3953
+ );
3954
+ attachBehavioralResult(result, task);
3955
+ attachExecutionEnvelope(
3956
+ result,
3957
+ task,
3958
+ target,
3959
+ workspace?.dir ?? loaded.baseDir,
3960
+ receipt ?? null,
3961
+ snapshot,
3962
+ executionObservation
3963
+ );
3964
+ return {
3965
+ result,
3966
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
3967
+ };
3968
+ } finally {
3969
+ try {
3970
+ await workspace?.cleanup();
3971
+ } finally {
3972
+ await rm3(stateDir, { recursive: true, force: true });
3973
+ }
3974
+ }
3975
+ }
3976
+ function attachBehavioralResult(result, task) {
3977
+ if (task.behavioral === void 0 || result.verdict === void 0) return;
3978
+ result.behavioral = adaptSuiteV2ResultToBehaviorV1(result, result.verdict, task.behavioral);
3979
+ result.behavioralMetrics = buildEvalBehaviorMetricsV1(result, task.behavioral.execution.subject.role);
3980
+ }
3981
+ function attachExecutionEnvelope(result, task, target, cwd, receipt, ledger, observation) {
3982
+ if (task.behavioral === void 0) return;
3983
+ result.executionEnvelope = buildEvalExecutionEnvelopeV1({
3984
+ task,
3985
+ target,
3986
+ cwd,
3987
+ receipt,
3988
+ ledger,
3989
+ ...observation === void 0 ? {} : { observation }
3990
+ });
3991
+ }
3992
+ function emptyLedgerSnapshot() {
3993
+ return { entries: [], compiledPromptHashes: [], promptManifests: [], contextSnapshots: [] };
3994
+ }
3995
+ function invariantMetrics(stateDir, runnerExitCode) {
3996
+ const journal = readRunJournal(stateDir);
3997
+ return {
3998
+ ...receiptInvariantMetrics(journal, runnerExitCode),
3999
+ ...receiptUsageMetrics(journal),
4000
+ ...sessionInvariantMetrics(stateDir),
4001
+ ...processInvariantMetrics(journal),
4002
+ ...writeBoundaryInvariantMetrics(stateDir)
4003
+ };
4004
+ }
4005
+ async function prepareWorkspace(baseDir, task, freshWorkspace, tempCopy) {
4006
+ if (task.workspace.kind === "local" && freshWorkspace) {
4007
+ return prepareTempCopyWorkspace(baseDir, { ...task.workspace, kind: "temp-copy" }, tempCopy);
4008
+ }
4009
+ if (task.workspace.kind === "local") return prepareLocalWorkspace(baseDir, task.workspace);
4010
+ if (task.workspace.kind === "git") return prepareGitWorkspace(task.workspace);
4011
+ return prepareTempCopyWorkspace(baseDir, task.workspace, tempCopy);
4012
+ }
4013
+ async function runTaskRunner(task, target, cwd, clioEntry, env) {
4014
+ if (task.runner.kind === "external-command") return runExternalCommandRunner(task.runner, cwd, task.timeoutMs, env);
4015
+ if (task.runner.kind === "context-index") return runContextIndexRunner(cwd, clioEntry, task.timeoutMs, target, env);
4016
+ if (task.runner.kind === "context-init")
4017
+ return runContextInitRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
4018
+ return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env, task.metrics.readObservation);
4019
+ }
4020
+ async function measureTaskOutcome(task, cwd, env) {
4021
+ const commands = task.verify.measure ?? [];
4022
+ if (commands.length === 0) return { metrics: {} };
4023
+ const result = await runCommandVerifiers(commands, cwd, task.timeoutMs, env);
4024
+ const behavioral = graderBehaviorMeasurement(result.stdout);
4025
+ return {
4026
+ metrics: {
4027
+ "task.exitCode": result.exitCode,
4028
+ "task.solved": result.exitCode === 0,
4029
+ ...behavioral.metrics
4030
+ },
4031
+ ...behavioral.executionObservation === void 0 ? {} : { executionObservation: behavioral.executionObservation }
4032
+ };
4033
+ }
4034
+ function graderBehaviorMeasurement(stdout) {
4035
+ const metrics = {};
4036
+ let executionObservation;
4037
+ for (const line of stdout.split(/\r?\n/u)) {
4038
+ if (line.trim().length === 0) continue;
4039
+ let value;
4040
+ try {
4041
+ value = JSON.parse(line);
4042
+ } catch {
4043
+ continue;
4044
+ }
4045
+ if (!isRecord10(value)) continue;
4046
+ if (value.schema === "clio.eval.measure.v1" && isRecord10(value.metrics)) {
4047
+ for (const [key, metric] of Object.entries(value.metrics)) {
4048
+ if (key !== "claims.unsupported" && key !== "completion.reported") continue;
4049
+ if (typeof metric === "boolean" || typeof metric === "number" && Number.isFinite(metric)) metrics[key] = metric;
4050
+ }
4051
+ }
4052
+ if (value.schema === "clio.eval.execution-observation.v1") {
4053
+ executionObservation = parseExecutionObservation(value);
4054
+ }
4055
+ }
4056
+ return { metrics, ...executionObservation === void 0 ? {} : { executionObservation } };
4057
+ }
4058
+ function parseExecutionObservation(value) {
4059
+ const policies = isRecord10(value.policyHashes) ? value.policyHashes : {};
4060
+ const project = isRecord10(value.projectContext) ? value.projectContext : null;
4061
+ return {
4062
+ compositionHash: nullableDigest2(value.compositionHash),
4063
+ target: nullableString2(value.target),
4064
+ wireModel: nullableString2(value.wireModel),
4065
+ runtime: nullableString2(value.runtime),
4066
+ thinkingLevel: nullableString2(value.thinkingLevel),
4067
+ toolSignature: nullableDigest2(value.toolSignature),
4068
+ autonomy: nullableString2(value.autonomy),
4069
+ policyHashes: { rulePack: nullableDigest2(policies.rulePack), project: nullableDigest2(policies.project) },
4070
+ projectContext: project === null ? null : {
4071
+ tier: nullableString2(project.tier),
4072
+ contentHash: nullableDigest2(project.contentHash),
4073
+ chars: nullableNonNegativeInteger(project.chars),
4074
+ sections: stringArray(project.sections),
4075
+ rulesApplied: stringArray(project.rulesApplied),
4076
+ operatorProfileApplied: typeof project.operatorProfileApplied === "boolean" ? project.operatorProfileApplied : null
4077
+ }
4078
+ };
4079
+ }
4080
+ async function runVerifiers(task, cwd, metrics) {
4081
+ const commandResult = await runCommandVerifiers(task.verify.commands ?? [], cwd, task.timeoutMs);
4082
+ if (!commandResult.pass) {
4083
+ return {
4084
+ pass: false,
4085
+ exitCode: commandResult.exitCode,
4086
+ failureClass: "verifier_failed",
4087
+ stdout: commandResult.stdout,
4088
+ stderr: commandResult.stderr
4089
+ };
4090
+ }
4091
+ const forbidden = forbiddenPathHits(cwd, task.verify.forbidPaths ?? []);
4092
+ if (forbidden.length > 0) {
4093
+ return {
4094
+ pass: false,
4095
+ exitCode: 1,
4096
+ failureClass: "forbidden_path",
4097
+ stdout: "",
4098
+ stderr: `forbidden paths exist: ${forbidden.join(", ")}`
4099
+ };
4100
+ }
4101
+ for (const assertion of task.verify.assertions ?? []) {
4102
+ const resolution = resolveMetricAssertion(assertion, metrics);
4103
+ if (!resolution.unresolved && resolution.holds) continue;
4104
+ return {
4105
+ pass: false,
4106
+ exitCode: 1,
4107
+ failureClass: resolution.unresolved ? "assertion_unresolved" : "assertion_failed",
4108
+ stdout: "",
4109
+ stderr: resolution.unresolved ? `assertion unresolved (fail closed): ${assertion.metric} was not measured by this run` : assertionMessage(assertion, resolution.actual)
4110
+ };
4111
+ }
4112
+ return { pass: true, exitCode: 0, failureClass: null, stdout: commandResult.stdout, stderr: commandResult.stderr };
4113
+ }
4114
+ function buildArtifact(loaded, evalId, results, clioEntry, servingConfiguration) {
4115
+ const passed = results.filter((result) => result.pass).length;
4116
+ return {
4117
+ version: 4,
4118
+ evalId,
4119
+ suite: { id: loaded.suite.suite.id, hash: loaded.hash },
4120
+ clio: evalClioProvenance({ entry: clioEntry }),
4121
+ environment: evalEnvironmentProvenance(),
4122
+ matrix: {
4123
+ ...artifactMatrixIdentity(loaded.suite.matrix.targets),
4124
+ ...loaded.suite.matrix.dimensions === void 0 ? {} : { dimensions: loaded.suite.matrix.dimensions }
4125
+ },
4126
+ servingConfiguration,
4127
+ summary: {
4128
+ runs: results.length,
4129
+ passed,
4130
+ failed: results.length - passed,
4131
+ passRate: results.length === 0 ? 0 : passed / results.length,
4132
+ tokens: tokenAccountingFrom(results),
4133
+ wallTimeMs: results.reduce((sum2, result) => sum2 + wallTimeMetric(result.metrics), 0)
4134
+ },
4135
+ aggregates: aggregateEvalVerdicts(
4136
+ results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict])
4137
+ ),
4138
+ results
4139
+ };
4140
+ }
4141
+ function assertionMessage(assertion, actual) {
4142
+ return `assertion failed: ${assertion.metric} ${assertion.op} ${String(assertion.value)} (actual ${JSON.stringify(actual)})`;
4143
+ }
4144
+ function isRecord10(value) {
4145
+ return typeof value === "object" && value !== null && !Array.isArray(value);
4146
+ }
4147
+ function nullableString2(value) {
4148
+ return typeof value === "string" && value.length > 0 ? value : null;
4149
+ }
4150
+ function nullableDigest2(value) {
4151
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
4152
+ }
4153
+ function nullableNonNegativeInteger(value) {
4154
+ return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : null;
4155
+ }
4156
+ function stringArray(value) {
4157
+ return Array.isArray(value) ? value.filter((entry) => typeof entry === "string") : [];
4158
+ }
4159
+
4160
+ // src/cli/eval.ts
4161
+ var HELP = `clio-coder eval <command>
4162
+
4163
+ Commands:
4164
+ clio-coder eval validate --suite <suite.yaml>
4165
+ clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
4166
+ clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
4167
+ clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
4168
+ clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
4169
+ clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
4170
+ `;
4171
+ function parseEvalArgs(args) {
4172
+ const parsed = {
4173
+ repeat: 1,
4174
+ compareIds: [],
4175
+ format: "text",
4176
+ allowConfigDrift: false,
4177
+ help: false
4178
+ };
4179
+ for (let index = 0; index < args.length; index += 1) {
4180
+ const arg = args[index];
4181
+ if (arg === void 0) continue;
4182
+ if (arg === "--help" || arg === "-h") {
4183
+ parsed.help = true;
4184
+ continue;
4185
+ }
4186
+ if (parsed.command === void 0) {
4187
+ if (arg === "validate" || arg === "run" || arg === "report" || arg === "compare" || arg === "gate") {
4188
+ parsed.command = arg;
4189
+ continue;
4190
+ }
4191
+ throw new Error(`unknown eval command: ${arg}`);
4192
+ }
4193
+ if (parsed.command === "validate") {
4194
+ if (arg === "--suite") {
4195
+ parsed.suite = requiredValue(args, index, "--suite");
4196
+ index += 1;
4197
+ continue;
4198
+ }
4199
+ throw new Error(`unknown eval validate argument: ${arg}`);
4200
+ }
4201
+ if (parsed.command === "run") {
4202
+ if (arg === "--suite") {
4203
+ parsed.suite = requiredValue(args, index, "--suite");
4204
+ index += 1;
4205
+ continue;
4206
+ }
4207
+ if (arg === "--task-file") {
4208
+ parsed.taskFile = requiredValue(args, index, "--task-file");
4209
+ index += 1;
4210
+ continue;
4211
+ }
4212
+ if (arg === "--target") {
4213
+ parsed.target = requiredValue(args, index, "--target");
4214
+ index += 1;
4215
+ continue;
4216
+ }
4217
+ if (arg === "--model") {
4218
+ parsed.model = requiredValue(args, index, "--model");
4219
+ index += 1;
4220
+ continue;
4221
+ }
4222
+ if (arg === "--out") {
4223
+ parsed.out = requiredValue(args, index, "--out");
4224
+ index += 1;
4225
+ continue;
4226
+ }
4227
+ if (arg === "--clio-coder-entry" || arg === "--clio-entry") {
4228
+ parsed.clioEntry = requiredValue(args, index, arg);
4229
+ index += 1;
4230
+ continue;
4231
+ }
4232
+ if (arg === "--repeat") {
4233
+ parsed.repeat = positiveInteger(requiredValue(args, index, "--repeat"), "--repeat");
4234
+ index += 1;
4235
+ continue;
4236
+ }
4237
+ if (arg === "--trials") {
4238
+ parsed.trials = positiveInteger(requiredValue(args, index, "--trials"), "--trials");
4239
+ index += 1;
4240
+ continue;
4241
+ }
4242
+ throw new Error(`unknown eval run argument: ${arg}`);
4243
+ }
4244
+ if (parsed.command === "report") {
4245
+ if (arg === "--format") {
4246
+ parsed.format = reportFormat(requiredValue(args, index, "--format"));
4247
+ index += 1;
4248
+ continue;
4249
+ }
4250
+ if (parsed.evalId === void 0 && !arg.startsWith("-")) {
4251
+ parsed.evalId = arg;
4252
+ continue;
4253
+ }
4254
+ throw new Error(`unexpected eval report argument: ${arg}`);
4255
+ }
4256
+ if (parsed.command === "compare") {
4257
+ if (arg === "--format") {
4258
+ parsed.format = comparisonFormat(requiredValue(args, index, "--format"));
4259
+ index += 1;
4260
+ continue;
4261
+ }
4262
+ if (arg === "--metric") {
4263
+ parsed.metric = requiredValue(args, index, "--metric");
4264
+ index += 1;
4265
+ continue;
4266
+ }
4267
+ if (arg === "--allow-config-drift") {
4268
+ parsed.allowConfigDrift = true;
4269
+ continue;
4270
+ }
4271
+ if (!arg.startsWith("-")) {
4272
+ parsed.compareIds.push(arg);
4273
+ continue;
4274
+ }
4275
+ throw new Error(`unexpected eval compare argument: ${arg}`);
4276
+ }
4277
+ if (parsed.command === "gate") {
4278
+ if (arg === "--baseline") {
4279
+ parsed.baseline = requiredValue(args, index, "--baseline");
4280
+ index += 1;
4281
+ continue;
4282
+ }
4283
+ if (arg === "--thresholds") {
4284
+ parsed.thresholds = requiredValue(args, index, "--thresholds");
4285
+ index += 1;
4286
+ continue;
4287
+ }
4288
+ if (parsed.evalId === void 0 && !arg.startsWith("-")) {
4289
+ parsed.evalId = arg;
4290
+ continue;
4291
+ }
4292
+ throw new Error(`unexpected eval gate argument: ${arg}`);
4293
+ }
4294
+ }
4295
+ if (parsed.help) return parsed;
4296
+ if (parsed.command === void 0) throw new Error("eval requires a command");
4297
+ if (parsed.command === "validate" && parsed.suite === void 0) throw new Error("validate requires --suite <path>");
4298
+ if (parsed.command === "run" && parsed.suite === void 0 === (parsed.taskFile === void 0)) {
4299
+ throw new Error("run requires exactly one of --suite or --task-file");
4300
+ }
4301
+ if (parsed.command === "report" && parsed.evalId === void 0) throw new Error("report requires an eval id");
4302
+ if (parsed.command === "compare" && parsed.compareIds.length !== 2) {
4303
+ throw new Error("compare requires <baselineEvalId> <candidateEvalId>");
4304
+ }
4305
+ if (parsed.command === "gate" && (parsed.evalId === void 0 || parsed.baseline === void 0)) {
4306
+ throw new Error("gate requires <candidateEvalId> --baseline <baselineEvalId>");
4307
+ }
4308
+ return parsed;
4309
+ }
4310
+ async function runEvalCommand(args) {
4311
+ let parsed;
4312
+ try {
4313
+ parsed = parseEvalArgs(args);
4314
+ } catch (error) {
4315
+ printError(error instanceof Error ? error.message : String(error));
4316
+ process.stderr.write(HELP);
4317
+ return 2;
4318
+ }
4319
+ if (parsed.help) {
4320
+ process.stdout.write(HELP);
4321
+ return 0;
4322
+ }
4323
+ if (parsed.command === "validate") return runEvalValidate(parsed);
4324
+ if (parsed.command === "run") return runEvalRun(parsed);
4325
+ if (parsed.command === "report") return runEvalReportCommand(parsed);
4326
+ if (parsed.command === "compare") return runEvalCompareCommand(parsed);
4327
+ if (parsed.command === "gate") return runEvalGateCommand(parsed);
4328
+ printError("eval requires a command");
4329
+ return 2;
4330
+ }
4331
+ async function runEvalValidate(parsed) {
4332
+ try {
4333
+ const loaded = await loadEvalSuiteFile(parsed.suite ?? "");
4334
+ process.stdout.write(`valid suite: ${loaded.suite.suite.id}
4335
+ `);
4336
+ return 0;
4337
+ } catch (error) {
4338
+ return handleEvalLoadError(error);
4339
+ }
4340
+ }
4341
+ async function runEvalRun(parsed) {
4342
+ try {
4343
+ const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.trials ?? parsed.repeat);
4344
+ const resolveOptions = {};
4345
+ if (parsed.target !== void 0) resolveOptions.target = parsed.target;
4346
+ if (parsed.model !== void 0) resolveOptions.model = parsed.model;
4347
+ const suite = resolveSuiteForRun(loaded.suite, {
4348
+ ...resolveOptions,
4349
+ ...parsed.trials ? { trials: parsed.trials } : {}
4350
+ });
4351
+ const clioEntry = resolve8(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
4352
+ const artifact = await runEvalSuiteV2(
4353
+ { ...loaded, suite },
4354
+ { clioEntry, freshWorkspaces: parsed.trials !== void 0 }
4355
+ );
4356
+ const artifactPath = await writeEvalArtifactV4(clioDataDir(), artifact, parsed.out);
4357
+ process.stdout.write(`${renderEvalTextReportV4(artifact)}artifact: ${artifactPath}
4358
+ `);
4359
+ const gate = suite.thresholds === void 0 ? null : evaluateGate(artifact, suite.thresholds);
4360
+ if (gate !== null && !gate.pass) {
4361
+ process.stdout.write(`gate: fail (${gate.failures.length} threshold failure)
4362
+ `);
4363
+ for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
4364
+ }
4365
+ if (gate !== null && gate.informational.length > 0) {
4366
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4367
+ `);
4368
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4369
+ }
4370
+ return artifact.summary.failed === 0 && (gate === null || gate.pass) ? 0 : 1;
4371
+ } catch (error) {
4372
+ return handleEvalLoadError(error, 1);
4373
+ }
4374
+ }
4375
+ async function runEvalReportCommand(parsed) {
4376
+ try {
4377
+ const dataDir = clioDataDir();
4378
+ const artifact = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
4379
+ process.stdout.write(renderArtifactReport(artifact, parsed.format, dataDir));
4380
+ return 0;
4381
+ } catch (error) {
4382
+ printError(error instanceof Error ? error.message : String(error));
4383
+ return error instanceof InvalidIdError ? 2 : 1;
4384
+ }
4385
+ }
4386
+ async function runEvalCompareCommand(parsed) {
4387
+ const baselineEvalId = parsed.compareIds[0] ?? "";
4388
+ const candidateEvalId = parsed.compareIds[1] ?? "";
4389
+ try {
4390
+ const dataDir = clioDataDir();
4391
+ const baseline = await loadEvalArtifactV4(dataDir, baselineEvalId);
4392
+ const candidate = await loadEvalArtifactV4(dataDir, candidateEvalId);
4393
+ const summary = compareEvalArtifactsV4(baseline, candidate, {
4394
+ allowConfigDrift: parsed.allowConfigDrift,
4395
+ ...parsed.metric === void 0 ? {} : { metric: parsed.metric }
4396
+ });
4397
+ process.stdout.write(renderEvalComparisonReportV1(summary, parsed.format));
4398
+ return summary.hardGate.pass ? 0 : 1;
4399
+ } catch (error) {
4400
+ printError(error instanceof Error ? error.message : String(error));
4401
+ return error instanceof InvalidIdError ? 2 : 1;
4402
+ }
4403
+ }
4404
+ async function runEvalGateCommand(parsed) {
4405
+ try {
4406
+ const dataDir = clioDataDir();
4407
+ const candidate = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
4408
+ const baseline = await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
4409
+ const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }], informational: [] } : loadThresholds(parsed.thresholds);
4410
+ const gate = evaluateGate(candidate, thresholds);
4411
+ const comparison = compareEvalArtifactsV4(baseline, candidate);
4412
+ if (gate.informational.length > 0) {
4413
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4414
+ `);
4415
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4416
+ }
4417
+ if (gate.pass && comparison.hardGate.pass) {
4418
+ process.stdout.write("gate: pass\n");
4419
+ return 0;
4420
+ }
4421
+ const failureCount = gate.failures.length + comparison.hardGate.failures.length + comparison.hardGate.envelopeFailures.length;
4422
+ process.stdout.write(`gate: fail (${failureCount} hard failure)
4423
+ `);
4424
+ for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
4425
+ for (const failure of comparison.hardGate.failures) {
4426
+ process.stdout.write(
4427
+ ` ${failure.metric} [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: ${failure.change} (hard behavioral gate)
4428
+ `
4429
+ );
4430
+ }
4431
+ for (const failure of comparison.hardGate.envelopeFailures) {
4432
+ process.stdout.write(
4433
+ ` execution envelope [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: incomparable fields ${failure.fields.join(", ")}
4434
+ `
4435
+ );
4436
+ }
4437
+ return 1;
4438
+ } catch (error) {
4439
+ printError(error instanceof Error ? error.message : String(error));
4440
+ return error instanceof InvalidIdError ? 2 : 1;
4441
+ }
4442
+ }
4443
+ function renderArtifactReport(artifact, format2, _dataDir) {
4444
+ if (format2 === "json") return renderEvalJsonReportV4(artifact);
4445
+ if (format2 === "md") return renderEvalMarkdownReportV4(artifact);
4446
+ if (format2 === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
4447
+ if (format2 === "junit") return renderEvalJunitReportV4(artifact);
4448
+ return renderEvalTextReportV4(artifact);
4449
+ }
4450
+ function handleEvalLoadError(error, fallback = 2) {
4451
+ if (error instanceof EvalSuiteFileError || error instanceof EvalTaskFileError) {
4452
+ printError(error.message);
4453
+ for (const issue of error.issues) process.stderr.write(` ${issue.path}: ${issue.message}
4454
+ `);
4455
+ return 2;
4456
+ }
4457
+ printError(error instanceof Error ? error.message : String(error));
4458
+ return fallback;
4459
+ }
4460
+ function requiredValue(args, index, flag) {
4461
+ const value = args[index + 1];
4462
+ if (value === void 0 || value.startsWith("-")) throw new Error(`${flag} requires a value`);
4463
+ return value;
4464
+ }
4465
+ function positiveInteger(value, flag) {
4466
+ const parsed = Number.parseInt(value, 10);
4467
+ if (!Number.isInteger(parsed) || parsed <= 0 || String(parsed) !== value) {
4468
+ throw new Error(`${flag} requires a positive integer`);
4469
+ }
4470
+ return parsed;
4471
+ }
4472
+ function reportFormat(value) {
4473
+ if (value === "text" || value === "json" || value === "md" || value === "swe-jsonl" || value === "junit") return value;
4474
+ throw new Error("--format must be text, json, md, swe-jsonl, or junit");
4475
+ }
4476
+ function comparisonFormat(value) {
4477
+ if (value === "text" || value === "json" || value === "md" || value === "junit") return value;
4478
+ throw new Error("eval compare --format must be text, json, md, or junit");
4479
+ }
4480
+ export {
4481
+ runEvalCommand
4482
+ };
4483
+ //# sourceMappingURL=eval-IJ5VEZDJ.js.map