@iowarp/clio-coder 0.3.8 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (333) hide show
  1. package/CHANGELOG.md +49 -0
  2. package/README.md +7 -3
  3. package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
  4. package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
  5. package/dist/assets/codewiki.json +1 -1
  6. package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
  7. package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
  8. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  9. package/dist/chunk-3DPEIQKN.js +113 -0
  10. package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
  11. package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
  12. package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
  13. package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
  14. package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
  15. package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
  16. package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
  17. package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
  18. package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
  19. package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
  20. package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
  21. package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
  22. package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
  23. package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
  24. package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
  25. package/dist/chunk-7C6RYZGQ.js +391 -0
  26. package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
  27. package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
  28. package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
  29. package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
  30. package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
  31. package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
  32. package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
  33. package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
  34. package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
  35. package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
  36. package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
  37. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  38. package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
  39. package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
  40. package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
  41. package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
  42. package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
  43. package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
  44. package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
  45. package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
  46. package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
  47. package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
  48. package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
  49. package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
  50. package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
  51. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  52. package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
  53. package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
  54. package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
  55. package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
  56. package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
  57. package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
  58. package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
  59. package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
  60. package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
  61. package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
  62. package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
  63. package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
  64. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  65. package/dist/chunk-RKKLTLYB.js +45 -0
  66. package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
  67. package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
  68. package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
  69. package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
  70. package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
  71. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  72. package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
  73. package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
  74. package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
  75. package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
  76. package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
  77. package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
  78. package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
  79. package/dist/chunk-YSEHGPCT.js +127 -0
  80. package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
  81. package/dist/cli/index.js +27 -27
  82. package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
  83. package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
  84. package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
  85. package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
  86. package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
  87. package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
  88. package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
  89. package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
  90. package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
  91. package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
  92. package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
  93. package/dist/doctor-TWBWFK5V.js +165 -0
  94. package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
  95. package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
  96. package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
  97. package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
  98. package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
  99. package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
  100. package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
  101. package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
  102. package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
  103. package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
  104. package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
  105. package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
  106. package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
  107. package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
  108. package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
  109. package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
  110. package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
  111. package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
  112. package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
  113. package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
  114. package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
  115. package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
  116. package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
  117. package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
  118. package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
  119. package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
  120. package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
  121. package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
  122. package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
  123. package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
  124. package/dist/worker/entry.js +29 -33
  125. package/docs/alcf-provider.md +1 -1
  126. package/docs/architecture.md +1 -1
  127. package/docs/artifact-versions.md +6 -1
  128. package/docs/built-in-agents.md +1 -1
  129. package/docs/capacity-and-scheduling.md +23 -2
  130. package/docs/commands-and-modes.md +1 -1
  131. package/docs/configuration-and-targets.md +30 -3
  132. package/docs/context-engine.md +63 -4
  133. package/docs/documentation-coverage.md +3 -3
  134. package/docs/documentation-guide.md +1 -1
  135. package/docs/environment-variables.md +2 -0
  136. package/docs/eval-runner.md +262 -11
  137. package/docs/evals-internal.md +72 -2
  138. package/docs/evidence-and-memory.md +11 -10
  139. package/docs/evolution.md +1 -1
  140. package/docs/extensions-and-sharing.md +3 -1
  141. package/docs/fleet-dispatch.md +4 -4
  142. package/docs/installation-and-lifecycle.md +1 -1
  143. package/docs/middleware-and-components.md +1 -1
  144. package/docs/model-catalog.md +1 -1
  145. package/docs/observability.md +53 -2
  146. package/docs/proactive-memory.md +127 -14
  147. package/docs/prompt-envelope-and-tools.md +19 -1
  148. package/docs/provider-adapter-cookbook.md +1 -1
  149. package/docs/release-cut-checklist.md +19 -3
  150. package/docs/safety-model.md +1 -1
  151. package/docs/scientific-validation.md +1 -1
  152. package/docs/skills-marketplace.md +1 -1
  153. package/docs/tool-usage.md +1 -1
  154. package/docs/trace-store.md +1 -1
  155. package/docs/troubleshooting.md +87 -0
  156. package/docs/tui-design.md +1 -1
  157. package/docs/worker-dispatch-mechanics.md +1 -1
  158. package/package.json +2 -1
  159. package/src/cli/agents.ts +1 -1
  160. package/src/cli/config-inspect.ts +33 -6
  161. package/src/cli/config.ts +1 -1
  162. package/src/cli/doctor-state-size.ts +82 -0
  163. package/src/cli/doctor.ts +3 -1
  164. package/src/cli/eval.ts +80 -16
  165. package/src/cli/extensions.ts +5 -1
  166. package/src/cli/fleet.ts +32 -3
  167. package/src/cli/targets.ts +44 -13
  168. package/src/cli/trace.ts +63 -4
  169. package/src/cli/usage.ts +63 -14
  170. package/src/core/bus-events.ts +29 -1
  171. package/src/core/cache-telemetry.ts +42 -0
  172. package/src/core/config.ts +18 -0
  173. package/src/core/defaults.ts +36 -6
  174. package/src/core/endpoint-key.ts +27 -0
  175. package/src/core/residency-target-key.ts +25 -0
  176. package/src/core/response-schema.ts +36 -2
  177. package/src/domains/config/classify.ts +3 -0
  178. package/src/domains/context/codewiki/coordinator.ts +12 -4
  179. package/src/domains/dispatch/admission.ts +40 -3
  180. package/src/domains/dispatch/capacity-lease.ts +98 -9
  181. package/src/domains/dispatch/contract.ts +11 -0
  182. package/src/domains/dispatch/execution-plan.ts +44 -4
  183. package/src/domains/dispatch/extension.ts +166 -42
  184. package/src/domains/dispatch/fleet-run.ts +23 -3
  185. package/src/domains/dispatch/heartbeat.ts +32 -8
  186. package/src/domains/dispatch/index.ts +3 -0
  187. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  188. package/src/domains/dispatch/reservation-store.ts +116 -8
  189. package/src/domains/dispatch/state.ts +4 -0
  190. package/src/domains/dispatch/worker-spawn.ts +25 -11
  191. package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
  192. package/src/domains/dispatch/write-boundary.ts +62 -1
  193. package/src/domains/eval/artifacts/store.ts +62 -0
  194. package/src/domains/eval/compare/behavioral.ts +224 -0
  195. package/src/domains/eval/compare/compare.ts +355 -2
  196. package/src/domains/eval/compare/envelope.ts +128 -0
  197. package/src/domains/eval/compare/gates.ts +24 -6
  198. package/src/domains/eval/compare/thresholds.ts +30 -3
  199. package/src/domains/eval/execution-provenance.ts +240 -0
  200. package/src/domains/eval/metrics/aggregate.ts +136 -0
  201. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  202. package/src/domains/eval/metrics/tracked.ts +413 -0
  203. package/src/domains/eval/provenance.ts +117 -0
  204. package/src/domains/eval/reports/comparison.ts +128 -0
  205. package/src/domains/eval/reports/junit.ts +17 -3
  206. package/src/domains/eval/reports/markdown.ts +3 -3
  207. package/src/domains/eval/reports/text.ts +14 -0
  208. package/src/domains/eval/run-compare.ts +20 -0
  209. package/src/domains/eval/runners/clio-run.ts +127 -0
  210. package/src/domains/eval/runners/external-command.ts +28 -3
  211. package/src/domains/eval/schema/adapter.ts +111 -0
  212. package/src/domains/eval/schema/artifact.ts +20 -0
  213. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  214. package/src/domains/eval/schema/behavioral.ts +520 -0
  215. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  216. package/src/domains/eval/schema/serving.ts +74 -0
  217. package/src/domains/eval/schema/suite.ts +38 -8
  218. package/src/domains/eval/schema/validate.ts +58 -3
  219. package/src/domains/eval/schema/verdict.ts +237 -0
  220. package/src/domains/eval/suites/resolve.ts +2 -0
  221. package/src/domains/eval/suites/run.ts +264 -33
  222. package/src/domains/eval/verifiers/command.ts +2 -1
  223. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  224. package/src/domains/evidence/build.ts +2 -13
  225. package/src/domains/evidence/eval.ts +2 -12
  226. package/src/domains/evidence/findings-markdown.ts +33 -0
  227. package/src/domains/evidence/run-trust.ts +7 -113
  228. package/src/domains/evidence/trust-projection.ts +2 -2
  229. package/src/domains/extensions/compatibility.ts +285 -0
  230. package/src/domains/extensions/discovery.ts +38 -3
  231. package/src/domains/extensions/resources.ts +1 -1
  232. package/src/domains/extensions/state.ts +12 -3
  233. package/src/domains/extensions/types.ts +2 -0
  234. package/src/domains/lifecycle/doctor.ts +69 -1
  235. package/src/domains/memory/index.ts +14 -0
  236. package/src/domains/memory/task-bank-promotion.ts +64 -0
  237. package/src/domains/memory/task-memory-policy.ts +77 -8
  238. package/src/domains/memory/task-memory-spend.ts +131 -0
  239. package/src/domains/memory/task-memory-status.ts +7 -0
  240. package/src/domains/memory/task-memory-telemetry.ts +2 -0
  241. package/src/domains/middleware/index.ts +1 -0
  242. package/src/domains/middleware/memory-intervention.ts +69 -5
  243. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  244. package/src/domains/observability/background-memory-usage.ts +140 -0
  245. package/src/domains/observability/cost.ts +1 -1
  246. package/src/domains/observability/index.ts +7 -0
  247. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  248. package/src/domains/observability/trace-store.ts +192 -2
  249. package/src/domains/prompts/compiler.ts +100 -13
  250. package/src/domains/providers/endpoint-capacity.ts +96 -0
  251. package/src/domains/providers/index.ts +10 -0
  252. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  253. package/src/domains/providers/runtime-resolution.ts +8 -1
  254. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  255. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  256. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  257. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  258. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  259. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  260. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  261. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  262. package/src/domains/providers/types/capability-flags.ts +2 -0
  263. package/src/domains/providers/types/target-descriptor.ts +2 -0
  264. package/src/domains/resources/prompts/loader.ts +95 -33
  265. package/src/domains/safety/call-target.ts +52 -0
  266. package/src/domains/safety/run-effects.ts +35 -4
  267. package/src/domains/session/context-accounting.ts +52 -1
  268. package/src/domains/session/context-ledger.ts +37 -13
  269. package/src/domains/session/index.ts +6 -0
  270. package/src/domains/session/prompt-cache.ts +140 -0
  271. package/src/domains/session/prompt-manifest.ts +42 -0
  272. package/src/engine/acp/adapter.ts +18 -3
  273. package/src/engine/ai.ts +35 -0
  274. package/src/engine/apis/llamacpp-residency.ts +55 -3
  275. package/src/engine/apis/lmstudio.ts +25 -5
  276. package/src/engine/apis/ollama-native.ts +2 -1
  277. package/src/engine/apis/openai-completions.ts +80 -17
  278. package/src/engine/apis/residency-lock.ts +3 -1
  279. package/src/engine/apis/residency.ts +34 -1
  280. package/src/engine/provider-payload.ts +29 -1
  281. package/src/entry/orchestrator.ts +176 -30
  282. package/src/interactive/chat-loop-messages.ts +26 -7
  283. package/src/interactive/chat-loop.ts +318 -41
  284. package/src/interactive/chat-panel.ts +62 -8
  285. package/src/interactive/clio-editor.ts +45 -8
  286. package/src/interactive/context-activity.ts +5 -1
  287. package/src/interactive/context-meter.ts +1 -1
  288. package/src/interactive/context-overlay.ts +40 -10
  289. package/src/interactive/cost-overlay.ts +64 -6
  290. package/src/interactive/dispatch-board.ts +84 -12
  291. package/src/interactive/fleet-run-preview.ts +41 -15
  292. package/src/interactive/handoff-round.ts +41 -2
  293. package/src/interactive/interactive-application.ts +24 -1
  294. package/src/interactive/interactive-input-runtime.ts +8 -0
  295. package/src/interactive/interactive-presentation.ts +4 -0
  296. package/src/interactive/interactive-shell.ts +20 -17
  297. package/src/interactive/interactive-slash-runtime.ts +27 -4
  298. package/src/interactive/memory-overlay.ts +8 -0
  299. package/src/interactive/mutation-preview.ts +295 -0
  300. package/src/interactive/overlay-general-openers.ts +16 -0
  301. package/src/interactive/overlay-key-routing.ts +38 -0
  302. package/src/interactive/overlay-lifecycle.ts +38 -5
  303. package/src/interactive/overlay-permission-lifecycle.ts +22 -2
  304. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  305. package/src/interactive/overlays/ask-user.ts +91 -19
  306. package/src/interactive/overlays/help-reference.ts +4 -0
  307. package/src/interactive/overlays/prompts.ts +11 -1
  308. package/src/interactive/overlays/settings.ts +35 -1
  309. package/src/interactive/permission-hint.ts +34 -2
  310. package/src/interactive/permission-overlay.ts +159 -9
  311. package/src/interactive/prewarm.ts +197 -0
  312. package/src/interactive/render-trace.ts +162 -15
  313. package/src/interactive/renderers/tool-execution.ts +4 -0
  314. package/src/interactive/side-question.ts +58 -1
  315. package/src/interactive/status/controller.ts +11 -0
  316. package/src/interactive/status/state-machine.ts +54 -2
  317. package/src/interactive/status/types.ts +7 -0
  318. package/src/interactive/terminal-lease.ts +2 -0
  319. package/src/interactive/turn-context.ts +299 -31
  320. package/src/interactive/turn-persistence.ts +14 -4
  321. package/src/interactive/turn-prewarm.ts +364 -0
  322. package/src/interactive/turn-queues.ts +7 -4
  323. package/src/interactive/turn-runtime.ts +8 -1
  324. package/src/interactive/turn-state.ts +23 -0
  325. package/src/interactive/view/view-overlay.ts +28 -3
  326. package/src/tools/ask-user.ts +43 -2
  327. package/src/tools/dispatch-plan.ts +17 -9
  328. package/src/tools/dispatch-scout.ts +1 -1
  329. package/src/tools/registry.ts +16 -0
  330. package/dist/chunk-AOCYTWAV.js +0 -449
  331. package/dist/chunk-HWUFFB6L.js +0 -83
  332. package/dist/chunk-R346GLFC.js +0 -31
  333. package/dist/doctor-M7YEDGAE.js +0 -91
@@ -1,29 +1,69 @@
1
1
  import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
2
2
  import {
3
+ discoverAgentRecipes
4
+ } from "./chunk-HCBCAYZU.js";
5
+ import "./chunk-AD7Y7STJ.js";
6
+ import "./chunk-AMKHQW3C.js";
7
+ import {
8
+ loadFragments,
3
9
  renderCodewikiDigest
4
- } from "./chunk-5WIGXA4T.js";
10
+ } from "./chunk-47CMYGET.js";
5
11
  import {
12
+ EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
13
+ EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
14
+ EVAL_TRACKED_METRIC_NAMES,
15
+ EVAL_VERDICT_SCHEMA_V1,
6
16
  EvalTaskFileError,
7
17
  TRUST_STATUS_AXES,
8
18
  adaptRunReceiptTrustStatus,
19
+ assertComparableTrackedMetricSources,
20
+ assertEvalBehaviorReferencesVerdictV1,
21
+ buildEvalBehaviorMetricsV1,
9
22
  createEvalId,
10
23
  evalClioProvenance,
11
24
  evalEnvironmentProvenance,
25
+ evalHarnessMetricsFromReceipt,
26
+ evalServingConfiguration,
27
+ evalServingObservationFrom,
12
28
  formatTrustSummary,
13
29
  inspectRunReceiptTrustStatus,
30
+ judgeEvalBehaviorV1,
14
31
  loadEvalArtifactV4,
15
32
  loadEvalTaskFile,
33
+ parseEvalBehaviorScenarioV1,
34
+ parseEvalExecutionMatrixDimensionsV1,
35
+ parseEvalVerdictEnvelopeV1,
36
+ renderEvalServingConfiguration,
37
+ sameEvalServingConfiguration,
16
38
  summarizeTrustStatus,
17
39
  verifyReceiptIntegrity,
18
40
  writeEvalArtifactV4
19
- } from "./chunk-K4XHGFR5.js";
41
+ } from "./chunk-MLOK6ZOS.js";
42
+ import {
43
+ listSessionLedgerRefs,
44
+ parseSessionEntries
45
+ } from "./chunk-LXPJXFM5.js";
46
+ import "./chunk-3DPEIQKN.js";
47
+ import "./chunk-W6GROXXM.js";
48
+ import "./chunk-RKKLTLYB.js";
49
+ import "./chunk-5QKCQQ3E.js";
20
50
  import {
21
51
  shellQuote
22
52
  } from "./chunk-TXOTCRLG.js";
23
- import "./chunk-P43ETTHK.js";
53
+ import "./chunk-HVDIIIQW.js";
54
+ import "./chunk-VPTUJU4P.js";
55
+ import "./chunk-2JDWVJND.js";
24
56
  import "./chunk-H7IXIC72.js";
25
- import "./chunk-AOCYTWAV.js";
57
+ import {
58
+ createSafetyPolicyEngine
59
+ } from "./chunk-N3PBVRTZ.js";
60
+ import "./chunk-A2NJGIB3.js";
61
+ import {
62
+ agentSpecFingerprint,
63
+ normalizeAgentSpec
64
+ } from "./chunk-S4COXYBG.js";
26
65
  import "./chunk-MV3K5QF2.js";
66
+ import "./chunk-RAPCMZL4.js";
27
67
  import "./chunk-UL3WSD3F.js";
28
68
  import "./chunk-ECH6PKUQ.js";
29
69
  import "./chunk-CGKSTWHD.js";
@@ -38,15 +78,29 @@ import {
38
78
  enumerateWorkspaceFiles
39
79
  } from "./chunk-33YXPOE3.js";
40
80
  import "./chunk-7CR24IG7.js";
81
+ import "./chunk-XPLRXC72.js";
41
82
  import "./chunk-IFBNV6H6.js";
42
83
  import {
43
84
  printError
44
85
  } from "./chunk-XK56QHLX.js";
45
86
  import "./chunk-5TSRNF4G.js";
87
+ import "./chunk-CFGTUFWB.js";
88
+ import {
89
+ extractReasoningTokens
90
+ } from "./chunk-AEYBF3TB.js";
91
+ import "./chunk-IHXBNWMM.js";
92
+ import "./chunk-B5CSFE7B.js";
93
+ import "./chunk-PNY46YEY.js";
94
+ import "./chunk-FQ4SKYE4.js";
95
+ import "./chunk-RKRLDWD3.js";
46
96
  import {
47
97
  InvalidIdError
48
- } from "./chunk-R346GLFC.js";
98
+ } from "./chunk-KV2AOLDF.js";
99
+ import "./chunk-6EJMN2Y3.js";
49
100
  import "./chunk-IWHMRKLL.js";
101
+ import "./chunk-XDOQXGFO.js";
102
+ import "./chunk-LL4KHSZI.js";
103
+ import "./chunk-4ZG3XFUR.js";
50
104
  import "./chunk-EQ63NRB7.js";
51
105
  import "./chunk-SST6Z5JA.js";
52
106
  import "./chunk-IKCO5N3L.js";
@@ -67,31 +121,585 @@ import {
67
121
 
68
122
  // src/cli/eval.ts
69
123
  init_esm_shims();
70
- import { resolve as resolve7 } from "node:path";
124
+ import { resolve as resolve8 } from "node:path";
71
125
 
72
126
  // src/domains/eval/compare/compare.ts
73
127
  init_esm_shims();
74
- function compareEvalArtifactsV4(baseline, candidate) {
128
+
129
+ // src/domains/eval/metrics/aggregate.ts
130
+ init_esm_shims();
131
+ function aggregateEvalVerdicts(verdicts) {
132
+ const byScenario = /* @__PURE__ */ new Map();
133
+ for (const verdict of verdicts) {
134
+ const group = byScenario.get(verdict.scenarioId) ?? [];
135
+ group.push(verdict);
136
+ byScenario.set(verdict.scenarioId, group);
137
+ }
138
+ return [...byScenario.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([scenarioId, group]) => aggregateScenario(scenarioId, group));
139
+ }
140
+ function aggregateScenario(scenarioId, verdicts) {
141
+ const ordered = [...verdicts].sort((left, right) => left.trialIndex - right.trialIndex);
142
+ const passed = ordered.filter((verdict) => verdict.outcome === "pass").length;
143
+ const failed = ordered.filter((verdict) => verdict.outcome === "fail").length;
144
+ const unmeasured = ordered.filter((verdict) => verdict.outcome === "unmeasured").length;
145
+ const machineryFailures = ordered.filter((verdict) => verdict.machinery === "infrastructure_failure").length;
146
+ const fixed = Object.fromEntries(
147
+ EVAL_TRACKED_METRIC_NAMES.map((name) => [name, distribution(ordered.map((verdict) => verdict.trackedMetrics[name]))])
148
+ );
149
+ const reasons = new Set(ordered.flatMap((verdict) => Object.keys(verdict.trackedMetrics.expectedColdReasons)));
150
+ const expectedColdReasons = Object.fromEntries(
151
+ [...reasons].sort((left, right) => left.localeCompare(right)).map((reason) => [
152
+ reason,
153
+ distribution(
154
+ ordered.map(
155
+ (verdict) => verdict.trackedMetrics.expectedColdReasons[reason] ?? { value: 0, source: "ledger" }
156
+ )
157
+ )
158
+ ])
159
+ );
160
+ const k = ordered.length;
161
+ return {
162
+ scenarioId,
163
+ trials: k,
164
+ k,
165
+ passed,
166
+ failed,
167
+ unmeasured,
168
+ machineryFailures,
169
+ passAtK: k > 0 && passed > 0 ? 1 : 0,
170
+ passPowK: k > 0 && passed === k ? 1 : 0,
171
+ trackedMetrics: { ...fixed, expectedColdReasons }
172
+ };
173
+ }
174
+ function distribution(metrics) {
175
+ const values = metrics.flatMap((metric) => metric.value === null ? [] : [metric.value]);
176
+ const sources = [...new Set(metrics.map((metric) => metric.source))].sort(compareSources);
177
+ if (values.length === 0) {
178
+ return {
179
+ observations: metrics.length,
180
+ measured: 0,
181
+ unmeasured: metrics.length,
182
+ mean: null,
183
+ min: null,
184
+ max: null,
185
+ p90: null,
186
+ variance: null,
187
+ standardDeviation: null,
188
+ sources
189
+ };
190
+ }
191
+ const ordered = [...values].sort((left, right) => left - right);
192
+ const p90Index = Math.max(0, Math.ceil(ordered.length * 0.9) - 1);
193
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
194
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
195
+ return {
196
+ observations: metrics.length,
197
+ measured: values.length,
198
+ unmeasured: metrics.length - values.length,
199
+ mean,
200
+ min: ordered[0] ?? null,
201
+ max: ordered.at(-1) ?? null,
202
+ p90: ordered[p90Index] ?? null,
203
+ variance,
204
+ standardDeviation: Math.sqrt(variance),
205
+ sources
206
+ };
207
+ }
208
+ function compareSources(left, right) {
209
+ return sourceOrder(left) - sourceOrder(right);
210
+ }
211
+ function sourceOrder(source) {
212
+ if (source === "ledger") return 0;
213
+ if (source === "receipt") return 1;
214
+ return 2;
215
+ }
216
+
217
+ // src/domains/eval/compare/behavioral.ts
218
+ init_esm_shims();
219
+
220
+ // src/domains/eval/compare/envelope.ts
221
+ init_esm_shims();
222
+ function compareEvalExecutionEnvelopesV1(identity, baseline, candidate, baselineDimensions, candidateDimensions) {
223
+ const leftDimensions = [...baselineDimensions].sort();
224
+ const rightDimensions = [...candidateDimensions].sort();
225
+ if (stableJson(leftDimensions) !== stableJson(rightDimensions)) {
226
+ return { ...identity, fields: ["matrix.dimensions"] };
227
+ }
228
+ if (baseline.some((result) => result.executionEnvelope !== void 0) && baseline.some((result) => result.executionEnvelope === void 0) || candidate.some((result) => result.executionEnvelope !== void 0) && candidate.some((result) => result.executionEnvelope === void 0)) {
229
+ return { ...identity, fields: ["executionEnvelope.missingTrial"] };
230
+ }
231
+ const ignored = new Set(leftDimensions);
232
+ const baselineEnvelopes = uniqueEnvelopes(baseline, ignored);
233
+ const candidateEnvelopes = uniqueEnvelopes(candidate, ignored);
234
+ if (baselineEnvelopes.length === 0 && candidateEnvelopes.length === 0) return null;
235
+ if (baselineEnvelopes.length === 0 || candidateEnvelopes.length === 0) {
236
+ return { ...identity, fields: ["executionEnvelope"] };
237
+ }
238
+ if (baselineEnvelopes.length > 1 || candidateEnvelopes.length > 1) {
239
+ return { ...identity, fields: ["executionEnvelope.withinRunVariance"] };
240
+ }
241
+ const left = baselineEnvelopes[0];
242
+ const right = candidateEnvelopes[0];
243
+ if (left === void 0 || right === void 0 || stableJson(left) === stableJson(right)) return null;
244
+ return { ...identity, fields: differingFields(left, right, ignored) };
245
+ }
246
+ function uniqueEnvelopes(results, ignored) {
247
+ const byIdentity = /* @__PURE__ */ new Map();
248
+ for (const result of results) {
249
+ if (result.executionEnvelope === void 0) continue;
250
+ const normalized = normalizedEnvelope(result.executionEnvelope, ignored);
251
+ byIdentity.set(stableJson(normalized), normalized);
252
+ }
253
+ return [...byIdentity.values()];
254
+ }
255
+ function normalizedEnvelope(envelope, ignored) {
256
+ return {
257
+ ...envelope,
258
+ prompt: ignored.has("prompt") ? { fragments: [], compositionHash: null } : envelope.prompt,
259
+ recipe: ignored.has("recipe") ? null : envelope.recipe,
260
+ target: ignored.has("target") ? "<matrix>" : envelope.target,
261
+ wireModel: ignored.has("wireModel") ? null : envelope.wireModel,
262
+ runtime: ignored.has("runtime") ? null : envelope.runtime,
263
+ thinkingLevel: ignored.has("thinkingLevel") ? null : envelope.thinkingLevel,
264
+ toolSignature: ignored.has("toolSignature") ? null : envelope.toolSignature,
265
+ autonomy: ignored.has("autonomy") ? null : envelope.autonomy,
266
+ policyHashes: ignored.has("policy") ? { rulePack: null, project: null } : envelope.policyHashes,
267
+ projectContext: ignored.has("projectContext") ? {
268
+ kind: "none",
269
+ tier: null,
270
+ contentHash: null,
271
+ chars: null,
272
+ sections: [],
273
+ rulesApplied: [],
274
+ operatorProfileApplied: null
275
+ } : envelope.projectContext,
276
+ corpus: ignored.has("corpus") ? { id: "<matrix>", version: "<matrix>" } : envelope.corpus
277
+ };
278
+ }
279
+ function differingFields(left, right, ignored) {
280
+ const fields = [
281
+ ["prompt", "prompt", left.prompt, right.prompt],
282
+ ["recipe", "recipe", left.recipe, right.recipe],
283
+ ["target", "target", left.target, right.target],
284
+ ["wireModel", "wireModel", left.wireModel, right.wireModel],
285
+ ["runtime", "runtime", left.runtime, right.runtime],
286
+ ["thinkingLevel", "thinkingLevel", left.thinkingLevel, right.thinkingLevel],
287
+ ["toolSignature", "toolSignature", left.toolSignature, right.toolSignature],
288
+ ["autonomy", "autonomy", left.autonomy, right.autonomy],
289
+ ["policy", "policyHashes", left.policyHashes, right.policyHashes],
290
+ ["projectContext", "projectContext", left.projectContext, right.projectContext],
291
+ ["corpus", "corpus", left.corpus, right.corpus]
292
+ ];
293
+ return fields.flatMap(
294
+ ([dimension, field, baseline, candidate]) => ignored.has(dimension) || stableJson(baseline) === stableJson(candidate) ? [] : [field]
295
+ );
296
+ }
297
+ function stableJson(value) {
298
+ if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
299
+ if (typeof value === "object" && value !== null) {
300
+ return `{${Object.entries(value).filter(([, entry]) => entry !== void 0).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
301
+ }
302
+ return JSON.stringify(value);
303
+ }
304
+
305
+ // src/domains/eval/compare/behavioral.ts
306
+ function compareEvalBehaviorMetricsV1(baseline, candidate) {
307
+ const baselineGroups = behaviorGroups(baseline);
308
+ const candidateGroups = behaviorGroups(candidate);
309
+ const keys = /* @__PURE__ */ new Set([...baselineGroups.keys(), ...candidateGroups.keys()]);
310
+ const comparisons = [];
311
+ const envelopeMismatches = [];
312
+ const baselineDimensions = baseline.matrix.dimensions ?? [];
313
+ const candidateDimensions = candidate.matrix.dimensions ?? [];
314
+ for (const key of [...keys].sort((left, right) => left.localeCompare(right))) {
315
+ const baselineGroup = baselineGroups.get(key);
316
+ const candidateGroup = candidateGroups.get(key);
317
+ const identity = baselineGroup ?? candidateGroup;
318
+ if (identity === void 0) continue;
319
+ const envelopeMismatch = compareEvalExecutionEnvelopesV1(
320
+ identity,
321
+ baselineGroup?.results ?? [],
322
+ candidateGroup?.results ?? [],
323
+ baselineDimensions,
324
+ candidateDimensions
325
+ );
326
+ if (envelopeMismatch !== null) envelopeMismatches.push(envelopeMismatch);
327
+ const comparability = {
328
+ comparable: envelopeMismatch === null,
329
+ mismatchedFields: envelopeMismatch?.fields ?? []
330
+ };
331
+ for (const definition of EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1) {
332
+ const baselineDistribution = behaviorDistribution(baselineGroup?.results ?? [], definition);
333
+ const candidateDistribution = behaviorDistribution(candidateGroup?.results ?? [], definition);
334
+ comparisons.push({
335
+ scenarioId: identity.scenarioId,
336
+ role: identity.role,
337
+ target: identity.target,
338
+ metric: definition.name,
339
+ family: definition.family,
340
+ direction: definition.direction,
341
+ hardGate: definition.hardGate,
342
+ baseline: baselineDistribution,
343
+ candidate: candidateDistribution,
344
+ change: envelopeMismatch === null ? classifyChange(baselineDistribution.mean, candidateDistribution.mean, definition.direction) : "incomparable",
345
+ meanDelta: subtractNullable(candidateDistribution.mean, baselineDistribution.mean),
346
+ varianceChange: envelopeMismatch === null ? classifyChange(baselineDistribution.variance, candidateDistribution.variance, "lower") : "incomparable",
347
+ varianceDelta: subtractNullable(candidateDistribution.variance, baselineDistribution.variance),
348
+ comparability
349
+ });
350
+ }
351
+ }
352
+ const failures = comparisons.flatMap(
353
+ (comparison) => comparison.hardGate && (comparison.change === "regressed" || comparison.change === "incomparable" && comparison.baseline.mean !== null && comparison.candidate.mean === null) ? [
354
+ {
355
+ scenarioId: comparison.scenarioId,
356
+ role: comparison.role,
357
+ target: comparison.target,
358
+ metric: comparison.metric,
359
+ change: comparison.change
360
+ }
361
+ ] : []
362
+ );
363
+ return {
364
+ comparisons,
365
+ hardGate: {
366
+ pass: failures.length === 0 && envelopeMismatches.length === 0,
367
+ failures,
368
+ envelopeFailures: envelopeMismatches
369
+ },
370
+ envelopeMismatches
371
+ };
372
+ }
373
+ function classifyChange(baseline, candidate, direction) {
374
+ if (baseline === null || candidate === null) return "incomparable";
375
+ if (baseline === candidate) return "unchanged";
376
+ if (direction === "higher") return candidate > baseline ? "improved" : "regressed";
377
+ return candidate < baseline ? "improved" : "regressed";
378
+ }
379
+ function behaviorGroups(artifact) {
380
+ const groups = /* @__PURE__ */ new Map();
381
+ for (const result of artifact.results) {
382
+ const behavioral = result.behavioralMetrics;
383
+ if (behavioral === void 0) continue;
384
+ const key = groupKey(behavioral.scenarioId, behavioral.role, behavioral.target);
385
+ const group = groups.get(key) ?? {
386
+ scenarioId: behavioral.scenarioId,
387
+ role: behavioral.role,
388
+ target: behavioral.target,
389
+ results: []
390
+ };
391
+ group.results.push(result);
392
+ groups.set(key, group);
393
+ }
394
+ return groups;
395
+ }
396
+ function behaviorDistribution(results, definition) {
397
+ const observations = results.map((result) => result.behavioralMetrics?.metrics[definition.name].value ?? null);
398
+ const values = observations.flatMap((value) => value === null ? [] : [value]);
399
+ if (values.length === 0) {
400
+ return {
401
+ observations: observations.length,
402
+ measured: 0,
403
+ unmeasured: observations.length,
404
+ mean: null,
405
+ min: null,
406
+ max: null,
407
+ p90: null,
408
+ variance: null,
409
+ standardDeviation: null,
410
+ source: definition.source
411
+ };
412
+ }
413
+ const ordered = [...values].sort((left, right) => left - right);
414
+ const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
415
+ const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
416
+ return {
417
+ observations: observations.length,
418
+ measured: values.length,
419
+ unmeasured: observations.length - values.length,
420
+ mean,
421
+ min: ordered[0] ?? null,
422
+ max: ordered.at(-1) ?? null,
423
+ p90: ordered[Math.max(0, Math.ceil(ordered.length * 0.9) - 1)] ?? null,
424
+ variance,
425
+ standardDeviation: Math.sqrt(variance),
426
+ source: definition.source
427
+ };
428
+ }
429
+ function groupKey(scenarioId, role, target) {
430
+ return JSON.stringify([scenarioId, role, target.id, target.model]);
431
+ }
432
+ function subtractNullable(left, right) {
433
+ return left === null || right === null ? null : left - right;
434
+ }
435
+
436
+ // src/domains/eval/compare/compare.ts
437
+ var EvalServingConfigurationDriftError = class extends Error {
438
+ baseline;
439
+ candidate;
440
+ constructor(baseline, candidate) {
441
+ super(
442
+ [
443
+ "serving configuration drift; pass --allow-config-drift to compare these runs",
444
+ `baseline serving: ${renderEvalServingConfiguration(baseline)}`,
445
+ `candidate serving: ${renderEvalServingConfiguration(candidate)}`
446
+ ].join("\n")
447
+ );
448
+ this.name = "EvalServingConfigurationDriftError";
449
+ this.baseline = baseline;
450
+ this.candidate = candidate;
451
+ }
452
+ };
453
+ function compareEvalArtifactsV4(baseline, candidate, options = {}) {
75
454
  const baselineTokens = baseline.summary.tokens;
76
455
  const candidateTokens = candidate.summary.tokens;
456
+ const baselineServing = servingConfigurationOf(baseline);
457
+ const candidateServing = servingConfigurationOf(candidate);
458
+ const configDrift = !sameEvalServingConfiguration(baselineServing, candidateServing);
459
+ if (configDrift && options.allowConfigDrift !== true) {
460
+ throw new EvalServingConfigurationDriftError(baselineServing, candidateServing);
461
+ }
462
+ const behavioral = compareEvalBehaviorMetricsV1(baseline, candidate);
463
+ const trackedMetrics = compareTrackedMetrics(baseline, candidate, options.metric);
464
+ const normalizedFilter = normalizeMetricFilter(options.metric);
465
+ const behavioralMetrics = normalizedFilter === void 0 ? behavioral.comparisons : behavioral.comparisons.filter((row) => row.metric === normalizedFilter || row.family === normalizedFilter);
466
+ if (options.metric !== void 0 && trackedMetrics.length === 0 && behavioralMetrics.length === 0) {
467
+ throw new Error(`eval metric not found: ${options.metric}`);
468
+ }
77
469
  return {
78
470
  baselineEvalId: baseline.evalId,
79
471
  candidateEvalId: candidate.evalId,
472
+ baselineServingConfiguration: baselineServing,
473
+ candidateServingConfiguration: candidateServing,
474
+ configDrift,
80
475
  passRateDelta: candidate.summary.passRate - baseline.summary.passRate,
81
476
  tokenDelta: baselineTokens.measured && candidateTokens.measured ? candidateTokens.total - baselineTokens.total : null,
82
- wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs
477
+ wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs,
478
+ trackedMetrics,
479
+ behavioralMetrics,
480
+ hardGate: behavioral.hardGate,
481
+ envelopeMismatches: behavioral.envelopeMismatches,
482
+ scenarioReports: behaviorRollups(behavioralMetrics, (row) => row.scenarioId),
483
+ roleReports: behaviorRollups(behavioralMetrics, (row) => row.role),
484
+ affectedCorpusResults: behavioral.envelopeMismatches.flatMap((mismatch) => {
485
+ const changedFields = mismatch.fields.filter((field) => field === "prompt" || field === "recipe");
486
+ return changedFields.length === 0 ? [] : [{ scenarioId: mismatch.scenarioId, role: mismatch.role, changedFields }];
487
+ })
83
488
  };
84
489
  }
85
490
  function renderEvalComparisonV4(summary) {
491
+ const envelopeFailures = summary.envelopeMismatches.map(
492
+ (mismatch) => ` incomparable envelope: ${mismatch.scenarioId} ${mismatch.role} ${mismatch.target.id}/${mismatch.target.model ?? "none"} fields=${mismatch.fields.join(",")}`
493
+ );
494
+ const affected = summary.affectedCorpusResults.map(
495
+ (result) => ` affected corpus result: ${result.scenarioId} role=${result.role} changed=${result.changedFields.join(",")}`
496
+ );
497
+ const scenarioReports = renderRollups("per-scenario baseline/candidate report", summary.scenarioReports);
498
+ const roleReports = renderRollups("per-role baseline/candidate report", summary.roleReports);
499
+ const hardFailures = summary.hardGate.failures.map(
500
+ (failure) => ` hard failure: ${failure.scenarioId} ${failure.role} ${failure.target.id}/${failure.target.model ?? "none"} ${failure.metric} ${failure.change}`
501
+ );
502
+ const tracked = summary.trackedMetrics.flatMap((row, index) => [
503
+ ...index === 0 ? [
504
+ "tracked metrics:",
505
+ "scenario metric baseline_mean baseline_p90 baseline_variance candidate_mean candidate_p90 candidate_variance mean_delta p90_delta variance_delta change variance_change sources"
506
+ ] : [],
507
+ [
508
+ row.scenarioId,
509
+ row.metric,
510
+ formatMetric(row.baseline.mean),
511
+ formatMetric(row.baseline.p90),
512
+ formatMetric(row.baseline.variance ?? null),
513
+ formatMetric(row.candidate.mean),
514
+ formatMetric(row.candidate.p90),
515
+ formatMetric(row.candidate.variance ?? null),
516
+ formatSignedMetric(row.meanDelta),
517
+ formatSignedMetric(row.p90Delta),
518
+ formatSignedMetric(row.varianceDelta),
519
+ row.change,
520
+ row.varianceChange,
521
+ `${row.baseline.sources.join("+") || "none"}->${row.candidate.sources.join("+") || "none"}`
522
+ ].join(" ")
523
+ ]);
524
+ const behavioral = summary.behavioralMetrics.flatMap((row, index) => [
525
+ ...index === 0 ? [
526
+ "behavioral metrics:",
527
+ "scenario role target model family metric baseline_mean baseline_variance baseline_coverage candidate_mean candidate_variance candidate_coverage mean_delta variance_delta change variance_change comparability gate source"
528
+ ] : [],
529
+ [
530
+ row.scenarioId,
531
+ row.role,
532
+ row.target.id,
533
+ row.target.model ?? "none",
534
+ row.family,
535
+ row.metric,
536
+ formatMetric(row.baseline.mean),
537
+ formatMetric(row.baseline.variance),
538
+ `${row.baseline.measured}/${row.baseline.observations}`,
539
+ formatMetric(row.candidate.mean),
540
+ formatMetric(row.candidate.variance),
541
+ `${row.candidate.measured}/${row.candidate.observations}`,
542
+ formatSignedMetric(row.meanDelta),
543
+ formatSignedMetric(row.varianceDelta),
544
+ row.change,
545
+ row.varianceChange,
546
+ row.comparability.comparable ? "comparable" : `incomparable:${row.comparability.mismatchedFields.join(",")}`,
547
+ row.hardGate ? "hard" : "informational",
548
+ row.baseline.source
549
+ ].join(" ")
550
+ ]);
86
551
  return [
87
552
  `baseline eval: ${summary.baselineEvalId}`,
88
553
  `candidate eval: ${summary.candidateEvalId}`,
554
+ `baseline serving: ${renderEvalServingConfiguration(summary.baselineServingConfiguration)}`,
555
+ `candidate serving: ${renderEvalServingConfiguration(summary.candidateServingConfiguration)}`,
556
+ `config drift: ${summary.configDrift ? "allowed" : "none"}`,
89
557
  `pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
90
558
  `token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
91
559
  `wall-time delta ms: ${summary.wallTimeDelta}`,
560
+ `behavioral hard gate: ${summary.hardGate.pass ? "pass" : `fail (${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length})`}`,
561
+ ...hardFailures,
562
+ ...envelopeFailures,
563
+ ...affected,
564
+ ...scenarioReports,
565
+ ...roleReports,
566
+ ...tracked,
567
+ ...behavioral,
92
568
  ""
93
569
  ].join("\n");
94
570
  }
571
+ function behaviorRollups(rows, keyOf) {
572
+ const groups = /* @__PURE__ */ new Map();
573
+ for (const row of rows) groups.set(keyOf(row), [...groups.get(keyOf(row)) ?? [], row]);
574
+ return [...groups.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([id, grouped]) => ({
575
+ id,
576
+ metrics: changeCounts(grouped.map((row) => row.change)),
577
+ variance: changeCounts(grouped.map((row) => row.varianceChange))
578
+ }));
579
+ }
580
+ function changeCounts(changes) {
581
+ return {
582
+ improved: changes.filter((change) => change === "improved").length,
583
+ regressed: changes.filter((change) => change === "regressed").length,
584
+ unchanged: changes.filter((change) => change === "unchanged").length,
585
+ incomparable: changes.filter((change) => change === "incomparable").length
586
+ };
587
+ }
588
+ function renderRollups(title, reports) {
589
+ if (reports.length === 0) return [];
590
+ return [
591
+ `${title}:`,
592
+ ...reports.map(
593
+ (report) => ` ${report.id}: metrics ${renderChangeCounts(report.metrics)}; variance ${renderChangeCounts(report.variance)}`
594
+ )
595
+ ];
596
+ }
597
+ function renderChangeCounts(counts) {
598
+ return `improved=${counts.improved} regressed=${counts.regressed} unchanged=${counts.unchanged} incomparable=${counts.incomparable}`;
599
+ }
600
+ function compareTrackedMetrics(baseline, candidate, metricFilter) {
601
+ const baselineAggregates = baseline.aggregates ?? aggregateEvalVerdicts(baseline.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
602
+ const candidateAggregates = candidate.aggregates ?? aggregateEvalVerdicts(candidate.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
603
+ const baselineByScenario = new Map(baselineAggregates.map((entry) => [entry.scenarioId, entry]));
604
+ const candidateByScenario = new Map(candidateAggregates.map((entry) => [entry.scenarioId, entry]));
605
+ const scenarioIds = [...baselineByScenario.keys()].filter((scenarioId) => candidateByScenario.has(scenarioId)).sort((left, right) => left.localeCompare(right));
606
+ const filter = normalizeMetricFilter(metricFilter);
607
+ const rows = [];
608
+ for (const scenarioId of scenarioIds) {
609
+ const baselineAggregate = baselineByScenario.get(scenarioId);
610
+ const candidateAggregate = candidateByScenario.get(scenarioId);
611
+ if (baselineAggregate === void 0 || candidateAggregate === void 0) continue;
612
+ for (const metric of EVAL_TRACKED_METRIC_NAMES) {
613
+ if (filter !== void 0 && filter !== metric) continue;
614
+ rows.push(
615
+ metricComparison(
616
+ scenarioId,
617
+ metric,
618
+ baselineAggregate.trackedMetrics[metric],
619
+ candidateAggregate.trackedMetrics[metric]
620
+ )
621
+ );
622
+ }
623
+ const reasons = /* @__PURE__ */ new Set([
624
+ ...Object.keys(baselineAggregate.trackedMetrics.expectedColdReasons),
625
+ ...Object.keys(candidateAggregate.trackedMetrics.expectedColdReasons)
626
+ ]);
627
+ for (const reason of [...reasons].sort((left, right) => left.localeCompare(right))) {
628
+ const metric = `expectedColdReasons.${reason}`;
629
+ if (filter !== void 0 && filter !== metric && filter !== "expectedColdReasons") continue;
630
+ rows.push(
631
+ metricComparison(
632
+ scenarioId,
633
+ metric,
634
+ baselineAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(baselineAggregate.k),
635
+ candidateAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(candidateAggregate.k)
636
+ )
637
+ );
638
+ }
639
+ }
640
+ return rows;
641
+ }
642
+ function metricComparison(scenarioId, metric, baseline, candidate) {
643
+ assertComparableTrackedMetricSources(`${scenarioId}.${metric}`, baseline.sources, candidate.sources);
644
+ const direction = trackedMetricDirection(metric);
645
+ return {
646
+ scenarioId,
647
+ metric,
648
+ baseline,
649
+ candidate,
650
+ meanDelta: subtractNullable2(candidate.mean, baseline.mean),
651
+ p90Delta: subtractNullable2(candidate.p90, baseline.p90),
652
+ varianceDelta: subtractNullable2(candidate.variance ?? null, baseline.variance ?? null),
653
+ change: classifyChange(baseline.mean, candidate.mean, direction),
654
+ varianceChange: classifyChange(baseline.variance ?? null, candidate.variance ?? null, "lower")
655
+ };
656
+ }
657
+ function servingConfigurationOf(artifact) {
658
+ return artifact.servingConfiguration ?? {
659
+ targetId: artifact.matrix.target,
660
+ runtimeId: null,
661
+ modelId: artifact.matrix.model,
662
+ serverBuild: null,
663
+ total_slots: null,
664
+ thinkingLevel: artifact.matrix.thinking,
665
+ compiledPromptHash: null
666
+ };
667
+ }
668
+ function normalizeMetricFilter(metric) {
669
+ if (metric === void 0) return void 0;
670
+ const trimmed = metric.trim();
671
+ if (trimmed.startsWith("trackedMetrics.")) return trimmed.slice("trackedMetrics.".length);
672
+ if (trimmed.startsWith("behavioralMetrics.")) return trimmed.slice("behavioralMetrics.".length);
673
+ return trimmed;
674
+ }
675
+ function trackedMetricDirection(metric) {
676
+ return metric === "cacheReadTokens" ? "higher" : "lower";
677
+ }
678
+ function zeroDistribution(observations) {
679
+ return {
680
+ observations,
681
+ measured: observations,
682
+ unmeasured: 0,
683
+ mean: 0,
684
+ min: 0,
685
+ max: 0,
686
+ p90: 0,
687
+ variance: 0,
688
+ standardDeviation: 0,
689
+ sources: ["ledger"]
690
+ };
691
+ }
692
+ function subtractNullable2(left, right) {
693
+ return left === null || right === null ? null : left - right;
694
+ }
695
+ function formatMetric(value) {
696
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(2);
697
+ }
698
+ function formatSignedMetric(value) {
699
+ if (value === null) return "null";
700
+ const formatted = formatMetric(value);
701
+ return value > 0 ? `+${formatted}` : formatted;
702
+ }
95
703
 
96
704
  // src/domains/eval/compare/gates.ts
97
705
  init_esm_shims();
@@ -102,12 +710,34 @@ var import_yaml = __toESM(require_dist(), 1);
102
710
  import { readFileSync } from "node:fs";
103
711
  function loadThresholds(path) {
104
712
  const parsed = (0, import_yaml.parse)(readFileSync(path, "utf8"));
105
- if (isRecord(parsed) && Array.isArray(parsed.fail)) return { fail: parsed.fail };
106
- if (isRecord(parsed) && isRecord(parsed.thresholds) && Array.isArray(parsed.thresholds.fail)) {
107
- return { fail: parsed.thresholds.fail };
713
+ const root = isRecord(parsed) && isRecord(parsed.thresholds) ? parsed.thresholds : parsed;
714
+ if (isRecord(root) && (Array.isArray(root.fail) || Array.isArray(root.informational))) {
715
+ return {
716
+ fail: parseAssertions(root.fail, `${path}.fail`),
717
+ informational: parseAssertions(root.informational, `${path}.informational`)
718
+ };
108
719
  }
109
720
  throw new Error(`invalid thresholds file: ${path}`);
110
721
  }
722
+ function parseAssertions(value, source) {
723
+ if (value === void 0) return [];
724
+ if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
725
+ return value.map((entry, index) => {
726
+ if (!isRecord(entry)) throw new Error(`${source}[${index}]: expected object`);
727
+ if (typeof entry.metric !== "string" || entry.metric.length === 0) {
728
+ throw new Error(`${source}[${index}].metric: expected non-empty string`);
729
+ }
730
+ if (!isOp(entry.op)) throw new Error(`${source}[${index}].op: expected lt, lte, gt, gte, eq, or neq`);
731
+ if (!isScalar(entry.value)) throw new Error(`${source}[${index}].value: expected scalar`);
732
+ return { metric: entry.metric, op: entry.op, value: entry.value };
733
+ });
734
+ }
735
+ function isOp(value) {
736
+ return value === "lt" || value === "lte" || value === "gt" || value === "gte" || value === "eq" || value === "neq";
737
+ }
738
+ function isScalar(value) {
739
+ return typeof value === "number" && Number.isFinite(value) || typeof value === "string" || typeof value === "boolean";
740
+ }
111
741
  function resolveMetricAssertion(assertion, metrics, artifact) {
112
742
  const actual = metricValue(assertion.metric, metrics, artifact);
113
743
  return { actual, unresolved: actual === null, holds: comparisonHolds(assertion, actual) };
@@ -152,21 +782,26 @@ function isRecord(value) {
152
782
 
153
783
  // src/domains/eval/compare/gates.ts
154
784
  function evaluateGate(artifact, thresholds) {
155
- const failures = [];
156
- for (const assertion of thresholds.fail) {
785
+ const failures = evaluateAssertions(artifact, thresholds.fail);
786
+ const informational = evaluateAssertions(artifact, thresholds.informational ?? []);
787
+ return { pass: failures.length === 0, failures, informational };
788
+ }
789
+ function evaluateAssertions(artifact, assertions) {
790
+ const findings = [];
791
+ for (const assertion of assertions) {
157
792
  const whole = resolveMetricAssertion(assertion, {}, artifact);
158
793
  if (!whole.unresolved) {
159
- if (whole.holds) failures.push({ assertion, actual: whole.actual, unresolved: false });
794
+ if (whole.holds) findings.push({ assertion, actual: whole.actual, unresolved: false });
160
795
  continue;
161
796
  }
162
797
  if (artifact.results.length === 0) {
163
- failures.push({ assertion, actual: null, unresolved: true });
798
+ findings.push({ assertion, actual: null, unresolved: true });
164
799
  continue;
165
800
  }
166
801
  for (const result of artifact.results) {
167
802
  const perRun = resolveMetricAssertion(assertion, result.metrics);
168
803
  if (!perRun.unresolved && !perRun.holds) continue;
169
- failures.push({
804
+ findings.push({
170
805
  assertion,
171
806
  actual: perRun.actual,
172
807
  unresolved: perRun.unresolved,
@@ -175,7 +810,7 @@ function evaluateGate(artifact, thresholds) {
175
810
  });
176
811
  }
177
812
  }
178
- return { pass: failures.length === 0, failures };
813
+ return findings;
179
814
  }
180
815
  function renderGateFailure(failure) {
181
816
  const run = failure.taskId === void 0 ? "" : ` [${failure.taskId}#${failure.repeatIndex ?? 0}]`;
@@ -184,6 +819,121 @@ function renderGateFailure(failure) {
184
819
  ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(failure.actual)}
185
820
  `;
186
821
  }
822
+ function renderInformationalBudget(finding) {
823
+ const run = finding.taskId === void 0 ? "" : ` [${finding.taskId}#${finding.repeatIndex ?? 0}]`;
824
+ const { metric, op, value } = finding.assertion;
825
+ return finding.unresolved ? ` ${metric}${run}: unmeasured informational budget
826
+ ` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(finding.actual)}
827
+ `;
828
+ }
829
+
830
+ // src/domains/eval/reports/comparison.ts
831
+ init_esm_shims();
832
+ function renderEvalComparisonReportV1(summary, format2) {
833
+ if (format2 === "json") return `${JSON.stringify(summary, null, 2)}
834
+ `;
835
+ if (format2 === "md") return renderMarkdown(summary);
836
+ if (format2 === "junit") return renderJunit(summary);
837
+ return renderEvalComparisonV4(summary);
838
+ }
839
+ function renderMarkdown(summary) {
840
+ const rows = summary.behavioralMetrics.map(
841
+ (row) => `| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`
842
+ );
843
+ const scenarioRows = summary.scenarioReports.map(
844
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
845
+ );
846
+ const roleRows = summary.roleReports.map(
847
+ (report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
848
+ );
849
+ return [
850
+ `# Eval comparison ${summary.baselineEvalId} \u2192 ${summary.candidateEvalId}`,
851
+ "",
852
+ `Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
853
+ ...summary.hardGate.failures.map(
854
+ (failure) => `- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`
855
+ ),
856
+ ...summary.envelopeMismatches.map(
857
+ (mismatch) => `- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`
858
+ ),
859
+ ...summary.affectedCorpusResults.map(
860
+ (result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`
861
+ ),
862
+ `Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
863
+ `Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
864
+ `Wall-time delta ms: ${summary.wallTimeDelta}`,
865
+ "",
866
+ "| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
867
+ "|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
868
+ ...rows,
869
+ "",
870
+ "## Per-scenario baseline/candidate report",
871
+ "",
872
+ "| Scenario | Metric changes | Variance changes |",
873
+ "|---|---|---|",
874
+ ...scenarioRows,
875
+ "",
876
+ "## Per-role baseline/candidate report",
877
+ "",
878
+ "| Role | Metric changes | Variance changes |",
879
+ "|---|---|---|",
880
+ ...roleRows,
881
+ ""
882
+ ].join("\n");
883
+ }
884
+ function renderJunit(summary) {
885
+ const failures = new Set(
886
+ summary.hardGate.failures.map(
887
+ (failure) => JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric])
888
+ )
889
+ );
890
+ const represented = /* @__PURE__ */ new Set();
891
+ const cases = summary.behavioralMetrics.map((row) => {
892
+ const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
893
+ const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
894
+ represented.add(key);
895
+ const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
896
+ return failures.has(key) ? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>` : ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
897
+ });
898
+ for (const failure of summary.hardGate.failures) {
899
+ const key = JSON.stringify([
900
+ failure.scenarioId,
901
+ failure.role,
902
+ failure.target.id,
903
+ failure.target.model,
904
+ failure.metric
905
+ ]);
906
+ if (represented.has(key)) continue;
907
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
908
+ cases.push(
909
+ ` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`
910
+ );
911
+ }
912
+ for (const failure of summary.hardGate.envelopeFailures) {
913
+ const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
914
+ cases.push(
915
+ ` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`
916
+ );
917
+ }
918
+ return [
919
+ `<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
920
+ ...cases,
921
+ "</testsuite>",
922
+ ""
923
+ ].join("\n");
924
+ }
925
+ function changeCounts2(counts) {
926
+ return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
927
+ }
928
+ function format(value) {
929
+ return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
930
+ }
931
+ function cell(value) {
932
+ return value.replaceAll("|", "\\|");
933
+ }
934
+ function escapeXml(value) {
935
+ return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
936
+ }
187
937
 
188
938
  // src/domains/eval/reports/json.ts
189
939
  init_esm_shims();
@@ -195,21 +945,35 @@ function renderEvalJsonReportV4(artifact) {
195
945
  // src/domains/eval/reports/junit.ts
196
946
  init_esm_shims();
197
947
  function renderEvalJunitReportV4(artifact) {
948
+ let failures = 0;
949
+ let skipped = 0;
198
950
  const cases = artifact.results.map((result) => {
199
- const name = escapeXml(
951
+ const name = escapeXml2(
200
952
  `${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`
201
953
  );
202
- if (result.pass) return ` <testcase name="${name}" />`;
203
- return ` <testcase name="${name}"><failure message="${escapeXml(result.failureClass ?? "failed")}" /></testcase>`;
954
+ if (!result.pass) {
955
+ failures += 1;
956
+ return ` <testcase name="${name}"><failure message="${escapeXml2(result.failureClass ?? "failed")}" /></testcase>`;
957
+ }
958
+ const outcome = result.behavioral?.outcome;
959
+ if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
960
+ failures += 1;
961
+ return ` <testcase name="${name}"><failure message="${escapeXml2(outcome)}" /></testcase>`;
962
+ }
963
+ if (outcome === "unknown" || outcome === "unmeasured") {
964
+ skipped += 1;
965
+ return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml2(outcome)}" /></testcase>`;
966
+ }
967
+ return ` <testcase name="${name}" />`;
204
968
  }).join("\n");
205
969
  return [
206
- `<testsuite name="${escapeXml(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${artifact.summary.failed}">`,
970
+ `<testsuite name="${escapeXml2(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
207
971
  cases,
208
972
  "</testsuite>",
209
973
  ""
210
974
  ].join("\n");
211
975
  }
212
- function escapeXml(value) {
976
+ function escapeXml2(value) {
213
977
  return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll('"', "&quot;");
214
978
  }
215
979
 
@@ -223,10 +987,10 @@ function renderEvalMarkdownReportV4(artifact) {
223
987
  `Target: ${artifact.matrix.target}`,
224
988
  `Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
225
989
  "",
226
- "| Task | Target | Model | Repeat | Pass | Failure |",
227
- "|---|---|---|---:|---|---|",
990
+ "| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
991
+ "|---|---|---|---|---:|---|---|---|",
228
992
  ...artifact.results.map(
229
- (result) => `| ${result.taskId} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.failureClass ?? ""} |`
993
+ (result) => `| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`
230
994
  ),
231
995
  ""
232
996
  ];
@@ -251,6 +1015,12 @@ function renderEvalSweJsonlReportV4(artifact) {
251
1015
  init_esm_shims();
252
1016
  function renderEvalTextReportV4(artifact) {
253
1017
  const tokens = artifact.summary.tokens;
1018
+ const behavioral = artifact.results.flatMap(
1019
+ (result) => result.behavioral === void 0 ? [] : [result.behavioral.outcome]
1020
+ );
1021
+ const behavioralSummary = behavioral.length === 0 ? [] : [
1022
+ `behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`
1023
+ ];
254
1024
  return [
255
1025
  `eval: ${artifact.evalId}`,
256
1026
  `suite: ${artifact.suite.id}`,
@@ -265,9 +1035,13 @@ function renderEvalTextReportV4(artifact) {
265
1035
  // reported next to how many runs it actually covers.
266
1036
  !tokens.measured ? `tokens total: unmeasured (0 of ${tokens.runs} runs reported usage)` : tokens.measuredRuns === tokens.runs ? `tokens total: ${tokens.total}` : `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
267
1037
  `wall time ms: ${artifact.summary.wallTimeMs}`,
1038
+ ...behavioralSummary,
268
1039
  ""
269
1040
  ].join("\n");
270
1041
  }
1042
+ function count(values, wanted) {
1043
+ return values.filter((value) => value === wanted).length;
1044
+ }
271
1045
 
272
1046
  // src/domains/eval/suites/load.ts
273
1047
  init_esm_shims();
@@ -351,12 +1125,26 @@ function readMatrix(value, path, issues) {
351
1125
  ];
352
1126
  });
353
1127
  if (repeats === null || targets.length === 0) return null;
1128
+ let dimensions;
1129
+ if (value.dimensions !== void 0) {
1130
+ try {
1131
+ dimensions = parseEvalExecutionMatrixDimensionsV1(value.dimensions, `${path}.dimensions`);
1132
+ } catch (error) {
1133
+ issues.push({ path: `${path}.dimensions`, message: error instanceof Error ? error.message : String(error) });
1134
+ return null;
1135
+ }
1136
+ }
354
1137
  const maxCostUsd = value.maxCostUsd;
355
1138
  if (maxCostUsd !== void 0 && (typeof maxCostUsd !== "number" || !Number.isFinite(maxCostUsd) || maxCostUsd < 0)) {
356
1139
  issues.push({ path: `${path}.maxCostUsd`, message: "expected non-negative number" });
357
1140
  return null;
358
1141
  }
359
- return { targets, repeats, ...maxCostUsd === void 0 ? {} : { maxCostUsd } };
1142
+ return {
1143
+ targets,
1144
+ repeats,
1145
+ ...dimensions === void 0 ? {} : { dimensions },
1146
+ ...maxCostUsd === void 0 ? {} : { maxCostUsd }
1147
+ };
360
1148
  }
361
1149
  function readTasks(value, path, issues) {
362
1150
  if (!Array.isArray(value) || value.length === 0) {
@@ -383,10 +1171,12 @@ function readTask(value, path, issues) {
383
1171
  const workspace = readWorkspace(value.workspace, `${path}.workspace`, issues);
384
1172
  const runner = readRunner(value.runner, `${path}.runner`, issues);
385
1173
  const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
1174
+ const behavioral = readBehavioral(value.behavioral, `${path}.behavioral`, issues);
386
1175
  if (id === null || workspace === null || runner === null || timeoutMs === null) return null;
387
1176
  return {
388
1177
  id,
389
1178
  tags: readOptionalStringArray(value, "tags", `${path}.tags`, issues),
1179
+ ...behavioral === void 0 ? {} : { behavioral },
390
1180
  workspace,
391
1181
  runner,
392
1182
  verify: readVerify(value.verify, `${path}.verify`, issues),
@@ -394,6 +1184,15 @@ function readTask(value, path, issues) {
394
1184
  timeoutMs
395
1185
  };
396
1186
  }
1187
+ function readBehavioral(value, path, issues) {
1188
+ if (value === void 0) return void 0;
1189
+ try {
1190
+ return parseEvalBehaviorScenarioV1(value, path);
1191
+ } catch (error) {
1192
+ issues.push({ path, message: error instanceof Error ? error.message : String(error) });
1193
+ return void 0;
1194
+ }
1195
+ }
397
1196
  function readWorkspace(value, path, issues) {
398
1197
  if (!isRecord2(value)) {
399
1198
  issues.push({ path, message: "expected object" });
@@ -430,6 +1229,10 @@ function readRunner(value, path, issues) {
430
1229
  }
431
1230
  const prompt = optionalString(value, "prompt");
432
1231
  const agent = optionalString(value, "agent");
1232
+ const autonomy = optionalString(value, "autonomy");
1233
+ if (autonomy !== void 0 && !["read-only", "suggest", "auto-edit", "full-auto"].includes(autonomy)) {
1234
+ issues.push({ path: `${path}.autonomy`, message: "expected read-only, suggest, auto-edit, or full-auto" });
1235
+ }
433
1236
  if (agent !== void 0 && kind !== "clio-run") {
434
1237
  issues.push({ path: `${path}.agent`, message: "agent is only valid on the clio-run runner" });
435
1238
  }
@@ -437,6 +1240,7 @@ function readRunner(value, path, issues) {
437
1240
  return {
438
1241
  kind,
439
1242
  ...prompt === void 0 ? {} : { prompt },
1243
+ ...autonomy === void 0 ? {} : { autonomy },
440
1244
  ...agent === void 0 ? {} : { agent },
441
1245
  ...command === void 0 ? {} : { command },
442
1246
  commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
@@ -463,7 +1267,22 @@ function readMetrics(value, path, issues) {
463
1267
  issues.push({ path, message: "expected object" });
464
1268
  return { collect: [] };
465
1269
  }
466
- return { collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues) };
1270
+ const observation = value.readObservation;
1271
+ let readObservation;
1272
+ if (observation !== void 0) {
1273
+ if (!isRecord2(observation)) {
1274
+ issues.push({ path: `${path}.readObservation`, message: "expected object" });
1275
+ } else {
1276
+ readObservation = {
1277
+ allowedPaths: readOptionalStringArray(observation, "allowedPaths", `${path}.readObservation.allowedPaths`, issues),
1278
+ decoyPaths: readOptionalStringArray(observation, "decoyPaths", `${path}.readObservation.decoyPaths`, issues)
1279
+ };
1280
+ }
1281
+ }
1282
+ return {
1283
+ collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues),
1284
+ ...readObservation === void 0 ? {} : { readObservation }
1285
+ };
467
1286
  }
468
1287
  function readThresholds(value, path, issues) {
469
1288
  if (value === void 0) return void 0;
@@ -471,7 +1290,10 @@ function readThresholds(value, path, issues) {
471
1290
  issues.push({ path, message: "expected object" });
472
1291
  return void 0;
473
1292
  }
474
- return { fail: readAssertions(value.fail, `${path}.fail`, issues) };
1293
+ return {
1294
+ fail: readAssertions(value.fail, `${path}.fail`, issues),
1295
+ informational: readAssertions(value.informational, `${path}.informational`, issues)
1296
+ };
475
1297
  }
476
1298
  function readAssertions(value, path, issues) {
477
1299
  if (value === void 0) return [];
@@ -620,7 +1442,8 @@ function resolveSuiteForRun(suite, options) {
620
1442
  ...suite,
621
1443
  matrix: {
622
1444
  ...suite.matrix,
623
- targets
1445
+ targets,
1446
+ ...options.trials === void 0 ? {} : { repeats: options.trials }
624
1447
  }
625
1448
  };
626
1449
  }
@@ -646,9 +1469,166 @@ function resolveTargets(targets, options) {
646
1469
 
647
1470
  // src/domains/eval/suites/run.ts
648
1471
  init_esm_shims();
649
- import { mkdtemp as mkdtemp3, rm as rm3 } from "node:fs/promises";
1472
+ import { mkdtemp as mkdtemp3, rm as rm3, writeFile } from "node:fs/promises";
650
1473
  import { tmpdir as tmpdir3 } from "node:os";
651
- import { resolve as resolve6 } from "node:path";
1474
+ import { resolve as resolve7 } from "node:path";
1475
+
1476
+ // src/domains/eval/execution-provenance.ts
1477
+ init_esm_shims();
1478
+ import { createHash as createHash2 } from "node:crypto";
1479
+ function buildEvalExecutionEnvelopeV1(input) {
1480
+ const scenario = input.task.behavioral;
1481
+ if (scenario === void 0) throw new Error(`behavioral task ${input.task.id} has no behavioral scenario`);
1482
+ const manifest = input.ledger.promptManifests.at(-1) ?? null;
1483
+ const contextSnapshot = input.ledger.contextSnapshots.at(-1) ?? null;
1484
+ const recipe = recipeIdentity(
1485
+ input,
1486
+ scenario.execution.subject.kind === "worker" ? scenario.execution.subject.role : null
1487
+ );
1488
+ const policy = policyIdentity(input.cwd, input.receipt, input.observation);
1489
+ const autonomy = input.receipt?.autonomyEnforcement?.autonomy ?? input.observation?.autonomy ?? input.task.runner.autonomy ?? null;
1490
+ const promptFragments = promptFragmentIdentities(manifest, recipe, autonomy);
1491
+ const compositionHash = input.receipt?.staticCompositionHash ?? input.observation?.compositionHash ?? manifest?.systemPromptHash ?? contextSnapshot?.promptHash ?? null;
1492
+ const projectContext = projectContextIdentity(input, manifest, promptFragments);
1493
+ return {
1494
+ schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
1495
+ prompt: { fragments: promptFragments, compositionHash },
1496
+ recipe: recipe === null ? null : { id: recipe.id, version: recipe.version, contentHash: recipe.contentHash },
1497
+ target: input.receipt?.targetId ?? input.observation?.target ?? input.target.id,
1498
+ wireModel: input.receipt?.wireModelId ?? input.observation?.wireModel ?? contextSnapshot?.modelId ?? input.target.model ?? null,
1499
+ runtime: input.receipt?.runtimeId ?? input.observation?.runtime ?? contextSnapshot?.runtimeId ?? null,
1500
+ thinkingLevel: input.receipt?.runtimeResolution?.effectiveThinkingLevel ?? input.observation?.thinkingLevel ?? manifest?.thinkingLevel ?? input.target.thinking ?? null,
1501
+ toolSignature: input.receipt?.toolSignature ?? input.observation?.toolSignature ?? contextSnapshot?.toolSignature ?? null,
1502
+ autonomy,
1503
+ policyHashes: policy,
1504
+ projectContext,
1505
+ corpus: { ...scenario.corpus }
1506
+ };
1507
+ }
1508
+ function recipeIdentity(input, role) {
1509
+ const id = input.receipt?.agentId ?? input.task.runner.agent ?? role;
1510
+ if (id === null || input.cwd === null) return null;
1511
+ try {
1512
+ const recipe = discoverAgentRecipes(input.cwd).find((entry) => entry.id === id);
1513
+ if (recipe === void 0) return null;
1514
+ return {
1515
+ id: recipe.id,
1516
+ version: recipe.version,
1517
+ contentHash: agentSpecFingerprint(normalizeAgentSpec(recipe)),
1518
+ personaHash: sha256(recipe.body)
1519
+ };
1520
+ } catch {
1521
+ return null;
1522
+ }
1523
+ }
1524
+ function promptFragmentIdentities(manifest, recipe, autonomy) {
1525
+ let versions = /* @__PURE__ */ new Map();
1526
+ try {
1527
+ versions = new Map([...loadFragments().byId.values()].map((fragment) => [fragment.id, fragment.version]));
1528
+ } catch {
1529
+ }
1530
+ if (manifest !== null) {
1531
+ return manifest.fragments.map((fragment) => ({
1532
+ id: fragment.id,
1533
+ version: versions.get(fragment.id) ?? "unversioned",
1534
+ contentHash: fragment.contentHash
1535
+ })).sort((left, right) => left.id.localeCompare(right.id));
1536
+ }
1537
+ if (recipe === null) return [];
1538
+ const selected = ["identity.clio-worker", "operating.contract", "operating.worker"];
1539
+ if (autonomy !== null) selected.push(`safety.${autonomy}`);
1540
+ const fragments = [];
1541
+ try {
1542
+ const table = loadFragments();
1543
+ for (const id of selected) {
1544
+ const fragment = table.byId.get(id);
1545
+ if (fragment !== void 0) {
1546
+ fragments.push({ id, version: fragment.version, contentHash: fragment.contentHash });
1547
+ }
1548
+ }
1549
+ } catch {
1550
+ }
1551
+ fragments.push({ id: `persona.${recipe.id}`, version: recipe.version, contentHash: recipe.personaHash });
1552
+ return fragments.sort((left, right) => left.id.localeCompare(right.id));
1553
+ }
1554
+ function policyIdentity(cwd, receipt, observation) {
1555
+ const sealed = receipt?.reproducibility?.safetyPolicy;
1556
+ if (sealed !== void 0) return { rulePack: sealed.rulePackHash, project: sealed.projectPolicyHash };
1557
+ if (observation !== void 0) return { ...observation.policyHashes };
1558
+ if (cwd === null) return { rulePack: null, project: null };
1559
+ try {
1560
+ const metadata = createSafetyPolicyEngine({ cwd }).metadata();
1561
+ return { rulePack: metadata.rulePackHash, project: metadata.projectPolicyHash };
1562
+ } catch {
1563
+ return { rulePack: null, project: null };
1564
+ }
1565
+ }
1566
+ function projectContextIdentity(input, manifest, fragments) {
1567
+ const receipt = input.receipt;
1568
+ if (receipt?.projectContext !== void 0) {
1569
+ const sections = [...receipt.projectContext.sections ?? []].sort();
1570
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1571
+ return {
1572
+ kind: "worker",
1573
+ tier: receipt.projectContext.tier,
1574
+ contentHash: hasContentBearingContext ? receipt.projectContext.contentHash ?? null : null,
1575
+ chars: hasContentBearingContext ? receipt.projectContext.chars ?? null : null,
1576
+ sections,
1577
+ rulesApplied: [...receipt.rulesApplied ?? []].sort(),
1578
+ operatorProfileApplied: receipt.operatorProfileApplied ?? null
1579
+ };
1580
+ }
1581
+ const observed = input.observation?.projectContext;
1582
+ if (observed !== void 0 && observed !== null) {
1583
+ const sections = [...observed.sections].sort();
1584
+ const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
1585
+ return {
1586
+ kind: "worker",
1587
+ tier: observed.tier,
1588
+ contentHash: hasContentBearingContext ? observed.contentHash : null,
1589
+ chars: hasContentBearingContext ? observed.chars : null,
1590
+ sections,
1591
+ rulesApplied: [...observed.rulesApplied].sort(),
1592
+ operatorProfileApplied: observed.operatorProfileApplied
1593
+ };
1594
+ }
1595
+ if (manifest !== null) {
1596
+ const contextFragments = fragments.filter((fragment) => fragment.id.startsWith("context."));
1597
+ const preload = manifest.projectPreload;
1598
+ const identity = {
1599
+ preload,
1600
+ fragments: contextFragments.map((fragment) => [fragment.id, fragment.contentHash])
1601
+ };
1602
+ return {
1603
+ kind: "session",
1604
+ tier: preload?.mode ?? null,
1605
+ contentHash: sha256(stableJson2(identity)),
1606
+ chars: preload?.chars ?? null,
1607
+ sections: contextFragments.map((fragment) => fragment.id).sort(),
1608
+ rulesApplied: [],
1609
+ operatorProfileApplied: contextFragments.some((fragment) => fragment.id === "context.operator-profile")
1610
+ };
1611
+ }
1612
+ return {
1613
+ kind: "none",
1614
+ tier: null,
1615
+ contentHash: null,
1616
+ chars: null,
1617
+ sections: [],
1618
+ rulesApplied: [],
1619
+ operatorProfileApplied: null
1620
+ };
1621
+ }
1622
+ function stableJson2(value) {
1623
+ if (Array.isArray(value)) return `[${value.map(stableJson2).join(",")}]`;
1624
+ if (typeof value === "object" && value !== null) {
1625
+ return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson2(entry)}`).join(",")}}`;
1626
+ }
1627
+ return JSON.stringify(value);
1628
+ }
1629
+ function sha256(value) {
1630
+ return createHash2("sha256").update(value, "utf8").digest("hex");
1631
+ }
652
1632
 
653
1633
  // src/domains/eval/metrics/context.ts
654
1634
  init_esm_shims();
@@ -1295,19 +2275,317 @@ function zeroToolCallMetrics() {
1295
2275
  };
1296
2276
  }
1297
2277
 
2278
+ // src/domains/eval/metrics/tracked.ts
2279
+ init_esm_shims();
2280
+ import { readFile as readFile2 } from "node:fs/promises";
2281
+ import { dirname as dirname2, join as join2 } from "node:path";
2282
+ async function readEvalLedgerSnapshot(stateDir) {
2283
+ const refs = await listSessionLedgerRefs(stateDir);
2284
+ const entries = [];
2285
+ const compiledPromptHashes = [];
2286
+ const promptManifests = [];
2287
+ const contextSnapshots = [];
2288
+ for (const ref of refs) {
2289
+ try {
2290
+ const raw = await readFile2(ref.path, "utf8");
2291
+ entries.push(...parseSessionEntries(raw, ref.path).entries);
2292
+ } catch {
2293
+ }
2294
+ try {
2295
+ const manifest = await readFile2(join2(dirname2(ref.path), "prompt-manifest.jsonl"), "utf8");
2296
+ for (const line of manifest.split(/\r?\n/u)) {
2297
+ const record = parseJsonRecord2(line);
2298
+ if (record === null) continue;
2299
+ const hash = record?.systemPromptHash;
2300
+ if (typeof hash !== "string" || !/^[a-f0-9]{64}$/u.test(hash)) continue;
2301
+ compiledPromptHashes.push(hash);
2302
+ const observation = promptManifestObservation(record);
2303
+ if (observation !== null) promptManifests.push(observation);
2304
+ }
2305
+ } catch {
2306
+ }
2307
+ try {
2308
+ const snapshots = await readFile2(join2(dirname2(ref.path), "context-snapshots.jsonl"), "utf8");
2309
+ for (const line of snapshots.split(/\r?\n/u)) {
2310
+ const record = parseJsonRecord2(line);
2311
+ if (record === null) continue;
2312
+ contextSnapshots.push({
2313
+ runtimeId: nullableString(record.runtimeId),
2314
+ modelId: nullableString(record.modelId),
2315
+ promptHash: nullableDigest(record.promptHash),
2316
+ toolSignature: nullableDigest(record.toolSignature)
2317
+ });
2318
+ }
2319
+ } catch {
2320
+ }
2321
+ }
2322
+ return { entries, compiledPromptHashes: [...new Set(compiledPromptHashes)], promptManifests, contextSnapshots };
2323
+ }
2324
+ function buildEvalTrackedMetrics(input) {
2325
+ const calls = assistantCalls(input.ledgerEntries);
2326
+ const compactionEntries = input.ledgerEntries.filter((entry) => entry.kind === "compactionSummary");
2327
+ const compactionUsage = compactionEntries.flatMap((entry) => {
2328
+ if (entry.kind !== "compactionSummary" || !isRecord5(entry.usage)) return [];
2329
+ return [entry.usage];
2330
+ });
2331
+ const modelCallReadings = calls.map(() => ledgerReading(1));
2332
+ for (const entry of compactionEntries) {
2333
+ if (entry.kind !== "compactionSummary") continue;
2334
+ const apiCalls = isRecord5(entry.usage) ? nonNegativeNumber(entry.usage.apiCalls) : null;
2335
+ modelCallReadings.push(apiCalls === null ? estimatedReading(1) : ledgerReading(apiCalls));
2336
+ }
2337
+ const uncachedReadings = calls.map(uncachedPrefillForCall);
2338
+ const cacheReadings = calls.map(cacheReadForCall);
2339
+ const generatedReadings = calls.map(generatedForCall);
2340
+ for (const usage of compactionUsage) {
2341
+ uncachedReadings.push(readingFromUsage(usage, "input"));
2342
+ cacheReadings.push(readingFromUsage(usage, "cacheRead"));
2343
+ generatedReadings.push(readingFromUsage(usage, "output"));
2344
+ }
2345
+ const reasoning = reasoningMetric(input.receipt, calls, compactionUsage);
2346
+ const receiptToolMetrics = input.receipt === null ? null : evalHarnessMetricsFromReceipt(input.receipt);
2347
+ const ledgerToolCalls = input.ledgerEntries.filter(
2348
+ (entry) => entry.kind === "message" && entry.role === "tool_call"
2349
+ ).length;
2350
+ const ledgerToolErrors = input.ledgerEntries.filter((entry) => {
2351
+ if (entry.kind !== "message" || entry.role !== "tool_result" || !isRecord5(entry.payload)) return false;
2352
+ return entry.payload.isError === true || entry.payload.outcome === "error";
2353
+ }).length;
2354
+ const receiptToolErrors = input.receipt?.toolStats.reduce((sum2, stat) => sum2 + finiteNonNegative(stat.errors), 0);
2355
+ const expectedColdReasons = expectedColdReasonMetrics(calls);
2356
+ return {
2357
+ modelCalls: sumReadings(modelCallReadings, "ledger"),
2358
+ uncachedPrefillTokens: sumReadings(uncachedReadings, "estimated"),
2359
+ cacheReadTokens: sumReadings(cacheReadings, "estimated"),
2360
+ generatedTokens: sumReadings(generatedReadings, "estimated"),
2361
+ reasoningTokens: reasoning,
2362
+ toolCalls: receiptToolMetrics === null ? { value: ledgerToolCalls, source: "ledger" } : { value: receiptToolMetrics.toolCalls, source: "receipt" },
2363
+ toolErrors: receiptToolErrors === void 0 ? { value: ledgerToolErrors, source: "ledger" } : { value: receiptToolErrors, source: "receipt" },
2364
+ ttftMsFirstCall: firstCallTtft(calls),
2365
+ wallClockMs: wallClockMetric(input.receipt, input.fallbackWallClockMs),
2366
+ contextTokensAtEnd: contextTokensAtEnd(calls, compactionEntries),
2367
+ compactions: { value: compactionEntries.length, source: "ledger" },
2368
+ expectedColdReasons
2369
+ };
2370
+ }
2371
+ function emptyEvalTrackedMetrics(source = "estimated") {
2372
+ const zero = () => ({ value: 0, source });
2373
+ return {
2374
+ modelCalls: zero(),
2375
+ uncachedPrefillTokens: zero(),
2376
+ cacheReadTokens: zero(),
2377
+ generatedTokens: zero(),
2378
+ reasoningTokens: { value: null, source },
2379
+ toolCalls: zero(),
2380
+ toolErrors: zero(),
2381
+ ttftMsFirstCall: zero(),
2382
+ wallClockMs: zero(),
2383
+ contextTokensAtEnd: zero(),
2384
+ compactions: zero(),
2385
+ expectedColdReasons: {}
2386
+ };
2387
+ }
2388
+ function assistantCalls(entries) {
2389
+ return entries.flatMap((entry) => {
2390
+ if (entry.kind !== "message" || entry.role !== "assistant" || !isRecord5(entry.payload)) return [];
2391
+ const promptCache = recordField(entry.payload, "promptCache");
2392
+ const timing = recordField(entry.payload, "timing");
2393
+ const usage = recordField(entry.payload, "usage");
2394
+ if (promptCache === null && timing === null && usage === null) return [];
2395
+ return [
2396
+ {
2397
+ payload: entry.payload,
2398
+ promptCache,
2399
+ backend: promptCache === null ? null : recordField(promptCache, "backend"),
2400
+ timing,
2401
+ usage
2402
+ }
2403
+ ];
2404
+ });
2405
+ }
2406
+ function uncachedPrefillForCall(call) {
2407
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2408
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2409
+ if (promptTokens !== null && cachedTokens !== null && cachedTokens <= promptTokens) {
2410
+ return ledgerReading(promptTokens - cachedTokens);
2411
+ }
2412
+ const piInput = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2413
+ if (piInput !== null) return ledgerReading(piInput);
2414
+ const legacyInput = call.usage === null ? null : nonNegativeNumber(call.usage.input);
2415
+ return legacyInput === null ? estimatedReading(0) : estimatedReading(legacyInput);
2416
+ }
2417
+ function cacheReadForCall(call) {
2418
+ const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
2419
+ if (cachedTokens !== null) return ledgerReading(cachedTokens);
2420
+ const piCacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2421
+ if (piCacheRead !== null) return ledgerReading(piCacheRead);
2422
+ const legacyCacheRead = call.usage === null ? null : nonNegativeNumber(call.usage.cacheRead);
2423
+ return legacyCacheRead === null ? estimatedReading(0) : estimatedReading(legacyCacheRead);
2424
+ }
2425
+ function generatedForCall(call) {
2426
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2427
+ if (predictedTokens !== null) return ledgerReading(predictedTokens);
2428
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2429
+ return output === null ? estimatedReading(0) : ledgerReading(output);
2430
+ }
2431
+ function readingFromUsage(usage, field) {
2432
+ const value = nonNegativeNumber(usage[field]);
2433
+ return value === null ? estimatedReading(0) : ledgerReading(value);
2434
+ }
2435
+ function reasoningMetric(receipt, calls, compactionUsage) {
2436
+ if (receipt !== null && typeof receipt.reasoningTokenCount === "number") {
2437
+ return { value: finiteNonNegative(receipt.reasoningTokenCount), source: "receipt" };
2438
+ }
2439
+ let total = 0;
2440
+ let measured = false;
2441
+ for (const call of calls) {
2442
+ const value = extractReasoningTokens(call.usage);
2443
+ if (value === null) continue;
2444
+ measured = true;
2445
+ total += finiteNonNegative(value);
2446
+ }
2447
+ for (const usage of compactionUsage) {
2448
+ const value = nonNegativeNumber(usage.reasoning);
2449
+ if (value === null) continue;
2450
+ measured = true;
2451
+ total += value;
2452
+ }
2453
+ return measured ? { value: total, source: "ledger" } : { value: null, source: "estimated" };
2454
+ }
2455
+ function firstCallTtft(calls) {
2456
+ const first = calls[0];
2457
+ const value = first?.timing === null || first?.timing === void 0 ? null : nonNegativeNumber(first.timing.ttftMs);
2458
+ return value === null ? { value: 0, source: "estimated" } : { value, source: "ledger" };
2459
+ }
2460
+ function wallClockMetric(receipt, fallback) {
2461
+ if (receipt !== null) {
2462
+ const started = Date.parse(receipt.startedAt);
2463
+ const ended = Date.parse(receipt.endedAt);
2464
+ if (Number.isFinite(started) && Number.isFinite(ended) && ended >= started) {
2465
+ return { value: ended - started, source: "receipt" };
2466
+ }
2467
+ }
2468
+ return { value: finiteNonNegative(fallback), source: "estimated" };
2469
+ }
2470
+ function contextTokensAtEnd(calls, compactions) {
2471
+ const lastCall = calls.at(-1);
2472
+ if (lastCall !== void 0) {
2473
+ const lastReading = contextForCall(lastCall);
2474
+ if (lastReading !== null && lastReading > 0) return { value: lastReading, source: "ledger" };
2475
+ if (lastReading === 0) {
2476
+ for (const call of [...calls.slice(0, -1)].reverse()) {
2477
+ const reading = contextForCall(call);
2478
+ if (reading !== null && reading > 0) return { value: reading, source: "ledger" };
2479
+ }
2480
+ }
2481
+ }
2482
+ const lastCompaction = compactions.at(-1);
2483
+ if (lastCompaction?.kind === "compactionSummary") {
2484
+ const tokensAfter = nonNegativeNumber(lastCompaction.tokensAfter);
2485
+ if (tokensAfter !== null) return { value: tokensAfter, source: "ledger" };
2486
+ }
2487
+ return { value: 0, source: "estimated" };
2488
+ }
2489
+ function contextForCall(call) {
2490
+ const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
2491
+ const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
2492
+ if (promptTokens !== null && predictedTokens !== null) return promptTokens + predictedTokens;
2493
+ const input = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
2494
+ const cacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
2495
+ const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
2496
+ return input === null || cacheRead === null || output === null ? null : input + cacheRead + output;
2497
+ }
2498
+ function expectedColdReasonMetrics(calls) {
2499
+ const counts = /* @__PURE__ */ new Map();
2500
+ for (const call of calls) {
2501
+ const reasons = call.promptCache?.expectedColdReasons;
2502
+ if (!Array.isArray(reasons)) continue;
2503
+ const unique = new Set(reasons.filter((reason) => typeof reason === "string" && reason.length > 0));
2504
+ for (const reason of unique) counts.set(reason, (counts.get(reason) ?? 0) + 1);
2505
+ }
2506
+ return Object.fromEntries(
2507
+ [...counts.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([reason, value]) => [reason, { value, source: "ledger" }])
2508
+ );
2509
+ }
2510
+ function sumReadings(readings, emptySource) {
2511
+ if (readings.length === 0) return { value: 0, source: emptySource };
2512
+ return {
2513
+ value: readings.reduce((sum2, reading) => sum2 + reading.value, 0),
2514
+ source: readings.some((reading) => reading.source === "estimated") ? "estimated" : "ledger"
2515
+ };
2516
+ }
2517
+ function ledgerReading(value) {
2518
+ return { value: finiteNonNegative(value), source: "ledger" };
2519
+ }
2520
+ function estimatedReading(value) {
2521
+ return { value: finiteNonNegative(value), source: "estimated" };
2522
+ }
2523
+ function finiteNonNegative(value) {
2524
+ return Number.isFinite(value) && value >= 0 ? value : 0;
2525
+ }
2526
+ function nonNegativeNumber(value) {
2527
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2528
+ }
2529
+ function recordField(record, field) {
2530
+ return isRecord5(record[field]) ? record[field] : null;
2531
+ }
2532
+ function parseJsonRecord2(line) {
2533
+ if (line.trim().length === 0) return null;
2534
+ try {
2535
+ const parsed = JSON.parse(line);
2536
+ return isRecord5(parsed) ? parsed : null;
2537
+ } catch {
2538
+ return null;
2539
+ }
2540
+ }
2541
+ function promptManifestObservation(record) {
2542
+ const systemPromptHash = nullableDigest(record.systemPromptHash);
2543
+ if (systemPromptHash === null || !Array.isArray(record.fragments)) return null;
2544
+ const fragments = record.fragments.flatMap((entry) => {
2545
+ if (!isRecord5(entry) || typeof entry.id !== "string") return [];
2546
+ const contentHash = nullableDigest(entry.contentHash);
2547
+ return contentHash === null ? [] : [{ id: entry.id, contentHash }];
2548
+ });
2549
+ const preload = record.projectPreload;
2550
+ const projectPreload = preload === null ? null : isRecord5(preload) && (preload.mode === "full" || preload.mode === "synopsis" || preload.mode === "none") && typeof preload.chars === "number" && Number.isInteger(preload.chars) && typeof preload.lines === "number" && Number.isInteger(preload.lines) && typeof preload.nearLimit === "boolean" && typeof preload.label === "string" ? {
2551
+ mode: preload.mode,
2552
+ chars: preload.chars,
2553
+ lines: preload.lines,
2554
+ reason: nullableString(preload.reason),
2555
+ nearLimit: preload.nearLimit,
2556
+ label: preload.label
2557
+ } : null;
2558
+ return {
2559
+ systemPromptHash,
2560
+ thinkingLevel: nullableString(record.thinkingLevel),
2561
+ projectPreload,
2562
+ fragments
2563
+ };
2564
+ }
2565
+ function nullableString(value) {
2566
+ return typeof value === "string" && value.length > 0 ? value : null;
2567
+ }
2568
+ function nullableDigest(value) {
2569
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
2570
+ }
2571
+ function isRecord5(value) {
2572
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2573
+ }
2574
+
1298
2575
  // src/domains/eval/runners/clio-run.ts
1299
2576
  init_esm_shims();
2577
+ import { isAbsolute, relative, resolve as resolve2, sep } from "node:path";
1300
2578
 
1301
2579
  // src/domains/eval/metrics/evidence.ts
1302
2580
  init_esm_shims();
1303
2581
  import { readFileSync as readFileSync3 } from "node:fs";
1304
- import { join as join2 } from "node:path";
2582
+ import { join as join3 } from "node:path";
1305
2583
  function dispatchScopeMetrics(receipt) {
1306
2584
  const scope = receipt.pathScope;
1307
2585
  if (scope === void 0) return {};
1308
2586
  const entries = [...scope.workingContextPaths, ...scope.writeBoundaries];
1309
2587
  const evidence = entries.flatMap((entry) => entry.evidence);
1310
- const count = (source) => evidence.filter((entry) => entry.source === source).length;
2588
+ const count2 = (source) => evidence.filter((entry) => entry.source === source).length;
1311
2589
  return {
1312
2590
  "dispatch.scope.mode": scope.mode,
1313
2591
  "dispatch.scope.inferredPathCount": entries.filter(
@@ -1316,9 +2594,9 @@ function dispatchScopeMetrics(receipt) {
1316
2594
  "dispatch.scope.derivedPathCount": entries.filter(
1317
2595
  (entry) => entry.evidence.some((item) => item.provenance === "derived")
1318
2596
  ).length,
1319
- "dispatch.scope.source.task": count("task"),
1320
- "dispatch.scope.source.briefing": count("briefing"),
1321
- "dispatch.scope.source.writeRoots": count("writeRoots")
2597
+ "dispatch.scope.source.task": count2("task"),
2598
+ "dispatch.scope.source.briefing": count2("briefing"),
2599
+ "dispatch.scope.source.writeRoots": count2("writeRoots")
1322
2600
  };
1323
2601
  }
1324
2602
  function receiptFromRunJsonStdout(stdout) {
@@ -1365,7 +2643,7 @@ function evidenceTrustMetrics(receipt, envelope) {
1365
2643
  }
1366
2644
  function readRunEnvelopeForReceipt(receipt, stateDir) {
1367
2645
  try {
1368
- const parsed = JSON.parse(readFileSync3(join2(stateDir, "runs.json"), "utf8"));
2646
+ const parsed = JSON.parse(readFileSync3(join3(stateDir, "runs.json"), "utf8"));
1369
2647
  if (!Array.isArray(parsed)) return null;
1370
2648
  const row = parsed.find(
1371
2649
  (entry) => typeof entry === "object" && entry !== null && entry.id === receipt.runId
@@ -1396,10 +2674,10 @@ function createTokenUsageFold() {
1396
2674
  } catch {
1397
2675
  return;
1398
2676
  }
1399
- if (!isRecord5(event) || event.type !== "message_end") return;
1400
- const message = isRecord5(event.message) ? event.message : void 0;
2677
+ if (!isRecord6(event) || event.type !== "message_end") return;
2678
+ const message = isRecord6(event.message) ? event.message : void 0;
1401
2679
  if (message === void 0 || message.role !== "assistant") return;
1402
- const usage = isRecord5(message.usage) ? message.usage : void 0;
2680
+ const usage = isRecord6(message.usage) ? message.usage : void 0;
1403
2681
  if (usage === void 0) return;
1404
2682
  measured = true;
1405
2683
  const input = numberField2(usage, "input");
@@ -1412,7 +2690,7 @@ function createTokenUsageFold() {
1412
2690
  tokens.cacheRead += cacheRead;
1413
2691
  tokens.cacheWrite += cacheWrite;
1414
2692
  tokens.total += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
1415
- if (isRecord5(usage.cost)) costUsd += numberField2(usage.cost, "total");
2693
+ if (isRecord6(usage.cost)) costUsd += numberField2(usage.cost, "total");
1416
2694
  };
1417
2695
  return {
1418
2696
  push(chunk) {
@@ -1462,14 +2740,115 @@ function numberField2(record, field) {
1462
2740
  const value = record[field];
1463
2741
  return typeof value === "number" && Number.isFinite(value) ? value : 0;
1464
2742
  }
1465
- function isRecord5(value) {
2743
+ function isRecord6(value) {
1466
2744
  return typeof value === "object" && value !== null && !Array.isArray(value);
1467
2745
  }
1468
2746
 
1469
2747
  // src/domains/eval/runners/external-command.ts
1470
2748
  init_esm_shims();
1471
2749
  import { spawn } from "node:child_process";
2750
+ import { performance as performance2 } from "node:perf_hooks";
2751
+
2752
+ // src/domains/eval/metrics/call-ledger-stream.ts
2753
+ init_esm_shims();
1472
2754
  import { performance } from "node:perf_hooks";
2755
+ function createEvalCallLedgerFold(now = () => performance.now()) {
2756
+ const entries = [];
2757
+ let pending = "";
2758
+ let activeStartedAt = null;
2759
+ let activeFirstOutputAt = null;
2760
+ const consume = (line) => {
2761
+ const event = parseRecord(line);
2762
+ if (event === null) return;
2763
+ if (event.type === "message_start" && isAssistantMessage(event.message)) {
2764
+ activeStartedAt = now();
2765
+ activeFirstOutputAt = null;
2766
+ return;
2767
+ }
2768
+ if (event.type === "message_update" && activeStartedAt !== null && activeFirstOutputAt === null) {
2769
+ activeFirstOutputAt = now();
2770
+ return;
2771
+ }
2772
+ if (event.type !== "message_end" || !isAssistantMessage(event.message)) return;
2773
+ const message = event.message;
2774
+ const usage = isRecord7(message.usage) ? message.usage : null;
2775
+ if (usage === null) {
2776
+ activeStartedAt = null;
2777
+ activeFirstOutputAt = null;
2778
+ return;
2779
+ }
2780
+ const endedAt = now();
2781
+ const promptCache = {
2782
+ input: nonNegativeNumber2(usage.input) ?? 0,
2783
+ cacheRead: nonNegativeNumber2(usage.cacheRead) ?? 0,
2784
+ cacheWrite: nonNegativeNumber2(usage.cacheWrite) ?? 0,
2785
+ backendVerdict: "unknown"
2786
+ };
2787
+ if (isRecord7(message.backendTimings)) promptCache.backend = structuredClone(message.backendTimings);
2788
+ const previous = entries.at(-1);
2789
+ entries.push({
2790
+ kind: "message",
2791
+ role: "assistant",
2792
+ turnId: `eval-call-${entries.length + 1}`,
2793
+ parentTurnId: previous?.turnId ?? null,
2794
+ timestamp: messageTimestamp(message.timestamp),
2795
+ payload: {
2796
+ promptCache,
2797
+ timing: {
2798
+ ttftMs: activeStartedAt === null ? null : Math.round(Math.max(0, (activeFirstOutputAt ?? endedAt) - activeStartedAt)),
2799
+ apiMs: activeStartedAt === null ? 0 : Math.round(Math.max(0, endedAt - activeStartedAt))
2800
+ },
2801
+ usage: structuredClone(usage)
2802
+ }
2803
+ });
2804
+ activeStartedAt = null;
2805
+ activeFirstOutputAt = null;
2806
+ };
2807
+ return {
2808
+ push(chunk) {
2809
+ pending += chunk;
2810
+ for (; ; ) {
2811
+ const newline = pending.indexOf("\n");
2812
+ if (newline === -1) break;
2813
+ consume(pending.slice(0, newline).replace(/\r$/u, ""));
2814
+ pending = pending.slice(newline + 1);
2815
+ }
2816
+ },
2817
+ entries() {
2818
+ if (pending.length > 0) {
2819
+ consume(pending.replace(/\r$/u, ""));
2820
+ pending = "";
2821
+ }
2822
+ return structuredClone(entries);
2823
+ }
2824
+ };
2825
+ }
2826
+ function isAssistantMessage(value) {
2827
+ return isRecord7(value) && value.role === "assistant";
2828
+ }
2829
+ function messageTimestamp(value) {
2830
+ const milliseconds = nonNegativeNumber2(value);
2831
+ if (milliseconds === null) return (/* @__PURE__ */ new Date(0)).toISOString();
2832
+ const date = new Date(milliseconds);
2833
+ return Number.isFinite(date.getTime()) ? date.toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
2834
+ }
2835
+ function nonNegativeNumber2(value) {
2836
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
2837
+ }
2838
+ function parseRecord(line) {
2839
+ if (line.trim().length === 0) return null;
2840
+ try {
2841
+ const value = JSON.parse(line);
2842
+ return isRecord7(value) ? value : null;
2843
+ } catch {
2844
+ return null;
2845
+ }
2846
+ }
2847
+ function isRecord7(value) {
2848
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2849
+ }
2850
+
2851
+ // src/domains/eval/runners/external-command.ts
1473
2852
  var OUTPUT_LIMIT = 2e5;
1474
2853
  var OUTPUT_HEAD_LIMIT = 2e4;
1475
2854
  var OUTPUT_TRUNCATION_MARKER = "\n[output middle truncated; tail preserved]\n";
@@ -1484,6 +2863,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1484
2863
  let usage = UNMEASURED_TOKEN_USAGE;
1485
2864
  let streamInvariants = EMPTY_STREAM_INVARIANTS;
1486
2865
  let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
2866
+ const ledgerEntries = [];
1487
2867
  for (const command of commands) {
1488
2868
  const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
1489
2869
  stdout = appendLimited(stdout, result.stdout);
@@ -1492,6 +2872,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1492
2872
  usage = addTokenStreamUsage(usage, result.usage);
1493
2873
  streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
1494
2874
  fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
2875
+ ledgerEntries.push(...result.ledgerEntries);
1495
2876
  if (result.exitCode !== 0) {
1496
2877
  return {
1497
2878
  assignmentId: null,
@@ -1507,7 +2888,8 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1507
2888
  ...fleetLoopMetricEntries(fleetLoops),
1508
2889
  "verifier.exitCode": result.exitCode
1509
2890
  },
1510
- artifacts: {}
2891
+ artifacts: {},
2892
+ ledgerEntries
1511
2893
  };
1512
2894
  }
1513
2895
  }
@@ -1525,18 +2907,20 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
1525
2907
  ...fleetLoopMetricEntries(fleetLoops),
1526
2908
  "verifier.exitCode": 0
1527
2909
  },
1528
- artifacts: {}
2910
+ artifacts: {},
2911
+ ledgerEntries
1529
2912
  };
1530
2913
  }
1531
2914
  function runShellCommand(command, cwd, timeoutMs, env) {
1532
- const started = performance.now();
1533
- return new Promise((resolve8) => {
2915
+ const started = performance2.now();
2916
+ return new Promise((resolve9) => {
1534
2917
  let stdout = "";
1535
2918
  let stderr = "";
1536
2919
  const metricCapture = createJsonlMetricCapture();
1537
2920
  const usageFold = createTokenUsageFold();
1538
2921
  const streamFold = createStreamInvariantFold();
1539
2922
  const fleetLoopFold = createFleetLoopFold();
2923
+ const callLedgerFold = createEvalCallLedgerFold();
1540
2924
  let timedOut = false;
1541
2925
  let settled = false;
1542
2926
  const child = spawn(command, {
@@ -1555,6 +2939,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1555
2939
  usageFold.push(chunk);
1556
2940
  streamFold.push(chunk);
1557
2941
  fleetLoopFold.push(chunk);
2942
+ callLedgerFold.push(chunk);
1558
2943
  });
1559
2944
  child.stderr.on("data", (chunk) => {
1560
2945
  stderr = appendLimited(stderr, chunk);
@@ -1568,7 +2953,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1568
2953
  if (settled) return;
1569
2954
  settled = true;
1570
2955
  clearTimeout(timer);
1571
- resolve8({
2956
+ resolve9({
1572
2957
  command,
1573
2958
  exitCode,
1574
2959
  stdout,
@@ -1576,8 +2961,9 @@ function runShellCommand(command, cwd, timeoutMs, env) {
1576
2961
  usage: usageFold.usage(),
1577
2962
  streamInvariants: streamFold.invariants(),
1578
2963
  fleetLoops: fleetLoopFold.observation(),
2964
+ ledgerEntries: callLedgerFold.entries(),
1579
2965
  stderr,
1580
- wallTimeMs: Math.round(performance.now() - started),
2966
+ wallTimeMs: Math.round(performance2.now() - started),
1581
2967
  timedOut
1582
2968
  });
1583
2969
  };
@@ -1609,7 +2995,7 @@ function createJsonlMetricCapture() {
1609
2995
  } catch {
1610
2996
  return;
1611
2997
  }
1612
- if (!isRecord6(parsed)) return;
2998
+ if (!isRecord8(parsed)) return;
1613
2999
  const compact = compactMetricEvent(parsed);
1614
3000
  if (compact === null) return;
1615
3001
  const encoded = JSON.stringify(compact);
@@ -1663,7 +3049,7 @@ function compactMetricEvent(event) {
1663
3049
  type,
1664
3050
  ...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
1665
3051
  toolName,
1666
- ...toolName === "dispatch" && isRecord6(event.args) ? { args: event.args } : toolName === "code_nav" && isRecord6(event.args) ? { args: { mode: event.args.mode } } : {}
3052
+ ...toolName === "dispatch" && isRecord8(event.args) ? { args: event.args } : toolName === "read" && isRecord8(event.args) ? { args: boundedReadArgs(event.args) } : toolName === "code_nav" && isRecord8(event.args) ? { args: { mode: event.args.mode } } : {}
1667
3053
  };
1668
3054
  }
1669
3055
  if (type === "tool_execution_end") {
@@ -1675,7 +3061,7 @@ function compactMetricEvent(event) {
1675
3061
  ...stringField(event, "outcome") !== void 0 ? { outcome: stringField(event, "outcome") } : {}
1676
3062
  };
1677
3063
  }
1678
- if (type !== "clio_tool_finish" || !isRecord6(event.payload)) return null;
3064
+ if (type !== "clio_tool_finish" || !isRecord8(event.payload)) return null;
1679
3065
  return {
1680
3066
  type,
1681
3067
  payload: {
@@ -1685,7 +3071,14 @@ function compactMetricEvent(event) {
1685
3071
  }
1686
3072
  };
1687
3073
  }
1688
- function isRecord6(value) {
3074
+ function boundedReadArgs(args) {
3075
+ for (const field of ["path", "filePath", "file_path"]) {
3076
+ const value = args[field];
3077
+ if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4096) };
3078
+ }
3079
+ return {};
3080
+ }
3081
+ function isRecord8(value) {
1689
3082
  return typeof value === "object" && value !== null && !Array.isArray(value);
1690
3083
  }
1691
3084
  function stringField(record, field) {
@@ -1694,7 +3087,7 @@ function stringField(record, field) {
1694
3087
  }
1695
3088
 
1696
3089
  // src/domains/eval/runners/clio-run.ts
1697
- async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
3090
+ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env, readObservation) {
1698
3091
  const prompt = runner.prompt ?? "";
1699
3092
  const args = [
1700
3093
  shellQuote(clioEntry),
@@ -1705,12 +3098,14 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1705
3098
  shellQuote(target.id),
1706
3099
  ...target.model === void 0 ? [] : ["--model", shellQuote(target.model)],
1707
3100
  ...target.thinking === void 0 ? [] : ["--thinking", shellQuote(target.thinking)],
3101
+ ...runner.autonomy === void 0 ? [] : ["--autonomy", runner.autonomy],
1708
3102
  shellQuote(prompt)
1709
3103
  ];
1710
3104
  const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
1711
3105
  const tokens = result.usage;
1712
3106
  const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
1713
3107
  const tools = toolCallMetricsFromJsonl(toolMetricStream);
3108
+ const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
1714
3109
  const receipt = receiptFromRunJsonStdout(result.stdout);
1715
3110
  const envelope = receipt === null ? null : readRunEnvelopeForReceipt(receipt, env?.CLIO_CODER_STATE_DIR ?? clioStateDir());
1716
3111
  return {
@@ -1729,6 +3124,7 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1729
3124
  "tools.totalCalls": tools.totalCalls,
1730
3125
  "tools.failed": tools.failed,
1731
3126
  "tools.blocked": tools.blocked,
3127
+ ...behavioralTools,
1732
3128
  "verifier.exitCode": result.exitCode,
1733
3129
  ...receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope }),
1734
3130
  ...receipt === null ? {} : { "evidence.qualityLabel": receipt.quality.typedValidations.length > 0 ? "measured" : "unmeasured" }
@@ -1736,8 +3132,11 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
1736
3132
  artifacts: {
1737
3133
  stdout: result.stdout,
1738
3134
  stderr: result.stderr,
3135
+ callLedger: JSON.stringify(result.ledgerEntries),
1739
3136
  ...receipt === null ? {} : { receipt: JSON.stringify(receipt) }
1740
- }
3137
+ },
3138
+ receipt,
3139
+ ledgerEntries: result.ledgerEntries
1741
3140
  };
1742
3141
  }
1743
3142
  function toolCallMetricsFromJsonl(stdout) {
@@ -1750,7 +3149,7 @@ function toolCallMetricsFromJsonl(stdout) {
1750
3149
  let event;
1751
3150
  try {
1752
3151
  const parsed = JSON.parse(line);
1753
- if (!isRecord7(parsed)) continue;
3152
+ if (!isRecord9(parsed)) continue;
1754
3153
  event = parsed;
1755
3154
  } catch {
1756
3155
  continue;
@@ -1764,7 +3163,7 @@ function toolCallMetricsFromJsonl(stdout) {
1764
3163
  recordToolOutcome(executionEnds, toolOutcome(event) ?? (event.isError === true ? "error" : "ok"));
1765
3164
  continue;
1766
3165
  }
1767
- if (event.type !== "clio_tool_finish" || !isRecord7(event.payload)) continue;
3166
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
1768
3167
  const outcome = toolOutcome(event.payload);
1769
3168
  if (outcome === void 0) continue;
1770
3169
  const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId");
@@ -1776,6 +3175,95 @@ function toolCallMetricsFromJsonl(stdout) {
1776
3175
  }
1777
3176
  return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
1778
3177
  }
3178
+ function toolBehaviorMetricEntriesFromJsonl(stdout, cwd, readObservation) {
3179
+ const starts = /* @__PURE__ */ new Map();
3180
+ const readPaths = /* @__PURE__ */ new Set();
3181
+ const executionEnds = [];
3182
+ const canonicalFinishes = [];
3183
+ const seenExecution = /* @__PURE__ */ new Set();
3184
+ const seenCanonical = /* @__PURE__ */ new Set();
3185
+ for (const line of stdout.split(/\r?\n/)) {
3186
+ if (line.trim().length === 0) continue;
3187
+ let event;
3188
+ try {
3189
+ const parsed = JSON.parse(line);
3190
+ if (!isRecord9(parsed)) continue;
3191
+ event = parsed;
3192
+ } catch {
3193
+ continue;
3194
+ }
3195
+ if (event.type === "tool_execution_start") {
3196
+ const callId2 = stringField2(event, "toolCallId");
3197
+ const tool2 = stringField2(event, "toolName");
3198
+ if (callId2 === void 0 || tool2 === void 0) continue;
3199
+ const path = tool2 === "read" && isRecord9(event.args) ? toolPath(event.args) : null;
3200
+ starts.set(callId2, { tool: tool2, path });
3201
+ if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
3202
+ continue;
3203
+ }
3204
+ if (event.type === "tool_execution_end") {
3205
+ const callId2 = stringField2(event, "toolCallId") ?? null;
3206
+ if (callId2 !== null && seenExecution.has(callId2)) continue;
3207
+ if (callId2 !== null) seenExecution.add(callId2);
3208
+ const tool2 = stringField2(event, "toolName") ?? (callId2 === null ? void 0 : starts.get(callId2)?.tool);
3209
+ if (tool2 === void 0) continue;
3210
+ executionEnds.push({ callId: callId2, tool: tool2, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
3211
+ continue;
3212
+ }
3213
+ if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
3214
+ const outcome = toolOutcome(event.payload);
3215
+ const tool = stringField2(event.payload, "tool");
3216
+ if (outcome === void 0 || tool === void 0) continue;
3217
+ const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId") ?? null;
3218
+ if (callId !== null && seenCanonical.has(callId)) continue;
3219
+ if (callId !== null) seenCanonical.add(callId);
3220
+ canonicalFinishes.push({ callId, tool, outcome });
3221
+ }
3222
+ const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
3223
+ const calls = /* @__PURE__ */ new Map();
3224
+ const blocked = /* @__PURE__ */ new Map();
3225
+ for (const terminal of terminals) {
3226
+ const tool = metricToolName(terminal.tool);
3227
+ calls.set(tool, (calls.get(tool) ?? 0) + 1);
3228
+ if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
3229
+ }
3230
+ const namedTools = /* @__PURE__ */ new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
3231
+ const entries = { "tools.read.distinctPaths": readPaths.size };
3232
+ for (const tool of [...namedTools].sort()) {
3233
+ entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
3234
+ entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
3235
+ }
3236
+ if (readObservation !== void 0) {
3237
+ const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
3238
+ const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
3239
+ entries["tools.read.outsideAllowed"] = [...readPaths].filter(
3240
+ (path) => !allowed.some((root) => pathWithin(path, root))
3241
+ ).length;
3242
+ entries["tools.read.decoyHits"] = [...readPaths].filter(
3243
+ (path) => decoys.some((root) => pathWithin(path, root))
3244
+ ).length;
3245
+ }
3246
+ return entries;
3247
+ }
3248
+ function toolPath(args) {
3249
+ for (const field of ["path", "filePath", "file_path"]) {
3250
+ const value = args[field];
3251
+ if (typeof value === "string" && value.length > 0 && value.length <= 4096) return value;
3252
+ }
3253
+ return null;
3254
+ }
3255
+ function normalizeObservedPath(cwd, path) {
3256
+ const absolute = resolve2(cwd, path);
3257
+ const local = relative(cwd, absolute);
3258
+ return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".").split(sep).join("/");
3259
+ }
3260
+ function pathWithin(path, root) {
3261
+ if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
3262
+ return path === root || path.startsWith(`${root}/`);
3263
+ }
3264
+ function metricToolName(tool) {
3265
+ return tool.toLowerCase().replaceAll(/[^a-z0-9_-]/gu, "_").slice(0, 64) || "unknown";
3266
+ }
1779
3267
  function recordToolOutcome(metrics, outcome) {
1780
3268
  metrics.totalCalls += 1;
1781
3269
  if (outcome === "error") metrics.failed += 1;
@@ -1789,7 +3277,7 @@ function stringField2(record, field) {
1789
3277
  const value = record[field];
1790
3278
  return typeof value === "string" && value.length > 0 ? value : void 0;
1791
3279
  }
1792
- function isRecord7(value) {
3280
+ function isRecord9(value) {
1793
3281
  return typeof value === "object" && value !== null && !Array.isArray(value);
1794
3282
  }
1795
3283
 
@@ -1840,7 +3328,7 @@ function parseContextIndexOutput(stdout) {
1840
3328
  // src/domains/eval/runners/context-init.ts
1841
3329
  init_esm_shims();
1842
3330
  import { existsSync as existsSync2, statSync } from "node:fs";
1843
- import { join as join3 } from "node:path";
3331
+ import { join as join4 } from "node:path";
1844
3332
  async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
1845
3333
  const extraArgs = runner.args ?? [];
1846
3334
  const command = [
@@ -1858,22 +3346,22 @@ async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, e
1858
3346
  ].map(shellQuote).join(" ");
1859
3347
  const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
1860
3348
  const payload = parseInitPayload(result.stdout);
1861
- const candidateGeneration = recordField(payload, "generation");
3349
+ const candidateGeneration = recordField2(payload, "generation");
1862
3350
  const generation = isValidGenerationPayload(payload, candidateGeneration) ? candidateGeneration : null;
1863
3351
  const routeError = generation ? generationRouteError(generation, target) : null;
1864
3352
  const payloadError = generation ? routeError : "context-init runner did not receive a valid JSON generation result";
1865
3353
  const exitCode = result.exitCode === 0 && payloadError ? 1 : result.exitCode;
1866
3354
  const stderr = payloadError ? `${result.stderr}${result.stderr.endsWith("\n") || result.stderr.length === 0 ? "" : "\n"}${payloadError}
1867
3355
  ` : result.stderr;
1868
- const run = recordField(generation, "run");
1869
- const tokens = recordField(run, "tokens");
3356
+ const run = recordField2(generation, "run");
3357
+ const tokens = recordField2(run, "tokens");
1870
3358
  const effectiveTarget = stringField3(run, "targetId");
1871
3359
  const effectiveModel = stringField3(run, "wireModelId");
1872
3360
  const effectiveRuntime = stringField3(run, "runtimeId");
1873
3361
  const effectiveRuntimeKind = stringField3(run, "runtimeKind");
1874
3362
  const effectiveThinking = stringField3(run, "thinkingLevel");
1875
3363
  const structuredOutputMode = stringField3(run, "structuredOutputMode");
1876
- const clioMdPath = join3(cwd, "CLIO-CODER.md");
3364
+ const clioMdPath = join4(cwd, "CLIO-CODER.md");
1877
3365
  const clioMdBytes = existsSync2(clioMdPath) ? statSync(clioMdPath).size : 0;
1878
3366
  return {
1879
3367
  assignmentId: null,
@@ -1927,7 +3415,7 @@ function isNonnegativeFiniteNumber(value) {
1927
3415
  return typeof value === "number" && Number.isFinite(value) && value >= 0;
1928
3416
  }
1929
3417
  function isValidRunPayload(value) {
1930
- const run = recordField(value);
3418
+ const run = recordField2(value);
1931
3419
  if (!run) return false;
1932
3420
  for (const key of ["durationMs", "promptBytes", "outputBytes"]) {
1933
3421
  if (!isNonnegativeFiniteNumber(run[key])) return false;
@@ -1936,7 +3424,7 @@ function isValidRunPayload(value) {
1936
3424
  for (const key of ["toolCalls", "toolFailures", "toolBlocked"]) {
1937
3425
  if (run[key] !== void 0 && !isNonnegativeFiniteNumber(run[key])) return false;
1938
3426
  }
1939
- const tokens = recordField(run, "tokens");
3427
+ const tokens = recordField2(run, "tokens");
1940
3428
  if (run.tokens !== void 0 && !tokens) return false;
1941
3429
  if (tokens) {
1942
3430
  for (const key of ["total", "input", "output", "cacheRead", "cacheWrite", "reasoning"]) {
@@ -1962,14 +3450,14 @@ function isValidGenerationPayload(payload, generation) {
1962
3450
  }
1963
3451
  const runPresent = generation.run !== void 0;
1964
3452
  if (runPresent && !isValidRunPayload(generation.run)) return false;
1965
- const run = recordField(generation, "run");
3453
+ const run = recordField2(generation, "run");
1966
3454
  if (mode === "model" && (parserOutcome !== "parsed" || !hasReceiptIdentity(run))) return false;
1967
3455
  if ((parserOutcome === "parsed" || parserOutcome === "rejected") && !runPresent) return false;
1968
3456
  if (parserOutcome === "rejected" && !hasReceiptIdentity(run)) return false;
1969
3457
  return true;
1970
3458
  }
1971
3459
  function generationRouteError(generation, target) {
1972
- const run = recordField(generation, "run");
3460
+ const run = recordField2(generation, "run");
1973
3461
  if (!run) return null;
1974
3462
  const actualTarget = stringField3(run, "targetId");
1975
3463
  const actualModel = stringField3(run, "wireModelId");
@@ -1988,34 +3476,101 @@ function generationRouteError(generation, target) {
1988
3476
  function parseInitPayload(stdout) {
1989
3477
  try {
1990
3478
  const parsed = JSON.parse(stdout);
1991
- return recordField(parsed);
3479
+ return recordField2(parsed);
1992
3480
  } catch {
1993
3481
  return null;
1994
3482
  }
1995
3483
  }
1996
- function recordField(value, field) {
3484
+ function recordField2(value, field) {
1997
3485
  const selected = field && typeof value === "object" && value !== null && !Array.isArray(value) ? value[field] : value;
1998
3486
  return typeof selected === "object" && selected !== null && !Array.isArray(selected) ? selected : null;
1999
3487
  }
2000
3488
  function numberField3(value, field) {
2001
- const record = recordField(value);
3489
+ const record = recordField2(value);
2002
3490
  const selected = record?.[field];
2003
3491
  return typeof selected === "number" && Number.isFinite(selected) ? selected : null;
2004
3492
  }
2005
3493
  function stringField3(value, field) {
2006
- const record = recordField(value);
3494
+ const record = recordField2(value);
2007
3495
  const selected = record?.[field];
2008
3496
  return typeof selected === "string" && selected.length > 0 ? selected : null;
2009
3497
  }
2010
3498
 
3499
+ // src/domains/eval/schema/adapter.ts
3500
+ init_esm_shims();
3501
+ import { createHash as createHash3 } from "node:crypto";
3502
+ function adaptSuiteV2ResultToVerdictV1(result, trackedMetrics) {
3503
+ const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
3504
+ const outcome = result.pass ? "pass" : "fail";
3505
+ const graderExitCode = result.metrics["task.exitCode"];
3506
+ return parseEvalVerdictEnvelopeV1({
3507
+ schema: EVAL_VERDICT_SCHEMA_V1,
3508
+ scenarioId: result.taskId,
3509
+ trialIndex: result.repeatIndex,
3510
+ outcome,
3511
+ machinery,
3512
+ reason: result.pass ? null : result.failureClass ?? "result_failed",
3513
+ trackedMetrics,
3514
+ behavioral: null,
3515
+ evidence: {
3516
+ assignmentId: result.assignmentId,
3517
+ terminalReceiptDigest: result.terminalReceiptDigest,
3518
+ graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null
3519
+ }
3520
+ });
3521
+ }
3522
+ function adaptSuiteV2ResultToBehaviorV1(result, verdict, scenario) {
3523
+ const requestedFacts = new Set(
3524
+ [...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
3525
+ (rule) => `${rule.fact.source}\0${rule.fact.key}`
3526
+ )
3527
+ );
3528
+ const observedSources = /* @__PURE__ */ new Set();
3529
+ const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
3530
+ if (value === null) return [];
3531
+ const source = metricFactSource(key);
3532
+ observedSources.add(source);
3533
+ if (!requestedFacts.has(`${source}\0${key}`)) return [];
3534
+ const serialized = JSON.stringify({ source, key, value });
3535
+ const digest = createHash3("sha256").update(serialized, "utf8").digest("hex");
3536
+ const fact = {
3537
+ id: `metric-${digest.slice(0, 16)}`,
3538
+ source,
3539
+ key,
3540
+ value,
3541
+ evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1e3) }
3542
+ };
3543
+ return [fact];
3544
+ });
3545
+ const allSources = ["transcript", "tool", "receipt", "grader"];
3546
+ const unavailableSources = allSources.filter(
3547
+ (source) => !observedSources.has(source) || source === "tool" && scenario.execution.toolTarget === "none"
3548
+ );
3549
+ const behavior = judgeEvalBehaviorV1(scenario, verdict, {
3550
+ facts,
3551
+ unavailableSources,
3552
+ infrastructureFailure: verdict.machinery === "infrastructure_failure"
3553
+ });
3554
+ assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
3555
+ return behavior;
3556
+ }
3557
+ function metricFactSource(key) {
3558
+ if (key.startsWith("tools.")) return "tool";
3559
+ if (key.startsWith("task.") || key.startsWith("claims.") || key.startsWith("completion.") || key === "result.pass" || key === "verifier.exitCode")
3560
+ return "grader";
3561
+ if (key.startsWith("receipt.") || key.startsWith("evidence.") || key.startsWith("boundary.") || key.startsWith("loop.") || key.startsWith("cost."))
3562
+ return "receipt";
3563
+ return "transcript";
3564
+ }
3565
+
2011
3566
  // src/domains/eval/verifiers/command.ts
2012
3567
  init_esm_shims();
2013
- async function runCommandVerifiers(commands, cwd, timeoutMs) {
3568
+ async function runCommandVerifiers(commands, cwd, timeoutMs, env) {
2014
3569
  let stdout = "";
2015
3570
  let stderr = "";
2016
3571
  let wallTimeMs = 0;
2017
3572
  for (const command of commands) {
2018
- const result = await runShellCommand(command, cwd, timeoutMs);
3573
+ const result = await runShellCommand(command, cwd, timeoutMs, env);
2019
3574
  stdout += result.stdout;
2020
3575
  stderr += result.stderr;
2021
3576
  wallTimeMs += result.wallTimeMs;
@@ -2027,9 +3582,9 @@ async function runCommandVerifiers(commands, cwd, timeoutMs) {
2027
3582
  // src/domains/eval/verifiers/file-exists.ts
2028
3583
  init_esm_shims();
2029
3584
  import { existsSync as existsSync3 } from "node:fs";
2030
- import { resolve as resolve2 } from "node:path";
3585
+ import { resolve as resolve3 } from "node:path";
2031
3586
  function forbiddenPathHits(cwd, paths) {
2032
- return paths.filter((path) => existsSync3(resolve2(cwd, path)));
3587
+ return paths.filter((path) => existsSync3(resolve3(cwd, path)));
2033
3588
  }
2034
3589
 
2035
3590
  // src/domains/eval/verifiers/patch.ts
@@ -2051,10 +3606,10 @@ init_esm_shims();
2051
3606
  import { spawn as spawn2 } from "node:child_process";
2052
3607
  import { mkdtemp, rm } from "node:fs/promises";
2053
3608
  import { tmpdir } from "node:os";
2054
- import { resolve as resolve3 } from "node:path";
3609
+ import { resolve as resolve4 } from "node:path";
2055
3610
  async function prepareGitWorkspace(workspace) {
2056
3611
  if (workspace.url === void 0) throw new Error("git workspace requires url");
2057
- const dest = await mkdtemp(resolve3(tmpdir(), "clio-eval-git-"));
3612
+ const dest = await mkdtemp(resolve4(tmpdir(), "clio-eval-git-"));
2058
3613
  try {
2059
3614
  await runGit(["clone", "--quiet", workspace.url, dest], process.cwd());
2060
3615
  const ref = workspace.checkout ?? workspace.commit;
@@ -2089,9 +3644,9 @@ function runGit(args, cwd) {
2089
3644
  // src/domains/eval/workspaces/local.ts
2090
3645
  init_esm_shims();
2091
3646
  import { access } from "node:fs/promises";
2092
- import { resolve as resolve4 } from "node:path";
3647
+ import { resolve as resolve5 } from "node:path";
2093
3648
  async function prepareLocalWorkspace(baseDir, workspace) {
2094
- const dir = resolve4(baseDir, workspace.path ?? ".");
3649
+ const dir = resolve5(baseDir, workspace.path ?? ".");
2095
3650
  await access(dir);
2096
3651
  return { dir, cleanup: async () => {
2097
3652
  } };
@@ -2099,28 +3654,113 @@ async function prepareLocalWorkspace(baseDir, workspace) {
2099
3654
 
2100
3655
  // src/domains/eval/workspaces/temp-copy.ts
2101
3656
  init_esm_shims();
2102
- import { cp, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
3657
+ import { execFile } from "node:child_process";
3658
+ import { cp, lstat, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
2103
3659
  import { tmpdir as tmpdir2 } from "node:os";
2104
- import { relative, resolve as resolve5 } from "node:path";
2105
- async function prepareTempCopyWorkspace(baseDir, workspace) {
2106
- const source = resolve5(baseDir, workspace.path ?? ".");
2107
- const dest = await mkdtemp2(resolve5(tmpdir2(), "clio-eval-workspace-"));
2108
- const excludes = workspace.excludes ?? [];
2109
- await cp(source, dest, {
2110
- recursive: true,
2111
- filter: (path) => !isExcluded(relative(source, path), excludes)
2112
- });
2113
- return {
2114
- dir: dest,
2115
- cleanup: async () => {
2116
- await rm2(dest, { recursive: true, force: true });
2117
- }
2118
- };
3660
+ import { relative as relative2, resolve as resolve6 } from "node:path";
3661
+ import { promisify } from "node:util";
3662
+ var execFileAsync = promisify(execFile);
3663
+ var GIT_FILE_LIST_LIMIT_BYTES = 128 * 1024 * 1024;
3664
+ async function prepareTempCopyWorkspace(baseDir, workspace, options = {}) {
3665
+ const source = resolve6(baseDir, workspace.path ?? ".");
3666
+ const dest = await mkdtemp2(resolve6(options.tempRoot ?? tmpdir2(), "clio-eval-workspace-"));
3667
+ try {
3668
+ const selection = await gitCopySelection(source);
3669
+ const excludes = workspace.excludes ?? [];
3670
+ const copyWorkspace = options.copy ?? defaultCopy;
3671
+ await copyWorkspace(source, dest, {
3672
+ recursive: true,
3673
+ filter: (path) => shouldCopy(relative2(source, path), excludes, selection)
3674
+ });
3675
+ return {
3676
+ dir: dest,
3677
+ cleanup: async () => {
3678
+ await rm2(dest, { recursive: true, force: true });
3679
+ }
3680
+ };
3681
+ } catch (error) {
3682
+ await rm2(dest, { recursive: true, force: true });
3683
+ throw error;
3684
+ }
2119
3685
  }
2120
3686
  function isExcluded(rel, excludes) {
2121
3687
  const normalized = rel.replaceAll("\\", "/");
2122
3688
  return excludes.some((entry) => normalized === entry || normalized.startsWith(`${entry.replaceAll("\\", "/")}/`));
2123
3689
  }
3690
+ async function defaultCopy(source, destination, options) {
3691
+ await cp(source, destination, options);
3692
+ }
3693
+ function shouldCopy(relativePath, excludes, selection) {
3694
+ const normalized = relativePath.replaceAll("\\", "/");
3695
+ if (normalized.length === 0) return true;
3696
+ if (isExcluded(normalized, excludes)) return false;
3697
+ if (selection === null) return true;
3698
+ return selection.files.has(normalized) || selection.directories.has(normalized);
3699
+ }
3700
+ async function gitCopySelection(source) {
3701
+ let inside;
3702
+ try {
3703
+ inside = await gitOutput(source, ["rev-parse", "--is-inside-work-tree"]);
3704
+ } catch (error) {
3705
+ if (await hasGitMarker(source)) throw error;
3706
+ return null;
3707
+ }
3708
+ if (inside.trim() !== "true") return null;
3709
+ const output = await gitOutput(source, [
3710
+ "--literal-pathspecs",
3711
+ "ls-files",
3712
+ "-z",
3713
+ "--cached",
3714
+ "--others",
3715
+ "--exclude-standard",
3716
+ "--",
3717
+ "."
3718
+ ]);
3719
+ const files = /* @__PURE__ */ new Set();
3720
+ const directories = /* @__PURE__ */ new Set();
3721
+ for (const path of output.split("\0")) {
3722
+ if (path.length === 0) continue;
3723
+ const normalized = normalizeGitPath(path);
3724
+ if (normalized === null) throw new Error("git ls-files returned a path outside the eval workspace");
3725
+ files.add(normalized);
3726
+ let separator = normalized.lastIndexOf("/");
3727
+ while (separator >= 0) {
3728
+ directories.add(normalized.slice(0, separator));
3729
+ separator = normalized.lastIndexOf("/", separator - 1);
3730
+ }
3731
+ }
3732
+ return { files, directories };
3733
+ }
3734
+ async function gitOutput(cwd, args) {
3735
+ const { stdout } = await execFileAsync("git", [...args], {
3736
+ cwd,
3737
+ encoding: "utf8",
3738
+ maxBuffer: GIT_FILE_LIST_LIMIT_BYTES
3739
+ });
3740
+ return stdout;
3741
+ }
3742
+ function normalizeGitPath(path) {
3743
+ const normalized = path.replaceAll("\\", "/").replace(/^\.\//u, "");
3744
+ if (normalized.length === 0 || normalized.startsWith("/") || /^[A-Za-z]:\//u.test(normalized)) return null;
3745
+ const segments = normalized.split("/");
3746
+ if (segments.some((segment) => segment.length === 0 || segment === "." || segment === "..")) return null;
3747
+ return normalized;
3748
+ }
3749
+ async function hasGitMarker(source) {
3750
+ let current = resolve6(source);
3751
+ while (true) {
3752
+ try {
3753
+ await lstat(resolve6(current, ".git"));
3754
+ return true;
3755
+ } catch (error) {
3756
+ const code = typeof error === "object" && error !== null && "code" in error ? error.code : void 0;
3757
+ if (code !== "ENOENT" && code !== "ENOTDIR") throw error;
3758
+ }
3759
+ const parent = resolve6(current, "..");
3760
+ if (parent === current) return false;
3761
+ current = parent;
3762
+ }
3763
+ }
2124
3764
 
2125
3765
  // src/domains/eval/suites/matrix.ts
2126
3766
  init_esm_shims();
@@ -2148,28 +3788,39 @@ async function runEvalSuiteV2(loaded, options) {
2148
3788
  const started = now();
2149
3789
  const evalId = createEvalId(started, loaded.hash);
2150
3790
  const results = [];
3791
+ const servingObservations = [];
2151
3792
  const maxCostUsd = loaded.suite.matrix.maxCostUsd;
2152
3793
  let spentUsd = 0;
2153
3794
  for (const item of expandEvalMatrix(loaded.suite)) {
2154
3795
  if (maxCostUsd !== void 0 && spentUsd > maxCostUsd) {
2155
- results.push(budgetExhaustedResult(item.task.id, item.target, item.repeatIndex, spentUsd, maxCostUsd));
3796
+ results.push(budgetExhaustedResult(loaded, item.task, item.target, item.repeatIndex, spentUsd, maxCostUsd));
2156
3797
  continue;
2157
3798
  }
2158
- const result = await runMatrixItem(loaded, item.task, item.target, item.repeatIndex, options.clioEntry);
2159
- spentUsd += resultCostUsd(result);
2160
- results.push(result);
3799
+ const completed = await runMatrixItem(
3800
+ loaded,
3801
+ item.task,
3802
+ item.target,
3803
+ item.repeatIndex,
3804
+ options.clioEntry,
3805
+ options.freshWorkspaces === true,
3806
+ options.tempCopy
3807
+ );
3808
+ spentUsd += resultCostUsd(completed.result);
3809
+ results.push(completed.result);
3810
+ servingObservations.push(completed.serving);
2161
3811
  }
2162
- return buildArtifact(loaded, evalId, results, options.clioEntry);
3812
+ const serving = await evalServingConfiguration(loaded.suite.matrix.targets, servingObservations);
3813
+ return buildArtifact(loaded, evalId, results, options.clioEntry, serving);
2163
3814
  }
2164
3815
  function resultCostUsd(result) {
2165
3816
  const value = result.metrics["cost.usd"];
2166
3817
  return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0;
2167
3818
  }
2168
- function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd) {
2169
- return {
3819
+ function budgetExhaustedResult(loaded, task, target, repeatIndex, spentUsd, maxCostUsd) {
3820
+ const result = {
2170
3821
  assignmentId: null,
2171
3822
  terminalReceiptDigest: null,
2172
- taskId,
3823
+ taskId: task.id,
2173
3824
  repeatIndex,
2174
3825
  target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
2175
3826
  pass: false,
@@ -2184,21 +3835,36 @@ function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd
2184
3835
  error: `matrix cost budget exhausted: spent $${spentUsd.toFixed(4)} of max $${maxCostUsd.toFixed(4)} before this item`
2185
3836
  }
2186
3837
  };
3838
+ result.verdict = adaptSuiteV2ResultToVerdictV1(result, emptyEvalTrackedMetrics());
3839
+ attachBehavioralResult(result, task);
3840
+ attachExecutionEnvelope(result, task, target, loaded.baseDir, null, emptyLedgerSnapshot());
3841
+ return result;
2187
3842
  }
2188
- async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
3843
+ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry, freshWorkspace, tempCopy) {
2189
3844
  let workspace = null;
2190
- const stateDir = await mkdtemp3(resolve6(tmpdir3(), "clio-eval-state-"));
3845
+ let receipt = null;
3846
+ let runnerWallTimeMs = 0;
3847
+ let executionObservation;
3848
+ const stateDir = await mkdtemp3(resolve7(tempCopy?.tempRoot ?? tmpdir3(), "clio-eval-state-"));
2191
3849
  try {
2192
- workspace = await prepareWorkspace(loaded.baseDir, task);
3850
+ workspace = await prepareWorkspace(loaded.baseDir, task, freshWorkspace, tempCopy);
2193
3851
  const setup = await runCommandVerifiers(task.workspace.setup ?? [], workspace.dir, task.timeoutMs);
2194
3852
  if (!setup.pass) throw new EvalWorkspaceSetupError(setup.exitCode, setup.stderr);
2195
3853
  const runner = await runTaskRunner(task, target, workspace.dir, clioEntry, {
2196
3854
  CLIO_CODER_STATE_DIR: stateDir,
2197
3855
  CLIO_CODER_ENTRY: clioEntry
2198
3856
  });
3857
+ const runnerStdoutFile = resolve7(stateDir, "eval-runner-output.jsonl");
3858
+ await writeFile(runnerStdoutFile, runner.stdout, "utf8");
3859
+ receipt = runner.receipt ?? null;
3860
+ runnerWallTimeMs = runner.wallTimeMs;
2199
3861
  const patch = collectPatchMetrics(workspace.dir);
2200
3862
  const receiptExitCode = runner.exitCode;
2201
3863
  const journalMetrics = invariantMetrics(stateDir, receiptExitCode);
3864
+ const measurement = await measureTaskOutcome(task, workspace.dir, {
3865
+ CLIO_EVAL_RUNNER_STDOUT_FILE: runnerStdoutFile
3866
+ });
3867
+ executionObservation = measurement.executionObservation;
2202
3868
  const metrics = {
2203
3869
  ...zeroToolCallMetrics(),
2204
3870
  ...collectContextMetrics(workspace.dir),
@@ -2214,15 +3880,16 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2214
3880
  "patch.testFilesModified": patch.testFilesModified,
2215
3881
  "result.pass": runner.exitCode === 0,
2216
3882
  "result.failureClass": runner.exitCode === 0 ? null : "runner_failed",
2217
- ...await measureTaskOutcome(task, workspace.dir)
3883
+ ...measurement.metrics
2218
3884
  };
2219
3885
  const verifier = await runVerifiers(task, workspace.dir, metrics);
2220
- const pass = runner.exitCode === 0 && verifier.pass;
2221
- const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : verifier.failureClass;
3886
+ const graderFailed = metrics["task.solved"] === false;
3887
+ const pass = runner.exitCode === 0 && verifier.pass && !graderFailed;
3888
+ const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : !verifier.pass ? verifier.failureClass : "grader_failed";
2222
3889
  metrics["verifier.exitCode"] = verifier.exitCode;
2223
3890
  metrics["result.pass"] = pass;
2224
3891
  metrics["result.failureClass"] = failureClass;
2225
- return {
3892
+ const result = {
2226
3893
  assignmentId: runner.assignmentId,
2227
3894
  terminalReceiptDigest: runner.terminalReceiptDigest,
2228
3895
  taskId: task.id,
@@ -2233,13 +3900,30 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2233
3900
  metrics,
2234
3901
  artifacts: {
2235
3902
  ...runner.artifacts,
3903
+ workspace: workspace.dir,
2236
3904
  ...verifier.stdout.length > 0 ? { verifierStdout: verifier.stdout } : {},
2237
3905
  ...verifier.stderr.length > 0 ? { verifierStderr: verifier.stderr } : {}
2238
3906
  }
2239
3907
  };
3908
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3909
+ const ledgerEntries = [...snapshot.entries, ...runner.ledgerEntries ?? []];
3910
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3911
+ result,
3912
+ buildEvalTrackedMetrics({
3913
+ ledgerEntries,
3914
+ receipt: receipt ?? null,
3915
+ fallbackWallClockMs: runner.wallTimeMs
3916
+ })
3917
+ );
3918
+ attachBehavioralResult(result, task);
3919
+ attachExecutionEnvelope(result, task, target, workspace.dir, receipt ?? null, snapshot, executionObservation);
3920
+ return {
3921
+ result,
3922
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
3923
+ };
2240
3924
  } catch (error) {
2241
3925
  const failureClass = error instanceof EvalWorkspaceSetupError ? "setup_failed" : "command_error";
2242
- return {
3926
+ const result = {
2243
3927
  assignmentId: null,
2244
3928
  terminalReceiptDigest: null,
2245
3929
  taskId: task.id,
@@ -2253,13 +3937,61 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
2253
3937
  "verifier.exitCode": 1,
2254
3938
  "latency.wallMs": 0
2255
3939
  },
2256
- artifacts: { error: error instanceof Error ? error.message : String(error) }
3940
+ artifacts: {
3941
+ error: error instanceof Error ? error.message : String(error),
3942
+ ...workspace === null ? {} : { workspace: workspace.dir }
3943
+ }
3944
+ };
3945
+ const snapshot = await readEvalLedgerSnapshot(stateDir);
3946
+ result.verdict = adaptSuiteV2ResultToVerdictV1(
3947
+ result,
3948
+ buildEvalTrackedMetrics({
3949
+ ledgerEntries: snapshot.entries,
3950
+ receipt: receipt ?? null,
3951
+ fallbackWallClockMs: runnerWallTimeMs
3952
+ })
3953
+ );
3954
+ attachBehavioralResult(result, task);
3955
+ attachExecutionEnvelope(
3956
+ result,
3957
+ task,
3958
+ target,
3959
+ workspace?.dir ?? loaded.baseDir,
3960
+ receipt ?? null,
3961
+ snapshot,
3962
+ executionObservation
3963
+ );
3964
+ return {
3965
+ result,
3966
+ serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
2257
3967
  };
2258
3968
  } finally {
2259
- await workspace?.cleanup();
2260
- await rm3(stateDir, { recursive: true, force: true });
3969
+ try {
3970
+ await workspace?.cleanup();
3971
+ } finally {
3972
+ await rm3(stateDir, { recursive: true, force: true });
3973
+ }
2261
3974
  }
2262
3975
  }
3976
+ function attachBehavioralResult(result, task) {
3977
+ if (task.behavioral === void 0 || result.verdict === void 0) return;
3978
+ result.behavioral = adaptSuiteV2ResultToBehaviorV1(result, result.verdict, task.behavioral);
3979
+ result.behavioralMetrics = buildEvalBehaviorMetricsV1(result, task.behavioral.execution.subject.role);
3980
+ }
3981
+ function attachExecutionEnvelope(result, task, target, cwd, receipt, ledger, observation) {
3982
+ if (task.behavioral === void 0) return;
3983
+ result.executionEnvelope = buildEvalExecutionEnvelopeV1({
3984
+ task,
3985
+ target,
3986
+ cwd,
3987
+ receipt,
3988
+ ledger,
3989
+ ...observation === void 0 ? {} : { observation }
3990
+ });
3991
+ }
3992
+ function emptyLedgerSnapshot() {
3993
+ return { entries: [], compiledPromptHashes: [], promptManifests: [], contextSnapshots: [] };
3994
+ }
2263
3995
  function invariantMetrics(stateDir, runnerExitCode) {
2264
3996
  const journal = readRunJournal(stateDir);
2265
3997
  return {
@@ -2270,23 +4002,80 @@ function invariantMetrics(stateDir, runnerExitCode) {
2270
4002
  ...writeBoundaryInvariantMetrics(stateDir)
2271
4003
  };
2272
4004
  }
2273
- async function prepareWorkspace(baseDir, task) {
4005
+ async function prepareWorkspace(baseDir, task, freshWorkspace, tempCopy) {
4006
+ if (task.workspace.kind === "local" && freshWorkspace) {
4007
+ return prepareTempCopyWorkspace(baseDir, { ...task.workspace, kind: "temp-copy" }, tempCopy);
4008
+ }
2274
4009
  if (task.workspace.kind === "local") return prepareLocalWorkspace(baseDir, task.workspace);
2275
4010
  if (task.workspace.kind === "git") return prepareGitWorkspace(task.workspace);
2276
- return prepareTempCopyWorkspace(baseDir, task.workspace);
4011
+ return prepareTempCopyWorkspace(baseDir, task.workspace, tempCopy);
2277
4012
  }
2278
4013
  async function runTaskRunner(task, target, cwd, clioEntry, env) {
2279
4014
  if (task.runner.kind === "external-command") return runExternalCommandRunner(task.runner, cwd, task.timeoutMs, env);
2280
4015
  if (task.runner.kind === "context-index") return runContextIndexRunner(cwd, clioEntry, task.timeoutMs, target, env);
2281
4016
  if (task.runner.kind === "context-init")
2282
4017
  return runContextInitRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
2283
- return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
4018
+ return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env, task.metrics.readObservation);
2284
4019
  }
2285
- async function measureTaskOutcome(task, cwd) {
4020
+ async function measureTaskOutcome(task, cwd, env) {
2286
4021
  const commands = task.verify.measure ?? [];
2287
- if (commands.length === 0) return {};
2288
- const result = await runCommandVerifiers(commands, cwd, task.timeoutMs);
2289
- return { "task.exitCode": result.exitCode, "task.solved": result.exitCode === 0 };
4022
+ if (commands.length === 0) return { metrics: {} };
4023
+ const result = await runCommandVerifiers(commands, cwd, task.timeoutMs, env);
4024
+ const behavioral = graderBehaviorMeasurement(result.stdout);
4025
+ return {
4026
+ metrics: {
4027
+ "task.exitCode": result.exitCode,
4028
+ "task.solved": result.exitCode === 0,
4029
+ ...behavioral.metrics
4030
+ },
4031
+ ...behavioral.executionObservation === void 0 ? {} : { executionObservation: behavioral.executionObservation }
4032
+ };
4033
+ }
4034
+ function graderBehaviorMeasurement(stdout) {
4035
+ const metrics = {};
4036
+ let executionObservation;
4037
+ for (const line of stdout.split(/\r?\n/u)) {
4038
+ if (line.trim().length === 0) continue;
4039
+ let value;
4040
+ try {
4041
+ value = JSON.parse(line);
4042
+ } catch {
4043
+ continue;
4044
+ }
4045
+ if (!isRecord10(value)) continue;
4046
+ if (value.schema === "clio.eval.measure.v1" && isRecord10(value.metrics)) {
4047
+ for (const [key, metric] of Object.entries(value.metrics)) {
4048
+ if (key !== "claims.unsupported" && key !== "completion.reported") continue;
4049
+ if (typeof metric === "boolean" || typeof metric === "number" && Number.isFinite(metric)) metrics[key] = metric;
4050
+ }
4051
+ }
4052
+ if (value.schema === "clio.eval.execution-observation.v1") {
4053
+ executionObservation = parseExecutionObservation(value);
4054
+ }
4055
+ }
4056
+ return { metrics, ...executionObservation === void 0 ? {} : { executionObservation } };
4057
+ }
4058
+ function parseExecutionObservation(value) {
4059
+ const policies = isRecord10(value.policyHashes) ? value.policyHashes : {};
4060
+ const project = isRecord10(value.projectContext) ? value.projectContext : null;
4061
+ return {
4062
+ compositionHash: nullableDigest2(value.compositionHash),
4063
+ target: nullableString2(value.target),
4064
+ wireModel: nullableString2(value.wireModel),
4065
+ runtime: nullableString2(value.runtime),
4066
+ thinkingLevel: nullableString2(value.thinkingLevel),
4067
+ toolSignature: nullableDigest2(value.toolSignature),
4068
+ autonomy: nullableString2(value.autonomy),
4069
+ policyHashes: { rulePack: nullableDigest2(policies.rulePack), project: nullableDigest2(policies.project) },
4070
+ projectContext: project === null ? null : {
4071
+ tier: nullableString2(project.tier),
4072
+ contentHash: nullableDigest2(project.contentHash),
4073
+ chars: nullableNonNegativeInteger(project.chars),
4074
+ sections: stringArray(project.sections),
4075
+ rulesApplied: stringArray(project.rulesApplied),
4076
+ operatorProfileApplied: typeof project.operatorProfileApplied === "boolean" ? project.operatorProfileApplied : null
4077
+ }
4078
+ };
2290
4079
  }
2291
4080
  async function runVerifiers(task, cwd, metrics) {
2292
4081
  const commandResult = await runCommandVerifiers(task.verify.commands ?? [], cwd, task.timeoutMs);
@@ -2322,7 +4111,7 @@ async function runVerifiers(task, cwd, metrics) {
2322
4111
  }
2323
4112
  return { pass: true, exitCode: 0, failureClass: null, stdout: commandResult.stdout, stderr: commandResult.stderr };
2324
4113
  }
2325
- function buildArtifact(loaded, evalId, results, clioEntry) {
4114
+ function buildArtifact(loaded, evalId, results, clioEntry, servingConfiguration) {
2326
4115
  const passed = results.filter((result) => result.pass).length;
2327
4116
  return {
2328
4117
  version: 4,
@@ -2330,7 +4119,11 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
2330
4119
  suite: { id: loaded.suite.suite.id, hash: loaded.hash },
2331
4120
  clio: evalClioProvenance({ entry: clioEntry }),
2332
4121
  environment: evalEnvironmentProvenance(),
2333
- matrix: artifactMatrixIdentity(loaded.suite.matrix.targets),
4122
+ matrix: {
4123
+ ...artifactMatrixIdentity(loaded.suite.matrix.targets),
4124
+ ...loaded.suite.matrix.dimensions === void 0 ? {} : { dimensions: loaded.suite.matrix.dimensions }
4125
+ },
4126
+ servingConfiguration,
2334
4127
  summary: {
2335
4128
  runs: results.length,
2336
4129
  passed,
@@ -2339,26 +4132,50 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
2339
4132
  tokens: tokenAccountingFrom(results),
2340
4133
  wallTimeMs: results.reduce((sum2, result) => sum2 + wallTimeMetric(result.metrics), 0)
2341
4134
  },
4135
+ aggregates: aggregateEvalVerdicts(
4136
+ results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict])
4137
+ ),
2342
4138
  results
2343
4139
  };
2344
4140
  }
2345
4141
  function assertionMessage(assertion, actual) {
2346
4142
  return `assertion failed: ${assertion.metric} ${assertion.op} ${String(assertion.value)} (actual ${JSON.stringify(actual)})`;
2347
4143
  }
4144
+ function isRecord10(value) {
4145
+ return typeof value === "object" && value !== null && !Array.isArray(value);
4146
+ }
4147
+ function nullableString2(value) {
4148
+ return typeof value === "string" && value.length > 0 ? value : null;
4149
+ }
4150
+ function nullableDigest2(value) {
4151
+ return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
4152
+ }
4153
+ function nullableNonNegativeInteger(value) {
4154
+ return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : null;
4155
+ }
4156
+ function stringArray(value) {
4157
+ return Array.isArray(value) ? value.filter((entry) => typeof entry === "string") : [];
4158
+ }
2348
4159
 
2349
4160
  // src/cli/eval.ts
2350
4161
  var HELP = `clio-coder eval <command>
2351
4162
 
2352
4163
  Commands:
2353
4164
  clio-coder eval validate --suite <suite.yaml>
2354
- clio-coder eval run --suite <suite.yaml> [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
2355
- clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
4165
+ clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
4166
+ clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
2356
4167
  clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
2357
- clio-coder eval compare <baselineEvalId> <candidateEvalId>
4168
+ clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
2358
4169
  clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
2359
4170
  `;
2360
4171
  function parseEvalArgs(args) {
2361
- const parsed = { repeat: 1, compareIds: [], format: "text", help: false };
4172
+ const parsed = {
4173
+ repeat: 1,
4174
+ compareIds: [],
4175
+ format: "text",
4176
+ allowConfigDrift: false,
4177
+ help: false
4178
+ };
2362
4179
  for (let index = 0; index < args.length; index += 1) {
2363
4180
  const arg = args[index];
2364
4181
  if (arg === void 0) continue;
@@ -2417,6 +4234,11 @@ function parseEvalArgs(args) {
2417
4234
  index += 1;
2418
4235
  continue;
2419
4236
  }
4237
+ if (arg === "--trials") {
4238
+ parsed.trials = positiveInteger(requiredValue(args, index, "--trials"), "--trials");
4239
+ index += 1;
4240
+ continue;
4241
+ }
2420
4242
  throw new Error(`unknown eval run argument: ${arg}`);
2421
4243
  }
2422
4244
  if (parsed.command === "report") {
@@ -2432,6 +4254,20 @@ function parseEvalArgs(args) {
2432
4254
  throw new Error(`unexpected eval report argument: ${arg}`);
2433
4255
  }
2434
4256
  if (parsed.command === "compare") {
4257
+ if (arg === "--format") {
4258
+ parsed.format = comparisonFormat(requiredValue(args, index, "--format"));
4259
+ index += 1;
4260
+ continue;
4261
+ }
4262
+ if (arg === "--metric") {
4263
+ parsed.metric = requiredValue(args, index, "--metric");
4264
+ index += 1;
4265
+ continue;
4266
+ }
4267
+ if (arg === "--allow-config-drift") {
4268
+ parsed.allowConfigDrift = true;
4269
+ continue;
4270
+ }
2435
4271
  if (!arg.startsWith("-")) {
2436
4272
  parsed.compareIds.push(arg);
2437
4273
  continue;
@@ -2504,13 +4340,19 @@ async function runEvalValidate(parsed) {
2504
4340
  }
2505
4341
  async function runEvalRun(parsed) {
2506
4342
  try {
2507
- const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.repeat);
4343
+ const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.trials ?? parsed.repeat);
2508
4344
  const resolveOptions = {};
2509
4345
  if (parsed.target !== void 0) resolveOptions.target = parsed.target;
2510
4346
  if (parsed.model !== void 0) resolveOptions.model = parsed.model;
2511
- const suite = resolveSuiteForRun(loaded.suite, resolveOptions);
2512
- const clioEntry = resolve7(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
2513
- const artifact = await runEvalSuiteV2({ ...loaded, suite }, { clioEntry });
4347
+ const suite = resolveSuiteForRun(loaded.suite, {
4348
+ ...resolveOptions,
4349
+ ...parsed.trials ? { trials: parsed.trials } : {}
4350
+ });
4351
+ const clioEntry = resolve8(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
4352
+ const artifact = await runEvalSuiteV2(
4353
+ { ...loaded, suite },
4354
+ { clioEntry, freshWorkspaces: parsed.trials !== void 0 }
4355
+ );
2514
4356
  const artifactPath = await writeEvalArtifactV4(clioDataDir(), artifact, parsed.out);
2515
4357
  process.stdout.write(`${renderEvalTextReportV4(artifact)}artifact: ${artifactPath}
2516
4358
  `);
@@ -2520,6 +4362,11 @@ async function runEvalRun(parsed) {
2520
4362
  `);
2521
4363
  for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
2522
4364
  }
4365
+ if (gate !== null && gate.informational.length > 0) {
4366
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4367
+ `);
4368
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4369
+ }
2523
4370
  return artifact.summary.failed === 0 && (gate === null || gate.pass) ? 0 : 1;
2524
4371
  } catch (error) {
2525
4372
  return handleEvalLoadError(error, 1);
@@ -2543,8 +4390,12 @@ async function runEvalCompareCommand(parsed) {
2543
4390
  const dataDir = clioDataDir();
2544
4391
  const baseline = await loadEvalArtifactV4(dataDir, baselineEvalId);
2545
4392
  const candidate = await loadEvalArtifactV4(dataDir, candidateEvalId);
2546
- process.stdout.write(renderEvalComparisonV4(compareEvalArtifactsV4(baseline, candidate)));
2547
- return 0;
4393
+ const summary = compareEvalArtifactsV4(baseline, candidate, {
4394
+ allowConfigDrift: parsed.allowConfigDrift,
4395
+ ...parsed.metric === void 0 ? {} : { metric: parsed.metric }
4396
+ });
4397
+ process.stdout.write(renderEvalComparisonReportV1(summary, parsed.format));
4398
+ return summary.hardGate.pass ? 0 : 1;
2548
4399
  } catch (error) {
2549
4400
  printError(error instanceof Error ? error.message : String(error));
2550
4401
  return error instanceof InvalidIdError ? 2 : 1;
@@ -2554,27 +4405,46 @@ async function runEvalGateCommand(parsed) {
2554
4405
  try {
2555
4406
  const dataDir = clioDataDir();
2556
4407
  const candidate = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
2557
- await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
2558
- const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }] } : loadThresholds(parsed.thresholds);
4408
+ const baseline = await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
4409
+ const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }], informational: [] } : loadThresholds(parsed.thresholds);
2559
4410
  const gate = evaluateGate(candidate, thresholds);
2560
- if (gate.pass) {
4411
+ const comparison = compareEvalArtifactsV4(baseline, candidate);
4412
+ if (gate.informational.length > 0) {
4413
+ process.stdout.write(`informational budgets: ${gate.informational.length} notice
4414
+ `);
4415
+ for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
4416
+ }
4417
+ if (gate.pass && comparison.hardGate.pass) {
2561
4418
  process.stdout.write("gate: pass\n");
2562
4419
  return 0;
2563
4420
  }
2564
- process.stdout.write(`gate: fail (${gate.failures.length} threshold failure)
4421
+ const failureCount = gate.failures.length + comparison.hardGate.failures.length + comparison.hardGate.envelopeFailures.length;
4422
+ process.stdout.write(`gate: fail (${failureCount} hard failure)
2565
4423
  `);
2566
4424
  for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
4425
+ for (const failure of comparison.hardGate.failures) {
4426
+ process.stdout.write(
4427
+ ` ${failure.metric} [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: ${failure.change} (hard behavioral gate)
4428
+ `
4429
+ );
4430
+ }
4431
+ for (const failure of comparison.hardGate.envelopeFailures) {
4432
+ process.stdout.write(
4433
+ ` execution envelope [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: incomparable fields ${failure.fields.join(", ")}
4434
+ `
4435
+ );
4436
+ }
2567
4437
  return 1;
2568
4438
  } catch (error) {
2569
4439
  printError(error instanceof Error ? error.message : String(error));
2570
4440
  return error instanceof InvalidIdError ? 2 : 1;
2571
4441
  }
2572
4442
  }
2573
- function renderArtifactReport(artifact, format, _dataDir) {
2574
- if (format === "json") return renderEvalJsonReportV4(artifact);
2575
- if (format === "md") return renderEvalMarkdownReportV4(artifact);
2576
- if (format === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
2577
- if (format === "junit") return renderEvalJunitReportV4(artifact);
4443
+ function renderArtifactReport(artifact, format2, _dataDir) {
4444
+ if (format2 === "json") return renderEvalJsonReportV4(artifact);
4445
+ if (format2 === "md") return renderEvalMarkdownReportV4(artifact);
4446
+ if (format2 === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
4447
+ if (format2 === "junit") return renderEvalJunitReportV4(artifact);
2578
4448
  return renderEvalTextReportV4(artifact);
2579
4449
  }
2580
4450
  function handleEvalLoadError(error, fallback = 2) {
@@ -2603,7 +4473,11 @@ function reportFormat(value) {
2603
4473
  if (value === "text" || value === "json" || value === "md" || value === "swe-jsonl" || value === "junit") return value;
2604
4474
  throw new Error("--format must be text, json, md, swe-jsonl, or junit");
2605
4475
  }
4476
+ function comparisonFormat(value) {
4477
+ if (value === "text" || value === "json" || value === "md" || value === "junit") return value;
4478
+ throw new Error("eval compare --format must be text, json, md, or junit");
4479
+ }
2606
4480
  export {
2607
4481
  runEvalCommand
2608
4482
  };
2609
- //# sourceMappingURL=eval-BEC2WHDA.js.map
4483
+ //# sourceMappingURL=eval-IJ5VEZDJ.js.map