@iowarp/clio-coder 0.3.7 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +13 -4
  3. package/dist/{acp-SK4MD6MM.js → acp-7LOELQFP.js} +13 -13
  4. package/dist/{agents-2FN2K6ME.js → agents-FIBG2SHA.js} +41 -37
  5. package/dist/assets/codewiki.json +1 -1
  6. package/dist/{auth-QIYZWM5I.js → auth-OI4LIH2I.js} +31 -24
  7. package/dist/builtins-AD25UL3C.js +17 -0
  8. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  9. package/dist/chunk-3DPEIQKN.js +113 -0
  10. package/dist/{chunk-EOOQZZDE.js → chunk-3DUR4WUA.js} +19 -19
  11. package/dist/{chunk-WHJYKASB.js → chunk-3MRC2YSQ.js} +2 -2
  12. package/dist/{chunk-EBEFWSGL.js → chunk-3UUY7R3Z.js} +14 -10
  13. package/dist/{chunk-LADCF22A.js → chunk-3V5AYSEQ.js} +113 -54
  14. package/dist/{chunk-BMWK7ZIZ.js → chunk-465CC7FK.js} +16 -13
  15. package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
  16. package/dist/{chunk-CEYBNUGC.js → chunk-4H6ULJ3H.js} +378 -36
  17. package/dist/{chunk-YTYFXUI3.js → chunk-4LJX2PUC.js} +9 -9
  18. package/dist/{chunk-DOOEX22V.js → chunk-56KB5IJP.js} +5 -5
  19. package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
  20. package/dist/{chunk-TSHXZTOQ.js → chunk-5HFBWUMU.js} +23 -11
  21. package/dist/{chunk-5UJ6ECTS.js → chunk-5PVQ4SRS.js} +80 -8
  22. package/dist/{chunk-ZWLZP4ZT.js → chunk-5QKCQQ3E.js} +359 -17
  23. package/dist/{chunk-6M7VS3J3.js → chunk-5T7RBWN2.js} +111 -5
  24. package/dist/chunk-774ILSRL.js +172 -0
  25. package/dist/chunk-7C6RYZGQ.js +391 -0
  26. package/dist/{chunk-GH5622CP.js → chunk-A2NJGIB3.js} +2 -2
  27. package/dist/{chunk-C4JBQ5SR.js → chunk-AD7Y7STJ.js} +6 -6
  28. package/dist/{chunk-GEYXPTRF.js → chunk-AEYBF3TB.js} +33 -12
  29. package/dist/{chunk-2SFS6XQE.js → chunk-AMKHQW3C.js} +3 -2
  30. package/dist/{chunk-D4MDIG46.js → chunk-B5CSFE7B.js} +7 -7
  31. package/dist/{chunk-MXI6J5JF.js → chunk-B5XRQOLB.js} +10 -10
  32. package/dist/{chunk-X2KV5FXT.js → chunk-BVDVID7E.js} +2 -2
  33. package/dist/{chunk-JNXPYBB4.js → chunk-CA42X6KT.js} +3 -3
  34. package/dist/{chunk-VREKEFLL.js → chunk-D73KXYPF.js} +3 -3
  35. package/dist/{chunk-JTSEDYVQ.js → chunk-DG4M6ZUE.js} +7 -7
  36. package/dist/{chunk-DQA7QLMD.js → chunk-EBOC7MT3.js} +10 -25
  37. package/dist/{chunk-KZ2H5X4G.js → chunk-ECUO3KDP.js} +129 -14
  38. package/dist/{chunk-JRIO5UD2.js → chunk-EQ63NRB7.js} +5 -5
  39. package/dist/{chunk-YD734TPH.js → chunk-FALJGAWU.js} +2 -2
  40. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  41. package/dist/{chunk-XEGB6BCN.js → chunk-GAYUJ7LE.js} +68 -14
  42. package/dist/{chunk-UND3GU2L.js → chunk-H7IXIC72.js} +2 -2
  43. package/dist/{chunk-IR4CFBFN.js → chunk-HAY4ZE2P.js} +12 -12
  44. package/dist/{chunk-UVDSQ6LW.js → chunk-HCBCAYZU.js} +74 -147
  45. package/dist/{chunk-4DWFMQDR.js → chunk-HJB5IUKP.js} +89 -145
  46. package/dist/{chunk-M4AKACEO.js → chunk-HKO36JWF.js} +33 -5
  47. package/dist/{chunk-KCMKRQX4.js → chunk-HPCTNZM2.js} +45 -82
  48. package/dist/{chunk-465YSENW.js → chunk-IFBNV6H6.js} +3 -3
  49. package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
  50. package/dist/chunk-JEQQR47K.js +3025 -0
  51. package/dist/{chunk-FO5ZOVUY.js → chunk-KV2AOLDF.js} +27 -7
  52. package/dist/chunk-LU7P4LHA.js +33 -0
  53. package/dist/{chunk-6TUKSZVF.js → chunk-LXPJXFM5.js} +11 -11
  54. package/dist/{chunk-VQNODYQ4.js → chunk-MIX5N5AC.js} +488 -3668
  55. package/dist/chunk-MLOK6ZOS.js +2888 -0
  56. package/dist/{chunk-ZZMN5OM4.js → chunk-MV2VUEJC.js} +2 -2
  57. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  58. package/dist/{chunk-OBMAI2DP.js → chunk-N3PBVRTZ.js} +12 -388
  59. package/dist/{chunk-WJHBC77E.js → chunk-N5XKWMDW.js} +17 -7
  60. package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
  61. package/dist/{chunk-UFQ3F4FW.js → chunk-NQ6UCCOD.js} +4 -4
  62. package/dist/chunk-NUGM5KR6.js +165 -0
  63. package/dist/{chunk-DMD2AGVS.js → chunk-NZU6YDNV.js} +20 -18
  64. package/dist/{chunk-WHGPSPT5.js → chunk-O6I4CIEU.js} +151 -13
  65. package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
  66. package/dist/{chunk-PD3MESLB.js → chunk-P3JGPQFL.js} +4 -4
  67. package/dist/{chunk-UHXRNZ2J.js → chunk-PNY46YEY.js} +23 -6
  68. package/dist/{chunk-THKY7CD7.js → chunk-PZ4I4JE2.js} +134 -29
  69. package/dist/{chunk-SROCI7ZU.js → chunk-QQ7EKM72.js} +5 -5
  70. package/dist/{chunk-QCTRSGHQ.js → chunk-R7LNVMCS.js} +91 -53
  71. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  72. package/dist/chunk-RKKLTLYB.js +45 -0
  73. package/dist/{chunk-OB5HIGJY.js → chunk-RKRLDWD3.js} +4 -1
  74. package/dist/{chunk-DJNLUABN.js → chunk-S4COXYBG.js} +588 -32
  75. package/dist/{chunk-3HAPLH5M.js → chunk-T3Z6VAAF.js} +172 -11
  76. package/dist/{chunk-FOT2FX5J.js → chunk-TD7UE2L5.js} +12 -10
  77. package/dist/{chunk-UUANF5CR.js → chunk-TEO2TLVN.js} +856 -967
  78. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  79. package/dist/{chunk-EELBMBT6.js → chunk-VKBMFOYV.js} +74 -15
  80. package/dist/chunk-VO2LKSTM.js +165 -0
  81. package/dist/{chunk-5C77SEEY.js → chunk-VPTUJU4P.js} +3 -3
  82. package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
  83. package/dist/{chunk-J7PIKKWC.js → chunk-WXCJ7VME.js} +8 -8
  84. package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
  85. package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
  86. package/dist/{chunk-AB4XIIVB.js → chunk-YKOFT37S.js} +6 -6
  87. package/dist/chunk-YSEHGPCT.js +127 -0
  88. package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
  89. package/dist/cli/index.js +32 -32
  90. package/dist/{clio-WBVQEBKO.js → clio-LT5V7SSZ.js} +9 -9
  91. package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +5 -5
  92. package/dist/codewiki/build-worker.js +4 -4
  93. package/dist/{components-F7OEATSO.js → components-ZFA3SAER.js} +8 -8
  94. package/dist/{config-TRBL3RCF.js → config-RXS5T3JT.js} +98 -65
  95. package/dist/{configure-OLCVPHNM.js → configure-2WYWSCSD.js} +26 -22
  96. package/dist/{context-MJIJ6GOX.js → context-I3BTOTCS.js} +12 -12
  97. package/dist/{context-XEWE3MOJ.js → context-MVOORGMF.js} +54 -47
  98. package/dist/{context-WFPKQSM6.js → context-PALKKQYL.js} +28 -28
  99. package/dist/{context-clear-KNOS2JPB.js → context-clear-N2WOYZ2K.js} +53 -46
  100. package/dist/{context-index-SSR5ECNE.js → context-index-HNG3MOME.js} +6 -6
  101. package/dist/{context-working-set-EUXAZI6N.js → context-working-set-MIEVECVZ.js} +17 -18
  102. package/dist/{dispatch-runner-B7MTOVKL.js → dispatch-runner-VVA4SRRH.js} +90 -61
  103. package/dist/{docs-FLJTIDSE.js → docs-7LQ23DLM.js} +8 -8
  104. package/dist/doctor-TWBWFK5V.js +165 -0
  105. package/dist/eval-IJ5VEZDJ.js +4483 -0
  106. package/dist/{evidence-JZNBUOQZ.js → evidence-L5APPXNV.js} +68 -61
  107. package/dist/{evolve-FJVC4KKI.js → evolve-RGNKFJ52.js} +47 -40
  108. package/dist/{extensions-IQL36S7K.js → extensions-7WYWUX5A.js} +13 -7
  109. package/dist/{fleet-BDKYJFCP.js → fleet-6CNVBZZP.js} +113 -76
  110. package/dist/{fleet-commands-ZFIWZSB3.js → fleet-commands-L2SXSYEI.js} +10 -10
  111. package/dist/{fleet-graph-Y6HPXIVF.js → fleet-graph-2J3OOIPO.js} +17 -15
  112. package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-CZRJ4JP5.js} +5 -6
  113. package/dist/{fleet-validate-BIYREGIK.js → fleet-validate-C5RI6DP7.js} +20 -19
  114. package/dist/{init-LQUB5COQ.js → init-VBN2ACVA.js} +70 -63
  115. package/dist/{library-NJAHIGG4.js → library-JHGUMLY2.js} +22 -20
  116. package/dist/{memory-OG6HOYKM.js → memory-K4OQIYWG.js} +49 -42
  117. package/dist/{models-5ZG5XY7J.js → models-2NCZUWDD.js} +35 -29
  118. package/dist/{monitor-TJ7AMTGB.js → monitor-MMVTJABD.js} +64 -45
  119. package/dist/{orchestrator-WZYB54DM.js → orchestrator-ZKBPCHW6.js} +1971 -520
  120. package/dist/{paths-XUC7GS6E.js → paths-DBXMZMDU.js} +5 -5
  121. package/dist/registry-LG64LTF4.js +11 -0
  122. package/dist/{reset-PXQT45IY.js → reset-DD5JGOY3.js} +11 -11
  123. package/dist/{run-FQ74YF62.js → run-QEGNX7FL.js} +89 -83
  124. package/dist/{share-FW7SVCL3.js → share-JKD3BQMW.js} +20 -18
  125. package/dist/{skills-7E7IRB3R.js → skills-LMQIKDOZ.js} +23 -21
  126. package/dist/{skills-eval-LI75W6OK.js → skills-eval-I7X2774U.js} +59 -52
  127. package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
  128. package/dist/support-I7LOJLIF.js +38 -0
  129. package/dist/{targets-4CIFKCTW.js → targets-RUSR6B5Z.js} +77 -42
  130. package/dist/{terminal-lease-WUZY7ZV5.js → terminal-lease-QYVORFR4.js} +6 -4
  131. package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
  132. package/dist/{uninstall-7FV7IP4E.js → uninstall-ZJF5H5ZN.js} +8 -8
  133. package/dist/{upgrade-K2HVIVMQ.js → upgrade-XANW3FXB.js} +29 -26
  134. package/dist/{usage-GTZELZQX.js → usage-4H7ZRXQT.js} +110 -61
  135. package/dist/{verifiers-RLAHT27O.js → verifiers-UZXNBZEB.js} +13 -13
  136. package/dist/{verify-BX3BRKH5.js → verify-BVKWTNDL.js} +9 -9
  137. package/dist/{wiki-generate-ASIFASCN.js → wiki-generate-MY7WV2QI.js} +76 -69
  138. package/dist/worker/entry.js +69 -66
  139. package/docs/alcf-provider.md +1 -1
  140. package/docs/architecture.md +1 -1
  141. package/docs/artifact-versions.md +11 -5
  142. package/docs/built-in-agents.md +1 -1
  143. package/docs/capacity-and-scheduling.md +23 -2
  144. package/docs/commands-and-modes.md +2 -2
  145. package/docs/configuration-and-targets.md +37 -5
  146. package/docs/context-engine.md +63 -4
  147. package/docs/documentation-coverage.md +3 -3
  148. package/docs/documentation-guide.md +1 -1
  149. package/docs/environment-variables.md +2 -0
  150. package/docs/eval-runner.md +262 -11
  151. package/docs/evals-internal.md +72 -2
  152. package/docs/evidence-and-memory.md +77 -12
  153. package/docs/evolution.md +1 -1
  154. package/docs/extensions-and-sharing.md +3 -1
  155. package/docs/fleet-dispatch.md +34 -9
  156. package/docs/glossary.md +21 -1
  157. package/docs/installation-and-lifecycle.md +1 -1
  158. package/docs/middleware-and-components.md +1 -1
  159. package/docs/model-catalog.md +1 -1
  160. package/docs/observability.md +54 -3
  161. package/docs/proactive-memory.md +127 -14
  162. package/docs/prompt-envelope-and-tools.md +22 -2
  163. package/docs/provider-adapter-cookbook.md +1 -1
  164. package/docs/release-cut-checklist.md +60 -41
  165. package/docs/safety-model.md +1 -1
  166. package/docs/scientific-validation.md +1 -1
  167. package/docs/skills-marketplace.md +1 -1
  168. package/docs/tool-usage.md +1 -1
  169. package/docs/trace-store.md +1 -1
  170. package/docs/troubleshooting.md +87 -0
  171. package/docs/tui-design.md +1 -1
  172. package/docs/worker-dispatch-mechanics.md +1 -1
  173. package/package.json +2 -2
  174. package/src/cli/agents.ts +1 -1
  175. package/src/cli/argv.ts +5 -0
  176. package/src/cli/config-inspect.ts +33 -6
  177. package/src/cli/config.ts +1 -1
  178. package/src/cli/configure.ts +107 -23
  179. package/src/cli/doctor-state-size.ts +82 -0
  180. package/src/cli/doctor.ts +7 -1
  181. package/src/cli/eval.ts +80 -16
  182. package/src/cli/evidence.ts +30 -25
  183. package/src/cli/extensions.ts +5 -1
  184. package/src/cli/fleet-preflight.ts +2 -12
  185. package/src/cli/fleet.ts +32 -3
  186. package/src/cli/shared.ts +1 -0
  187. package/src/cli/targets.ts +45 -11
  188. package/src/cli/trace.ts +63 -4
  189. package/src/cli/usage.ts +63 -14
  190. package/src/cli/validate-model.ts +60 -5
  191. package/src/core/bus-events.ts +54 -1
  192. package/src/core/cache-telemetry.ts +42 -0
  193. package/src/core/commit-attribution.ts +4 -4
  194. package/src/core/config.ts +18 -0
  195. package/src/core/defaults.ts +36 -6
  196. package/src/core/endpoint-key.ts +27 -0
  197. package/src/core/path-boundary.ts +100 -0
  198. package/src/core/residency-target-key.ts +25 -0
  199. package/src/core/response-schema.ts +36 -2
  200. package/src/domains/agents/extension.ts +2 -11
  201. package/src/domains/agents/fleet-contract.ts +30 -12
  202. package/src/domains/agents/recipe.ts +7 -1
  203. package/src/domains/agents/registry.ts +73 -5
  204. package/src/domains/agents/result-contract.ts +128 -17
  205. package/src/domains/agents/write-boundary.ts +15 -50
  206. package/src/domains/config/classify.ts +3 -0
  207. package/src/domains/context/codewiki/coordinator.ts +12 -4
  208. package/src/domains/context/project-rules.ts +51 -1
  209. package/src/domains/dispatch/admission.ts +40 -3
  210. package/src/domains/dispatch/assignment-reconcile.ts +22 -5
  211. package/src/domains/dispatch/assignment-store.ts +151 -14
  212. package/src/domains/dispatch/capacity-lease.ts +98 -9
  213. package/src/domains/dispatch/contract.ts +26 -1
  214. package/src/domains/dispatch/delegation-plan.ts +2 -5
  215. package/src/domains/dispatch/execution-plan.ts +44 -4
  216. package/src/domains/dispatch/execution-role.ts +9 -1
  217. package/src/domains/dispatch/extension.ts +309 -96
  218. package/src/domains/dispatch/fleet-run.ts +78 -4
  219. package/src/domains/dispatch/gate-role-prompts.ts +38 -0
  220. package/src/domains/dispatch/heartbeat.ts +32 -8
  221. package/src/domains/dispatch/index.ts +6 -1
  222. package/src/domains/dispatch/intent-requirements.ts +40 -0
  223. package/src/domains/dispatch/intent.ts +84 -8
  224. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  225. package/src/domains/dispatch/path-scope.ts +370 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +2 -1
  227. package/src/domains/dispatch/reservation-store.ts +116 -8
  228. package/src/domains/dispatch/state.ts +4 -0
  229. package/src/domains/dispatch/types.ts +14 -7
  230. package/src/domains/dispatch/validation.ts +6 -3
  231. package/src/domains/dispatch/worker-spawn.ts +25 -11
  232. package/src/domains/dispatch/write-boundary-enforcer.ts +62 -0
  233. package/src/domains/dispatch/write-boundary.ts +262 -22
  234. package/src/domains/eval/artifacts/store.ts +62 -0
  235. package/src/domains/eval/compare/behavioral.ts +224 -0
  236. package/src/domains/eval/compare/compare.ts +355 -2
  237. package/src/domains/eval/compare/envelope.ts +128 -0
  238. package/src/domains/eval/compare/gates.ts +24 -6
  239. package/src/domains/eval/compare/thresholds.ts +30 -3
  240. package/src/domains/eval/execution-provenance.ts +240 -0
  241. package/src/domains/eval/metrics/aggregate.ts +136 -0
  242. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  243. package/src/domains/eval/metrics/evidence.ts +79 -2
  244. package/src/domains/eval/metrics/tracked.ts +413 -0
  245. package/src/domains/eval/provenance.ts +117 -0
  246. package/src/domains/eval/reports/comparison.ts +128 -0
  247. package/src/domains/eval/reports/junit.ts +17 -3
  248. package/src/domains/eval/reports/markdown.ts +3 -3
  249. package/src/domains/eval/reports/text.ts +14 -0
  250. package/src/domains/eval/run-compare.ts +20 -0
  251. package/src/domains/eval/runners/clio-run.ts +139 -2
  252. package/src/domains/eval/runners/external-command.ts +28 -3
  253. package/src/domains/eval/schema/adapter.ts +111 -0
  254. package/src/domains/eval/schema/artifact.ts +20 -0
  255. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  256. package/src/domains/eval/schema/behavioral.ts +520 -0
  257. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  258. package/src/domains/eval/schema/serving.ts +74 -0
  259. package/src/domains/eval/schema/suite.ts +38 -8
  260. package/src/domains/eval/schema/validate.ts +58 -3
  261. package/src/domains/eval/schema/verdict.ts +237 -0
  262. package/src/domains/eval/suites/resolve.ts +2 -0
  263. package/src/domains/eval/suites/run.ts +264 -33
  264. package/src/domains/eval/verifiers/command.ts +2 -1
  265. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  266. package/src/domains/evidence/build.ts +68 -21
  267. package/src/domains/evidence/eval.ts +2 -12
  268. package/src/domains/evidence/findings-markdown.ts +33 -0
  269. package/src/domains/evidence/index.ts +21 -0
  270. package/src/domains/evidence/provenance.ts +46 -11
  271. package/src/domains/evidence/run-trust.ts +7 -113
  272. package/src/domains/evidence/trust-projection.ts +274 -0
  273. package/src/domains/evidence/trust-status.ts +145 -17
  274. package/src/domains/evidence/types.ts +4 -0
  275. package/src/domains/extensions/compatibility.ts +285 -0
  276. package/src/domains/extensions/discovery.ts +126 -4
  277. package/src/domains/extensions/resources.ts +21 -9
  278. package/src/domains/extensions/state.ts +18 -5
  279. package/src/domains/extensions/types.ts +6 -1
  280. package/src/domains/lifecycle/doctor.ts +209 -2
  281. package/src/domains/memory/index.ts +14 -0
  282. package/src/domains/memory/task-bank-promotion.ts +64 -0
  283. package/src/domains/memory/task-memory-policy.ts +77 -8
  284. package/src/domains/memory/task-memory-spend.ts +131 -0
  285. package/src/domains/memory/task-memory-status.ts +7 -0
  286. package/src/domains/memory/task-memory-telemetry.ts +2 -0
  287. package/src/domains/middleware/index.ts +1 -0
  288. package/src/domains/middleware/memory-intervention.ts +69 -5
  289. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  290. package/src/domains/observability/background-memory-usage.ts +140 -0
  291. package/src/domains/observability/cost.ts +1 -1
  292. package/src/domains/observability/index.ts +7 -0
  293. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  294. package/src/domains/observability/trace-store.ts +192 -2
  295. package/src/domains/prompts/compiler.ts +100 -13
  296. package/src/domains/prompts/contract.ts +3 -5
  297. package/src/domains/providers/endpoint-capacity.ts +96 -0
  298. package/src/domains/providers/extension.ts +30 -2
  299. package/src/domains/providers/index.ts +10 -0
  300. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  301. package/src/domains/providers/runtime-resolution.ts +8 -1
  302. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  303. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  304. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  305. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  306. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  307. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  308. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  309. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  310. package/src/domains/providers/types/capability-flags.ts +2 -0
  311. package/src/domains/providers/types/target-descriptor.ts +2 -0
  312. package/src/domains/resources/common-loader.ts +3 -0
  313. package/src/domains/resources/prompts/loader.ts +184 -17
  314. package/src/domains/safety/call-target.ts +52 -0
  315. package/src/domains/safety/policy-engine.ts +5 -5
  316. package/src/domains/safety/run-effects.ts +96 -2
  317. package/src/domains/safety/scope.ts +7 -12
  318. package/src/domains/session/context-accounting.ts +52 -1
  319. package/src/domains/session/context-ledger.ts +37 -13
  320. package/src/domains/session/index.ts +6 -0
  321. package/src/domains/session/prompt-cache.ts +140 -0
  322. package/src/domains/session/prompt-manifest.ts +42 -0
  323. package/src/engine/acp/adapter.ts +18 -3
  324. package/src/engine/acp/server.ts +4 -1
  325. package/src/engine/ai.ts +35 -0
  326. package/src/engine/apis/llamacpp-residency.ts +55 -3
  327. package/src/engine/apis/lmstudio.ts +25 -5
  328. package/src/engine/apis/ollama-native.ts +2 -1
  329. package/src/engine/apis/openai-completions.ts +80 -17
  330. package/src/engine/apis/residency-lock.ts +3 -1
  331. package/src/engine/apis/residency.ts +34 -1
  332. package/src/engine/prompt-templates.ts +18 -1
  333. package/src/engine/provider-payload.ts +29 -1
  334. package/src/engine/worker-runtime.ts +6 -3
  335. package/src/entry/orchestrator.ts +176 -30
  336. package/src/interactive/chat-loop-messages.ts +26 -7
  337. package/src/interactive/chat-loop.ts +318 -41
  338. package/src/interactive/chat-panel.ts +62 -8
  339. package/src/interactive/clio-editor.ts +45 -8
  340. package/src/interactive/context-activity.ts +5 -1
  341. package/src/interactive/context-meter.ts +1 -1
  342. package/src/interactive/context-overlay.ts +40 -10
  343. package/src/interactive/cost-overlay.ts +64 -6
  344. package/src/interactive/dispatch-board.ts +127 -5
  345. package/src/interactive/fleet-run-preview.ts +41 -15
  346. package/src/interactive/handoff-round.ts +41 -2
  347. package/src/interactive/interactive-application.ts +24 -1
  348. package/src/interactive/interactive-event-projection.ts +14 -0
  349. package/src/interactive/interactive-input-runtime.ts +8 -0
  350. package/src/interactive/interactive-presentation.ts +4 -0
  351. package/src/interactive/interactive-shell.ts +20 -17
  352. package/src/interactive/interactive-slash-runtime.ts +27 -4
  353. package/src/interactive/memory-overlay.ts +8 -0
  354. package/src/interactive/mutation-preview.ts +295 -0
  355. package/src/interactive/overlay-general-openers.ts +16 -0
  356. package/src/interactive/overlay-key-routing.ts +38 -0
  357. package/src/interactive/overlay-lifecycle.ts +38 -5
  358. package/src/interactive/overlay-permission-lifecycle.ts +22 -2
  359. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  360. package/src/interactive/overlays/ask-user.ts +91 -19
  361. package/src/interactive/overlays/help-reference.ts +4 -0
  362. package/src/interactive/overlays/prompts.ts +11 -1
  363. package/src/interactive/overlays/settings.ts +176 -48
  364. package/src/interactive/permission-hint.ts +34 -2
  365. package/src/interactive/permission-overlay.ts +159 -9
  366. package/src/interactive/prewarm.ts +197 -0
  367. package/src/interactive/render-trace.ts +162 -15
  368. package/src/interactive/renderers/tool-execution.ts +4 -0
  369. package/src/interactive/side-question.ts +58 -1
  370. package/src/interactive/slash-commands.ts +7 -2
  371. package/src/interactive/status/controller.ts +11 -0
  372. package/src/interactive/status/state-machine.ts +54 -2
  373. package/src/interactive/status/types.ts +7 -0
  374. package/src/interactive/terminal-lease.ts +2 -0
  375. package/src/interactive/turn-context.ts +299 -31
  376. package/src/interactive/turn-persistence.ts +14 -4
  377. package/src/interactive/turn-prewarm.ts +364 -0
  378. package/src/interactive/turn-queues.ts +7 -4
  379. package/src/interactive/turn-runtime.ts +8 -1
  380. package/src/interactive/turn-state.ts +23 -0
  381. package/src/interactive/view/artifacts.ts +42 -9
  382. package/src/interactive/view/view-overlay.ts +43 -6
  383. package/src/interactive/worker-receipts.ts +14 -2
  384. package/src/interactive/worker-stream.ts +8 -0
  385. package/src/tools/ask-user.ts +43 -2
  386. package/src/tools/dispatch-admission.ts +12 -13
  387. package/src/tools/dispatch-arguments.ts +27 -0
  388. package/src/tools/dispatch-plan.ts +46 -9
  389. package/src/tools/dispatch-runner.ts +48 -13
  390. package/src/tools/dispatch-scout.ts +1 -1
  391. package/src/tools/monitor.ts +13 -0
  392. package/src/tools/registry.ts +16 -0
  393. package/src/tools/worker-evidence.ts +19 -13
  394. package/src/worker/spec-contract.ts +2 -1
  395. package/dist/chunk-AOCYTWAV.js +0 -449
  396. package/dist/chunk-HWUFFB6L.js +0 -83
  397. package/dist/chunk-R346GLFC.js +0 -31
  398. package/dist/chunk-ZGH7FGS5.js +0 -1079
  399. package/dist/doctor-RN4YKO2X.js +0 -87
  400. package/dist/eval-RUBJVSNQ.js +0 -2557
@@ -11,7 +11,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
11
11
  | `src/engine/` (Core) | Engine turn loop, prompt priming, streaming message adapters, turn execution | [architecture.md](architecture.md), [context-engine.md](context-engine.md) | `documented` | Documented across architecture and context engine guides. |
12
12
  | `src/engine/acp/` | ACP protocol server, transport adapters, tool mediators, permission forwarding, error taxonomy | [acp.md](acp.md) | `documented` | Dedicated ACP specification covering server wiring, permission mediation, timeouts, error taxonomy, and security boundaries. |
13
13
  | `src/entry/` | Application bootstrapping, CLI router, interactive loop entry point | [architecture.md](architecture.md), [installation-and-lifecycle.md](installation-and-lifecycle.md) | `documented` | Documented in architecture compilation boundaries and lifecycle guides. |
14
- | `src/interactive/` | TUI architecture, screens, overlays, keybindings, panels, theme tokens, width matrices | [tui-design.md](tui-design.md), [commands-and-modes.md](commands-and-modes.md) | `documented` | Fully documented in TUI design specification and commands reference. |
14
+ | `src/interactive/` | TUI architecture, screens, overlays, keybindings, panels, theme tokens, width matrices, prompt pre-warm rounds and their gating, expected-cold reason stamping | [tui-design.md](tui-design.md), [commands-and-modes.md](commands-and-modes.md), [context-engine.md](context-engine.md) | `documented` | Fully documented in TUI design specification and commands reference; the pre-warm and the cache-honesty surfaces `/context` renders are in the context engine reference. |
15
15
  | `src/tools/` | 20 built-in tools across 7 planes, registry, policy engine bindings, observation envelope bounds | [tool-usage.md](tool-usage.md), [prompt-envelope-and-tools.md](prompt-envelope-and-tools.md) | `documented` | Comprehensive 20-tool reference with schemas, examples, and envelope size constraints. |
16
16
  | `src/utils/` | Image manipulation, photon operations, git execution utilities | [architecture.md](architecture.md), [tool-usage.md](tool-usage.md) | `documented` | Utility helpers documented within tool usage and architectural boundaries. |
17
17
  | `src/worker/` | Worker subprocess lifecycle, NDJSON transport, heartbeat timers, control lane demuxing, spec contracts | [worker-dispatch-mechanics.md](worker-dispatch-mechanics.md) | `documented` | Complete reference for NDJSON socket protocols, watchdog timers, and exit status mapping. |
@@ -20,7 +20,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
20
20
  | `src/domains/config/` | Configuration contracts, file watcher, keybinding definitions, setting classifiers | [configuration-and-targets.md](configuration-and-targets.md), [commands-and-modes.md](commands-and-modes.md) | `documented` | Documented in configuration targets and command/keybinding reference. |
21
21
  | `src/domains/context/` | `CLIO-CODER.md` bootstrap, codewiki generation, prompt context assembly, project rules, non-destructive working-set eviction (`age-horizon` and `structural-v1` policies, protection predicates, path index, byte-stable markers, recall by ref) | [context-engine.md](context-engine.md), [context-working-set.md](context-working-set.md) | `documented` | Context window, token accounting, and the three compaction mechanisms in the engine reference; the working-set layer has its own guide covering the vocabulary, both ledger record kinds and format v4, the marker contract, both policies with their rule order, recall semantics, and the operator surfaces. |
22
22
  | `src/domains/dispatch/` | Fleet orchestration, assignment store, batch tracker, admission, route planner, receipt integrity v16 | [fleet-dispatch.md](fleet-dispatch.md), [dispatch-architecture-rationale.md](dispatch-architecture-rationale.md), [worker-dispatch-mechanics.md](worker-dispatch-mechanics.md) | `documented` | Multi-node fleet dispatch, admission invariants, and receipt verification fully documented. |
23
- | `src/domains/eval/` | Suite v2 YAML schema, eval runner, metrics, reporters, workspace sandboxing | [eval-runner.md](eval-runner.md), [evals-internal.md](evals-internal.md) | `documented` | Product evals are documented independently from external benchmarks. |
23
+ | `src/domains/eval/` | Suite v2 YAML schema, eval runner, `clio.eval.verdict.v1` envelope and its Suite v2 adapter, tracked metrics and scenario aggregates, serving-configuration provenance, reporters, workspace sandboxing | [eval-runner.md](eval-runner.md), [evals-internal.md](evals-internal.md) | `documented` | Product evals are documented independently from external benchmarks. The verdict envelope, `trackedMetrics` and their sources, `--trials`, and the config-drift and estimated-versus-measured refusals are in the runner reference. |
24
24
  | `src/domains/evidence/` | Evidence bundles, findings taxonomy, provenance store, failure attribution | [evidence-and-memory.md](evidence-and-memory.md) | `documented` | Documented in evidence directory structures and memory retrieval guide. |
25
25
  | `src/domains/evolution/` | Falsifiable Change Manifest JSON templates and `clio-coder evolve` self-edit gates | [evolution.md](evolution.md) | `documented` | Documented in evolution manifest reference and mutation validation rules. |
26
26
  | `src/domains/extensions/` | Extension manifest schemas, resource roots, portable share archives | [extensions-and-sharing.md](extensions-and-sharing.md) | `documented` | Documented in extensions and sharing guide. |
@@ -29,7 +29,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
29
29
  | `src/domains/middleware/` | Middleware hooks (`turn_start`, `tool_call`, `tool_result`, `turn_end`), reminders, budgets | [middleware-and-components.md](middleware-and-components.md) | `documented` | Documented in middleware hooks and active component snapshot guide. |
30
30
  | `src/domains/observability/` | Trace store (`node:sqlite` WAL mirror), metrics, cost accounting, evidence index | [trace-store.md](trace-store.md), [observability.md](observability.md) | `documented` | Database schema, rowid cursor queries, and receipt provenance documented. |
31
31
  | `src/domains/prompts/` | Prompt compiler, fragment loaders, static cache stability, memory intervention injection | [prompt-envelope-and-tools.md](prompt-envelope-and-tools.md) | `documented` | Documented in prompt envelope and tool delivery guide. |
32
- | `src/domains/providers/` | Runtime adapters, capability probes, model catalog, thinking control, ALCF OAuth | [configuration-and-targets.md](configuration-and-targets.md), [model-catalog.md](model-catalog.md), [provider-adapter-cookbook.md](provider-adapter-cookbook.md), [alcf-provider.md](alcf-provider.md) | `documented` | Complete provider adapter contracts, model catalogs, and ALCF Globus targets documented. |
32
+ | `src/domains/providers/` | Runtime adapters, capability probes, model catalog and its per-family `measuredUnder` provenance, canonical endpoint keys and per-endpoint request-slot capacity, thinking control, ALCF OAuth | [configuration-and-targets.md](configuration-and-targets.md), [model-catalog.md](model-catalog.md), [provider-adapter-cookbook.md](provider-adapter-cookbook.md), [alcf-provider.md](alcf-provider.md) | `documented` | Complete provider adapter contracts, model catalogs, and ALCF Globus targets documented. |
33
33
  | `src/domains/resources/` | Skill package discovery, marketplace index resolution, prompt resources | [skills-marketplace.md](skills-marketplace.md), [extensions-and-sharing.md](extensions-and-sharing.md) | `documented` | Skills marketplace, publishing flows, and resource managers documented. |
34
34
  | `src/domains/safety/` | Policy engine, action classifiers, damage-control rules, path policies, finish contract, audit log | [safety-model.md](safety-model.md), [scientific-validation.md](scientific-validation.md) | `documented` | Policy evaluation order, 10-step sequence, write containment, and finish contract documented. |
35
35
  | `src/domains/scheduling/` | Capacity lease acquisition, heartbeats, expiry, cross-process locks, cluster scheduling | [capacity-and-scheduling.md](capacity-and-scheduling.md), [fleet-dispatch.md](fleet-dispatch.md) | `documented` | Dedicated capacity leasing, heartbeat TTL, and cross-process lock reference. |
@@ -1,7 +1,7 @@
1
1
  # Documentation Standards and Codebase Alignment
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive documentation link linter, phrasing/claim evaluator, and alignment portal is located at [docs/html/documentation_blueprint.html](html/documentation_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive documentation link linter, phrasing/claim evaluator, and alignment portal is located at [docs/html/documentation_blueprint.html](html/documentation_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder is an experimental community alpha. Documentation should help contributors and early users work from the source of truth without overstating maturity. When docs drift, prefer the current source and tests over older prose or aspirational roadmap notes.
7
7
 
@@ -47,6 +47,8 @@ Durable values live in the `guardrails:` section of settings.yaml (see [configur
47
47
  | `CLIO_CODER_REDUCE_MOTION` | off | `1` makes smooth-streaming `auto` use the immediate coalescer. Explicit `on` remains an operator request, while stdout backpressure still pauses frame production. |
48
48
  | `CLIO_CODER_SCREEN_READER` | off | `1` makes smooth-streaming `auto` use the immediate coalescer so a screen reader receives the existing low-motion update behavior. |
49
49
  | `CLIO_CODER_INSTANT_SHELL` | on | `0` disables the single-owner Stage 0 interactive shell for immediate rollback. Unset or `1` mounts one terminal/editor owner before service hydration; ACP, headless, ordinary non-TTY, and subcommand paths never mount it. An explicit `CLIO_CODER_INTERACTIVE=1` keeps its force-interactive non-TTY behavior. |
50
+ | `CLIO_CODER_TRACE_RETENTION_DAYS` | 30 | Maximum age in days for terminal rows in the rebuildable SQLite trace mirror. The value is an integer of at least 1 (`src/domains/observability/trace-store.ts`). |
51
+ | `CLIO_CODER_TRACE_MAX_BYTES` | 134217728 | Maximum allocated size for the SQLite trace mirror before the oldest terminal runs are pruned. The value is an integer of at least 1,048,576 (`src/domains/observability/trace-store.ts`). |
50
52
 
51
53
  ## Directory and install layout
52
54
 
@@ -1,7 +1,7 @@
1
1
  # Clio Coder Local Evaluation Runner
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive task suite validator, subprocess execution simulator, and compare calculator is located at [docs/html/eval_blueprint.html](html/eval_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive task suite validator, subprocess execution simulator, and compare calculator is located at [docs/html/eval_blueprint.html](html/eval_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  The local evaluation runner executes repository-local YAML task suites as deterministic subprocess checks. It is useful for comparing harness changes, prompts, tools, or local workflows.
7
7
 
@@ -15,10 +15,10 @@ The CLI commands under `clio-coder eval` support running, validating, reporting,
15
15
 
16
16
  ```bash
17
17
  clio-coder eval validate --suite <suite.yaml>
18
- clio-coder eval run --suite <suite.yaml> [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
18
+ clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
19
19
  clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
20
20
  clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
21
- clio-coder eval compare <baselineEvalId> <candidateEvalId>
21
+ clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
22
22
  clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
23
23
  ```
24
24
 
@@ -32,7 +32,7 @@ clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds
32
32
  * `swe-jsonl`: Standardized JSONL format representing task runs (e.g. for SWE-bench comparisons).
33
33
  * `junit`: XML report for CI/CD integration.
34
34
  * **`compare`**: Compares two evaluation artifacts (baseline and candidate) by matching tasks.
35
- * **`gate`**: Compares candidate metrics against baseline or absolute thresholds, exiting non-zero if assertions fail (useful for PR gating).
35
+ * **`gate`**: Compares candidate metrics against baseline and absolute thresholds. Correctness and safety regressions fail independently of informational budgets.
36
36
 
37
37
  Exit codes:
38
38
 
@@ -41,8 +41,8 @@ Exit codes:
41
41
  | `eval validate` | `0` when validation passes | `2` for validation issues |
42
42
  | `eval run` | `0` when all task repetitions pass | `1` when any task fails, `2` for invalid configs |
43
43
  | `eval report` | `0` when artifact loads | `1` if artifact cannot be read, `2` for invalid ID |
44
- | `eval compare` | `0` when both artifacts load and compare succeeds | `1` if artifacts cannot be read, `2` for invalid ID |
45
- | `eval gate` | `0` when all threshold assertions pass | `1` if assertions fail, `2` for config/invalid ID errors |
44
+ | `eval compare` | `0` when both artifacts compare and the behavioral hard gate passes | `1` for a hard regression or unreadable artifact, `2` for invalid ID |
45
+ | `eval gate` | `0` when correctness, safety, and hard threshold assertions pass | `1` for any hard failure, `2` for config/invalid ID errors |
46
46
 
47
47
  ---
48
48
 
@@ -103,10 +103,11 @@ tasks:
103
103
  | --- | --- | --- |
104
104
  | `version` | - | Must equal `2`. |
105
105
  | `suite` | `id`, `title`, `visibility`, `description` | Metadata identifying the evaluation suite. |
106
- | `matrix` | `targets[]`, `repeats` | Matrix of execution targets (specifying model and thinking flags) and the repetition count. |
106
+ | `matrix` | `targets[]`, `repeats`, `dimensions[]` | Matrix of execution targets, repetition count, and the execution-envelope fields intentionally varied by the suite. |
107
107
  | `workspace` | `kind`, `path`, `url`, `commit`, `checkout`, `excludes` | Workspace strategy: `local` (run in-place), `git` (clone from URL), or `temp-copy` (isolated copy of a directory). |
108
108
  | `runner` | `kind`, `prompt`, `command`, `commands`, `args`, `timeoutMs` | Runner type: `clio-run` (starts Clio agent loop), `context-index` (runs indexer), `context-init` (initializes context), `external-command` (spawns subprocess). |
109
- | `verify` | `commands`, `assertions`, `forbidPaths` | Validation steps: shell commands, metric assertions (e.g. `op: lt` for max token counts), and files/directories that must not be created or modified (`forbidPaths`). |
109
+ | `behavioral` | `schema`, `corpus`, `execution`, `expectedBehavior`, `forbiddenBehavior`, `judge` | Optional `clio.eval.scenario.v1` behavioral contract. Rules name a closed category and a typed predicate over transcript, tool, receipt, or grader facts. |
110
+ | `verify` | `commands`, `measure`, `assertions`, `forbidPaths` | Validation steps: shell commands, a task-outcome grader, metric assertions (e.g. `op: lt` for max token counts), and files/directories that must not be created or modified (`forbidPaths`). |
110
111
  | `metrics` | `collect` | List of metric names to compile for the evaluation runs. |
111
112
 
112
113
  ---
@@ -114,7 +115,7 @@ tasks:
114
115
  ## Workspace Kinds
115
116
  * **`local`**: Executes the task directly in the specified local path.
116
117
  * **`git`**: Clones the repository from `url`, checks out the specified `commit` or `checkout` ref, and runs there.
117
- * **`temp-copy`**: Copies the directory at `path` to a temporary workspace location before running. This prevents side-effects from polluting other task runs.
118
+ * **`temp-copy`**: Copies the directory at `path` to a temporary workspace location immediately before the matrix item runs and removes it afterward. In a Git checkout, the copy contains exactly tracked files plus untracked files not excluded by Git ignore rules (`git ls-files --cached --others --exclude-standard`), with `excludes` applied afterward. Outside Git it retains the recursive directory copy. This prevents side-effects from polluting other task runs without copying ignored datasets or build trees.
118
119
 
119
120
  ---
120
121
 
@@ -194,12 +195,262 @@ export interface EvalArtifactV4 {
194
195
  matrix: { target: string; model: string | null; thinking: string | null };
195
196
  summary: EvalArtifactSummaryV4;
196
197
  results: EvalArtifactResultV4[];
198
+ servingConfiguration?: EvalServingConfigurationV1;
199
+ aggregates?: EvalScenarioAggregateV1[];
197
200
  }
198
201
  ```
199
202
 
203
+ `servingConfiguration` and `aggregates` are additive. The v4 reader still accepts an artifact that omits them, and each result's `verdict` is optional for the same reason, so an artifact written before this release loads unchanged.
204
+
200
205
  ---
201
206
 
202
- ## Task Outcome Measurement (`verify.measure`)
207
+ ## The verdict envelope
208
+
209
+ Every result carries a strictly parsed `clio.eval.verdict.v1` envelope (`src/domains/eval/schema/verdict.ts`). Suite v2 results are adapted into it at one explicit boundary (`src/domains/eval/schema/adapter.ts`) rather than by widening the artifact version, because the envelope carries no information a v4 artifact cannot hold.
210
+
211
+ ```json
212
+ {
213
+ "schema": "clio.eval.verdict.v1",
214
+ "scenarioId": "latency-nonnegative",
215
+ "trialIndex": 0,
216
+ "outcome": "pass",
217
+ "machinery": "ok",
218
+ "reason": null,
219
+ "trackedMetrics": { "...": "see below" },
220
+ "behavioral": null,
221
+ "evidence": {
222
+ "assignmentId": "qcy5rfopdrfw",
223
+ "terminalReceiptDigest": "d85a3ad4f8ae...",
224
+ "graderExitCode": 0
225
+ }
226
+ }
227
+ ```
228
+
229
+ The envelope is fail-closed by construction. `outcome` is one of `pass`, `fail`, or `unmeasured`; `machinery` is `ok` or `infrastructure_failure`; `reason` is null for a pass or unmeasured outcome and names the rule or failure class for every failure; the original `behavioral` reservation remains exactly `null`; and an envelope claiming both `infrastructure_failure` and `pass` is rejected at parse rather than recorded. A run whose harness broke therefore cannot be read as a model that succeeded. Behavioral results use the separately versioned sibling document below rather than changing this persisted schema.
230
+
231
+ ### Behavioral scenario and verdict documents
232
+
233
+ Behavioral evaluation is additive and does not change the persisted `clio.eval.verdict.v1` reader. A Suite v2 task may declare a `clio.eval.scenario.v1` block, and its Artifact v4 result then carries a sibling `clio.eval.behavior.v1` document whose `verdictRef` names the verdict schema, scenario id, and trial index. This preserves existing verdicts and the tracked-metrics baseline while making a cross-linked behavioral document independently parseable.
234
+
235
+ The closed categories are `tool_choice`, `exploration`, `delegation`, `safety_comprehension`, `claim_grounding`, `denied_tool_recovery`, `completion_behavior`, and `task_correctness`. Each category result is exactly one of `satisfied`, `violated`, `unknown`, or `unmeasured`. The document outcome is `pass`, `behavioral_failure`, `unknown`, `unmeasured`, or `infrastructure_failure`; missing facts are never invented as successes, and an infrastructure failure cannot become a behavioral pass.
236
+
237
+ Expected and forbidden rules contain typed predicates over facts sourced from `transcript`, `tool`, `receipt`, or `grader`. Facts cite a locator, SHA-256 digest, and optional bounded excerpt. The parser caps rules, facts, evidence per category, ids, and explanations. Before judging, facts and unavailable sources are sorted into a canonical representation and hashed as `judgeInputDigest`, so input order cannot change the judge result. Duplicate or conflicting facts, missing categories, malformed evidence, contradictory outcomes, and a behavioral document that references a different result are refused.
238
+
239
+ Suite execution adapts scalar run metrics into these observable facts at the Suite v2 to Artifact v4 boundary. A declared no-tool target leaves tool-dependent rules `unmeasured`, while an available evidence source that omits a required fact produces `unknown`. Categories a role-specific scenario does not claim to measure remain `unmeasured`; they are not numeric zero and do not silently satisfy a rule.
240
+
241
+ ### Public built-in behavioral corpus
242
+
243
+ The repository ships corpus `public-built-in-behavior` version `1.0.0` under
244
+ `benchmarks/eval/`. It contains no private prompts, endpoints, credentials, or
245
+ mutable external dataset:
246
+
247
+ - `behavioral-machinery.yaml` provides one positive and one adversarial
248
+ machinery-only check for each of the 13 shipped built-in worker recipes. Its
249
+ deterministic driver loads the production recipe catalog, admits a real
250
+ dispatch through the production gate, runs a scripted worker, and verifies
251
+ the sealed receipt and result-contract outcome. The 26 scenarios require no
252
+ model; they do not infer behavior by grepping recipe frontmatter.
253
+ - `behavioral-model.yaml` provides four isolated main-agent scenarios on the
254
+ `mini` target: a focused edit, adversarial scope control, required
255
+ delegation, and recovery after Bash is denied. Together they cover all eight
256
+ behavioral categories with per-tool call and blocked-call counts, distinct
257
+ and allowlisted read-path counts, declared decoy hits, and grader-emitted
258
+ claim-support and completion facts.
259
+ - `behavioral-model-negative-control.yaml` intentionally reads a declared
260
+ decoy. A healthy run solves its literal task while recording
261
+ `behavioral_failure` with violated exploration and safety labels, proving
262
+ that the rules can reject observed model behavior rather than merely restate
263
+ aggregate success counters.
264
+
265
+ Build once, then run either focused suite from the repository root:
266
+
267
+ ```sh
268
+ node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-machinery.yaml --clio-coder-entry dist/cli/index.js
269
+ node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model.yaml --target mini --clio-coder-entry dist/cli/index.js
270
+ node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model-negative-control.yaml --target mini --clio-coder-entry dist/cli/index.js
271
+ ```
272
+
273
+ The machinery tasks use the repository read-only and create only private
274
+ scratch state under `TMPDIR`; model tasks use a fresh `temp-copy` workspace and
275
+ remove it after the matrix item settles. The machinery suite is the fast
276
+ admission, worker, and receipt contract. The model suite is the live behavioral
277
+ measurement: keep its Artifact v4 output as evidence for the exact target and
278
+ serving configuration that ran, rather than treating one observed model result
279
+ as a universal guarantee. Behavioral facts and their evidence store only
280
+ bounded read counters, not path strings. As with other eval runs, the artifact's
281
+ bounded diagnostic stdout may retain the underlying tool event stream.
282
+
283
+ ### `trackedMetrics`
284
+
285
+ Eleven numbers plus a reason histogram, each carrying the source it came from. `source` is `ledger` (the per-call ledger folded from the worker's own JSON stream), `receipt` (the sealed run receipt), or `estimated`, and `estimated` is what a missing observation is marked as rather than being silently counted as measured.
286
+
287
+ | Metric | Usual source |
288
+ | --- | --- |
289
+ | `modelCalls` | ledger |
290
+ | `uncachedPrefillTokens` | ledger, from `promptCache.backend` |
291
+ | `cacheReadTokens` | ledger, from `promptCache.backend`, falling back to pi-ai cache reads |
292
+ | `generatedTokens` | ledger |
293
+ | `reasoningTokens` | receipt; nullable, because absent and zero are different claims |
294
+ | `toolCalls`, `toolErrors` | ledger when present, otherwise receipt |
295
+ | `ttftMsFirstCall` | ledger |
296
+ | `wallClockMs` | receipt |
297
+ | `contextTokensAtEnd` | ledger |
298
+ | `compactions` | ledger |
299
+ | `expectedColdReasons` | ledger, one sourced count per reason |
300
+
301
+ A dispatched worker's receipt reports `sessionId: null` and writes no session archive, which is why the ledger source exists at all: the runner folds structured usage, backend timing, cache, and monotonic TTFT facts out of the worker's `message_end` events. It keeps no prompt text, no model prose, and no tool-result content in that fold.
302
+
303
+ ### Scenario aggregates
304
+
305
+ `aggregates` groups verdicts by `scenarioId`, sets `k` to the trial count, and records `passAtK` (any trial passed) and `passPowK` (every trial passed). Each tracked numeric metric reports observation, measured, and unmeasured counts, mean, min, max, nearest-rank p90, population variance, standard deviation, and the set of sources observed. A metric with no observation keeps every numeric statistic `null`; it never becomes zero. At `k: 1`, variance and standard deviation are zero only when the value was actually measured.
306
+
307
+ ### Behavioral multi-metric results
308
+
309
+ A result with a `clio.eval.behavior.v1` verdict also carries the additive
310
+ `clio.eval.behavior.metrics.v1` projection. The projection binds the scenario
311
+ to its role and target/model envelope and records one `number | null`
312
+ observation for each closed metric. The source travels beside every value:
313
+
314
+ | Family | Metric | Direction | Gate | Source |
315
+ |---|---|---|---|---|
316
+ | correctness | `correctness.taskSolved` | higher | hard | grader |
317
+ | safety | `safety.violations` | lower | hard | behavioral label |
318
+ | behavior | `behavior.labelViolations` | lower | informational | behavioral labels |
319
+ | efficiency | `efficiency.toolCalls` | lower | informational | terminal tool events |
320
+ | exploration | `exploration.unnecessaryReads` | lower | informational | read observation counters |
321
+ | delegation | `delegation.quality` | higher | informational | behavioral label |
322
+ | claims | `claims.unsupported` | lower | informational | grader |
323
+ | tokens | `tokens.total` | lower | informational | runner usage stream |
324
+ | latency | `latency.wallMs` | lower | informational | monotonic runner clock |
325
+ | cost | `cost.usd` | lower | informational | sealed receipt |
326
+
327
+ Label metrics are numeric projections only when the category is `satisfied` or
328
+ `violated`; `unknown` and `unmeasured` remain null. A missing grader, token
329
+ stream, receipt, read observation, or category label likewise remains null.
330
+ The projection therefore records observation coverage without claiming that
331
+ silence was success, safety, or zero cost.
332
+
333
+ ### Behavioral comparisons and variance
334
+
335
+ `eval compare` reduces the behavioral projection independently for every
336
+ scenario, role, target id, and model id. Each row contains the baseline and
337
+ candidate distributions, coverage, mean delta, variance delta, and two closed
338
+ classifications: `improved`, `regressed`, `unchanged`, or `incomparable` for the
339
+ mean and for variability. Lower variance is the improvement direction for the
340
+ variability classification.
341
+
342
+ Correctness and safety rows are hard. A measured regression fails the hard
343
+ gate even when pass rate, tokens, latency, or cost improved. Losing a
344
+ correctness or safety measurement that existed in the baseline is also a hard
345
+ failure; a category explicitly unmeasured on both sides stays incomparable but
346
+ does not invent a regression. Other families remain visible informational
347
+ tradeoffs. `--metric` accepts either a behavioral metric or family as well as a
348
+ tracked metric, but filtering displayed rows never filters the hard-gate
349
+ decision.
350
+
351
+ Comparison output supports `text`, `json`, `md`, and `junit`. All four carry
352
+ the same hard-gate result and closed classifications. JUnit failures represent
353
+ only hard behavioral failures; an informational efficiency or cost regression
354
+ is emitted as testcase output rather than a failed testcase.
355
+
356
+ ### Execution-envelope provenance and comparability
357
+
358
+ Every newly written behavioral result carries an additive
359
+ `clio.eval.execution-envelope.v1` sibling. Artifact v4,
360
+ `clio.eval.verdict.v1`, and `clio.eval.behavior.metrics.v1` retain their
361
+ existing identities. The envelope records the selected prompt fragment ids,
362
+ authored versions or `unversioned` marker, fragment content hashes, prompt
363
+ composition hash, recipe id/version/fingerprint when a worker recipe applies,
364
+ target, wire model, runtime, thinking level, tool signature, effective
365
+ autonomy, rule-pack and project-policy hashes, bounded project-context
366
+ provenance, and corpus id/version. A machinery-only scenario uses explicit
367
+ nulls for model concepts that did not apply; null is not substituted for a
368
+ fact that was observed.
369
+
370
+ Suite v2 may declare `matrix.dimensions` from `prompt`, `recipe`, `target`,
371
+ `wireModel`, `runtime`, `thinkingLevel`, `toolSignature`, `autonomy`, `policy`,
372
+ `projectContext`, and `corpus`. Comparison ignores only dimensions declared by
373
+ both artifacts. Any other envelope difference marks every metric row for that
374
+ scenario/role/target incomparable and fails the behavioral gate. A missing
375
+ envelope on only one side is also incomparable. Two older artifacts that both
376
+ predate the sibling remain readable and compare under their existing data.
377
+
378
+ Text, JSON, Markdown, and JUnit comparison reports carry the same envelope
379
+ mismatch. Text and Markdown also include independent per-scenario and per-role
380
+ baseline/candidate counts for improved, regressed, unchanged, and incomparable
381
+ metric means and variances. When the prompt or recipe identity changes, the
382
+ generated evidence names each affected corpus scenario and role instead of
383
+ hiding it behind an aggregate score.
384
+
385
+ ### Checked behavioral release baseline
386
+
387
+ The checked deterministic baseline is
388
+ `benchmarks/eval/behavioral-machinery-baseline.json`. The release gate runs all
389
+ 26 machinery-only scenarios through the built CLI and compares a stable
390
+ projection of their labels, metrics, and execution envelopes with that file.
391
+ It requires no model, private endpoint, credential, or mutable dataset.
392
+
393
+ When an intentional prompt, recipe, policy, or expected-behavior change moves
394
+ the evidence, run the same machinery suite first, inspect the failing diff and
395
+ the named affected corpus results, then update explicitly:
396
+
397
+ ```sh
398
+ npm run build
399
+ node benchmarks/eval/check-behavioral-release.mjs --update
400
+ git diff -- benchmarks/eval/behavioral-machinery-baseline.json
401
+ ```
402
+
403
+ The baseline update belongs in the reviewed change that caused it. Do not use
404
+ the update command merely to make a red gate green. The model-required and
405
+ negative-control suites remain manual release evidence because their outputs
406
+ depend on a live target; they are never folded into the deterministic baseline.
407
+ The projection excludes `latency.wallMs` because scheduler timing is not stable
408
+ evidence. Behavioral labels, deterministic metrics, and the execution envelope
409
+ remain checked byte for byte.
410
+
411
+ ### Hard thresholds and informational budgets
412
+
413
+ Suite and external threshold files keep two separate assertion lists:
414
+
415
+ ```yaml
416
+ thresholds:
417
+ fail:
418
+ - metric: task.solved
419
+ op: eq
420
+ value: false
421
+ informational:
422
+ - metric: cost.usd
423
+ op: gt
424
+ value: 0.25
425
+ ```
426
+
427
+ `fail` is the backwards-compatible hard list. A firing or unresolved hard
428
+ assertion makes `eval run` or `eval gate` exit nonzero. `informational` uses the
429
+ same typed predicates and reports every firing budget or missing measurement,
430
+ but never changes the exit status. `eval gate` additionally evaluates the
431
+ baseline-to-candidate correctness and safety hard gate, so a cheaper candidate
432
+ cannot offset a task or safety regression.
433
+
434
+ ### `--trials N`
203
435
 
204
- Task outcome commands declared under `verify.measure` evaluate whether the model solved the workload and record metrics (`task.solved`, `task.exitCode`). A non-zero exit from `verify.measure` is recorded as data and **never fails the evaluation item**. Task solution outcome is a measurement, while only machinery invariant behavior operates as a gate.
436
+ `--trials N` overrides the suite's `matrix.repeats` and asks for an isolated workspace per matrix item. A `local` workspace is converted to a temporary copy immediately before that item runs, so an explicit trial run never mutates the directory it was pointed at; `git` and `temp-copy` workspaces already produce a distinct preparation directory per item. Workspace and state directories are removed on the item's `finally` path, including runner, setup, and copy failures. The trial index rides through to each verdict's `trialIndex`.
437
+
438
+ ### Serving-configuration provenance and drift refusal
439
+
440
+ `servingConfiguration` records what the numbers were measured against: `targetId`, `runtimeId`, `modelId`, `serverBuild`, `total_slots`, `thinkingLevel`, and `compiledPromptHash`. The build string and slot count are read from the server after the matrix has run while it is still awake, by fetching `/props` and falling back to the model-qualified slots query when `/props` exposes no `total_slots`. The prompt hash is the receipt's static composition hash, so a prompt change is visible as a configuration change rather than as a mysterious metric shift.
441
+
442
+ `eval compare` prints both configurations and refuses outright when they differ:
443
+
444
+ ```text
445
+ serving configuration drift; pass --allow-config-drift to compare these runs
446
+ baseline serving: target=mini runtime=llamacpp model=... server_build=b226-2115b73d8 total_slots=1 thinking=off compiled_prompt_hash=...
447
+ candidate serving: ...
448
+ ```
449
+
450
+ `--allow-config-drift` proceeds and labels the comparison `config drift: allowed`. There is a second refusal that has no override: a metric whose baseline distribution contains an `estimated` observation and whose candidate does not, or the reverse, raises `EvalTrackedMetricSourceMismatchError` rather than printing a delta, because subtracting a measurement from an estimate produces a number that looks like evidence and is not. `--metric <name>` filters tracked or behavioral rows, accepts `expectedColdReasons`, a specific `expectedColdReasons.<reason>`, a behavioral family, or a behavioral metric, and errors when the name matches nothing.
451
+
452
+ ---
453
+
454
+ ## Task Outcome Measurement (`verify.measure`)
205
455
 
456
+ Task outcome commands declared under `verify.measure` are the code grader for whether the model solved the workload and record metrics (`task.solved`, `task.exitCode`). A non-zero exit fails the final result and is named on its verdict as `reason: grader_failed`, while `machinery` remains `ok` when the runner and machinery verifiers succeeded. This keeps the artifact's `pass`, verdict outcome, scenario aggregates, and summary on one pass decision without misreporting a grader failure as broken machinery.
@@ -1,7 +1,7 @@
1
1
  # Internal Eval Suites
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive blueprint is available at [docs/html/evals_internal_blueprint.html](html/evals_internal_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive blueprint is available at [docs/html/evals_internal_blueprint.html](html/evals_internal_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Private suites should live outside this repository. Keep datasets, prompts,
7
7
  live fleet coordinates, calibration outputs, and raw run artifacts in a private
@@ -19,6 +19,77 @@ data directory. Product eval artifacts and external benchmark campaigns are
19
19
  separate: public benchmark adapters live under `benchmarks/community/` and do
20
20
  not use the eval runner.
21
21
 
22
+ The public behavioral corpus is the deliberate exception to the otherwise
23
+ private Suite v2 data policy. Its reviewable, synthetic suites live under
24
+ `benchmarks/eval/`: a model-free positive/adversarial authority pair for every
25
+ built-in worker recipe, four tiny main-agent model scenarios covering all eight
26
+ behavioral categories with event- and grader-derived facts, and an intentional
27
+ decoy negative control. The model-free driver uses the shipped recipe catalog,
28
+ real dispatch admission, scripted workers, and sealed receipts rather than
29
+ frontmatter inspection. See
30
+ [eval-runner.md](eval-runner.md#public-built-in-behavioral-corpus) for the
31
+ focused commands. Private prompts, calibration cases, fleet coordinates, and
32
+ campaign artifacts still belong outside this repository and must not be copied
33
+ into the public corpus.
34
+
35
+ ## Running a private suite as a measurement
36
+
37
+ A private suite is usually run to answer whether a harness change moved
38
+ something, which makes it a measurement rather than a pass or fail. Three
39
+ mechanics matter for that, all documented in full in
40
+ [eval-runner.md](eval-runner.md#the-verdict-envelope).
41
+
42
+ Run repeated trials with `--trials N` rather than by editing `matrix.repeats`.
43
+ The flag overrides the suite's repeat count and asks for an isolated workspace
44
+ per matrix item, so a `local` workspace is copied for the run instead of being
45
+ mutated across trials. Each result's verdict carries its `trialIndex`, and the
46
+ artifact's `aggregates` reduce them per scenario: `k`, `passAtK` (any trial
47
+ passed), `passPowK` (every trial passed), and a mean and nearest-rank p90 for
48
+ every tracked metric. A single trial produces a `k: 1` aggregate whose mean and
49
+ p90 are the same observed value, which is a fact to state in a report rather
50
+ than a distribution to reason about.
51
+
52
+ Read the tracked metrics with their sources attached. A private suite on a
53
+ local target is measuring prefill economics as much as correctness, so
54
+ `uncachedPrefillTokens`, `cacheReadTokens`, `ttftMsFirstCall`, and the
55
+ `expectedColdReasons` histogram are the interesting columns, and each one says
56
+ whether it came from the ledger, from the receipt, or was `estimated`. A metric
57
+ marked `estimated` on one side of a comparison and measured on the other is
58
+ refused rather than differenced.
59
+
60
+ Behavioral suites add a second projection beside those tracked performance
61
+ metrics. Compare it per scenario, role, and target/model envelope rather than
62
+ reducing unlike roles into one pass rate. Correctness and safety are hard
63
+ regression gates; tool efficiency, unnecessary exploration, delegation
64
+ quality, unsupported claims, tokens, latency, and receipt cost remain separate
65
+ families with their own measured coverage and repeat variance. A missing value
66
+ is null and makes that row incomparable, never zero.
67
+
68
+ Put release-blocking assertions under `thresholds.fail` and non-blocking spend
69
+ or latency budgets under `thresholds.informational`. Informational findings are
70
+ printed in every gate run but do not change its exit status. Do not put a cost
71
+ budget in the hard list to compensate for weak correctness, and do not turn a
72
+ correctness rule into an informational budget; the comparison gate evaluates
73
+ correctness and safety before either kind of operator-authored threshold.
74
+
75
+ Record the serving configuration or the comparison is not one. The artifact
76
+ captures `targetId`, `runtimeId`, `modelId`, `serverBuild`, `total_slots`,
77
+ `thinkingLevel`, and `compiledPromptHash`, read from the server after the matrix
78
+ has run while it is still awake. `eval compare` refuses two artifacts whose
79
+ configurations differ unless `--allow-config-drift` is passed, and prints both
80
+ either way. Treat that refusal as the useful behavior it is: a private suite
81
+ compared across a server restart that changed a flag, a quantization, or the
82
+ thinking level is measuring the server, not the change under test.
83
+
84
+ The verdict envelope keeps its original `behavioral: null` field for compatibility.
85
+ A suite that declares a versioned behavioral scenario records the result as a
86
+ separate `clio.eval.behavior.v1` document on the Artifact v4 result, cross-linked
87
+ to the unchanged verdict identity. Its labels come only from bounded transcript,
88
+ tool, receipt, or grader facts, never from an ungrounded judge paragraph. A run whose harness broke records
89
+ `machinery: "infrastructure_failure"`, which the parser refuses to pair with a
90
+ `pass`, so a private suite cannot report a passing rate that includes runs
91
+ nothing measured.
92
+
22
93
  ## Context Regression Seed
23
94
 
24
95
  ```yaml
@@ -267,4 +338,3 @@ thresholds:
267
338
  op: gt
268
339
  value: 0
269
340
  ```
270
-