@iowarp/clio-coder 0.3.7 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +13 -4
  3. package/dist/{acp-SK4MD6MM.js → acp-7LOELQFP.js} +13 -13
  4. package/dist/{agents-2FN2K6ME.js → agents-FIBG2SHA.js} +41 -37
  5. package/dist/assets/codewiki.json +1 -1
  6. package/dist/{auth-QIYZWM5I.js → auth-OI4LIH2I.js} +31 -24
  7. package/dist/builtins-AD25UL3C.js +17 -0
  8. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  9. package/dist/chunk-3DPEIQKN.js +113 -0
  10. package/dist/{chunk-EOOQZZDE.js → chunk-3DUR4WUA.js} +19 -19
  11. package/dist/{chunk-WHJYKASB.js → chunk-3MRC2YSQ.js} +2 -2
  12. package/dist/{chunk-EBEFWSGL.js → chunk-3UUY7R3Z.js} +14 -10
  13. package/dist/{chunk-LADCF22A.js → chunk-3V5AYSEQ.js} +113 -54
  14. package/dist/{chunk-BMWK7ZIZ.js → chunk-465CC7FK.js} +16 -13
  15. package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
  16. package/dist/{chunk-CEYBNUGC.js → chunk-4H6ULJ3H.js} +378 -36
  17. package/dist/{chunk-YTYFXUI3.js → chunk-4LJX2PUC.js} +9 -9
  18. package/dist/{chunk-DOOEX22V.js → chunk-56KB5IJP.js} +5 -5
  19. package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
  20. package/dist/{chunk-TSHXZTOQ.js → chunk-5HFBWUMU.js} +23 -11
  21. package/dist/{chunk-5UJ6ECTS.js → chunk-5PVQ4SRS.js} +80 -8
  22. package/dist/{chunk-ZWLZP4ZT.js → chunk-5QKCQQ3E.js} +359 -17
  23. package/dist/{chunk-6M7VS3J3.js → chunk-5T7RBWN2.js} +111 -5
  24. package/dist/chunk-774ILSRL.js +172 -0
  25. package/dist/chunk-7C6RYZGQ.js +391 -0
  26. package/dist/{chunk-GH5622CP.js → chunk-A2NJGIB3.js} +2 -2
  27. package/dist/{chunk-C4JBQ5SR.js → chunk-AD7Y7STJ.js} +6 -6
  28. package/dist/{chunk-GEYXPTRF.js → chunk-AEYBF3TB.js} +33 -12
  29. package/dist/{chunk-2SFS6XQE.js → chunk-AMKHQW3C.js} +3 -2
  30. package/dist/{chunk-D4MDIG46.js → chunk-B5CSFE7B.js} +7 -7
  31. package/dist/{chunk-MXI6J5JF.js → chunk-B5XRQOLB.js} +10 -10
  32. package/dist/{chunk-X2KV5FXT.js → chunk-BVDVID7E.js} +2 -2
  33. package/dist/{chunk-JNXPYBB4.js → chunk-CA42X6KT.js} +3 -3
  34. package/dist/{chunk-VREKEFLL.js → chunk-D73KXYPF.js} +3 -3
  35. package/dist/{chunk-JTSEDYVQ.js → chunk-DG4M6ZUE.js} +7 -7
  36. package/dist/{chunk-DQA7QLMD.js → chunk-EBOC7MT3.js} +10 -25
  37. package/dist/{chunk-KZ2H5X4G.js → chunk-ECUO3KDP.js} +129 -14
  38. package/dist/{chunk-JRIO5UD2.js → chunk-EQ63NRB7.js} +5 -5
  39. package/dist/{chunk-YD734TPH.js → chunk-FALJGAWU.js} +2 -2
  40. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  41. package/dist/{chunk-XEGB6BCN.js → chunk-GAYUJ7LE.js} +68 -14
  42. package/dist/{chunk-UND3GU2L.js → chunk-H7IXIC72.js} +2 -2
  43. package/dist/{chunk-IR4CFBFN.js → chunk-HAY4ZE2P.js} +12 -12
  44. package/dist/{chunk-UVDSQ6LW.js → chunk-HCBCAYZU.js} +74 -147
  45. package/dist/{chunk-4DWFMQDR.js → chunk-HJB5IUKP.js} +89 -145
  46. package/dist/{chunk-M4AKACEO.js → chunk-HKO36JWF.js} +33 -5
  47. package/dist/{chunk-KCMKRQX4.js → chunk-HPCTNZM2.js} +45 -82
  48. package/dist/{chunk-465YSENW.js → chunk-IFBNV6H6.js} +3 -3
  49. package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
  50. package/dist/chunk-JEQQR47K.js +3025 -0
  51. package/dist/{chunk-FO5ZOVUY.js → chunk-KV2AOLDF.js} +27 -7
  52. package/dist/chunk-LU7P4LHA.js +33 -0
  53. package/dist/{chunk-6TUKSZVF.js → chunk-LXPJXFM5.js} +11 -11
  54. package/dist/{chunk-VQNODYQ4.js → chunk-MIX5N5AC.js} +488 -3668
  55. package/dist/chunk-MLOK6ZOS.js +2888 -0
  56. package/dist/{chunk-ZZMN5OM4.js → chunk-MV2VUEJC.js} +2 -2
  57. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  58. package/dist/{chunk-OBMAI2DP.js → chunk-N3PBVRTZ.js} +12 -388
  59. package/dist/{chunk-WJHBC77E.js → chunk-N5XKWMDW.js} +17 -7
  60. package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
  61. package/dist/{chunk-UFQ3F4FW.js → chunk-NQ6UCCOD.js} +4 -4
  62. package/dist/chunk-NUGM5KR6.js +165 -0
  63. package/dist/{chunk-DMD2AGVS.js → chunk-NZU6YDNV.js} +20 -18
  64. package/dist/{chunk-WHGPSPT5.js → chunk-O6I4CIEU.js} +151 -13
  65. package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
  66. package/dist/{chunk-PD3MESLB.js → chunk-P3JGPQFL.js} +4 -4
  67. package/dist/{chunk-UHXRNZ2J.js → chunk-PNY46YEY.js} +23 -6
  68. package/dist/{chunk-THKY7CD7.js → chunk-PZ4I4JE2.js} +134 -29
  69. package/dist/{chunk-SROCI7ZU.js → chunk-QQ7EKM72.js} +5 -5
  70. package/dist/{chunk-QCTRSGHQ.js → chunk-R7LNVMCS.js} +91 -53
  71. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  72. package/dist/chunk-RKKLTLYB.js +45 -0
  73. package/dist/{chunk-OB5HIGJY.js → chunk-RKRLDWD3.js} +4 -1
  74. package/dist/{chunk-DJNLUABN.js → chunk-S4COXYBG.js} +588 -32
  75. package/dist/{chunk-3HAPLH5M.js → chunk-T3Z6VAAF.js} +172 -11
  76. package/dist/{chunk-FOT2FX5J.js → chunk-TD7UE2L5.js} +12 -10
  77. package/dist/{chunk-UUANF5CR.js → chunk-TEO2TLVN.js} +856 -967
  78. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  79. package/dist/{chunk-EELBMBT6.js → chunk-VKBMFOYV.js} +74 -15
  80. package/dist/chunk-VO2LKSTM.js +165 -0
  81. package/dist/{chunk-5C77SEEY.js → chunk-VPTUJU4P.js} +3 -3
  82. package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
  83. package/dist/{chunk-J7PIKKWC.js → chunk-WXCJ7VME.js} +8 -8
  84. package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
  85. package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
  86. package/dist/{chunk-AB4XIIVB.js → chunk-YKOFT37S.js} +6 -6
  87. package/dist/chunk-YSEHGPCT.js +127 -0
  88. package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
  89. package/dist/cli/index.js +32 -32
  90. package/dist/{clio-WBVQEBKO.js → clio-LT5V7SSZ.js} +9 -9
  91. package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +5 -5
  92. package/dist/codewiki/build-worker.js +4 -4
  93. package/dist/{components-F7OEATSO.js → components-ZFA3SAER.js} +8 -8
  94. package/dist/{config-TRBL3RCF.js → config-RXS5T3JT.js} +98 -65
  95. package/dist/{configure-OLCVPHNM.js → configure-2WYWSCSD.js} +26 -22
  96. package/dist/{context-MJIJ6GOX.js → context-I3BTOTCS.js} +12 -12
  97. package/dist/{context-XEWE3MOJ.js → context-MVOORGMF.js} +54 -47
  98. package/dist/{context-WFPKQSM6.js → context-PALKKQYL.js} +28 -28
  99. package/dist/{context-clear-KNOS2JPB.js → context-clear-N2WOYZ2K.js} +53 -46
  100. package/dist/{context-index-SSR5ECNE.js → context-index-HNG3MOME.js} +6 -6
  101. package/dist/{context-working-set-EUXAZI6N.js → context-working-set-MIEVECVZ.js} +17 -18
  102. package/dist/{dispatch-runner-B7MTOVKL.js → dispatch-runner-VVA4SRRH.js} +90 -61
  103. package/dist/{docs-FLJTIDSE.js → docs-7LQ23DLM.js} +8 -8
  104. package/dist/doctor-TWBWFK5V.js +165 -0
  105. package/dist/eval-IJ5VEZDJ.js +4483 -0
  106. package/dist/{evidence-JZNBUOQZ.js → evidence-L5APPXNV.js} +68 -61
  107. package/dist/{evolve-FJVC4KKI.js → evolve-RGNKFJ52.js} +47 -40
  108. package/dist/{extensions-IQL36S7K.js → extensions-7WYWUX5A.js} +13 -7
  109. package/dist/{fleet-BDKYJFCP.js → fleet-6CNVBZZP.js} +113 -76
  110. package/dist/{fleet-commands-ZFIWZSB3.js → fleet-commands-L2SXSYEI.js} +10 -10
  111. package/dist/{fleet-graph-Y6HPXIVF.js → fleet-graph-2J3OOIPO.js} +17 -15
  112. package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-CZRJ4JP5.js} +5 -6
  113. package/dist/{fleet-validate-BIYREGIK.js → fleet-validate-C5RI6DP7.js} +20 -19
  114. package/dist/{init-LQUB5COQ.js → init-VBN2ACVA.js} +70 -63
  115. package/dist/{library-NJAHIGG4.js → library-JHGUMLY2.js} +22 -20
  116. package/dist/{memory-OG6HOYKM.js → memory-K4OQIYWG.js} +49 -42
  117. package/dist/{models-5ZG5XY7J.js → models-2NCZUWDD.js} +35 -29
  118. package/dist/{monitor-TJ7AMTGB.js → monitor-MMVTJABD.js} +64 -45
  119. package/dist/{orchestrator-WZYB54DM.js → orchestrator-ZKBPCHW6.js} +1971 -520
  120. package/dist/{paths-XUC7GS6E.js → paths-DBXMZMDU.js} +5 -5
  121. package/dist/registry-LG64LTF4.js +11 -0
  122. package/dist/{reset-PXQT45IY.js → reset-DD5JGOY3.js} +11 -11
  123. package/dist/{run-FQ74YF62.js → run-QEGNX7FL.js} +89 -83
  124. package/dist/{share-FW7SVCL3.js → share-JKD3BQMW.js} +20 -18
  125. package/dist/{skills-7E7IRB3R.js → skills-LMQIKDOZ.js} +23 -21
  126. package/dist/{skills-eval-LI75W6OK.js → skills-eval-I7X2774U.js} +59 -52
  127. package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
  128. package/dist/support-I7LOJLIF.js +38 -0
  129. package/dist/{targets-4CIFKCTW.js → targets-RUSR6B5Z.js} +77 -42
  130. package/dist/{terminal-lease-WUZY7ZV5.js → terminal-lease-QYVORFR4.js} +6 -4
  131. package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
  132. package/dist/{uninstall-7FV7IP4E.js → uninstall-ZJF5H5ZN.js} +8 -8
  133. package/dist/{upgrade-K2HVIVMQ.js → upgrade-XANW3FXB.js} +29 -26
  134. package/dist/{usage-GTZELZQX.js → usage-4H7ZRXQT.js} +110 -61
  135. package/dist/{verifiers-RLAHT27O.js → verifiers-UZXNBZEB.js} +13 -13
  136. package/dist/{verify-BX3BRKH5.js → verify-BVKWTNDL.js} +9 -9
  137. package/dist/{wiki-generate-ASIFASCN.js → wiki-generate-MY7WV2QI.js} +76 -69
  138. package/dist/worker/entry.js +69 -66
  139. package/docs/alcf-provider.md +1 -1
  140. package/docs/architecture.md +1 -1
  141. package/docs/artifact-versions.md +11 -5
  142. package/docs/built-in-agents.md +1 -1
  143. package/docs/capacity-and-scheduling.md +23 -2
  144. package/docs/commands-and-modes.md +2 -2
  145. package/docs/configuration-and-targets.md +37 -5
  146. package/docs/context-engine.md +63 -4
  147. package/docs/documentation-coverage.md +3 -3
  148. package/docs/documentation-guide.md +1 -1
  149. package/docs/environment-variables.md +2 -0
  150. package/docs/eval-runner.md +262 -11
  151. package/docs/evals-internal.md +72 -2
  152. package/docs/evidence-and-memory.md +77 -12
  153. package/docs/evolution.md +1 -1
  154. package/docs/extensions-and-sharing.md +3 -1
  155. package/docs/fleet-dispatch.md +34 -9
  156. package/docs/glossary.md +21 -1
  157. package/docs/installation-and-lifecycle.md +1 -1
  158. package/docs/middleware-and-components.md +1 -1
  159. package/docs/model-catalog.md +1 -1
  160. package/docs/observability.md +54 -3
  161. package/docs/proactive-memory.md +127 -14
  162. package/docs/prompt-envelope-and-tools.md +22 -2
  163. package/docs/provider-adapter-cookbook.md +1 -1
  164. package/docs/release-cut-checklist.md +60 -41
  165. package/docs/safety-model.md +1 -1
  166. package/docs/scientific-validation.md +1 -1
  167. package/docs/skills-marketplace.md +1 -1
  168. package/docs/tool-usage.md +1 -1
  169. package/docs/trace-store.md +1 -1
  170. package/docs/troubleshooting.md +87 -0
  171. package/docs/tui-design.md +1 -1
  172. package/docs/worker-dispatch-mechanics.md +1 -1
  173. package/package.json +2 -2
  174. package/src/cli/agents.ts +1 -1
  175. package/src/cli/argv.ts +5 -0
  176. package/src/cli/config-inspect.ts +33 -6
  177. package/src/cli/config.ts +1 -1
  178. package/src/cli/configure.ts +107 -23
  179. package/src/cli/doctor-state-size.ts +82 -0
  180. package/src/cli/doctor.ts +7 -1
  181. package/src/cli/eval.ts +80 -16
  182. package/src/cli/evidence.ts +30 -25
  183. package/src/cli/extensions.ts +5 -1
  184. package/src/cli/fleet-preflight.ts +2 -12
  185. package/src/cli/fleet.ts +32 -3
  186. package/src/cli/shared.ts +1 -0
  187. package/src/cli/targets.ts +45 -11
  188. package/src/cli/trace.ts +63 -4
  189. package/src/cli/usage.ts +63 -14
  190. package/src/cli/validate-model.ts +60 -5
  191. package/src/core/bus-events.ts +54 -1
  192. package/src/core/cache-telemetry.ts +42 -0
  193. package/src/core/commit-attribution.ts +4 -4
  194. package/src/core/config.ts +18 -0
  195. package/src/core/defaults.ts +36 -6
  196. package/src/core/endpoint-key.ts +27 -0
  197. package/src/core/path-boundary.ts +100 -0
  198. package/src/core/residency-target-key.ts +25 -0
  199. package/src/core/response-schema.ts +36 -2
  200. package/src/domains/agents/extension.ts +2 -11
  201. package/src/domains/agents/fleet-contract.ts +30 -12
  202. package/src/domains/agents/recipe.ts +7 -1
  203. package/src/domains/agents/registry.ts +73 -5
  204. package/src/domains/agents/result-contract.ts +128 -17
  205. package/src/domains/agents/write-boundary.ts +15 -50
  206. package/src/domains/config/classify.ts +3 -0
  207. package/src/domains/context/codewiki/coordinator.ts +12 -4
  208. package/src/domains/context/project-rules.ts +51 -1
  209. package/src/domains/dispatch/admission.ts +40 -3
  210. package/src/domains/dispatch/assignment-reconcile.ts +22 -5
  211. package/src/domains/dispatch/assignment-store.ts +151 -14
  212. package/src/domains/dispatch/capacity-lease.ts +98 -9
  213. package/src/domains/dispatch/contract.ts +26 -1
  214. package/src/domains/dispatch/delegation-plan.ts +2 -5
  215. package/src/domains/dispatch/execution-plan.ts +44 -4
  216. package/src/domains/dispatch/execution-role.ts +9 -1
  217. package/src/domains/dispatch/extension.ts +309 -96
  218. package/src/domains/dispatch/fleet-run.ts +78 -4
  219. package/src/domains/dispatch/gate-role-prompts.ts +38 -0
  220. package/src/domains/dispatch/heartbeat.ts +32 -8
  221. package/src/domains/dispatch/index.ts +6 -1
  222. package/src/domains/dispatch/intent-requirements.ts +40 -0
  223. package/src/domains/dispatch/intent.ts +84 -8
  224. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  225. package/src/domains/dispatch/path-scope.ts +370 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +2 -1
  227. package/src/domains/dispatch/reservation-store.ts +116 -8
  228. package/src/domains/dispatch/state.ts +4 -0
  229. package/src/domains/dispatch/types.ts +14 -7
  230. package/src/domains/dispatch/validation.ts +6 -3
  231. package/src/domains/dispatch/worker-spawn.ts +25 -11
  232. package/src/domains/dispatch/write-boundary-enforcer.ts +62 -0
  233. package/src/domains/dispatch/write-boundary.ts +262 -22
  234. package/src/domains/eval/artifacts/store.ts +62 -0
  235. package/src/domains/eval/compare/behavioral.ts +224 -0
  236. package/src/domains/eval/compare/compare.ts +355 -2
  237. package/src/domains/eval/compare/envelope.ts +128 -0
  238. package/src/domains/eval/compare/gates.ts +24 -6
  239. package/src/domains/eval/compare/thresholds.ts +30 -3
  240. package/src/domains/eval/execution-provenance.ts +240 -0
  241. package/src/domains/eval/metrics/aggregate.ts +136 -0
  242. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  243. package/src/domains/eval/metrics/evidence.ts +79 -2
  244. package/src/domains/eval/metrics/tracked.ts +413 -0
  245. package/src/domains/eval/provenance.ts +117 -0
  246. package/src/domains/eval/reports/comparison.ts +128 -0
  247. package/src/domains/eval/reports/junit.ts +17 -3
  248. package/src/domains/eval/reports/markdown.ts +3 -3
  249. package/src/domains/eval/reports/text.ts +14 -0
  250. package/src/domains/eval/run-compare.ts +20 -0
  251. package/src/domains/eval/runners/clio-run.ts +139 -2
  252. package/src/domains/eval/runners/external-command.ts +28 -3
  253. package/src/domains/eval/schema/adapter.ts +111 -0
  254. package/src/domains/eval/schema/artifact.ts +20 -0
  255. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  256. package/src/domains/eval/schema/behavioral.ts +520 -0
  257. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  258. package/src/domains/eval/schema/serving.ts +74 -0
  259. package/src/domains/eval/schema/suite.ts +38 -8
  260. package/src/domains/eval/schema/validate.ts +58 -3
  261. package/src/domains/eval/schema/verdict.ts +237 -0
  262. package/src/domains/eval/suites/resolve.ts +2 -0
  263. package/src/domains/eval/suites/run.ts +264 -33
  264. package/src/domains/eval/verifiers/command.ts +2 -1
  265. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  266. package/src/domains/evidence/build.ts +68 -21
  267. package/src/domains/evidence/eval.ts +2 -12
  268. package/src/domains/evidence/findings-markdown.ts +33 -0
  269. package/src/domains/evidence/index.ts +21 -0
  270. package/src/domains/evidence/provenance.ts +46 -11
  271. package/src/domains/evidence/run-trust.ts +7 -113
  272. package/src/domains/evidence/trust-projection.ts +274 -0
  273. package/src/domains/evidence/trust-status.ts +145 -17
  274. package/src/domains/evidence/types.ts +4 -0
  275. package/src/domains/extensions/compatibility.ts +285 -0
  276. package/src/domains/extensions/discovery.ts +126 -4
  277. package/src/domains/extensions/resources.ts +21 -9
  278. package/src/domains/extensions/state.ts +18 -5
  279. package/src/domains/extensions/types.ts +6 -1
  280. package/src/domains/lifecycle/doctor.ts +209 -2
  281. package/src/domains/memory/index.ts +14 -0
  282. package/src/domains/memory/task-bank-promotion.ts +64 -0
  283. package/src/domains/memory/task-memory-policy.ts +77 -8
  284. package/src/domains/memory/task-memory-spend.ts +131 -0
  285. package/src/domains/memory/task-memory-status.ts +7 -0
  286. package/src/domains/memory/task-memory-telemetry.ts +2 -0
  287. package/src/domains/middleware/index.ts +1 -0
  288. package/src/domains/middleware/memory-intervention.ts +69 -5
  289. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  290. package/src/domains/observability/background-memory-usage.ts +140 -0
  291. package/src/domains/observability/cost.ts +1 -1
  292. package/src/domains/observability/index.ts +7 -0
  293. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  294. package/src/domains/observability/trace-store.ts +192 -2
  295. package/src/domains/prompts/compiler.ts +100 -13
  296. package/src/domains/prompts/contract.ts +3 -5
  297. package/src/domains/providers/endpoint-capacity.ts +96 -0
  298. package/src/domains/providers/extension.ts +30 -2
  299. package/src/domains/providers/index.ts +10 -0
  300. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  301. package/src/domains/providers/runtime-resolution.ts +8 -1
  302. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  303. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  304. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  305. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  306. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  307. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  308. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  309. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  310. package/src/domains/providers/types/capability-flags.ts +2 -0
  311. package/src/domains/providers/types/target-descriptor.ts +2 -0
  312. package/src/domains/resources/common-loader.ts +3 -0
  313. package/src/domains/resources/prompts/loader.ts +184 -17
  314. package/src/domains/safety/call-target.ts +52 -0
  315. package/src/domains/safety/policy-engine.ts +5 -5
  316. package/src/domains/safety/run-effects.ts +96 -2
  317. package/src/domains/safety/scope.ts +7 -12
  318. package/src/domains/session/context-accounting.ts +52 -1
  319. package/src/domains/session/context-ledger.ts +37 -13
  320. package/src/domains/session/index.ts +6 -0
  321. package/src/domains/session/prompt-cache.ts +140 -0
  322. package/src/domains/session/prompt-manifest.ts +42 -0
  323. package/src/engine/acp/adapter.ts +18 -3
  324. package/src/engine/acp/server.ts +4 -1
  325. package/src/engine/ai.ts +35 -0
  326. package/src/engine/apis/llamacpp-residency.ts +55 -3
  327. package/src/engine/apis/lmstudio.ts +25 -5
  328. package/src/engine/apis/ollama-native.ts +2 -1
  329. package/src/engine/apis/openai-completions.ts +80 -17
  330. package/src/engine/apis/residency-lock.ts +3 -1
  331. package/src/engine/apis/residency.ts +34 -1
  332. package/src/engine/prompt-templates.ts +18 -1
  333. package/src/engine/provider-payload.ts +29 -1
  334. package/src/engine/worker-runtime.ts +6 -3
  335. package/src/entry/orchestrator.ts +176 -30
  336. package/src/interactive/chat-loop-messages.ts +26 -7
  337. package/src/interactive/chat-loop.ts +318 -41
  338. package/src/interactive/chat-panel.ts +62 -8
  339. package/src/interactive/clio-editor.ts +45 -8
  340. package/src/interactive/context-activity.ts +5 -1
  341. package/src/interactive/context-meter.ts +1 -1
  342. package/src/interactive/context-overlay.ts +40 -10
  343. package/src/interactive/cost-overlay.ts +64 -6
  344. package/src/interactive/dispatch-board.ts +127 -5
  345. package/src/interactive/fleet-run-preview.ts +41 -15
  346. package/src/interactive/handoff-round.ts +41 -2
  347. package/src/interactive/interactive-application.ts +24 -1
  348. package/src/interactive/interactive-event-projection.ts +14 -0
  349. package/src/interactive/interactive-input-runtime.ts +8 -0
  350. package/src/interactive/interactive-presentation.ts +4 -0
  351. package/src/interactive/interactive-shell.ts +20 -17
  352. package/src/interactive/interactive-slash-runtime.ts +27 -4
  353. package/src/interactive/memory-overlay.ts +8 -0
  354. package/src/interactive/mutation-preview.ts +295 -0
  355. package/src/interactive/overlay-general-openers.ts +16 -0
  356. package/src/interactive/overlay-key-routing.ts +38 -0
  357. package/src/interactive/overlay-lifecycle.ts +38 -5
  358. package/src/interactive/overlay-permission-lifecycle.ts +22 -2
  359. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  360. package/src/interactive/overlays/ask-user.ts +91 -19
  361. package/src/interactive/overlays/help-reference.ts +4 -0
  362. package/src/interactive/overlays/prompts.ts +11 -1
  363. package/src/interactive/overlays/settings.ts +176 -48
  364. package/src/interactive/permission-hint.ts +34 -2
  365. package/src/interactive/permission-overlay.ts +159 -9
  366. package/src/interactive/prewarm.ts +197 -0
  367. package/src/interactive/render-trace.ts +162 -15
  368. package/src/interactive/renderers/tool-execution.ts +4 -0
  369. package/src/interactive/side-question.ts +58 -1
  370. package/src/interactive/slash-commands.ts +7 -2
  371. package/src/interactive/status/controller.ts +11 -0
  372. package/src/interactive/status/state-machine.ts +54 -2
  373. package/src/interactive/status/types.ts +7 -0
  374. package/src/interactive/terminal-lease.ts +2 -0
  375. package/src/interactive/turn-context.ts +299 -31
  376. package/src/interactive/turn-persistence.ts +14 -4
  377. package/src/interactive/turn-prewarm.ts +364 -0
  378. package/src/interactive/turn-queues.ts +7 -4
  379. package/src/interactive/turn-runtime.ts +8 -1
  380. package/src/interactive/turn-state.ts +23 -0
  381. package/src/interactive/view/artifacts.ts +42 -9
  382. package/src/interactive/view/view-overlay.ts +43 -6
  383. package/src/interactive/worker-receipts.ts +14 -2
  384. package/src/interactive/worker-stream.ts +8 -0
  385. package/src/tools/ask-user.ts +43 -2
  386. package/src/tools/dispatch-admission.ts +12 -13
  387. package/src/tools/dispatch-arguments.ts +27 -0
  388. package/src/tools/dispatch-plan.ts +46 -9
  389. package/src/tools/dispatch-runner.ts +48 -13
  390. package/src/tools/dispatch-scout.ts +1 -1
  391. package/src/tools/monitor.ts +13 -0
  392. package/src/tools/registry.ts +16 -0
  393. package/src/tools/worker-evidence.ts +19 -13
  394. package/src/worker/spec-contract.ts +2 -1
  395. package/dist/chunk-AOCYTWAV.js +0 -449
  396. package/dist/chunk-HWUFFB6L.js +0 -83
  397. package/dist/chunk-R346GLFC.js +0 -31
  398. package/dist/chunk-ZGH7FGS5.js +0 -1079
  399. package/dist/doctor-RN4YKO2X.js +0 -87
  400. package/dist/eval-RUBJVSNQ.js +0 -2557
@@ -16,25 +16,29 @@ import {
16
16
  resolveDeliveryTools,
17
17
  sanitizeLockedSynthesisMessage,
18
18
  workerLoopBlockBudget
19
- } from "../chunk-QCTRSGHQ.js";
19
+ } from "../chunk-R7LNVMCS.js";
20
20
  import "../chunk-K7VKOLQQ.js";
21
21
  import {
22
22
  createMiddlewareContractFromSnapshot,
23
23
  createMiddlewareToolChoiceControl,
24
24
  shouldRequestStalledTurnContinuation
25
- } from "../chunk-XEGB6BCN.js";
25
+ } from "../chunk-GAYUJ7LE.js";
26
26
  import {
27
27
  CONFIRMED_SCOPE,
28
28
  READONLY_SCOPE,
29
29
  WORKSPACE_SCOPE,
30
30
  isSubset
31
- } from "../chunk-IR4CFBFN.js";
31
+ } from "../chunk-HAY4ZE2P.js";
32
32
  import {
33
33
  createLoopState,
34
34
  observe
35
35
  } from "../chunk-UOV2BYIW.js";
36
- import "../chunk-EELBMBT6.js";
37
- import "../chunk-WHGPSPT5.js";
36
+ import "../chunk-CA42X6KT.js";
37
+ import "../chunk-OEDBCISO.js";
38
+ import "../chunk-5DQRIYDZ.js";
39
+ import "../chunk-VKBMFOYV.js";
40
+ import "../chunk-O6I4CIEU.js";
41
+ import "../chunk-VG7TBQIY.js";
38
42
  import {
39
43
  DEFAULT_ESCALATION_FALLBACK,
40
44
  DEFAULT_ESCALATION_TIMEOUT_MS,
@@ -50,13 +54,35 @@ import {
50
54
  startAntigravityWorkerRun,
51
55
  startClaudeCodeWorkerRun,
52
56
  validateRehydratedWorkerRuntime
53
- } from "../chunk-CEYBNUGC.js";
54
- import "../chunk-5FR74PWO.js";
57
+ } from "../chunk-4H6ULJ3H.js";
58
+ import "../chunk-VPTUJU4P.js";
59
+ import "../chunk-2JDWVJND.js";
60
+ import "../chunk-H7IXIC72.js";
61
+ import "../chunk-7C6RYZGQ.js";
55
62
  import {
56
63
  createSafetyPolicyEngine
57
- } from "../chunk-OBMAI2DP.js";
58
- import "../chunk-5C77SEEY.js";
59
- import "../chunk-UND3GU2L.js";
64
+ } from "../chunk-N3PBVRTZ.js";
65
+ import "../chunk-A2NJGIB3.js";
66
+ import "../chunk-OQE5J4C6.js";
67
+ import {
68
+ RESULT_CONTRACT_REPAIR_LIMIT,
69
+ classify,
70
+ parseWorkerResultContract,
71
+ resultContractRepairMessages,
72
+ validateResultContract
73
+ } from "../chunk-S4COXYBG.js";
74
+ import {
75
+ protectedArtifactMutationBlockReason
76
+ } from "../chunk-MV3K5QF2.js";
77
+ import {
78
+ DEFAULT_AUTONOMY_LEVEL,
79
+ agentSkillToolPolicy
80
+ } from "../chunk-RAPCMZL4.js";
81
+ import "../chunk-UL3WSD3F.js";
82
+ import "../chunk-ODFEOB4F.js";
83
+ import "../chunk-OZNBF4L3.js";
84
+ import "../chunk-4BJ5BYCE.js";
85
+ import "../chunk-ECH6PKUQ.js";
60
86
  import {
61
87
  UNKNOWN_RESOURCE,
62
88
  WORKER_PROTOCOL_VERSION,
@@ -76,35 +102,6 @@ import {
76
102
  deleteInjectedCompileCacheFrom
77
103
  } from "../chunk-I4HZDVNP.js";
78
104
  import "../chunk-ZI647VB5.js";
79
- import "../chunk-JNXPYBB4.js";
80
- import "../chunk-GH5622CP.js";
81
- import "../chunk-XN3L4EYL.js";
82
- import "../chunk-VG7TBQIY.js";
83
- import "../chunk-OQE5J4C6.js";
84
- import "../chunk-SPULKLCF.js";
85
- import {
86
- agentSkillToolPolicy
87
- } from "../chunk-GOXNB3AO.js";
88
- import {
89
- DEFAULT_AUTONOMY_LEVEL
90
- } from "../chunk-HWUFFB6L.js";
91
- import {
92
- RESULT_CONTRACT_REPAIR_LIMIT,
93
- parseResultContract,
94
- resultContractRepairMessages,
95
- validateResultContract
96
- } from "../chunk-DJNLUABN.js";
97
- import {
98
- classify
99
- } from "../chunk-AOCYTWAV.js";
100
- import {
101
- protectedArtifactMutationBlockReason
102
- } from "../chunk-MV3K5QF2.js";
103
- import "../chunk-UL3WSD3F.js";
104
- import "../chunk-ODFEOB4F.js";
105
- import "../chunk-OZNBF4L3.js";
106
- import "../chunk-4BJ5BYCE.js";
107
- import "../chunk-ECH6PKUQ.js";
108
105
  import "../chunk-6XLNIQDB.js";
109
106
  import "../chunk-5B2AEOW5.js";
110
107
  import {
@@ -114,56 +111,62 @@ import "../chunk-65DEGPJ6.js";
114
111
  import {
115
112
  FileKnowledgeBase,
116
113
  applyModelCapabilityPatch,
117
- getRuntimeRegistry,
118
114
  loadPluginRuntimes,
119
- registerBuiltinRuntimes,
120
115
  registerClioApiProviders,
121
116
  resolveModelRuntimeCapabilitiesForModel,
122
117
  resolveProviderKnowledgeBaseRoots,
123
118
  setGlobalDefaultMaxOutputTokens,
124
119
  setProtectedModelsProvider,
125
120
  setResidencyNoticeSink
126
- } from "../chunk-VQNODYQ4.js";
121
+ } from "../chunk-MIX5N5AC.js";
127
122
  import "../chunk-CFGTUFWB.js";
128
- import "../chunk-JRIO5UD2.js";
123
+ import "../chunk-LU7P4LHA.js";
124
+ import "../chunk-AEYBF3TB.js";
129
125
  import "../chunk-IHXBNWMM.js";
130
- import "../chunk-GEYXPTRF.js";
131
- import {
132
- readSettings,
133
- validateSettingsFile
134
- } from "../chunk-UHXRNZ2J.js";
135
- import "../chunk-4DGYLA73.js";
136
- import "../chunk-LL4KHSZI.js";
137
- import "../chunk-SST6Z5JA.js";
138
- import "../chunk-FO5ZOVUY.js";
139
- import "../chunk-R346GLFC.js";
140
126
  import {
141
127
  setGitCommitAttributionEnabled,
142
128
  withManagedGitCommitAttributionEnvironment
143
- } from "../chunk-D4MDIG46.js";
129
+ } from "../chunk-B5CSFE7B.js";
130
+ import {
131
+ readSettings,
132
+ validateSettingsFile
133
+ } from "../chunk-PNY46YEY.js";
144
134
  import "../chunk-FQ4SKYE4.js";
135
+ import "../chunk-ZGVHUX3M.js";
136
+ import "../chunk-RKRLDWD3.js";
137
+ import "../chunk-KV2AOLDF.js";
138
+ import {
139
+ AI_AGENT_NAME
140
+ } from "../chunk-6EJMN2Y3.js";
141
+ import {
142
+ readClioVersion
143
+ } from "../chunk-IWHMRKLL.js";
144
+ import {
145
+ registerBuiltinRuntimes
146
+ } from "../chunk-JEQQR47K.js";
147
+ import "../chunk-XDOQXGFO.js";
148
+ import "../chunk-LL4KHSZI.js";
145
149
  import {
146
150
  configureGuardrails,
147
151
  isWorkerToolCallCapExceededReason,
148
152
  isWorkerToolCallCapSynthesisReason
149
153
  } from "../chunk-4ZG3XFUR.js";
154
+ import "../chunk-774ILSRL.js";
155
+ import "../chunk-EQ63NRB7.js";
156
+ import "../chunk-SST6Z5JA.js";
150
157
  import "../chunk-IKCO5N3L.js";
151
158
  import "../chunk-3I7MS7N2.js";
152
- import {
153
- registerFauxFromEnv
154
- } from "../chunk-FCSXB6T2.js";
155
159
  import "../chunk-APJ265NV.js";
156
- import "../chunk-YXLYO42X.js";
157
160
  import "../chunk-BNAZZHFG.js";
158
- import {
159
- AI_AGENT_NAME
160
- } from "../chunk-6EJMN2Y3.js";
161
161
  import "../chunk-WEPFGWHJ.js";
162
- import "../chunk-ZGVHUX3M.js";
163
- import "../chunk-OB5HIGJY.js";
164
162
  import {
165
- readClioVersion
166
- } from "../chunk-IWHMRKLL.js";
163
+ getRuntimeRegistry
164
+ } from "../chunk-NUGM5KR6.js";
165
+ import "../chunk-VO2LKSTM.js";
166
+ import {
167
+ registerFauxFromEnv
168
+ } from "../chunk-UOSL25KY.js";
169
+ import "../chunk-YXLYO42X.js";
167
170
  import {
168
171
  init_esm_shims
169
172
  } from "../chunk-3R73A4XB.js";
@@ -943,7 +946,7 @@ function resolveWorkerRuntimeBudget(input) {
943
946
  }
944
947
  function startWorkerRun(input, emit) {
945
948
  assertResponseSchemaRuntime(input);
946
- if (input.resultContract !== void 0) parseResultContract(input.resultContract, "WorkerSpec.resultContract");
949
+ if (input.resultContract !== void 0) parseWorkerResultContract(input.resultContract, "WorkerSpec.resultContract");
947
950
  if (input.runtime.id === "claude-sdk") {
948
951
  return startClaudeSdkWorkerRun(input, emit);
949
952
  }
@@ -1,7 +1,7 @@
1
1
  # ALCF Inference Provider
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive target configurator and Globus OAuth flow diagram is located at [docs/html/alcf_blueprint.html](html/alcf_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive target configurator and Globus OAuth flow diagram is located at [docs/html/alcf_blueprint.html](html/alcf_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio can use Argonne's ALCF inference gateway as an OpenAI-compatible target
7
7
  backed by Globus OAuth. The runtime id is `alcf`; each configured target points
@@ -1,7 +1,7 @@
1
1
  # Clio Coder Architecture and Boundaries
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/architecture_blueprint.html](html/architecture_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/architecture_blueprint.html](html/architecture_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder is an experimental, terminal-first coding harness for the CLIO ecosystem. CLIO stands for Context Layer for Input/Output; the project is named for the Greek muse of history and developed by the Gnosis Research Center at Illinois Tech. Its architecture favors small, auditable subsystems over a single monolithic agent loop: CLI entry points, the interactive TUI, provider/runtime code, worker subprocesses, tools, and feature domains are kept separate so local-model support and scientific-software workflows can evolve without collapsing safety boundaries.
7
7
 
@@ -1,6 +1,6 @@
1
1
  # Artifact Versions & Serialization Contracts
2
2
 
3
- This document is the canonical registry of all versioned file formats, serialized data structures, integrity digests, and migration rules across Clio Coder in `v0.3.7`.
3
+ This document is the canonical registry of all versioned file formats, serialized data structures, integrity digests, and migration rules across Clio Coder in `v0.3.9`.
4
4
 
5
5
  ---
6
6
 
@@ -10,7 +10,7 @@ Clio Coder strictly versions every persistent or network-transported data struct
10
10
 
11
11
  | Artifact / Subsystem | Current Version | Symbol / Type & Source Location | Persisted Path / Wire Location | Schema Semantics & Version Differences | Mismatch Handling |
12
12
  | :--- | :--- | :--- | :--- | :--- | :--- |
13
- | **Run Receipt** | `19` | `RUN_RECEIPT_INTEGRITY_VERSION = 19`<br>`src/domains/dispatch/receipt-integrity.ts:13` | `<stateDir>/receipts/<runId>.json` | Cryptographically sealed run record. Version 19 covers all base provenance fields, routing intent, quality labels, `validationGrounding`, `capabilityMismatch`, council provenance, and fleet gate provenance. | Fail-closed. Incompatible receipts fail verification and are never read as evidence. |
13
+ | **Run Receipt** | `20` | `RUN_RECEIPT_INTEGRITY_VERSION = 20`<br>`src/domains/dispatch/receipt-integrity.ts:13` | `<stateDir>/receipts/<runId>.json` | Cryptographically sealed run record. Version 20 adds `pathProvenance` on dispatch intent and the resolved `pathScope`, over the v19 base of provenance fields, routing intent, quality labels, `validationGrounding`, `capabilityMismatch`, council provenance, and fleet gate provenance. | Fail-closed. A receipt below v20 is reported as retired rather than invalid, is never read as evidence, and is never migrated; a malformed or tampered v20 receipt fails verification. |
14
14
  | **Session Ledger** | `3` | `CURRENT_SESSION_FORMAT_VERSION = 3`<br>`src/engine/session.ts:66` | `<stateDir>/sessions/<cwdHash>/<sessionId>/` (`meta.json`, `current.jsonl`, `tree.json`) | Append-only ledger format with UUIDv7 turn IDs, session header line, and tree graph linkage. | Automated migration via `src/domains/session/migrations/` on `/resume`. Earlier unmigratable versions rejected. |
15
15
  | **Worker Spec** | `3` | `WORKER_SPEC_VERSION = 3`<br>`src/worker/spec-contract.ts:22` | Subprocess `stdin` control plane JSON payload | Worker invocation parameters, tool surface profile, and execution bounds. | Fail-closed preflight rejection before worker activation. |
16
16
  | **Worker Runtime Descriptor** | `2` | `WORKER_RUNTIME_DESCRIPTOR_VERSION = 2`<br>`src/worker/spec-contract.ts:23` | Worker attestation descriptor payload | Attestation descriptor for worker runtime environment and hardware facts. | Attestation mismatch causes immediate process termination. |
@@ -18,10 +18,16 @@ Clio Coder strictly versions every persistent or network-transported data struct
18
18
  | **Fleet Contract** | `1 \| 2 \| 3 \| 4 \| 5` (Current: `5`) | `FleetContractVersion = 1 \| 2 \| 3 \| 4 \| 5`<br>`FLEET_WRITE_BOUNDARY_VERSION = 4`<br>`FLEET_DYNAMIC_STEP_VERSION = 5`<br>`src/domains/agents/fleet-contract.ts` | `.clio-coder/fleets/<name>.yaml`, `.clio-coder/fleets/<name>.yml`, or built-in recipes | Multi-agent workflow contract. v1 is agent-only; v2 adds deterministic code steps; v3 adds bounded loops and commit steps; v4 adds declared per-step write boundaries; v5 adds plan steps, gate steps, per-step target or profile routing, and the single-writer declaration. | Reader refuses contracts whose version features it does not support. |
19
19
  | **Execution Plan** | `4` | `version: 4` in `interface ExecutionPlan`<br>`src/domains/dispatch/execution-plan.ts:98` | Statically compiled DAG representation in dispatch memory and receipts | Statically unrolled, deterministically hashed execution plan. v4 adds bounded loop nodes, verification staleness tracking, and commit nodes. | Preflight validation rejects unsupported plan versions. |
20
20
  | **Eval Artifact** | `4` | `version: 4` in `interface EvalArtifactV4`<br>`src/domains/eval/schema/artifact.ts:51-52` | `<stateDir>/evals/<evalId>.json` | Stored eval results with suite provenance, matrix parameters, and itemized metric outcomes. Note: `EVAL_ARTIFACT_VERSION = 1` in `src/domains/eval/types.ts:2` is legacy/dead code. | Incompatible eval artifacts are rejected during `clio-coder eval report` and `compare`. |
21
+ | **Prompt Manifest** | `2` | `PROMPT_MANIFEST_VERSION = 2`<br>`src/domains/session/prompt-manifest.ts:30` | `<stateDir>/sessions/<cwdHash>/<sessionId>/prompt-manifest.jsonl` | Per-session record of the compiled system prompt: fragment ids, relative paths, content hashes, section token estimates, and the composition hash. Version 2 is the stable-prefix-first ordering with one `# Memory` header and records `contextWindowSource` beside the window the prompt states (#249). | Additive. A record without a `version` field predates the field and reads as version 1, so a 0.3.8 manifest still parses; the version explains the single `promptRecompiled` entry a resumed session's first compile writes. |
22
+ | **Eval Verdict Envelope** | `clio.eval.verdict.v1` | `EVAL_VERDICT_SCHEMA_V1`<br>`src/domains/eval/schema/verdict.ts:1` | Inside the Eval Artifact at `<stateDir>/evals/<evalId>.json` | Fail-closed verdict identity with ledger- and receipt-sourced performance metrics, per-scenario pass and distribution aggregates, and serving-configuration provenance (#252). One pass decision: the code grader's outcome is part of `result.pass`. | Fail-closed. `eval compare` refuses serving-configuration drift unless explicitly allowed; an artifact without a parseable verdict envelope is rejected rather than read as passing. |
23
+ | **Behavioral Scenario & Result** | `clio.eval.scenario.v1`, `clio.eval.behavior.v1` | `EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1`, `EVAL_BEHAVIOR_SCHEMA_V1`<br>`src/domains/eval/schema/behavioral.ts:4-5` | Suite v2 task declarations; additive sibling inside the Eval Artifact | Versioned behavioral contract: bounded expected and forbidden rules across tool choice, exploration, delegation, safety comprehension, claim grounding, denied-tool recovery, completion behavior, and task correctness, with deterministic judge inputs canonicalized from transcript, tool, receipt, and grader facts (#156). References the unchanged `clio.eval.verdict.v1` identity. | Fail-closed. `unknown`, `unmeasured`, `behavioral_failure`, and `infrastructure_failure` stay distinct; a malformed, partial, contradictory, or cross-linked verdict cannot parse as a pass. Existing artifact readers are unaffected because the sibling is additive. |
24
+ | **Behavioral Metrics Projection** | `clio.eval.behavior.metrics.v1` | `EVAL_BEHAVIOR_METRICS_SCHEMA_V1`<br>`src/domains/eval/schema/behavioral-metrics.ts:4` | Additive role- and target/model-bound projection inside the Eval Artifact | Ten sourced metric families (correctness, safety, label violations, tool-call efficiency, unnecessary exploration, delegation quality, unsupported claims, tokens, latency, cost, and repeat variability) with coverage, min/max, p90, population variance, and standard deviation per distribution (#161). | Unmeasured observations stay typed `null` and never become zero violations. A baseline hard metric that becomes unmeasured fails the comparison closed, and `--metric` filtering cannot hide a hard failure. |
25
+ | **Execution Envelope** | `clio.eval.execution-envelope.v1` | `EVAL_EXECUTION_ENVELOPE_SCHEMA_V1`<br>`src/domains/eval/schema/execution-envelope.ts:1` | On every new behavioral result inside the Eval Artifact | Strictly parsed binding of prompt fragment ids, versions, and content hashes, composition hash, recipe identity and content hash, target, wire model, runtime, thinking level, tool signature, autonomy, policy hashes, project-context provenance, and corpus id and version (#164). Suites declare which matrix dimensions may vary. | Fail-closed. Comparisons mark rows incomparable on any undeclared envelope drift, refuse one-sided envelopes and within-run variance, and name every prompt- or recipe-affected corpus result. |
21
26
  | **Trace Database** | `1` | `TRACE_SCHEMA_VERSION = 1`<br>`src/domains/observability/trace-store.ts:23` | `<stateDir>/trace.sqlite` (`meta` table `schema_version`) | Schema version for the 7 SQLite trace mirror tables (`runs`, `phases`, `events`, `envelopes`, `gate_results`, `agent_sessions`, `processes`). | Log warning (`[clio:trace]`), trace writing degrades without failing the parent run. |
22
27
  | **Capacity State File** | `2` | `version: 2` in `interface CapacityStateFile`<br>`src/domains/dispatch/capacity-lease.ts:40` | `<stateDir>/dispatch-admission.json` | Active capacity leases, drain status, and cross-process lock state. | Corrupted or unparseable state file causes admission to fail closed. |
23
28
  | **Protected Artifact Journal** | `1` | `version: 1` in `interface PendingProtectedArtifactRecord`<br>`src/domains/session/protected-artifact-journal.ts:22` | `<stateDir>/protected-artifact-pending/<key>/<id>.json` | Write-ahead durability records for pending protected artifacts. | Leftover records reconciled during session initialization. |
24
29
  | **Fleet Run Record** | `1` | `version: 1` in `interface FleetRunRecord`<br>`src/domains/dispatch/fleet-run.ts` | `<stateDir>/fleet-runs/<runId>.json` | Durable record of one fleet run: contract name, plan hash, static step ids and steps, `--var` values, replayed and settled step results, and the delegation plan hash a `kind: plan` step produced. Read by `fleet run --resume`. | Resume refuses a changed plan hash with a per-step diff and refuses differing `--var` values. |
30
+ | **Durable Assignment Store** | `1` | `version: 1` in `interface AssignmentStoreFile`<br>`DurableAssignmentRecord`<br>`src/domains/dispatch/assignment-store.ts` | `<stateDir>/assignments.json` | Machine-wide logical-dispatch records: assignment id, attempt ids, terminal run id, status, optional fleet verdict owner, and—while running—`processOwner {pid, processBirthToken, acquiredAt}`. The owner is cleared on a true terminal transition. | A live sibling owner keeps the row running; a genuinely dead or legacy ownerless row is reconciled. An unsupported or unreadable store is treated as empty, and malformed records are ignored. |
25
31
  | **Checkout Writer Lease** | `1` | `version: 1` in `interface CheckoutWriterLeaseRecord`<br>`src/domains/dispatch/checkout-writer-lease.ts` | `<stateDir>/checkout-writer-leases/<key>.json` (key derived from the canonical checkout path) | Cross-process single-writer lease: checkout path, pid, process birth token, acquisition time. | A live sibling holder is refused with `checkout_writer_lease_held`; a dead owner is reclaimed; a malformed record is treated as absent. |
26
32
  | **Out-of-turn Usage Ledger** | unversioned JSONL | `OutOfTurnUsageRow`<br>`src/domains/observability/out-of-turn-usage.ts` | `<stateDir>/usage/out-of-turn.jsonl` | One row per priced `/btw` or `/handoff` call: label, session id, repo identity, timestamp, target, attributed model, provider usage. Bounded ring of `MAX_OUT_OF_TURN_USAGE_ROWS = 1000`, rewritten atomically under the state-file lock. | Unparseable rows are skipped and counted by `usage report`; the session ledger is never affected. |
27
33
  | **Library Pins** | unversioned YAML map | `readLibraryPins`<br>`src/domains/resources/library.ts` | `<configDir>/library-pins.yaml` | Typed ref (`skill:x`, `agent:y`, `prompt:p`, `fleet:z`) to `{sha256, sourceUrl}` for every resource `library add` or the Skills Hub installed. | A non-map document reads as empty; an entry whose installed file is missing is reported as available, not installed. |
@@ -30,7 +36,7 @@ Clio Coder strictly versions every persistent or network-transported data struct
30
36
 
31
37
  ## 2. Integrity Verification Contracts
32
38
 
33
- ### Receipt Integrity (Version 19)
39
+ ### Receipt Integrity (Version 20)
34
40
 
35
41
  Receipt integrity authenticates that a sealed receipt matches its ledger envelope without modification. Verification reproduces the canonical JSON serialization and computes the SHA-256 digest:
36
42
 
@@ -42,9 +48,9 @@ export function computeReceiptDigest(receipt: RunReceiptV15): string {
42
48
  ```
43
49
 
44
50
  Receipt verification checks:
45
- 1. `integrity.version === 19`.
51
+ 1. `integrity.version === 20`. A receipt sealed at a lower version is reported as retired rather than invalid: it is intact, but is not read as evidence and is never migrated.
46
52
  2. Calculated SHA-256 matches `integrity.digest`.
47
- 3. All optional fields present in the schema (`validationGrounding`, `capabilityMismatch`, `steering`, `gate`, `fleetGate`, `council`, `plan`, `briefing`) conform to the strict v19 specification.
53
+ 3. All optional fields present in the schema (`validationGrounding`, `capabilityMismatch`, `steering`, `gate`, `fleetGate`, `council`, `plan`, `briefing`) conform to the strict v20 specification.
48
54
 
49
55
  ---
50
56
 
@@ -3,7 +3,7 @@
3
3
  Clio Coder dispatches focused fleet agents from Markdown recipes. Recipes are data files, not hidden code plugins: YAML frontmatter declares identity, mode, tools, optional target/model hints, and thinking level; the Markdown body is the agent instruction text.
4
4
 
5
5
  > [!TIP]
6
- > **Interactive Spec Available:** An interactive dashboard for the agent registry and dispatch admission check gates is located at [docs/html/agents_blueprint.html](html/agents_blueprint.html) (Version: 0.3.7).
6
+ > **Interactive Spec Available:** An interactive dashboard for the agent registry and dispatch admission check gates is located at [docs/html/agents_blueprint.html](html/agents_blueprint.html) (Version: 0.3.9).
7
7
 
8
8
  The source of truth is `src/domains/agents/**`. Clio's agent dispatch engine and execution boundaries are built upon the [@earendil-works/pi-agent-core](https://www.npmjs.com/package/@earendil-works/pi-agent-core) library.
9
9
 
@@ -8,19 +8,29 @@ Source implementations: `src/domains/scheduling/` and `src/domains/dispatch/capa
8
8
 
9
9
  ## 1. Capacity Model & Admission Invariants
10
10
 
11
- Fleet dispatch manages compute resources across local and remote execution nodes as a unified capacity pool. Dispatched workers must acquire a durable capacity lease before they are spawned.
11
+ Fleet dispatch manages compute resources across local and remote execution nodes as a unified capacity pool. Dispatched workers must acquire a durable capacity lease before they are spawned. Admission checks global, node, and inference-endpoint limits independently.
12
12
 
13
13
  ```mermaid
14
14
  graph TD
15
15
  req[Dispatch Request] --> lock[Acquire Cross-Process Lock: dispatch-admission.json.lock]
16
16
  lock --> reap[Reap Expired Leases & Dead PIDs]
17
- reap --> check[Check Capacity Limits: global & per-node]
17
+ reap --> check[Check Capacity Limits: global, per-node, and per-endpoint]
18
18
  check -->|Within Limits| grant[Grant Capacity Lease & Write State]
19
19
  check -->|Limits Exceeded| queue[Queue / Reject Request]
20
20
  grant --> unlock[Release Lock]
21
21
  unlock --> spawn[Spawn Worker Process]
22
22
  ```
23
23
 
24
+ | Dimension | Identity | Limit resolution |
25
+ | :--- | :--- | :--- |
26
+ | Global | All dispatches using the state directory. | `budget.concurrency: auto` remains four. |
27
+ | Node | The local node or one configured fleet node. | The configured node limit applies. An unset local node cap remains unbounded. |
28
+ | Inference endpoint | A normalized scheme, host, port, and base path. | A target's `maxConcurrentRequests` override wins, followed by cached probe discovery. Other local-native targets default to one slot. vLLM and SGLang remain unbounded. |
29
+
30
+ The conventional final `/v1` mount and a trailing slash normalize to the same endpoint. Host aliases are not collapsed because Clio cannot prove they address the same server. For example, `http://localhost:8080/` and `http://127.0.0.1:8080/v1` remain distinct, while two target descriptors that use the same normalized URL share one endpoint limit.
31
+
32
+ llama.cpp discovery reads `total_slots` from cached probe results. A router can expose the selected worker's value from `/props?model=<id>` even when router `/props` has no slot count. The selected model's `/v1/models` argv supplies a `--parallel` fallback. LM Studio defaults to one slot when its REST response supplies no concurrency fact. Ollama defaults to one unless `OLLAMA_NUM_PARALLEL` is visible to the local process.
33
+
24
34
  ### State Storage & Format
25
35
 
26
36
  All capacity state is stored in a single durable JSON file:
@@ -49,6 +59,7 @@ export interface CapacityLease {
49
59
  leaseId: string; // Unique lease identifier
50
60
  assignmentId: string; // Owning dispatch assignment ID
51
61
  nodeId: string; // Execution node identifier ("local" or remote ID)
62
+ endpointKey?: string; // Canonical inference endpoint identifier
52
63
  ownerPid: number; // Process ID of the orchestrator/worker owner
53
64
  processBirthToken: string; // OS-level token preventing PID reuse collisions
54
65
  acquiredAt: string; // ISO-8601 acquisition timestamp
@@ -59,6 +70,16 @@ export interface CapacityLease {
59
70
  }
60
71
  ```
61
72
 
73
+ The orchestrator's active model stream is registered in memory against the same endpoint key, so its own turn consumes one endpoint slot before a worker is admitted. This foreground count is not written to `dispatch-admission.json`; process exit releases it. Durable leases and held reservation members carry `endpointKey`, and held members count their peak per wave for the endpoint just as they do for a node.
74
+
75
+ Execution-plan waves also honor the endpoint bound. A plan with four available worker positions targeting one two-slot server packs at most two of them into a wave, or one when the orchestrator already holds the other slot. Endpoint saturation is refused rather than queued, because an endpoint-specific request queue would hold a dispatch open behind a stream whose length nobody knows. The refusal names the endpoint, both slot counts, why one slot is already gone, and the two moves that actually free capacity:
76
+
77
+ ```text
78
+ dispatch: admission denied: endpoint '192.168.86.141:8080' capacity reached (1/1 slots): the orchestrator's own turn holds one; collect in-flight runs or point workers at a second server
79
+ ```
80
+
81
+ That is the exact text on all three paths that can refuse for this reason: lease acquisition (`src/domains/dispatch/capacity-lease.ts`), the admission gate (`src/domains/dispatch/admission.ts`), and reservation preflight (`src/domains/dispatch/reservation-store.ts`). The `1/1` above is the common local case rather than an example: a llama.cpp router started with `--parallel 1` discovers one slot, so any dispatch raised while the orchestrator is streaming is refused before a worker process starts.
82
+
62
83
  ### Constants & Operational Bounds
63
84
 
64
85
  | Constant | Value | Description | Source Reference |
@@ -1,7 +1,7 @@
1
1
  # Commands and Modes
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/commands_blueprint.html](html/commands_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/commands_blueprint.html](html/commands_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
 
7
7
  Clio Coder is a terminal-first alpha harness. This page keeps the command
@@ -131,7 +131,7 @@ Example:
131
131
  clio-coder run \
132
132
  "Find the test command and summarize the project structure." \
133
133
  --target local-lmstudio \
134
- --model your-model-id
134
+ --model qwen3.8-27b
135
135
  ```
136
136
 
137
137
  ## Interactive Slash Commands
@@ -1,7 +1,7 @@
1
1
  # Configuration, Targets, Runtimes, and Auth
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive configuration validator, target resolver, and CLI command generator is located at [docs/html/configuration_blueprint.html](html/configuration_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive configuration validator, target resolver, and CLI command generator is located at [docs/html/configuration_blueprint.html](html/configuration_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder is target-first: chat and fleet dispatch resolve through configured targets in `settings.yaml`, not through provider-specific ad hoc flags. Chat and print targets are HTTP and native engine-backed runtimes. Fleet dispatch can also target the sanctioned Claude Code subscription runtimes described below.
7
7
 
@@ -82,11 +82,16 @@ clio-coder configure \
82
82
  --id local-lmstudio \
83
83
  --runtime lmstudio \
84
84
  --url http://127.0.0.1:1234 \
85
- --model your-model-id \
85
+ --model qwen3.8-27b \
86
86
  --set-orchestrator \
87
87
  --set-fleet-default
88
88
  ```
89
89
 
90
+ `--model` must be an id the server advertises. `configure` fetches the
91
+ server's model list and refuses an id that is not on it, printing the ids it
92
+ found and which of them are loaded; `--force` saves the target anyway. Replace
93
+ `qwen3.8-27b` with an id from `lms ls` (LM Studio) or your server's model list.
94
+
90
95
  Use the id you chose, probe it, then launch the TUI:
91
96
 
92
97
  ```bash
@@ -134,6 +139,9 @@ targets:
134
139
  runtime: lmstudio
135
140
  url: http://127.0.0.1:1234
136
141
  defaultModel: your-model-id
142
+ # Optional. Request slots this inference endpoint can serve at once.
143
+ # Overrides live discovery; omit it and Clio reads the server's own count.
144
+ maxConcurrentRequests: 2
137
145
  capabilities:
138
146
  reasoning: true # optional; only if your model/runtime supports it
139
147
  lmstudio:
@@ -168,7 +176,7 @@ memory:
168
176
  everyNTools: 10
169
177
  windowSteps: 8
170
178
  maxTokens: 400
171
- timeoutMs: 180000 # shipped operator default in src/core/defaults.ts; library fallback in task-memory-policy.ts is 20000 ms
179
+ timeoutMs: 30000 # shipped operator default in src/core/defaults.ts; the library fallback in task-memory-policy.ts is the same value
172
180
 
173
181
  workers:
174
182
  default:
@@ -251,6 +259,9 @@ context:
251
259
  protectLastTurns: 6
252
260
  minEvictableTokens: 200
253
261
 
262
+ prewarm:
263
+ enabled: true # send the next turn's prefix early; local-native targets only
264
+
254
265
  retry:
255
266
  enabled: true
256
267
  maxRetries: 3
@@ -268,6 +279,26 @@ guardrails:
268
279
 
269
280
  Target capability overrides may include `chat`, `tools`, `toolCallFormat`, `reasoning`, `thinkingFormat`, `structuredOutputs`, `vision`, `audio`, `embeddings`, `rerank`, `fim`, `contextWindow`, and `maxTokens`.
270
281
 
282
+ ### `maxConcurrentRequests`
283
+
284
+ `maxConcurrentRequests` is a per-target integer of at least 1, validated with the rest of the target block, and it is the operator's override for how many requests the inference endpoint behind that target can serve at once. It is not a settings-file default and has no shipped value, so it does not appear in the settings inventory below.
285
+
286
+ Set it only when discovery is wrong. Clio resolves the limit in this order: this override; then a `parallelSlots` count cached on the target's probe result; then one slot for any other `local-native` runtime; then no bound at all for a cloud runtime, vLLM, or SGLang. llama.cpp discovery reads `total_slots` from the router's `/props`, falls back to the selected worker's `/props?model=<id>` when the router reports none, and falls back again to the `--parallel` argv on the selected `/v1/models` entry. LM Studio reads `config.parallel` off the loaded instance and otherwise reports one; Ollama reads `OLLAMA_NUM_PARALLEL` from the environment the Clio process can see and otherwise reports one.
287
+
288
+ The limit is keyed on the endpoint rather than the target, so two targets pointed at the same normalized URL share it. Raising it above what the server will actually serve does not create capacity; it removes the refusal that would have told you the server was full. See [capacity-and-scheduling.md](capacity-and-scheduling.md) for the admission model and the exact denial text.
289
+
290
+ ### The `local-native` tier
291
+
292
+ Three behaviors in this release are gated on a runtime's tier being `local-native` rather than on a target id or a server name, so it is worth stating what the tier is. It is a property of the runtime descriptor (`RuntimeTier` in `src/domains/providers/types/runtime-descriptor.ts`), and the runtimes that carry it are `llamacpp` with its completion, embedding, rerank, and Anthropic-surface variants, `lmstudio`, `ollama-native`, `vllm`, `sglang`, and the two `lemonade` surfaces. Everything else is `cloud`, `protocol`, or `subscription`.
293
+
294
+ The tier means "an inference server the operator runs, whose prefix cache and resident model Clio's own behavior can displace." That is what the three gates are actually asking:
295
+
296
+ - **Pre-warm** runs only here, whatever `prewarm.enabled` says, because a cloud provider bills the request and caches on its own schedule. The check is made twice, once from configuration before any runtime is resolved and once against the resolved runtime, so an unreachable target does not pay for a capability probe at boot just to be told no.
297
+ - **Endpoint capacity** defaults to one slot here when discovery reports nothing, and to unbounded elsewhere. vLLM and SGLang are the deliberate exceptions inside the tier: both serve genuinely concurrent requests, so an undiscovered limit is left unbounded rather than guessed at one.
298
+ - **Five of the eight expected-cold reasons** are stamped only here, because a single-slot local cache is the only one an interleaved run actually displaces. The other three moved the prompt bytes themselves and are stamped on every tier. The full split is in [context-engine.md](context-engine.md#cache-divergence-honesty).
299
+
300
+ The tool-prose-loop detector is keyed on the same tier, for the same reason: narrating a tool call instead of emitting one is a behavior of open-weight models served locally, and a list of server names would have left an Ollama or vLLM run with no cutoff at all.
301
+
271
302
  ### LM Studio transport and settings
272
303
 
273
304
  The canonical runtime id is `lmstudio`. The former `lmstudio-native` id remains an accepted alias,
@@ -637,6 +668,7 @@ Every one of these has an environment override for a single process; see [enviro
637
668
  | `context.workingSet.target` | `0.6` | number greater than 0 and less than 1 | next turn |
638
669
  | `context.workingSet.protectLastTurns` | `6` | integer ≥ 1 | next turn |
639
670
  | `context.workingSet.minEvictableTokens` | `200` | integer ≥ 0 | next turn |
671
+ | `prewarm.enabled` | `true` | boolean | next turn |
640
672
  | `defaults.maxTokens` | `32768` | integer ≥ 0 | next turn |
641
673
  | `budget.sessionCeilingUsd` | `5` | number ≥ 0 | immediately |
642
674
  | `budget.concurrency` | `auto` | `auto` or integer ≥ 1 | next dispatch |
@@ -658,7 +690,7 @@ Generic provider and transport errors are classified by transient retry rules, i
658
690
  | `memory.intervention.everyNTools` | `10` | integer ≥ 2 | next turn |
659
691
  | `memory.intervention.windowSteps` | `8` | integer ≥ 1 | next turn |
660
692
  | `memory.intervention.maxTokens` | `400` | integer ≥ 1 | next turn |
661
- | `memory.intervention.timeoutMs` | `180000` | integer ≥ 1 | next turn |
693
+ | `memory.intervention.timeoutMs` | `30000` | integer ≥ 1 | next turn |
662
694
 
663
695
  ### Turn-end watchdog
664
696
 
@@ -768,7 +800,7 @@ clio-coder configure \
768
800
  --id local-llamacpp \
769
801
  --runtime llamacpp \
770
802
  --url http://127.0.0.1:8080 \
771
- --model your-model-id \
803
+ --model qwen3.8-27b \
772
804
  --set-orchestrator \
773
805
  --set-fleet-default
774
806
  ```
@@ -1,7 +1,7 @@
1
1
  # Context Engine
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/context_blueprint.html](html/context_blueprint.html) (Version: 0.3.7).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/context_blueprint.html](html/context_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder tracks context pressure, records per-turn snapshots, and protects the provider context with bounded tool results plus single-threshold compaction.
7
7
 
@@ -15,6 +15,8 @@ Each target has a declared, desired, and effective context window. The effective
15
15
 
16
16
  The loaded window outranks the declared one because it is the only figure describing what the backend will serve. LM Studio routinely opens a model well below its `max_context_length`, and a run planned against the larger number overruns the server before compaction ever fires. Discovery carries that number per model in `discoveredModelStates[<model>].contextLength`, and the residency notice reads the same entry, so a model Clio is budgeting a loaded window for is never announced as absent.
17
17
 
18
+ A resumed session carries the loaded window it already recorded. A resume re-resolves its target before discovery has reported what the backend has open, so the first turn used to budget against the probed figure, which on a multi-slot or multi-copy backend can be several times the real headroom, and corrected a turn later. `lastLoadedContextWindow` reads the last `loaded` window the session's own `context-snapshots.jsonl` recorded for the same target and model and hands it to resolution as `knownLoadedContextWindow`. It is used only when live discovery reports nothing, and it is scoped to that target and model, so a different selection re-probes and a model reloaded at a new size corrects as soon as discovery names the live window.
19
+
18
20
  Local-native runtimes use a recommended minimum desired window of 128,000 tokens. If the live model reports a smaller loaded context window, Clio re-resolves the target so accounting uses the actual ceiling.
19
21
 
20
22
  The `/context` overlay states which layer answered, next to the token total: `loaded`, `probed`, `configured`, `declared`, or `assumed`.
@@ -27,13 +29,17 @@ The estimator in `context-accounting.ts` uses a four-characters-per-token family
27
29
 
28
30
  At submit time, Clio captures a context snapshot and persists a slim JSONL record under the session directory as `context-snapshots.jsonl`. The slim record keeps token counts, segment metadata, signatures, and hashes, not the heavy prompt or transcript text. When provider usage arrives, `reconcileSnapshot` folds actual input and output counts back into the ledger.
29
31
 
32
+ Every snapshot records the divergence between the two accountings. `estimatedTokens` is the chars/4 prompt-side total the snapshot was captured with and is never rewritten by a reconcile; `reconciledTokens` is the provider's own prompt count for the call, with cached prompt tokens folded back in; `divergenceRatio` is the second over the first. A ratio above 1 means the estimator is under-counting what the backend charges for the same messages.
33
+
34
+ The reconciled figure is not only a display value. Once a provider has answered, the compaction verdict budgets against `max(estimate, reconciled + estimate of everything appended since)`, at all three evaluation points: the pre-submit trigger, the post-tool continuation guard, and the preflight overflow check. The estimate stays a floor because it prices material the attested call never saw; the provider count can only raise the figure, never lower it. A working-set projection subtracts the tokens the eviction planner priced out and re-anchors on the projected message list rather than discarding the attestation, so post-eviction accounting is still provider-anchored. A summary compaction rewrites the conversation the attestation described, so it drops the anchor and the next call re-establishes it.
35
+
30
36
  Session metadata enforces session format version 4 (`CURRENT_SESSION_FORMAT_VERSION = 4`). Version 4 is additive: it adds the `contextEviction` and `contextRecall` records and changes no existing entry. A version 3 session therefore migrates to 4 in place when Clio opens it, and no entry is rewritten. Only a session written by a newer build is refused, with an error naming the version it read and pointing at upgrading. The bump is one-way for the operator: a 0.3.3 binary cannot open a session this release wrote.
31
37
 
32
- The `/context` overlay and footer meter read the same ledger categories: `system`, `tools`, `agents`, `skills`, `memory`, `project`, `messages`, `pending`, `reserve`, `free`, and `streaming`.
38
+ The `/context` overlay and footer meter read the same ledger categories in display order: `system`, `tools`, `agents`, `skills`, `memory`, `project`, `messages`, `pending`, `streaming`, `free`, and `reserve`.
33
39
 
34
40
  ## Single-threshold compaction
35
41
 
36
- Auto-compaction is controlled by one pressure threshold. Pressure is `estimated_tokens / context_window`. The default threshold is `0.8`.
42
+ Auto-compaction is controlled by one pressure threshold. Pressure is `budgeted_tokens / context_window`, where the budgeted figure is the reconciled total when the provider has attested one and the chars/4 estimate otherwise. The default threshold is `0.8`.
37
43
 
38
44
  Crossing that threshold engages three mechanisms in a fixed order. The first two are cheap, reversible, and call no model. Only the third rewrites what the session says about itself.
39
45
 
@@ -81,14 +87,64 @@ Every provider Clio targets caches by exact prefix. Anthropic hashes the cumulat
81
87
 
82
88
  The procedural replay target sweep measured 0.4, 0.5, 0.6, and an exhaustive rung-6 stop over 24 traces. Target 0.4 and exhaustive selection converged because un-evictable residue exhausted the candidate pool. Against 0.6, target 0.4 cut cold-prefix tokens by 2.8% at 64k and 7.3% at 128k, with no summary reduction and a 0.00072 reduction in retention covered at 128k. That is below the 10% cache-saving threshold set for changing a cross-tier default, so the default remains 0.6. The complete sweep and reopening rule are in the replay README.
83
89
 
90
+ The same arithmetic governs the compiled system prompt, which sits ahead of every message. Its sections are ordered stable prefix first, so a section that can change between two turns never sits ahead of one that cannot; the order and the rule behind it are in [prompt-envelope-and-tools.md](prompt-envelope-and-tools.md#section-order-stable-prefix-first).
91
+
84
92
  Compaction and eviction both change the replayed history. On a local backend with a single prefix-cache slot, the next turn after either one is expected to be cold because the byte prefix moved. Dispatch traffic can disturb the same slot.
85
93
 
86
- Clio records these disturbances once on the next assistant entry as `promptCache.expectedColdReasons`. The recorded reasons are `working_set_evict` for an applied eviction event, `compaction` for the summary path, and `dispatch` for interleaved worker traffic. `compaction` and `dispatch` are stamped only on `local-native` targets, because a single-slot local cache is the one an interleaved run actually disturbs. `working_set_evict` is stamped on every tier: the eviction moved the byte prefix itself, so the cloud prefix cache is cold for the same reason. The user sees one dim notice, and the same reasons persist on that entry in the session ledger next to the per-call cache data.
94
+ Clio records these disturbances once on the next assistant entry as `promptCache.expectedColdReasons`. There are eight recorded reasons, and they split into two groups by what they disturb.
95
+
96
+ | Reason | Stamped when | Tier |
97
+ | --- | --- | --- |
98
+ | `working_set_evict` | An eviction event was applied to the replayed history. | every tier |
99
+ | `tool_surface_change` | The session's tool signature differs from the last completed run's. | every tier |
100
+ | `prompt_recompiled` | A recompile changed the prompt text and the manifest holds a previous hash to name, or an in-process session switch replaced the prefix after this process had applied a prompt. | every tier |
101
+ | `compaction` | The summary compaction path ran. | `local-native` |
102
+ | `dispatch` | A dispatch started, completed, or failed between turns. | `local-native` |
103
+ | `residency` | A residency load or eviction succeeded on this session's own serving endpoint. | `local-native` |
104
+ | `thinking_change` | The resolved thinking level for this run differs from the last completed backend run's. | `local-native` |
105
+ | `background_memory` | A proactive-memory step completed against the endpoint this session streams to. | `local-native` |
106
+
107
+ The three tier-independent reasons moved the byte prefix itself, so a cloud prefix cache is cold for exactly the same reason a local one is. The other five disturb a local server or the template it renders, and a single-slot local cache is the only one an interleaved run actually displaces, so they are stamped only when the runtime's tier is `local-native`. Two of them are gated on identity as well as tier: `residency` compares the mutation's target key against this session's own runtime and base URL, and `background_memory` compares the memory step's canonical endpoint key against the target this session streams to, so work on a second server never explains a cold prefix on the first. `prompt_recompiled` deliberately does not fire on a process's first compile: a fresh or resumed session has no previous hash to have diverged from, and stamping it there would mark every session's opening turn as expected-cold. An in-process switch (`/resume`, `/new`, a fork) stamps it from the switch itself rather than from the manifest. Manifest provenance follows the session, so the incoming session's `previousHash` is its own last recorded hash and usually equals what it compiles now, which leaves the manifest with nothing to report; the backend's slot meanwhile still holds the outgoing session's prompt and history, so the first turn after the switch is cold on every tier.
108
+
109
+ The user sees one dim notice per reason, and the same reasons persist on the run's first assistant entry in the session ledger next to the per-call cache data. The `/context` overlay renders each one in prose (`working-set eviction`, `dispatch traffic`, `residency change`, `thinking-level change`, `tool-surface change`, `prompt recompile`, `compaction`, `background memory step`) and falls through to the wire value only for an unknown reason.
87
110
 
88
111
  Per-call cache verdicts are `hot`, `partial`, `cold`, and `small`. They are derived from provider usage and persisted with `timing { ttftMs, apiMs }` and `promptCache { input, cacheRead, cacheWrite, backendVerdict }` when available.
89
112
 
113
+ ### What the serving backend reports
114
+
115
+ On a llama.cpp or LM Studio target, Clio also persists the server's own prefill accounting rather than inferring it from pi-ai's token counts. The observer reads the last complete timing object off the final ordinary SSE event of a stream, or the top-level one on a non-streaming response, from the response the turn already makes; it opens no second connection and sets no extra payload flag. What lands on the assistant entry is `promptCache.backend`:
116
+
117
+ | Field | Meaning |
118
+ | --- | --- |
119
+ | `promptTokens` | The whole prompt the server accounted for. On the observed llama.cpp build that is `prompt_n + cache_n`, since `prompt_n` counts only newly evaluated work. |
120
+ | `cachedTokens` | `cache_n`, the prompt work the slot reused. `null` when the server reports no cache figure at all. |
121
+ | `predictedTokens` | `predicted_n`, tokens generated. |
122
+ | `promptMs` | `prompt_ms`, wall-clock milliseconds spent in prefill. |
123
+ | `predictedMs` | `predicted_ms`, wall-clock milliseconds spent generating. |
124
+ | `source` | `llamacpp-timings` or `lmstudio-timings`. |
125
+
126
+ `uncachedPrefillTokens` is derived centrally as `promptTokens - cachedTokens`, and only when both figures are present and consistent. That distinction carries all the way to the surfaces: a missing `cache_n` persists `cachedTokens: null` and leaves the pi-ai verdict in force, so `/context` says `server does not report cache reads` instead of calling the backend cold. LM Studio 2.29.0 is that case today. Its OpenAI-compatible port returns `usage`, `stats`, and `system_fingerprint` and no `timings` object, on both the streaming and non-streaming shapes and with `timings_per_token` explicitly requested, so `lmstudio-timings` is a shape Clio accepts and has not yet observed.
127
+
128
+ The verdict keeps its existing pi-ai path unless pi-ai reports `cacheRead === 0` while the backend reports a numeric `cachedTokens`. In that one case the same hot, partial, cold, and small thresholds are applied to the measured counts instead. No timing ratio or wall-clock heuristic participates in a verdict.
129
+
130
+ `/context` renders the last call as `prefill: N uncached · M cached · X ms`, and falls back to `prefill: N prompt · X ms` when the server gave no cache figure. `/cost` folds every durable call in the session into a total uncached prefill plus the four verdict counts, `clio-coder usage report` carries the same two facts per session, and `clio-coder doctor` reports the latest session's verdict counts and its most frequent expected-cold reason without opening the TUI.
131
+
90
132
  The `/context` overlay closes the loop. When the last settled run came back `cold` and Clio had recorded a reason for it, the overlay adds a line naming that reason, for example `last cold turn: working-set eviction (expected)`, and reports the cache line without the warning token. A reused prompt shell with a cold backend and no recorded reason stays a warning: Clio kept the bytes stable and the provider re-prefilled anyway, which is a disagreement worth surfacing.
91
133
 
134
+ ## Prompt pre-warm
135
+
136
+ On a local server prefill is the cost. A fresh session's first turn prefills the whole compiled prompt plus the tool schemas before the model emits a token, and a resumed session's first turn prefills the entire replayed history. Both are paid after the operator presses Enter, and both are fully determined before they type anything. Since llama.cpp picks the slot with the longest common prefix and re-evaluates only the suffix, sending that prefix early leaves the processed KV where the real turn will land.
137
+
138
+ Clio sends it at three moments: after the session prompt compiles at session start, after a resume rebuilds the message array, and after a compaction settles. The third is included because the next turn is known to be cold and the operator is usually reading the summary rather than typing.
139
+
140
+ The payload is the request the next turn would send minus the operator's text: the same system prompt, the same tool schemas, the same replayed messages, the same thinking level, and the same `cache_prompt`, with one single-character user message appended so the chat template renders the prefix up to the user turn, and `max_tokens: 1`. It is built through the same `streamSimple` dispatcher `createEngineAgent` hands the engine as its `streamFn`, not a hand-assembled payload, because any byte that differs ahead of the user turn defeats the purpose.
141
+
142
+ The pre-warm is refused rather than queued whenever it would compete with real work. It runs only on `local-native` targets, whatever `prewarm.enabled` says, because a cloud provider bills the request and caches on its own schedule. It never runs while a turn is in flight, while any dispatch is outstanding, on a worker, or in headless `run`. The dispatch guard is a stand-in: without per-endpoint capacity accounting the pre-warm cannot tell whether a worker already occupies the server it would warm, so it stands down for all worker traffic. The round already claims one endpoint slot for as long as its request is out and releases it in a `finally`, through the `registerEndpointSlot` seam the chat loop wires from the endpoint-capacity registry, so capacity counts a pre-warm the same way it counts the orchestrator's streaming turn.
143
+
144
+ Pressing Enter lets go of an in-flight pre-warm at the keystroke, before the admission gate. Whether it also aborts the HTTP request is gated on what the backend does with a cancelled one, and the measured backend does nothing. On the operator's llama.cpp router (build `b226-2115b73d8`, Qwen3.8-27B, `--parallel 1`), aborting 1.5 s into a 47,620-token prefill did not cancel the server's work: the server finished prefilling, so the prefix did survive the abort and the next request read 47,596 of 47,620 tokens from cache with `prompt_ms 927`, but that request also waited 89.5 s of wall clock for the abandoned one to leave the single slot. Letting the pre-warm complete instead cost 89.3 s plus a 1.3 s turn, the same wall clock. The abort therefore frees no slot and saves no time on this backend; all it does is discard the usage and timings of prefill the server performed. So a submit detaches the round instead: Clio stops calling it the current pre-warm, never waits on it, withholds its `/context` line because it no longer describes the prefix the next turn will send, and still records what it cost. `ABORT_ROUND_ON_SUBMIT` in `src/interactive/turn-prewarm.ts` carries the measurement and flips the behavior for a backend that honors cancellation.
145
+
146
+ Each round appends one `prewarm` custom ledger entry carrying its trigger, the backend prompt tokens, `timing`, and `promptCache`. The entry is never rendered and never becomes a model message, so it contributes zero tokens to the context estimate. `/context` shows `prewarmed: N tokens in X ms` until the next settled run answers the question it asked. `prewarm` is never an expected-cold reason: a pre-warm is the opposite of a disturbance. Its provider usage is real spend and is reported to `/cost` and `clio-coder usage report` under its own row, the way a `/btw` side question is.
147
+
92
148
  ## Settings
93
149
 
94
150
  The public settings use one compaction threshold plus a non-destructive working-set stage:
@@ -108,6 +164,9 @@ context:
108
164
  target: 0.6
109
165
  protectLastTurns: 6
110
166
  minEvictableTokens: 200
167
+
168
+ prewarm:
169
+ enabled: true
111
170
  ```
112
171
 
113
172
  `compaction.auto` controls the pre-request trigger. Manual `/context compact` still runs when `auto` is false. `compaction.model` optionally selects a dedicated summarization model, and `compaction.systemPrompt` optionally points at a prompt override file. `compaction.excludeLastTurns` only governs the temporary legacy mask path; working-set protection uses `context.workingSet.protectLastTurns`.