@iowarp/clio-coder 0.3.8 → 0.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (333) hide show
  1. package/CHANGELOG.md +49 -0
  2. package/README.md +7 -3
  3. package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
  4. package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
  5. package/dist/assets/codewiki.json +1 -1
  6. package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
  7. package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
  8. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  9. package/dist/chunk-3DPEIQKN.js +113 -0
  10. package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
  11. package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
  12. package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
  13. package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
  14. package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
  15. package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
  16. package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
  17. package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
  18. package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
  19. package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
  20. package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
  21. package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
  22. package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
  23. package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
  24. package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
  25. package/dist/chunk-7C6RYZGQ.js +391 -0
  26. package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
  27. package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
  28. package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
  29. package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
  30. package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
  31. package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
  32. package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
  33. package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
  34. package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
  35. package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
  36. package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
  37. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  38. package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
  39. package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
  40. package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
  41. package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
  42. package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
  43. package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
  44. package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
  45. package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
  46. package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
  47. package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
  48. package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
  49. package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
  50. package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
  51. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  52. package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
  53. package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
  54. package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
  55. package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
  56. package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
  57. package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
  58. package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
  59. package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
  60. package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
  61. package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
  62. package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
  63. package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
  64. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  65. package/dist/chunk-RKKLTLYB.js +45 -0
  66. package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
  67. package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
  68. package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
  69. package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
  70. package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
  71. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  72. package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
  73. package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
  74. package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
  75. package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
  76. package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
  77. package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
  78. package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
  79. package/dist/chunk-YSEHGPCT.js +127 -0
  80. package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
  81. package/dist/cli/index.js +27 -27
  82. package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
  83. package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
  84. package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
  85. package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
  86. package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
  87. package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
  88. package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
  89. package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
  90. package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
  91. package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
  92. package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
  93. package/dist/doctor-TWBWFK5V.js +165 -0
  94. package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
  95. package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
  96. package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
  97. package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
  98. package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
  99. package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
  100. package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
  101. package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
  102. package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
  103. package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
  104. package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
  105. package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
  106. package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
  107. package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
  108. package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
  109. package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
  110. package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
  111. package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
  112. package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
  113. package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
  114. package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
  115. package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
  116. package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
  117. package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
  118. package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
  119. package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
  120. package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
  121. package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
  122. package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
  123. package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
  124. package/dist/worker/entry.js +29 -33
  125. package/docs/alcf-provider.md +1 -1
  126. package/docs/architecture.md +1 -1
  127. package/docs/artifact-versions.md +6 -1
  128. package/docs/built-in-agents.md +1 -1
  129. package/docs/capacity-and-scheduling.md +23 -2
  130. package/docs/commands-and-modes.md +1 -1
  131. package/docs/configuration-and-targets.md +30 -3
  132. package/docs/context-engine.md +63 -4
  133. package/docs/documentation-coverage.md +3 -3
  134. package/docs/documentation-guide.md +1 -1
  135. package/docs/environment-variables.md +2 -0
  136. package/docs/eval-runner.md +262 -11
  137. package/docs/evals-internal.md +72 -2
  138. package/docs/evidence-and-memory.md +11 -10
  139. package/docs/evolution.md +1 -1
  140. package/docs/extensions-and-sharing.md +3 -1
  141. package/docs/fleet-dispatch.md +4 -4
  142. package/docs/installation-and-lifecycle.md +1 -1
  143. package/docs/middleware-and-components.md +1 -1
  144. package/docs/model-catalog.md +1 -1
  145. package/docs/observability.md +53 -2
  146. package/docs/proactive-memory.md +127 -14
  147. package/docs/prompt-envelope-and-tools.md +19 -1
  148. package/docs/provider-adapter-cookbook.md +1 -1
  149. package/docs/release-cut-checklist.md +19 -3
  150. package/docs/safety-model.md +1 -1
  151. package/docs/scientific-validation.md +1 -1
  152. package/docs/skills-marketplace.md +1 -1
  153. package/docs/tool-usage.md +1 -1
  154. package/docs/trace-store.md +1 -1
  155. package/docs/troubleshooting.md +87 -0
  156. package/docs/tui-design.md +1 -1
  157. package/docs/worker-dispatch-mechanics.md +1 -1
  158. package/package.json +2 -1
  159. package/src/cli/agents.ts +1 -1
  160. package/src/cli/config-inspect.ts +33 -6
  161. package/src/cli/config.ts +1 -1
  162. package/src/cli/doctor-state-size.ts +82 -0
  163. package/src/cli/doctor.ts +3 -1
  164. package/src/cli/eval.ts +80 -16
  165. package/src/cli/extensions.ts +5 -1
  166. package/src/cli/fleet.ts +32 -3
  167. package/src/cli/targets.ts +44 -13
  168. package/src/cli/trace.ts +63 -4
  169. package/src/cli/usage.ts +63 -14
  170. package/src/core/bus-events.ts +29 -1
  171. package/src/core/cache-telemetry.ts +42 -0
  172. package/src/core/config.ts +18 -0
  173. package/src/core/defaults.ts +36 -6
  174. package/src/core/endpoint-key.ts +27 -0
  175. package/src/core/residency-target-key.ts +25 -0
  176. package/src/core/response-schema.ts +36 -2
  177. package/src/domains/config/classify.ts +3 -0
  178. package/src/domains/context/codewiki/coordinator.ts +12 -4
  179. package/src/domains/dispatch/admission.ts +40 -3
  180. package/src/domains/dispatch/capacity-lease.ts +98 -9
  181. package/src/domains/dispatch/contract.ts +11 -0
  182. package/src/domains/dispatch/execution-plan.ts +44 -4
  183. package/src/domains/dispatch/extension.ts +166 -42
  184. package/src/domains/dispatch/fleet-run.ts +23 -3
  185. package/src/domains/dispatch/heartbeat.ts +32 -8
  186. package/src/domains/dispatch/index.ts +3 -0
  187. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  188. package/src/domains/dispatch/reservation-store.ts +116 -8
  189. package/src/domains/dispatch/state.ts +4 -0
  190. package/src/domains/dispatch/worker-spawn.ts +25 -11
  191. package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
  192. package/src/domains/dispatch/write-boundary.ts +62 -1
  193. package/src/domains/eval/artifacts/store.ts +62 -0
  194. package/src/domains/eval/compare/behavioral.ts +224 -0
  195. package/src/domains/eval/compare/compare.ts +355 -2
  196. package/src/domains/eval/compare/envelope.ts +128 -0
  197. package/src/domains/eval/compare/gates.ts +24 -6
  198. package/src/domains/eval/compare/thresholds.ts +30 -3
  199. package/src/domains/eval/execution-provenance.ts +240 -0
  200. package/src/domains/eval/metrics/aggregate.ts +136 -0
  201. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  202. package/src/domains/eval/metrics/tracked.ts +413 -0
  203. package/src/domains/eval/provenance.ts +117 -0
  204. package/src/domains/eval/reports/comparison.ts +128 -0
  205. package/src/domains/eval/reports/junit.ts +17 -3
  206. package/src/domains/eval/reports/markdown.ts +3 -3
  207. package/src/domains/eval/reports/text.ts +14 -0
  208. package/src/domains/eval/run-compare.ts +20 -0
  209. package/src/domains/eval/runners/clio-run.ts +127 -0
  210. package/src/domains/eval/runners/external-command.ts +28 -3
  211. package/src/domains/eval/schema/adapter.ts +111 -0
  212. package/src/domains/eval/schema/artifact.ts +20 -0
  213. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  214. package/src/domains/eval/schema/behavioral.ts +520 -0
  215. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  216. package/src/domains/eval/schema/serving.ts +74 -0
  217. package/src/domains/eval/schema/suite.ts +38 -8
  218. package/src/domains/eval/schema/validate.ts +58 -3
  219. package/src/domains/eval/schema/verdict.ts +237 -0
  220. package/src/domains/eval/suites/resolve.ts +2 -0
  221. package/src/domains/eval/suites/run.ts +264 -33
  222. package/src/domains/eval/verifiers/command.ts +2 -1
  223. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  224. package/src/domains/evidence/build.ts +2 -13
  225. package/src/domains/evidence/eval.ts +2 -12
  226. package/src/domains/evidence/findings-markdown.ts +33 -0
  227. package/src/domains/evidence/run-trust.ts +7 -113
  228. package/src/domains/evidence/trust-projection.ts +2 -2
  229. package/src/domains/extensions/compatibility.ts +285 -0
  230. package/src/domains/extensions/discovery.ts +38 -3
  231. package/src/domains/extensions/resources.ts +1 -1
  232. package/src/domains/extensions/state.ts +12 -3
  233. package/src/domains/extensions/types.ts +2 -0
  234. package/src/domains/lifecycle/doctor.ts +69 -1
  235. package/src/domains/memory/index.ts +14 -0
  236. package/src/domains/memory/task-bank-promotion.ts +64 -0
  237. package/src/domains/memory/task-memory-policy.ts +77 -8
  238. package/src/domains/memory/task-memory-spend.ts +131 -0
  239. package/src/domains/memory/task-memory-status.ts +7 -0
  240. package/src/domains/memory/task-memory-telemetry.ts +2 -0
  241. package/src/domains/middleware/index.ts +1 -0
  242. package/src/domains/middleware/memory-intervention.ts +69 -5
  243. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  244. package/src/domains/observability/background-memory-usage.ts +140 -0
  245. package/src/domains/observability/cost.ts +1 -1
  246. package/src/domains/observability/index.ts +7 -0
  247. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  248. package/src/domains/observability/trace-store.ts +192 -2
  249. package/src/domains/prompts/compiler.ts +100 -13
  250. package/src/domains/providers/endpoint-capacity.ts +96 -0
  251. package/src/domains/providers/index.ts +10 -0
  252. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  253. package/src/domains/providers/runtime-resolution.ts +8 -1
  254. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  255. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  256. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  257. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  258. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  259. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  260. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  261. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  262. package/src/domains/providers/types/capability-flags.ts +2 -0
  263. package/src/domains/providers/types/target-descriptor.ts +2 -0
  264. package/src/domains/resources/prompts/loader.ts +95 -33
  265. package/src/domains/safety/call-target.ts +52 -0
  266. package/src/domains/safety/run-effects.ts +35 -4
  267. package/src/domains/session/context-accounting.ts +52 -1
  268. package/src/domains/session/context-ledger.ts +37 -13
  269. package/src/domains/session/index.ts +6 -0
  270. package/src/domains/session/prompt-cache.ts +140 -0
  271. package/src/domains/session/prompt-manifest.ts +42 -0
  272. package/src/engine/acp/adapter.ts +18 -3
  273. package/src/engine/ai.ts +35 -0
  274. package/src/engine/apis/llamacpp-residency.ts +55 -3
  275. package/src/engine/apis/lmstudio.ts +25 -5
  276. package/src/engine/apis/ollama-native.ts +2 -1
  277. package/src/engine/apis/openai-completions.ts +80 -17
  278. package/src/engine/apis/residency-lock.ts +3 -1
  279. package/src/engine/apis/residency.ts +34 -1
  280. package/src/engine/provider-payload.ts +29 -1
  281. package/src/entry/orchestrator.ts +176 -30
  282. package/src/interactive/chat-loop-messages.ts +26 -7
  283. package/src/interactive/chat-loop.ts +318 -41
  284. package/src/interactive/chat-panel.ts +62 -8
  285. package/src/interactive/clio-editor.ts +45 -8
  286. package/src/interactive/context-activity.ts +5 -1
  287. package/src/interactive/context-meter.ts +1 -1
  288. package/src/interactive/context-overlay.ts +40 -10
  289. package/src/interactive/cost-overlay.ts +64 -6
  290. package/src/interactive/dispatch-board.ts +84 -12
  291. package/src/interactive/fleet-run-preview.ts +41 -15
  292. package/src/interactive/handoff-round.ts +41 -2
  293. package/src/interactive/interactive-application.ts +24 -1
  294. package/src/interactive/interactive-input-runtime.ts +8 -0
  295. package/src/interactive/interactive-presentation.ts +4 -0
  296. package/src/interactive/interactive-shell.ts +20 -17
  297. package/src/interactive/interactive-slash-runtime.ts +27 -4
  298. package/src/interactive/memory-overlay.ts +8 -0
  299. package/src/interactive/mutation-preview.ts +295 -0
  300. package/src/interactive/overlay-general-openers.ts +16 -0
  301. package/src/interactive/overlay-key-routing.ts +38 -0
  302. package/src/interactive/overlay-lifecycle.ts +38 -5
  303. package/src/interactive/overlay-permission-lifecycle.ts +22 -2
  304. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  305. package/src/interactive/overlays/ask-user.ts +91 -19
  306. package/src/interactive/overlays/help-reference.ts +4 -0
  307. package/src/interactive/overlays/prompts.ts +11 -1
  308. package/src/interactive/overlays/settings.ts +35 -1
  309. package/src/interactive/permission-hint.ts +34 -2
  310. package/src/interactive/permission-overlay.ts +159 -9
  311. package/src/interactive/prewarm.ts +197 -0
  312. package/src/interactive/render-trace.ts +162 -15
  313. package/src/interactive/renderers/tool-execution.ts +4 -0
  314. package/src/interactive/side-question.ts +58 -1
  315. package/src/interactive/status/controller.ts +11 -0
  316. package/src/interactive/status/state-machine.ts +54 -2
  317. package/src/interactive/status/types.ts +7 -0
  318. package/src/interactive/terminal-lease.ts +2 -0
  319. package/src/interactive/turn-context.ts +299 -31
  320. package/src/interactive/turn-persistence.ts +14 -4
  321. package/src/interactive/turn-prewarm.ts +364 -0
  322. package/src/interactive/turn-queues.ts +7 -4
  323. package/src/interactive/turn-runtime.ts +8 -1
  324. package/src/interactive/turn-state.ts +23 -0
  325. package/src/interactive/view/view-overlay.ts +28 -3
  326. package/src/tools/ask-user.ts +43 -2
  327. package/src/tools/dispatch-plan.ts +17 -9
  328. package/src/tools/dispatch-scout.ts +1 -1
  329. package/src/tools/registry.ts +16 -0
  330. package/dist/chunk-AOCYTWAV.js +0 -449
  331. package/dist/chunk-HWUFFB6L.js +0 -83
  332. package/dist/chunk-R346GLFC.js +0 -31
  333. package/dist/doctor-M7YEDGAE.js +0 -91
@@ -1,7 +1,7 @@
1
1
  # Observability Viewer
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  `/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
7
7
 
@@ -13,6 +13,35 @@
13
13
 
14
14
  `/view` opens a full-screen split viewer. The left pane groups artifacts by category and supports type-to-filter. The right pane renders the selected artifact with pager controls. `Tab` or `Shift+Tab` switches between the artifact list and details. `Left` and `Right` jump to the previous or next non-empty category from either pane; `Up` and `Down` select artifacts in the list or scroll details in the content pane. Category jumps honor the active filter and wrap at the ends. `v` verifies a selected receipt. `o` shows the absolute backing path through the notice channel when the selected artifact has one; pathless artifacts produce a warning notice instead. In the list pane, `Esc` clears a non-empty filter before a second `Esc` closes the viewer.
15
15
 
16
+ ## Trace retention and state usage
17
+
18
+ The SQLite trace mirror at `<state-dir>/trace.sqlite` is rebuildable and bounded. By default Clio retains terminal runs for 30 days and limits the allocated database to 128 MiB (134,217,728 bytes), whichever limit is reached first. The policy runs after each dispatch or interactive turn becomes terminal. It deletes a run as one unit across `runs`, `phases`, `events`, `envelopes`, `gate_results`, `agent_sessions`, and `processes`. A `queued` or `running` run is never a candidate, even when its start time is older than the age cutoff or its rows put the store over the byte limit.
19
+
20
+ Two environment variables configure the automatic policy:
21
+
22
+ | Variable | Default | Valid values |
23
+ | --- | ---: | --- |
24
+ | `CLIO_CODER_TRACE_RETENTION_DAYS` | `30` | An integer of at least 1. |
25
+ | `CLIO_CODER_TRACE_MAX_BYTES` | `134217728` | An integer of at least 1,048,576. |
26
+
27
+ An operator can apply the current policy immediately or supply one-command overrides:
28
+
29
+ ```bash
30
+ clio-coder trace prune
31
+ clio-coder trace prune --max-age-days 14 --max-bytes 67108864
32
+ clio-coder trace prune --json
33
+ ```
34
+
35
+ The command reports terminal runs removed, total rows removed across the seven run-owned tables, physical bytes reclaimed from `trace.sqlite` and its WAL sidecars, whether `VACUUM` ran, and how many live runs were protected. Age pruning uses a terminal run's `ended_at`. Size pruning removes the oldest terminal runs until the live database pages fit or no terminal candidate remains.
36
+
37
+ Deleting SQLite rows creates reusable pages but does not normally reduce the file. Clio runs `VACUUM` when at least 20 percent of allocated pages are reclaimable, or whenever reclaiming deleted pages is necessary to enforce the 128 MiB bound. It then truncates the WAL. Smaller deletions remain available for SQLite to reuse and avoid rewriting the whole database on every completed run.
38
+
39
+ `clio-coder doctor` includes a `state storage` row with the recursive byte total for the state directory and the largest top-level contributor. For example:
40
+
41
+ ```text
42
+ OK state storage 96.4 MiB (101,082,624 bytes); largest contributor trace.sqlite at 89.9 MiB (94,248,960 bytes)
43
+ ```
44
+
16
45
  ---
17
46
 
18
47
  ## The Evidence Spine End-to-End
@@ -69,8 +98,28 @@ An `EvidenceIndexRow` has the following schema:
69
98
  | `turns` | `turns in window` and the `tokens` fact | Folded calls that were turns, so labelled calls are subtracted exactly as `/cost` subtracts them. |
70
99
  | `sideQuestions` | `side questions in window` and the `tokens` fact | `/btw` rounds in the window. |
71
100
  | `handoffs` | `handoffs in window` and the `tokens` fact | `/handoff` extraction rounds in the window. |
101
+ | `prewarms` | `pre-warms in window` and the `tokens` fact | Prompt pre-warm rounds in the window. |
102
+ | `backgroundMemorySteps` | `background memory steps in window` and the `tokens` fact | Proactive-memory model steps in the window. |
103
+
104
+ The last five fields appear only when at least one labelled call falls in the window, and each individual line is printed only when its own count is above zero. An archive with no labelled call in it renders exactly as it did before those fields existed, so their presence is itself the signal that money was spent beside a session. All four labelled kinds are subtracted from `turns` the same way, so a session's turn count never includes a round the operator did not take.
105
+
106
+ The report also prints a prompt-cache block, one row per session that recorded any cache telemetry:
107
+
108
+ ```text
109
+ prompt cache by session (from backend timings and persisted verdicts)
110
+ session uncached prefill hot/partial/cold/small
111
+ 3vpu6z19ee7t 130353 4/3/2/0
112
+ ```
72
113
 
73
- The last three fields appear only when at least one labelled call falls in the window. An archive with no `/btw` or `/handoff` round in it renders exactly as it did before those fields existed, so their presence is itself the signal that money was spent beside a session.
114
+ `uncached prefill` is the sum of the backend's own newly evaluated prompt tokens across every persisted call in that session, and reads `n/a` rather than `0` when the server reported no cache figure to subtract. The four counts are the per-call verdicts. Both facts are also in `--json` under a `session-cache` fact per session.
115
+
116
+ `clio-coder doctor` reports the same evidence for the latest session only, as one row, so a cache problem is visible without opening the TUI or a report:
117
+
118
+ ```text
119
+ OK cache telemetry last session 3vpu6z19ee7t: hot 4 · partial 3 · cold 2 · small 0; top expected reason dispatch (3)
120
+ ```
121
+
122
+ The row reads `top expected reason none` when the session recorded verdicts but no expected-cold reason, and it degrades to a warning saying `no prompt-cache telemetry recorded` when the latest session has none at all, which is the honest answer for a target whose backend reports nothing rather than a claim of a perfect cache. "Latest" selects the most recent `current.jsonl` by its newest entry timestamp, falling back to the file's mtime, so the other diagnostic JSONL files in a session directory cannot be mistaken for the conversation.
74
123
 
75
124
  ### The Out-of-Turn Usage Store
76
125
 
@@ -102,6 +151,8 @@ A row has the following schema:
102
151
 
103
152
  `repoIdentity` is the same cwd hash the session ledger is filed under, which is what lets `usage report --repo <path>` select these rows with the hash it already computes for the ledgers.
104
153
 
154
+ `label` is one of `side-question`, `handoff`, `prewarm`, or `background-memory`. The last two joined for the same reason as the first two: a prompt pre-warm and a proactive-memory step are provider calls the operator did not ask for and would otherwise never see, and neither appends anything to the session JSONL. A row may also carry `timing { durationMs }` and a `promptCache` block built from the backend's own prefill facts when the server reported them; a backend that reports no timings simply omits the block, as LM Studio's OpenAI-compatible port does.
155
+
105
156
  ---
106
157
 
107
158
  ## Artifact Categories and Path Layouts
@@ -1,6 +1,6 @@
1
1
  # Proactive task memory
2
2
 
3
- > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.8).
3
+ > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.9).
4
4
 
5
5
  Clio's proactive task memory protects long-running work from behavioral state
6
6
  decay: a requirement, environment fact, failed attempt, or diagnosis can still
@@ -120,7 +120,7 @@ malformed response, or telemetry failure is silent and never blocks a tool.
120
120
  - `memory.intervention.everyNTools` (default `10`): Minimum completed-tool interval between background interventions.
121
121
  - `memory.intervention.windowSteps` (default `8`): Completed tool-trajectory window analyzed during background evaluation.
122
122
  - `memory.intervention.maxTokens` (default `400`): Bounds the rendered memory-bank and reminder context budget; the policy model output cap is a separate fixed `4,000`-token contract in `task-memory-policy.ts`, sized so that a model which reasons anyway still reaches its envelope.
123
- - `memory.intervention.timeoutMs` (default `180000`): Wall-clock limit for one background memory-policy request. The step is detached, so this deadline never delays a turn; set it above the observed step time for your route or finished work is discarded as a timeout. Step latency on a small local route is long-tailed rather than tightly clustered, so size this off a high percentile and not off a median.
123
+ - `memory.intervention.timeoutMs` (default `30000`): Wall-clock limit for one background memory-policy request. The step is detached, so this deadline never delays a turn, but it does hold a request slot on a real inference endpoint that your own turns and your dispatched workers queue against. The default is what a turn boundary can wait for rather than what a long-tailed route eventually answers in: on the reference route below, 23 of 60 steps ran past 30 seconds and 531 of the measured 1,666 seconds were spent beyond that mark. Raise it only if you have measured that your route's slow steps are the ones producing reminders, and read the trade in "Cost and the default decision" first.
124
124
 
125
125
  ## Trigger semantics
126
126
 
@@ -203,6 +203,88 @@ Thus `last` remains `injected` across such continuations until a later
203
203
  tool-bearing or explicitly triggered memory step produces a new outcome (e.g.,
204
204
  a healthy tool leading to `silent`).
205
205
 
206
+ ## Cost and the default decision
207
+
208
+ The LLM tier costs real tokens, real seconds of model time, and a request slot on
209
+ a server that is usually the same machine the operator's own turns run on. Every
210
+ step is therefore accounted for the way a `/btw` side question is: one cost entry
211
+ under the `background-memory` label, which `/cost` shows as its own `memory steps`
212
+ row, and one durable row in `<stateDir>/usage/out-of-turn.jsonl` carrying the
213
+ usage, the call's duration, and the backend's prefill facts, which
214
+ `clio-coder usage report` folds after the process exits. `/memory` shows the
215
+ lifetime figures folded from `steps.jsonl`: steps, tokens, model time, and the
216
+ hit rate.
217
+
218
+ ### The measurement
219
+
220
+ From one operator's `steps.jsonl`, 274 rows spanning 2026-08-14 to 2026-08-29 on
221
+ a small local background route:
222
+
223
+ | Figure | Value |
224
+ | --- | --- |
225
+ | Model-tier steps | 60 |
226
+ | Tokens | 137,205 |
227
+ | Model time | 1,666.6 s |
228
+ | Step latency | median 18.7 s, p90 70.2 s, max 102.5 s |
229
+ | Injections produced by the model tier | 6 |
230
+ | Hit rate | 10.0 percent |
231
+ | Cost per injection | 22,868 tokens and 278 s of model time |
232
+ | Model-tier injections in the last 5 days | 0 of 4 steps |
233
+
234
+ Four further injections in the same window came from the free rules tier, so the
235
+ lifetime total of 10 injections is not the model tier's score. Rules-tier
236
+ injections cost nothing.
237
+
238
+ ### The decision
239
+
240
+ The default does not change, and it is a deliberate default rather than an
241
+ unexamined one:
242
+
243
+ - `memory.intervention.enabled` stays `true`. It runs the rules tier, which makes
244
+ no model calls, spends no tokens, and produced 4 of the 10 injections.
245
+ - The LLM tier stays opt-in through `background.target` and `background.model`,
246
+ which is already the case: an unset background role never resolves a client.
247
+ A 10 percent hit rate at 22,868 tokens per injection does not earn a default-on
248
+ position, and it is not so poor that it earns removal from an operator who has
249
+ measured their own route and wants it.
250
+ - The step deadline drops from 180 s to 30 s. This is the one behavioral change,
251
+ and it is a genuine trade: at 30 s, two of the six observed injections, at
252
+ 53.6 s and 57.7 s, would have been cut, while 531 s of the 1,666 s spent would
253
+ not have been spent at all. The deadline is the bound on what one optional call
254
+ may hold a shared local server for, not a prediction of when a route answers.
255
+ - A step that would run on the endpoint the chat target is streaming against is
256
+ skipped with reason `endpoint_busy`, and the skip is recorded. On a single-slot
257
+ llama.cpp router the alternative is queueing behind the operator's own decoding
258
+ or evicting the resident model, and neither is a cost an optional call may
259
+ impose.
260
+
261
+ ### What a background target costs on a shared local server
262
+
263
+ If `background.target` names the same server as `orchestrator.target`, that
264
+ server's slots are shared. On a llama.cpp router started with `--parallel 1`
265
+ there is exactly one, and the memory step and the operator's turn contend for it.
266
+
267
+ The consequence is worth stating plainly: a shared endpoint suppresses the model
268
+ tier rather than merely delaying it. A step is started from the `turn_end` hook,
269
+ which fires inside the streaming run at `agent_end`
270
+ (`src/interactive/turn-runtime.ts`), while the chat loop still holds its
271
+ foreground registration on that endpoint; the loop releases the hold afterwards,
272
+ in the `finally` around the run (`src/interactive/chat-loop.ts`). Every boundary
273
+ therefore finds the endpoint busy and records `dropped`/`endpoint_busy`. That is
274
+ the intended trade: an optional call may not take the one slot the operator's own
275
+ turn is using, and it may not make the server swap the resident model out. The
276
+ `/memory` step list and `steps.jsonl` say so on every boundary, so the tier is
277
+ visibly declining rather than quietly idle.
278
+
279
+ The second mechanism is an `expected cold` stamp, for the case where a step did
280
+ run on the chat endpoint. Its prompt is a trajectory rather than the chat prefix,
281
+ so the next turn's prefill is expected to be cold; `/context` names
282
+ `background_memory` as the reason instead of reporting an unexplained cold
283
+ prefix.
284
+
285
+ Pointing the background role at a second machine avoids both effects and is the
286
+ arrangement the tier is designed for.
287
+
206
288
  ## Choosing a background model
207
289
 
208
290
  Memory reads a trajectory and writes a fixed envelope. It does not plan, and it
@@ -257,7 +339,7 @@ memory:
257
339
  everyNTools: 10
258
340
  windowSteps: 8
259
341
  maxTokens: 400
260
- timeoutMs: 180000
342
+ timeoutMs: 30000
261
343
  ```
262
344
 
263
345
  With `background.target` and `background.model` unset, Clio stays in the
@@ -289,12 +371,12 @@ percentile of 79.9, and a 95th of 131.6. Capability is not the constraint;
289
371
  latency is, its spread is wide, and the detached step above is what makes the
290
372
  tier usable anyway.
291
373
 
292
- Size `timeoutMs` off that tail rather than off the median. The shipped 180000
293
- captures roughly the whole distribution on this route. A 20000 setting looks
294
- generous against an 18.6-second median and in practice discarded about half of
295
- all steps, since the request is aborted on timeout and its work is thrown
296
- away. A route whose steps mostly record `timeout` is a misconfigured deadline
297
- before it is a slow model.
374
+ The deadline is a bound on what an optional call may hold that server for, not a
375
+ figure sized to capture the tail. The shipped 30000 sits above the median and
376
+ below the tail deliberately, and a step that exceeds it records `timeout` with
377
+ its work discarded. A route whose steps mostly record `timeout` is a
378
+ misconfigured deadline before it is a slow model, so read the ledger before
379
+ raising it: `/memory` shows the hit rate the raise would be buying.
298
380
 
299
381
  The target ID is not hard-coded. Any configured orchestrator-eligible local
300
382
  target and wire model can fill the background role. Before enabling it, use the
@@ -316,6 +398,25 @@ For an immediate kill switch, set `memory.intervention.enabled` to `false` in
316
398
  `/settings`. Removing the background target instead returns to rules-only
317
399
  operation while leaving deterministic protection active.
318
400
 
401
+ ## Where what the tier writes ends up
402
+
403
+ A bank entry lives and dies with its session. When a reminder actually reaches
404
+ the operator, the entries it cited are also proposed into the durable store at
405
+ `<dataDir>/memory/records.json`, unapproved, scoped to the repository the session
406
+ is working in, with provenance naming the session and the source entry. That is
407
+ the one automatic writer of that file; everything else about it is unchanged.
408
+ `/memory` and `clio-coder memory list` show the proposal, and
409
+ `clio-coder memory approve <id>` is still a separate operator action, so nothing
410
+ the background plane produced reaches a system prompt without review. A step with
411
+ no session, or one running outside a canonical repository, proposes nothing:
412
+ global scope broadens applicability to every future session and is not a claim a
413
+ background step may make on the operator's behalf.
414
+
415
+ Rules-tier reminders are not proposed. Their entries are this middleware's own
416
+ one-line records of a repeated tool failure, and filing each one as a durable
417
+ lesson would fill the review queue with rows nobody asked for. They remain
418
+ promotable by hand from `/memory`.
419
+
319
420
  ## What the LLM tier actually writes
320
421
 
321
422
  Measured on the shipped prompt against `google/gemma-4-26b-a4b-qat`, across ten
@@ -392,11 +493,23 @@ keeps one previous generation as `steps.jsonl.1`. Every exact-schema record has:
392
493
  - `silent`, `injected`, `gated`, `timeout`, `malformed`, or `dropped` decision;
393
494
  - count of cited entries, input/output/total memory-model tokens, and latency.
394
495
 
395
- `dropped` is the one outcome that ran no step: the boundary triggered while an
396
- earlier step still held the single in-flight slot. It costs no tokens and no
397
- latency, its triggers survive to the next free boundary, and it does not replace
398
- the operator-visible last decision. Counting `dropped` rows against `llm` rows
399
- over a session is how a starved cadence becomes visible.
496
+ The same steps are also billed. See "Cost and the default decision" for the
497
+ `/cost` row, the durable out-of-turn usage row, and the lifetime figures `/memory`
498
+ folds out of this file.
499
+
500
+ `dropped` is the one outcome that ran no step. It has two causes, separated by
501
+ the row's reason: `step_in_flight` means the boundary triggered while an earlier
502
+ step still held the single in-flight slot, and `endpoint_busy` means the step
503
+ would have called the endpoint the chat target was streaming against. Both cost
504
+ no tokens and no latency, both leave their triggers pending for the next free
505
+ boundary, and neither replaces the operator-visible last decision. Counting
506
+ `dropped` rows against `llm` rows over a session is how a starved cadence becomes
507
+ visible.
508
+
509
+ A step that exceeds the deadline records `timeout`, never `silent`: reason
510
+ `deadline` when the policy's own race fired first, and `timed_out` when the
511
+ transport aborted at its deadline. Both are distinct from `client_error`, which
512
+ is a route that refused rather than a route that was slow.
400
513
 
401
514
  The log contains no task, trajectory, bank, error, or reminder text. File creation,
402
515
  rotation, serialization, and injected sinks are all best effort; a read-only
@@ -1,7 +1,7 @@
1
1
  # Prompt Envelope and Tools
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder keeps the model-facing envelope stable and moves enforcement into the runtime registry and safety policy.
7
7
 
@@ -13,6 +13,24 @@ The chat loop compiles one provider-facing system prompt for a session. The comp
13
13
 
14
14
  The compiled prompt is reused byte-for-byte on ordinary submits. It recompiles only when that key changes or when config hot-reload invalidates the prompt cache. Path-scoped project rules can therefore recompile the prompt when a matching file enters working context. When recompilation changes the text, the session ledger records a `promptRecompiled` entry with the previous hash, new hash, and token estimate.
15
15
 
16
+ ## Section order: stable prefix first
17
+
18
+ The compiled prompt lays its sections down in `SESSION_PROMPT_SECTION_ORDER` (`src/domains/prompts/compiler.ts`): identity, operating contract, delegation, skills, safety, tool contract, fleet, retrieval hints, project context, memory, runtime, then the operator-editable tail fragments (workspace root, Clio repo awareness, project rules, operator profile) in their own order.
19
+
20
+ One rule fixes that list. A section goes as late as its volatility, and anything that reads a clock, a probe, or a mutable store goes after everything that does not. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so a section that can change between two turns must not sit ahead of sections that cannot. The runtime block is last of the compiled sections because its `Context window: N` moves when the backend reloads a model or a co-residency clamp lands; memory sits just ahead of it because an approved memory record rewrites that section mid-session; project rules are dead last because path-scoped rules join the prompt when a matching file enters working context.
21
+
22
+ `Context window: N` is the window the backend will actually serve. A recorded loaded window outranks a probe, which reports a figure the target advertises without saying it is what is open, so a resumed session states the window its ledger measured rather than a re-probed server-wide number. Each prompt-manifest record carries that window and the layer that answered it (`contextWindow`, `contextWindowSource`) alongside a `version` for the prompt layout itself, so a recompile whose only cause was the window moving is explained by the record rather than inferred.
23
+
24
+ `PROMPT_MANIFEST_VERSION` (`src/domains/session/prompt-manifest.ts`) is `2` as of this release, and the reordering above is what moved it. The field is additive: a record written by 0.3.8 carries no `version` and reads back as version 1, so a `prompt-manifest.jsonl` from an older session still parses. The rule for the field is that it tracks the layout rather than the inputs. Bump it when the compiled text moves for a reason other than a changed fragment, a changed tool surface, or a changed setting, so that a resumed session has the version in hand to explain the single `promptRecompiled` entry its first compile writes.
25
+
26
+ ### What not to add to the prefix
27
+
28
+ Two additions look free and are not.
29
+
30
+ The first is a terseness rule. It is tempting to cap the prose a model emits between tool calls, because that text is generated tokens on every hop of a long turn. Anthropic measured that exact change on Claude Code and reported a 3 percent quality regression, so a word-count or verbosity limit on inter-tool text is a bad trade: the tokens it saves are the cheapest ones in the turn, and the model's own narration of what it is about to do is load-bearing for what it then does. Bound tool results instead, where a single `grep` can cost thousands of tokens and the envelope caps already do the work.
31
+
32
+ The second is anything that varies with the wall clock or the working tree. No timestamp, no `git status`, no branch name, no session id, and no run id belongs anywhere in the compiled prefix. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so one such field turns the whole prompt into a cache miss on every turn for no information the model could not have asked a tool for. On the sprint's measurement server that is a whole 2,778-token prompt re-prefilled at 2.6 s where the same change behind the stable sections cost 516 tokens and 0.72 s. Volatile facts belong in the user message, in a tool result, or in the runtime block, which is last for this reason.
33
+
16
34
  The disk fragments under `src/domains/prompts/fragments/` are layered by who reads them. `identity.clio` and `operating.contract` are constitutional: they render for every reader, name no tool, and state what is always true about Clio and her harness. `operating.delegation` (fleet coordination, receipts, spot-checks, shared `[worker result]` notes) renders only when `dispatch` is on the session's tool surface, and `operating.skills` (skill-shaped tasks, `/skill <name>` suggestions) only when `context` is; a fragment that teaches a tool is absent when the tool is, the same rule the Fleet block follows. `identity.docs-routing`, the directive to call `context(scope="docs")` before answering a question about Clio herself, follows the `context` gate too, while `identity.self-awareness` (installed paths, code outranks docs, configuration locations) names no tool and is unconditional. `operating.worker` (the assigned-task contract) renders only for dispatched workers, which never see the coordinator fragments. `safety.<level>` states what runs, what is approval-required, and what is blocked at the effective autonomy, in the safety net's action-class vocabulary (read, write, command, `system_modify`, `git_destructive`) and never by tool name, so the same body is true on every surface; the session and every worker read that one body, and what "approval-required" resolves to is the only role text (one operator confirmation for the session, the worker's `onPermission` routing for a worker).
17
35
 
18
36
  Prompt extensions can add dynamic fragments for project rules, the operator profile, and Clio source-tree awareness. Pending skill requests and middleware reminders are visible text in the user message, not hidden prompt machinery.
@@ -1,7 +1,7 @@
1
1
  # Provider Adapter Cookbook
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  This cookbook guides developers through implementing custom model runtimes and inference server integrations within Clio Coder. It explains the runtime descriptor interfaces, probing protocols, model synthesis, and how to configure reasoning and thinking behaviors.
7
7
 
@@ -39,8 +39,14 @@ Run against the exact final candidate with `NO_COLOR` unset and
39
39
  7. `npm run ci` (runs 1 through 6)
40
40
  8. `npm run ci:release` (7 plus `scripts/check-release.mjs`: dist shebang
41
41
  integrity, version coherence between `package.json` and the top
42
- `CHANGELOG.md` heading, the forbidden-file list, the required runtime
43
- resources, and the tarball and unpacked size budgets)
42
+ `CHANGELOG.md` heading, the deterministic 26-scenario behavioral machinery
43
+ corpus against its checked baseline, the forbidden-file list, the required
44
+ runtime resources, and the tarball and unpacked size budgets). A baseline
45
+ mismatch prints reviewable evidence and names prompt- or recipe-affected
46
+ corpus results. For an intentional change, inspect that diff, run
47
+ `node benchmarks/eval/check-behavioral-release.mjs --update` (with `TMPDIR` on a disk-backed path if `/tmp` is a small tmpfs),
48
+ review `benchmarks/eval/behavioral-machinery-baseline.json`, and commit it
49
+ with the change.
44
50
  9. Optional: step 8 again under Node 24. Hosted CI gates on Node 22 alone,
45
51
  the `engines` floor; the weekly `flake-hunt` workflow carries Node 24.
46
52
  Repeat locally only when the cut touches runtime-sensitive code.
@@ -58,7 +64,17 @@ Run against the exact final candidate with `NO_COLOR` unset and
58
64
  12. Install that tarball into a clean temporary prefix with an empty
59
65
  `CLIO_CODER_HOME` and verify `--version`, `--help`, an empty-state non-TTY
60
66
  launch, `doctor`, and `uninstall --dry-run` without developer-local state.
61
- 13. Interactive release testing, which this cut added because the release is
67
+ 13. Before interactive release testing, run the model-required public
68
+ behavioral corpus manually against the release target and built CLI:
69
+ `node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model.yaml --target mini --clio-coder-entry dist/cli/index.js`
70
+ and
71
+ `node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model-negative-control.yaml --target mini --clio-coder-entry dist/cli/index.js`.
72
+ Retain both Artifact v4 files as release evidence. The positive corpus must
73
+ report its scenario and role rows without an undeclared envelope mismatch;
74
+ the negative control must still record violated exploration and safety
75
+ labels. These model-dependent runs are manual and are never required by
76
+ ordinary deterministic CI. Continue with interactive release testing,
77
+ which this cut added because the release is
62
78
  almost entirely interactive surface: a tester agent drives the step-12
63
79
  install through real TUI sessions in a throwaway repository, one session
64
80
  per shipped feature, against local targets for the main session and a
@@ -1,7 +1,7 @@
1
1
  # Clio Coder Safety Model
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio Coder's safety posture is code-enforced, not prompt-only. As the orchestrator coding agent in the [IOWarp](https://iowarp.ai) ecosystem developed by the [Gnosis Research Center](https://grc.iit.edu) at Illinois Tech under NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318), Clio gates execution by target capabilities, the tool registry, the safety policy engine, project policies, protected-artifact checks, and audit receipts.
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Clio Coder Scientific Validation Contracts
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Scientific software development cannot treat simple file presence as proof of correctness. A simulation script that crashes on rank 48, or writes out NetCDF arrays filled with `NaN`s, may still successfully write a file to the disk.
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Skills Marketplace
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  The Skills Hub (`/skill`) shows project skills, user skills, and the marketplace. Every marketplace row comes from the same local lookup that `clio-coder skills install <name>` and `/skill <name>` resolve through, so the hub lists nothing it cannot install.
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Tool Usage Reference
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Trace store contract
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  Clio's trace database is a rebuildable, queryable mirror. Receipts, session
7
7
  ledgers, gate artifacts, and evidence remain the source of truth. Removing
@@ -26,6 +26,93 @@ This guide provides concrete, actionable remediation procedures for operational
26
26
 
27
27
  ---
28
28
 
29
+ ## Reading a cold cache
30
+
31
+ On a local server prefill is most of what a turn costs, so a cold prefix cache is the difference between a first token in under a second and one in fifty. This is how to find out why a turn went cold, starting from what the TUI shows.
32
+
33
+ **1. Read the `/context` cache lines.** Two lines answer different questions. The prompt-cache line says what the provider reported and whether the compiled prompt shell was reused. The prefill line says what the server itself did:
34
+
35
+ ```text
36
+ prefill: 34,951 uncached · 0 cached · 48,617 ms
37
+ ```
38
+
39
+ Those are the server's own numbers, not Clio's estimate. `server does not report cache reads` in place of the cached figure means the backend gave no `cache_n` at all, which is LM Studio 2.29.0's OpenAI-compatible port today; on that target the verdict comes from the provider's `cached_tokens` instead and the prefill line reports only total prompt work and milliseconds.
40
+
41
+ **2. Look for the expected-cold line.** When the last settled run came back `cold` and Clio had recorded a cause, `/context` names it rather than warning:
42
+
43
+ ```text
44
+ last cold turn: working-set eviction (expected)
45
+ ```
46
+
47
+ The eight causes and what stamps each one are in [context-engine.md](context-engine.md#cache-divergence-honesty). `background_memory` renders in prose as `last cold turn: background memory step (expected)`.
48
+
49
+ **3. Confirm it in the ledger.** The reasons are durable, so a finished session answers the same question without the TUI. Open `current.jsonl` under the session directory `clio-coder paths` reports and read the run's first assistant entry:
50
+
51
+ ```json
52
+ {
53
+ "timing": { "ttftMs": 53194, "apiMs": 56770 },
54
+ "promptCache": {
55
+ "input": 34951, "cacheRead": 0, "cacheWrite": 0,
56
+ "backendVerdict": "cold",
57
+ "expectedColdReasons": ["dispatch", "residency"],
58
+ "backend": {
59
+ "promptTokens": 34951, "cachedTokens": 0, "predictedTokens": 24,
60
+ "promptMs": 48617, "predictedMs": 373, "source": "llamacpp-timings"
61
+ }
62
+ }
63
+ }
64
+ ```
65
+
66
+ `expectedColdReasons` is stamped once per run, on its first persisted call, so a turn with several model calls carries it on the first one only. `clio-coder doctor` folds the latest session for you and prints the verdict counts plus the most frequent reason, and `clio-coder usage report` gives per-session uncached prefill and verdict counts across the window.
67
+
68
+ **4. When there is no reason, the warning is the finding.** A cold backend with a reused prompt shell and no recorded reason is a real disagreement: Clio kept the bytes stable and the server re-prefilled anyway. `/context` leaves the warning in place for exactly that case. Four causes are worth checking in order, and none of them is a Clio bug:
69
+
70
+ - **The server slept.** A llama.cpp router started with `--sleep-idle-seconds N` drops the slot's prefix cache when it sleeps, and `--cache-ram` does not reliably restore a large state. A gap longer than that setting between two turns explains a cold turn completely. Raise the flag, or accept that a session left idle pays for its first turn back.
71
+ - **Something else used the endpoint.** A worker, a second Clio session, or another client on the same server evicts the slot. Clio stamps `dispatch`, `residency`, and `background_memory` only for work it can attribute to itself on that endpoint; a foreign process leaves no stamp. `clio-coder targets --probe` reports the endpoint's slot count, and `/fleet` settings show active slots per endpoint.
72
+ - **The model was swapped.** A router serving one model at a time reloads on a residency change, and everything the previous model had cached is gone. This normally does stamp `residency`, but only when the mutation went through Clio.
73
+ - **The prompt moved for a reason Clio did not classify.** Compare the run's `promptHash` and `toolSignature` in `context-snapshots.jsonl` against the previous run's. Equal hashes with a cold backend point at the server; different hashes with no `prompt_recompiled` or `tool_surface_change` stamp is worth an issue.
74
+
75
+ One case is expected on hybrid architectures and looks like a bug. Qwen3.8 keeps recurrent state that llama.cpp cannot roll back to an arbitrary token, so a change anywhere inside a cached prefix re-prefills from the last context checkpoint rather than from the changed byte. A small edit to old history can therefore cost thousands of tokens of prefill with the prompt hash otherwise stable. The server's checkpoint count and its `--checkpoint-min-step` are the levers; see the `qwen3.8-27b` family's `serving` and `measuredUnder` notes in `src/domains/providers/models/local-models/clio-local-coding-targets.yaml` for the measured figures and the exact argv they were taken under.
76
+
77
+ ---
78
+
79
+ ## A TUI that stops answering the keyboard
80
+
81
+ When an interactive session stops responding to typing, the question worth
82
+ answering before anything else is which half of the input pipeline stopped: the
83
+ stdin reader that hands bytes to the application, or the renderer that turns
84
+ them into a frame on stdout. Clio keeps that evidence without being asked. Every
85
+ interactive process holds a bounded in-memory ring of the last 256 input-ingress
86
+ records and the last 256 committed frames, and writes it out when the process
87
+ receives `SIGTERM`, which is the signal a `kill` of the stuck pane sends.
88
+
89
+ The dump lands in the state directory `clio-coder paths` reports:
90
+
91
+ ```text
92
+ <stateDir>/input-wedge/<ISO timestamp>-<pid>.json
93
+ ```
94
+
95
+ The five newest dumps are kept and older ones are removed as new ones land.
96
+ Read `classification` first:
97
+
98
+ | `classification` | What it means |
99
+ | :--- | :--- |
100
+ | `input-not-committed` | Bytes reached the application and no frame carrying them ever reached stdout. The renderer is the stuck half. |
101
+ | `no-input-recorded` | Nothing was delivered at all. If the operator was typing, the stdin reader is the stuck half. |
102
+ | `input-committed` | Both halves were moving. Whatever the session was doing, it was not this pipeline. |
103
+
104
+ `msSinceLastInputIngress` and `msSinceLastCommittedFrame` say how long each half
105
+ had been quiet when the signal arrived, and the `inputIngress` and `frames`
106
+ arrays carry the records themselves. Frames are kept only when they reached
107
+ stdout, so an empty `frames` array is itself a finding.
108
+
109
+ For a full session trace rather than the tail, set `CLIO_CODER_RENDER_TRACE` to
110
+ a file path before starting the session. That writes every record, including
111
+ provider deltas and terminal writes, as JSONL. The ring is the always-on subset
112
+ of the same records, for the case where nobody armed the trace first.
113
+
114
+ ---
115
+
29
116
  ## Diagnostic Commands
30
117
 
31
118
  When encountering unexpected system behavior:
@@ -1,7 +1,7 @@
1
1
  # Clio TUI Design System
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  This document is the reference specification for the Clio Coder TUI visual layout, styling, and behavior. It describes color semantics, the glyph vocabulary, structural recipes, and state choreography for all surfaces under [src/interactive/](../src/interactive/).
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Worker Dispatch Mechanics
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.8).
4
+ > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.9).
5
5
 
6
6
  This document describes the design and lifecycle of Clio Coder dispatched workers, focusing on the spawning sequence, execution isolation, the standard input/output NDJSON communication loop, and permission escalation routing.
7
7
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iowarp/clio-coder",
3
- "version": "0.3.8",
3
+ "version": "0.3.9",
4
4
  "description": "Coding agent for HPC and scientific-software developers, part of IOWarp's CLIO ecosystem of agentic science.",
5
5
  "keywords": [
6
6
  "ai",
@@ -92,6 +92,7 @@
92
92
  "//": "below here: a real model target, chosen with --target <id>; costs money and/or GPU time, never run in CI",
93
93
  "live:smoke": "node --import tsx benchmarks/internal/live-smoke.ts",
94
94
  "live:fleet-dispatch": "node --import tsx benchmarks/internal/live-fleet-dispatch.ts",
95
+ "live:release-residue": "node --import tsx tests/live/release-residue.ts",
95
96
  "live:tui": "node --import tsx benchmarks/internal/pty-drive.ts",
96
97
  "live:home": "node --import tsx benchmarks/internal/live-home.ts"
97
98
  },
package/src/cli/agents.ts CHANGED
@@ -9,7 +9,7 @@ import { printError } from "./shared.js";
9
9
 
10
10
  const HELP = `clio-coder agents [--json] [--all]
11
11
 
12
- List user-facing agent specs from built-in, user, and project recipes.
12
+ List user-facing agent specs from built-in, extension, user, and project recipes.
13
13
 
14
14
  Flags:
15
15
  --json emit specs as JSON instead of the formatted table