@iowarp/clio-coder 0.3.3 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/CHANGELOG.md +74 -0
  2. package/CONTRIBUTING.md +7 -7
  3. package/README.md +3 -3
  4. package/dist/{acp-P2AQILE2.js → acp-2BEHC4DL.js} +9 -8
  5. package/dist/{agents-72W3BI7I.js → agents-LNNFTM53.js} +29 -24
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5TWEIYDN.js → auth-KXXFI2VS.js} +14 -10
  8. package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
  9. package/dist/{chunk-YCWGATWI.js → chunk-24I7BN55.js} +2 -2
  10. package/dist/{chunk-EKMEHE4H.js → chunk-33YXPOE3.js} +2 -3
  11. package/dist/chunk-3BPUFZDL.js +37 -0
  12. package/dist/chunk-43AOLP7E.js +375 -0
  13. package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
  14. package/dist/{chunk-6SGHMWE3.js → chunk-5JGRAMKL.js} +5 -5
  15. package/dist/{chunk-V6RTAOC2.js → chunk-6US73PDB.js} +572 -51
  16. package/dist/{chunk-5UFT4SUX.js → chunk-6XXKFVSN.js} +3 -3
  17. package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
  18. package/dist/chunk-AOCYTWAV.js +449 -0
  19. package/dist/chunk-CFGTUFWB.js +67 -0
  20. package/dist/chunk-CJUB2JJ2.js +1478 -0
  21. package/dist/{chunk-FNTMWMX5.js → chunk-CKXWIANG.js} +14 -12
  22. package/dist/{chunk-PIWWS5BL.js → chunk-CYQKWTG3.js} +63 -78
  23. package/dist/{chunk-LZSJBIVT.js → chunk-DJVECN66.js} +271 -762
  24. package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
  25. package/dist/{chunk-ZWMF7253.js → chunk-E2ER4LJF.js} +304 -9
  26. package/dist/{chunk-STBPMHSX.js → chunk-EKY57CSP.js} +51 -84
  27. package/dist/{chunk-DUYJ5IO6.js → chunk-EYPA3EGJ.js} +12 -4
  28. package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
  29. package/dist/chunk-FYYLNIL5.js +313 -0
  30. package/dist/{chunk-OQ33BKR3.js → chunk-G7MUEIGA.js} +3 -60
  31. package/dist/chunk-GEYXPTRF.js +613 -0
  32. package/dist/chunk-GOXNB3AO.js +261 -0
  33. package/dist/{chunk-G4BMMOKF.js → chunk-HVDIIIQW.js} +2 -2
  34. package/dist/chunk-HWUFFB6L.js +83 -0
  35. package/dist/{chunk-4XUGQOHA.js → chunk-K7T3E2SR.js} +15 -8
  36. package/dist/chunk-K7VKOLQQ.js +15 -0
  37. package/dist/{chunk-UFIIWP2H.js → chunk-KHSFENX2.js} +8 -8
  38. package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
  39. package/dist/chunk-LCGCVYZ4.js +57 -0
  40. package/dist/chunk-LL4KHSZI.js +22 -0
  41. package/dist/{chunk-PAJK6MAQ.js → chunk-LYF7OHWH.js} +42 -15
  42. package/dist/{chunk-POHLU5DW.js → chunk-M6L6IDJG.js} +3 -3
  43. package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
  44. package/dist/{chunk-X4RCMKVQ.js → chunk-NDINPTJ4.js} +2 -2
  45. package/dist/{chunk-TZK7PACC.js → chunk-NILBFAPG.js} +14 -8
  46. package/dist/chunk-ODFEOB4F.js +1082 -0
  47. package/dist/{chunk-AGYYIBLL.js → chunk-OH3TOQTB.js} +6 -2
  48. package/dist/chunk-OZNBF4L3.js +23 -0
  49. package/dist/{chunk-DSELYM6W.js → chunk-PBTHKCPN.js} +30 -10
  50. package/dist/{verify-375KUB3Y.js → chunk-PCZJO5TI.js} +127 -42
  51. package/dist/{chunk-ED4KHGC3.js → chunk-PPAMZ32Z.js} +9 -2
  52. package/dist/{chunk-SRDMMSEP.js → chunk-QM3F2GKX.js} +1063 -1645
  53. package/dist/{chunk-X6IAEBZR.js → chunk-QNQHSOLF.js} +7 -7
  54. package/dist/{chunk-OC7FIQPC.js → chunk-R46L2BIR.js} +10 -7
  55. package/dist/{chunk-2TLUCQVG.js → chunk-RD5U66HV.js} +3 -3
  56. package/dist/{chunk-6N5PTWMY.js → chunk-RY3LY4J5.js} +50 -13
  57. package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
  58. package/dist/{chunk-OOJYHWRB.js → chunk-TSHXZTOQ.js} +6 -5
  59. package/dist/chunk-TZSKNMZG.js +434 -0
  60. package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
  61. package/dist/{chunk-VJWL6YS5.js → chunk-UUVG37B4.js} +2 -2
  62. package/dist/{chunk-COU2UHX6.js → chunk-VEZEGCGW.js} +170 -2
  63. package/dist/chunk-W6GROXXM.js +69 -0
  64. package/dist/{chunk-OAO4GE4M.js → chunk-WHGPSPT5.js} +2 -2
  65. package/dist/chunk-WHJYKASB.js +677 -0
  66. package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
  67. package/dist/{chunk-YHZX5GEU.js → chunk-XAKHZX5N.js} +2 -2
  68. package/dist/{chunk-TZTZS7QK.js → chunk-XE2VEJHX.js} +5 -3
  69. package/dist/{chunk-LM5TQCJZ.js → chunk-XF5N4U5A.js} +8 -7
  70. package/dist/{chunk-LWLEKMDQ.js → chunk-XXQNGV4M.js} +1073 -552
  71. package/dist/{chunk-KZWTDYJF.js → chunk-XYDYPRZI.js} +7 -7
  72. package/dist/chunk-ZGVHUX3M.js +66 -0
  73. package/dist/{chunk-LW6DSM3M.js → chunk-ZRGEBJ4T.js} +1192 -1119
  74. package/dist/{chunk-2DJ2KNFG.js → chunk-ZXF4XRKW.js} +202 -40
  75. package/dist/chunk-ZZMN5OM4.js +122 -0
  76. package/dist/cli/index.js +34 -30
  77. package/dist/{clio-JOU4FXVA.js → clio-M2KGYUFZ.js} +7 -6
  78. package/dist/{code-nav-7AX6FYE6.js → code-nav-GQNL7XA6.js} +8 -6
  79. package/dist/codewiki/build-worker.js +4 -4
  80. package/dist/{components-KELWS457.js → components-5TTYYX6G.js} +3 -3
  81. package/dist/{config-XCDVKR23.js → config-XUUYQIWO.js} +47 -35
  82. package/dist/{configure-4GAP54ZW.js → configure-IHJ7YOMV.js} +18 -15
  83. package/dist/{context-77FM5DV5.js → context-74JLXAWD.js} +18 -10
  84. package/dist/{context-4UOGGLQ5.js → context-75MIWW3U.js} +41 -29
  85. package/dist/{context-5VKGUVJJ.js → context-ZQ7SIFJV.js} +85 -9
  86. package/dist/{context-clear-XXJRLCJJ.js → context-clear-GYKWNUML.js} +41 -29
  87. package/dist/{context-index-BZ4UYMTC.js → context-index-SSR5ECNE.js} +3 -3
  88. package/dist/context-working-set-UX5KEP4J.js +1553 -0
  89. package/dist/{dispatch-runner-QPRDDBDX.js → dispatch-runner-GIJBHNFL.js} +47 -32
  90. package/dist/{docs-2C2LTVT2.js → docs-6FZSCG5B.js} +3 -3
  91. package/dist/{doctor-HR46URBJ.js → doctor-SVJ5BZCW.js} +12 -12
  92. package/dist/{eval-XSSNATB4.js → eval-CG6LLBLD.js} +54 -238
  93. package/dist/{evidence-6HG2PY2B.js → evidence-ZYFIEN42.js} +57 -28
  94. package/dist/{evolve-K7YU3NCY.js → evolve-QGEXEMDW.js} +36 -25
  95. package/dist/{extensions-QVDOHDGJ.js → extensions-ADGNCJJD.js} +3 -3
  96. package/dist/{fleet-VY3HHKN6.js → fleet-S5R4ZOQY.js} +73 -44
  97. package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-BHSNPBMH.js} +3 -3
  98. package/dist/{init-JYGXI3FK.js → init-5DRU55YR.js} +49 -37
  99. package/dist/memory-7YKKR6UC.js +467 -0
  100. package/dist/{models-I5QWSEOM.js → models-ZPOLRU2C.js} +24 -21
  101. package/dist/{monitor-GE4ID3IA.js → monitor-US5F5YGZ.js} +73 -46
  102. package/dist/{orchestrator-EM5MC3HM.js → orchestrator-E2AL4T5N.js} +1624 -1007
  103. package/dist/{paths-UXLN5YYZ.js → paths-E7KYAQWE.js} +3 -3
  104. package/dist/{reset-L2FQEE3E.js → reset-KZ652EK6.js} +6 -5
  105. package/dist/{run-ZU3QMZPZ.js → run-SRNBKDWD.js} +76 -54
  106. package/dist/{share-S5BZQC5I.js → share-CGZE33UP.js} +7 -6
  107. package/dist/{skills-X5VXCRNQ.js → skills-S2X4DLY5.js} +4 -4
  108. package/dist/{skills-eval-WKIHWTHR.js → skills-eval-W2GGIC4R.js} +40 -29
  109. package/dist/{targets-SNCPI2NR.js → targets-54SWINWB.js} +28 -23
  110. package/dist/{terminal-lease-BNAHVHBS.js → terminal-lease-SAIF2OGY.js} +6 -4
  111. package/dist/{uninstall-FZCQCDKC.js → uninstall-BVLWXKBT.js} +3 -3
  112. package/dist/{upgrade-JQHHPQ4K.js → upgrade-JKAR27XC.js} +20 -19
  113. package/dist/{usage-OR4O5SMZ.js → usage-MSAWCLX4.js} +79 -36
  114. package/dist/verifiers-NCBTHHN2.js +1220 -0
  115. package/dist/verify-X5HDROLA.js +25 -0
  116. package/dist/{wiki-generate-UEXP2ARI.js → wiki-generate-GUSOQ6ZP.js} +50 -37
  117. package/dist/worker/entry.js +90 -70
  118. package/dist/{workspace-G4ZWUIPR.js → workspace-ZJ6BFM3Q.js} +4 -4
  119. package/docs/README.md +8 -7
  120. package/docs/acp.md +1 -1
  121. package/docs/alcf-provider.md +1 -1
  122. package/docs/architecture.md +2 -2
  123. package/docs/artifact-placement.md +1 -2
  124. package/docs/artifact-versions.md +1 -1
  125. package/docs/built-in-agents.md +1 -1
  126. package/docs/capacity-and-scheduling.md +1 -1
  127. package/docs/commands-and-modes.md +60 -26
  128. package/docs/config-knobs-audit.md +1 -2
  129. package/docs/configuration-and-targets.md +26 -1
  130. package/docs/context-engine.md +67 -13
  131. package/docs/context-working-set.md +194 -0
  132. package/docs/development-pipeline.md +1 -1
  133. package/docs/documentation-coverage.md +6 -6
  134. package/docs/documentation-guide.md +7 -6
  135. package/docs/environment-variables.md +2 -1
  136. package/docs/eval-runner.md +1 -1
  137. package/docs/evals-internal.md +4 -32
  138. package/docs/evidence-and-memory.md +139 -7
  139. package/docs/evolution.md +1 -1
  140. package/docs/exit-codes-and-output.md +1 -1
  141. package/docs/extensions-and-sharing.md +2 -2
  142. package/docs/fleet-dispatch.md +49 -8
  143. package/docs/glossary.md +21 -1
  144. package/docs/installation-and-lifecycle.md +2 -2
  145. package/docs/middleware-and-components.md +19 -2
  146. package/docs/model-catalog.md +7 -9
  147. package/docs/observability.md +4 -4
  148. package/docs/proactive-memory.md +26 -16
  149. package/docs/prompt-envelope-and-tools.md +7 -5
  150. package/docs/provider-adapter-cookbook.md +1 -1
  151. package/docs/release-cut-checklist.md +43 -40
  152. package/docs/safety-model.md +49 -8
  153. package/docs/scientific-validation.md +21 -3
  154. package/docs/session-lifecycle.md +3 -3
  155. package/docs/skills-marketplace.md +1 -1
  156. package/docs/tool-usage.md +79 -12
  157. package/docs/trace-store.md +1 -1
  158. package/docs/troubleshooting.md +1 -1
  159. package/docs/tui-design.md +38 -4
  160. package/docs/worker-dispatch-mechanics.md +11 -1
  161. package/package.json +13 -13
  162. package/skills/meta/clio-test/SKILL.md +20 -17
  163. package/skills/meta/clio-test/evals.md +3 -3
  164. package/skills/meta/clio-test/references/harness.md +35 -6
  165. package/skills/meta/clio-test/references/test-map.md +20 -10
  166. package/skills/registry.yaml +2 -2
  167. package/skills/skill-marketplace.json +1 -1
  168. package/src/cli/agents.ts +2 -3
  169. package/src/cli/argv.ts +14 -1
  170. package/src/cli/context-working-set.ts +513 -0
  171. package/src/cli/context.ts +8 -0
  172. package/src/cli/evidence.ts +20 -2
  173. package/src/cli/fleet.ts +15 -0
  174. package/src/cli/index.ts +5 -1
  175. package/src/cli/memory.ts +272 -10
  176. package/src/cli/modes/json-stream.ts +2 -2
  177. package/src/cli/modes/print.ts +12 -1
  178. package/src/cli/run.ts +22 -2
  179. package/src/cli/targets.ts +12 -3
  180. package/src/cli/usage.ts +55 -7
  181. package/src/cli/verifiers.ts +325 -0
  182. package/src/core/bash-exec.ts +39 -14
  183. package/src/core/bus-events.ts +22 -4
  184. package/src/core/config.ts +54 -0
  185. package/src/core/defaults.ts +50 -3
  186. package/src/core/response-model-id.ts +134 -0
  187. package/src/core/toml.ts +62 -0
  188. package/src/core/verification-scripts.ts +6 -0
  189. package/src/core/workspace-files.ts +0 -1
  190. package/src/domains/agents/builtins/architect.md +1 -1
  191. package/src/domains/agents/builtins/verifier.md +3 -0
  192. package/src/domains/agents/catalog.ts +5 -4
  193. package/src/domains/agents/recipe.ts +54 -14
  194. package/src/domains/agents/result-contract.ts +7 -4
  195. package/src/domains/config/classify.ts +1 -0
  196. package/src/domains/context/bootstrap.ts +36 -27
  197. package/src/domains/context/project-metadata.ts +19 -63
  198. package/src/domains/context/prompt-context.ts +8 -0
  199. package/src/domains/context/working-set/contract.ts +161 -0
  200. package/src/domains/context/working-set/defaults.ts +28 -0
  201. package/src/domains/context/working-set/engine.ts +203 -0
  202. package/src/domains/context/working-set/fold.ts +62 -0
  203. package/src/domains/context/working-set/horizon.ts +38 -0
  204. package/src/domains/context/working-set/marker.ts +103 -0
  205. package/src/domains/context/working-set/path-index.ts +436 -0
  206. package/src/domains/context/working-set/payload.ts +152 -0
  207. package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
  208. package/src/domains/context/working-set/policies/index.ts +20 -0
  209. package/src/domains/context/working-set/policies/structural.ts +160 -0
  210. package/src/domains/context/working-set/project.ts +132 -0
  211. package/src/domains/context/working-set/protect.ts +109 -0
  212. package/src/domains/context/working-set/recall.ts +177 -0
  213. package/src/domains/context/working-set/replay/controls.ts +112 -0
  214. package/src/domains/context/working-set/replay/load-clio.ts +199 -0
  215. package/src/domains/context/working-set/replay/metrics.ts +185 -0
  216. package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
  217. package/src/domains/context/working-set/replay/report.ts +139 -0
  218. package/src/domains/context/working-set/replay/runner.ts +325 -0
  219. package/src/domains/context/working-set/replay/synthetic.ts +422 -0
  220. package/src/domains/context/working-set/replay/trace.ts +21 -0
  221. package/src/domains/context/working-set/visible.ts +54 -0
  222. package/src/domains/dispatch/budget-envelope.ts +396 -0
  223. package/src/domains/dispatch/contract.ts +2 -0
  224. package/src/domains/dispatch/extension.ts +81 -27
  225. package/src/domains/dispatch/orphan-recovery.ts +1 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +4 -0
  227. package/src/domains/dispatch/state.ts +1 -0
  228. package/src/domains/dispatch/types.ts +10 -3
  229. package/src/domains/dispatch/validation.ts +14 -0
  230. package/src/domains/dispatch/worker-spawn.ts +14 -3
  231. package/src/domains/eval/metrics/evidence.ts +0 -116
  232. package/src/domains/eval/metrics/invariants.ts +1 -1
  233. package/src/domains/eval/runners/clio-run.ts +1 -10
  234. package/src/domains/eval/runners/external-command.ts +2 -29
  235. package/src/domains/eval/schema/suite.ts +0 -7
  236. package/src/domains/eval/suites/run.ts +1 -7
  237. package/src/domains/evidence/build.ts +112 -45
  238. package/src/domains/evidence/eval.ts +24 -7
  239. package/src/domains/evidence/index.ts +53 -0
  240. package/src/domains/evidence/ordering.ts +12 -0
  241. package/src/domains/evidence/run-trust.ts +221 -0
  242. package/src/domains/evidence/store.ts +46 -6
  243. package/src/domains/evidence/trust-status.ts +854 -0
  244. package/src/domains/evidence/types.ts +26 -0
  245. package/src/domains/memory/index.ts +22 -0
  246. package/src/domains/memory/operations.ts +58 -1
  247. package/src/domains/memory/promotion.ts +281 -0
  248. package/src/domains/memory/prompt-section.ts +25 -5
  249. package/src/domains/memory/proposal.ts +51 -7
  250. package/src/domains/memory/task-bank.ts +3 -2
  251. package/src/domains/memory/task-memory-handoff.ts +181 -24
  252. package/src/domains/memory/task-memory-policy.ts +3 -1
  253. package/src/domains/memory/types.ts +37 -0
  254. package/src/domains/memory/validate.ts +178 -0
  255. package/src/domains/middleware/memory-intervention.ts +38 -25
  256. package/src/domains/middleware/runtime.ts +6 -0
  257. package/src/domains/middleware/skills-reminder.ts +19 -4
  258. package/src/domains/middleware/stalled-turn.ts +208 -5
  259. package/src/domains/middleware/types.ts +10 -0
  260. package/src/domains/observability/contract.ts +6 -1
  261. package/src/domains/observability/cost.ts +20 -4
  262. package/src/domains/observability/extension.ts +2 -2
  263. package/src/domains/providers/index.ts +3 -0
  264. package/src/domains/providers/model-discovery.ts +9 -0
  265. package/src/domains/providers/runtime-resolution.ts +38 -1
  266. package/src/domains/providers/runtimes/common/probe-helpers.ts +97 -16
  267. package/src/domains/providers/types/context-window-slots.ts +18 -0
  268. package/src/domains/providers/types/runtime-descriptor.ts +3 -1
  269. package/src/domains/safety/autonomy.ts +1 -1
  270. package/src/domains/safety/call-target.ts +211 -14
  271. package/src/domains/safety/decision-presentation.ts +268 -0
  272. package/src/domains/safety/default-path-policy.ts +8 -0
  273. package/src/domains/safety/finish-contract.ts +4 -3
  274. package/src/domains/safety/policy-engine.ts +48 -6
  275. package/src/domains/safety/redaction.ts +73 -0
  276. package/src/domains/session/compaction/compact.ts +23 -1
  277. package/src/domains/session/compaction/cut-point.ts +2 -0
  278. package/src/domains/session/compaction/tokens.ts +16 -1
  279. package/src/domains/session/context-ledger.ts +12 -1
  280. package/src/domains/session/decision-board.ts +4 -0
  281. package/src/domains/session/entries.ts +110 -1
  282. package/src/domains/session/history.ts +68 -19
  283. package/src/domains/session/manager.ts +9 -2
  284. package/src/domains/session/migrations/index.ts +22 -3
  285. package/src/domains/session/usage.ts +24 -7
  286. package/src/engine/acp/event-mapper.ts +7 -0
  287. package/src/engine/acp/server.ts +32 -2
  288. package/src/engine/agent.ts +18 -1
  289. package/src/engine/apis/lmstudio.ts +25 -4
  290. package/src/engine/apis/openai-completions.ts +147 -22
  291. package/src/engine/claude/sdk-runtime.ts +8 -2
  292. package/src/engine/claude/tool-safety.ts +13 -0
  293. package/src/engine/loop-guard.ts +27 -3
  294. package/src/engine/session.ts +9 -3
  295. package/src/engine/worker-events.ts +4 -3
  296. package/src/engine/worker-runtime.ts +59 -54
  297. package/src/entry/orchestrator.ts +34 -5
  298. package/src/interactive/chat-loop-messages.ts +40 -6
  299. package/src/interactive/chat-loop.ts +13 -0
  300. package/src/interactive/chat-panel.ts +17 -1
  301. package/src/interactive/chat-renderer.ts +49 -24
  302. package/src/interactive/clio-editor.ts +44 -7
  303. package/src/interactive/context-meter.ts +10 -0
  304. package/src/interactive/context-overlay.ts +120 -7
  305. package/src/interactive/context-recall-command.ts +110 -0
  306. package/src/interactive/cost-overlay.ts +39 -8
  307. package/src/interactive/dispatch-board.ts +212 -35
  308. package/src/interactive/footer/widgets.ts +13 -0
  309. package/src/interactive/interactive-application.ts +6 -1
  310. package/src/interactive/interactive-input-runtime.ts +11 -1
  311. package/src/interactive/interactive-presentation.ts +11 -1
  312. package/src/interactive/interactive-slash-runtime.ts +37 -1
  313. package/src/interactive/memory-overlay.ts +89 -4
  314. package/src/interactive/model-session-replay.ts +21 -0
  315. package/src/interactive/overlay-ask-user-lifecycle.ts +1 -1
  316. package/src/interactive/overlay-frame.ts +5 -2
  317. package/src/interactive/overlay-general-openers.ts +46 -1
  318. package/src/interactive/overlay-key-routing.ts +41 -1
  319. package/src/interactive/overlay-lifecycle.ts +11 -4
  320. package/src/interactive/overlay-permission-lifecycle.ts +23 -8
  321. package/src/interactive/overlay-session-lifecycle.ts +8 -4
  322. package/src/interactive/overlay-transitions.ts +11 -0
  323. package/src/interactive/overlays/ask-user.ts +74 -30
  324. package/src/interactive/overlays/decisions.ts +3 -1
  325. package/src/interactive/permission-hint.ts +35 -0
  326. package/src/interactive/permission-overlay.ts +95 -45
  327. package/src/interactive/renderers/tool-execution.ts +37 -51
  328. package/src/interactive/session-last-turn.ts +8 -1
  329. package/src/interactive/session-transcript.ts +2 -2
  330. package/src/interactive/session-usage-reseed.ts +36 -10
  331. package/src/interactive/slash-commands.ts +31 -4
  332. package/src/interactive/status/summary.ts +5 -0
  333. package/src/interactive/status/types.ts +5 -0
  334. package/src/interactive/terminal-lease.ts +1 -0
  335. package/src/interactive/turn-context.ts +333 -110
  336. package/src/interactive/turn-middleware.ts +7 -6
  337. package/src/interactive/turn-runtime.ts +37 -8
  338. package/src/interactive/turn-state.ts +3 -0
  339. package/src/interactive/worker-progress.ts +440 -0
  340. package/src/interactive/worker-stream.ts +51 -110
  341. package/src/tools/agent-tools.ts +39 -7
  342. package/src/tools/ask-user.ts +21 -1
  343. package/src/tools/bash.ts +144 -82
  344. package/src/tools/builtin-tool-catalog.ts +11 -5
  345. package/src/tools/context/index.ts +107 -5
  346. package/src/tools/context/surface.ts +3 -2
  347. package/src/tools/core-bootstrap.ts +21 -0
  348. package/src/tools/dispatch-arguments.ts +8 -0
  349. package/src/tools/dispatch-event-text.ts +19 -0
  350. package/src/tools/dispatch-runner.ts +9 -7
  351. package/src/tools/dispatch.ts +24 -1
  352. package/src/tools/monitor.ts +43 -20
  353. package/src/tools/registry.ts +72 -10
  354. package/src/tools/result-disposition.ts +706 -0
  355. package/src/tools/result-shaping.ts +321 -20
  356. package/src/tools/safe-exec.ts +2 -0
  357. package/src/tools/verify/authoring.ts +1120 -0
  358. package/src/tools/verify/catalog.ts +346 -0
  359. package/src/tools/verify/index.ts +13 -3
  360. package/src/tools/verify/scripts.ts +135 -37
  361. package/src/tools/verify/surface.ts +9 -5
  362. package/src/tools/worker-evidence.ts +54 -12
  363. package/src/worker/spec-contract.ts +43 -3
  364. package/dist/chunk-J7CWMCQD.js +0 -255
  365. package/dist/chunk-T6YILFSB.js +0 -80
  366. package/dist/chunk-VAKQQHWR.js +0 -434
  367. package/dist/chunk-VPAYEGVX.js +0 -184
  368. package/dist/chunk-XBXAASKX.js +0 -18
  369. package/dist/memory-WFZMGYHX.js +0 -236
  370. package/src/domains/eval/metrics/chaos-stream.ts +0 -93
@@ -1,7 +1,7 @@
1
1
  # Commands and Modes
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/commands_blueprint.html](html/commands_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/commands_blueprint.html](html/commands_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
 
7
7
  Clio Coder is a terminal-first alpha harness. This page keeps the command
@@ -56,7 +56,7 @@ For process exit codes, stdout deliverable guarantees, and machine-readable JSON
56
56
  | `clio-coder dev components diff --from <a> --to <b> [--json]` | Compare component snapshots. |
57
57
  | `clio-coder evidence build\|inspect\|list` | Build and inspect deterministic evidence artifacts. |
58
58
  | `clio-coder eval validate\|run\|report\|compare\|gate` | Validate, run, report, compare, and gate local evaluation suites (Suite v2). |
59
- | `clio-coder memory list\|propose\|approve\|reject\|prune` | Manage scoped, evidence-linked memory records. |
59
+ | `clio-coder memory list\|propose\|promote\|approve\|reject\|prune` | Manage scoped, evidence-linked memory records. |
60
60
  | `clio-coder trace runs [--db PATH] [--limit N] [--json]` | List runs recorded in the durable trace mirror beside the ledger. |
61
61
  | `clio-coder trace phases <runId> [--db PATH]` | Show one run's recorded phases. |
62
62
  | `clio-coder trace tail <runId> [--follow] [--db PATH]` | Tail one run's recorded events; `--follow` streams as they land. |
@@ -78,6 +78,8 @@ For process exit codes, stdout deliverable guarantees, and machine-readable JSON
78
78
  | `clio-coder context wiki [--update] [--status] [--depth auto\|simple\|medium\|detailed] [--target <id>] [--model <id>] [--thinking off\|low\|medium\|high]` | Generate, update, or inspect the agent-authored Markdown wiki under `.clio-coder/wiki/`. |
79
79
  | `clio-coder context reset [--all] [--yes]` | Clear accumulated project context artifacts; `--all` also removes `CLIO-CODER.md`. `--yes` (or `-y`) answers every confirmation and is required when stdin is not a terminal. |
80
80
  | `clio-coder context index [--json]` | Build the structural codewiki index without model calls; writes `.clio-coder/codewiki.json` and `.clio-coder/state.json` and prints coverage plus a structural hash. |
81
+ | `clio-coder context replay (--sessions <path>... \| --synthetic <ids>) [--policies <ids>] [--budgets <tokens>] [--threshold <ratio>] [--target <ratio>] [--protect-last-turns <n>] [--min-evictable-tokens <n>] [--seed <n>] [--no-filter] [--json <out>] [--md <out>]` | Replay working-set policies over Clio session ledgers or the seeded procedural corpora and report retention, precision, token savings, recall cost, cold-prefix cost, saturation, and summary headroom. |
82
+ | `clio-coder context working-set --session <id\|path>` | Inspect one session's durable working-set fold and path-index summary without modifying the ledger. |
81
83
 
82
84
  ## Headless Run Flags
83
85
 
@@ -152,11 +154,11 @@ The registry table below lists the available interactive slash commands. On a ba
152
154
  | `/agents` | `/agents` | List Clio agents and ACP delegation agents |
153
155
  | `/targets` | `/targets` | Open Settings → Targets: health, use, connect, probe, remove |
154
156
  | `/cost` | `/cost` | Show session token and cost totals |
155
- | `/context` | `/context compact [instructions] \| /context init \| /context refresh \| /context reset` | Context hub: window overlay plus compact, init, refresh, and reset |
157
+ | `/context` | `/context compact [instructions] \| /context recall <ref> \| /context init \| /context refresh \| /context reset` | Context hub: window overlay plus compact, recall, init, refresh, and reset |
156
158
  | `/fleet` | `/fleet` | Open Settings → Fleet: defaults, profiles, agent bindings, nodes |
157
159
  | `/decisions` | `/decisions` | Show settled interview decisions and operator revisions |
158
160
  | `/tasks` | `/tasks add <text> \| /tasks hand <id> \| /tasks done <id> \| /tasks drop <id>` | Show the session board or manage project operator tasks |
159
- | `/memory` | `/memory seed` | Inspect task memory or seed it from the newest handoff |
161
+ | `/memory` | `/memory seed` | Inspect, promote, or seed task memory |
160
162
  | `/view` | `/view [filter] \| /view verify <runId>` | Browse session artifacts and verify receipts |
161
163
  | `/thinking` | `/thinking [level]` | Set the chat thinking level, or open Settings → Orchestrator |
162
164
  | `/output` | `/output [verbosity]` | Set transcript detail (minimal, default, verbose), or open Settings → Terminal |
@@ -169,9 +171,13 @@ The registry table below lists the available interactive slash commands. On a ba
169
171
  | `/fork` | `/fork` | Fork from an assistant turn |
170
172
  | `/export` | `/export [path]` | Export a self-contained HTML transcript by default; a `.md` path writes Markdown |
171
173
 
172
- `/context` with no arguments opens the context-window ledger overlay. The
173
- subcommands own the durable project-context noun: `compact` summarizes older
174
- turns in the session window, `init` bootstraps or updates `CLIO-CODER.md` and the
174
+ `/context` with no arguments opens the context-window ledger overlay, including
175
+ the working-set section (policy, evicted items and tokens, events, recalls, churn).
176
+ The subcommands own the durable project-context noun: `compact` summarizes older
177
+ turns in the session window, `recall <ref>` prints an evicted tool-result body
178
+ back into the transcript by the ref its `[evicted ...]` marker names (it never
179
+ enters model context; the model recalls with `context(scope="recall", ref=...)`),
180
+ `init` bootstraps or updates `CLIO-CODER.md` and the
175
181
  codewiki, `refresh` re-indexes the codewiki and refreshes `.clio-coder/state.json`
176
182
  without touching `CLIO-CODER.md`, and `reset` deletes accumulated
177
183
  context artifacts (`.clio-coder/codewiki.json`, `.clio-coder/state.json`,
@@ -218,7 +224,7 @@ Configuration lives in one place: the `/settings` overlay. `/settings <section>`
218
224
 
219
225
  Settings → Targets presents an operational console table (`HEALTH`, `ID`, `ROLES`, `RUNTIME`, `LATENCY`) with an in-place action/detail drawer for URL, default model, last probe error, and reachability. `Enter` opens actions for `Use` (switches active chat target and rebases model), `Connect` (runs the API-key or OAuth flow then probes), `Probe`, and `Remove` (with preflight analysis of affected routes/profiles). Probing runs live when the overlay opens or when explicitly requested. Target creation is initiated via `clio-coder targets add`.
220
226
 
221
- Settings → Fleet is an entity workbench organized with dim group headers (`Defaults`, `Profiles`, `Agent routes`, `Placement`). Dispatched worker defaults and profile rows render as compact summaries (`fast-local node-a/example-coder-model high auto`), drilling into fields (`target`, `model`, `thinkingLevel`, `node`) on `Enter`. Profile removal is a named destructive action with affected-route preflight. Running and retrying dispatches live in the `Alt+W` Fleet Runs board, which also steers and cancels them.
227
+ Settings → Fleet is an entity workbench organized with dim group headers (`Defaults`, `Profiles`, `Agent routes`, `Placement`). Dispatched worker defaults and profile rows render as compact summaries (`fast-local node-a/example-coder-model high auto`), drilling into fields (`target`, `model`, `thinkingLevel`, `node`) on `Enter`. Profile removal is a named destructive action with affected-route preflight. Running and retrying dispatches live in the `Alt+W` Fleet Runs board, which also steers and cancels them. `Enter` opens the selected run's worker detail: the phase, the running call with its redacted action descriptor, and the bounded tail of the worker's own prose.
222
228
 
223
229
  `/run` and `/delegate` put the worker's answer on screen. Both echo the typed
224
230
  line dim above the block, then stream the run into the transcript as an attributed
@@ -344,7 +350,7 @@ editor reserves and can be rebound through `settings.yaml.keybindings`.
344
350
  | `Alt+U` | Toggle the footer dashboard between compact (quiet 2-zone) and expanded (4-zone urgency) layouts. |
345
351
  | `Alt+L` | Open the model and targets selector. |
346
352
  | `Alt+J` / `Alt+K` | Cycle forward / backward through the scoped model set (when empty, displays a notice directing the operator to `/scoped-models`). |
347
- | `Alt+W` | Toggle the Fleet Runs board (task, run ID, live telemetry, retry, and terminal history). |
353
+ | `Alt+W` | Toggle the Fleet Runs board (task, run ID, live telemetry, retry, and terminal history). Inside it, `Enter` opens the selected run's live worker detail, `s` steers, and `x` cancels. |
348
354
  | `Alt+B` | Open the composite session and operator task board (`/tasks`). Approved application-boundary override of editor word-back. |
349
355
  | `Alt+D` | Open the settled interview decision board (`/decisions`). Approved application-boundary override of editor word-delete. |
350
356
  | `Alt+S` / `Ctrl+Alt+B` | Convert an active attached dispatch to a detached background batch. |
@@ -409,7 +415,9 @@ Tool and command execution is governed by:
409
415
  - **Safety Net:** Granular rule packs loaded from `damage-control-rules.yaml`, project policies, and protected artifact paths; always on, identical at every autonomy level.
410
416
  - **Autonomy Mapping:** Once the net passes a call, the level decides whether it runs, asks, or is denied. See [safety-model.md](safety-model.md) for the full matrix.
411
417
 
412
- When an action asks for confirmation, whether from a safety-net rail or from the autonomy level, the call parks and the TUI displays a queued permission dialog whose `Asked by:` line names the asking axis. The operator can approve or deny that single action without changing the level.
418
+ When an action asks for confirmation, whether from a safety-net rail or from the autonomy level, the call parks and three surfaces say so at once. The transcript row reads `⏸ awaiting approval` with `action ·`, `axis ·`, and `target ·` lines under it; the footer phase pill reads `⏸ confirm`; and a consequence-tier dialog opens with the tool, target, action, authenticated requester, one-shot authority, reversibility, and deny and stop effects. Titles distinguish workspace authority, outward consequences, safety-net confirmation, system changes, and worker escalations. The dialog sits at bottom center with five rows reserved for the composer and footer, and it re-anchors on resize. The composer rail switches to `CONFIRM` and repeats the keys while the prompt owns the keyboard.
419
+
420
+ The keys are the same on both surfaces: `Enter` allows this one call, `Esc` denies it, and `s` denies it and stops the turn so nothing asks again. `Enter` allows only from an empty composer. While the composer holds a draft, the habitual send key does nothing, the rail and the dialog footer read `[Backspace] clear draft` instead of `[Enter] allow`, and only the deletion keys (`Backspace`, `Delete`, `Ctrl+U`, `Ctrl+W`, `Ctrl+K`) reach the editor until the draft is gone. Every other key is swallowed. A call that parks while another overlay holds the screen is announced with an `[approval]` notice and the dialog opens as soon as that overlay closes; the dialog lays itself out for any terminal width, so no width is too narrow for it. Approving or denying never changes the level.
413
421
 
414
422
  Notice vocabulary, one prefix per mechanism: `[safety-net]` for level-independent blocks, `[approval]` for parked calls, `[autonomy]` for read-only denials, and `[middleware]` for hook diagnostics.
415
423
 
@@ -459,7 +467,7 @@ to execute through the existing engine worker path, the sanctioned Claude Code w
459
467
  | --- | --- |
460
468
  | `npm run ci` | Local and GitHub PR gate: typecheck, lint, skills pin check, build, the deterministic test suite, and the trace-viewer suite. |
461
469
  | `npm run ci:release` | Maintainer release gate: `npm run ci`, then the `check-release` dist and packaging audit. |
462
- | `npm run test:live` | Local manual live-model smoke. Requires `CLIO_CODER_LIVE_SMOKE=1` and a configured real model target. Add `-- --delegation` for `opencode` and `copilot` ACP delegation checks. |
470
+ | `npm run live:smoke -- --target <id>` | One real headless turn against a configured target. Add `--delegation` for the `opencode` and `copilot` ACP agents. The other operator-run drivers (`live:fleet-dispatch`, `live:tui`, `live:home`) are listed in `benchmarks/internal/README.md`. |
463
471
  | `npm run typecheck` | Strict TypeScript pass. |
464
472
  | `npm run lint` | Biome checks plus `scripts/check-hygiene.ts`, which runs the boundary invariants, the skills pin check, and the README and docs drift rules. |
465
473
  | `npm run test` | Contract and smoke tests through the sharded runner. |
@@ -467,28 +475,21 @@ to execute through the existing engine worker path, the sanctioned Claude Code w
467
475
  | `npm run dev` | `tsup --watch`. |
468
476
  | `npm run clean` | Remove `dist/`. |
469
477
 
470
- Live smoke example:
471
-
472
- ```bash
473
- CLIO_CODER_LIVE_SMOKE=1 \
474
- CLIO_CODER_LIVE_TARGET=openai-compat \
475
- CLIO_CODER_LIVE_RUNTIME=openai-compat \
476
- CLIO_CODER_LIVE_MODEL=your-model \
477
- CLIO_CODER_LIVE_BASE_URL=http://localhost:8080/v1 \
478
- npm run test:live
479
- ```
480
-
481
- Delegation validation is a separate opt-in flag because it depends on local
482
- `opencode` and `copilot` commands:
478
+ Live drivers pick their model with `--target <id>`, naming one of the targets
479
+ `clio-coder targets` lists; `--model <wireId>` and `--thinking <level>`
480
+ override that target's defaults for the run. The run happens in a scratch Clio
481
+ home holding only that target, so nothing touches the operator's real state:
483
482
 
484
483
  ```bash
485
- CLIO_CODER_LIVE_SMOKE=1 npm run test:live -- --delegation
484
+ npm run build
485
+ npm run live:smoke -- --target local-lmstudio --model qwen3.8-27b
486
+ npm run live:smoke -- --target anthropic --delegation
486
487
  ```
487
488
 
488
489
  Live checks cost tokens or local GPU time and are not deterministic CI. They
489
490
  are useful for OpenAI-compatible local gateways such as llama.cpp, LM Studio
490
491
  with Dynamo-backed workers, vLLM, and SGLang, plus cloud targets when
491
- credentials are available.
492
+ credentials are configured.
492
493
 
493
494
  ## Environment Variables
494
495
 
@@ -519,6 +520,39 @@ structural hash. The same builder is used by `clio-coder context init`, `clio-co
519
520
  refresh`, session freshness checks, tool-demand backfill, and in-session
520
521
  incremental updates.
521
522
 
523
+ ### Working-set replay
524
+
525
+ `clio-coder context replay --sessions <path>...` accepts individual session directories,
526
+ Clio sessions roots, and ledger JSONL files; `--synthetic <ids>` adds one or more procedural
527
+ corpora (`science-long`, `refactor`, `exploration`) generated in memory from a fixed seed, so
528
+ the committed tables can be rebuilt byte for byte on any checkout without private transcripts.
529
+ Clio traces remove prior eviction/recall sidecars and select the active branch. Both sources
530
+ drive the live fold, projection, policy, and eviction planner at deterministic turn
531
+ boundaries. The default inclusion cascade for ledgers requires at least eight turns, eight
532
+ tool results, and one file re-read; `--no-filter` retains every otherwise-readable trace.
533
+ Markdown goes to stdout unless `--md` names a file, while `--json` writes a stable report
534
+ including the configuration, the corpus, git revision when available, and exact command line.
535
+ `--protect-last-turns` and `--min-evictable-tokens` override those two working-set settings
536
+ for the replay only; they never update saved settings. Saturated events is pooled over
537
+ applied eviction events and reports how often a policy exhausted its usable candidates,
538
+ which distinguishes a budget that measures policy choice from one that simply runs out of
539
+ evictable material. Recall tokens is the token-weighted complement of precision: what a
540
+ perfect recall would read back for evicted items the session referenced again. Cold prefix
541
+ tokens is the projected working set after the earliest evicted position of each event, which
542
+ is what an exact-prefix cache re-prefills on the next request. The replay also models the
543
+ summary stage: when the projection is still over the threshold after an eviction, it applies
544
+ the same `findCutPoint(keepRecentTokens)` cut the live path uses, appends a stand-in
545
+ `compactionSummary`, counts it, and treats what the cut removed as lost for retention.
546
+ Summaries (mean) is therefore the number of lossy, token-spending compactions a policy forced
547
+ per trace. The summary-headroom mean always carries its contributing trace count because
548
+ traces that never require summary compaction do not enter that nullable mean.
549
+
550
+ `clio-coder context working-set --session <id|path>` is a read-only inspection command for
551
+ one ledger. It prints evicted refs with reason, superseding ref, and token count; aggregate
552
+ event, recall, and churn facts; path-observation counts by operation; and paths whose earlier
553
+ reads were followed by writes or edits. A persisted `/tree` pin is honored when the session
554
+ metadata is available, so the report does not resurrect an abandoned branch.
555
+
522
556
  The current artifact is schema v5. It records files with path, language, line
523
557
  count, role, content hash, imports, and optional summary; declaration-only
524
558
  symbols with name, kind, file id, line, and optional signature; and import edges
@@ -96,8 +96,7 @@ All default off; all enabled with `1`.
96
96
 
97
97
  | Env var(s) | Used by |
98
98
  |---|---|
99
- | `CLIO_CODER_LIVE_SMOKE`, `CLIO_CODER_LIVE_TARGET`, `CLIO_CODER_LIVE_RUNTIME`, `CLIO_CODER_LIVE_BASE_URL`, `CLIO_CODER_LIVE_MODEL`, `CLIO_CODER_LIVE_API_KEY`, `CLIO_CODER_LIVE_KEEP` | `scripts/live-smoke.mjs` |
100
- | `CLIO_CODER_MAIN_TARGET/_MODEL/_URL/_THINKING`, `CLIO_CODER_WORKER_TARGET/_MODEL/_URL/_THINKING`, `CLIO_CODER_FLEET`, `CLIO_CODER_FLEET_PROFILE`, `CLIO_CODER_PRED_MODEL`, `CLIO_CODER_LLAMACPP_KEY`, `CLIO_CODER_LMSTUDIO_KEY` | benchmark fleet config (`benchmarks/community/`) |
99
+ | `CLIO_CODER_MAIN_TARGET/_MODEL/_URL/_RUNTIME/_THINKING`, `CLIO_CODER_WORKER_TARGET/_MODEL/_URL/_RUNTIME/_THINKING`, `CLIO_CODER_LLAMACPP_KEY`, `CLIO_CODER_LMSTUDIO_KEY` | Terminal-Bench installed agent (`benchmarks/community/terminal-bench/`), which renders its own settings.yaml inside the task container |
101
100
  | `CLIO_CODER_AUTONOMY`, `CLIO_CODER_TASK_TIMEOUT` | terminal-bench agent wrapper (writes settings.yaml / its own timeout; not a Clio knob) |
102
101
  | `CLIO_CODER_BIN`, `CLIO_CODER_ENTRY`, `CLIO_CODER_TARBALL_URL` | install/dev scripts |
103
102
  | `CLIO_CODER_NO_UPDATE_NOTIFIER` | **Dead.** Set by four benchmark harnesses, read by nothing in `src/`. Either an update notifier was removed and the setters were left behind, or the feature was never built. |
@@ -1,7 +1,7 @@
1
1
  # Configuration, Targets, Runtimes, and Auth
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive configuration validator, target resolver, and CLI command generator is located at [docs/html/configuration_blueprint.html](html/configuration_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive configuration validator, target resolver, and CLI command generator is located at [docs/html/configuration_blueprint.html](html/configuration_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder is target-first: chat and fleet dispatch resolve through configured targets in `settings.yaml`, not through provider-specific ad hoc flags. Chat and print targets are HTTP and native engine-backed runtimes. Fleet dispatch can also target the sanctioned Claude Code subscription runtimes described below.
7
7
 
@@ -224,6 +224,15 @@ compaction:
224
224
  excludeLastTurns: 6
225
225
  # model: provider/summary-model-id
226
226
  # systemPrompt: ~/.config/clio-coder/prompts/compaction.md
227
+
228
+ context:
229
+ workingSet:
230
+ enabled: true
231
+ policy: structural-v1
232
+ target: 0.6
233
+ protectLastTurns: 6
234
+ minEvictableTokens: 200
235
+
227
236
  retry:
228
237
  enabled: true
229
238
  maxRetries: 3
@@ -297,6 +306,17 @@ LM Studio can require bearer authentication for its HTTP APIs
297
306
 
298
307
  A model id on an LM Studio target is resolved against that host's loaded instances. A key with a loaded instance is never sent bare (which would JIT-load a second copy). An instance id reported loaded by two configured LM Studio targets on different hosts is an LM Link peer projection. When a bare model key is requested and multiple instances of it are loaded, Clio selects an instance in this order: the target's configured `defaultModel`, then an instance not cross-listed by another configured LM Studio target, and finally the first loaded instance. This behavior tracks issue #113.
299
308
 
309
+ When the selected instance is also loaded on a peer, a request may be answered by that peer (#185). Clio separates the requested model id, the response observation, and the model id used for accounting. Every new assistant ledger entry carries `responseModelIdObservation` in one of these explicit shapes:
310
+
311
+ | State | Meaning | Accounting attribution |
312
+ | --- | --- | --- |
313
+ | `{ "state": "reported", "reportedModelId": "<id>" }` | Clio observed an OpenAI-compatible event stream and the provider reported a model id. | The reported id. |
314
+ | `{ "state": "not-reported" }` | Clio observed the event stream and it contained no model id. | `unknown`, because the provider did not identify the responding model. |
315
+ | `{ "state": "not-observed" }` | This provider path did not expose response model-id presence to the stream tap. | A differing `responseModel` when available, otherwise the requested model id. |
316
+ | `{ "state": "legacy-difference-only", "differingModelId": "<id>" }` or the same shape with `null` | The ledger predates #193 and recorded only whether the response `model` differed from the request. This state is produced while reading historical rows; new rows do not write it. | The historical differing id when available, otherwise the requested model id. |
317
+
318
+ The adapter retains `responseModel` as the differing response id because providers outside the stream tap still supply that fact. `clio-coder usage report` emits `attributedModelId`, `requestedModelIds`, and `responseModelIdObservationCounts`. Its text table and the `/cost` overlay use the labels `attributed model`, `requested model ids`, and `response model id observation`; requested ids are printed as ids rather than as `same`. The footer's last-turn line uses `response model id observation <state>`, with the id after `reported` or a historical `legacy difference-only` state. Dispatch receipt `upstreamResponses` entries carry `requestedModelId`, `responseModelIdObservation`, `differingResponseModelId`, and `providerResponseId`. The peer warning is said once per process per distinct fact (target, requested id, resolved instance, peer set), not once per turn.
319
+
300
320
 
301
321
  Prompt-template overrides, system prompts, GPU-offload ratios, KV-cache quantization, parallel slots,
302
322
  context checkpoints, and speculative-decoding variants are not writable through this Clio settings
@@ -575,6 +595,11 @@ Every one of these has an environment override for a single process; see [enviro
575
595
  | `compaction.auto` | `true` | boolean | next turn |
576
596
  | `compaction.threshold` | `0.8` | number in 0 to 1 | next turn |
577
597
  | `compaction.excludeLastTurns` | `6` | integer ≥ 1 | next turn |
598
+ | `context.workingSet.enabled` | `true` | boolean | next turn |
599
+ | `context.workingSet.policy` | `structural-v1` | `age-horizon` or `structural-v1` | next turn |
600
+ | `context.workingSet.target` | `0.6` | number greater than 0 and less than 1 | next turn |
601
+ | `context.workingSet.protectLastTurns` | `6` | integer ≥ 1 | next turn |
602
+ | `context.workingSet.minEvictableTokens` | `200` | integer ≥ 0 | next turn |
578
603
  | `defaults.maxTokens` | `32768` | integer ≥ 0 | next turn |
579
604
  | `budget.sessionCeilingUsd` | `5` | number ≥ 0 | immediately |
580
605
  | `budget.concurrency` | `auto` | `auto` or integer ≥ 1 | next dispatch |
@@ -1,11 +1,13 @@
1
1
  # Context Engine
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/context_blueprint.html](html/context_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/context_blueprint.html](html/context_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder tracks context pressure, records per-turn snapshots, and protects the provider context with bounded tool results plus single-threshold compaction.
7
7
 
8
- Source of truth lives in `src/domains/session/context-accounting.ts`, `src/domains/session/context-ledger.ts`, `src/domains/session/compaction/`, `src/domains/session/migrations/index.ts`, and the chat-loop integration in `src/interactive/chat-loop.ts`.
8
+ Source of truth lives in `src/domains/session/context-accounting.ts`, `src/domains/session/context-ledger.ts`, `src/domains/session/compaction/`, `src/domains/context/working-set/`, `src/domains/session/migrations/index.ts`, and the chat-loop integration in `src/interactive/chat-loop.ts`.
9
+
10
+ The non-destructive eviction layer has its own guide: [context-working-set.md](context-working-set.md).
9
11
 
10
12
  ## Context window resolution
11
13
 
@@ -17,13 +19,15 @@ Local-native runtimes use a recommended minimum desired window of 128,000 tokens
17
19
 
18
20
  The `/context` overlay states which layer answered, next to the token total: `loaded`, `probed`, `configured`, `declared`, or `assumed`.
19
21
 
22
+ A probed llama.cpp window is the share one request gets, not the server's total. llama.cpp splits `--ctx-size` evenly across `--parallel` slots unless `--kv-unified` is set, so a server started with `--ctx-size 786432 --parallel 4 --no-kv-unified` admits 196,608 tokens per request, and that is the figure autocompact and the meter plan against. The probe reads the flags (long and short forms, `-c`, `-np`, `-kvu`, and the last of `--kv-unified` or `--no-kv-unified` given) off the router's per-model status, keeps the split on the model's discovery state, and `/context` prints the derivation next to the share: `196,608 (786,432 / 4 slots)`. `clio-coder targets` does the same in its `ctx` note for the target's default model and adds a probe note naming the flags.
23
+
20
24
  ## Token accounting and snapshots
21
25
 
22
26
  The estimator in `context-accounting.ts` uses a four-characters-per-token family for hot-path accounting. It estimates system prompt, tools, messages, pending input, and runtime categories without calling a model tokenizer on every TUI refresh.
23
27
 
24
28
  At submit time, Clio captures a context snapshot and persists a slim JSONL record under the session directory as `context-snapshots.jsonl`. The slim record keeps token counts, segment metadata, signatures, and hashes, not the heavy prompt or transcript text. When provider usage arrives, `reconcileSnapshot` folds actual input and output counts back into the ledger.
25
29
 
26
- Session metadata enforces session format version 3 (`CURRENT_SESSION_FORMAT_VERSION = 3`). Before resuming any session, Clio checks `sessionFormatVersion`; earlier formats are rejected outright with an error rather than silently migrated.
30
+ Session metadata enforces session format version 4 (`CURRENT_SESSION_FORMAT_VERSION = 4`). Version 4 is additive: it adds the `contextEviction` and `contextRecall` records and changes no existing entry. A version 3 session therefore migrates to 4 in place when Clio opens it, and no entry is rewritten. Only a session written by a newer build is refused, with an error naming the version it read and pointing at upgrading. The bump is one-way for the operator: a 0.3.3 binary cannot open a session this release wrote.
27
31
 
28
32
  The `/context` overlay and footer meter read the same ledger categories: `system`, `tools`, `agents`, `skills`, `memory`, `project`, `messages`, `pending`, `reserve`, `free`, and `streaming`.
29
33
 
@@ -31,31 +35,63 @@ The `/context` overlay and footer meter read the same ledger categories: `system
31
35
 
32
36
  Auto-compaction is controlled by one pressure threshold. Pressure is `estimated_tokens / context_window`. The default threshold is `0.8`.
33
37
 
34
- When `compaction.auto` is enabled and pressure crosses the threshold before a request, Clio first masks stale tool observations and stale thinking older than `excludeLastTurns`. This is a cheap local rewrite. Tool call and result structure remain present, but the observation body is replaced with a marker and stale assistant thinking content is dropped from replay.
38
+ Crossing that threshold engages three mechanisms in a fixed order. The first two are cheap, reversible, and call no model. Only the third rewrites what the session says about itself.
39
+
40
+ ### 1. Working-set eviction
41
+
42
+ When `compaction.auto` is enabled and pressure crosses the threshold before a request, Clio applies the configured working-set policy first. The policy selects tool-result bodies and closed-turn thinking blocks, `runAutoCompact` appends one `contextEviction` ledger entry, and `refreshAgentMessagesFromSession` projects those units out of model replay behind a one-line marker. Nothing is deleted: the ledger keeps the original bodies, the transcript keeps showing them, and `/resume`, `/tree`, `/fork`, and the HTML export are unaffected.
43
+
44
+ Already-evicted units are never selected again. Recent turns keep their full observations and thinking, governed by `context.workingSet.protectLastTurns`. Results whose estimated body is below `context.workingSet.minEvictableTokens` (200 tokens by default) are kept whatever their age as a low-yield churn guard. The engine separately refuses any candidate whose marker would save no tokens. The `age-horizon` policy is therefore the selection the old destructive mask made minus those small results, not a byte-identical reproduction of it; the default `structural-v1` policy applies its structural rules before any age rule.
45
+
46
+ If the projection drops pressure below the threshold, Clio sends the request and no summary runs. The policies, the protection predicates, the marker format, and the ledger records are documented in [context-working-set.md](context-working-set.md).
47
+
48
+ ### 2. Recall
49
+
50
+ An evicted body comes back on demand and only on demand. The marker names the exact call: `context(scope="recall", ref="<turnId>")` returns the original body byte-exact through the observation envelope and appends a `contextRecall` entry. Operators use `/context recall <ref>`, which prints the body to the transcript without putting it into model context.
51
+
52
+ A recall does not un-evict. The marker stays byte-identical where it was, so the provider prefix cache is untouched, and repeated recalls of the same ref are the churn signal the `/context` overlay reports.
53
+
54
+ Offline replay does not infer those explicit decisions from a later read of the same path. A reread already returns current content, while recall returns a selected historical ref. The replay tables keep the token-weighted `recallTokens` demand bound and reserve recall count, churn, and tail-growth simulation for ledgers or corpora that record which refs were actually recalled.
55
+
56
+ ### 3. LLM summary, as a last resort
35
57
 
36
- Marker format:
58
+ If pressure remains above the threshold after eviction, Clio runs the summary compaction path: it calls the summarization model, appends a `compactionSummary` entry, refreshes projected replay messages from the session, and continues. This is the only mechanism that spends tokens and the only one whose output is a lossy paraphrase, which is why it runs last.
59
+
60
+ Iterative compaction has one raw-history boundary. The first pass searches from the start of the active path. A later pass searches strictly after the previous `compactionSummary`, while the prior checkpoint and its retained suffix are fed to the summarizer as canonical context for one cumulative replacement. The replay benchmark uses that same boundary. It must not run `findCutPoint` over the visible retained suffix again: doing so re-prices history already captured by the previous checkpoint and overstates repeated summary churn.
61
+
62
+ Manual `/context compact`, `CLIO_CODER_FORCE_COMPACT=1`, and overflow recovery force the summary path directly and skip every pre-stage. The overflow guard runs before the user turn is committed, so a blocked oversized request does not leave an unanswered user entry in the ledger.
63
+
64
+ ### The legacy mask escape hatch
65
+
66
+ `CLIO_CODER_LEGACY_MASK=1` restores the destructive pre-stage working-set eviction replaced. It calls `session.replaceEntries` and rewrites the persisted bodies, so masked content is gone for the operator as well as the model. It uses the old marker format:
37
67
 
38
68
  ```text
39
69
  [Observation masked: <tool> output was <lines> lines, <chars> chars - contents masked to save context. Re-run the tool for current content.] Preview: <preview>
40
70
  ```
41
71
 
42
- Already-compacted entries are not masked again. Recent turns keep their full observations and thinking. If masking drops pressure below the threshold, Clio sends the request without an LLM summary. If pressure remains above the threshold, Clio runs the summary compaction path, appends a compaction summary entry, refreshes replay messages from the session, and continues.
72
+ It exists for one release as a compatibility diagnosis path and is removed in the next.
43
73
 
44
- When the ledger is replayed to the model, compaction summaries, branch summaries, and bash executions become standardized user-role message text. Clio imports `COMPACTION_SUMMARY_PREFIX`, `BRANCH_SUMMARY_PREFIX`, their suffixes, and `bashExecutionToText` through `src/engine/messages.ts`; `src/interactive/chat-renderer.ts` maps Clio's entry shapes onto them and applies replay truncation.
74
+ ### Replay text
45
75
 
46
- Manual `/context compact`, `CLIO_CODER_FORCE_COMPACT=1`, and overflow recovery force the LLM summary path directly. The overflow guard runs before the user turn is committed, so a blocked oversized request does not leave an unanswered user entry in the ledger.
76
+ When the ledger is replayed to the model, compaction summaries, branch summaries, and bash executions become standardized user-role message text. Clio imports `COMPACTION_SUMMARY_PREFIX`, `BRANCH_SUMMARY_PREFIX`, their suffixes, and `bashExecutionToText` through `src/engine/messages.ts`; `src/interactive/chat-renderer.ts` maps Clio's entry shapes onto them and applies replay truncation. The working-set projection runs before that builder, so markers are what the replay text is built from.
47
77
 
48
78
  ## Cache-divergence honesty
49
79
 
50
- Compaction rewrites the replayed history. On a local backend with a single prefix-cache slot, the next turn after compaction is expected to be cold because the byte prefix changed. Dispatch traffic can disturb the same slot.
80
+ Every provider Clio targets caches by exact prefix. Anthropic hashes the cumulative prefix up to a `cache_control` breakpoint and looks back at most 20 blocks for an earlier write; the minimum cacheable prefix is 512 to 4,096 tokens by model, reads cost 0.1x input and writes 1.25x. OpenAI caches automatically from 1,024 tokens in 128-token increments on exact prefix matches at 0.1x. vLLM hashes each KV block from its parent block's hash, so a change in one block invalidates every later block. llama.cpp (and LM Studio on top of it) picks the slot with the longest common prefix and re-evaluates only the suffix, and `--cache-reuse` can shift later KV chunks back into place after a mid-prompt removal. The consequence is the same everywhere except on llama.cpp with cache reuse: whatever bytes change, everything after the earliest changed position is re-prefilled. That is why a marker is byte-stable, why a recall rides the tail instead of restoring the body in place, why `structural-v1` batches evictions down to `target` instead of trimming on every turn, and why the replay tables report cold prefix tokens per event next to tokens evicted: at a 32k budget one event re-prefills most of the window whichever policy chose the items, so the lever that protects a cloud cache is the number of events, not their contents. A local backend with cache reuse pays less for the same removal, which is where finer-grained eviction and recall earn their keep.
81
+
82
+ The procedural replay target sweep measured 0.4, 0.5, 0.6, and an exhaustive rung-6 stop over 24 traces. Target 0.4 and exhaustive selection converged because un-evictable residue exhausted the candidate pool. Against 0.6, target 0.4 cut cold-prefix tokens by 2.8% at 64k and 7.3% at 128k, with no summary reduction and a 0.00072 reduction in retention covered at 128k. That is below the 10% cache-saving threshold set for changing a cross-tier default, so the default remains 0.6. The complete sweep and reopening rule are in the replay README.
51
83
 
52
- Clio records these disturbances once on the next assistant entry as `promptCache.expectedColdReasons`. The user sees one dim notice, and the same reasons persist on that entry in the session ledger next to the per-call cache data.
84
+ Compaction and eviction both change the replayed history. On a local backend with a single prefix-cache slot, the next turn after either one is expected to be cold because the byte prefix moved. Dispatch traffic can disturb the same slot.
85
+
86
+ Clio records these disturbances once on the next assistant entry as `promptCache.expectedColdReasons`. The recorded reasons are `working_set_evict` for an applied eviction event, `compaction` for the summary path, and `dispatch` for interleaved worker traffic. `compaction` and `dispatch` are stamped only on `local-native` targets, because a single-slot local cache is the one an interleaved run actually disturbs. `working_set_evict` is stamped on every tier: the eviction moved the byte prefix itself, so the cloud prefix cache is cold for the same reason. The user sees one dim notice, and the same reasons persist on that entry in the session ledger next to the per-call cache data.
53
87
 
54
88
  Per-call cache verdicts are `hot`, `partial`, `cold`, and `small`. They are derived from provider usage and persisted with `timing { ttftMs, apiMs }` and `promptCache { input, cacheRead, cacheWrite, backendVerdict }` when available.
55
89
 
90
+ The `/context` overlay closes the loop. When the last settled run came back `cold` and Clio had recorded a reason for it, the overlay adds a line naming that reason, for example `last cold turn: working-set eviction (expected)`, and reports the cache line without the warning token. A reused prompt shell with a cold backend and no recorded reason stays a warning: Clio kept the bytes stable and the provider re-prefilled anyway, which is a disagreement worth surfacing.
91
+
56
92
  ## Settings
57
93
 
58
- The public settings block has one threshold and one recent-turn horizon:
94
+ The public settings use one compaction threshold plus a non-destructive working-set stage:
59
95
 
60
96
  ```yaml
61
97
  compaction:
@@ -64,9 +100,27 @@ compaction:
64
100
  excludeLastTurns: 6
65
101
  # model: provider/summary-model-id
66
102
  # systemPrompt: ~/.config/clio-coder/prompts/compaction.md
103
+
104
+ context:
105
+ workingSet:
106
+ enabled: true
107
+ policy: structural-v1
108
+ target: 0.6
109
+ protectLastTurns: 6
110
+ minEvictableTokens: 200
67
111
  ```
68
112
 
69
- `auto` controls the pre-request trigger. Manual `/context compact` still runs when `auto` is false. `model` optionally selects a dedicated summarization model. `systemPrompt` optionally points at a prompt override file for compaction.
113
+ `compaction.auto` controls the pre-request trigger. Manual `/context compact` still runs when `auto` is false. `compaction.model` optionally selects a dedicated summarization model, and `compaction.systemPrompt` optionally points at a prompt override file. `compaction.excludeLastTurns` only governs the temporary legacy mask path; working-set protection uses `context.workingSet.protectLastTurns`.
114
+
115
+ | Key | Default | Accepted | Meaning |
116
+ | --- | --- | --- | --- |
117
+ | `context.workingSet.enabled` | `true` | boolean | `false` skips eviction and goes directly to summary compaction. It does not restore the destructive mask. |
118
+ | `context.workingSet.policy` | `structural-v1` | `age-horizon`, `structural-v1` | Candidate selection rule set. `age-horizon` is the pre-layer age selection. |
119
+ | `context.workingSet.target` | `0.6` | number greater than 0 and less than 1 | Used-over-window ratio an applied eviction event batches down to. |
120
+ | `context.workingSet.protectLastTurns` | `6` | integer ≥ 1 | Recent turns whose observations and thinking are never evicted. |
121
+ | `context.workingSet.minEvictableTokens` | `200` | integer ≥ 0 | Results below this body estimate are never evicted. The default is a measured low-yield churn guard; marker break-even is enforced separately. |
122
+
123
+ Set `CLIO_CODER_LEGACY_MASK=1` only as a temporary compatibility escape hatch for the old destructive mask stage. See [context-working-set.md](context-working-set.md) for what each policy selects and why.
70
124
 
71
125
  Settings validation is strict: an older file still carrying the removed `compaction.thresholds` block fails to load with the exact key path during normal startup. Edit removed or unknown keys deliberately; `clio-coder doctor --fix` does not transform settings into the current schema.
72
126
 
@@ -138,7 +192,7 @@ In Git workspaces, the indexer uses the same visible file set across full builds
138
192
  incremental updates, fingerprints, and project profiles: tracked files plus
139
193
  untracked, unignored work in progress. It excludes symlinks, submodule gitlinks,
140
194
  generated output, scratch space, and local-state directories such as `.git`,
141
- `.clio-coder`, `.superpowers`, `.codex`, `.claude`, `.clio-coder-benchmark`, `node_modules`,
195
+ `.clio-coder`, `.superpowers`, `.codex`, `.claude`, `node_modules`,
142
196
  `dist`, `build`, `coverage`, virtualenvs, `target`, and `vendor`. Non-Git
143
197
  workspaces use a bounded filesystem walk with the same directory exclusions.
144
198
  Source coverage spans TypeScript, JavaScript, Python, Rust, Go, C, C++, CUDA