@iowarp/clio-coder 0.3.2 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (259) hide show
  1. package/CHANGELOG.md +269 -458
  2. package/CONTRIBUTING.md +1 -1
  3. package/README.md +3 -3
  4. package/dist/{acp-BIYHVZIM.js → acp-S5R4RR5B.js} +7 -6
  5. package/dist/{agents-YT6SSRIT.js → agents-P6DMMVZY.js} +24 -21
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5TWEIYDN.js → auth-2XCZLPKS.js} +12 -8
  8. package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
  9. package/dist/{chunk-WMSVI4G2.js → chunk-2LZI5CAG.js} +133 -13
  10. package/dist/{chunk-OAO4GE4M.js → chunk-2TZWSW76.js} +2 -2
  11. package/dist/{chunk-OOJYHWRB.js → chunk-34475P3I.js} +2 -2
  12. package/dist/{chunk-WVO7V2QY.js → chunk-35MKKU5R.js} +4 -4
  13. package/dist/{chunk-LBNRH5WM.js → chunk-3HZ5RWN2.js} +5 -5
  14. package/dist/{chunk-AGYYIBLL.js → chunk-3JLKSKD7.js} +2 -2
  15. package/dist/{chunk-MBS4V7ZP.js → chunk-4JUF2NNX.js} +7 -7
  16. package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
  17. package/dist/chunk-5M54SPOL.js +926 -0
  18. package/dist/{chunk-STBPMHSX.js → chunk-7RXG6QRZ.js} +51 -11
  19. package/dist/{chunk-77VKQEHF.js → chunk-A2GZF7DC.js} +5 -5
  20. package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
  21. package/dist/chunk-AOCYTWAV.js +449 -0
  22. package/dist/chunk-BEY543CS.js +258 -0
  23. package/dist/{chunk-6N5PTWMY.js → chunk-BP4OYD6A.js} +32 -13
  24. package/dist/chunk-BPGS2WCQ.js +612 -0
  25. package/dist/{chunk-J5HN4RYU.js → chunk-BRXQQJFP.js} +8 -8
  26. package/dist/chunk-CFGTUFWB.js +67 -0
  27. package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
  28. package/dist/{chunk-G2DE3C7R.js → chunk-EDRHSCIE.js} +4 -4
  29. package/dist/{chunk-4KLWL3UC.js → chunk-EFADSJET.js} +2 -2
  30. package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
  31. package/dist/chunk-FYYLNIL5.js +313 -0
  32. package/dist/{chunk-EPVUXGXG.js → chunk-HV5X7OR2.js} +14 -12
  33. package/dist/{chunk-TZTZS7QK.js → chunk-HXG4IURW.js} +5 -3
  34. package/dist/{chunk-IGLFWIYI.js → chunk-K6WL7QZT.js} +3 -3
  35. package/dist/chunk-K7VKOLQQ.js +15 -0
  36. package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
  37. package/dist/{chunk-V4RXGQ5Q.js → chunk-KRPY7NTG.js} +10 -7
  38. package/dist/chunk-LL4KHSZI.js +22 -0
  39. package/dist/{chunk-KJ5LWLOE.js → chunk-MEQ45TQ4.js} +15 -9
  40. package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
  41. package/dist/{chunk-AO4RKG4M.js → chunk-N4CZJQRK.js} +5 -5
  42. package/dist/{chunk-ARBGF5F7.js → chunk-NILBFAPG.js} +14 -8
  43. package/dist/chunk-OZNBF4L3.js +23 -0
  44. package/dist/{verify-G6V4D2G7.js → chunk-PCZJO5TI.js} +127 -42
  45. package/dist/chunk-QQK64KLB.js +1360 -0
  46. package/dist/{chunk-6EJV5X2W.js → chunk-QQL5RT5M.js} +979 -1619
  47. package/dist/{chunk-LZSJBIVT.js → chunk-QWU7ZBO7.js} +70 -720
  48. package/dist/{chunk-2EHAIA3X.js → chunk-RD5U66HV.js} +3 -3
  49. package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
  50. package/dist/{chunk-OQ33BKR3.js → chunk-TTNYS3EA.js} +3 -60
  51. package/dist/chunk-TW3WDMVS.js +677 -0
  52. package/dist/chunk-TZSKNMZG.js +434 -0
  53. package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
  54. package/dist/{chunk-7EYHLWU7.js → chunk-UZHIZC5S.js} +7 -7
  55. package/dist/{chunk-QTYWRVRA.js → chunk-VAWWTKDP.js} +8 -8
  56. package/dist/{chunk-X75S7HFS.js → chunk-VEZEGCGW.js} +214 -20
  57. package/dist/{chunk-OHHN2SO4.js → chunk-VMNQ6OZA.js} +98 -202
  58. package/dist/chunk-VSNATDE6.js +122 -0
  59. package/dist/chunk-W6GROXXM.js +69 -0
  60. package/dist/chunk-WPQLXFOZ.js +375 -0
  61. package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
  62. package/dist/{chunk-3ZXDFGR5.js → chunk-X6COSD2O.js} +5 -5
  63. package/dist/chunk-ZGVHUX3M.js +66 -0
  64. package/dist/{chunk-MAW544W2.js → chunk-ZWMF7253.js} +4 -4
  65. package/dist/{chunk-MQSRRFWA.js → chunk-ZYKPLLNQ.js} +563 -546
  66. package/dist/cli/index.js +27 -23
  67. package/dist/{clio-4LY5K2AC.js → clio-J5JIOIDS.js} +7 -6
  68. package/dist/{code-nav-7AX6FYE6.js → code-nav-AXCXSBHX.js} +5 -3
  69. package/dist/{config-GTLUW2PR.js → config-OEBMIN2U.js} +37 -27
  70. package/dist/{configure-R6A64DHX.js → configure-PUQOSIXQ.js} +16 -13
  71. package/dist/{context-5VKGUVJJ.js → context-EKDCKUUZ.js} +82 -7
  72. package/dist/{context-RW5HC47S.js → context-MGSE4Z2T.js} +33 -23
  73. package/dist/{context-JFZEJ7W5.js → context-URSXPBCK.js} +17 -9
  74. package/dist/{context-clear-6ZHBAZZT.js → context-clear-KDAJRNUK.js} +33 -23
  75. package/dist/context-working-set-SBKMPPI2.js +1552 -0
  76. package/dist/{dispatch-runner-VKBRCWQC.js → dispatch-runner-MSWN72NK.js} +43 -29
  77. package/dist/{doctor-KI767GSN.js → doctor-7BSE27PJ.js} +10 -10
  78. package/dist/{eval-XSSNATB4.js → eval-IZGDOO4H.js} +9 -8
  79. package/dist/{evidence-UA6AWDQQ.js → evidence-SR7WXB5B.js} +51 -23
  80. package/dist/{evolve-QNTFGV6Z.js → evolve-K7VE2CBX.js} +30 -20
  81. package/dist/{fleet-Q7UOMUSG.js → fleet-7XMJNQNF.js} +48 -38
  82. package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-AQNAH644.js} +3 -3
  83. package/dist/{init-WBB65ZHQ.js → init-JGNPAYXT.js} +41 -31
  84. package/dist/{memory-MD3O64RI.js → memory-4ALKDJ4Q.js} +32 -22
  85. package/dist/{models-BZU34YWD.js → models-ZMMLFJNN.js} +22 -19
  86. package/dist/{monitor-MEQA5C3I.js → monitor-2F3T5KHP.js} +55 -43
  87. package/dist/{orchestrator-CGFKEP27.js → orchestrator-ORHT43JB.js} +2507 -1896
  88. package/dist/{reset-L2FQEE3E.js → reset-NXGTYNUO.js} +4 -3
  89. package/dist/{run-IV4Q6RLN.js → run-RF4WJGMT.js} +51 -41
  90. package/dist/{share-S5BZQC5I.js → share-UT3W6E4M.js} +5 -4
  91. package/dist/{skills-LQEKRDTN.js → skills-PSACKC5Q.js} +2 -2
  92. package/dist/{skills-eval-3DC4HEWS.js → skills-eval-WJSI55RZ.js} +34 -24
  93. package/dist/{targets-C4SSGQOB.js → targets-PIIRAOYS.js} +23 -20
  94. package/dist/{terminal-lease-IT5JW2NR.js → terminal-lease-ULWXWNVY.js} +5 -4
  95. package/dist/{upgrade-7TT7SQ3G.js → upgrade-346TZ6AV.js} +18 -17
  96. package/dist/{usage-GV4PKT3M.js → usage-6KKXR32N.js} +34 -24
  97. package/dist/verifiers-4UUM6TEE.js +1214 -0
  98. package/dist/verify-X5HDROLA.js +25 -0
  99. package/dist/{wiki-generate-DQF6Z66B.js → wiki-generate-7STOCIFZ.js} +42 -31
  100. package/dist/worker/entry.js +33 -24
  101. package/docs/README.md +8 -7
  102. package/docs/acp.md +1 -1
  103. package/docs/alcf-provider.md +1 -1
  104. package/docs/architecture.md +2 -2
  105. package/docs/artifact-versions.md +1 -1
  106. package/docs/built-in-agents.md +1 -1
  107. package/docs/capacity-and-scheduling.md +1 -1
  108. package/docs/commands-and-modes.md +53 -21
  109. package/docs/config-knobs-audit.md +1 -2
  110. package/docs/configuration-and-targets.md +15 -1
  111. package/docs/context-engine.md +64 -12
  112. package/docs/context-working-set.md +194 -0
  113. package/docs/development-pipeline.md +1 -1
  114. package/docs/documentation-coverage.md +5 -5
  115. package/docs/documentation-guide.md +6 -5
  116. package/docs/environment-variables.md +2 -1
  117. package/docs/eval-runner.md +1 -1
  118. package/docs/evals-internal.md +14 -1
  119. package/docs/evidence-and-memory.md +74 -2
  120. package/docs/evolution.md +1 -1
  121. package/docs/exit-codes-and-output.md +1 -1
  122. package/docs/extensions-and-sharing.md +2 -2
  123. package/docs/fleet-dispatch.md +22 -7
  124. package/docs/glossary.md +21 -1
  125. package/docs/installation-and-lifecycle.md +6 -6
  126. package/docs/middleware-and-components.md +1 -1
  127. package/docs/model-catalog.md +7 -9
  128. package/docs/observability.md +4 -4
  129. package/docs/performance-methodology.md +2 -2
  130. package/docs/proactive-memory.md +1 -1
  131. package/docs/prompt-envelope-and-tools.md +4 -4
  132. package/docs/provider-adapter-cookbook.md +1 -1
  133. package/docs/release-cut-checklist.md +35 -35
  134. package/docs/safety-model.md +23 -4
  135. package/docs/scientific-validation.md +21 -3
  136. package/docs/session-lifecycle.md +3 -3
  137. package/docs/skills-marketplace.md +1 -1
  138. package/docs/tool-usage.md +79 -12
  139. package/docs/trace-store.md +1 -1
  140. package/docs/troubleshooting.md +1 -1
  141. package/docs/tui-design.md +2 -2
  142. package/docs/worker-dispatch-mechanics.md +11 -1
  143. package/package.json +8 -11
  144. package/skills/meta/clio-test/SKILL.md +20 -17
  145. package/skills/meta/clio-test/evals.md +3 -3
  146. package/skills/meta/clio-test/references/harness.md +35 -6
  147. package/skills/meta/clio-test/references/test-map.md +20 -10
  148. package/skills/registry.yaml +2 -2
  149. package/skills/skill-marketplace.json +1 -1
  150. package/src/cli/context-working-set.ts +513 -0
  151. package/src/cli/context.ts +8 -0
  152. package/src/cli/evidence.ts +20 -2
  153. package/src/cli/index.ts +4 -0
  154. package/src/cli/verifiers.ts +325 -0
  155. package/src/core/bash-exec.ts +39 -14
  156. package/src/core/bus-events.ts +19 -4
  157. package/src/core/config.ts +54 -0
  158. package/src/core/defaults.ts +50 -3
  159. package/src/core/git-commit-attribution.ts +46 -21
  160. package/src/core/verification-scripts.ts +6 -0
  161. package/src/domains/agents/builtins/verifier.md +3 -0
  162. package/src/domains/config/classify.ts +1 -0
  163. package/src/domains/config/keybindings.ts +3 -3
  164. package/src/domains/context/working-set/contract.ts +161 -0
  165. package/src/domains/context/working-set/defaults.ts +28 -0
  166. package/src/domains/context/working-set/engine.ts +203 -0
  167. package/src/domains/context/working-set/fold.ts +62 -0
  168. package/src/domains/context/working-set/horizon.ts +38 -0
  169. package/src/domains/context/working-set/marker.ts +103 -0
  170. package/src/domains/context/working-set/path-index.ts +436 -0
  171. package/src/domains/context/working-set/payload.ts +152 -0
  172. package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
  173. package/src/domains/context/working-set/policies/index.ts +21 -0
  174. package/src/domains/context/working-set/policies/structural.ts +160 -0
  175. package/src/domains/context/working-set/project.ts +132 -0
  176. package/src/domains/context/working-set/protect.ts +109 -0
  177. package/src/domains/context/working-set/recall.ts +177 -0
  178. package/src/domains/context/working-set/replay/controls.ts +112 -0
  179. package/src/domains/context/working-set/replay/load-clio.ts +199 -0
  180. package/src/domains/context/working-set/replay/metrics.ts +185 -0
  181. package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
  182. package/src/domains/context/working-set/replay/report.ts +139 -0
  183. package/src/domains/context/working-set/replay/runner.ts +325 -0
  184. package/src/domains/context/working-set/replay/synthetic.ts +422 -0
  185. package/src/domains/context/working-set/replay/trace.ts +21 -0
  186. package/src/domains/context/working-set/visible.ts +54 -0
  187. package/src/domains/evidence/build.ts +112 -45
  188. package/src/domains/evidence/eval.ts +24 -7
  189. package/src/domains/evidence/index.ts +53 -0
  190. package/src/domains/evidence/ordering.ts +12 -0
  191. package/src/domains/evidence/run-trust.ts +221 -0
  192. package/src/domains/evidence/store.ts +46 -6
  193. package/src/domains/evidence/trust-status.ts +854 -0
  194. package/src/domains/evidence/types.ts +26 -0
  195. package/src/domains/middleware/memory-intervention.ts +3 -0
  196. package/src/domains/middleware/stalled-turn.ts +165 -4
  197. package/src/domains/safety/autonomy.ts +1 -1
  198. package/src/domains/safety/default-path-policy.ts +8 -0
  199. package/src/domains/safety/finish-contract.ts +4 -3
  200. package/src/domains/safety/policy-engine.ts +48 -6
  201. package/src/domains/session/compaction/compact.ts +23 -1
  202. package/src/domains/session/compaction/cut-point.ts +2 -0
  203. package/src/domains/session/compaction/tokens.ts +16 -1
  204. package/src/domains/session/context-ledger.ts +2 -0
  205. package/src/domains/session/entries.ts +107 -1
  206. package/src/domains/session/manager.ts +9 -2
  207. package/src/domains/session/migrations/index.ts +22 -3
  208. package/src/engine/acp/server.ts +3 -0
  209. package/src/engine/agent.ts +18 -1
  210. package/src/engine/session.ts +9 -3
  211. package/src/entry/orchestrator.ts +16 -4
  212. package/src/interactive/chat-loop-messages.ts +18 -6
  213. package/src/interactive/chat-panel.ts +571 -244
  214. package/src/interactive/chat-renderer.ts +79 -39
  215. package/src/interactive/context-meter.ts +10 -0
  216. package/src/interactive/context-overlay.ts +81 -6
  217. package/src/interactive/context-recall-command.ts +110 -0
  218. package/src/interactive/editor-submit.ts +26 -1
  219. package/src/interactive/footer/widgets.ts +22 -20
  220. package/src/interactive/footer-panel.ts +6 -1
  221. package/src/interactive/interactive-application.ts +2 -0
  222. package/src/interactive/interactive-event-projection.ts +12 -0
  223. package/src/interactive/interactive-slash-runtime.ts +49 -8
  224. package/src/interactive/model-session-replay.ts +21 -0
  225. package/src/interactive/overlay-general-openers.ts +6 -0
  226. package/src/interactive/overlay-session-lifecycle.ts +8 -4
  227. package/src/interactive/overlays/ask-user.ts +146 -24
  228. package/src/interactive/renderers/tool-execution.ts +167 -56
  229. package/src/interactive/session-transcript.ts +2 -2
  230. package/src/interactive/slash-commands.ts +29 -2
  231. package/src/interactive/status/index.ts +12 -1
  232. package/src/interactive/status/reasoning.ts +87 -0
  233. package/src/interactive/status/summary.ts +13 -2
  234. package/src/interactive/transcript-detail.ts +120 -0
  235. package/src/interactive/turn-context.ts +238 -88
  236. package/src/interactive/turn-middleware.ts +6 -6
  237. package/src/tools/agent-tools.ts +11 -4
  238. package/src/tools/bash.ts +144 -82
  239. package/src/tools/builtin-tool-catalog.ts +18 -6
  240. package/src/tools/context/index.ts +105 -3
  241. package/src/tools/context/surface.ts +3 -2
  242. package/src/tools/core-bootstrap.ts +21 -0
  243. package/src/tools/dispatch-runner.ts +9 -7
  244. package/src/tools/monitor.ts +28 -20
  245. package/src/tools/presentation.ts +107 -0
  246. package/src/tools/registry.ts +65 -7
  247. package/src/tools/result-disposition.ts +550 -0
  248. package/src/tools/result-shaping.ts +262 -19
  249. package/src/tools/safe-exec.ts +2 -0
  250. package/src/tools/verify/authoring.ts +1119 -0
  251. package/src/tools/verify/catalog.ts +346 -0
  252. package/src/tools/verify/index.ts +13 -3
  253. package/src/tools/verify/scripts.ts +135 -37
  254. package/src/tools/verify/surface.ts +9 -5
  255. package/src/tools/worker-evidence.ts +35 -12
  256. package/dist/chunk-MNA4JGU4.js +0 -255
  257. package/dist/chunk-SRF2PJNW.js +0 -184
  258. package/dist/chunk-T6YILFSB.js +0 -80
  259. package/dist/chunk-VAKQQHWR.js +0 -434
package/CHANGELOG.md CHANGED
@@ -1,528 +1,339 @@
1
1
  # Changelog
2
2
 
3
- Public release notes for Clio Coder are documented here. This file is kept
4
- intentionally short for users, operators, and community contributors. The
5
- implementation detail and verification notes behind each entry live in the
6
- commit history, where every change carries the transcript and gates that
7
- produced it.
8
-
9
- Versions follow semantic versioning for a pre-1.0 project: minor versions may
10
- still change interfaces.
3
+ All notable changes to Clio Coder are documented in this file. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and versions follow Semantic Versioning; pre-1.0 minor releases may include incompatible changes.
4
+
5
+ ## 0.3.4 - 2026-08-22
6
+
7
+ ### Added
8
+ - A typed project verifier catalog at `.clio-coder/verifiers.yaml` (#170). Each check declares a stable id, description, exact argv vector, repository-relative cwd, bounded timeout, and tags. `verify()` lists package scripts and catalog checks through one `DeclaredCheck` projection, and `verify(check=<id>)` runs the admitted argv through safe-exec with no shell, no model-text interpolation, and no widening of workspace authority. Shell strings, escaping cwds, duplicate ids, unknown fields, unsupported versions, and over-cap values fail closed with diagnostics that name the field and the cap.
9
+ - Guided verifier authoring through `clio-coder verifiers discover|author|validate|dry-run|add|edit|rename|remove` (#174). Discovery recognizes package scripts, Cargo manifests, CMake presets, declared Python runners, Go modules, and existing validation-contract commands with source provenance, and proposes argv vectors labeled project-declared or toolchain-defined. Every preview shows path, cwd, timeout, tags, and effective execution authority; nothing is written or executed before `--yes`, validation uses the production catalog parser, and dry runs use the production verify path. Projects with no declared command get an explicit manual-entry path instead of a guessed command.
10
+ - One canonical tool-result disposition contract with independent presentation and model-context axes (#165). A tool declares how the operator sees a result and, separately, whether the model receives full content, a bounded excerpt, a deterministic code-produced summary, or metadata only. Typed result metadata records captured, displayed, and context byte counts, the requested versus applied mode, truncation, the offload path, and summary provenance. Exit status, error state, safety facts, and retrieval instructions survive every context mode.
11
+ - Canonical Bash output dispositions with a tail-biased bounded default, deterministic redacted diagnostic summaries, metadata-only retrieval, budget-admitted full context, and explicit byte and termination facts (#172). `output_policy` is optional on the Bash tool; omitting it preserves the previous tail behavior.
12
+ - A six-axis canonical trust status for runs (#154): artifact integrity, validation grounding, independent review, context provenance, autonomy enforcement, and completion evidence, each with explicit `absent`, `unknown`, and `not_applicable` states and a named source and authority. Composition never promotes one axis from another; the no-promotion rules are enforced at the adapters and the evidence composition boundary. Evidence bundles gain `trust-status.json`.
13
+ - Non-destructive working-set eviction. When context pressure crosses `compaction.threshold`, Clio now records which tool-result bodies and closed-turn thinking blocks leave the model's working set instead of rewriting them out of the session. The bodies stay in the ledger, the transcript keeps showing them, and each one is replaced in model replay by a one-line marker naming the ref, the reason, the size, and the exact call that brings it back.
14
+ - Exact recall by ref. The model reads an evicted body back with `context(scope="recall", ref="<turnId>")`; the operator reads one into the transcript with `/context recall <ref>`, which never enters model context. A recall does not un-evict: the marker stays byte-identical so the provider prefix cache is untouched, and repeated recalls of one ref are the churn signal.
15
+ - Two eviction policies. `structural-v1` is the default: it selects by what the session did since (`stale_after_mutation`, `superseded_read`, `failure_resolved`, `listing_consumed`, `thinking_turn_closed`) and falls back to age only under pressure. `age-horizon` reproduces the previous age-based selection, minus results whose body is below `context.workingSet.minEvictableTokens`. Replayed over the seeded procedural corpora with the summary stage modeled, `structural-v1` cuts the number of lossy summary compactions per 300-turn science trace from 21.5 to 8.8 at a 64k budget and from 8.9 to 3.1 at 128k, retains at or above `age-horizon` at every budget, and beats random eviction on precision by 2.3x or more; the default-policy rule, the cold-prefix cost it pays for finer batching, and the full grid are under `benchmarks/results/context-replay/`.
16
+ - `/context` reports the working set: policy, evicted items, evicted tokens, events, recalls, and churn. Evicted tool rows carry a dim `evicted · <reason>` tag in the transcript.
17
+ - Cache-honesty attribution for eviction. An applied event stamps `working_set_evict` on the next assistant entry's `promptCache.expectedColdReasons`, and `/context` reports `last cold turn: working-set eviction (expected)` instead of warning about a cold backend it caused itself.
18
+ - `clio-coder context replay --sessions <path>...` replays Clio ledgers, and `--synthetic <ids>` replays seeded procedural science-coding corpora (`science-long`, `refactor`, `exploration`), through the live eviction code with `none`, `random`, and `oracle` controls and reports retention, precision, tokens evicted, recall tokens, cold prefix tokens, saturation, turns to first summary, and summaries per trace under a modeled summary stage; `clio-coder context working-set --session <id|path>` prints one session's working-set fold and path index. The procedural corpora replace the private Claude Code transcripts the first tables were built on, so the committed tables under `benchmarks/results/context-replay/` rebuild byte for byte on any checkout.
19
+ - New guide: `docs/context-working-set.md`.
20
+
21
+ ### Changed
22
+ - Tool offload files are content-addressed: `<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt` instead of the tool call id or a timestamp, so identical captures share one file and the `retrieve=` header, the eviction marker, and the resume transcript carry the same bytes every time.
23
+ - Editing an existing `.clio-coder/verifiers.yaml` through `clio-coder verifiers add|edit|rename|remove` now mutates the operator's file in place: comments and on-disk order survive, only changed fields move, and a revision that changes nothing writes the file back byte for byte.
24
+ - Evidence rows, the latest gate decision and finish contract, and the verifier catalog order by code point instead of locale collation, so two machines rebuild the same bundle in the same order.
25
+ - Receipt inspection, evidence bundles, monitor output, and worker evidence derive their trust facts from one canonical derivation (#157). The public `evidence_verification=<state>/<basis>` token is unchanged. An integrity-failed receipt now renders `receipt_integrity=FAILED reason=…` and labels its claims `worker claims (unverified prose)` in both dispatch and monitor output.
26
+ - Session format version 4. The bump is additive: it adds the `contextEviction` and `contextRecall` records and changes no existing entry, so a version 3 session migrates to 4 in place on open with nothing rewritten. Only a session written by a newer build is refused. The bump is one-way for the operator, and a 0.3.3 binary cannot open a session this release wrote.
27
+ - New settings under `context.workingSet`: `enabled` (default `true`), `policy` (default `structural-v1`), `target` (default `0.6`), `protectLastTurns` (default `6`), and `minEvictableTokens` (default `200`). `compaction.excludeLastTurns` now governs only the legacy mask path.
28
+ - Compaction reports a `working_set` stage on `ContextPruned`, and the middleware `on_compaction` hook gains the `working_set_evict` and `working_set_recall` stages.
29
+
30
+ ### Removed
31
+ - The Claude Code transcript loader for `context replay` and its `--format` flag. Replay inputs are Clio ledgers and the seeded procedural corpora.
32
+ - `TRUST_STATUS_NO_PROMOTION_RULES`, `adaptEvidenceFindingsValidationStatus`, and `adaptEvidenceLinkContextStatus` from the evidence barrel. The no-promotion rules are enforced at the adapters and the composition boundary and are written out in `docs/evidence-and-memory.md`; the two adapters were reachable from no builder.
33
+
34
+ ### Security
35
+ - A project-catalog `verify(check=<id>)` no longer sits in the no-prompt set. The policy engine resolves the check against `.clio-coder/verifiers.yaml`, scans the declared argv with the damage-control rules and the zero-access read guard, and tags it unrecognized, so `auto-edit` confirms it once with the argv shown and `full-auto` runs it. Package-script checks and `verify(check="frontend")` are unchanged. `.clio-coder/verifiers.yaml` and `.clio-coder/safety.yaml` are read-only to the model's `write`, `edit`, and bash redirect paths by default: before this, a model at the default `auto-edit` level could write the catalog and run any argv through `verify` without a prompt.
36
+
37
+ ### Fixed
38
+ - A completed conversational offer such as "point me at it and I'll get moving" no longer triggers an automatic continuation or the "turn still has open work" footer warning (#178). The stalled-turn detector now requires an announced concrete action: it recognizes inflected verbs ("I'll be running the tests"), announced paths and commands ("Let me open src/cli/index.ts", "I'll npm run build"), and keeps conditional offers, "let me know" phrasing, questions, and wait statements ("I'll wait for your go-ahead before I touch index.ts") suppressed. The one-continuation cap for genuine stalls is unchanged.
39
+ - A completion-contract audit row can no longer ground validation. The row is the run's own self-report and feeds completion evidence only; validation grounding is filled by executions the session ledger observed. A run whose receipt fails integrity contributes no verified field, its completion self-report downgrades to `unknown`, and the `no-validation` warning is restored. A blank or whitespace-only identifier in one audit row no longer aborts the whole evidence build.
40
+ - Tool-result summaries are honest about omission. Head and tail slices are disjoint, a scratch offload is written only when the model projection actually omits content, whitespace-only output is complete rather than truncated, NUL bytes are removed from model context while presentation and the offload keep the captured bytes, the head-tail strategy honors `redact`, and a throwing disposition resolver fails closed to metadata-only with the cause recorded in the result metadata, the model header, and the transcript row.
41
+ - `clio-coder verifiers … --yes` no longer prints "no file has been written" immediately before writing the catalog; the confirmed preview is rendered after authorization.
42
+ - Auto-compaction no longer destroys observations. The stale-observation mask rewrote persisted bodies through `session.replaceEntries`, so masked content was gone from `/resume`, `/tree`, `/fork`, and the HTML export as well as from the model. `CLIO_CODER_LEGACY_MASK=1` restores that stage for one release as a compatibility escape hatch; it is removed in the next release.
43
+
44
+ ## 0.3.3 - 2026-08-21
45
+
46
+ ### Changed
47
+ - Unified transcript detail under `/output minimal|default|verbose`, with consistent per-block and all-block tool/thinking overrides that reset when the output level is reapplied.
48
+ - Folded Bash execution bodies by default while retaining concise command, outcome, timing, size, and bounded failure evidence on the transcript row (#166, #177).
49
+ - Rendered reasoning as stream-ordered thinking segments and made interview prompts true fullscreen workspaces (#171).
50
+
51
+ ### Fixed
52
+ - Preserved live, interrupted, and replayed reasoning order and token provenance, including provider-reported zero-output turns.
53
+ - Preserved complete replay bodies for HTML export and aggregated multi-call replay receipts.
54
+ - Kept failure excerpts and mutation diffs inside narrow terminal frames, including at 40 columns.
55
+ - Replaced internal tool-call labels such as `bash(...)` with operator-facing action descriptions in live, replayed, blocked, and exported transcript rows.
56
+ - Refreshed the footer immediately after `/output` changes and kept explicit fold choices scoped to the intended tool or thinking stretch.
57
+ - Rechecked commit-attribution repository state and repaired missing or damaged cached hook wrappers before reuse.
11
58
 
12
59
  ## 0.3.2 - 2026-08-20
13
60
 
14
- This release upgrades the engine and hardens the terminal agent. The three engine SDK libraries move to 0.84.0, and a pass over the interactive surface ported the behaviors listed first below. A day of audits against the live code and a Qwen3.8-27B fleet on llama.cpp and LM Studio then produced the defects listed after them; every one is fixed here, each code fix carries a regression test, and the tickets (#110 through #136) carry the evidence.
15
-
16
- - Added default-on, evidence-aware Git commit provenance. Commits created through Clio receive only the `Assisted-by`, `Tested-by`, `Reviewed-by`, and contributor-compatible `Co-authored-by` trailers justified by trusted material authorship, successful validation commands, and independent passing review; the human author and committer remain unchanged, normal external commits are untouched, replayed history is skipped, and Settings -> Advanced can disable attribution with byte-for-byte message preservation.
17
- - Made the `/settings` filter narrow the catalog per keystroke (#135), matching `/model` and `/resume`. The `[/]` editor's draft is now the live query, so rows, section match counts, and the empty state track each character; Enter commits the query and Esc restores the committed one.
18
- - Named the winning project handbook in `/context` (#136). The overlay showed the preload class only (`full (0.3kB, 11 lines)`) and never said which `CLIO-CODER.md` or `CLIO-CODER.override.md` produced it, which the override semantics make ambiguous; it now lists the effective chain, ancestor to nearest, with workspace-relative paths.
19
- - Proved at the process boundary that the real built worker entry consumes Clio's injected compile-cache pair before any descendant it spawns can see it (#148). A hermetic smoke drives `dist/worker/entry.js` through one stubbed inference turn whose tool call is a real bash descendant launching a generic Node child; neither inherits Clio's cache directory or marker, while an operator-supplied `NODE_COMPILE_CACHE` reaches both unchanged through matching and mismatched spoofed markers on the direct `clio-coder worker` path.
20
- - Admitted prompts in the order they were typed during the instant shell's boot window. Two submissions in that window could be answered out of order: `ChatLoop.submit` had no admission gate, so a second submit arriving while the first was still awaiting its target probe and prompt compile ran the same pipeline concurrently, reached the engine first, and left the first recorded in the ledger but never sent. Each submit now waits for the previous one to own the stream or return, then lands as the next steer or follow-up, and the engine's raw active-prompt invariant no longer leaks into the transcript.
21
- - Made the fullscreen TUI render its transcript. `terminal.tuiMode: fullscreen` painted the header, composer, and footer but left the transcript region blank because the Stage 0 to Stage 1 terminal-lease handoff never attached the transcript to the alternate-screen root.
22
- - Made the footer's live `↓` and `⚡` counters report the streaming turn instead of the previous turn's totals. A turn cancelled while the model was still reasoning now settles the footer, the `turn · in/out` receipt, and the ledger's closing row to the same estimated spend instead of `↑0 ↓0`.
23
- - Kept the draft editable when an unknown slash command is rejected. `/compact` and `/skill:foo` failed closed correctly but cleared the editor; rejected input now stays in place and never enters prompt history.
24
- - Stopped shipping `docs/html` in the npm package. The Markdown guides under `docs/` remain bundled; `clio-coder docs` serves the interactive blueprints from a source checkout and, from an npm install, says where they are instead of serving nothing.
25
- - Replaced the bold experimental warning in the launchpad, the one-line session header, and the CLI banner with facts. The session header now names where Clio is working, which route answers, and what the context is ready for; the maturity caveat is stated once, in the README.
26
-
27
- - Upgraded the three Pi SDK libraries to 0.84.0, bringing Clio the release's provider, OAuth cancellation, terminal rendering, and platform fixes while preserving its own agent loop, session store, and regular-screen TUI behavior. Clio now uses Pi's explicit `TuiMainScreen` renderer and forwards concrete abort signals through built-in and ALCF OAuth refreshes, so a cancelled authentication flow no longer leaves token work running in the background.
28
- - Added the redacted action class, asking safety axis, and target beneath a tool row parked for approval. The facts use the same transient view as the permission overlay, disappear when the call resumes or settles, and never enter the session ledger.
29
- - Added Page Up and Page Down navigation to the `/resume` picker, moving by the 12 visible rows while preserving Clio's existing session filtering and ledger behavior.
30
- - Rendered successful `edit` and `write` calls as numbered diffs in their live tool rows, with word-level emphasis and Clio's add/remove theme colors in regular and fullscreen modes. Replayed and exported diffs remain plain text.
31
- - Made `/export` write a themed, self-contained, size-bounded HTML transcript by default, with terminal styling converted to HTML and each tool call kept in a semantic row. Passing a `.md` path preserves the existing plain Markdown export.
32
- - Added durable task, decision, and workspace-output inspection. `/tasks` now combines live work, terminal board history, recorded workspace outputs, and a project operator inbox; `/decisions` preserves settled interviews and operator corrections; `/view` renders task-ledger snapshots and active-branch workspace outputs. The model-facing task ledger and inbox bookkeeping intentionally stays audited as read class without granting workspace mutation authority, and `Alt+B` plus `Alt+D` are approved application-boundary shortcuts for the two boards. Workspace output loading re-resolves the current workspace root and target before every read, refuses symlink swaps outside that root, and retains the recorded missing-file result.
33
- - Ranked `/model` search with fuzzy matching over provider-qualified search text. Direct `target/model` matches outrank proxy-carried model ids while Clio's availability, authentication, health, runtime, favorites, and recent-model facts remain intact.
34
- - Routed prompt-template arguments through the engine's command-argument parser and substitution, retiring Clio's second parser. Templates support quoted arguments, `$1` through `$9`, `$@`, `$ARGUMENTS`, and argument slices.
35
- - Added a maintained engine SDK ownership table, thin-wrapper watch list, regression net, and five-step dependency-upgrade checklist (`docs/pi-boundary.md`) so each release identifies upstream fixes that Clio should inherit and Clio-owned deltas that must remain.
36
- - Added a checked engine declaration-surface snapshot (`docs/pi-surface.json`). Dependency upgrades now fail lint when an imported SDK symbol changes or disappears and report newly exported primitives for review.
37
- - Set `AI_AGENT=clio-coder` for spawned commands, fleet workers, registered code steps, and hooks so generic developer tooling can attribute child processes to Clio Coder.
38
- - Rebuilt the tool transcript as one stable lifecycle block from streamed arguments through live progress and settlement. Expanded calls now expose typed secondary arguments, exit status, counts, displayed and total bytes, truncation, timeouts, usage, added tools, context exclusion, and full-output paths; mixed image results never print base64, admission refusals never masquerade as executions, and operator `!`/`!!` bash commands stream and settle with the same grammar as model tools.
39
- - Let the engine assemble Anthropic thinking requests from the active thinking level, including adaptive effort and bounded token budgets. Clio removed its competing Anthropic payload rewrite and retains only the OpenAI Responses summary field that the engine's agent loop does not expose.
40
- - Aligned resumed and forked custom session entries with the engine's model-facing message text. Compaction summaries, branch summaries, and local bash executions now reach the model through the engine's message-text helpers instead of Clio-maintained copies, while Clio keeps its own durable ledger format.
41
- - Replaced Clio's TypeBox `anyOf` string-enum helper with the engine's compact string-enum schema. Tool schemas keep their exact accepted values and descriptions while avoiding a duplicate schema implementation and preserving the simpler wire form across providers.
42
- - Connected native tool progress to the engine's cumulative tool-update callback. Bash now streams a throttled, bounded output tail into its existing live transcript row while it runs, so operators see useful command progress without changing approval decisions, final result shaping, or the session ledger.
43
- - Replaced Clio's copied tool-output truncation implementation with the engine's UTF-8-safe head, tail, line, and size-formatting primitives. Read, search, context, and shell output retain Clio's 16 KiB per-observation ceiling while sharing the engine's maintained byte and line handling.
44
- - Gave every slash operation one canonical spelling across parsing, autocomplete, help, documentation, and skill guidance. `/quit`, `/context compact`, `/model`, `/settings`, `/skill <name>`, `/run --agent-profile`, and `/run --runtime` are now the only accepted forms; retired aliases fail closed and remain editable, and prompt templates that collide with built-in commands are diagnosed and ignored consistently in interactive and headless runs.
45
- - Completed fullscreen prompt navigation by marking every Clio user turn with the OSC 133 semantic prompt sequence, so `Ctrl+Shift+Up` and `Ctrl+Shift+Down` now move between real transcript prompts. Clio's editor chrome and transcript export now use the engine's terminal-sequence stripper instead of partial local ANSI regexes, and the optional presentation-settings path retains regular-screen defaults.
46
- - Made OAuth cancellation reach the complete production credential path. Active agent and background-request signals now cross `providers.auth`, credential mutations carry the engine's signal-aware operation options, cancelled lock waiters leave the live holder alone, and a token aborted before persistence never appears in Clio's in-memory credential view.
47
- - Fixed refused local commands and unresolved worker steering so the editor's pre-submit clearing restores the draft instead of losing it or recording rejected input in prompt history.
48
- - Replaced Clio's copied transient-provider regex with the engine's retryable-error classifier. DNS and WebSocket failures now enter the bounded retry ladder, quota, usage-limit, and billing failures stop immediately even when they contain an HTTP retry status, and Clio retains only its longer local-model loading delay and cancellable countdown.
49
- - Added directory-scoped `CLIO-CODER.override.md` handbooks with an explicit subtree boundary. An override replaces inherited guidance for its directory and descendants without affecting siblings, deeper handbooks may add narrower instructions, prompt blocks retain source paths, malformed overrides fail closed, and reset never deletes an override.
50
- - Added `Ctrl+P` and `Ctrl+N` prompt history through the editor's dedicated history actions. Accepted chat, slash, local-command, steering, follow-up, and interrupt inputs are available in process-local history, consecutive duplicates collapse, rejected inputs remain editable without polluting history, and browsing forward restores the unfinished draft.
51
- - Added an opt-in fullscreen TUI built on the engine's alternate-screen layout. Settings → Terminal can select a sticky composer and footer beneath an independently scrollable transcript, with follow-end behavior and a mouse-draggable scrollbar whose visibility is configurable; regular terminal scrollback remains the default.
52
- - Added terminal-native Mermaid diagrams and Unicode LaTeX to finalized assistant transcripts. Supported `mermaid` fences now become themed box-drawing diagrams when they fit, inline and display math render without a browser, and unsupported or oversized diagrams remain readable as their original source.
53
- - Made generic OpenAI-compatible, llama.cpp, LM Studio, and vLLM streams tolerant of servers that omit `finish_reason`. The engine now infers a normal or tool-use stop at the clean end of those local streams, so a complete answer is no longer reported as a provider failure solely because its server omitted the optional marker.
54
- - Routed local OpenAI-compatible sampling through the engine's typed sampling-parameter request path and enabled its bounded `thinking_token_budget` support for vLLM. Family-specific temperature, top-p, top-k, min-p, presence-penalty, and repetition-penalty values still reach llama.cpp and LM Studio byte for byte, while vLLM now reserves room for the final answer instead of allowing reasoning to consume the complete response ceiling.
55
- - Kept the local `!` bash row live until its command settles. The row froze at its first frame (`running · 7ms`, `no output yet`) because the chat panel classified every replay block as settled; a replay block can now declare itself live, so the elapsed counter ticks, output streams in, and the row leaves the frozen history prefix only when the execution record says it finished.
56
- - Kept a one-column gutter between a truncated model id and the context column in the `/model` overlay, so a long id no longer runs into the ctx figure and reads as one token.
57
- - Made the persisted post-compaction token count report what compaction freed. The `/tree` compaction row echoed the pre-compaction figure (`~16276 -> ~16276 tokens`) because the estimator anchored on an assistant message whose usage described the old prompt; the after figure now drops the summarized history, matching the footer's reclaimed-context notice.
58
- - Made an LM Studio model id mean one loaded instance on one host (#113). LM Studio lists a downloadable key next to its loaded instances and, with LM Link, next to a peer's instances too, and sending the bare key made the server load a second copy of a 27B model behind the operator's back. Clio now resolves a requested id against the target host's loaded instances, never sends a bare key that already has an instance, prefers the target's `defaultModel` and then an instance no other configured LM Studio target reports, flags an id reported loaded by two hosts as an LM Link peer projection and never auto-selects it, and never triggers a load for a key that already has a resident instance. A key with no instance still loads just in time as before.
59
- - Stopped the llama.cpp residency path from taking a node out of service (#127, #134). Asking for a model id the router does not know used to unload the resident model first and then fail the load, leaving the router empty; the reconciler now refuses the eviction when the replacement is not in the router catalog and reloads what it evicted when a load is rejected for any other reason. A router that reports an idle model as `sleeping` is read as resident instead of being asked to load it again, which had failed every turn with `400 model is already running`.
60
- - Fixed `/new`, `/resume`, `/tree`, and `/fork` issued while a turn was streaming (#114). The session was switched before the turn settled, so the aborted assistant record landed in the new session as a phantom root and the original session kept a question with no answer; each path now cancels and awaits the in-flight turn so it seals into the session that owns it.
61
- - Made two dispatch processes stop overwriting each other's run ledger (#118). A ledger persisted every row it had read at open time, reverting a sibling's settled run to `running`, and orphan recovery then stamped that run `dead` with exit 1 without reading its own receipt, which said it succeeded. Only rows a process mutated win on merge, and recovery seals from a verified receipt before declaring a run abandoned.
62
- - Accepted a target whose settings still say `runtime: lmstudio-native` in the worker spec contract (#119), resolving the alias instead of failing three attempts with an internal mismatch string, and stopped retrying a spec-contract rejection at full cost.
63
- - Stopped a damaged `credentials.yaml` from wedging `clio-coder upgrade` forever (#121): the provider-rename migration no longer throws on a damaged store when it has nothing to rename, each successful migration is recorded before the next runs, and the upgrade error names `--skip-migrations`.
64
- - Made the headless JSON stream honor its documented contract (#122). `--json` no longer re-emits every message in full after streaming its deltas, and `--json-events terminal` emits only `session`, `turn_start`, `agent_end`, `turn_end`, and `notice` instead of 40 KB of message bodies for a one-word answer.
65
- - Brought CLI exit codes back to the documented table (#123): `trace` validates its command and arguments before the no-database courtesy message, `trace sql` mutation refusal and bare `extensions`/`skills` exit 2 with usage on stderr, `trace runs --json` exists, `fleet` rejects unknown flags, `$COLUMNS` widens piped `targets` output, and a real `upgrade` no longer says "would write it".
66
- - Restored the gemma-4 channel-marker stream filter (#126), now applied on every runtime whenever the resolved family is gemma-4 rather than only on the removed LM Studio SDK path, so chain-of-thought and tool-call channel text no longer leak into the transcript.
67
- - Fixed `--thinking max` resolving to the lowest active level on families whose effort map stops at `high` (#128); the effective level is now monotonic across the seven configured levels for every catalog family.
68
- - Made the probed context window win over the catalog's in every consumer (#129). The orchestrator's engine model was sized from the catalog's 262144 while the server served 131072, so `max_tokens` overran the real window on long turns and auto-compaction never fired before the limit; deployment limits now override model metadata at resolution time.
69
- - Made the LM Studio request carry the resolver's effective reasoning effort (#130), so `xhigh` reaches models that accept it, `mechanism: none` families are not asked to reason, and always-on families are not told `none`. A target's literal `lmstudio.request.reasoning` override of `low`, `medium`, or `high` now outranks the thinking dial but is clamped to the efforts the model advertises, so a model that reports only on and off receives `low`; `auto` keeps the resolver's mapping.
70
- - Made a stalled stream fail with its reason instead of `Request aborted` (#131). The reclassified stall (`stream stalled: no output from <target> for <n>s`) is now the assistant message and the receipt's `outcomeDetail`, each stalled attempt leaves one assistant row rather than an aborted row beside a corrected one, and a headless provider failure on stderr names the target, runtime, and endpoint it was talking to.
71
- - Bounded the estimated reasoning-token count to the reported output tokens (#132), gave every receipt the same `node` block on success and failure (#120), made `/export` reject unknown flags and extra arguments like every other command (#115), and marked truncated rows in the expanded notices panel with an ellipsis (#116).
72
- - Rewrote the Qwen3.8-27B catalog entry's prose as portable hardware classes (#133), removing one operator's node names, one-off timings, and a citation of a deleted source file, and added a catalog test that every cited source path exists.
73
- - Corrected two documentation claims (#117): `docs/session-lifecycle.md` described `/resume <sessionId>`, an argument the command does not take, and the configuration knob table said `defaults.maxTokens` must be at least 1 while 0 is accepted.
74
- - Isolated the test suite from the developer's machine (#110, #111). A contract test resolved the real `~/.config/clio-coder/settings.yaml`, overwrote it with malformed YAML, and deleted it; the shared harness now redirects every Clio root into the run's scratch home with `CLIO_CODER_REQUIRE_HOME_PREFIX`, guards it with a canary test, and the welcome-dashboard test no longer reads the checkout it runs in.
75
- - Gated `npm publish` on version coherence (#124): `scripts/check-release.mjs` fails when `package.json` and the top `CHANGELOG.md` heading disagree or the heading is still `Unreleased`, and the release checklist is rewritten for this version.
76
- - Replaced the LM Studio SDK path with one HTTP adapter. The canonical runtime is now `lmstudio`, with `lmstudio-native` retained as an in-memory compatibility alias and an upgrade migration for persisted runtime ids, websocket URLs, and stored credentials. Chat, reasoning streams, tools, and usage use `/v1/chat/completions`; probing and residency prefer `/api/v1/models` with v0 and OpenAI listing fallbacks; explicit load and unload use native REST and carry bearer authentication. A target without `lmstudio.load` keeps server just-in-time defaults, while an explicit block controls context length, flash attention, eval batch size, expert count, and KV-cache GPU offload. Every successful load records its returned instance id for this process, and unload or fallback swap refuses every pre-existing or LM Link instance. Same-key LM Link instances are independent, not duplicates. Request settings cover TTL, draft model, and reasoning effort. The LM Studio path never sends `chat_template_kwargs`; its seven Clio thinking levels map to `none`, `low`, `medium`, or `high`, with on/off-only models clamped to `low`. Configure verifies the exact LM Studio greeting before saving, direct and legacy ids both persist canonically, the model catalog no longer promises SDK behavior, and scripted HTTP contract tests cover v1, v0, auth, migration, configuration, loading, ownership-safe release, reasoning defaults, split tool calls, and usage without live model calls.
77
- - Extended the strict ACP v1 server into the durable control surface an external session-managing client needs without weakening generic clients. A process may now bind its one chat runtime with either `session/new` or standard `session/load`; load authorizes against canonical-workspace history, refuses any unended record it cannot prove is free, resets provider context to the durable pinned branch, and emits only bounded original transcript replay before its response. Namespaced methods list, label, and delete workspace sessions; expose and atomically patch only target, model, thinking level, and global autonomy; list and explicitly probe credential-free target projections; and read or override a bound session's autonomy for its next prompt. New/load results and durable metadata record bind-time target/model attribution, including the TUI `/new` path. Clients may opt into one permanent versioned `clio-coder/event` envelope whose first kind forwards loop-block detector counts without a fabricated tool id, command shape, path, or timestamp authority; clients that do not opt in receive no extension notification. A server-side approval timeout is now audited as `expired`, aborts the turn, suppresses model continuation, and fails the prompt with `permission_expired` instead of inventing a denial the model can retry. Workspace Git identity treats ignored nested scratch roots as non-Git and path-scopes dirty status and recent history for legitimate monorepo subdirectories while preserving the exact workspace as the authority boundary. ACP output additionally bounds live tool titles and location paths, session-list aggregate frames, live tool-call cardinality, and replay tool-call cardinality to fixed reader limits a client can rely on.
78
- - Hardened `clio-coder acp` into a strict, truthful ACP v1 server so an external client can drive it safely, and corrected `docs/acp.md` to the methods that actually exist. `initialize` must come first, exactly once, with `protocolVersion: 1`; the launch `--cwd` is realpath-canonicalized and is the process's only workspace, so every session opener must carry an absolute `cwd` that canonicalizes to it and the server never falls back to another directory or calls `chdir` after boot (the first `session/new` used to retarget the whole process to any existing directory it named). One successful new or load per process, since one chat instance backs the server and a second session id would share provider context and ledger ancestry. A prompt Clio cannot start (no orchestrator target, no model, unknown target, runtime resolution failure) now fails the `session/prompt` request with `prompt_not_admitted` and a closed-set reason instead of returning an empty `end_turn` with zero usage, which the ACP client could not tell from a model that chose to say nothing. Every error carries its machine-readable detail under `error.data._meta["clio-coder/error"]` and nothing else: no `Error.stack`, no echoed frame, no absolute path, no provider prose (`turn_failed` and `internal_error` messages are fixed host text and the original goes to stderr). Text, thought, tool `content`, `rawInput`, `rawOutput`, and `toolCallId` are bounded in UTF-8 bytes; each tool call has one wire identity across `tool_call`, `tool_call_update`, and `session/request_permission`, the permission request reuses the exact `rawInput`/`locations` snapshot the client already rendered and binds only to a currently open call, and only the exact `allow-once` option grants. `session/cancel` while a permission is parked settles the permission, the parked tool, and the prompt; `session/close` refuses under an active prompt and is idempotent otherwise; a cancelled or failed turn closes every open tool call with a terminal update; stdin EOF waits for the in-flight prompt to settle before the session store stops. The stdio transport bounds an input line at 1 MiB and answers `id: null` requests instead of dropping them. Protected by the ACP contract suites plus a spawned real-server smoke test that drives the built CLI with an isolated `CLIO_CODER_HOME` and a loopback provider stub through a text turn, an admission failure, a tool call with permission allow and reject, and cancel while a permission is parked.
79
- - Loaded the two wiki dispatch prompts through the same fragment loader every other prompt fragment uses instead of a hand-rolled `readFileSync` (#92). `prompts/fragments/wiki/{page,plan}.md` sat inside the directory `fragment-loader.ts` walks, but `walk()` explicitly skipped any directory named `wiki`; `context/wiki/prompts.ts` read the two files directly and did its own `{{token}}` substitution, with no id, no version, no content hash, and no hot reload. The skip was not an oversight: neither file had real YAML frontmatter, only a decorative leading and trailing `---` line, and `page.md`'s body has its own literal `---` fences (the front matter format the wiki *writer* must produce) that the loader's first-`---`-to-next-`---` frontmatter regex would have matched against instead of a real closing delimiter, corrupting the split and throwing at every Clio startup. Both files now carry real `id`/`version`/`description` frontmatter above their unchanged body, so the loader's existing delimiter search closes on the new marker instead; the old decorative `---` lines are kept as ordinary body text rather than stripped, so the substituted, trimmed prompt text `context/wiki/prompts.ts` sends as a dispatch's `task` is byte-identical to before, verified by comparing the old hand-rolled substitution against the new loader-backed one for both files. `{{token}}` substitution still has no home in the loader itself, the same division `identity.self-awareness`'s `{TOKEN}` placeholders already use in `compiler.ts`: the loader hands back a raw body and the one caller that needs live per-dispatch values fills them in. Grepped `src/` for other hand-rolled prompt-text `readFileSync` calls; the only other `{{token}}` users found (`agents/fleet-contract.ts`, `dispatch/code-step.ts`, the fleet `.md` templates) already have their own strict, validated template contracts and are a different system. No builtin worker system prompt changes size: the wiki fragments were never part of `compileWorker`'s output, only ever a dispatch's dynamic task text.
80
- - Gave `compileWorker` the `additionalFragments` channel `compile()` already had, so active project rules and the operator profile reach a dispatched worker (#96). Both are built once per compile in `prompts/extension.ts` and were injected for the main session only; a coder worker editing a file under an active `.clio-coder/rules/**` path never saw the rule that governed it, the layer inversion the user-editable configuration is supposed to prevent. Project rules are scoped to the worker's inferred working context, so a worker whose task never touches a ruled path does not carry that rule's text. That context is `writeRoots`, when the caller sets them, plus path-like tokens recalled from the task and briefing text, since the model-facing `dispatch` tool has no structured path field; a missed path token means a rule can go unseen, never fabricated, because `selectActiveRules` still requires a real glob match. The operator profile renders unconditionally, the same as it does for the session, because it governs how the worker should do the task (validation preference, commit-message style, local-only paths), not only how the orchestrator talks to the operator. The channel is `additionalFragments`, rendered last after persona: nothing splices into an existing worker section, the pattern a7faa133 removed. Measured with a matching rule and a profile both firing: +289 bytes per builtin (+3,468 across the twelve, 66,472 to 69,940); the operator profile alone is +178 bytes per builtin; a worker whose task touches nothing ruled pays 0.
81
- - Made `/tree` switch and `/fork` of the same turn reconstruct the same state, and fixed the task board following the abandoned branch after a switch (#94). Live replay after a `/tree` switch kept every unanchored sidecar (`taskLedger`, routing notices, a leafless `workerRun`) regardless of where it sat in the file, while `/fork` dropped anything written after the fork point; `filterEntriesToActivePath` (`src/domains/session/tree/active-path.ts`) and `fork.ts`'s `sessionEntryBelongsToPath` now share one `entryBelongsToPath` verdict, so both surfaces exclude a sidecar written after the leaf's position. The task board's own fold (`task-board.ts:181-187`) was a second, independent copy of the same bug: it took the last `taskLedger` entry in raw file order with no branch filter at all, so a plain `/resume` on a branched session could show whichever branch happened to write last, not just a `/tree` switch; `readEntries` now runs through the same active-path filter (`src/entry/orchestrator.ts`). The board's session-id-keyed cache (`task-board.ts:327-336`) also never noticed a `/tree` switch, since that only moves the append point inside the same session; a new `SessionTurnSwitched` bus event and a `TaskBoardStore.invalidate()` method fix that. Also: the `/tree` pin is now persisted to `meta.pinnedLeafTurnId` and cleared on the next append, so a switch made without a follow-up message survives quit and resume instead of reverting to the abandoned tip via timestamp inference; `/tree` marks the active tip with its own glyph; and Enter on a compaction or branch row is inert with a status message instead of throwing `turn not found`.
82
- - Replaced the doc routing table in the system prompt with a directive to call `context(scope="docs")`. `identity.self-awareness` carried a 42-row "topic -> docs/file.md" table, 2,410 bytes of every session prompt, guarded by its own first line ("read these only when the user asks about Clio herself; never for ordinary coding work") and growing by one row per doc. The docs engine already indexes every bundled `docs/*.md` and every search response lists the whole corpus, so the table was resident knowledge the model could fetch in one call; measured on the current index, 39 of the 42 rows rank their document in the top five from the bare row label and all 42 from a natural question. The new `identity.docs-routing` fragment tells the model to call `context(scope="docs", query=<the question>)` before answering and before any grep, find, or read when asked about Clio herself, phrased as an instruction rather than a note that docs exist, because a note is what once sent the model grepping the workspace for a skill; it renders only when `context` is on the surface, the same gate the Skills passage uses. The paths, the code-outranks-docs rule, and the configuration locations stay unconditional. Session prompt drops from 15,724 to 13,314 bytes; workers are unchanged.
83
- - Added a `skills-pin` check to `scripts/check-hygiene.ts`, run by `npm run lint`. `skills/registry.yaml` drifting from an edited skill's content was only caught by `npm run skills:check`, which runs inside the full `ci` chain and nowhere else; a5d50940 landed a stale `clio-test` pin from `be2b5ccb` that sat unnoticed for two days because nobody's local lint run ever exercised it. The new check shells out to `pin-skills.ts --check`, the same command `skills:check` runs, so lint and the full gate share one definition of "stale" instead of two. `npm run lint` wall time moves from 4.4s to 5.1s.
84
- - Made the worker read the same safety text the session reads, and corrected two things that text said. The four `safety.<level>` fragments were rendered for the session only; a dispatched worker got a hand-written copy of the four levels from a `switch` in `compiler.ts`, and the copy drifted: its full-auto branch said "Writes, dispatches, and ordinary commands run" for workers none of which admit dispatch, and its auto-edit branch never said which commands are recognized or that `$(...)` is always approval-required, so a worker learned those only from its first denied call. The `switch` is gone; both renderers read the one fragment body, and the fragments now speak in the safety net's action classes and never name a tool, so nothing in them can be false for a surface that lacks one. Two session-side statements were wrong against the engine and are fixed in passing. `safety/auto-edit.md` said "Commands with pipes, `&&`, or redirects count as unrecognized and ask", while `policy-engine.ts` recognizes and runs a `&&` chain whose every step is recognized; it now says "a `&&` chain whose every step is recognized runs too" and lists pipes, `;`, `||`, redirects, newlines, and a chain with an unrecognized step as the unrecognized forms. `safety/read-only.md` said only "The safety net applies at every autonomy level" where the other three levels name `git_destructive`; it now carries the same sentence. Workers gain the auto-edit detail (about 500 bytes per builtin) and no authority: the worker's engine loads the same builtin allowlist and the same `.clio-coder/safety.yaml`, and already denies the same calls.
85
- - Removed `test:isolated` from the package scripts. It ran the same contracts and smoke globs `test:coverage` runs, minus the coverage flags, through the same per-file `test:file` invocation, so `npm run test:file -- 'tests/contracts/**/*.test.ts' 'tests/smoke/**/*.test.ts'` reproduces it exactly. `test:repeat` stays: it is the one lane that reruns smoke shuffled to catch order- and timing-dependent flakes a single deterministic pass cannot, and `.github/workflows/ci.yml` runs it on every push. The CI workflow already runs one release gate on both node lanes with no `if: matrix.node-version` guard hiding a check from either one; the single remaining guard skips the coverage summary step on node 24 because coverage is a report about the suite rather than a check of it, and paying node's per-file instrumentation twice would only post the same numbers to the job summary twice.
86
- - Said each routing rule once. The session prompt told the model seven times, in five places, that broad repository exploration goes to a worker, and the phrasings disagreed: the operating contract and two Tool Contract lines said use `agent:"auto"`, while `FLEET_ROUTING_GUIDANCE` said auto "is a fallback, not a router" and the Fleet roster said broad recon goes to `scout`. Skills guidance was stated in five places. The Tool Contract now keeps only what nothing else says (complete surface, harness model, inventory rule, narrow-orientation tools, validate before claims, retry-shape recovery, the fleet pin rule); the Retrieval Hints keep only what is and is not preloaded; and the single surviving broad-exploration rule, in `operating.delegation`, is the pinned one: hand it to a worker before repo-wide reads, pinning `scout` when the roster lists it. The session prompt loses 1,360 bytes and no capability: every deleted sentence has one surviving statement in a passage that renders exactly when its tool does.
87
- - Split the operating contract into a constitutional layer and role layers. One file, `operating/contract.md`, was carrying the coordinator's delegation and receipt discipline, the `/share` note, the skills passage, and the shared posture, and the worker compiler was regex-splicing it into a second document; every one of the 12 builtin workers was told to use `dispatch` and `agent:"auto"` although no builtin admits `dispatch` (#91). `operating.contract` now holds only what is true for every reader; `operating.delegation` renders for the session only when `dispatch` is on its tool surface and `operating.skills` only when `context` is, the same rule the Fleet block already followed; `operating.worker` holds the assigned-task contract that used to be inline TypeScript with no fragment provenance. The worker operating contract shrinks from 3,638 to 1,575 bytes for every builtin, the twelve worker prompts from 84,952 to 60,196 bytes total, and no session sentence was removed: the one-shot approval semantics moved from the contract paragraph into the session's safety section, where the level fragments they qualify are.
88
- - Gave an installed Clio a skills marketplace she can see. Asked how a catalog skill worked, an installed Clio grepped the operator's own project for the skill's name, found nothing, and gave up; asked to list the marketplace, she returned the operator's personal skills. Three defects produced that. `discoverMarketplaceSkills` resolved its index to `<configDir>/skill-marketplace.json`, which nothing ever writes, and never consulted `resolvePackageRoot`, so the `skill-marketplace.json` the tarball ships was read by nothing and an install had no marketplace at all; both the index and the package's own catalog now fall back to the package root, so bare-name search and install work offline against local files. `context(scope="skills")` read installed roots only and never consulted the marketplace, so the one skills surface the model has could not show an installable skill however well configured the marketplace was; it now renders a marketplace section, and loading an uninstalled name explains how to install it rather than reporting an unknown skill. Every prompt string said "installed skills" and the retrieval hint routed questions about where skills live to grep, which is exactly what the model did; the prompt text now names installable skills and routes those questions to `context`. The package ships all 31 skills rather than 7, plus `skills/registry.yaml` so provenance pinning resolves for an installed copy, costing 0.2MB unpacked and 0.08MB on the wire. A new fresh-install suite exercises the path from a foreign working directory with the catalog environment variables blanked, which is the configuration every existing skills test missed and the reason this shipped.
89
- - Kept the marketplace out of workers and made the skills reminder fire on a fresh install. An unbound worker holding the context tool began seeing marketplace rows it could do nothing with, alongside prose telling it to suggest the skill to an operator it has no channel to address; marketplace rows are now suppressed for every worker registry and the skills passage is stripped for all workers, leaving a worker the skills it was given and nothing else. Bound workers were never affected and the capability boundary is unchanged. Separately, the once-per-session skills reminder is the one channel that reliably makes a model consider a skill, and it was gated on the installed count being above zero, so the operator most likely to benefit from learning that skills exist was the one operator never told; it now counts installed plus installable and says both.
90
- - Moved the static-analysis checks out of the test runner. Nine files under `tests/` read source, docs, README, config, and script text off disk and asserted on their structure while executing no product code; `export-hygiene` alone cost 57.9 seconds, the second most expensive file in the suite. `scripts/check-hygiene.ts` now runs all ten rules in one process against one read of the tree, wired into `npm run lint`, and every assertion moved with them. `npm test` drops from 1,760 CPU-seconds to 1,465 and from 160 seconds wall to 124. The boundaries test lane is gone; `check-boundaries.ts` remains as a library the lint imports.
91
- - Restored the streaming display coalescer. The raw engine wrapper that precedes every derived text and thinking delta was classified as a synchronous event, so each provider chunk cancelled the pending 16 ms display frame and forced an immediate render request; the coalescer was defeated exactly while a response streamed, which is the one time it exists for. Wrappers carrying text or thinking deltas, which the panel has always ignored, now stop at the renderer instead of reaching it; tool-call formation keeps its synchronous path, and every event the panel acts on arrives in the same order as before.
92
- - Enabled Node's V8 compile cache for the boot-path module graphs: interactive chat, `run`, `acp`, and native fleet workers. Locally spawned workers inherit the directory through their spawn environment and consume it before any of their children can see it; SSH-placed workers enable the remote install's own cache in-process at the worker entry, with nothing exported to their children. Read-only and dry-run commands never enable it, `paths` and bare `doctor` keep their promise that nothing is created, and the cache enables only on an initialized install so a home Clio never set up stays untouched. `NODE_COMPILE_CACHE` and `NODE_DISABLE_COMPILE_CACHE` always win, and a cache failure never affects a command. The import-only and real-PTY fill/hit observations, with their deliberately narrower endpoint names, are recorded in `docs/performance-methodology.md` rather than generalized into a per-command saving.
93
- - Corrected boot and TUI performance instrumentation before using it for 0.3.2 decisions. `first TUI paint` is now marked only after the first real frame has issued all of its stdout writes; explicit frame ids group diff, ANSI, hardware-cursor, and IME writes; canonical event and input ids correlate queue, panel, and committed-frame high-water marks; and stdout return, backpressure, and drain are recorded. The trace writer is bounded, asynchronous, awaited at shutdown, and nonfatal on storage failure. A deterministic contract plus a real built-CLI PTY harness cover first frame, input, resize, grouped writes, paused output, and process reaping. The documented metrics are stdout/PTY endpoints, never an in-process claim of literal glass latency.
94
- - Added the adaptive stream pacer as `terminal.smoothStreaming: off | auto | on`, with the conservative 0.3.2 default left at `off`. Only derived visible text and thinking are paced, through one ordered generation queue; raw text/thinking wrappers remain transparent and public events, transcript persistence, tool formation, cumulative tool state, and ordered boundaries remain synchronous. Pacing is grapheme-safe, arrival-credit and oldest-age bounded, self-stopping, folded-thinking aware, and forced to settle before abort, retry, submit, mode change, final return, or teardown. Fullscreen frozen scrolling and resize survive paced updates, and stdout backpressure gates frame construction rather than creating a second unbounded SSH buffer. `auto` bypasses non-TTY, remote/multiplexed, CI, accessibility-marked, and backpressured sessions; `CLIO_CODER_SMOOTH_STREAM=off` is the immediate per-process escape hatch. Deterministic fake-clock, ordering, grapheme, reset, folded-thinking, scroll/resize, no-drain, and built-CLI PTY contracts cover the rollout.
95
- - Added a measured instant interactive shell through one `TerminalLease`, enabled by default with `CLIO_CODER_INSTANT_SHELL=0` as the immediate rollback. Stage 0 and Stage 1 share the exact terminal, TUI, root host, editor, raw mode, decoder, resize subscription, terminal queries, signal router, and stop lifecycle; hydration swaps roots and delegates atomically instead of adopting reconstructed state. Typed drafts and cursor state survive, early submissions are visible immutable FIFO admissions dispatched exactly once after the application attaches, and diagnostics are serialized into the hydrated TUI or emitted after terminal restoration. Boot failure, Ctrl+C before or after attachment, SIGTERM, and a stale hydration all converge on exactly-once terminal restoration and recover accepted input. ACP, headless, ordinary non-TTY invocation, help, and subcommands never mount Stage 0; the explicit force-interactive non-TTY override is preserved. Built-graph contracts keep its closure at five chunks/90,654 bytes and out of provider/tool/codewiki graphs; real PTY acceptance covers early input, multiple submits, resize during hydration, protocol initialization once, failure recovery, signals, and raw-mode restoration. Corrected Stage 0 commit, PTY receipt, and Stage 1 hydration observations are reported separately in `docs/performance-methodology.md`.
96
- - Removed the eager userland Undici graph from `web_fetch` and use the Node 22.19+ built-in Fetch, Headers, Request, Response, stream, and abort implementation. Localhost contracts preserve request headers/body, redirects, UTF-8 streaming and cancellation at the byte ceiling, external abort, timeout, HTTP previews, binary rejection, and transport errors on both supported Node lines; an installed-tarball turn invokes the real tool from a foreign working directory. The built graph contains no Undici source or dependency, while the measured size and deliberately mixed import-time observations are recorded without turning host noise into a boot claim.
97
- - Made codewiki's tree-sitter graph genuinely lazy and moved runtime indexing off the interactive event loop. Lightweight schema, artifact, and path modules preserve synchronous cached reads without evaluating the builder; actual full, stale, and incremental builds run in a dedicated worker and load only the required grammars. Session startup, parallel `code_nav` demand, mutation batches, explicit index/refresh, bootstrap, wiki grounding, and reset now share one per-workspace generation queue and cross-process lease, so an older build cannot overwrite newer state or resurrect a reset artifact, and shutdown drains admitted work. Built-source and installed-tarball coverage prove nested help is tree-sitter-free and write-free while a real foreign-cwd build loads the runtime and its vendored grammar; measurements and deliberately scoped import observations are in `docs/performance-methodology.md`.
98
- - Split `context`, `code_nav`, `verify`, `web_fetch`, `dispatch`, `monitor`, and `steer` into immutable lightweight tool surfaces and first-use implementation chunks. Registration order, provider schemas and descriptions, policy metadata, execution modes, argument normalization, admission, permissions, middleware, result shaping, and worker surface attestation remain registry-owned and eager; only an admitted runner imports code. Dispatch keeps its trusted plan and capacity-reservation identities in one synchronous admission controller shared with the lazy runner, deeply freezes every execution-affecting control before middleware or approval can observe it, binds the `apply_winner` repository destination into the approval hash and text, and releases a guard-blocked prepared admission exactly once. Workers import only the core tool bootstrap, so their built entry never evaluates the three orchestrator-only runners. Concurrent first calls share one import, an implementation whose surface drifted fails closed, unrelated tools remain absent, and a missing Clio-owned chunk carries the same named reinstall guidance as a missing command chunk. V8 coverage repeats the absent-before-use/present-on-invocation proof for every tool and the worker exclusion against both the source build and an installed tarball from a foreign working directory.
99
- - Removed the eager `pi-ai/compat` and all-provider aggregates from ordinary engine startup. Clio's engine boundary now owns the ordered API dispatcher and its pinned synchronous environment-key lookup while composing only the nine Pi provider factories backed by current Clio runtimes; Pi still owns their public lazy APIs, catalog data, authentication and headers, Bedrock loader, OAuth flows, faux provider, and image registrar. A configured out-of-tree runtime activates a one-time compatibility bridge immediately before plugin evaluation, preserving the existing process-global registry identity and last-writer-wins overrides; a failed bridge prevents that plugin from evaluating. Without a plugin the aggregate, unconfigured provider catalogs, legacy aliases, unrelated provider bodies, OAuth flows, and image generator remain unevaluated. Source and installed foreign-cwd coverage prove both sides of the graph boundary, provider contracts preserve payload and cancellation behavior, every credential mapping is parity-checked against Pi 0.84, and an installed plugin overrides a known API through the shared external Pi instance.
61
+ ### Added
62
+ - Evidence-aware Git commit attribution, with an Advanced setting to disable it without changing commit messages.
63
+ - Fullscreen terminal mode, terminal-native Mermaid and LaTex rendering, smooth-streaming controls, instant-shell startup, prompt-history navigation, and improved model, settings, resume, task, decision, and workspace-output views.
64
+ - HTML transcript export (with Markdown export retained), live tool-progress and numbered edit/write diffs, and richer tool lifecycle details.
65
+ - Directory-scoped `CLIO-CODER.override.md` instructions, project-rule propagation to workers, an installed-skills marketplace, and source/codewiki assets in published packages.
66
+ - Compile-cache support for interactive, run, ACP, and worker boot paths; codewiki indexing and several tool implementations now load on demand.
67
+ - A Pi SDK boundary/upgrade checklist and declaration-surface checks; Pi SDK libraries are updated to 0.84.0.
68
+
69
+ ### Changed
70
+ - Consolidated slash-command spelling and prompt-template argument handling; retired aliases now fail closed while preserving the editor draft.
71
+ - Replaced the LM Studio SDK path with an HTTP adapter. `lmstudio` is canonical; `lmstudio-native` remains a compatibility alias for persisted settings.
72
+ - Improved local-model selection, residency, context sizing, reasoning controls, retries, OpenAI-compatible streaming, and token accounting.
73
+ - Strengthened ACP v1 session, workspace, permission, output-bound, and error contracts for external clients.
74
+ - Improved terminal rendering, transcript streaming, timing, export, and session replay behavior.
75
+ - `docs/html` is no longer included in npm packages; Markdown guides remain available.
76
+
77
+ ### Fixed
78
+ - Preserved prompt submission order during instant-shell startup and restored transcript rendering in fullscreen mode.
79
+ - Kept rejected slash-command drafts editable and aligned footer, receipt, and ledger usage for completed or cancelled turns.
80
+ - Prevented session switches during streaming from creating phantom entries; fixed task-board branch selection and persisted `/tree` pins.
81
+ - Prevented concurrent dispatch processes from overwriting run-ledger rows (#118), and hardened worker compile-cache isolation (#148).
82
+ - Fixed LM Studio duplicate-load behavior (#113), llama.cpp residency failures (#127, #134), runtime alias handling (#119), and probed context-window precedence (#129).
83
+ - Restored documented headless JSON/event output and CLI exit-code behavior (#122, #123); corrected thinking-level resolution (#128), stalled-stream reporting (#131), and reasoning-token estimates (#132).
84
+ - Isolated tests from user configuration (#110, #111), and corrected documentation claims (#117).
85
+
86
+ ### Security
87
+ - Hardened worker and session safety-policy consistency, OAuth cancellation, workspace output reads, and ACP permission/cancellation handling.
88
+ - Added publish-time version-coherence checks (#124) and removed unsafe or misleading default behaviors in credentials migration and package serving.
100
89
 
101
90
  ## 0.3.1 - 2026-08-16
102
91
 
103
- - Made a dispatched worker visible while it runs. `/run` and `/delegate` stream into the transcript as a worker block: an attributed header naming the agent, its route or protocol, and its run id; the worker's prose down a rail; a coalesced tool line; and a receipt footer whose failure reason wraps as prose above it. Dispatch lifecycle events fold into one entry per assignment, so a failover reads as one run with an ↻ line per earlier attempt, and a late event from a superseded attempt moves nothing. A model-started run lands folded under the tool call that spawned it, so a fan-out of three reads top to bottom in spawn order, folded cards stack without blank rows, and `Alt+O` folds or unfolds the newest tool call or worker block. Every board row, footer chip, and transcript block leads with who asked for the run: hollow for the operator, filled for the model, a quiet dot for Clio's own. The session records a `workerRun` entry where a block opens, so a `/resume` redraws the block from that entry plus the sealed receipt, bounded exactly like a live one, and a run whose receipt is gone settles as `receipt unavailable`. `--share` on `/run` and `/delegate`, and `/share [runId]` afterwards, hand a worker's bounded receipt text to the main agent as operator text under a `[worker result] <agent> · run <id> · <outcome> · shared by the operator` header, which the operating contract now names so the model treats it as steering rather than an unattributed tool result; bare `/share` picks the newest finished run the operator started and skips agent-origin runs the model already holds. A structured worker answer renders as prose on the rail instead of one line-wrapped JSON string, a mutation report reading as its changed paths, validations with verdict glyphs, summary, and commit line, while the model still receives the raw payload. A replayed operator turn after `/fork` or `/resume` shows what the operator typed rather than the composed prompt with its system-reminder scaffolding. The transcript ledger prints a written file's byte count rather than the length of its confirmation sentence.
104
- - Added the interop discovery domain. `src/domains/interop` owns knowing which other coding agents are installed: a registry with one pure-data entry per known agent, bounded detection that resolves binaries without a shell and reports `unknown` when a probe cannot answer, and a decision record in `state/interop.json` that only an operator decision writes. `clio-coder configure --interop` and the first-run wizard turn a detection report into consent proposals, show the exact `delegation.agents` YAML before asking, and append an accepted peer through the serialized settings writer; a decline is remembered by fingerprint and stays silent until the binary's version or path moves, and one review that answers n, n, y keeps all three decisions. `clio-coder doctor` gains a row per detected agent, a warning for a configured peer whose command no longer resolves, and one aggregate row for foreign skills, and still writes nothing. The `/interop` overlay lists agents as Detected, Configured, and Declined with one key each to connect or decline, the boot line names installed peers that are unconfigured and undecided, and the `delegation.agents` row in `/settings` gains add and remove actions. Foreign agent directories become a `noWritePaths` policy class from the registry, so Clio never writes into another agent's directory, and foreign prompt roots such as `~/.claude/commands`, `~/.codex/prompts`, and `~/.config/opencode/command` load into `/prompts` trusted at user scope while their project counterparts stay behind `skills.trustProjectCompatRoots`. Detection never spawns a foreign work command, never reads a foreign session store, and never wires a peer without an answer.
105
- - Fixed every prompt template being unreachable from the TUI. `/demo` and any other `/name` that a loaded template owns now submits through the expansion path instead of failing as `is not a command`; an untrusted project template refuses with a named reason, and headless prints that refusal and exits 1.
106
- - Implemented the documented `clio-coder acp --cwd PATH` and `--permission-timeout MS`. Both had appeared in `docs/acp.md` since v0.3.0 and were rejected as unknown options. `--cwd` changes into the workspace before boot and `--permission-timeout` overrides `delegation.defaults.permissionTimeoutMs` for that server only; a rejected invocation writes usage to stderr and leaves the protocol channel untouched. ACP `_meta` keys now live under the `clio-coder/` namespace rather than `clio.coder/`.
107
- - Made the agent ledger reachable and made the main model read it. No builtin recipe declared the ledger tool, so a fanned-out worker was never offered it, and three seams between the dispatch call site, the local spawner, and the fleet transports each dropped the control-lane callback that carries a post; a live three-scout fan-out sealed a board with zero entries. Every builtin that can be fanned out now declares the ledger, every spawn path forwards the full option set, a post that lands before the run has attribution is held in order and appended once it does, a backgrounded parallel batch keeps its board open until collect, and a detached batch closes its board on the collection that settles it. A settled parallel dispatch, a compete result, and monitor collect for a detached batch now carry one `agent ledger (<n> entries, sequence <w>)` section after the per-run lines with the same bounded render, attribution, and corroboration and dispute marks a worker sees, reserved out of the output ceiling so truncation never drops it; a board nobody posted to is omitted, so a single-run dispatch is byte-identical to before. Corroboration keys citations in workspace-relative form, so one file cited by relative and absolute path reads as one corroborated finding. A new composition suite drives the real recipes, bundle, and dispatch and monitor tools with only the worker process faked.
108
- - Refused an unrunnable review fan-out at admission rather than after approval. A review gate with more than one task, a review under a non-parallel mode, a compete with review or several tasks, an unparseable request set, or an unknown mode never parks for approval and returns the executor's own shape error, so the operator no longer approves a plan the executor rejects in milliseconds and the model is told how to gate a fan-out: run it without review, then dispatch one integration task with review.
109
- - Answered an identical worker escalation with the operator's earlier decision. The escalate posture keeps a per-run memory keyed by tool plus canonical arguments; a parked call already decided in this run is released with the remembered grant or cancelled with a reason naming the earlier request, both audited as `operator:remembered`, ending the loop where each approved card only bought the next identical card. A different command is still a new decision.
110
- - Fixed a coder run losing its entire final answer on the synthesis-locked repair round. The lock only set `tool_choice: none`, llama.cpp honoured that by disabling its tool-call parser while the template still rendered every tool schema, and the model's tool markup came back as content the loop guard removed. A locked worker round on OpenAI-family APIs now removes tools from the request, so plain text is the only move left; Anthropic keeps `tool_choice: none`. When the loop guard does remove a reply, the repair rounds, the receipt, and the fallback notice now say that the reply held only tool-call markup and was removed, instead of `result must be valid JSON`.
111
- - Made the mutation-contract refusal show the value that would have passed. An empty `validations` array is told it was empty and given one shaped entry; wrong keys are told which keys an entry carries; the coder recipe states the same requirement where the model reads its result guidance.
112
- - Fixed a `/run --model` override stranding a llama.cpp router with one instance slot. A model that will come back tag-pinned may only take an unprotected slot; when the load would fit only by evicting the configured worker model it declines up front with a will-not-fit notice naming that resident and its role, so the swap that could not be undone is never made.
113
- - Corrected tool timing across the transcript, receipts, and `toolStats`. The registry reports how long a call sat parked on a permission card and the tool adapter subtracts it, and the transcript prefers the registry's own span over the engine frame, so an operator's approval wait is no longer sealed as tool execution time. The park closes at the operator's decision rather than at the verdict, so an approved call's own execution is no longer subtracted with the wait; a review dispatch that fanned out for 224 seconds had settled as a 15ms line (#82).
114
- - Read every model JSON payload through one shared reader. `parseJsonObjectPayload` in `src/core/json-payload.ts` walks the whole message, the first fenced block, and the outermost brace span, and tries every candidate span in turn, so a fenced code report, judge verdict, verifier report, terminal result contract, or context handbook whose body nests its own fence is no longer refused for its wrapping. Conformance did not widen: a candidate still has to parse to an object and every field check stays where it was.
115
- - Fixed `clio-coder context init` losing complete handbooks. Citations carrying a line anchor (`src/cart.ts:40-42`) or written as a call (`priceCart()`) are grounded by the path or name underneath, so a handbook whose Gotchas and Architecture sections cited real files no longer falls under the grounding floor; the local import convention is measured from the relative specifiers the indexer extracted, reported only on a clear majority, instead of asserting `.js` because a `tsconfig.json` exists; and the fallback notice names `context-bootstrap` and says the handbook was not produced instead of `Scout unavailable`.
116
- - Named reachability as reachability. An LM Studio turn that failed with `connect ENETUNREACH` no longer advises resizing a model that was never asked to load; `describeLoadFailure` recognises connect-level errnos and says the server did not answer.
117
- - Fixed the benchmark adapters. HumanEval manifests count generation errors and failed tasks separately, keep the model and target profile through grading, and record where each completion came from, and `regrade --out <run-dir>` rescores a finished 164-task run from its stored attempts in seconds without calling a model. SciCode steps are graded with the dependency header they were promised and the return line upstream supplies, the prompt reaches the model as prose rather than an indented code block, and `snapshot-step` lets a step solved by hand or through the TUI reach the grader. The Terminal-Bench container mirrors the operator's own fleet per node, runtime and thinking level included, probes the worker endpoint in preflight, writes one manifest per episode, and its installer installs `curl` when neither downloader is present, prints the output that produced a failure, and returns rather than exits when sourced so the harness's tmux signal still fires.
118
- - Renamed the remaining surfaces the 0.3.0 rename missed. The process title is `clio-coder` and a worker reads `clio-coder-worker:<agent>` in `ps` and `htop`; the fleet node key is `clioCoderEntry` and the eval flag is `--clio-coder-entry`, with both old spellings still read; `CLIO_CODER_MAX_DISPATCH_RUNS` is the canonical guardrail spelling and `CLIO_CODER_MAX_RUNS` still reads. Removed the `identity` setting that shipped in every generated `settings.yaml` and that nothing read; an existing file with the key still loads. Declared the package as a CLI with an `exports` map naming only `./package.json`, so the 778 shipped source files are no longer an unversioned deep-import surface. Corrected the runtime plugin path to `~/.config/clio-coder/runtimes/`, the tool count to twenty in every place it was stated, and the glossary. Replaced one operator's node names, model ids, and addresses in source comments, docs, scripts, soak configurations, and benchmark adapters with neutral placeholders and `CLIO_CODER_*` environment variables, and attributed the identity fragment to the Gnosis Research Center at Illinois Tech and the IOWarp team.
119
- - Reworked the settings and configuration surfaces of the TUI. `/settings` is a transactional Settings Center grouped as CORE, ROUTING, RUNTIME, and EXPERIENCE; every edit builds a change plan and asks Apply this session, Apply and save globally, or Cancel, a global save writes only the leaves the mutation touched, and `/output` and `/thinking` apply to the session rather than silently rewriting the file. Fleet profiles, agent bindings, and targets are editable rows: Settings → Fleet is an entity workbench with add, remove, node-pin, and unbind flows, and Settings → Targets is a console with use, probe, connect, and remove per row. `/targets`, `/fleet`, `/scoped-models`, bare `/thinking`, and bare `/output` deep-link into the matching section, `/settings <section>` reaches every other one, and the bespoke overlays those commands opened are gone; running and retrying dispatches stay on the `Alt+W` Fleet Runs board. Below 72 columns the overlay drills down one page at a time under a breadcrumb, Esc moves up exactly one level at every width, and `/` filters rows by label, path, and description. A scoped-model reference whose target is unconfigured is kept and disclosed rather than silently dropped at load.
120
- - Recast the rest of the TUI around the same doctrine: an adaptive welcome launchpad, a composer that reads unmistakably as the place to type, a voice-first transcript with a two-cell hanging indent, slash commands presented as a grouped palette, a quiet two-zone footer that discloses status progressively and degrades notices without losing their dismiss key, one orange activity signal per surface, and a decision prompt that wears the orange frame and names its accept key. An `ask_user` prompt is now the size of the question it asks: a two-option confirmation is a bottom-anchored bar rather than a full-height modal, and only a multi-question round or a later interview round takes the full surface. `/memory` reaches every entry through a filterable master-detail overlay, a narrow list footer keeps the key that opens its detail pane, a skill suggestion line no longer steals the reply glyph from the answer, and the footer says nothing about reasoning tokens when a turn spent none.
121
- - Rebuilt the trace viewer on the docs blueprint design system: brand assets and a version chip on the masthead, semantic palette tokens that reserve action orange for running state, interactive waterfall lanes with tool tick markers, a run search filter that survives live polling, truthful `$0.00` spend, and contract tests for tool durations, payload truncation honesty, and process status badges.
122
- - Made time correct across the runtime. Durations are measured on `performance.now()` and never on two wall-clock reads, so an NTP step or a laptop resume cannot publish a time-to-first-token that never happened or abort a healthy stream through the stall watchdog; a backwards clock is counted rather than clamped into silence. Every operator-facing timestamp renders through one `format-time.ts` module in the operator's own timezone, so `/tree`, `/fork`, `/resume`, `/memory` step rows, the banner's handoff freshness, and the `/export` filename no longer read a UTC instant as local and land a whole calendar day off for anyone far from UTC. Lockfiles and capacity leases record host, pid, and birth token, so a sibling host sharing a state directory never steals a live holder's lock or slot and a live holder keeps its lock however long it takes; the four lock implementations that disagreed on staleness by a factor of 26 are one. `runs.json` eviction spends its cap on finished rows and keeps live ones, eval ids can no longer collide within a millisecond, and a run whose owner died mid-dispatch is finalized on store open. The test harness gained a frozen, steppable clock, and the panels that raced the wall clock read it.
123
- - Made a stalled stream abort and retry instead of waiting forever. `retry.streamStallMs` (default 180000, 0 disables) arms a deadline at `agent_start` that every engine event pushes out; when it passes with nothing from the stream the run is aborted through the same path Esc uses and walks the existing retry ladder, so a headless run, a fleet worker, or an ACP client no longer needs a human at the keyboard. Silence during tool execution does not count. The await-watchdog counts a worker's own stream as progress, so a dispatch that produced thousands of events is no longer reported as `no progress`.
124
- - Fixed the result-contract repair round. The directive now rides as a tool result paired with a synthetic assistant call, which every served template treats as a history append instead of re-rendering the whole prompt (11,352 tokens reprocessed as a user turn against 407 as a tool result on a warmed 15k prefix), which OpenAI's and Anthropic's validators accept, and whose zero usage the context estimator skips; before that last change every dispatched worker died one round after its terminal message with `Cannot read properties of undefined (reading 'totalTokens')`. LM Studio no longer reports `cacheRead: 0` for a stat it does not have.
125
- - Made LM Studio residency fit the context and keep one instance. A just-in-time load beside another resident model is capped at 131,072 tokens and the clamp is reported before the first turn, raised with `CLIO_CODER_LMSTUDIO_CORESIDENT_CONTEXT`; a model that is already resident is reused rather than loaded a second time; configured models carry the plane they serve so a forced eviction names its role; and a turn under 2 tok/s for 30 seconds emits a `degraded` notice listing the resident set. The planner budgets against the window the backend has actually loaded rather than a catalog maximum, so the 80% compaction trigger no longer sits above the limit a backend fails at.
126
- - Added the outward exposure tier to `ask_user`. A skill marks a gate whose confirmation leaves the machine as `exposure: outward`, and that gate parks for the operator at both supervised levels instead of being auto-answered, so filing a public issue can no longer go through in 24ms without a human; `full-auto` is untouched and an unrecognised tier reads as outward. The read-only exploration nudge advises once per turn instead of forcing a second model round, and the footer tools row leads with the registered tool count so a session whose only call was a dispatch no longer reads `tools none`. A turn that claims worker results without a dispatch call to back them draws an honesty notice.
127
- - Moved generated artifacts out of the working tree. The `artifact` tool's defaults resolve to `.clio-coder/artifacts/PLAN.md`, `REVIEW.md`, and `REPORT.md`, which is gitignored, so a report-writing turn no longer drops a root-level file into a repository a second agent is working in; an explicit `path` still writes exactly where it says. `docs/artifact-placement.md` states the contract.
128
- - Made the package carry Clio's own source and code map. `files` gains `src/**` and the build ends by indexing exactly the packed file set into `dist/assets/codewiki.json`, so an installed Clio can read her own source and answer questions about herself; the twelve tree-sitter grammars are vendored into `dist/assets/grammars/` in place of two grammar packages costing ~72MB per install, and `chalk`, `diff`, `uuid`, `yaml`, `typebox`, and `undici` are bundled into `dist/` and moved to devDependencies. A pack-install smoke installs the tarball into a scratch prefix and proves `--version` and `context index` from there. The release workflow verifies artifacts and never publishes; `npm publish` is a manual maintainer step, and a contract test guards that no publish path is re-added.
129
- - Collapsed the git skill pipeline to three stages, `file-ticket`, `fix-issue`, and `ship`, retiring `commit-crafting`, `create-pr`, `investigate-issue`, and `review-changes`; the coder and git-master builtins rebind to the survivors and the catalog is re-pinned.
130
- - Fixed a pasted slash command with a trailing newline landing in the composer without running, shadow workers reading as `sh:`-prefixed agent names rather than as internal processes with their own footer rows and receipt ids, and evidence attribution claiming an overlapping sibling run's entry as its own; an entry inside two run windows now lists both candidates and appears in each bundle.
131
- - Made `clio-coder doctor` treat a home Clio has never written to as not set up rather than broken. It prints one `WARN installation not set up yet` row naming `clio-coder`, `clio-coder configure`, and `doctor --fix`, exits 0, and creates nothing, so the `--fix` that follows is stamped as an install rather than a repair; once any root exists the per-row report and exit 1 are unchanged.
132
- - Made an upgrade say so. `install.json` records `upgradedFrom` when the version changes, `clio-coder doctor` shows `upgraded <when> from <version>`, and the first interactive launch on the new version shows one footer notice naming the transition, the commands and skills that moved, and the CHANGELOG section, then records `noticedVersion` so it never repeats; headless and ACP boots stay silent. `clio-coder upgrade --dry-run` names the recorded and current versions it would move between and says plainly when no migrations are registered, and the post-install and local paths report `0.3.0 -> 0.3.1` rather than the binary's own version twice. Nothing about the 0.3.0 to 0.3.1 upgrade needs a hand step: `identity:` still loads and is ignored, `clioEntry` and `CLIO_CODER_MAX_RUNS` still read.
92
+ ### Added
93
+ - Live worker transcript blocks, receipts, sharing, folding, and durable replay for `/run` and `/delegate`.
94
+ - Interoperability discovery and opt-in configuration for compatible coding agents, with protected foreign-agent directories and prompt roots.
95
+ - An agent-ledger surface for coordinated worker findings and a transactional Settings Center for routing, runtime, and experience settings.
96
+ - Stream-stall retries, authoritative timing/timezone handling, packaged source/code maps, improved trace viewing, and clearer upgrade notices.
97
+
98
+ ### Changed
99
+ - Reworked the TUI around adaptive launch, composer, transcript, footer, permission, and narrow-terminal layouts.
100
+ - Improved fleet admission, result contracts, benchmark adapters, local-model residency, artifacts, and release packaging.
101
+ - Renamed remaining user-facing runtime identifiers to `clio-coder`; legacy settings and environment spellings remain readable where noted.
102
+
103
+ ### Fixed
104
+ - Restored TUI prompt-template invocation and implemented documented ACP `--cwd` and `--permission-timeout` options.
105
+ - Preserved synthesis-locked worker answers, prevented unsafe llama.cpp model overrides, and corrected tool durations (#82).
106
+ - Fixed handbook grounding, reachability diagnostics, JSON result parsing, pasted slash commands, and resumed/forked prompt display.
107
+ - Fixed cancelled-turn replay/accounting, credentials corruption safeguards, doctor diagnostics, configuration layout, and missing-artifact errors.
108
+
109
+ ### Security
110
+ - Remembered identical per-run escalation decisions without widening different requests.
111
+ - Added outward-exposure confirmation, safer artifact defaults, protected foreign-agent paths, and stricter credential, URL-opening, and permission behavior.
133
112
 
134
113
  ## 0.3.0 - 2026-08-14
135
114
 
136
- - Renamed every namespace the tool occupies on a user's machine from `clio` to `clio-coder` ahead of the first npm publish, so the future clio-core, clio-agent, and clio-kit tools can coexist with it: the installed binary and every command example are now `clio-coder`, XDG config/state/cache directories use a `clio-coder` segment, the project context directory is `.clio-coder/`, the generated handbook is `CLIO-CODER.md`, environment variables are `CLIO_CODER_*`, and extension manifests are `clio-coder-extension.*`. The product name in prose remains Clio Coder, and code-internal identifiers, event names, and receipt fields are unchanged. There are no migration shims: this ships in the first published release, so there is no installed base to migrate.
137
- - Marked v0.3.0 prominently as experimental in the CLI/TUI startup experience and repository front page: behavior and interfaces may break or change without notice.
138
- - Added the agent ledger, a bounded coordination surface the concurrent workers of one dispatch share. A worker in a multi-run dispatch posts typed entries over its existing control lane: a claim staking path scopes so peers stop duplicating work, a cited finding a peer can corroborate, or a review of another entry by id. Every worker receives the board at spawn framed as untrusted peer data and live deltas over stdin, and reads answer from a local mirror with an explicit staleness watermark rather than blocking on a round trip. The orchestrator is the sole writer and stamps every attribution field from its own admission record, so a worker cannot forge authorship; posts are capped at twenty per run; rendering labels each finding corroborated, uncorroborated, or an ungrounded lead and never merges peer output into a consensus, so the single-worker finding that matters stays visible. Each run's receipt seals its contribution count and digest, and a run with no peers never sees the tool.
139
- - Made compete candidates diverge on purpose instead of racing four near-identical attempts. Each candidate receives one engineering stance from a closed set (minimal diff, test-first, refactor-tolerant, spec-literal) as a one-line dynamic prompt message that leaves the task, recipe, and result contract untouched, and the judge now reports where the candidates disagreed rather than only which one won.
140
- - Fixed active session branches staying active through production compaction, post-compaction replay, and editor `!command` sidecars. Switching to an older `/tree` branch no longer lets the most recently appended abandoned sibling silently replace provider context.
141
- - Made global CLI startup parsing one strict, order-independent pass. Interleaved `--skill`, `--api-key`, and context flags no longer expose a secret as a mistaken subcommand, command recognition and dispatch now share one registry, and misspelled leading options fail closed instead of disappearing.
142
- - Made ACP connect, turn, and permission deadlines mandatory schedulable bounds. Zero can no longer disable a request timer, oversized values can no longer overflow Node into an approximately 1 ms timeout, and CLI, settings, direct adapter, and transport paths share the same defaults and timer ceiling.
143
- - Stopped automatic retries after a failed attempt has executed a potentially state-changing tool call, including a tool that returned an error after partial work. Clio does not yet isolate ordinary retry workspaces, so launching a successor in the same checkout could consume partial edits; read-only, model-startup, and transport failures remain retryable. Removed the disconnected workspace-transaction implementation and direct-only tests that falsely implied this boundary was wired.
144
- - Removed unreachable pre-release scaffolds, obsolete barrels, empty tombstones, and the superseded auth selector, then enabled TypeScript unused-local and unused-parameter checks so dead implementation residue fails the build instead of accumulating.
145
- - Added the soak, a benchmark whose subject is Clio rather than the model. Every other suite measures what a model produced and passes a run whose receipt never sealed, whose seal does not authenticate, or whose stream republished its own transcript. `benchmarks/soak/clio-soak.yaml` inverts that: a weak model that never repairs the fixture passes because the machinery behaved, and a strong model that repairs it fails the moment Clio breaks a promise about itself. Three sibling suites gate what their own surface can answer, for write boundaries, bounded loops, and SIGINT chaos. It runs offline against a local target, because a gate on Clio's own invariants that needs a cloud key is not a gate.
146
- - Made invariant metrics fail closed by construction. `src/domains/eval/metrics/invariants.ts` reduces receipt, session, process, and write-boundary promises from the journal each eval item leaves behind, and every reader is total: a metric it could not compute is absent rather than false, because a threshold on an absent metric fails closed while a fabricated value is indistinguishable from a check that passed. Each metric has a test proving it fails on a corrupted artifact.
147
- - Made eval token accounting observed rather than assumed. Usage is folded out of a runner's live stdout as it arrives, so truncating the operator-facing artifact cannot erase it. A runner that observed no usage reports `tokens.measured: false` and no counts at all, never a zero, and reports say how many runs a total covers. Eval artifacts moved to version 4 with a discriminated `summary.tokens`; the parser refuses counts beside `measured: false`, a `tokens.total` threshold on an unmeasured artifact fails closed, and a comparison against an unmeasured side reports an unmeasured delta.
148
- - Added receipt-derived accounting for surfaces that publish no usage stream. `clio-coder fleet run --json` drains its workers' events, so a bounded loop's cost never reached the stream fold. `receiptUsage.*` reads what the run sealed and authenticated, carries its own provenance, and never merges with or masquerades as stream-observed `tokens.*`. An incomplete or unauthenticated receipt set reports unmeasured and no counts.
149
- - Added per-step write boundaries. Fleet contract v4 declares a `writes` path allowlist per step, `readonly` being the empty one, and the orchestrator verifies the claim after the fact by diffing the checkout against a pinned baseline, rolling back unauthorized changes, and sealing a verdict carrying its own digest and that baseline. This is detect-and-rollback, never sandboxing: nothing prevents a write, and confinement an agent cannot escape needs OS-level isolation this does not provide. A path that cannot be cleanly restored is reported and left for the operator rather than guessed at.
150
- - Added bounded check/repair loops and the shipped SDLC chains. A `loop` declares `maxAttempts`, a check, and an agent repair, and compilation unrolls it into conditional nodes so the plan stays one deterministic hashed DAG with a receipt per attempt. The declared bound is the promise: attempts never exceed it, every attempt after the first is `recovery`, and a spent bound reports `loop_bound_exhausted` rather than a green it did not earn. Verification staleness is scheduler-enforced.
151
- - Added deterministic `code` steps. A step names a command id from the repo-owned registry at `.clio-coder/fleets/commands.yaml`, never a shell string a model authored, and an unknown id or missing registry fails contract validation before anything dispatches. A code step holds no capacity lease and carries no execution role or authority grant.
152
- - Added headless session continuity. `clio-coder run --session <id>` appends to an existing session and `--continue` to the most recent one for the working directory. Continuation is a hard requirement rather than a hint: a session that cannot be resumed exits 2 before any model call, because an answer written without the history the caller asked for is worse than no answer. The session id is discoverable from the surface that ran the turn, and stdout stays the assistant's answer alone.
153
- - Made process-safe dispatch admission durable. Capacity leases are the expiring global and per-node authority, with acquisition, retry rebinding, heartbeat, drain, and reservation transfer serialized by one cross-process lock. The lease bound fails admission closed rather than dropping a lease, a plan slot belongs to an assignment so a retry never queues behind itself, and the operator's machine-wide drain is TTL-bounded so an abandoned drain cannot wedge the host.
154
- - Made context budgeting distinguish a target-reported limit from Clio's planning assumption. A router can answer `/props` with `n_ctx: 0` while a model-specific probe or LM Studio detail row records the selected model's context, including its currently loaded window rather than its larger theoretical maximum; selection is exact-id and target status names the source. A target that declares no window now receives Clio's explicit 131,072-token assumption with a human-visible warning to probe it, and a reported window below 128,000 warns on every runtime tier instead of quietly shrinking a session to the obsolete 8K fallback.
155
- - Fixed `clio-coder run --agent --json` republishing its segment transcript. Every message in an `agent_end` had already crossed the wire as its own `message_end`; on a two-minute run that was 24 KB restating 19 KB, and the ratio grew with the answer. Both `--json` wire projections now live in one module and make the same promise: content crosses exactly once.
156
- - Fixed a suite's declared `thresholds.fail` deciding nothing at run time. Only a later `eval gate` invocation read it, so a run that broke a declared threshold still exited zero. A gate now reaches per-run metrics, so the failure names the run rather than an aggregate, and an assertion whose metric was never measured fails closed in both layers.
157
- - Updated the Pi engine dependencies to 0.83.0. `src/engine/` is now the one place pi types enter the codebase; no file outside it imports `@earendil-works/*`, type-only included.
158
- - Made the default `clio-coder --help` describe the command surface a person needs to read while preserving the complete surface for scripts and agents. Harness-oriented commands are grouped under `clio-coder dev`, `clio-coder --help --all` reveals both sets, and the former top-level command forms continue to resolve unchanged.
159
- - Reorganized the interactive application into focused controller, presentation, input, event-projection, process-shell, transcript, ticker, editor, slash-command, and overlay lifecycle modules. The TUI's public behavior and command surface are unchanged; the decomposition gives lifecycle, input, overlay, and rendering contracts independent test seams instead of concentrating them in a 3,600-line entrypoint.
160
- - Fixed direct shell interpolation while opening provider URLs on macOS and Linux. Interactive OAuth authorization, device-code, and console URLs once entered an `exec()` command after only double-quote escaping, so backticks or command substitutions in provider-supplied text could execute before the browser opened; `open` and `xdg-open` now receive the URL as a distinct `spawn()` argument. Windows still starts its browser through `cmd /c start`, so this change removes the original `exec()` construction there but does not claim a shell-free Windows launch.
161
- - Made retries distinguish a self-hosted model that is loading from an ordinary rate limit. Unloaded or loading-model errors now qualify for retry with a 15-second minimum delay that still honors the configured maximum, so disk and VRAM loading does not spend the usual short backoff sequence before the target can serve a request.
162
- - Corrected damage-control matching so it evaluates commands that will execute rather than prose a file will contain. Writing documentation that quotes `rm -rf /`, a migration that contains `DROP TABLE`, or a classifier fixture no longer trips a command-pattern block; the same destructive command remains subject to policy when passed to a command-bearing tool, and destination paths still receive path-based checks.
163
- - Reworked project-context bootstrap around the dedicated `context-bootstrap` agent and a generated local handbook. `CLIO-CODER.md` is a gitignored runtime artifact rather than this repository's canonical instructions; the agent uses ordinary binding and default route resolution while honoring a legacy Scout binding, an existing handbook informs generation rather than silently cancelling it, and default initialization preserves it until an explicit `--apply` or `--rewrite` action. Bootstrap retains the previous handbook provenance when a run generates nothing and reads project identity from the manifests, build files, citation records, and README forms each ecosystem uses instead of assuming a Node package.
164
- - Made `clio-coder context refresh` update only the existing handbook sections the rebuilt codewiki owns, leaving model- and human-authored prose stable, and made context status name a partial generated wiki as incomplete rather than presenting it as coverage it has not earned. Context reset preserves the wiki by default and its help names that exception, so a destructive action is not hidden behind a generic reset.
165
- - Rebuilt generated wikis around a depth-scaled, index-derived page plan and isolated page writers. A bounded planner may refine the deterministic candidate, but each `wiki-writer` dispatch is confined to staging, receives a bounded source set and no Git tool, checkpoints one page at a time, and promotes the coherent pages it wrote while recording the rest for `clio-coder context wiki --update`. Deterministic assembly regenerates navigation from pages on disk, drops empty pages so they remain owed, and records unresolved links and citations for the next update rather than fabricating complete coverage.
166
- - Hardened the skill substrate from discovery through evaluation. A `SKILL.md` must be bounded valid UTF-8 text and a discovered root must contain the content it advertises; scalar or sequence tool declarations resolve to Clio's actual tool surface and warn on unenforceable names. `clio-coder skills eval` now measures the precedence-winning copy activation would load and reports its origin and hash, while install and update constrain GitHub paths to the cloned repository, validate a staged replacement before an atomic swap, preserve the installed copy on failure, and detect post-install drift against the recorded or catalog-pinned normalized hash.
167
- - Fixed worker budgets so late-run guardrails preserve the artifact a worker was admitted to produce. The reserve and soft limit now end broad discovery rather than read, write, edit, or the closed `orientation` product's `code_nav` delivery surface; executed calls remain bounded by the lifetime cap, refused calls do not spend it, free refusals carry their own bounded backstop, and loop escalation scales with the worker's admitted budget instead of the interactive threshold.
168
- - Made provider thinking controls and usage accounting describe what actually reached a model. `thinking: off` now sends an explicitly mapped off effort for supported effort-level families, while families with no such mapping continue to omit the field and LM Studio native reports its transport cannot send template thinking controls. Reasoning tokens remain accounted even when Clio suppresses the thought text, so an incorrect reasoning-never classification cannot make the server's observed usage disappear.
169
- - Made the npm release contract follow the runtime resources an installed CLI actually resolves. One shared release manifest now drives the package gate and tests, source analysis checks literal package-root paths against the allowlist, bundled HTML remains present for `clio-coder docs`, and generated repository-local handbooks cannot leak from a release checkout into the tarball.
170
- - Changed how the TUI treats a slash command it does not recognize. A command-shaped token now fails with `/<token> is not a command. Type /help for the list.` instead of being sent to the model as ordinary chat. This removes the fall-through that previously let four renamed spellings, `/status`, `/hotkeys`, `/skills`, `/connect`, `/disconnect`, `/receipts`, and every typo reach the model as a question about itself. The accepted cost is that one command-shaped word followed by prose now resolves as a command, so `/tmp is full` fails; the escape is a leading backslash, and `\/tmp is full` reaches the model unchanged.
171
- - Fixed `clio-coder uninstall --remove-binary` reporting that it removed a dangling launcher symlink while leaving it in place, which left a broken `clio-coder` on `PATH` after an uninstall that exited zero. Launcher ownership is now the resolved identity of this installation's own entry, so a link into a different clio-coder installation and a symlink whose target is a directory are both preserved with an actionable warning, and only a link that resolves to this installation, or a dangling link naming a clio-coder entry, is removed.
172
- - Fixed `clio-coder reset` and `clio-coder uninstall` reporting global success after a partial delete. Both now attempt every selected root, collect per-path failures instead of throwing the first one, rebuild the skeleton, name each surviving path with its reason, and exit 1 with the exact invocation to rerun.
173
- - Fixed `clio-coder trace --help` failing with `unknown trace flag: --help` and exiting 2 while every other subcommand answered on stdout with status 0. `clio-coder trace` usage also now says that `clio-coder trace ui` needs a source checkout, which is where the viewer ships; from an installed package the other trace subcommands read the same database.
174
- - Fixed `clio-coder configure --list` and the first-run runtime menu writing fixed-width rows sized for roughly 88 columns, which ran model hints to 141 columns on an 80-column terminal. The plain-stdout configure surfaces now measure the terminal and degrade by restacking rather than by dropping information.
175
- - Fixed error messages that named a command which could not change the outcome. An invalid `settings.yaml` now names the file, the keys, and `clio-coder reset --config --force` rather than `clio-coder doctor --fix`, which by design never rewrites settings content; an interrupted install that leaves a missing chunk reports reinstall instructions; and `clio-coder run --no-context-files` explains that global options precede the subcommand instead of reporting an unknown option.
176
- - Fixed a cancelled turn poisoning every later turn in the same session. The durable closing turn a cancel writes carries no provider `usage`, because no model call produced it, and it was recorded `stopReason: "stop"`, which walked it past the context estimator's aborted-or-error guard. The next turn then failed in roughly 30 ms with `Cannot read properties of undefined (reading 'totalTokens')` and no network call, and kept failing for the life of the session, because the process that wrote the record never re-read it and every later read did. The closing turn is now recorded `aborted`, which is what happened. Replay learned the same distinction: an aborted turn that carries its own text and no provider `errorMessage` is a notice Clio wrote itself and renders as that text alone, while a genuine mid-stream abort keeps its error line.
177
- - Fixed `clio-coder auth login` destroying the credentials it could not parse. Every write to `credentials.yaml` is a whole-file rewrite of the parsed view with no backup, and the parser answered a YAML error by returning an empty store, so a single login serialized that emptiness over the secrets and took the file from 211 bytes to 112. What made it likely rather than obscure is that `clio-coder doctor` called the file OK and `clio-coder auth list` showed every provider `disconnected`, the same word it uses for a provider you never logged into, so the operator is told they are logged out and then does the one thing that looks like recovery. Reading now reports what was lost on the way in, a write over invalid YAML or an unreadable shape refuses and names the file and the reason with a non-zero exit, the in-memory view is no longer updated ahead of the disk write, and `auth list` and `auth status` warn instead of understating. The refusal is bounded so a first login still works: absent, empty, comment-only, and empty-mapping files are clean rather than damaged, and only a non-empty mapping in neither known shape is refused.
178
- - Fixed `clio-coder doctor` reporting a root it cannot use as a healthy directory. The four root rows tested existence, which is true for a regular file and for a mode-000 directory, so doctor exited 0 and `clio-coder doctor --fix` one command later died on `Expected directory` and printed no report at all. Each root is now checked for being a directory that is readable, writable, and traversable, and the row names which of those failed; a `--fix` that throws records a `repair` row and still prints the rest of the report; and the metadata row no longer calls an `install.json` that is present and merely unreadable missing.
179
- - Added `[s] stop turn` at the permission prompt, which previously offered no exit from a repeating approval loop. `s` denies the call and ends the run through the same path the loop guard uses, closing the turn with a durable notice naming the tool, answering every call the turn had parked, and suppressing the re-notify that would otherwise open the next one. Escape is unchanged and still answers exactly one call.
180
- - Fixed `/cost` and the footer usage line rendering one process-lifetime accumulator that only a new session ever reset, so a resumed session showed the previous session's totals under the resumed session's id, and a process that resumed and sent nothing reported zero for a session holding tens of thousands of tokens on disk. Resume, `/tree` switch, and `/fork` reseed those totals from the session's own ledger, skipping turns marked aborted or error along with the all-zero usage block a cancelled partial persists.
181
- - Made `clio-coder doctor` report the session and credential damage its own loaders already see. A `session store` row fails when `state/sessions` is gone while state metadata remains, which previously left doctor byte-identical to a healthy run in both text and `--json`; damaged ledgers are parsed with the reader `clio-coder run --continue` uses and the row names up to three `path:line` damage points before summarizing the rest. The credentials row consults the same damage reason the storage layer reports, so an unparseable store fails the row with the YAML reason and an unreadable one names the path and the remedy rather than a raw `EACCES` string. The `invalid JSON skipped` line prints once per damaged line per process instead of six times.
182
- - Made `clio-coder usage report` distinguish a missing store from zero activity, and gave it the token and cost facts it never carried. Text names `session store missing at <path>` and `receipt store missing at <path>`, the opportunities section says the inputs are absent rather than concluding `none`, and `--json` emits `session-store-missing` and `receipt-store-missing` facts. Token and cost rows are folded from the same per-call ledger fold `/cost` reads, which now lives in `src/domains/session/usage.ts` and is imported by both surfaces, so the headless report and the TUI cannot drift into two answers for one ledger.
183
- - Fixed writers recreating state the operator had just removed. `clio-coder uninstall` reported `removed Clio Coder state` and then a live process resurrected `runs.json` from memory on quit, holding runs whose receipts no longer existed, so `clio-coder fleet status` billed them while `clio-coder usage report` said zero for the same week. One shared guard in `src/core/xdg.ts` now covers the session shutdown checkpoint, tree and meta persistence, the audit log, recent-models, and the dispatch ledger. It reads the resolved XDG cache rather than probing the path, because a bare existence check cannot tell a removed root from a first run that has not created one yet, and a probe-based guard would have silently disabled the audit writer on fresh installs. Separately, `clio-coder uninstall --dry-run` no longer leaves npm debug logs in `$HOME/.npm/_logs`, which the documented side-effect-free preview wrote on every invocation.
184
- - Fixed `clio-coder auth login` printing its success line and exiting 0 after a write the storage layer had refused, which reported a stored credential that was never written. Every credential-writing CLI path now consults the storage damage reason after the write and reports the refusal with a non-zero exit, and the interactive auth overlay consults the same reason instead of claiming success on the refused write.
185
- - Made `clio-coder reset` and `clio-coder uninstall` say what they are about to destroy and what survives. The reset preview lists each root's real children with counts and gives every scope a note, and the help defers to that listing rather than describing it. Uninstall enumerates the per-project `.clio-coder/` directories recorded in session metadata and names `clio-coder context reset --all` before the roots and the launcher are touched, because the record it reads lives inside a root it is about to remove and the cleaner is a subcommand of the binary `--remove-binary` is about to delete. Per-project directories are still not removed by uninstall; this makes the ordering visible rather than closing the gap.
186
- - Made a failed settings reload render as one width-clamped notice from the shared settings formatter instead of an inspect dump and a stack trace that corrupted the frame. Read, parse, and schema failures are distinguished, the remedy matches the kind and leads the detail so frame clamping cuts context rather than the fix, the raw YAML source fragment is gone, and the previous good configuration stays active. Doctor's settings row and the thrown error use the same helpers.
187
- - Fixed `clio-coder targets --probe` writing fixed-width rows that ignored the terminal. The id column never truncates, url and model truncate with an ellipsis, `--json` stays untruncated, and surplus width goes to url and model before the tier column, which had let four distinct hosts render identically at 120 columns while the group header already carried the tier.
188
- - Made the TUI tell the truth about a denied tool call and about usage the ledger was hiding. A denial renders as the dialog outcome and status alone, the collapsed row reads `bash(id) ✗ blocked` instead of claiming the call `ran`, and no parked-approval line survives the decision. `/context compact` persists its summarization usage on the compaction entry and moves `/cost` immediately, so a real model call stops being invisible to every usage surface. A cancelled turn now persists estimated usage marked `estimated: true` in the shape a completed turn uses, rather than the all-zero block the cancel path used to write beside real streamed text, and `/cost` readers skip it. `/tree` shows the fork parent and the compaction and branch nodes, and replay no longer dumps `custom:promptRecompiled` JSON into the transcript. The stop-turn notice no longer doubles the usage line onto the empty assistant entry it splits open, and a settled entry holding nothing renders nothing instead of an empty bubble.
189
- - Made overlay footers say what Esc will do from where the overlay is right now. The verb was derived from whether an overlay committed anything, which produced `cancel` on half of them and `close` on the other half while neither word described the key: a list overlay clears a typed filter on the first Esc and closes on the second, and the footer said `close` through both. The caller names the action now, out of a fixed vocabulary of `close`, `clear filter`, and `back`, and `/model` and `/resume`, which close on the first Esc, say so instead of advertising a filter they do not clear. Every overlay uses one filter vocabulary.
190
- - Fixed a repair masquerading as an install. `initializeClioHome` stamped `installedAt` when it had only repaired an existing home, so the install date moved every time doctor rebuilt the skeleton. A repair writes `repairedAt`, `install.json` accepts either stamp, and doctor prints whichever exist. The doctor damage row groups identical messages instead of repeating one error three times.
191
- - Made every cost surface claim only what something measured. The cost aggregate could not tell nothing-priced from priced-at-zero, so the footer asserted `$0.00` before any turn had run while `/cost` on the same session said no usage was recorded, and after a real turn against a target that publishes no pricing both printed the literal words `cost unknown` where a number belongs. The aggregate carries its call count, one formatter answers every surface, and a surface with no priced call shows no cost field at all. Usage and cost are separate claims: the tokens were counted whatever the pricing did, so the session total stays on screen and only the cost field disappears. A fixed-width cell that owns a cost column cannot drop the field, so it says `not measured` in words rather than inventing a number.
192
- - Made the footer describe the branch on screen rather than the whole ledger file. `current.jsonl` is append-only, so after a `/tree` switch the abandoned sibling turns are still in the file: the transcript stopped showing them while the session total, `/cost`, and the last-turn status line kept counting and describing them. All three now fold through the same active-path lineage the transcript replays, on `/tree`, `/resume`, and `/fork`. The footer's chip strip also ranks its chips, so the chip budget and the width budget both shed per-turn detail first and the session total last; it used to slice whatever sat at the end of the list, which cost an 80-column terminal both the total and the cost chip while spending the whole budget on per-turn detail. Two fields the ledger does not record are left neutral on a rescoped line rather than reconstructed: the watchdog peak, and reasoning tokens no provider reported.
193
- - Made the installer and uninstaller describe one `PATH` the same way. `scripts/install-local.sh` compared the resolved `clio-coder` against its own link path as raw strings, only when the bin directory was already on `PATH`, and never on a dry run, which is why a shadowing install could go unreported in exactly the case where the other `clio-coder` is the only one a bare name reaches. Both sides are resolved through their symlinks before they are compared, and the installer prints the same `another clio-coder is on your PATH at <path>` warning and check command that `clio-coder uninstall` prints. The README install block clones the release ref, carries verbatim the `export PATH` line the installer prints when the bin directory is not on `PATH`, verifies with the launcher's resolved path, and asks `command -v clio-coder` rather than comparing `--version` output, because two installs of the same release report the same version while the bare name still resolves to the other one. The warning names the divergence; it does not reorder anyone's `PATH`.
194
- - Added a reserved `clio:` frontmatter block so a Clio skill stays a portable skill. Catalog skills carried `registry-id`, `source-url`, and audit fields at the top level of their frontmatter, mixed in with the community-standard keys every agent loader reads. `name`, `description`, `version`, `license`, and `allowed-tools` now stay exactly as the wider ecosystem expects, and everything Clio-specific moves under one `clio:` mapping extended with provenance, origin, eval status, model size, and agent bindings. A Clio skill dropped into any `.claude/skills` directory loads unchanged with the block ignored, and loaded by Clio Coder it carries its full marketplace metadata. The loader reads the nested block first and falls back to the flat keys for already-installed copies, install writes the nested form, and the pinned content hash covers it.
195
- - Added sixteen ported skills and a `git-master` builtin agent, taking the catalog from sixteen skills to thirty-two. Every port was rewritten to the catalog bar rather than copied, with provenance and origin recorded in its `clio` block and eval status honest at `untested` for the ports that have not been evaluated. `git-master` owns bounded git operations end to end and binds the six rewritten git skills, while `coder` binds only commit-crafting and review-changes on role fit, because binding injects skill names as guidance and the rest would misdirect an implementation loop. Two limits are recorded rather than papered over: `ast-grep` is bound to the main agent only, because the scout role's read-only tool class cannot admit an execute-class tool and the clean path is a typed read-only tool that does not exist yet, and `worktree-create` stays sequential because no worker recipe carries bash. `coding-standards` and `tech-spec` enter as provisional under a documented trial field.
196
- - Reorganized the thirty-two skill catalog into seven category folders, `research`, `coding`, `git`, `planning`, `context`, `workflow`, and `meta`, and renamed the skills to carry their job instead of their source repository's branding. The `piv-` prefix is gone in favor of `commit-crafting`, `review-changes`, `investigate-issue`, and `create-pr`, `plan-create-prd` becomes `product-requirements`, `plan-architecture` becomes `architecture`, and `plan-create-stories` becomes `backlog`, with titles, descriptions, eval headers, cross-references, source URLs, and the builtin agent bindings all moving together. The pin script discovers packages recursively and renders `registry.yaml` grouped by category, and marketplace discovery probes one category level down. The move also surfaced a packaging defect it fixes: the release manifest shipped only `cut-it`, so an installed package could not have resolved `git-master`'s bound skills, and it now ships `workflow/cut-it` and the whole `git` category.
197
- - Smoke-checked the entire skill catalog against a 30B local model through the real eval harness, one transcript-verified scenario per skill. The campaign surfaced a judge that scored instruction prose instead of transcripts and a permission gate that blocked every git call headless, so the git scenarios gained seeded repository fixtures and the expected results were rephrased to observable behavior. Five skill bodies were corrected from what the runs showed, each skill records a dated smoke result, and `smoke-checked` is a documented eval status; skills the cutoff reached first remain honestly at `scenarios-recorded`.
198
- - Added `queued_user_turn` to the `--json` wire protocol. A steer or follow-up typed while a turn is still running used to enter the run with nothing on the stream to mark it, so a `--json` consumer saw the model react to text it had never been told about. The frame is emitted at injection time, when the queued text actually enters the run rather than when it was typed, and it carries the same text the model receives. It is a new frame on a public wire, so consumers that switch on frame kind should expect it.
199
- - Removed the four-second freeze on opening a workspace and took the redraw cost off the transcript's length. Codewiki startup ran a synchronous `statSync` sweep that blocked the event loop for 3342 ms on this repository and now costs 12 ms. TUI dirty-frame render fell from 10 ms to roughly 0.14 ms at 400 turns, because the frames above the viewport are a frozen prefix that is composed once instead of re-rendered every keystroke. Workspace enumeration no longer holds the loop for a 34 ms floor; a cooperative async walk yields between batches, and the worst block observed under validation is 9 ms to 15 ms.
200
- - Made the settings center's three-column layout start at a real 120-column terminal. The description earned its own column at a threshold of 116, but two nested frames sit above that body and each costs a border and a pad, so a 120-column terminal only ever presents 112 columns and the layout never appeared outside a 124-column window. The floor is 112, and the contract test pins both sides of the boundary at the body widths a 120- and a 119-column terminal actually produce.
201
- - Added a fleet roster to the compiled session prompt, so the orchestrator can see the agents it may dispatch instead of guessing. `agent: "auto"` now baselines from task shape, sending code work to `coder`, tests to `tester`, docs to `documenter`, review to `verifier`, and research to `researcher`, where it previously resolved to the read-only `scout` for everything.
202
- - Replaced the delegation incentive with a rule the model can evaluate: two or more independent file-scoped subtasks, or any broad exploration, goes to workers, while synthesis, validation, and a single narrow change stay with the orchestrator. A file handed to a worker is not the orchestrator's to edit, and an admission refusal is reported rather than silently worked around.
203
- - Made the run summary the orchestrator reads carry what the worker did. A mutation-class run whose mutating calls all failed now reports that nothing was written instead of a bare successful exit.
204
- - Made sealed results measurable against the run's own tool events. A reported changed path the run wrote verifies, one whose only write was refused fails, one that exists but was never touched seals quality unmeasured, and one that exists nowhere fails outright. A passing validation claim that no executed command supports downgrades to unmeasured with the claim named, while a check the recognizer cannot tie to a claim is reported unmatched rather than called fabricated.
205
- - Added capability-mismatch refusal at dispatch admission. Pointing a read-only recipe at mutation work is rejected before a worker exists, and only when the caller pinned the id, the task classifies as mutation, and the recipe's postcondition is a report. Anything short of that admits with a flag on the receipt, and `agent: auto` never refuses.
206
- - Gave workers the workspace root and a bounded top-level listing on every dispatch, closing the case of a worker spending a fifth of its call budget locating the repository.
207
- - Added operator-initiated backgrounding. `Alt+S` converts a running attached dispatch into a detached batch, the runs continue, and collection is unchanged. Review gates, compete, pipelines, Scout fleet plans, and time-boxed calls refuse the conversion and say why.
208
- - Added `monitor` with `mode="tools"`, which reports what a run executed from the event tail and the receipt's integrity-checked totals.
209
- - Made scratch directories under `/var` write like `/tmp`. `/var` stays protected with `/var/tmp` and `/var/folders` carved out, `/run` joins the protected list, and a pathless `artifact` call no longer writes into a system root without confirmation.
210
- - Made compound bash earn recognition. A `&&` chain is evaluated at its most restrictive recognized member and refused as a whole if any member is unrecognized, and an `sh -c` wrapper is read through to its inner command rather than around it.
211
- - Made a headless denial name the form that would have been recognized. A headless run cannot answer a permission prompt, so an agent that hit the wall used to retry variants until the loop guard ended the turn.
212
- - Fixed project-context bootstrap being dead on llama.cpp. That server refuses any request carrying both a response schema and tools, and every bootstrap dispatch carries tools; the conflict is now refused at admission and the run goes straight to the prompt-parser fallback. A failed bootstrap reports the real error instead of "did not return an assistant response", and a `--rewrite` that fell back to the existing handbook says so and exits 1.
213
- - Made interactive turns visible to `clio-coder trace`, recorded as runs under a `session` assignment id with one event per tool call. A missing database now reads as an empty state naming its path rather than raw sqlite text, and the `node:sqlite` experimental warning is gone from trace commands unless `--trace-warnings` is passed.
214
- - Made the TUI read the tool verdict the registry issues instead of pattern-matching output text, which had rendered a failing test whose output contained the word "cancelled" as a policy block. The verdict persists through `/export` and `/resume`.
215
- - Fixed a headless run whose turn ends on a terminating artifact printing nothing and exiting 0. The artifact content is the answer and now reaches stdout.
216
- - Made `clio-coder skills eval` measure the skill instead of the harness. Arms are hermetic and run at full-auto in disposable workspaces, so an exec-class call is no longer scored as a skill failure; `--allow-network` opts back into network tools. The judge scores transcript evidence only, and an unparseable judge verdict is `unmeasured` at exit 3 rather than a verdict.
217
- - Published `skills/skill-marketplace.json` beside the pinned registry, so an install can point `CLIO_CODER_SKILL_MARKETPLACE_INDEX` at it and resolve bare names. `clio-coder skills install` takes several names and `--category`, `search` matches that same category vocabulary, and `inspect` on a marketplace name says the skill is not installed rather than denying it exists.
218
- - Fixed the Skills Hub listing fifteen installable-looking rows in an environment with no marketplace configured. The rows came from a third source nothing else consulted; the hub now builds from the same discovery the installer uses, and a fresh environment gets an empty state naming the two real ways to get a skill.
219
- - Fixed `--help` executing, rejecting, or being swallowed across the `targets`, `context`, `fleet`, and `auth` subcommands. Every one answers with usage on stdout at exit 0 and executes nothing.
220
- - Gave the artifact loaders one not-found voice. `clio-coder eval report`, `clio-coder evidence inspect`, `clio-coder evolve manifest validate`, and `clio-coder components diff` named raw filesystem errors; they now name the id or path that was missing, and `clio-coder docs` prints its absolute served directory from any working directory.
221
- - Reconciled the `clio-coder auth login` and `clio-coder configure --list` runtime inventories, which drew from one source and looked like two products. Each screen now says what it shows and names the command that shows the rest.
222
- - Fixed builtin fleet contracts listing as invalid on a fresh checkout, where a missing command registry is setup and not corruption. Those entries render as setup with the exact file and keys to create, while a registry that binds wrong ids stays invalid.
223
- - Stopped `clio-coder configure` presenting two alphabetically-first catalog ids as recommendations and seeding `defaultModel: gpt-4` when none was given. A catalog-backed runtime reports its catalog size and requires `--model`.
224
- - Made overlays respect the width they are given. The permission overlay wraps its safety text and keeps `[s] stop` at every width, hints elide by priority rather than position, and `/view` drops to one pane rather than squeezing two into noise. Verified at 40, 80, and 120 columns.
225
- - Made `thinking: max` reachable. Dispatch validation rejected it and route selection threw on it mid-run, so the level was unreachable through the dispatch tool, `clio-coder run`, and the CLI; the level list now has one source and a drift test fails any file that respells it.
115
+ ### Added
116
+ - The first npm-published `clio-coder` command and namespace: binary, XDG roots, project directory, handbook, environment variables, and extension manifests use `clio-coder` naming.
117
+ - Agent ledgers, intentional compete stances, durable dispatch capacity leases, deterministic execution plans, typed worker contracts, and transactional worker attempts.
118
+ - Soak and invariant evaluation suites, receipt-derived accounting, per-step write-boundary checks, bounded check/repair loops, deterministic fleet code steps, and headless session continuation.
119
+ - Improved context bootstrap/refresh and generated-wiki workflows, hardened skill discovery/install/evaluation, and updated Pi engine dependencies to 0.83.0.
120
+
121
+ ### Changed
122
+ - `clio-coder --help` now emphasizes the human-facing command surface; `--help --all` retains the full scripting and harness surface.
123
+ - Reorganized TUI internals without changing its public command surface, and made context budgeting favor target-reported limits.
124
+ - Unknown slash commands now fail closed; use a leading backslash for command-shaped prose.
125
+
126
+ ### Fixed
127
+ - Preserved active session branches across compaction and replay, made startup option parsing strict, and bounded ACP deadlines.
128
+ - Stopped automatic retries after potentially state-changing tool calls.
129
+ - Fixed JSON transcript duplication, evaluation threshold enforcement, cancellation recovery, usage totals, uninstall/reset reporting, and narrow-terminal configuration output.
130
+ - Corrected provider URL launching, damage-control matching, credentials handling, doctor reporting, and trace help behavior.
131
+
132
+ ### Security
133
+ - Replaced shell interpolation when opening provider URLs on macOS/Linux with argument-safe process spawning.
134
+ - Made dispatch plans immutable and receipt-backed, enforced worker attestation and write-boundary recovery, and fail-closed on malformed durable contracts, unknown skills, and unsafe retries.
135
+
226
136
  ## 0.2.9 - 2026-08-05
227
137
 
228
- - Added deterministic `code` steps to fleet contracts and execution plans. A step names a command id from the repo-owned registry at `.clio/fleets/commands.yaml`, never a shell string a model authored, and an unknown id or a missing registry fails contract validation before anything dispatches. A code step holds no capacity lease, carries no execution role or authority grant, and returns the typed `code-report` contract; under `onFailure: continue` its verbatim output is the input to the step that repairs it.
229
- - Added bounded check/repair loops and shipped three SDLC fleets. A `loop` declares `maxAttempts`, a check (a registered command or a gate agent), and an agent repair; compilation unrolls it into conditional nodes so the plan stays one deterministic hashed DAG with whole-plan admission and a receipt per attempt. Verification staleness is scheduler-enforced: a workspace step landing after a green re-runs it before any dependent may rely on it. `build-test`, `build-review`, and `sdlc` ship from `src/domains/agents/fleets/`, and a project `.clio/fleets/<name>.md` shadows a builtin.
230
- - Added a durable trace store: every run is mirrored into a rebuildable WAL SQLite database beside the ledger, with a fail-closed schema version, seven Clio-mapped tables, and one documented rowid-cursor query. Writes are bounded and secret-redacted, run off the worker event pump, and drop display-only progress before any lifecycle, terminal, tool, attempt, or usage fact; a trace failure degrades the mirror, never the dispatch it observes.
231
- - Added `clio trace` (`runs`, `phases`, `tail --follow`, `procs`, a single read-only `sql` SELECT, and `ui`) plus a no-build localhost-only waterfall viewer under `apps/trace-viewer`, outside the published package. Component dollar columns stay empty where no authoritative producer supplies them, and the viewer says "not recorded" rather than inventing a breakdown.
232
- - Added per-step write boundaries. Fleet contract v4 declares a `writes` path allowlist per step, `readonly` being the empty one, and the orchestrator verifies the claim after the fact by diffing the checkout against a pinned baseline, rolling back unauthorized changes, and failing the step with `writes_boundary_violation` naming the paths and the declaration. This is detect-and-rollback rather than sandboxing; a path that cannot be restored cleanly is reported and left for the operator instead of guessed at.
233
- - Updated the Pi engine dependencies to 0.80.6, including native `max` thinking-level support and upstream runtime, accounting, and protocol fixes.
234
- - Made singular `dispatch({agent, task, briefing, detach})` first-class while preserving batch `tasks`, pinned task/briefing separation through approval, and rejected briefing-only or ambiguous `task`+`tasks` calls.
235
- - Made native-worker initialization fail closed on the same `worker_announce` wire-version handshake locally and over SSH, then required a full identity and resource attestation before the first model call.
236
- - Required a nonempty receipt-sealed final answer for successful native and ACP delegation, with deterministic `worker_final_output_missing` failure and no automatic retry; added prose-free steering provenance and advanced receipt integrity to strict v15, whose reader rejects every earlier format.
237
- - Made model-facing dispatch and collect output distinguish verified receipt integrity, evidence verification, briefing provenance, and bounded project-context provenance, and tightened collect-before-synthesis and bounded Scout spot-check guidance without restoring forced routing.
238
- - Unified ordinary synchronous and detached model dispatch over one per-tool run consumer: synchronous calls auto-wait on the registered drain, while detached calls return run ids and leave that same drain running so the parent model can monitor or steer mid-run. The interactive operator/TUI can monitor and steer an active synchronous native run through the dispatch contract; ACP runs remain monitorable but have no steering channel. Review and compete retain direct gate-sensitive drains so reviewer/judge output is staged before their receipt-facing settlement path.
239
- - Added one prompts-domain-compiled Clio worker harness with deterministic canonical tool slicing, stable/dynamic prompt separation, and read-only reviewer/judge parity; added strict per-agent `budget` frontmatter, Scout `18/4/true` and Coder `50/5/true` profiles, native and Claude SDK enforcement, and fail-closed explicit-budget admission for black-box subprocess runtimes while preserving the operator hard cap.
240
- - Hardened dispatch authority and provenance: approval now uses one registry-owned, deeply immutable resolved plan (effective agent, target, model, node id/kind/host, every bounded gate role, and cost ceiling); approval rendering and execution consume that same trusted artifact, execution fails closed on route or ceiling drift, forged plan fields are ignored, and receipts identify the actual one-shot approval or full-auto decision.
241
- - Added write-ahead, integrity-covered coordinator evidence for review verdicts, compete winners, and supervised/full-auto winner application; reviewer/judge output is staged before receipt settlement and reconstructed only from a verified receipt after restart, and evidence bundles now include `gate-decisions.json`.
242
- - Closed external-agent policy gaps by canonicalizing standardized ACP locations for path-bearing reads and mutations, rejecting contradictory tool metadata, unenforceable profiles, and authority narrowing before launch, representing external tool inventory as unknown, supporting bounded read-only ACP reviewer/judge roles, and bounding resistant ACP process groups on abort, stall, failure, and successful teardown.
243
- - Made protected-artifact boundaries durable across session append, synchronous flush, reload, reset, restart, local/shared-filesystem dispatch, and compete worktrees with a write-ahead recovery journal, merge-time protected-diff check, and fail-closed degraded mode.
244
- - Made iterative compaction cumulative, surfaced compaction failures distinctly from legitimate no-ops, and isolated repository-scoped memory by canonical repository identity in interactive and headless agent prompts.
245
- - Made compete worktree ownership transactional and segment-safe, with all admitted workers settled before cleanup, durable coordinator/worker process leases, PID-reuse-resistant termination of hard-crash orphans at orchestrator startup, and restart preservation of a pending or recovered winner while losers are removed.
246
- - Made broad repository exploration model-authored: the chat harness and middleware no longer force Scout routing or block direct reads, while the operating contract, Scout catalog description, and an advisory after 9 or more manual read-only calls steer delegation. Dispatched Scout workers retain the 18-call exploration-to-synthesis guardrail, and the task-aware Fleet Runs UI exposes live tools, tokens, priced cost, retries, steering acknowledgements, and per-run cancellation.
247
- - Added strict versioned agent recipes and typed terminal result contracts. Malformed custom recipes are quarantined, built-in schema failures stop startup, Scout citations must be grounded in the worker's own live reads, and a postcondition that was never reached is recorded as `not-reached` rather than fabricated as a quality failure.
248
- - Added one deterministic `ExecutionPlan` v2 DAG with whole-plan preflight, capacity-bounded waves, authenticated handoffs, requested/approved authority on every task, and a strict resolved-plan v3 boundary with explicit deadlines. Older or partial durable forms are rejected rather than migrated or accepted through aliases.
249
- - Added process-safe global and per-node admission with durable expiring leases, deterministic priority/FIFO queues, finite deadlines, retry reservation rebinding, TTL-bounded operator drain, and cross-process owner-liveness checks. Placement spreads by durable lease usage, but leases remain the authority under the state lock.
250
- - Added a two-lane worker protocol and pre-call attestation of protocol, process group, host, settings and WorkerSpec digests, runtime, target, endpoint, model, tool surface, and resource facts. Local and remote aborts terminate the whole process group; display backpressure cannot delay heartbeats, acknowledgements, or receipt-bearing frames.
251
- - Added measured joint route resolution across agent, target, model, runtime, and node. Hard constraints eliminate before deterministic Pareto ranking; route history v3 aggregates only compatible capability evidence, retires older files, and invalidates buckets on tool-surface or endpoint drift.
252
- - Added operator-scoped active routing for read-only-capability work in researcher, verifier, reviewer, and judge roles. Shadow remains the default; activation requires named role/posture settings and exact-route readiness evidence, and no-ready-candidate, manual pin, and `failover: none` paths fail closed.
253
- - Added bounded `agent: auto` evaluation and typed Scout phase escalation. Agent authority, tools, skills, result contract, locality, and governance remain hard filters; authority-changing transitions require an authenticated plan approval or existing full-auto authority, and agent automation stays shadow by default with per-agent/per-role readiness.
254
- - Added transactional attempt isolation for editing work. Every assignment owns baseline-pinned attempt worktrees, winning changes are checked for outcome, receipt integrity, result conformance, quality gate, protected artifacts, ancestry, and destination cleanliness before apply, and a refused winner is preserved with recovery instructions.
255
- - Advanced receipt integrity to strict v15, route policy to v4, route history to v3, ExecutionPlan to v2, and resolved dispatch plans to v3. Current readers accept only the current format for each boundary; earlier forms are rejected or explicitly retired.
256
- - Clarified the compact main and worker harness prompts with an explicit direct-tool inventory and a strict distinction between tools, fleet agents, and operator-activated skills, preventing capability questions from triggering irrelevant fleet queries.
257
- - Grounded generated wikis in detected repository instructions and their declared source-of-truth documents, added dirty-working-tree evidence and source-tree freshness metadata, tightened capability-claim guidance, let the bound documenter worker profile control its thinking level, and raised the local tool-bearing turn ceiling to the configured 32K default so long-context documenters are not artificially constrained to 16K.
138
+ ### Added
139
+ - Deterministic fleet code steps, bounded check/repair loops, shipped SDLC fleets, and a durable trace store with read-only trace commands and viewer.
140
+ - Per-step write-boundary verification, typed worker result contracts, strict worker attestation, and process-safe capacity/routing leases.
141
+ - One compiled worker harness with explicit tool/budget profiles, model-facing dispatch/collect provenance, and transactional editing attempts.
142
+
143
+ ### Changed
144
+ - Added first-class singular dispatch while retaining batch dispatch, and unified synchronous and detached run monitoring.
145
+ - Updated Pi engine dependencies to 0.80.6 and advanced receipt, route, plan, and policy formats to their strict current versions.
146
+ - Broadened model-authored repository exploration while retaining bounded Scout guidance and Fleet Runs visibility.
147
+
148
+ ### Fixed
149
+ - Made successful native and ACP delegation require receipt-sealed final output, and made protected-artifact recovery durable across restart and worktrees.
150
+ - Improved compaction, context provenance, routing, external-agent cancellation, and generated-wiki grounding.
151
+
152
+ ### Security
153
+ - Enforced immutable approved dispatch plans, strict external-agent policy checks, bounded worker protocol frames, and fail-closed handling of older or partial durable formats.
258
154
 
259
155
  ## 0.2.8 - 2026-07-07
260
156
 
261
- - Redesigned the tool surface into seven planes with consolidated observe,
262
- execute, orchestrate, retrieve, interact, mutate, and artifact tools.
263
- - Added session task tracking with the `tasks` tool, `/tasks`, open-task
264
- continuation nudges, and receipt-backed task evidence.
265
- - Added richer dispatch controls: `monitor`, `steer`, pipeline dispatch,
266
- ad-hoc specialist personas, and worker permission escalation.
267
- - Rebuilt context indexing around codewiki schema v4, a separate
268
- agent-authored wiki layer, `/context refresh`, worker context injection, and
269
- one ignore policy for grep/find visibility.
270
- - Added prompt manifests, eval provenance, public benchmark manifests, unified
271
- observation truncation envelopes, `/export`, and deeper per-tool usage docs.
272
- - Improved the interactive UI with a shared TUI design system, slash-command
273
- argument completion, a `/context` hub, clearer permission queues, and better
274
- skill-loading guidance.
275
- - Fixed safety and autonomy edge cases around approval denials, symlink path
276
- checks, loop-guard recovery, external worker tool profiles, reasoning-off
277
- models, and source-tree awareness in nested repositories.
278
- - Reworked local model residency: local inference targets are treated as
279
- multi-model servers with finite VRAM, one shared reconciler drives llama.cpp
280
- routers, LM Studio, and Ollama, co-resident models such as a scout beside
281
- the main coder are protected symmetrically, residency mutations against one
282
- server are serialized across processes, and the `CLIO_RESIDENCY=observe`
283
- and `lifecycle: user-managed` opt-outs now work on every runtime path.
284
- - Added native shadow-agent fleet routing: `/agents` lists shadow agents and
285
- `/fleet` binds native agents to target/model worker profiles, including
286
- changing a bound profile's model from the bindings tab.
287
- - Enforced loop-guard synthesis lockouts mechanically for the main agent and
288
- dispatched workers: locked turns can only answer (request-level
289
- `tool_choice: none`), dead tool-call markup is sanitized out of locked
290
- answers, and a result-stagnation detector blocks byte-identical retry
291
- escalations that evaded the verbatim detector.
292
- - Conditioned the harness for local models with measured fixes: a
293
- deterministic tool-routing order in the system prompt, task-board reminders
294
- on enumerated multi-step requests, a validation nudge on successful edits,
295
- recovery guidance with sanctioned pivots on blocked calls and denials,
296
- ask_user gated to genuine decisions, bundled-docs retrieval repairs, and
297
- observation-budget stubs that end retry traps.
298
- - Made the TUI truthful under pressure: context meters draw the autocompact
299
- reserve with its own glyph, the dispatch board renders at the terminal's
300
- real width, overflowing rows drop whole facts behind an ellipsis instead of
301
- clipping mid-number, permission overlays show the parked call's target
302
- (including worker escalations, sanitized at the trust boundary), and
303
- blocked or aborted tool calls settle instead of spinning forever.
304
- - Fixed accounting: aborted turns keep their real token usage, headless run
305
- receipts sum usage across all agent segments, guard blocks are recorded as
306
- blocked safety decisions, and worker tool-call caps count blocked attempts.
307
- - Fixed session integrity across `/tree`, `/fork`, resume, and compaction so
308
- abandoned sibling turns are never replayed, copied, or summarized.
309
- - Hardened worker IPC: subprocesses drain stdout before exit, streaming no
310
- longer amplifies quadratically, and internal generator dispatches (wiki
311
- update, context bootstrap) get a wall-clock deadline with progress output.
312
- - Fixed deadline timers that could silently never fire in a quiet process:
313
- ACP request timeouts, dispatch and internal-generator deadlines, the
314
- dispatch drain grace, and worker escalation timeouts now hold the event
315
- loop until they fire or are cleared.
316
- - Updated the model catalog: the Qwopus3.6 Coder entries carry the upstream
317
- presence-penalty sampler default, and reasoning-never model families no
318
- longer receive or replay thinking fields.
319
- - Aligned the documentation corpus with v0.2.8, added worker-dispatch and
320
- provider-adapter guides, and refreshed the HTML docs viewer with new
321
- interactive blueprints.
322
- - Breaking: legacy tool names such as `glob`, `workspace_context`,
323
- `docs_search`, `run_task`, `validate_frontend`, `write_plan`,
324
- `write_review`, `create_skill`, and `dispatch_batch` were removed in favor of
325
- the consolidated tool surface.
157
+ ### Added
158
+ - A consolidated seven-plane tool surface, task tracking, richer dispatch monitoring/steering, codewiki v4, exports, and improved interactive command hubs.
159
+ - Multi-model local residency management and native shadow-agent fleet routing.
160
+
161
+ ### Changed
162
+ - Improved local-model prompting, TUI pressure handling, accounting, worker IPC, deadlines, model catalog metadata, and documentation.
163
+
164
+ ### Fixed
165
+ - Corrected approval, symlink, loop-guard, worker-profile, reasoning, session-branch, and timeout edge cases.
166
+
167
+ ### Removed
168
+ - Legacy tools including `glob`, `workspace_context`, `docs_search`, `run_task`, `validate_frontend`, `write_plan`, `write_review`, `create_skill`, and `dispatch_batch`; use the consolidated tool surface.
326
169
 
327
170
  ## 0.2.7 - 2026-07-02
328
171
 
329
- - Added five reviewed marketplace skills: `scientific-debugging`,
330
- `experiment-protocol`, `design-council`, `credentials`, and
331
- `workflow-distiller`.
332
- - Added executable skill evals, enforced skill tool surfaces, and registry
333
- integrity pins for catalog skills.
334
- - Added credential damage control with zero-access credential storage,
335
- secret-value redaction in evidence bundles, and a read-only
336
- `credential_present` tool.
337
- - Added `clio usage report`, headless main-agent receipts, dispatch evidence
338
- bundles, high-rigor validation prompts, and evidence-linked change manifests.
339
- - Reduced package size, tightened the release workflow, refreshed
340
- documentation, and fixed several dispatch, lifecycle, skill, and loop-guard
341
- reliability issues.
172
+ ### Added
173
+ - Reviewed marketplace skills, executable skill evaluations, enforced skill tool surfaces, and registry integrity pins.
174
+ - Credential damage control, usage reports, headless receipts, dispatch evidence bundles, and high-rigor validation support.
175
+
176
+ ### Changed
177
+ - Reduced package size and refreshed release and documentation workflows.
178
+
179
+ ### Fixed
180
+ - Improved dispatch, lifecycle, skill, and loop-guard reliability.
181
+
182
+ ### Security
183
+ - Added zero-access credential storage and secret redaction in evidence bundles.
342
184
 
343
185
  ## 0.2.6 - 2026-06-24
344
186
 
345
- - Added VRAM-aware local model residency so interactive, headless, and worker
346
- runs reconcile model load/evict behavior through one path.
347
- - Added first-class customization surfaces: layered project settings,
348
- path-scoped rules, operator profiles, user hooks, and `clio config inspect`.
349
- - Added self-orientation through the `docs_search` tool and `clio docs`
350
- viewer.
351
- - Added SciCode benchmark support, coverage gates, deterministic repeat lanes,
352
- and refreshed guides and HTML blueprints.
353
- - Fixed a dispatched-run residency gap that could leave Ollama models resident
354
- and overflow VRAM.
187
+ ### Added
188
+ - VRAM-aware local-model residency, layered settings, path-scoped rules, operator profiles, hooks, configuration inspection, docs search/viewing, and SciCode benchmark support.
189
+
190
+ ### Fixed
191
+ - Prevented dispatched Ollama work from leaving models resident and overflowing VRAM.
355
192
 
356
193
  ## 0.2.5 - 2026-06-23
357
194
 
358
- - Added the `alcf` runtime for Argonne ALCF Sophia/Metis inference targets over
359
- Globus OAuth.
360
- - Added ALCF model metadata, gateway-specific documentation, authenticated
361
- provider discovery, and strict OpenAI-compatible payload handling.
362
- - Added contract coverage for ALCF OAuth selection, runtime discovery, and
363
- strict reasoning-payload behavior.
195
+ ### Added
196
+ - The `alcf` runtime for Argonne ALCF Sophia/Metis targets, including Globus OAuth, discovery, metadata, and gateway documentation.
197
+
198
+ ### Fixed
199
+ - Enforced strict OpenAI-compatible reasoning payloads for ALCF targets.
364
200
 
365
201
  ## 0.2.4 - 2026-06-23
366
202
 
367
- - Added agent fleet management with agent-to-profile bindings, profile CRUD,
368
- fault-tolerant dispatch, and a `/fleet` overlay.
369
- - Isolated dispatch tests from the real run ledger and pinned several dispatch
370
- invariants with regression coverage.
371
- - Made receipt digests deterministic across hosts and refreshed the pi engine,
372
- Claude SDK, Anthropic SDK, Biome, TypeBox, undici, uuid, and tsx
373
- dependencies.
203
+ ### Added
204
+ - Fleet management with agent/profile bindings, fault-tolerant dispatch, and a `/fleet` overlay.
205
+
206
+ ### Changed
207
+ - Refreshed Pi, Claude, Anthropic, Biome, TypeBox, Undici, UUID, and TSX dependencies.
208
+
209
+ ### Fixed
210
+ - Isolated dispatch tests and made receipt digests deterministic across hosts.
374
211
 
375
212
  ## 0.2.3 - 2026-06-17
376
213
 
377
- - Rebuilt the interactive command surface around a declarative slash-command
378
- registry and full-screen hubs for help, agents, prompts, extensions, targets,
379
- skills, settings, and receipts.
380
- - Introduced the enforced autonomy model with an always-on safety net, clearer
381
- approval and safety notices, and stricter tool admission across chat,
382
- workers, headless runs, and ACP delegations.
383
- - Added subscription and delegation runtimes including `anthropic-max`,
384
- `claude-code`, `claude-sdk`, Claude Code over ACP, and `antigravity-code`
385
- workers.
386
- - Added `clio context-index`, deterministic multi-language codewiki indexing,
387
- `code_nav`, scratch offloading for large tool results, middleware hooks, live
388
- agent steering, and richer receipt tool activity.
389
- - Reworked Clio's on-disk roots, settings ownership, lifecycle commands,
390
- model-target vocabulary, and observability workflows.
391
- - Removed several legacy slash commands; their workflows moved into `/skill`,
392
- `/targets`, `/help`, `/view`, and related hubs.
214
+ ### Added
215
+ - Declarative slash commands and full-screen hubs; enforced autonomy and safety notices; additional subscription/delegation runtimes; codewiki indexing, middleware, live steering, and richer receipts.
216
+
217
+ ### Changed
218
+ - Reworked on-disk roots, settings ownership, lifecycle commands, model-target vocabulary, and observability.
219
+
220
+ ### Removed
221
+ - Retired legacy slash commands; their workflows moved to `/skill`, `/targets`, `/help`, `/view`, and related hubs.
393
222
 
394
223
  ## 0.2.2 - 2026-06-11
395
224
 
396
- - Retired built-in CLI-subprocess runtimes in favor of direct HTTP/native/pi-ai
397
- targets and ACP delegation for external coding agents.
398
- - Added the context engine, single-threshold compaction, bounded tool results,
399
- prompt-cache telemetry, and session-owned live routing.
400
- - Added `clio acp`, ACP delegation support, a curated skills marketplace,
401
- richer skill activation, and local source install/uninstall scripts.
402
- - Upgraded `CLIO.md` into a project rulebook with custom sections and better
403
- source-tree awareness.
404
- - Improved prompt-prefix stability, session-ledger append behavior, permission
405
- overlays, and release verification.
225
+ ### Added
226
+ - Context engine, compaction, bounded tool results, prompt-cache telemetry, ACP support, a curated skills marketplace, and local install/uninstall scripts.
227
+ - A richer `CLIO.md` project rulebook and source-tree awareness.
228
+
229
+ ### Changed
230
+ - Replaced built-in CLI-subprocess runtimes with direct HTTP/native/Pi targets and ACP delegation.
231
+
232
+ ### Fixed
233
+ - Improved prompt-prefix stability, ledger appends, permission overlays, and release verification.
406
234
 
407
235
  ## 0.2.1 - 2026-06-05
408
236
 
409
- - Added live token-throughput telemetry, a larger context fill bar, hashed
410
- prompt-envelope delivery, and prompt diagnostics in `clio run --json`.
411
- - Narrowed per-turn tool exposure and bounded long tool outputs to reduce
412
- context pressure.
413
- - Retuned the footer dashboard for smaller terminals and refreshed README and
414
- operator documentation.
415
- - Fixed headless `clio run` argument handling, unknown-agent failures,
416
- dashboard layout, and prompt-diagnostic visibility.
237
+ ### Added
238
+ - Live token-throughput telemetry, prompt-envelope hashes, and `clio run --json` prompt diagnostics.
239
+
240
+ ### Changed
241
+ - Reduced context pressure through narrower tool exposure and bounded output; improved the footer for smaller terminals.
242
+
243
+ ### Fixed
244
+ - Corrected headless run arguments, unknown-agent handling, dashboard layout, and prompt-diagnostic visibility.
417
245
 
418
246
  ## 0.2.0 - 2026-06-03
419
247
 
420
- - First community alpha release for source-checkout users.
421
- - Added JIT skills, stronger prompt compaction, `clio init` / `/init`
422
- adoption of existing project instruction files, and centralized runtime
423
- target resolution.
424
- - Added runtime diagnostics, command-output routing, durable session JSONL
425
- coverage, expanded user docs, and a portable `Ctrl+G` leader-key fallback.
426
- - Hardened path policy handling, headless `clio run`, prompt cache boundaries,
427
- overlay rendering, session persistence, fork replay, and TUI startup.
248
+ ### Added
249
+ - First community alpha for source-checkout users, with JIT skills, stronger compaction, project-instruction adoption, runtime resolution, diagnostics, durable sessions, and expanded documentation.
250
+
251
+ ### Fixed
252
+ - Hardened path policy, headless runs, prompt-cache boundaries, overlays, session replay, and TUI startup.
428
253
 
429
254
  ## 0.1.9 - 2026-05-17
430
255
 
431
- - Added `dispatch` as a first-class fleet-agent handoff tool.
432
- - Added `validate_frontend` for HTML/CSS/JavaScript artifacts and finish
433
- evidence for typed validation tools.
434
- - Improved local model capability handling, GPT-OSS/Harmony parsing, active-run
435
- follow-ups, and cancellation behavior.
436
- - Fixed reasoning replay, Harmony stream parsing, OpenAI Codex file-tool
437
- schema aliases, lifecycle metadata repair, and duplicate model-capability
438
- paths.
256
+ ### Added
257
+ - First-class fleet `dispatch`, frontend artifact validation, typed finish evidence, and local-model capability improvements.
258
+
259
+ ### Fixed
260
+ - Corrected reasoning replay, Harmony parsing, Codex file-tool aliases, lifecycle metadata repair, and model-capability duplication.
439
261
 
440
262
  ## 0.1.8 - 2026-05-11
441
263
 
442
- - Added extensions, share archives, extension and share CLI/TUI workflows, and
443
- a redesigned welcome dashboard.
444
- - Added model and context-window validation to `clio configure`.
445
- - Added a Claude Code SDK safety bridge, supervised approval IPC, a TUI
446
- approval overlay, and receipt accounting for SDK safety decisions.
447
- - Fixed Gemini CLI token accounting and expanded tests for extensions, share
448
- archives, configure validation, and supervised SDK decisions.
264
+ ### Added
265
+ - Extensions, share archives, associated CLI/TUI workflows, a redesigned welcome dashboard, configure validation, and a Claude Code SDK safety bridge.
266
+
267
+ ### Fixed
268
+ - Corrected Gemini CLI token accounting and expanded extension, sharing, configuration, and supervised-SDK coverage.
449
269
 
450
270
  ## 0.1.7 - 2026-05-11
451
271
 
452
- - Added a shared safety policy engine for orchestrator and native workers.
453
- - Added strict project command policy parsing, typed execution tools, and
454
- receipt safety summaries.
455
- - Changed default-mode Bash to default-deny ordinary execution unless allowed
456
- by curated commands or project policy.
457
- - Hardened dispatch scope, external-runtime permission mapping, audit rows,
458
- and worker safety parity.
272
+ ### Added
273
+ - A shared safety-policy engine, strict project command policies, typed execution tools, and receipt safety summaries.
274
+
275
+ ### Changed
276
+ - Default Bash now denies ordinary execution unless allowed by curated commands or project policy.
277
+
278
+ ### Fixed
279
+ - Hardened dispatch scope, external-runtime permissions, audit rows, and worker safety parity.
459
280
 
460
281
  ## 0.1.6 - 2026-05-04
461
282
 
462
- - Added `clio --print` / `clio -p` for one non-interactive orchestrator turn.
463
- - Added stdin plus argv prompt composition and stdout guarding for scriptable
464
- print-mode use.
465
- - Reserved future JSON/RPC modes behind explicit errors and added focused CLI
466
- coverage.
283
+ ### Added
284
+ - `clio --print` / `clio -p` for one non-interactive turn, with stdin/argv composition and stdout safeguards.
285
+
286
+ ### Changed
287
+ - Reserved future JSON/RPC modes behind explicit errors.
467
288
 
468
289
  ## 0.1.5 - 2026-05-03
469
290
 
470
- - Public alpha release for developers and research-software teams testing a
471
- terminal-first coding agent from source.
472
- - Shipped the interactive TUI, target-first configuration, built-in coding
473
- agents, persistent sessions, project context, receipts, audit logs, evidence,
474
- evals, memory, and safety modes as an integrated product surface.
475
- - Added `clio init`, CLIO.md parsing, codewiki indexing, clearer `/cost`
476
- accounting, a redesigned `/model` popup, and CLIO-branded popup frames.
477
- - Documented alpha limits: source install remained the supported path, model
478
- behavior varied by target, and operators were expected to review privileged
479
- actions.
291
+ ### Added
292
+ - Public alpha for source-install developers and research-software teams: interactive TUI, target-first configuration, coding agents, sessions, project context, receipts, audits, evidence, evaluations, memory, and safety modes.
293
+ - `clio init`, CLIO.md parsing, codewiki indexing, improved cost/model UI, and documented alpha operating limits.
480
294
 
481
295
  ## 0.1.4 - 2026-04-30
482
296
 
483
- - Added the evolution plane: component inventory, typed change manifests,
484
- deterministic evidence building, local eval runs, memory records, middleware
485
- hooks, protected-artifact safety, and finish-contract checks.
486
- - Added workspace orientation, a `workspace_context` tool, eight specialist
487
- agent recipes, and a scientific-validation pack.
488
- - Unified llama.cpp runtime handling, improved the TUI, expanded compaction and
489
- context accounting, and hardened protected-artifact behavior.
297
+ ### Added
298
+ - Evolution tooling for inventories, change manifests, evidence, evaluations, memory, middleware, protected artifacts, finish checks, workspace orientation, specialist recipes, and scientific validation.
299
+
300
+ ### Changed
301
+ - Unified llama.cpp handling and improved TUI, compaction, context accounting, and protected-artifact behavior.
490
302
 
491
303
  ## 0.1.3 - 2026-04-27
492
304
 
493
- - Added live tool output, bash command echo, `Ctrl+T` thinking expansion, and a
494
- git-branch footer slot in the TUI.
495
- - Made `CLIO.md` the canonical project instruction file and improved local
496
- runtime detection for LM Studio and Ollama.
497
- - Aligned Clio Coder's identity with IOWarp's CLIO ecosystem, updated package
498
- docs, reorganized safety rule packs, and added a clean-clone smoke CI job.
499
- - Fixed slash autocomplete on Debian/Ubuntu, stabilized JSON envelopes for
500
- `doctor` and `targets`, and improved partial tool-output rendering.
305
+ ### Added
306
+ - Live tool output, Bash echo, thinking expansion, and a Git-branch footer slot.
307
+
308
+ ### Changed
309
+ - Made `CLIO.md` the canonical project instruction file and improved LM Studio/Ollama detection.
310
+
311
+ ### Fixed
312
+ - Corrected Debian/Ubuntu slash autocomplete, doctor/targets JSON envelopes, and partial tool-output rendering.
501
313
 
502
314
  ## 0.1.2 - 2026-04-25
503
315
 
504
- - Added visible retry handling for transient provider and stream failures.
505
- - Improved tool, bash, edit, dashboard, hotkey, resume, prompt, receipt,
506
- compaction, audit, and abort behavior in the interactive TUI.
507
- - Fixed retry duplication, cancellation races, oversized Bash output handling,
508
- resume/fork/new behavior during active runs, provider hot-swaps, and local
509
- OpenAI-compatible reasoning/tool schemas.
316
+ ### Added
317
+ - Visible retries for transient provider and stream failures.
318
+
319
+ ### Changed
320
+ - Improved interactive tool, Bash, dashboard, hotkey, resume, prompt, receipt, compaction, audit, and abort behavior.
321
+
322
+ ### Fixed
323
+ - Corrected retry duplication, cancellation races, oversized Bash output, active-run session operations, provider hot-swaps, and local OpenAI-compatible reasoning/tool schemas.
510
324
 
511
325
  ## 0.1.1 - 2026-04-24
512
326
 
513
- - Added deterministic loading of project context files from the current working
514
- directory upward.
515
- - Fixed resume, fork, and tree-switch replay for rich session entries and
516
- durable tool records.
517
- - Fixed subprocess worker dispatch, out-of-tree SDK runtime rehydration,
518
- receipt verification, dispatch heartbeat state, and the documented boundary
519
- check command.
327
+ ### Added
328
+ - Deterministic loading of project context files from the working directory upward.
329
+
330
+ ### Fixed
331
+ - Corrected rich session replay, subprocess dispatch, out-of-tree SDK rehydration, receipt verification, dispatch heartbeats, and boundary-check documentation.
520
332
 
521
333
  ## 0.1.0-exp - 2026-04-24
522
334
 
523
- - Initial experimental public release.
524
- - Shipped the interactive TUI, CLI lifecycle commands, target-first
525
- configuration, runtime coverage, seven built-in agents, dispatch workers,
526
- receipts, audit logs, safety modes, and XDG-aware state layout.
527
- - Known limits: Windows was best effort, and some remote fan-out and MCP
528
- surfaces were scaffolded but not yet admitted by dispatch.
335
+ ### Added
336
+ - Initial experimental public release with interactive TUI, lifecycle CLI, target-first configuration, runtime support, built-in agents, dispatch workers, receipts, audit logs, safety modes, and XDG-aware state.
337
+
338
+ ### Security
339
+ - Windows support was best effort; remote fan-out and MCP surfaces were scaffolded but not admitted by dispatch.