@iowarp/clio-coder 0.3.3 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/CHANGELOG.md +74 -0
  2. package/CONTRIBUTING.md +7 -7
  3. package/README.md +3 -3
  4. package/dist/{acp-P2AQILE2.js → acp-2BEHC4DL.js} +9 -8
  5. package/dist/{agents-72W3BI7I.js → agents-LNNFTM53.js} +29 -24
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5TWEIYDN.js → auth-KXXFI2VS.js} +14 -10
  8. package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
  9. package/dist/{chunk-YCWGATWI.js → chunk-24I7BN55.js} +2 -2
  10. package/dist/{chunk-EKMEHE4H.js → chunk-33YXPOE3.js} +2 -3
  11. package/dist/chunk-3BPUFZDL.js +37 -0
  12. package/dist/chunk-43AOLP7E.js +375 -0
  13. package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
  14. package/dist/{chunk-6SGHMWE3.js → chunk-5JGRAMKL.js} +5 -5
  15. package/dist/{chunk-V6RTAOC2.js → chunk-6US73PDB.js} +572 -51
  16. package/dist/{chunk-5UFT4SUX.js → chunk-6XXKFVSN.js} +3 -3
  17. package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
  18. package/dist/chunk-AOCYTWAV.js +449 -0
  19. package/dist/chunk-CFGTUFWB.js +67 -0
  20. package/dist/chunk-CJUB2JJ2.js +1478 -0
  21. package/dist/{chunk-FNTMWMX5.js → chunk-CKXWIANG.js} +14 -12
  22. package/dist/{chunk-PIWWS5BL.js → chunk-CYQKWTG3.js} +63 -78
  23. package/dist/{chunk-LZSJBIVT.js → chunk-DJVECN66.js} +271 -762
  24. package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
  25. package/dist/{chunk-ZWMF7253.js → chunk-E2ER4LJF.js} +304 -9
  26. package/dist/{chunk-STBPMHSX.js → chunk-EKY57CSP.js} +51 -84
  27. package/dist/{chunk-DUYJ5IO6.js → chunk-EYPA3EGJ.js} +12 -4
  28. package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
  29. package/dist/chunk-FYYLNIL5.js +313 -0
  30. package/dist/{chunk-OQ33BKR3.js → chunk-G7MUEIGA.js} +3 -60
  31. package/dist/chunk-GEYXPTRF.js +613 -0
  32. package/dist/chunk-GOXNB3AO.js +261 -0
  33. package/dist/{chunk-G4BMMOKF.js → chunk-HVDIIIQW.js} +2 -2
  34. package/dist/chunk-HWUFFB6L.js +83 -0
  35. package/dist/{chunk-4XUGQOHA.js → chunk-K7T3E2SR.js} +15 -8
  36. package/dist/chunk-K7VKOLQQ.js +15 -0
  37. package/dist/{chunk-UFIIWP2H.js → chunk-KHSFENX2.js} +8 -8
  38. package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
  39. package/dist/chunk-LCGCVYZ4.js +57 -0
  40. package/dist/chunk-LL4KHSZI.js +22 -0
  41. package/dist/{chunk-PAJK6MAQ.js → chunk-LYF7OHWH.js} +42 -15
  42. package/dist/{chunk-POHLU5DW.js → chunk-M6L6IDJG.js} +3 -3
  43. package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
  44. package/dist/{chunk-X4RCMKVQ.js → chunk-NDINPTJ4.js} +2 -2
  45. package/dist/{chunk-TZK7PACC.js → chunk-NILBFAPG.js} +14 -8
  46. package/dist/chunk-ODFEOB4F.js +1082 -0
  47. package/dist/{chunk-AGYYIBLL.js → chunk-OH3TOQTB.js} +6 -2
  48. package/dist/chunk-OZNBF4L3.js +23 -0
  49. package/dist/{chunk-DSELYM6W.js → chunk-PBTHKCPN.js} +30 -10
  50. package/dist/{verify-375KUB3Y.js → chunk-PCZJO5TI.js} +127 -42
  51. package/dist/{chunk-ED4KHGC3.js → chunk-PPAMZ32Z.js} +9 -2
  52. package/dist/{chunk-SRDMMSEP.js → chunk-QM3F2GKX.js} +1063 -1645
  53. package/dist/{chunk-X6IAEBZR.js → chunk-QNQHSOLF.js} +7 -7
  54. package/dist/{chunk-OC7FIQPC.js → chunk-R46L2BIR.js} +10 -7
  55. package/dist/{chunk-2TLUCQVG.js → chunk-RD5U66HV.js} +3 -3
  56. package/dist/{chunk-6N5PTWMY.js → chunk-RY3LY4J5.js} +50 -13
  57. package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
  58. package/dist/{chunk-OOJYHWRB.js → chunk-TSHXZTOQ.js} +6 -5
  59. package/dist/chunk-TZSKNMZG.js +434 -0
  60. package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
  61. package/dist/{chunk-VJWL6YS5.js → chunk-UUVG37B4.js} +2 -2
  62. package/dist/{chunk-COU2UHX6.js → chunk-VEZEGCGW.js} +170 -2
  63. package/dist/chunk-W6GROXXM.js +69 -0
  64. package/dist/{chunk-OAO4GE4M.js → chunk-WHGPSPT5.js} +2 -2
  65. package/dist/chunk-WHJYKASB.js +677 -0
  66. package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
  67. package/dist/{chunk-YHZX5GEU.js → chunk-XAKHZX5N.js} +2 -2
  68. package/dist/{chunk-TZTZS7QK.js → chunk-XE2VEJHX.js} +5 -3
  69. package/dist/{chunk-LM5TQCJZ.js → chunk-XF5N4U5A.js} +8 -7
  70. package/dist/{chunk-LWLEKMDQ.js → chunk-XXQNGV4M.js} +1073 -552
  71. package/dist/{chunk-KZWTDYJF.js → chunk-XYDYPRZI.js} +7 -7
  72. package/dist/chunk-ZGVHUX3M.js +66 -0
  73. package/dist/{chunk-LW6DSM3M.js → chunk-ZRGEBJ4T.js} +1192 -1119
  74. package/dist/{chunk-2DJ2KNFG.js → chunk-ZXF4XRKW.js} +202 -40
  75. package/dist/chunk-ZZMN5OM4.js +122 -0
  76. package/dist/cli/index.js +34 -30
  77. package/dist/{clio-JOU4FXVA.js → clio-M2KGYUFZ.js} +7 -6
  78. package/dist/{code-nav-7AX6FYE6.js → code-nav-GQNL7XA6.js} +8 -6
  79. package/dist/codewiki/build-worker.js +4 -4
  80. package/dist/{components-KELWS457.js → components-5TTYYX6G.js} +3 -3
  81. package/dist/{config-XCDVKR23.js → config-XUUYQIWO.js} +47 -35
  82. package/dist/{configure-4GAP54ZW.js → configure-IHJ7YOMV.js} +18 -15
  83. package/dist/{context-77FM5DV5.js → context-74JLXAWD.js} +18 -10
  84. package/dist/{context-4UOGGLQ5.js → context-75MIWW3U.js} +41 -29
  85. package/dist/{context-5VKGUVJJ.js → context-ZQ7SIFJV.js} +85 -9
  86. package/dist/{context-clear-XXJRLCJJ.js → context-clear-GYKWNUML.js} +41 -29
  87. package/dist/{context-index-BZ4UYMTC.js → context-index-SSR5ECNE.js} +3 -3
  88. package/dist/context-working-set-UX5KEP4J.js +1553 -0
  89. package/dist/{dispatch-runner-QPRDDBDX.js → dispatch-runner-GIJBHNFL.js} +47 -32
  90. package/dist/{docs-2C2LTVT2.js → docs-6FZSCG5B.js} +3 -3
  91. package/dist/{doctor-HR46URBJ.js → doctor-SVJ5BZCW.js} +12 -12
  92. package/dist/{eval-XSSNATB4.js → eval-CG6LLBLD.js} +54 -238
  93. package/dist/{evidence-6HG2PY2B.js → evidence-ZYFIEN42.js} +57 -28
  94. package/dist/{evolve-K7YU3NCY.js → evolve-QGEXEMDW.js} +36 -25
  95. package/dist/{extensions-QVDOHDGJ.js → extensions-ADGNCJJD.js} +3 -3
  96. package/dist/{fleet-VY3HHKN6.js → fleet-S5R4ZOQY.js} +73 -44
  97. package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-BHSNPBMH.js} +3 -3
  98. package/dist/{init-JYGXI3FK.js → init-5DRU55YR.js} +49 -37
  99. package/dist/memory-7YKKR6UC.js +467 -0
  100. package/dist/{models-I5QWSEOM.js → models-ZPOLRU2C.js} +24 -21
  101. package/dist/{monitor-GE4ID3IA.js → monitor-US5F5YGZ.js} +73 -46
  102. package/dist/{orchestrator-EM5MC3HM.js → orchestrator-E2AL4T5N.js} +1624 -1007
  103. package/dist/{paths-UXLN5YYZ.js → paths-E7KYAQWE.js} +3 -3
  104. package/dist/{reset-L2FQEE3E.js → reset-KZ652EK6.js} +6 -5
  105. package/dist/{run-ZU3QMZPZ.js → run-SRNBKDWD.js} +76 -54
  106. package/dist/{share-S5BZQC5I.js → share-CGZE33UP.js} +7 -6
  107. package/dist/{skills-X5VXCRNQ.js → skills-S2X4DLY5.js} +4 -4
  108. package/dist/{skills-eval-WKIHWTHR.js → skills-eval-W2GGIC4R.js} +40 -29
  109. package/dist/{targets-SNCPI2NR.js → targets-54SWINWB.js} +28 -23
  110. package/dist/{terminal-lease-BNAHVHBS.js → terminal-lease-SAIF2OGY.js} +6 -4
  111. package/dist/{uninstall-FZCQCDKC.js → uninstall-BVLWXKBT.js} +3 -3
  112. package/dist/{upgrade-JQHHPQ4K.js → upgrade-JKAR27XC.js} +20 -19
  113. package/dist/{usage-OR4O5SMZ.js → usage-MSAWCLX4.js} +79 -36
  114. package/dist/verifiers-NCBTHHN2.js +1220 -0
  115. package/dist/verify-X5HDROLA.js +25 -0
  116. package/dist/{wiki-generate-UEXP2ARI.js → wiki-generate-GUSOQ6ZP.js} +50 -37
  117. package/dist/worker/entry.js +90 -70
  118. package/dist/{workspace-G4ZWUIPR.js → workspace-ZJ6BFM3Q.js} +4 -4
  119. package/docs/README.md +8 -7
  120. package/docs/acp.md +1 -1
  121. package/docs/alcf-provider.md +1 -1
  122. package/docs/architecture.md +2 -2
  123. package/docs/artifact-placement.md +1 -2
  124. package/docs/artifact-versions.md +1 -1
  125. package/docs/built-in-agents.md +1 -1
  126. package/docs/capacity-and-scheduling.md +1 -1
  127. package/docs/commands-and-modes.md +60 -26
  128. package/docs/config-knobs-audit.md +1 -2
  129. package/docs/configuration-and-targets.md +26 -1
  130. package/docs/context-engine.md +67 -13
  131. package/docs/context-working-set.md +194 -0
  132. package/docs/development-pipeline.md +1 -1
  133. package/docs/documentation-coverage.md +6 -6
  134. package/docs/documentation-guide.md +7 -6
  135. package/docs/environment-variables.md +2 -1
  136. package/docs/eval-runner.md +1 -1
  137. package/docs/evals-internal.md +4 -32
  138. package/docs/evidence-and-memory.md +139 -7
  139. package/docs/evolution.md +1 -1
  140. package/docs/exit-codes-and-output.md +1 -1
  141. package/docs/extensions-and-sharing.md +2 -2
  142. package/docs/fleet-dispatch.md +49 -8
  143. package/docs/glossary.md +21 -1
  144. package/docs/installation-and-lifecycle.md +2 -2
  145. package/docs/middleware-and-components.md +19 -2
  146. package/docs/model-catalog.md +7 -9
  147. package/docs/observability.md +4 -4
  148. package/docs/proactive-memory.md +26 -16
  149. package/docs/prompt-envelope-and-tools.md +7 -5
  150. package/docs/provider-adapter-cookbook.md +1 -1
  151. package/docs/release-cut-checklist.md +43 -40
  152. package/docs/safety-model.md +49 -8
  153. package/docs/scientific-validation.md +21 -3
  154. package/docs/session-lifecycle.md +3 -3
  155. package/docs/skills-marketplace.md +1 -1
  156. package/docs/tool-usage.md +79 -12
  157. package/docs/trace-store.md +1 -1
  158. package/docs/troubleshooting.md +1 -1
  159. package/docs/tui-design.md +38 -4
  160. package/docs/worker-dispatch-mechanics.md +11 -1
  161. package/package.json +13 -13
  162. package/skills/meta/clio-test/SKILL.md +20 -17
  163. package/skills/meta/clio-test/evals.md +3 -3
  164. package/skills/meta/clio-test/references/harness.md +35 -6
  165. package/skills/meta/clio-test/references/test-map.md +20 -10
  166. package/skills/registry.yaml +2 -2
  167. package/skills/skill-marketplace.json +1 -1
  168. package/src/cli/agents.ts +2 -3
  169. package/src/cli/argv.ts +14 -1
  170. package/src/cli/context-working-set.ts +513 -0
  171. package/src/cli/context.ts +8 -0
  172. package/src/cli/evidence.ts +20 -2
  173. package/src/cli/fleet.ts +15 -0
  174. package/src/cli/index.ts +5 -1
  175. package/src/cli/memory.ts +272 -10
  176. package/src/cli/modes/json-stream.ts +2 -2
  177. package/src/cli/modes/print.ts +12 -1
  178. package/src/cli/run.ts +22 -2
  179. package/src/cli/targets.ts +12 -3
  180. package/src/cli/usage.ts +55 -7
  181. package/src/cli/verifiers.ts +325 -0
  182. package/src/core/bash-exec.ts +39 -14
  183. package/src/core/bus-events.ts +22 -4
  184. package/src/core/config.ts +54 -0
  185. package/src/core/defaults.ts +50 -3
  186. package/src/core/response-model-id.ts +134 -0
  187. package/src/core/toml.ts +62 -0
  188. package/src/core/verification-scripts.ts +6 -0
  189. package/src/core/workspace-files.ts +0 -1
  190. package/src/domains/agents/builtins/architect.md +1 -1
  191. package/src/domains/agents/builtins/verifier.md +3 -0
  192. package/src/domains/agents/catalog.ts +5 -4
  193. package/src/domains/agents/recipe.ts +54 -14
  194. package/src/domains/agents/result-contract.ts +7 -4
  195. package/src/domains/config/classify.ts +1 -0
  196. package/src/domains/context/bootstrap.ts +36 -27
  197. package/src/domains/context/project-metadata.ts +19 -63
  198. package/src/domains/context/prompt-context.ts +8 -0
  199. package/src/domains/context/working-set/contract.ts +161 -0
  200. package/src/domains/context/working-set/defaults.ts +28 -0
  201. package/src/domains/context/working-set/engine.ts +203 -0
  202. package/src/domains/context/working-set/fold.ts +62 -0
  203. package/src/domains/context/working-set/horizon.ts +38 -0
  204. package/src/domains/context/working-set/marker.ts +103 -0
  205. package/src/domains/context/working-set/path-index.ts +436 -0
  206. package/src/domains/context/working-set/payload.ts +152 -0
  207. package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
  208. package/src/domains/context/working-set/policies/index.ts +20 -0
  209. package/src/domains/context/working-set/policies/structural.ts +160 -0
  210. package/src/domains/context/working-set/project.ts +132 -0
  211. package/src/domains/context/working-set/protect.ts +109 -0
  212. package/src/domains/context/working-set/recall.ts +177 -0
  213. package/src/domains/context/working-set/replay/controls.ts +112 -0
  214. package/src/domains/context/working-set/replay/load-clio.ts +199 -0
  215. package/src/domains/context/working-set/replay/metrics.ts +185 -0
  216. package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
  217. package/src/domains/context/working-set/replay/report.ts +139 -0
  218. package/src/domains/context/working-set/replay/runner.ts +325 -0
  219. package/src/domains/context/working-set/replay/synthetic.ts +422 -0
  220. package/src/domains/context/working-set/replay/trace.ts +21 -0
  221. package/src/domains/context/working-set/visible.ts +54 -0
  222. package/src/domains/dispatch/budget-envelope.ts +396 -0
  223. package/src/domains/dispatch/contract.ts +2 -0
  224. package/src/domains/dispatch/extension.ts +81 -27
  225. package/src/domains/dispatch/orphan-recovery.ts +1 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +4 -0
  227. package/src/domains/dispatch/state.ts +1 -0
  228. package/src/domains/dispatch/types.ts +10 -3
  229. package/src/domains/dispatch/validation.ts +14 -0
  230. package/src/domains/dispatch/worker-spawn.ts +14 -3
  231. package/src/domains/eval/metrics/evidence.ts +0 -116
  232. package/src/domains/eval/metrics/invariants.ts +1 -1
  233. package/src/domains/eval/runners/clio-run.ts +1 -10
  234. package/src/domains/eval/runners/external-command.ts +2 -29
  235. package/src/domains/eval/schema/suite.ts +0 -7
  236. package/src/domains/eval/suites/run.ts +1 -7
  237. package/src/domains/evidence/build.ts +112 -45
  238. package/src/domains/evidence/eval.ts +24 -7
  239. package/src/domains/evidence/index.ts +53 -0
  240. package/src/domains/evidence/ordering.ts +12 -0
  241. package/src/domains/evidence/run-trust.ts +221 -0
  242. package/src/domains/evidence/store.ts +46 -6
  243. package/src/domains/evidence/trust-status.ts +854 -0
  244. package/src/domains/evidence/types.ts +26 -0
  245. package/src/domains/memory/index.ts +22 -0
  246. package/src/domains/memory/operations.ts +58 -1
  247. package/src/domains/memory/promotion.ts +281 -0
  248. package/src/domains/memory/prompt-section.ts +25 -5
  249. package/src/domains/memory/proposal.ts +51 -7
  250. package/src/domains/memory/task-bank.ts +3 -2
  251. package/src/domains/memory/task-memory-handoff.ts +181 -24
  252. package/src/domains/memory/task-memory-policy.ts +3 -1
  253. package/src/domains/memory/types.ts +37 -0
  254. package/src/domains/memory/validate.ts +178 -0
  255. package/src/domains/middleware/memory-intervention.ts +38 -25
  256. package/src/domains/middleware/runtime.ts +6 -0
  257. package/src/domains/middleware/skills-reminder.ts +19 -4
  258. package/src/domains/middleware/stalled-turn.ts +208 -5
  259. package/src/domains/middleware/types.ts +10 -0
  260. package/src/domains/observability/contract.ts +6 -1
  261. package/src/domains/observability/cost.ts +20 -4
  262. package/src/domains/observability/extension.ts +2 -2
  263. package/src/domains/providers/index.ts +3 -0
  264. package/src/domains/providers/model-discovery.ts +9 -0
  265. package/src/domains/providers/runtime-resolution.ts +38 -1
  266. package/src/domains/providers/runtimes/common/probe-helpers.ts +97 -16
  267. package/src/domains/providers/types/context-window-slots.ts +18 -0
  268. package/src/domains/providers/types/runtime-descriptor.ts +3 -1
  269. package/src/domains/safety/autonomy.ts +1 -1
  270. package/src/domains/safety/call-target.ts +211 -14
  271. package/src/domains/safety/decision-presentation.ts +268 -0
  272. package/src/domains/safety/default-path-policy.ts +8 -0
  273. package/src/domains/safety/finish-contract.ts +4 -3
  274. package/src/domains/safety/policy-engine.ts +48 -6
  275. package/src/domains/safety/redaction.ts +73 -0
  276. package/src/domains/session/compaction/compact.ts +23 -1
  277. package/src/domains/session/compaction/cut-point.ts +2 -0
  278. package/src/domains/session/compaction/tokens.ts +16 -1
  279. package/src/domains/session/context-ledger.ts +12 -1
  280. package/src/domains/session/decision-board.ts +4 -0
  281. package/src/domains/session/entries.ts +110 -1
  282. package/src/domains/session/history.ts +68 -19
  283. package/src/domains/session/manager.ts +9 -2
  284. package/src/domains/session/migrations/index.ts +22 -3
  285. package/src/domains/session/usage.ts +24 -7
  286. package/src/engine/acp/event-mapper.ts +7 -0
  287. package/src/engine/acp/server.ts +32 -2
  288. package/src/engine/agent.ts +18 -1
  289. package/src/engine/apis/lmstudio.ts +25 -4
  290. package/src/engine/apis/openai-completions.ts +147 -22
  291. package/src/engine/claude/sdk-runtime.ts +8 -2
  292. package/src/engine/claude/tool-safety.ts +13 -0
  293. package/src/engine/loop-guard.ts +27 -3
  294. package/src/engine/session.ts +9 -3
  295. package/src/engine/worker-events.ts +4 -3
  296. package/src/engine/worker-runtime.ts +59 -54
  297. package/src/entry/orchestrator.ts +34 -5
  298. package/src/interactive/chat-loop-messages.ts +40 -6
  299. package/src/interactive/chat-loop.ts +13 -0
  300. package/src/interactive/chat-panel.ts +17 -1
  301. package/src/interactive/chat-renderer.ts +49 -24
  302. package/src/interactive/clio-editor.ts +44 -7
  303. package/src/interactive/context-meter.ts +10 -0
  304. package/src/interactive/context-overlay.ts +120 -7
  305. package/src/interactive/context-recall-command.ts +110 -0
  306. package/src/interactive/cost-overlay.ts +39 -8
  307. package/src/interactive/dispatch-board.ts +212 -35
  308. package/src/interactive/footer/widgets.ts +13 -0
  309. package/src/interactive/interactive-application.ts +6 -1
  310. package/src/interactive/interactive-input-runtime.ts +11 -1
  311. package/src/interactive/interactive-presentation.ts +11 -1
  312. package/src/interactive/interactive-slash-runtime.ts +37 -1
  313. package/src/interactive/memory-overlay.ts +89 -4
  314. package/src/interactive/model-session-replay.ts +21 -0
  315. package/src/interactive/overlay-ask-user-lifecycle.ts +1 -1
  316. package/src/interactive/overlay-frame.ts +5 -2
  317. package/src/interactive/overlay-general-openers.ts +46 -1
  318. package/src/interactive/overlay-key-routing.ts +41 -1
  319. package/src/interactive/overlay-lifecycle.ts +11 -4
  320. package/src/interactive/overlay-permission-lifecycle.ts +23 -8
  321. package/src/interactive/overlay-session-lifecycle.ts +8 -4
  322. package/src/interactive/overlay-transitions.ts +11 -0
  323. package/src/interactive/overlays/ask-user.ts +74 -30
  324. package/src/interactive/overlays/decisions.ts +3 -1
  325. package/src/interactive/permission-hint.ts +35 -0
  326. package/src/interactive/permission-overlay.ts +95 -45
  327. package/src/interactive/renderers/tool-execution.ts +37 -51
  328. package/src/interactive/session-last-turn.ts +8 -1
  329. package/src/interactive/session-transcript.ts +2 -2
  330. package/src/interactive/session-usage-reseed.ts +36 -10
  331. package/src/interactive/slash-commands.ts +31 -4
  332. package/src/interactive/status/summary.ts +5 -0
  333. package/src/interactive/status/types.ts +5 -0
  334. package/src/interactive/terminal-lease.ts +1 -0
  335. package/src/interactive/turn-context.ts +333 -110
  336. package/src/interactive/turn-middleware.ts +7 -6
  337. package/src/interactive/turn-runtime.ts +37 -8
  338. package/src/interactive/turn-state.ts +3 -0
  339. package/src/interactive/worker-progress.ts +440 -0
  340. package/src/interactive/worker-stream.ts +51 -110
  341. package/src/tools/agent-tools.ts +39 -7
  342. package/src/tools/ask-user.ts +21 -1
  343. package/src/tools/bash.ts +144 -82
  344. package/src/tools/builtin-tool-catalog.ts +11 -5
  345. package/src/tools/context/index.ts +107 -5
  346. package/src/tools/context/surface.ts +3 -2
  347. package/src/tools/core-bootstrap.ts +21 -0
  348. package/src/tools/dispatch-arguments.ts +8 -0
  349. package/src/tools/dispatch-event-text.ts +19 -0
  350. package/src/tools/dispatch-runner.ts +9 -7
  351. package/src/tools/dispatch.ts +24 -1
  352. package/src/tools/monitor.ts +43 -20
  353. package/src/tools/registry.ts +72 -10
  354. package/src/tools/result-disposition.ts +706 -0
  355. package/src/tools/result-shaping.ts +321 -20
  356. package/src/tools/safe-exec.ts +2 -0
  357. package/src/tools/verify/authoring.ts +1120 -0
  358. package/src/tools/verify/catalog.ts +346 -0
  359. package/src/tools/verify/index.ts +13 -3
  360. package/src/tools/verify/scripts.ts +135 -37
  361. package/src/tools/verify/surface.ts +9 -5
  362. package/src/tools/worker-evidence.ts +54 -12
  363. package/src/worker/spec-contract.ts +43 -3
  364. package/dist/chunk-J7CWMCQD.js +0 -255
  365. package/dist/chunk-T6YILFSB.js +0 -80
  366. package/dist/chunk-VAKQQHWR.js +0 -434
  367. package/dist/chunk-VPAYEGVX.js +0 -184
  368. package/dist/chunk-XBXAASKX.js +0 -18
  369. package/dist/memory-WFZMGYHX.js +0 -236
  370. package/src/domains/eval/metrics/chaos-stream.ts +0 -93
@@ -1,9 +1,9 @@
1
1
  # Evidence Corpus and Long-Term Memory
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.3). Use it to design, validate, and simulate memory proposals, approval loops, pruning rules, and token budgets.
4
+ > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.6). Use it to design, validate, and simulate memory proposals, approval loops, pruning rules, and token budgets.
5
5
 
6
- Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.3, forensic evidence auto-builds on dispatch run completion: when a run finalizes, the observability domain automatically compiles the evidence bundle under `<dataDir>/evidence/run-<id>/` and updates a compact sidecar index row in `<stateDir>/evidence-index.json`. Long-term memory records are local, evidence-linked, and only injected after explicit approval. Use the TUI [`/view`](observability.md) command for interactive inspection of receipts, dispatch output, durable tool output, compaction summaries, and session accountability before building or citing evidence.
6
+ Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.6, forensic evidence auto-builds on dispatch run completion: when a run finalizes, the observability domain automatically compiles the evidence bundle under `<dataDir>/evidence/run-<id>/` and updates a compact sidecar index row in `<stateDir>/evidence-index.json`. Long-term memory records are local, evidence-linked, and only injected after explicit approval. Use the TUI [`/view`](observability.md) command for interactive inspection of receipts, dispatch output, durable tool output, compaction summaries, and session accountability before building or citing evidence.
7
7
 
8
8
  Source of truth: `src/domains/evidence/**`, `src/domains/memory/**`, `src/cli/evidence.ts`, and `src/cli/memory.ts`.
9
9
 
@@ -48,6 +48,7 @@ Run/session evidence files:
48
48
  ├── audit-linked.jsonl
49
49
  ├── receipt.json
50
50
  ├── gate-decisions.json
51
+ ├── trust-status.json
51
52
  ├── protected-artifacts.json
52
53
  ├── findings.json
53
54
  └── findings.md
@@ -67,6 +68,7 @@ Eval evidence adds `eval-result.json` and uses empty receipt/protected-artifact
67
68
  | `audit-linked.jsonl` | Audit rows linked to run/session context when available. |
68
69
  | `receipt.json` | Receipt bundle (`{ version: 1, receipts: [...] }`); only receipts that pass integrity verification contribute verified fields. |
69
70
  | `gate-decisions.json` | Integrity-verified review verdicts, compete winner selections, and winner confirmations discovered from linked receipt ids. |
71
+ | `trust-status.json` | Canonical per-run six-axis trust projections derived from authenticated receipts, gate decisions, grounded validation artifacts, and exact finish-contract audit rows. |
70
72
  | `protected-artifacts.json` | Protected artifact state/events. |
71
73
  | `findings.json` / `findings.md` | Structured and readable findings. |
72
74
 
@@ -159,6 +161,76 @@ whether applicable validation evidence was observed. Briefing provenance is
159
161
  also distinct from bounded project-context provenance: both can be absent or
160
162
  present independently, and neither hash is evidence for the other.
161
163
 
164
+ ### Canonical trust status
165
+
166
+ `src/domains/evidence/trust-status.ts` defines the version 1 canonical trust
167
+ status. It is a six-axis algebra, not an overall trust verdict, confidence
168
+ percentage, or pass/fail score. Consumers project only the axes needed for a
169
+ decision and preserve every other axis unchanged.
170
+
171
+ | Axis | Closed states | Question answered |
172
+ |---|---|---|
173
+ | Artifact integrity | `verified`, `failed`, `absent`, `unknown`, `not_applicable` | Did the integrity verifier authenticate the referenced artifact? |
174
+ | Validation grounding | `validated`, `failed`, `ungrounded`, `absent`, `unknown`, `not_applicable` | What correctness-bearing validation was observed and grounded? |
175
+ | Independent review | `passed`, `failed`, `inconclusive`, `not_independent`, `absent`, `unknown`, `not_applicable` | What outcome did an authenticated independent reviewer or judge record? |
176
+ | Context provenance | `recorded`, `invalid`, `absent`, `unknown`, `not_applicable` | Is the origin of briefing, project context, or linked evidence recorded consistently? |
177
+ | Autonomy enforcement | `enforced`, `approximated`, `bypassed`, `absent`, `unknown`, `not_applicable` | How faithfully did the runtime enforce the selected authority? |
178
+ | Completion evidence | `evidenced`, `incomplete`, `limited`, `absent`, `unknown`, `not_applicable` | What did the finish contract observe at the completion boundary? |
179
+
180
+ `absent` means no fact was recorded and carries a reason but no invented
181
+ attribution. `unknown` means a named source exists but cannot establish the
182
+ answer. `not_applicable` means a named authority determined that the axis does
183
+ not apply. Every non-absent state names both its source and its authority.
184
+ Sources may retain up to 16 typed artifact references. References contain an
185
+ artifact kind, identifier, and optional SHA-256 digest; they never embed the
186
+ artifact body. Normalization sorts the references and rejects duplicates,
187
+ unbounded lists, unknown fields, invalid identifiers, and sources that are not
188
+ permitted to speak for an axis.
189
+
190
+ The composition rules prohibit cross-axis promotion:
191
+
192
+ - Verified artifact integrity never promotes validation grounding.
193
+ - Recorded context provenance never promotes validation or correctness.
194
+ - A passing review never establishes authorship or context origin.
195
+ - Enforced autonomy never promotes completion evidence.
196
+ - A completion self-report never promotes validation grounding. The linked
197
+ `completion_contract` audit row is the run's own report of what it did, so it
198
+ reaches completion evidence and no other axis. Validation grounding is filled
199
+ only by independently observed executions the session ledger recorded.
200
+
201
+ The current adapters apply the following persisted-format compatibility rules.
202
+ They do not mutate receipt, gate-decision, evidence-bundle, or session formats.
203
+
204
+ | Existing persisted fact | Canonical mapping |
205
+ |---|---|
206
+ | Missing receipt | Every receipt-owned axis is `absent` with `artifact_missing`. |
207
+ | Current receipt present but integrity not checked | Artifact integrity is `unknown`; the receipt's own digest never authenticates itself. The other receipt-owned axes are `absent` with `not_observed` until authentication succeeds. |
208
+ | Historical receipt missing its integrity block | Receipt-owned axes are `unknown` through the compatibility source, even if a caller presents a contradictory positive verification result. |
209
+ | Integrity verification succeeds or fails | Artifact integrity is `verified` or `failed`. A failure leaves the receipt-owned validation grounding, context provenance, and autonomy enforcement `absent`; no untrusted receipt claim contributes a positive state. Validation the session ledger observed on its own (a validation command that ran and exited 0) still grounds the run, so a tampered run can read `artifactIntegrity: failed` beside `validationGrounding: validated`. The two axes name different artifacts and different authorities, and the bundle's `receipt-integrity` finding is what flags the pairing. |
210
+ | Receipt `verification.state: verified` | Validation grounding is `validated` unless a stronger typed failure or ungrounded claim is present. |
211
+ | Receipt `verification.state: unverified` | Validation grounding is `absent` with `not_observed`; lack of a validation tool is not a failed validation. |
212
+ | Receipt verification `unknown` or `not_applicable` | Validation grounding preserves `unknown` or `not_applicable`. A missing historical verification field maps to `unknown`. |
213
+ | Typed receipt validation or result-contract quality | A passing correctness-bearing fact maps to `validated`; a failing fact maps to `failed`; an ungrounded passing claim maps to `ungrounded`. |
214
+ | Valid bounded project context or valid briefing hash | Context provenance is `recorded`. Explicit project-context tier `none` with no briefing is `not_applicable`; a missing historical field is `unknown`; a contradictory block is `invalid`. |
215
+ | Gate decision | An authenticated independent pass or fail maps to `passed` or `failed`. Correlated review maps to `not_independent`. Unauthenticated artifacts map to `unknown`; operator or full-auto confirmation alone is `not_applicable` to independent review. |
216
+ | Receipt autonomy grade | `mediated`, `approximated`, and `bypassed` map to `enforced`, `approximated`, and `bypassed`. A dangerous-bypass flag always normalizes to `bypassed`; a missing historical block is `unknown`. |
217
+ | Finish-contract assessment | `validation_evidence`, `unvalidated_mutation`, `explicit_limitation`, and `no_mutation` map to `evidenced`, `incomplete`, `limited`, and `not_applicable`. A run whose receipt was presented and rejected downgrades `evidenced` to `unknown`: the row still points at its own record, but a rejected receipt authenticates nothing about the run it names. |
218
+ | Malformed audit row identifier | A blank or whitespace-only optional identifier is treated as absent. The row falls back to its derived correlation id or drops out of the trust projection; it never aborts the bundle. |
219
+ | Bundle without `trust-status.json` | Inspection reports `projection: historical_format` with no canonical run projections. It never reconstructs positive states from older summary tags. |
220
+
221
+ Receipt inspection, worker output, monitor details, and evidence rebuilding all
222
+ use the same authenticated receipt projection boundary. Evidence rebuilding
223
+ then composes independently authenticated gate decisions and exact
224
+ finish-contract records without changing receipt-owned axes. Findings such as
225
+ `no-validation`, `proxy-validation`, `external-approximation`, and
226
+ `external-bypass` are selected from the canonical states, while their detailed
227
+ domain artifacts remain in the receipt, gate, audit, and trace files.
228
+
229
+ The canonical aggregate is an additive projection for downstream work. Receipt
230
+ integrity remains version 15, evidence bundles remain version 1, gate decisions
231
+ remain version 2, and no persisted receipt field or cryptographic algorithm
232
+ changes.
233
+
162
234
  ### Mutation-Report Grounding
163
235
 
164
236
  Mutation-report receipts are grounded directly against observed tool events recorded in the run ledger:
@@ -174,7 +246,8 @@ Mutation-report receipts are grounded directly against observed tool events reco
174
246
 
175
247
  ```bash
176
248
  clio-coder memory list
177
- clio-coder memory propose --from-evidence <evidenceId>
249
+ clio-coder memory propose --from-evidence <evidenceId> [scope options]
250
+ clio-coder memory promote --from-handoff <path> [--entry <id>...] --scope <scope> [scope options]
178
251
  clio-coder memory approve <memoryId>
179
252
  clio-coder memory reject <memoryId>
180
253
  clio-coder memory prune --stale
@@ -195,6 +268,8 @@ The store is capped at `500` records and is sorted by scope, key, creation time,
195
268
  ```mermaid
196
269
  stateDiagram-v2
197
270
  evidence --> proposed: propose --from-evidence
271
+ taskBank --> proposed: /memory selected-entry action
272
+ redactedHandoff --> proposed: promote --from-handoff
198
273
  proposed --> approved: approve <id>
199
274
  proposed --> rejected: reject <id>
200
275
  approved --> rejected: reject <id>
@@ -205,6 +280,45 @@ stateDiagram-v2
205
280
 
206
281
  Records must cite at least one evidence ID to be considered for prompt injection. Rejected records remain in the store until stale pruning so the same bad lesson is not immediately re-proposed from the same evidence.
207
282
 
283
+ Task-bank promotion is a reviewed export from transient execution memory. The
284
+ `/memory` overlay offers repo and global proposal actions only on selected
285
+ knowledge and procedural rows. Status remains private and cannot enter the
286
+ promotion service. The first global action arms a warning, and the second
287
+ action acknowledges the broader applicability. A successful action writes an
288
+ unapproved record and names the separate `memory approve` command required to
289
+ make it injectable.
290
+
291
+ The CLI consumes a version 2 `clio-task-memory` handoff snapshot. Omitting
292
+ `--entry` proposes every knowledge and procedural entry; repeating `--entry`
293
+ selects exact entry IDs. Version 2 snapshots carry source session, evidence,
294
+ runtime, agent, timestamps, and export-redaction facts. Version 1 snapshots
295
+ remain seedable but cannot be promoted because they do not carry source
296
+ session or evidence provenance.
297
+
298
+ Every promotion redacts secret-shaped values before `records.json` is written.
299
+ The durable provenance block records the source kind, session, selected entry,
300
+ entry class and timestamps, plus the replacement count and source field paths.
301
+ Promotion never approves its own output.
302
+
303
+ ### Explicit scope selection
304
+
305
+ Reviewed scope options are closed to four choices:
306
+
307
+ | Scope | Required selection | Validation |
308
+ | --- | --- | --- |
309
+ | `repo` | `--repository <canonical-absolute-path>` | The path must exist and already equal its canonical absolute identity. Symlink aliases and paths containing unresolved segments are rejected. |
310
+ | `global` | `--acknowledge-global` | The acknowledgement is separate from `--scope global`. |
311
+ | `runtime` | `--runtime <id>` | The ID must be valid and must occur in the source provenance. |
312
+ | `agent` | `--agent <id>` | The ID must be valid and must occur in the source provenance. |
313
+
314
+ The same options may be added to `memory propose --from-evidence`. With no
315
+ scope option, evidence proposals keep the existing inference order. An
316
+ explicit repository may differ from the repository that produced the
317
+ evidence, which supports a reviewed lesson about repository A learned while
318
+ working in repository B. Runtime and agent overrides may only select an exact
319
+ identity already recorded by the evidence. Global scope always requires its
320
+ own acknowledgement. No inference path widens an explicit choice.
321
+
208
322
  ---
209
323
 
210
324
  ## Prompt injection rules
@@ -215,7 +329,7 @@ Defaults:
215
329
 
216
330
  | Constraint | Default |
217
331
  | --- | --- |
218
- | Scopes | `global`, `repo` |
332
+ | Base scopes | `global`, `repo` |
219
333
  | Token budget | `400` estimated tokens |
220
334
  | Max records | `5` |
221
335
  | Required status | `approved: true` |
@@ -224,6 +338,12 @@ Defaults:
224
338
 
225
339
  Rendered memory lines always cite record ID, scope, lesson, and evidence IDs. The prompt tells the model not to extrapolate beyond cited findings.
226
340
 
341
+ Interactive main-agent sessions additionally admit records for the exact
342
+ active runtime. `clio-coder run --agent` admits records for the exact resolved
343
+ runtime and selected agent. Runtime and agent records use structured identity
344
+ fields; `appliesWhen` text cannot grant either applicability. Missing,
345
+ malformed, or different active identities exclude those records.
346
+
227
347
  ### Repository-scoped identity
228
348
 
229
349
  Repository memory is selected by an exact canonical absolute-path identity. The interactive orchestrator and `clio-coder run --agent` compute that identity from the active working directory; symlink aliases collapse to the same key. A repository move, a different Git worktree path, a subdirectory launch, a malformed identity, or a missing identity does not inherit another repository's memory. Global records are unaffected.
@@ -236,15 +356,27 @@ Every `scope: "repo"` record must carry:
236
356
 
237
357
  The structured `repository` field is the only applicability mechanism: store validation rejects repo records without it, and `appliesWhen` tokens never grant repository applicability. There is intentionally no automatic path rewrite for moved repositories or worktrees: a filesystem move produces a different identity and the record simply stops applying until it is re-scoped with new evidence.
238
358
 
359
+ Runtime and agent records follow the same fail-closed shape:
360
+
361
+ ```json
362
+ { "runtime": { "kind": "runtime", "key": "openai" } }
363
+ ```
364
+
365
+ ```json
366
+ { "agent": { "kind": "agent", "key": "coder" } }
367
+ ```
368
+
369
+ Only the field matching the record scope is present.
370
+
239
371
  ---
240
372
 
241
373
  ## Recommended workflow
242
374
 
243
375
  1. Build evidence from the run/session/eval that taught the lesson.
244
376
  2. Inspect the evidence and findings.
245
- 3. Propose memory from the evidence.
246
- 4. Review the proposed lesson for correctness and scope.
247
- 5. Approve only if it is durable and useful.
377
+ 3. Propose memory from the evidence, or promote selected public task memory from `/memory` or a redacted handoff.
378
+ 4. Review the proposed lesson, source provenance, redaction facts, and exact scope.
379
+ 5. Approve only if it is durable and useful under that scope.
248
380
  6. Reject incorrect or overbroad records.
249
381
  7. Prune stale records periodically.
250
382
 
package/docs/evolution.md CHANGED
@@ -1,7 +1,7 @@
1
1
  # Evolution and Change Manifests
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder uses change manifests to make harness changes reviewable, falsifiable, and rollback-friendly. CLIO stands for Context Layer for Input/Output, named for the Greek muse of history. A manifest is JSON, generated or checked with `clio-coder evolve manifest`, and should describe what changed, why, what evidence supports it, what could regress, how to validate it, and how to roll it back.
7
7
 
@@ -1,6 +1,6 @@
1
1
  # Exit Codes & Machine-Readable Output Contracts
2
2
 
3
- This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.3`.
3
+ This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.6`.
4
4
 
5
5
  Source implementations: `src/cli/` and `src/entry/`.
6
6
 
@@ -1,7 +1,7 @@
1
1
  # Extensions, Prompt Templates, Skills, and Share Archives
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder has lightweight community-oriented resource packaging. Extensions are filesystem bundles that contribute prompts and skills. Share archives are portable JSON files for moving project/user Clio resources between machines or collaborators. Themes are built into the engine and are no longer loaded from extensions.
7
7
 
@@ -251,7 +251,7 @@ Share archives are single JSON files:
251
251
  "formatVersion": 1,
252
252
  "manifest": {
253
253
  "format": "clio.share.v1",
254
- "clioVersion": "0.3.3",
254
+ "clioVersion": "0.3.6",
255
255
  "createdAt": "...",
256
256
  "files": []
257
257
  },
@@ -1,6 +1,6 @@
1
1
  # Fleet Dispatch
2
2
 
3
- > **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.3).
3
+ > **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.6).
4
4
 
5
5
  Clio Coder dispatches bounded worker agents. With a fleet configured, those
6
6
  workers run on remote machines over SSH while the orchestrator keeps every
@@ -77,7 +77,11 @@ briefing wins. Supplying both `task` and `tasks` fails instead of choosing one.
77
77
  After approval, execution consumes only the registry-owned resolved plan, so
78
78
  later mutation of raw arguments cannot change either field.
79
79
 
80
- Recipes may declare `budget: {toolCalls, readReserve, synthesis}`. `toolCalls` is the admitted-call phase boundary; the final `readReserve` slots accept canonical `read` plus the agent's granted mutation tools, so a writer can still deliver inside its own reserve; `synthesis: true` forces a text-only final round, while `false` stops after the admitted phase. `guardrails.workerToolCallCap` is transported separately as the ceiling on executed calls and always wins when lower. Native workers and Claude SDK enforce this policy. Claude Code and Antigravity reject explicit-budget recipes because their black-box loops cannot provide equivalent per-call mediation. Before launch, every admitted WorkerSpec v3 contains one concrete effective budget and a settings fingerprint, including custom recipes whose source omitted a budget.
80
+ Recipes declare a default with `budget: {toolCalls, readReserve, synthesis}`. They may also declare `maximum: {toolCalls, readReserve}` inside that object. A recipe without `maximum` is an exact pin, which preserves the fixed behavior of existing recipes. A ranged recipe admits the optional dispatch request `budget: {toolCalls, readReserve, retryRevision?}` only when the request is inside its maximum. `retryRevision` has the same two integer fields and preauthorizes the ceiling that a later automatic retry, bounded result-contract revision, or review revision may select. The loop guard raises a result-contract revision boundary only in this case. A phase without that ceiling cannot grow and retains the existing text-only repair behavior.
81
+
82
+ `toolCalls` is the admitted-call phase boundary. The final `readReserve` slots accept canonical `read` plus the agent's granted mutation tools, so a writer can still deliver inside its own reserve. Admission requires integers and `0 <= readReserve < toolCalls` for every declared phase. `synthesis: true` forces a text-only final round, while `false` stops after the admitted phase. `guardrails.workerToolCallCap` remains the operator-controlled lifetime ceiling and always wins when lower. A default may be clamped by a lower operator cap so default callers retain their prior behavior; an explicit request outside the operator cap is denied.
83
+
84
+ Admission computes one immutable envelope with the recipe policy, invocation request, effective worker budget, and every clamp or escalation reason. Native workers and Claude SDK enforce the effective budget. Claude Code, Antigravity, and ACP delegation reject invocation envelopes because their black-box loops cannot provide equivalent per-call mediation. Before launch, every admitted WorkerSpec v3 still contains one concrete effective budget and a settings fingerprint. The envelope provenance is sealed in the run ledger and receipt and appears in monitor, fleet status, and the live fleet card.
81
85
 
82
86
  ## Node setup
83
87
 
@@ -579,6 +583,19 @@ basis unknown/not applicable). A read-only Scout can therefore report `receipt_i
579
583
  bounded `project_context` provenance are also rendered independently; neither
580
584
  hash substitutes for the other.
581
585
 
586
+ The canonical terminology for these facts is the six-axis trust status in
587
+ [`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
588
+ Receipt integrity projects onto artifact integrity; receipt verification,
589
+ typed quality, and validation grounding project onto validation grounding;
590
+ gate decisions project onto independent review; briefing and project context
591
+ project onto context provenance; and `autonomyEnforcement` projects onto
592
+ autonomy enforcement. A receipt does not contain independent-review or
593
+ completion-evidence outcomes merely because it is sealed. Those axes remain
594
+ `absent` until an authenticated gate artifact or finish assessment is composed.
595
+ In particular, verified integrity cannot validate claims, known provenance
596
+ cannot establish correctness, and a review verdict cannot establish
597
+ authorship.
598
+
582
599
  Gate references point backward: a reviewer references the builder it
583
600
  reviewed, a revise builder references the reviewer whose findings it
584
601
  received, and a judge references every candidate. Because a worker receipt
@@ -645,6 +662,27 @@ hard block.
645
662
  renders `local`), gate badges (`gate reviewer c2`), reroute badges, live
646
663
  tool activity (names only; arguments never cross the worker stdout seam),
647
664
  and a per-worker context meter.
665
+ - `Enter` on the selected Fleet Runs row opens its worker detail: the phase,
666
+ the running call with a redacted action descriptor (`bash running npm
667
+ test`), and the bounded tail of the worker's own prose. The default list
668
+ stays compact, so a fan-out of scouts costs one card each until an operator
669
+ opens one. Detail follows the cursor rather than pinning to a run.
670
+ - The board and the transcript worker block read one projection
671
+ (`src/interactive/worker-progress.ts`), so they cannot disagree about what a
672
+ worker is saying or touching. It keeps 40 lines and 4096 bytes of tail, 8
673
+ distinct tool names, 4 recent actions, and accepts 16 KB of delta bytes per
674
+ 250 ms; what the bounds refuse is counted and named on the card beside the
675
+ `/view dispatch:<runId>` deep link.
676
+ - Action descriptors are composed where the arguments are trusted: the tool
677
+ registry's admission path, the Claude tool mapper, and the ACP update
678
+ mapper. Each reads a fixed verb vocabulary and a fixed argument-field
679
+ allowlist, scrubs credentials, strips escape sequences, and bounds the
680
+ result to 64 characters before it crosses the worker stdout seam. Raw
681
+ argument objects never cross at all.
682
+ - Reasoning content is never displayed. The detail may name a `thinking`
683
+ phase and the usage facts the card already carries, never the text.
684
+ - Settlement replaces the provisional tail with the sealed receipt's answer;
685
+ a run whose receipt cannot be read keeps its own last durable message.
648
686
  - The context meter renders the worker's last-message context occupancy
649
687
  against the model's context window: healthy below 80 percent, warn from 80,
650
688
  critical from 95.
@@ -654,6 +692,7 @@ hard block.
654
692
  - The monitor tool reports the node and reroute lineage on `status`, `list`,
655
693
  and `collect`.
656
694
  - `clio-coder fleet status [--json]` shows the durable ledger view cross-process.
695
+ - A worker permission escalation uses the `Worker escalation` consequence tier in operator presentation. The tier names the worker agent and run and describes where the one-shot answer returns. It does not approve the request, change the worker's inherited autonomy, or weaken the safety net; the existing worker escalation protocol remains the only resolution path.
657
696
 
658
697
  ## Speculation observer
659
698
 
@@ -682,11 +721,13 @@ After `npm run build`, an operator with a configured model target can run the
682
721
  single-turn, read-only fleet lifecycle check explicitly:
683
722
 
684
723
  ```bash
685
- CLIO_CODER_LIVE_EVAL=1 npm run test:live-eval:fleet-dispatch
724
+ npm run live:fleet-dispatch -- --target <id> [--model <wireId>] [--thinking medium]
686
725
  ```
687
726
 
688
- It is not part of deterministic CI. The script copies the repository into an
689
- isolated committed workspace, sandboxes all Clio config/state/data/cache,
690
- exercises Scout, bounded spot-checking, detached Debugger briefing, steering,
691
- wait, and collect, and fails if any workspace content changes. Failures retain
692
- their isolated artifacts for diagnosis.
727
+ It is not part of deterministic CI. The driver
728
+ (`benchmarks/internal/live-fleet-dispatch.ts`) copies the repository into a
729
+ committed temporary workspace, sandboxes all Clio config, state, data, and
730
+ cache under a scratch home holding only the chosen target, exercises Scout,
731
+ bounded spot-checking, detached Debugger briefing, steering, wait, and
732
+ collect, and fails if any workspace content changes. A failed run retains its
733
+ scratch tree for diagnosis.
package/docs/glossary.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # Clio Coder Glossary
2
2
 
3
- This document defines core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
3
+ This document defines the 45 core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
4
4
 
5
5
  ---
6
6
 
@@ -165,3 +165,23 @@ This document defines core architectural concepts and terminology used throughou
165
165
  ### 40. Delegate
166
166
  - **Definition**: Another coding agent Clio drives over ACP stdio as if it were a worker, configured under `delegation.agents` and invoked with `/delegate`. A delegate is a foreign harness, not a model target.
167
167
  - **Owning Type**: `DelegationAgentConfig` in `src/core/defaults.ts`.
168
+
169
+ ### 41. Working Set
170
+ - **Definition**: The part of the session ledger the model receives on the next request. It is the ledger with the current eviction projection applied, and it exists only in memory; the ledger itself is never narrowed. Not to be confused with the context ledger, which is the accounting of how the window is spent.
171
+ - **Owning Type**: `WorkingSetView` in `src/domains/context/working-set/contract.ts`.
172
+
173
+ ### 42. Projection
174
+ - **Definition**: The pure, idempotent transform from ledger entries to the entries the replay builder hands the model. It substitutes markers for evicted bodies and drops thinking from closed turns, returning unaffected entries by reference. Nothing about it is persisted.
175
+ - **Owning Type**: `projectWorkingSet` in `src/domains/context/working-set/project.ts`.
176
+
177
+ ### 43. Eviction
178
+ - **Definition**: The decision that a tool-result body or an assistant turn's thinking leaves the working set, recorded as an append-only ledger entry with a typed reason. It removes nothing: the original entry stays in the ledger and stays visible in the transcript, `/resume`, `/fork`, and the HTML export.
179
+ - **Owning Type**: `ContextEvictionEntry` in `src/domains/session/entries.ts`.
180
+
181
+ ### 44. Recall
182
+ - **Definition**: Readmitting an evicted body by ref, through `context(scope="recall", ref=...)` for the model or `/context recall <ref>` for the operator. A recall does not un-evict: the marker stays where it was so the provider prefix cache is untouched, and repeated recalls of one ref are the churn signal.
183
+ - **Owning Type**: `ContextRecallEntry` in `src/domains/session/entries.ts`; resolution in `resolveRecall` in `src/domains/context/working-set/recall.ts`.
184
+
185
+ ### 45. Marker
186
+ - **Definition**: The byte-stable one-line stub the projection renders in place of an evicted body, naming the ref, the reason, the tool, the size, and the exact recall call. It carries no timestamp and no counter, because a marker whose bytes drifted between renders would cold-start the prefix cache on a turn that evicted nothing new.
187
+ - **Owning Type**: `renderMarker` in `src/domains/context/working-set/marker.ts`.
@@ -3,7 +3,7 @@
3
3
  Clio Coder is designed to be self-contained and platform-compliant. This document outlines the default directory paths, file purposes, permission levels, and lifecycle commands (`install`, `reset`, `upgrade`, and `uninstall`). Clio Coder installs from npm as `@iowarp/clio-coder` (`npm install -g @iowarp/clio-coder`, published since v0.3.0) or from a source checkout with a deterministic local symlink; the CLI classifies both install kinds and `clio-coder upgrade` handles each.
4
4
 
5
5
  > [!TIP]
6
- > **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.3). You can open it directly in any web browser to view details dynamically.
6
+ > **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.6). You can open it directly in any web browser to view details dynamically.
7
7
 
8
8
  ---
9
9
 
@@ -229,7 +229,7 @@ Upgrading from 0.3.1 to 0.3.3 is automated:
229
229
  clio-coder upgrade
230
230
  ```
231
231
 
232
- Key lifecycle and operational updates in v0.3.3:
232
+ Key lifecycle and operational updates in v0.3.6:
233
233
  - Upgraded the underlying engine SDK libraries to 0.84.0 with signal-aware OAuth cancellation.
234
234
  - Hardened migration resilience: damaged `credentials.yaml` files no longer block upgrades when no renames are needed (#121); `--skip-migrations` is available as a recovery override.
235
235
  - Fullscreen TUI mode (`terminal.tuiMode`, `terminal.fullscreenScrollbar`) is available via Settings → Terminal (restart required). Adaptive presentation pacing is the live `terminal.smoothStreaming` setting; 0.3.3 defaults it to `off`, with conservative `auto` and explicit `on` available from the same section.
@@ -1,7 +1,7 @@
1
1
  # Middleware and Component Registry
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder has two related but separate surfaces:
7
7
 
@@ -114,7 +114,24 @@ Middleware hook budgets are phase-aware through `DEFAULT_MIDDLEWARE_HOOK_BUDGETS
114
114
 
115
115
  Per-phase budgets can be overridden via `CLIO_CODER_HOOK_BUDGET_<PHASE>_MS` or global `CLIO_CODER_HOOK_BUDGET_MS`. Warmup grace exempts initial calls (`DEFAULT_HOOK_BUDGET_WARMUP_CALLS = 1`), and steady-state warnings trigger when at least 3 of the last 5 post-warmup calls exceed budget (`DEFAULT_HOOK_BUDGET_WINDOW = 5`, `DEFAULT_HOOK_BUDGET_THRESHOLD = 3`). Overruns are reported but do not abort the turn. The orchestrator and workers share the middleware contract, but worker guard state is process-local.
116
116
 
117
- Middleware reminders are visible request text, not hidden prompt state. `turn_start` reminders flush into the same accepted request; `turn_end` reminders flush once on the next request. The built-in stalled-turn rule can request one automatic continuation for a user prompt, then stops rather than looping forever.
117
+ Middleware reminders are visible request text, not hidden prompt state. `turn_start` reminders flush into the same accepted request; `turn_end` reminders flush once on the next request. A `request_continuation` from any producer is capped at one automatic continuation per user prompt; a second producer in the same prompt gets a footer notice that the nudge is spent, and the turn is handed back to the operator rather than looped.
118
+
119
+ ### Built-in registrations
120
+
121
+ These ship in every interactive session. Each is one bounded behavior with a visible reminder; none changes a tool policy or a safety verdict.
122
+
123
+ | Id | Hooks | What it does |
124
+ | --- | --- | --- |
125
+ | `nudge.stalled-turn` | `turn_end` | The one declarative rule. A turn that called no tools and ended on an announced action ("Next I will inspect `src/cli/index.ts`") is continued once with a reminder to perform it or say plainly that it is finished. Questions, "let me know", conditional offers ("if you want me to"), and completion statements are not announcements. |
126
+ | `observer.skills-reminder` | `turn_start`, `turn_end` | Once per session, on the first substantive turn, when installed or installable skills exist, injects one line teaching the suggestion protocol: list with `context(scope="skills")`, open the reply with `Suggested skill: /skill <name>` when one matches, then continue the task in the same turn. Only the operator loads a skill. At `turn_end`, a reply that made the suggestion and stopped with only listing calls behind it is continued once (#184): the suggestion is not the task. Greetings do not spend the session's one reminder; a resumed or forked session never gets one. |
127
+ | `observer.task-board-reminder` | `turn_start` | Once per session, when the operator's text literally enumerates three or more steps (`1)`, `2.`, `step 3:`, or three bulleted lines), injects one line asking for `tasks action="plan"` before the first edit. Prose that merely mentions numbers never counts. |
128
+ | `nudge.open-tasks` | `turn_end` | A settled work turn (one that called tools) that ends while the session task board still has pending or active tasks is continued once with the open list. Pure conversation turns, aborted or errored turns, and boards where every remaining task is blocked do not trigger. |
129
+ | `nudge.detached-dispatch` | `turn_end` | A settled turn that ends while a detached dispatch batch has every run terminal and uncollected is continued once, naming the ready batches; `monitor mode="collect"` clears it, including across resume. Batches with runs still in flight, and surfaces without `monitor`, do not trigger. |
130
+ | `nudge.read-only-exploration` | `after_tool`, `turn_end` | After nine or more read-only calls (`read`, `grep`, `find`, `ls`, `code_nav`, read-only shell) in one user turn without a successful Scout dispatch, injects one advisory to delegate broad reconnaissance to Scout. One advisory per user turn, and only on surfaces that have `dispatch`. |
131
+ | `rail.unbacked-worker-claim` | `after_tool`, `turn_end` | A reply that reports worker or Scout results in a turn with no `dispatch` call gets one warning that the claim is not backed by a receipt. A `[worker result]` note the operator shared is receipt-backed and exempt. No continuation: the operator decides. |
132
+ | `observer.memory-intervention` | `after_tool` | Every `memory.intervention.everyNTools` tool calls, asks a background model for a bounded reflection over the recent window and injects it as a reminder when it arrives. Governed by the `memory.intervention` settings block. |
133
+
134
+ Two coded controls sit beside the registrations rather than among them. `tool-choice-control` turns `require_tool` and `lock_tools` effects into the provider's tool-choice field for the next round: a required tool clears when that tool starts, a lock lasts until the next submitted turn and outranks later requirements. `hook-receipts` is the durable ring (200 entries, throttled to one write per two seconds) of user-defined hook executions that `clio-coder config inspect` reads.
118
135
 
119
136
  User-defined hook declarations load from three places: `<extensionRoot>/hooks.yaml`, `.clio-coder/hooks.yaml`, and `.clio-coder/hooks.local.yaml`. A hook can be `prompt`, `effect`, or `command`. Command hooks run an argv array without a shell, under the workspace with a timeout and bounded output, and every hook execution emits a receipt.
120
137
 
@@ -1,7 +1,7 @@
1
1
  # Model Catalog, Runtime Refresh, and Field Notes
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio Coder treats a selectable model as the intersection of three sources:
7
7
 
@@ -39,14 +39,12 @@ worker spec and receipt are written.
39
39
 
40
40
  ## Benchmarking Models
41
41
 
42
- Model and config benchmark adapters ship under [benchmarks/community/](../benchmarks/community/). These adapters (such as `bench:swe`, `bench:scicode`, and the fleet benchmark `bench:tb`) drive Clio through the CLI or `clio-coder eval`.
43
-
44
- For example, to run the fleet benchmark:
45
- ```sh
46
- npm run bench:tb -- --limit 3
47
- ```
48
-
49
- The benchmarks record context-window, thinking, sampling, weight quantization, and KV-cache settings so sweeps can be compared consistently.
42
+ The public benchmark adapters under [benchmarks/community/](../benchmarks/community/)
43
+ (SWE-bench Lite, Terminal-Bench, SciCode, HumanEval) drive Clio through
44
+ `clio-coder run --json` or `clio-coder eval run`, each taking `--target` and
45
+ `--model` from the configured targets. `benchmarks/README.md` has the
46
+ commands. The run manifests record the target profile (runtime, model,
47
+ thinking level) so sweeps can be compared consistently.
50
48
 
51
49
  ## What "sanctioned" means
52
50
 
@@ -1,7 +1,7 @@
1
1
  # Observability Viewer
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  `/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
7
7
 
@@ -69,7 +69,7 @@ Clio resolves directories under platform-specific XDG defaults (on Linux, these
69
69
  | **Dispatch outputs** | Logs and ledger records detailing worker execution. | `<stateDir>/runs.json` and `<stateDir>/receipts/<runId>.json` |
70
70
  | **Task ledgers** | Per-turn task-board goals, active runs, required validation evidence, and operator-task provenance when present. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
71
71
  | **Workspace outputs** | Latest successful `artifact`, `write`, or `edit` result for each normalized path on the active session branch. Missing files remain visible as durable recorded facts. | Recorded path beneath the session metadata `cwd` |
72
- | **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<toolCallId>.txt` |
72
+ | **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt` |
73
73
  | **Protected artifacts** | Validation-protected artifact metadata and its absolute artifact path when available. | Session ledger record plus the protected workspace path |
74
74
  | **Compaction** | Summaries of compacted history sessions. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
75
75
  | **Prompt manifests** | One validated record per prompt compile: `systemPromptHash`, previous hash, token estimate, thinking dial at compile time, per-section token estimates, and per-fragment content hashes. Identifies the exact compiled prompt and supports hash diffs without storing prompt text. Malformed records appear as an explicit read-error artifact. | `<stateDir>/sessions/<cwdHash>/<sessionId>/prompt-manifest.jsonl` |
@@ -159,8 +159,8 @@ The base provenance sets, steering, routing, quality, worker identity, and resul
159
159
  | `safety.toolTelemetry.ingestionErrors` | `number` | Current dispatch receipts | Malformed or lost frames, event-fold/source errors, and drain timeouts that make otherwise mediated telemetry incomplete | experimental |
160
160
  | `safety.toolTelemetry.unfinished` | `{ tool, count }[]` | Current dispatch receipts | Tool starts that had no matching finish when the receipt sealed | experimental |
161
161
  | `safety.toolTelemetry.workspaceMutationPossible` | `boolean` | Current dispatch receipts | Whether incomplete or unavailable telemetry could conceal a shared-workspace mutation; retry admission fails closed when true | experimental |
162
- | `autonomyEnforcement.grade` | `string` | Always in v0.3.3 | The autonomy grade level enforced for the run | experimental |
163
- | `autonomyEnforcement.autonomy` | `string` | Always in v0.3.3 | The effective autonomy level name (e.g. auto-edit, suggest, read-only, full-auto) | experimental |
162
+ | `autonomyEnforcement.grade` | `string` | Always in v0.3.6 | The autonomy grade level enforced for the run | experimental |
163
+ | `autonomyEnforcement.autonomy` | `string` | Always in v0.3.6 | The effective autonomy level name (e.g. auto-edit, suggest, read-only, full-auto) | experimental |
164
164
  | `autonomyEnforcement.externalMode` | `string` | When running external worker | The execution mode of the external worker runtime | experimental |
165
165
  | `autonomyEnforcement.dangerousBypass` | `boolean` | When running external worker | Whether a safety bypass was explicitly activated | experimental |
166
166
  | `validationGrounding.claimed` | `number` | Validation grounding evaluated | Count of validations claimed by worker | experimental |
@@ -1,6 +1,6 @@
1
1
  # Proactive task memory
2
2
 
3
- > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.3).
3
+ > **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.6).
4
4
 
5
5
  Clio's proactive task memory protects long-running work from behavioral state
6
6
  decay: a requirement, environment fact, failed attempt, or diagnosis can still
@@ -133,7 +133,7 @@ it runs detached from it.
133
133
  | Interval | After `memory.intervention.everyNTools` completed tools since the last prompted step; default 10. This is the nondeterministic/citation-gated path. |
134
134
  | Tool-error streak | Two consecutive error outcomes. A successful tool resets the streak. |
135
135
  | Loop signal | Reuses the orchestrator loop guard's verdict; it does not infer a second competing loop detector. |
136
- | Repeated failure | The rules tier records failed tool fingerprints and annotates the failing tool result once the same failure appears twice in the bounded trajectory. |
136
+ | Repeated failure | The rules tier records failed operation fingerprints and annotates the failing tool result once the same failure appears twice in the bounded trajectory. |
137
137
  | Post-compaction | The first turn start after compaction restores status and knowledge once, without a model call, because compaction is precisely where execution facts leave the active window. |
138
138
 
139
139
  ### Two delivery channels
@@ -145,11 +145,13 @@ repeated failure uses exactly one of them:
145
145
 
146
146
  - **Mid-turn annotation.** The second identical failure appends one cited
147
147
  `Memory:` advisory to that tool's own result, through the existing
148
- `annotate_tool_result` effect the loop guard already uses. The advisory digest
149
- takes the first line of the tool error that names a problem, falling back to
150
- the first line when no line names one. The model reads it on its very next round.
151
- This is spent once per fingerprint per turn and re-earned in a later turn, because
152
- the same command failing again after an operator turn is news again.
148
+ `annotate_tool_result` effect the loop guard already uses. The advisory uses
149
+ the canonical result-disposition digest when one is available. Older hook
150
+ producers fall back to the first tool-error line that names a problem. Every
151
+ digest is redacted and byte-capped before it reaches the task bank. The model
152
+ reads the advisory on its very next round. This is spent once per operation
153
+ fingerprint per turn and re-earned in a later turn, because the same command
154
+ failing again after an operator turn is news again.
153
155
  - **Next-turn reminder.** Post-compaction reactivation and any background-model
154
156
  reminder ride the `inject_reminder` buffer into the next submitted turn, inside
155
157
  the visible `<system-reminder>` block, and persist in the session ledger.
@@ -319,15 +321,23 @@ operation while leaving deterministic protection active.
319
321
  Measured on the shipped prompt against `google/gemma-4-26b-a4b-qat`, across ten
320
322
  live steps and forty controlled runs on the same route.
321
323
 
322
- The tier writes `update_status` reliably and `save_knowledge` rarely, and that is
323
- correct rather than broken. A trajectory step carries the tool name, a bounded
324
- call description, an outcome, and a result digest. On success the digest is an
325
- opaque result fingerprint, so a window of successful reads tells the model which
326
- files were touched and nothing about what is in them. There is no durable fact in
327
- that input, and a status line is the only faithful thing to write about it.
328
-
329
- Three candidate causes were ruled out by controlled runs that changed one
330
- variable at a time:
324
+ Earlier measurements found that the tier wrote `update_status` reliably and
325
+ `save_knowledge` rarely. At that time a successful trajectory step carried an
326
+ opaque result fingerprint, so a window of successful reads told the model which
327
+ files were touched and nothing about what was in them.
328
+
329
+ A current trajectory step keeps two fields with different jobs. The operation
330
+ fingerprint identifies repeated calls and remains derived only from the tool name
331
+ and arguments. The result digest is human-readable diagnostic content from the
332
+ canonical result-disposition projection, with explicit source provenance. Secret
333
+ redaction and a 240-byte cap apply before the digest reaches the task bank or the
334
+ background request. A metadata-only disposition contributes outcome facts and no
335
+ captured body. Results without a canonical disposition use a redacted deterministic
336
+ fallback, so older tool producers remain useful without gaining a second model
337
+ summarizer.
338
+
339
+ Three candidate causes were ruled out in the earlier implementation by
340
+ controlled runs that changed one variable at a time:
331
341
 
332
342
  - rewriting the prompt's second worked example to carry a `save_knowledge` moved
333
343
  nothing, and made the model emit no operations at all in four of five runs;