@iowarp/clio-coder 0.3.3 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/CHANGELOG.md +74 -0
  2. package/CONTRIBUTING.md +7 -7
  3. package/README.md +3 -3
  4. package/dist/{acp-P2AQILE2.js → acp-2BEHC4DL.js} +9 -8
  5. package/dist/{agents-72W3BI7I.js → agents-LNNFTM53.js} +29 -24
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5TWEIYDN.js → auth-KXXFI2VS.js} +14 -10
  8. package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
  9. package/dist/{chunk-YCWGATWI.js → chunk-24I7BN55.js} +2 -2
  10. package/dist/{chunk-EKMEHE4H.js → chunk-33YXPOE3.js} +2 -3
  11. package/dist/chunk-3BPUFZDL.js +37 -0
  12. package/dist/chunk-43AOLP7E.js +375 -0
  13. package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
  14. package/dist/{chunk-6SGHMWE3.js → chunk-5JGRAMKL.js} +5 -5
  15. package/dist/{chunk-V6RTAOC2.js → chunk-6US73PDB.js} +572 -51
  16. package/dist/{chunk-5UFT4SUX.js → chunk-6XXKFVSN.js} +3 -3
  17. package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
  18. package/dist/chunk-AOCYTWAV.js +449 -0
  19. package/dist/chunk-CFGTUFWB.js +67 -0
  20. package/dist/chunk-CJUB2JJ2.js +1478 -0
  21. package/dist/{chunk-FNTMWMX5.js → chunk-CKXWIANG.js} +14 -12
  22. package/dist/{chunk-PIWWS5BL.js → chunk-CYQKWTG3.js} +63 -78
  23. package/dist/{chunk-LZSJBIVT.js → chunk-DJVECN66.js} +271 -762
  24. package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
  25. package/dist/{chunk-ZWMF7253.js → chunk-E2ER4LJF.js} +304 -9
  26. package/dist/{chunk-STBPMHSX.js → chunk-EKY57CSP.js} +51 -84
  27. package/dist/{chunk-DUYJ5IO6.js → chunk-EYPA3EGJ.js} +12 -4
  28. package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
  29. package/dist/chunk-FYYLNIL5.js +313 -0
  30. package/dist/{chunk-OQ33BKR3.js → chunk-G7MUEIGA.js} +3 -60
  31. package/dist/chunk-GEYXPTRF.js +613 -0
  32. package/dist/chunk-GOXNB3AO.js +261 -0
  33. package/dist/{chunk-G4BMMOKF.js → chunk-HVDIIIQW.js} +2 -2
  34. package/dist/chunk-HWUFFB6L.js +83 -0
  35. package/dist/{chunk-4XUGQOHA.js → chunk-K7T3E2SR.js} +15 -8
  36. package/dist/chunk-K7VKOLQQ.js +15 -0
  37. package/dist/{chunk-UFIIWP2H.js → chunk-KHSFENX2.js} +8 -8
  38. package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
  39. package/dist/chunk-LCGCVYZ4.js +57 -0
  40. package/dist/chunk-LL4KHSZI.js +22 -0
  41. package/dist/{chunk-PAJK6MAQ.js → chunk-LYF7OHWH.js} +42 -15
  42. package/dist/{chunk-POHLU5DW.js → chunk-M6L6IDJG.js} +3 -3
  43. package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
  44. package/dist/{chunk-X4RCMKVQ.js → chunk-NDINPTJ4.js} +2 -2
  45. package/dist/{chunk-TZK7PACC.js → chunk-NILBFAPG.js} +14 -8
  46. package/dist/chunk-ODFEOB4F.js +1082 -0
  47. package/dist/{chunk-AGYYIBLL.js → chunk-OH3TOQTB.js} +6 -2
  48. package/dist/chunk-OZNBF4L3.js +23 -0
  49. package/dist/{chunk-DSELYM6W.js → chunk-PBTHKCPN.js} +30 -10
  50. package/dist/{verify-375KUB3Y.js → chunk-PCZJO5TI.js} +127 -42
  51. package/dist/{chunk-ED4KHGC3.js → chunk-PPAMZ32Z.js} +9 -2
  52. package/dist/{chunk-SRDMMSEP.js → chunk-QM3F2GKX.js} +1063 -1645
  53. package/dist/{chunk-X6IAEBZR.js → chunk-QNQHSOLF.js} +7 -7
  54. package/dist/{chunk-OC7FIQPC.js → chunk-R46L2BIR.js} +10 -7
  55. package/dist/{chunk-2TLUCQVG.js → chunk-RD5U66HV.js} +3 -3
  56. package/dist/{chunk-6N5PTWMY.js → chunk-RY3LY4J5.js} +50 -13
  57. package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
  58. package/dist/{chunk-OOJYHWRB.js → chunk-TSHXZTOQ.js} +6 -5
  59. package/dist/chunk-TZSKNMZG.js +434 -0
  60. package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
  61. package/dist/{chunk-VJWL6YS5.js → chunk-UUVG37B4.js} +2 -2
  62. package/dist/{chunk-COU2UHX6.js → chunk-VEZEGCGW.js} +170 -2
  63. package/dist/chunk-W6GROXXM.js +69 -0
  64. package/dist/{chunk-OAO4GE4M.js → chunk-WHGPSPT5.js} +2 -2
  65. package/dist/chunk-WHJYKASB.js +677 -0
  66. package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
  67. package/dist/{chunk-YHZX5GEU.js → chunk-XAKHZX5N.js} +2 -2
  68. package/dist/{chunk-TZTZS7QK.js → chunk-XE2VEJHX.js} +5 -3
  69. package/dist/{chunk-LM5TQCJZ.js → chunk-XF5N4U5A.js} +8 -7
  70. package/dist/{chunk-LWLEKMDQ.js → chunk-XXQNGV4M.js} +1073 -552
  71. package/dist/{chunk-KZWTDYJF.js → chunk-XYDYPRZI.js} +7 -7
  72. package/dist/chunk-ZGVHUX3M.js +66 -0
  73. package/dist/{chunk-LW6DSM3M.js → chunk-ZRGEBJ4T.js} +1192 -1119
  74. package/dist/{chunk-2DJ2KNFG.js → chunk-ZXF4XRKW.js} +202 -40
  75. package/dist/chunk-ZZMN5OM4.js +122 -0
  76. package/dist/cli/index.js +34 -30
  77. package/dist/{clio-JOU4FXVA.js → clio-M2KGYUFZ.js} +7 -6
  78. package/dist/{code-nav-7AX6FYE6.js → code-nav-GQNL7XA6.js} +8 -6
  79. package/dist/codewiki/build-worker.js +4 -4
  80. package/dist/{components-KELWS457.js → components-5TTYYX6G.js} +3 -3
  81. package/dist/{config-XCDVKR23.js → config-XUUYQIWO.js} +47 -35
  82. package/dist/{configure-4GAP54ZW.js → configure-IHJ7YOMV.js} +18 -15
  83. package/dist/{context-77FM5DV5.js → context-74JLXAWD.js} +18 -10
  84. package/dist/{context-4UOGGLQ5.js → context-75MIWW3U.js} +41 -29
  85. package/dist/{context-5VKGUVJJ.js → context-ZQ7SIFJV.js} +85 -9
  86. package/dist/{context-clear-XXJRLCJJ.js → context-clear-GYKWNUML.js} +41 -29
  87. package/dist/{context-index-BZ4UYMTC.js → context-index-SSR5ECNE.js} +3 -3
  88. package/dist/context-working-set-UX5KEP4J.js +1553 -0
  89. package/dist/{dispatch-runner-QPRDDBDX.js → dispatch-runner-GIJBHNFL.js} +47 -32
  90. package/dist/{docs-2C2LTVT2.js → docs-6FZSCG5B.js} +3 -3
  91. package/dist/{doctor-HR46URBJ.js → doctor-SVJ5BZCW.js} +12 -12
  92. package/dist/{eval-XSSNATB4.js → eval-CG6LLBLD.js} +54 -238
  93. package/dist/{evidence-6HG2PY2B.js → evidence-ZYFIEN42.js} +57 -28
  94. package/dist/{evolve-K7YU3NCY.js → evolve-QGEXEMDW.js} +36 -25
  95. package/dist/{extensions-QVDOHDGJ.js → extensions-ADGNCJJD.js} +3 -3
  96. package/dist/{fleet-VY3HHKN6.js → fleet-S5R4ZOQY.js} +73 -44
  97. package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-BHSNPBMH.js} +3 -3
  98. package/dist/{init-JYGXI3FK.js → init-5DRU55YR.js} +49 -37
  99. package/dist/memory-7YKKR6UC.js +467 -0
  100. package/dist/{models-I5QWSEOM.js → models-ZPOLRU2C.js} +24 -21
  101. package/dist/{monitor-GE4ID3IA.js → monitor-US5F5YGZ.js} +73 -46
  102. package/dist/{orchestrator-EM5MC3HM.js → orchestrator-E2AL4T5N.js} +1624 -1007
  103. package/dist/{paths-UXLN5YYZ.js → paths-E7KYAQWE.js} +3 -3
  104. package/dist/{reset-L2FQEE3E.js → reset-KZ652EK6.js} +6 -5
  105. package/dist/{run-ZU3QMZPZ.js → run-SRNBKDWD.js} +76 -54
  106. package/dist/{share-S5BZQC5I.js → share-CGZE33UP.js} +7 -6
  107. package/dist/{skills-X5VXCRNQ.js → skills-S2X4DLY5.js} +4 -4
  108. package/dist/{skills-eval-WKIHWTHR.js → skills-eval-W2GGIC4R.js} +40 -29
  109. package/dist/{targets-SNCPI2NR.js → targets-54SWINWB.js} +28 -23
  110. package/dist/{terminal-lease-BNAHVHBS.js → terminal-lease-SAIF2OGY.js} +6 -4
  111. package/dist/{uninstall-FZCQCDKC.js → uninstall-BVLWXKBT.js} +3 -3
  112. package/dist/{upgrade-JQHHPQ4K.js → upgrade-JKAR27XC.js} +20 -19
  113. package/dist/{usage-OR4O5SMZ.js → usage-MSAWCLX4.js} +79 -36
  114. package/dist/verifiers-NCBTHHN2.js +1220 -0
  115. package/dist/verify-X5HDROLA.js +25 -0
  116. package/dist/{wiki-generate-UEXP2ARI.js → wiki-generate-GUSOQ6ZP.js} +50 -37
  117. package/dist/worker/entry.js +90 -70
  118. package/dist/{workspace-G4ZWUIPR.js → workspace-ZJ6BFM3Q.js} +4 -4
  119. package/docs/README.md +8 -7
  120. package/docs/acp.md +1 -1
  121. package/docs/alcf-provider.md +1 -1
  122. package/docs/architecture.md +2 -2
  123. package/docs/artifact-placement.md +1 -2
  124. package/docs/artifact-versions.md +1 -1
  125. package/docs/built-in-agents.md +1 -1
  126. package/docs/capacity-and-scheduling.md +1 -1
  127. package/docs/commands-and-modes.md +60 -26
  128. package/docs/config-knobs-audit.md +1 -2
  129. package/docs/configuration-and-targets.md +26 -1
  130. package/docs/context-engine.md +67 -13
  131. package/docs/context-working-set.md +194 -0
  132. package/docs/development-pipeline.md +1 -1
  133. package/docs/documentation-coverage.md +6 -6
  134. package/docs/documentation-guide.md +7 -6
  135. package/docs/environment-variables.md +2 -1
  136. package/docs/eval-runner.md +1 -1
  137. package/docs/evals-internal.md +4 -32
  138. package/docs/evidence-and-memory.md +139 -7
  139. package/docs/evolution.md +1 -1
  140. package/docs/exit-codes-and-output.md +1 -1
  141. package/docs/extensions-and-sharing.md +2 -2
  142. package/docs/fleet-dispatch.md +49 -8
  143. package/docs/glossary.md +21 -1
  144. package/docs/installation-and-lifecycle.md +2 -2
  145. package/docs/middleware-and-components.md +19 -2
  146. package/docs/model-catalog.md +7 -9
  147. package/docs/observability.md +4 -4
  148. package/docs/proactive-memory.md +26 -16
  149. package/docs/prompt-envelope-and-tools.md +7 -5
  150. package/docs/provider-adapter-cookbook.md +1 -1
  151. package/docs/release-cut-checklist.md +43 -40
  152. package/docs/safety-model.md +49 -8
  153. package/docs/scientific-validation.md +21 -3
  154. package/docs/session-lifecycle.md +3 -3
  155. package/docs/skills-marketplace.md +1 -1
  156. package/docs/tool-usage.md +79 -12
  157. package/docs/trace-store.md +1 -1
  158. package/docs/troubleshooting.md +1 -1
  159. package/docs/tui-design.md +38 -4
  160. package/docs/worker-dispatch-mechanics.md +11 -1
  161. package/package.json +13 -13
  162. package/skills/meta/clio-test/SKILL.md +20 -17
  163. package/skills/meta/clio-test/evals.md +3 -3
  164. package/skills/meta/clio-test/references/harness.md +35 -6
  165. package/skills/meta/clio-test/references/test-map.md +20 -10
  166. package/skills/registry.yaml +2 -2
  167. package/skills/skill-marketplace.json +1 -1
  168. package/src/cli/agents.ts +2 -3
  169. package/src/cli/argv.ts +14 -1
  170. package/src/cli/context-working-set.ts +513 -0
  171. package/src/cli/context.ts +8 -0
  172. package/src/cli/evidence.ts +20 -2
  173. package/src/cli/fleet.ts +15 -0
  174. package/src/cli/index.ts +5 -1
  175. package/src/cli/memory.ts +272 -10
  176. package/src/cli/modes/json-stream.ts +2 -2
  177. package/src/cli/modes/print.ts +12 -1
  178. package/src/cli/run.ts +22 -2
  179. package/src/cli/targets.ts +12 -3
  180. package/src/cli/usage.ts +55 -7
  181. package/src/cli/verifiers.ts +325 -0
  182. package/src/core/bash-exec.ts +39 -14
  183. package/src/core/bus-events.ts +22 -4
  184. package/src/core/config.ts +54 -0
  185. package/src/core/defaults.ts +50 -3
  186. package/src/core/response-model-id.ts +134 -0
  187. package/src/core/toml.ts +62 -0
  188. package/src/core/verification-scripts.ts +6 -0
  189. package/src/core/workspace-files.ts +0 -1
  190. package/src/domains/agents/builtins/architect.md +1 -1
  191. package/src/domains/agents/builtins/verifier.md +3 -0
  192. package/src/domains/agents/catalog.ts +5 -4
  193. package/src/domains/agents/recipe.ts +54 -14
  194. package/src/domains/agents/result-contract.ts +7 -4
  195. package/src/domains/config/classify.ts +1 -0
  196. package/src/domains/context/bootstrap.ts +36 -27
  197. package/src/domains/context/project-metadata.ts +19 -63
  198. package/src/domains/context/prompt-context.ts +8 -0
  199. package/src/domains/context/working-set/contract.ts +161 -0
  200. package/src/domains/context/working-set/defaults.ts +28 -0
  201. package/src/domains/context/working-set/engine.ts +203 -0
  202. package/src/domains/context/working-set/fold.ts +62 -0
  203. package/src/domains/context/working-set/horizon.ts +38 -0
  204. package/src/domains/context/working-set/marker.ts +103 -0
  205. package/src/domains/context/working-set/path-index.ts +436 -0
  206. package/src/domains/context/working-set/payload.ts +152 -0
  207. package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
  208. package/src/domains/context/working-set/policies/index.ts +20 -0
  209. package/src/domains/context/working-set/policies/structural.ts +160 -0
  210. package/src/domains/context/working-set/project.ts +132 -0
  211. package/src/domains/context/working-set/protect.ts +109 -0
  212. package/src/domains/context/working-set/recall.ts +177 -0
  213. package/src/domains/context/working-set/replay/controls.ts +112 -0
  214. package/src/domains/context/working-set/replay/load-clio.ts +199 -0
  215. package/src/domains/context/working-set/replay/metrics.ts +185 -0
  216. package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
  217. package/src/domains/context/working-set/replay/report.ts +139 -0
  218. package/src/domains/context/working-set/replay/runner.ts +325 -0
  219. package/src/domains/context/working-set/replay/synthetic.ts +422 -0
  220. package/src/domains/context/working-set/replay/trace.ts +21 -0
  221. package/src/domains/context/working-set/visible.ts +54 -0
  222. package/src/domains/dispatch/budget-envelope.ts +396 -0
  223. package/src/domains/dispatch/contract.ts +2 -0
  224. package/src/domains/dispatch/extension.ts +81 -27
  225. package/src/domains/dispatch/orphan-recovery.ts +1 -0
  226. package/src/domains/dispatch/receipt-integrity.ts +4 -0
  227. package/src/domains/dispatch/state.ts +1 -0
  228. package/src/domains/dispatch/types.ts +10 -3
  229. package/src/domains/dispatch/validation.ts +14 -0
  230. package/src/domains/dispatch/worker-spawn.ts +14 -3
  231. package/src/domains/eval/metrics/evidence.ts +0 -116
  232. package/src/domains/eval/metrics/invariants.ts +1 -1
  233. package/src/domains/eval/runners/clio-run.ts +1 -10
  234. package/src/domains/eval/runners/external-command.ts +2 -29
  235. package/src/domains/eval/schema/suite.ts +0 -7
  236. package/src/domains/eval/suites/run.ts +1 -7
  237. package/src/domains/evidence/build.ts +112 -45
  238. package/src/domains/evidence/eval.ts +24 -7
  239. package/src/domains/evidence/index.ts +53 -0
  240. package/src/domains/evidence/ordering.ts +12 -0
  241. package/src/domains/evidence/run-trust.ts +221 -0
  242. package/src/domains/evidence/store.ts +46 -6
  243. package/src/domains/evidence/trust-status.ts +854 -0
  244. package/src/domains/evidence/types.ts +26 -0
  245. package/src/domains/memory/index.ts +22 -0
  246. package/src/domains/memory/operations.ts +58 -1
  247. package/src/domains/memory/promotion.ts +281 -0
  248. package/src/domains/memory/prompt-section.ts +25 -5
  249. package/src/domains/memory/proposal.ts +51 -7
  250. package/src/domains/memory/task-bank.ts +3 -2
  251. package/src/domains/memory/task-memory-handoff.ts +181 -24
  252. package/src/domains/memory/task-memory-policy.ts +3 -1
  253. package/src/domains/memory/types.ts +37 -0
  254. package/src/domains/memory/validate.ts +178 -0
  255. package/src/domains/middleware/memory-intervention.ts +38 -25
  256. package/src/domains/middleware/runtime.ts +6 -0
  257. package/src/domains/middleware/skills-reminder.ts +19 -4
  258. package/src/domains/middleware/stalled-turn.ts +208 -5
  259. package/src/domains/middleware/types.ts +10 -0
  260. package/src/domains/observability/contract.ts +6 -1
  261. package/src/domains/observability/cost.ts +20 -4
  262. package/src/domains/observability/extension.ts +2 -2
  263. package/src/domains/providers/index.ts +3 -0
  264. package/src/domains/providers/model-discovery.ts +9 -0
  265. package/src/domains/providers/runtime-resolution.ts +38 -1
  266. package/src/domains/providers/runtimes/common/probe-helpers.ts +97 -16
  267. package/src/domains/providers/types/context-window-slots.ts +18 -0
  268. package/src/domains/providers/types/runtime-descriptor.ts +3 -1
  269. package/src/domains/safety/autonomy.ts +1 -1
  270. package/src/domains/safety/call-target.ts +211 -14
  271. package/src/domains/safety/decision-presentation.ts +268 -0
  272. package/src/domains/safety/default-path-policy.ts +8 -0
  273. package/src/domains/safety/finish-contract.ts +4 -3
  274. package/src/domains/safety/policy-engine.ts +48 -6
  275. package/src/domains/safety/redaction.ts +73 -0
  276. package/src/domains/session/compaction/compact.ts +23 -1
  277. package/src/domains/session/compaction/cut-point.ts +2 -0
  278. package/src/domains/session/compaction/tokens.ts +16 -1
  279. package/src/domains/session/context-ledger.ts +12 -1
  280. package/src/domains/session/decision-board.ts +4 -0
  281. package/src/domains/session/entries.ts +110 -1
  282. package/src/domains/session/history.ts +68 -19
  283. package/src/domains/session/manager.ts +9 -2
  284. package/src/domains/session/migrations/index.ts +22 -3
  285. package/src/domains/session/usage.ts +24 -7
  286. package/src/engine/acp/event-mapper.ts +7 -0
  287. package/src/engine/acp/server.ts +32 -2
  288. package/src/engine/agent.ts +18 -1
  289. package/src/engine/apis/lmstudio.ts +25 -4
  290. package/src/engine/apis/openai-completions.ts +147 -22
  291. package/src/engine/claude/sdk-runtime.ts +8 -2
  292. package/src/engine/claude/tool-safety.ts +13 -0
  293. package/src/engine/loop-guard.ts +27 -3
  294. package/src/engine/session.ts +9 -3
  295. package/src/engine/worker-events.ts +4 -3
  296. package/src/engine/worker-runtime.ts +59 -54
  297. package/src/entry/orchestrator.ts +34 -5
  298. package/src/interactive/chat-loop-messages.ts +40 -6
  299. package/src/interactive/chat-loop.ts +13 -0
  300. package/src/interactive/chat-panel.ts +17 -1
  301. package/src/interactive/chat-renderer.ts +49 -24
  302. package/src/interactive/clio-editor.ts +44 -7
  303. package/src/interactive/context-meter.ts +10 -0
  304. package/src/interactive/context-overlay.ts +120 -7
  305. package/src/interactive/context-recall-command.ts +110 -0
  306. package/src/interactive/cost-overlay.ts +39 -8
  307. package/src/interactive/dispatch-board.ts +212 -35
  308. package/src/interactive/footer/widgets.ts +13 -0
  309. package/src/interactive/interactive-application.ts +6 -1
  310. package/src/interactive/interactive-input-runtime.ts +11 -1
  311. package/src/interactive/interactive-presentation.ts +11 -1
  312. package/src/interactive/interactive-slash-runtime.ts +37 -1
  313. package/src/interactive/memory-overlay.ts +89 -4
  314. package/src/interactive/model-session-replay.ts +21 -0
  315. package/src/interactive/overlay-ask-user-lifecycle.ts +1 -1
  316. package/src/interactive/overlay-frame.ts +5 -2
  317. package/src/interactive/overlay-general-openers.ts +46 -1
  318. package/src/interactive/overlay-key-routing.ts +41 -1
  319. package/src/interactive/overlay-lifecycle.ts +11 -4
  320. package/src/interactive/overlay-permission-lifecycle.ts +23 -8
  321. package/src/interactive/overlay-session-lifecycle.ts +8 -4
  322. package/src/interactive/overlay-transitions.ts +11 -0
  323. package/src/interactive/overlays/ask-user.ts +74 -30
  324. package/src/interactive/overlays/decisions.ts +3 -1
  325. package/src/interactive/permission-hint.ts +35 -0
  326. package/src/interactive/permission-overlay.ts +95 -45
  327. package/src/interactive/renderers/tool-execution.ts +37 -51
  328. package/src/interactive/session-last-turn.ts +8 -1
  329. package/src/interactive/session-transcript.ts +2 -2
  330. package/src/interactive/session-usage-reseed.ts +36 -10
  331. package/src/interactive/slash-commands.ts +31 -4
  332. package/src/interactive/status/summary.ts +5 -0
  333. package/src/interactive/status/types.ts +5 -0
  334. package/src/interactive/terminal-lease.ts +1 -0
  335. package/src/interactive/turn-context.ts +333 -110
  336. package/src/interactive/turn-middleware.ts +7 -6
  337. package/src/interactive/turn-runtime.ts +37 -8
  338. package/src/interactive/turn-state.ts +3 -0
  339. package/src/interactive/worker-progress.ts +440 -0
  340. package/src/interactive/worker-stream.ts +51 -110
  341. package/src/tools/agent-tools.ts +39 -7
  342. package/src/tools/ask-user.ts +21 -1
  343. package/src/tools/bash.ts +144 -82
  344. package/src/tools/builtin-tool-catalog.ts +11 -5
  345. package/src/tools/context/index.ts +107 -5
  346. package/src/tools/context/surface.ts +3 -2
  347. package/src/tools/core-bootstrap.ts +21 -0
  348. package/src/tools/dispatch-arguments.ts +8 -0
  349. package/src/tools/dispatch-event-text.ts +19 -0
  350. package/src/tools/dispatch-runner.ts +9 -7
  351. package/src/tools/dispatch.ts +24 -1
  352. package/src/tools/monitor.ts +43 -20
  353. package/src/tools/registry.ts +72 -10
  354. package/src/tools/result-disposition.ts +706 -0
  355. package/src/tools/result-shaping.ts +321 -20
  356. package/src/tools/safe-exec.ts +2 -0
  357. package/src/tools/verify/authoring.ts +1120 -0
  358. package/src/tools/verify/catalog.ts +346 -0
  359. package/src/tools/verify/index.ts +13 -3
  360. package/src/tools/verify/scripts.ts +135 -37
  361. package/src/tools/verify/surface.ts +9 -5
  362. package/src/tools/worker-evidence.ts +54 -12
  363. package/src/worker/spec-contract.ts +43 -3
  364. package/dist/chunk-J7CWMCQD.js +0 -255
  365. package/dist/chunk-T6YILFSB.js +0 -80
  366. package/dist/chunk-VAKQQHWR.js +0 -434
  367. package/dist/chunk-VPAYEGVX.js +0 -184
  368. package/dist/chunk-XBXAASKX.js +0 -18
  369. package/dist/memory-WFZMGYHX.js +0 -236
  370. package/src/domains/eval/metrics/chaos-stream.ts +0 -93
@@ -1,11 +1,11 @@
1
1
  # Tool Usage Reference
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
7
7
 
8
- In Clio Coder v0.3.3, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
8
+ In Clio Coder v0.3.6, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
9
9
 
10
10
  ## Observation envelope: truncation notices, offload, next hints, and the turn budget
11
11
 
@@ -21,7 +21,7 @@ Truncated text results append exactly one notice line:
21
21
 
22
22
  Segments that do not apply are omitted. `<total>` renders as `N+` when the search stopped early at its item limit, so the true total was never counted. `next:` is an exact argument fragment (for example `limit=200` or `offset=451`); re-issue the same call with that argument changed to continue.
23
23
 
24
- Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<toolCallId>.txt` and the notice's `full:` segment names the path. Read it with `read` using offset/limit. Tools offload only when the byte cap cut collected content; a bare item-limit truncation continues via `next` and does not offload. `read` never offloads, because the source file is directly re-addressable via `offset`.
24
+ Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<sha256 of the captured text>.txt` and the notice's `full:` segment names the path. Read it with `read` using offset/limit. Tools offload only when the byte cap cut collected content; a bare item-limit truncation continues via `next` and does not offload. `read` never offloads, because the source file is directly re-addressable via `offset`.
25
25
 
26
26
  JSON-format results (code_nav, context scope=docs/workspace) never get an appended notice. An oversize JSON payload is replaced whole by the parseable stub `{"error":"result exceeded <cap>","offloadPath":"...","next":"..."}` so the model never receives JSON cut mid-document. Empty results are also valid JSON with empty arrays and `next` populated.
27
27
 
@@ -102,17 +102,24 @@ Arguments:
102
102
  - `command` (required).
103
103
  - `cwd` (optional). Working directory; resolved against the session workspace and rejected when it escapes it. The safety net blocks an escaping cwd at admission, and the tool enforces the same rule itself.
104
104
  - `timeout_ms` (optional). Default 300000 (5 minutes).
105
+ - `output_policy` (optional). Canonical model-context disposition: `full`, `bounded`, `summary`, or `metadata-only`. Omission is exactly `bounded`.
105
106
 
106
107
  Workspace containment: commands whose filesystem targets resolve outside the session workspace escalate to `system_modify` and ask for one-shot confirmation at every autonomy level (headless runs deny asks). Recognized targets are shell redirects, `tee`/`mkdir`/`touch` path operands, `cp`/`mv`/`ln` destinations, in-place `sed -i` operands, and any `cd`/`pushd` whose directory leaves the workspace, since a `cd` outside re-bases every relative path that follows it. Inside-workspace equivalents stay plain `execute` with no new prompts.
107
108
 
108
- Output shaping is tail-biased: the display keeps the LAST 16KB / 2000 lines, because the failing assertion, compiler error, and exit summary live at the end. Before truncating, the full output is spilled to the per-session scratch file and the appended note names the path; read it with offset/limit. A command producing more than 16MB of output is stopped with an error. A timeout or nonzero exit returns the shaped output plus a status line (`bash: command timed out after <ms>ms`, `bash: command failed (exit N)`).
109
+ The default `bounded` policy keeps a tail-biased model excerpt under the 16KB result budget, because the failing assertion, compiler error, and exit summary usually live at the end. `summary` is useful for noisy builds and test runs: code deterministically selects a bounded head, tail, and error-like lines, applies Clio's repository secret redactor, and records the source hash and algorithm in summary provenance. `metadata-only` is appropriate when the model needs only outcome and termination facts; stdout and stderr stay out of model context while the operator presentation, retained byte size, and retrieval path remain available. `full` is for output known to be small. It is admitted only when the complete captured result and its facts fit the bounded result/context budget; otherwise the result explicitly records a typed downgrade to tail-biased `bounded` and provides retrieval. Do not use `full` as the routine default.
109
110
 
110
- Reach for bash for builds, git, package managers, and anything without a dedicated tool. Prefer the dedicated tools over their shell equivalents: grep/find/read/ls get envelope truncation, exact continuation hints, and the shared ignore policy that `cat`, shell `grep`, and shell `find` do not. Prefer `verify` over bash for declared package.json verification scripts, since verify produces typed evidence.
111
+ Presentation is independent from model context. The operator-facing display remains folded and tail-biased under every policy. When the display or selected context omits captured content, the terminal result writes one per-session scratch artifact and names it in the result. Live updates use the selected policy, remain bounded, and never write per-update artifacts. Every terminal result records requested and applied context modes, captured/displayed/context bytes, truncation or downgrade state, and any offload path. Exit code, signal, timeout, abort, and output-cap facts survive every policy. Scratch retrieval may contain the raw retained output; the deterministic `summary` projection is the redacted surface.
112
+
113
+ A command producing more than 16MB of combined output is stopped with an error. UTF-8 decoding spans process chunks, and a code point split by the hard byte cap is discarded rather than replaced with an invalid character. Raw NUL bytes are removed from model context under every policy, which leaves multi-byte code points and ANSI escape sequences whole; the operator presentation and the scratch artifact keep the captured bytes, and the result still records the omission and its retrieval path. A timeout, abort, output cap, or nonzero exit preserves captured diagnostics and appends a status line such as `bash: command timed out after <ms>ms` or `bash: command failed (exit N)` before canonical shaping.
114
+
115
+ Reach for bash for builds, git, package managers, and anything without a dedicated tool. Prefer the dedicated tools over their shell equivalents: grep/find/read/ls get envelope truncation, exact continuation hints, and the shared ignore policy that `cat`, shell `grep`, and shell `find` do not. Prefer `verify` over bash for declared package scripts and project-catalog entries, since verify produces typed evidence.
111
116
 
112
117
  ```text
113
118
  bash(command="git status --short")
114
119
  bash(command="git log --oneline -10")
115
120
  bash(command="npm run build", timeout_ms=600000)
121
+ bash(command="npm run test", timeout_ms=600000, output_policy="summary")
122
+ bash(command="make artifact", output_policy="metadata-only")
116
123
  ```
117
124
 
118
125
  ## grep: search file contents with ripgrep
@@ -266,27 +273,87 @@ dispatch(tasks=["Refactor step 1", "Refactor step 2"], mode="sequential", timeou
266
273
 
267
274
  ## verify: run declared verification checks
268
275
 
269
- One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/frontend.ts`.
276
+ One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/catalog.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/authoring.ts`, `src/tools/verify/frontend.ts`.
270
277
 
271
278
  Arguments:
272
279
 
273
- - `check` (optional). A declared package.json script name or `"frontend"`. Omit to list available checks.
280
+ - `check` (optional). A declared project-catalog ID, package.json script name, or `"frontend"`. Omit to list available checks.
274
281
  - `path` (check=frontend). Artifact file under the workspace root.
275
- - `args` (optional). Extra arguments passed to the script after `--`. A JSON-string array is tolerated and parsed.
282
+ - `args` (package scripts only). Extra arguments passed after `--`. A JSON-string array is tolerated and parsed. Project-catalog checks ignore this field.
276
283
  - `browser` (check=frontend). `auto` (default), `required`, or `off`.
277
- - `cwd` (optional). Working directory.
278
- - `timeout_ms` (optional). Default 120000.
284
+ - `cwd` (package scripts only). Package working directory. Project catalogs are always discovered at the session workspace root, and a project check uses its declared `cwd`.
285
+ - `timeout_ms` (package scripts and frontend only). Default 120000. A project check uses its declared `timeoutMs`.
286
+ - `max_output_bytes` (package scripts and frontend only). Default 600000. Project checks retain the safe-exec default cap.
287
+
288
+ `verify()` lists checks grouped as `package.json` and `.clio-coder/verifiers.yaml`. Both providers project through the same canonical metadata: `{id, description, command, cwd, timeoutMs, tags, source}`. Package scripts must match the verification family `test*/lint*/build*/typecheck*/check*/format*/ci*` (a family prefix, optionally followed by `:`, `.`, or `-` and a suffix, e.g. `test:unit`). `verify(check="typecheck")` runs `npm run typecheck` through the safe-exec spine with no shell. A package script name outside the family is rejected with a pointer to run it through bash.
289
+
290
+ ### Project verifier catalog
291
+
292
+ Projects may commit a versioned executable catalog at `.clio-coder/verifiers.yaml`:
293
+
294
+ ```yaml
295
+ version: 1
296
+ checks:
297
+ - id: rust-workspace
298
+ description: Run the Rust workspace tests
299
+ command: [cargo, test, --workspace]
300
+ cwd: .
301
+ timeoutMs: 600000
302
+ tags: [rust, test]
303
+ ```
304
+
305
+ Version 1 is strict. Every root and check field shown above is required, unknown fields fail, and duplicate IDs fail. A project ID uses lowercase letters, digits, `.`, `_`, `:`, or `-`, begins with a letter or digit, and is at most 64 UTF-8 bytes. `frontend` is reserved. Descriptions are trimmed single-line text capped at 512 bytes. `command` is a nonempty argv array with at most 64 entries and 4096 bytes per entry. A shell command string is invalid, and explicit shell executables such as `sh`, `bash`, `pwsh`, and `cmd` are rejected. `cwd` is a repository-relative existing directory capped at 512 bytes; absolute paths, `..` escapes, and symbolic-link escapes fail. `timeoutMs` is a positive integer capped at 900000. A check may carry at most 16 distinct lowercase tags of at most 32 bytes each. The whole file is capped at 262144 bytes and may contain at most 128 checks. YAML aliases are disabled.
306
+
307
+ Provider IDs share one namespace. If a catalog ID collides with a discovered package script, listing and execution fail and identify both source files. Catalog parsing also fails closed before any package or project check runs.
308
+
309
+ `verify(check="rust-workspace")` spawns exactly `cargo` with `test` and `--workspace`; it does not interpolate model text or invoke a shell. Model-supplied `args`, `cwd`, `timeout_ms`, `max_output_bytes`, or undeclared environment fields cannot widen or replace the catalog entry. Safe execution passes only Clio's small environment allowlist, applies cancellation and the declared timeout, and shapes output at the standard 600000-byte cap. Execution details retain the compatible command string plus exact `argv`, `cwd`, `exitCode`, `durationMs`, `aborted`, `timedOut`, and `outputCapped` evidence, along with the check's declared source, command, cwd, timeout, description, and tags.
310
+
311
+ ### Guided catalog authoring
312
+
313
+ An empty `verify()` result points to `clio-coder verifiers author`. The authoring command inspects only command-bearing files at the workspace root:
314
+
315
+ - verification-family package scripts, projected exactly as `npm run <script>` and shown as already active rather than duplicated into the catalog;
316
+ - `Cargo.toml`, projected to Cargo's package or workspace test vector;
317
+ - visible build and test entries in `CMakePresets.json`, projected to the corresponding `cmake --build --preset` or `ctest --preset` vector;
318
+ - declared Python runners in `pyproject.toml`, `pytest.ini`, `tox.ini`, `noxfile.py`, or the pytest section of `setup.cfg`;
319
+ - a module directive in `go.mod`, projected to `go test ./...`;
320
+ - top-level `validators` entries in the documented YAML scientific-validation files.
321
+
322
+ Every proposal records its source path and location and labels the command origin as `project-declared` or `toolchain-defined`. Project-declared examples include a package script, a Python entry point, and an exact validation-contract command. Toolchain-defined examples include Cargo's test command, a named CMake preset invocation, a configured Python runner, and Go's module test command. Validation command strings are converted to argv only when their quoting is complete and they contain no shell operators, expansion, redirection, or environment assignment. Ambiguous entries and `VALIDATION.md` prose receive a manual-entry diagnostic. Directory names such as `build`, `tests`, `python`, or `cargo` never imply a command.
323
+
324
+ `discover` and every mutating command first print an authority preview. Each check shows the destination or active source path, source provenance, exact JSON argv vector, repository-relative cwd, timeout, tags, and effective execution authority. Preview and discovery do not create `.clio-coder`, write a file, or run a check. A mutating command without `--yes` ends after the preview. Repeating the reviewed command with `--yes` is the explicit write decision; the serialized YAML must pass the production catalog parser before the atomic write is reachable.
325
+
326
+ ```text
327
+ clio-coder verifiers discover
328
+ clio-coder verifiers author
329
+ clio-coder verifiers author --exclude cmake-build-debug --rename go-test=go-suite
330
+ clio-coder verifiers author --dry-run go-suite --yes
331
+ clio-coder verifiers validate
332
+ ```
333
+
334
+ `validate` reads the committed file with the same parser used by `verify()`. `dry-run <id>` is an explicit request to execute one admitted check through the production `verify` path. `author --dry-run <id> --yes` writes only after confirmation and starts the selected dry run only after the write is accepted by production discovery.
335
+
336
+ Later changes use the same preview and confirmation boundary. `edit` preserves the ID unless `rename` is requested. Renames and additions reject collisions with catalog IDs and active package-script IDs. Removals state that the deleted command will no longer be executable through catalog authority. Generated IDs are stable for a stable ordered signal set; a collision receives the first available deterministic `-2`, `-3`, and later suffix.
337
+
338
+ ```text
339
+ clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]'
340
+ clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]' --tags scientific,netcdf --yes
341
+ clio-coder verifiers edit validate-grid --timeout-ms 300000
342
+ clio-coder verifiers rename validate-grid validate-regional-grid --yes
343
+ clio-coder verifiers remove validate-regional-grid --yes
344
+ ```
279
345
 
280
- `verify()` with no check lists declared checks grouped by source; today the only source is package.json scripts whose names match the verification family `test*/lint*/build*/typecheck*/check*/format*/ci*` (a family prefix, optionally followed by `:`, `.`, or `-` and a suffix, e.g. `test:unit`). `verify(check="typecheck")` runs `npm run typecheck` through the safe-exec spine with no shell; output is capped at 600000 bytes and `details = {command, cwd, exitCode, durationMs, timedOut, outputCapped}`. A script name outside the family is rejected with a pointer to run it through bash.
346
+ The `add` command is the explicit path for an unsupported or ambiguous project. `--command` must be a JSON argv array, so manual entry still cannot turn a shell command string into executable catalog authority.
281
347
 
282
348
  `verify(check="frontend", path=<file>)` validates an HTML, CSS, or JavaScript artifact without shell access. The path must stay inside the workspace root and end in `.html`, `.htm`, `.css`, `.js`, `.mjs`, or `.cjs`. Checks per type: HTML tag balance (comment-aware, HTML5 optional end tags honored), inline and referenced script syntax (classic scripts parsed in-process, modules via `node --check`), inline and linked CSS brace/string/comment balance, local script and stylesheet references resolved and existence-checked (external and root-relative references are skipped), and an optional headless browser load. `browser="auto"` warns when no chromium/chrome/edge executable is on PATH, `"required"` fails, `"off"` skips. Each check reports pass, warn, fail, or skip; any fail makes the whole result an error. `details = {action: "verify", check: "frontend", path, browserMode, status, checks}`.
283
349
 
284
- Prefer verify over bash for the verification family: the typed result feeds the finish contract as validation evidence.
350
+ Prefer verify over bash for the verification family and project catalog: the typed result feeds the finish contract as validation evidence.
285
351
 
286
352
  ```text
287
353
  verify()
288
354
  verify(check="typecheck")
289
355
  verify(check="test", args=["tests/contracts/dispatch.test.ts"])
356
+ verify(check="rust-workspace")
290
357
  verify(check="frontend", path="site/index.html", browser="off")
291
358
  ```
292
359
 
@@ -1,7 +1,7 @@
1
1
  # Trace store contract
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  Clio's trace database is a rebuildable, queryable mirror. Receipts, session
7
7
  ledgers, gate artifacts, and evidence remain the source of truth. Removing
@@ -1,6 +1,6 @@
1
1
  # Troubleshooting & Error Remediation
2
2
 
3
- This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.3`.
3
+ This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.6`.
4
4
 
5
5
  ---
6
6
 
@@ -1,7 +1,7 @@
1
1
  # Clio TUI Design System
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  This document is the reference specification for the Clio Coder TUI visual layout, styling, and behavior. It describes color semantics, the glyph vocabulary, structural recipes, and state choreography for all surfaces under [src/interactive/](../src/interactive/).
7
7
 
@@ -36,7 +36,7 @@ All color styling is defined in [src/interactive/theme/tokens.ts](../src/interac
36
36
  - Color is used functionally to indicate state. If removing a color does not lose information, the text is colored using `dim`, `muted`, or left unstyled.
37
37
  - `warning` amber is reserved for true warnings. Costs and neutral telemetry numbers use `muted`.
38
38
  - `accentDeep` is used only in section tags. Metric values (such as TTFT, tokens-per-second, and autonomy status) use `muted`.
39
- - `action` neon orange remains scarce and strictly disciplined: only while Clio is acting or a prompt owns the keyboard (e.g. running connect/probe operations, active dispatch/fleet execution, or the keyboard-owning confirmation border / `STEER` mode). It is never used for idle decoration or settled telemetry, and never appears on more than one element per screen region.
39
+ - `action` neon orange remains scarce and strictly disciplined: only while Clio is acting, for workspace-authority and worker-escalation decision frames, or in `STEER` mode. It is never used for idle decoration or settled telemetry, and never appears on more than one element per screen region. Outward, safety-net, and system decision frames use `warning`; conversational answers use `accent`.
40
40
  - Per-surface color budgets limit noise: chip strips use at most one non-neutral token per chip, and framed cards use at most one status token alongside neutral colors.
41
41
 
42
42
  ---
@@ -118,6 +118,8 @@ Overlay frames share the island's top border rules and include keyboard shortcut
118
118
  └─ [Tab] mode · [Esc] close ─────────────────┘
119
119
  ```
120
120
 
121
+ Fleet run cards add two bounded budget rows when native dispatch admission supplies an envelope. The `policy` row shows the recipe default or exact pin, its optional maximum, and the invocation request. The `budget` row shows the effective phase, the operator lifetime cap, and the clamp or retry/revision escalation reason. Historical or external-agent rows without this provenance omit both rows.
122
+
121
123
  ### 4.3 Section Headers
122
124
 
123
125
  - **Panel Section Tag**: Bold CAPS in `accentDeep`.
@@ -144,6 +146,21 @@ All TUI overlays and cards support compact widths down to 40 columns:
144
146
  - Keybinding hints, cards, and markdown detail text wrap fluidly without horizontal clipping.
145
147
  - Settings provides a dedicated drill-down stack below 72 columns.
146
148
 
149
+ ### 4.7 Decision Consequence Frames
150
+
151
+ Permission confirmation and `ask_user` use one pure consequence presentation classifier while keeping separate input and execution protocols. The classifier supplies the tier title, semantic frame token, consequence and reversibility copy, requester attribution, and display actions. Permission keeps allow-once, deny, and stop behavior. `ask_user` keeps selection, free-text, cancellation, and its compact, panel, or interview layout chosen from question shape.
152
+
153
+ | Tier | Title | Token | Plain-text identity |
154
+ | --- | --- | --- | --- |
155
+ | Conversation | `Answer a question` | `accent` | `Conversational answer` |
156
+ | Workspace | `Approve workspace action` | `action` | `Workspace authority` |
157
+ | Outward | `Confirm outward consequence` | `warning` | `Outward consequence` |
158
+ | Safety net | `Safety-net confirmation` | `warning` | `Safety-net confirmation` |
159
+ | System | `Approve system change` | `warning` | `System change` |
160
+ | Worker | `Worker needs approval` | `action` | `Worker escalation` |
161
+
162
+ The words carry the meaning when color is disabled. Permission copy states the exact one-shot authority, whether effects are reversible, the authenticated requester and axis, and what deny and stop do. The classifier never consumes question, reason, summary, option-label, or requested-title prose, so those strings cannot select or lower a tier.
163
+
147
164
  ---
148
165
 
149
166
  ## 5. Screen Surfaces & State Choreography
@@ -230,6 +247,7 @@ The collapsed form is one composed ledger line:
230
247
  - Expanded calls show the primary argument in the signature and every secondary argument as a typed field list. Multiline argument bodies become line and byte facts, nested objects retain structured rendering, and safety-sensitive values remain redacted.
231
248
  - Running calls label `live output` and replace the cumulative partial result in place. Settled calls label `output` and show available exit status, result or observation counts, line count, displayed and total byte sizes, truncation, timeout, tool-token usage, dynamically added tools, context exclusion, and the full-output path. A blocked or aborted admission instead labels its `decision` and does not claim that the tool ran.
232
249
  - A call parked for one-shot approval replaces its running timer with `awaiting approval` and shows the already-sanitized action class, asking safety axis, and target below the row. These facts are transient UI state: approval, denial, abort, or settlement clears them, and they are never reconstructed from the session ledger.
250
+ - The live permission frame derives its consequence tier from those typed facts and the authenticated origin. It anchors at bottom center with five rows reserved for the composer and footer, and it recomputes that anchor on resize. Each queued frame retains its own tier and requester.
233
251
  - Text and image tool results keep their text while rendering images as MIME and byte-size placeholders; base64 image data is never written to the terminal.
234
252
  - Successful `edit` and `write` calls render the bounded diff produced by the tool result. Live regular-screen and fullscreen rows color removed and added lines with the `error` and `success` tokens and emphasize changed words; `/resume` replay and `/export` keep the same numbered diff as plain text.
235
253
  - Operator `!` and `!!` bash commands use the same running and settled block as model-initiated bash. The block appears before the process starts, streams the throttled cumulative stdout/stderr tail, and settles in place while the existing `bashExecution` session entry remains the durable record. `!!` continues to exclude that record from model context and says so in the block.
@@ -294,13 +312,22 @@ The `/settings` overlay is a full-screen transactional control center:
294
312
  - **Scoped Models Checklist**: Settings → `Models` provides a provider-backed checklist subview with target-level and target/model items, checked current selections, `Space` to toggle, and capability details in the inspector. Unresolved model references are preserved under an `Unavailable` group.
295
313
  - **Narrow Terminal Drill-Down Navigation**: Below 72 columns, Settings transitions from a split view to a modal drill-down stack (section list → section rows → detail drawer) with a breadcrumb and `Esc` moving up one level before closing. Includes `/` filtering across label, path, and description, narrowing per keystroke like `/model` and `/resume`. Below 60 columns, side margins are removed for full-width presentation.
296
314
 
297
- ### 7.2 Task and Decision Boards
315
+ ### 7.2 Fleet Runs Board
316
+
317
+ The `Alt+W` board renders one card per run. The default list is compact: run id, route, task, status, telemetry, retry, tool names, and proof. `Enter` opens the selected run's worker detail, which adds two rows to that card and nothing to any other:
318
+
319
+ - **`doing`**: the phase (`◐ thinking` in `reason`, `◑ writing` in `accent`, `⚙ tool` in `action`, `◔ waiting` in `info`) followed by the running call as `<tool> <verb> <object>`, or the last finished call as `last <tool> <verb> <object>`. The verb and object come from a descriptor composed at the worker seam; raw arguments never reach the renderer.
320
+ - **`answer`**: the newest rows of the worker's bounded prose on a `│` rail with a hanging indent under the key, then a dim row naming the lines and bytes the bounds refused and the `/view dispatch:<runId>` deep link.
321
+
322
+ Wrapping happens before the row cap, so the block is at most six rows tall at any width and a streaming answer cannot make the card grow under the operator. Detail follows the cursor rather than pinning to a run, and closing the board closes it. Reasoning text is never rendered; the `thinking` phase word is the whole of what the board says about it.
323
+
324
+ ### 7.3 Task and Decision Boards
298
325
 
299
326
  - **Composite Tasks Board (`/tasks`, `Alt+B`)**: Presents four sections in one reopenable overlay: the live session board, terminal task history, successful workspace artifacts, and project-scoped operator tasks. Selecting a workspace artifact opens the filtered `/view` path. Operator rows support add, hand, done, and drop actions; refresh is explicit for captured history and artifacts, while lightweight repaint reads the current board snapshot.
300
327
  - **Settled Decisions Board (`/decisions`, `Alt+D`)**: Groups completed and cancelled interviews on the active branch, expands source questions and answers, and lets the operator supersede a value or submit a correction. Corrections travel through the ordinary operator-turn path after the durable decision snapshot is updated.
301
328
  - **Approved editor overrides**: `Alt+B` and `Alt+D` are deliberate application-input boundary overrides of Pi's editor word-back and word-delete chords. Clio routes them before the editor so the two global boards remain one chord away. They are explicit exceptions to the general rule that Clio app bindings avoid Pi editor reserves, and users may rebind the Clio actions in `settings.yaml`.
302
329
 
303
- ### 7.3 Slash Autocomplete Command Palette
330
+ ### 7.4 Slash Autocomplete Command Palette
304
331
  - **Grouped Palette**: Typing `/` opens a grouped command palette (ordered by `Run`, `Inspect`, `Configure`, `Sessions`) with compact argument hints and formatted descriptions.
305
332
  - **One Canonical Spelling**: Autocomplete, help, and parsing expose the same unique slash-command names; no alias rows compete with canonical commands.
306
333
 
@@ -330,3 +357,10 @@ Two shapes, both ending at something the user can act on.
330
357
  ### 8.2 Memory Step Rows
331
358
 
332
359
  `/memory` activity rows read `<trigger> <decision> <reason>`, followed by `<N>w` when the step wrote to the bank and `<N> cited` when it cited entries, then the tier and latency. `describeTaskMemoryActivity` is the one place that builds this string.
360
+
361
+ Knowledge and procedural task-bank rows expose `p` to propose the selected
362
+ entry for the active canonical repository and `g` to propose it globally.
363
+ Global scope requires a second `g` press on the same entry after the warning
364
+ line appears. Status rows are labeled private and neither action can promote
365
+ them. Both actions create unapproved durable proposals, show the resulting
366
+ memory ID, and leave approval to the separate reviewed memory lifecycle.
@@ -1,7 +1,7 @@
1
1
  # Worker Dispatch Mechanics
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.6).
5
5
 
6
6
  This document describes the design and lifecycle of Clio Coder dispatched workers, focusing on the spawning sequence, execution isolation, the standard input/output NDJSON communication loop, and permission escalation routing.
7
7
 
@@ -213,6 +213,16 @@ Receipts carry exactly one integrity version (`RUN_RECEIPT_INTEGRITY_VERSION = 1
213
213
  - **Strict Primitive Handling**: `undefined` object properties are omitted; non-finite numbers (`NaN`, `Infinity`) or `bigint` throw an explicit serialization error.
214
214
  - **Coverage**: Includes every current receipt field and reconstructible ledger field, including route intent/decision/quality, execution role, worker identity, result-contract conformance, node/reroute/gate/plan provenance, briefing, steering, and `outcomeCode`.
215
215
 
216
+ Integrity is only the artifact-integrity axis of the canonical trust status.
217
+ The other axes are validation grounding, independent review, context
218
+ provenance, autonomy enforcement, and completion evidence. Sealing proves that
219
+ the receipt matches its covered ledger facts; it does not verify correctness,
220
+ establish context authorship, turn a correlated review into an independent
221
+ one, or prove completion. Every non-absent canonical fact retains a named
222
+ source and authority plus bounded references to detailed artifacts. The full
223
+ state vocabulary and compatibility map are documented in
224
+ [`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
225
+
216
226
  ### 5.3 Acceptance Coverage
217
227
 
218
228
  The assignment contract's acceptance scenarios map to deterministic contract
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iowarp/clio-coder",
3
- "version": "0.3.3",
3
+ "version": "0.3.6",
4
4
  "description": "Coding agent for HPC and scientific-software developers, part of IOWarp's CLIO ecosystem of agentic science.",
5
5
  "keywords": [
6
6
  "ai",
@@ -76,8 +76,7 @@
76
76
  "test:file": "node --import tsx --import ./tests/harness/tmp-root.ts --test",
77
77
  "pretest": "test -f dist/assets/codewiki.json && [ -z \"$(find src -newer dist/assets/codewiki.json -type f -print -quit)\" ] || npm run build",
78
78
  "test": "node scripts/shard-tests.mjs",
79
- "test:coverage": "node scripts/test-coverage.mjs --experimental-test-coverage --test-coverage-include='src/**/*.ts' --test-coverage-exclude='src/**/*.d.ts' 'tests/contracts/**/*.test.ts' 'tests/smoke/**/*.test.ts'",
80
- "test:repeat": "node tests/harness/repeat-tests.mjs",
79
+ "test:repeat": "node scripts/repeat-tests.mjs",
81
80
  "test:trace-viewer": "npm --prefix apps/trace-viewer test",
82
81
  "trace:ui": "node apps/trace-viewer/server.mjs",
83
82
  "ci": "npm run typecheck && npm run lint && npm run skills:check && npm run build && npm run test && npm run test:trace-viewer",
@@ -86,15 +85,15 @@
86
85
  "prepublishOnly": "npm run ci:release",
87
86
  "skills:pin": "node --import tsx scripts/pin-skills.ts",
88
87
  "skills:check": "node --import tsx scripts/pin-skills.ts --check",
89
- "//": "below here: real providers or a live model target, cost money and/or time, never run in CI",
90
- "test:live": "node scripts/live-smoke.mjs",
91
- "test:live-eval": "node scripts/live-eval-recon.mjs",
92
- "test:live-eval:fleet-dispatch": "node scripts/live-eval-fleet-dispatch.mjs",
93
- "test:live-verify:dispatch-routing": "node scripts/live-verify-dispatch-routing.mjs",
94
- "test:lifecycle": "node --import tsx scripts/lifecycle-matrix.mjs",
95
- "bench:swe": "python3 benchmarks/community/swe-bench-lite/swebench_clio.py",
96
- "bench:scicode": "python3 benchmarks/community/scicode/scicode_clio.py",
97
- "bench:tb": "python3 benchmarks/community/clio_fleet.py"
88
+ "benchmark:typecheck": "tsc -p benchmarks/tsconfig.json",
89
+ "benchmark:check": "npm run benchmark:typecheck && node --import tsx --test benchmarks/internal/tests/*.test.ts",
90
+ "benchmark:campaign": "node --import tsx benchmarks/internal/campaign.ts",
91
+ "benchmark:report": "node --import tsx benchmarks/internal/report.ts",
92
+ "//": "below here: a real model target, chosen with --target <id>; costs money and/or GPU time, never run in CI",
93
+ "live:smoke": "node --import tsx benchmarks/internal/live-smoke.ts",
94
+ "live:fleet-dispatch": "node --import tsx benchmarks/internal/live-fleet-dispatch.ts",
95
+ "live:tui": "node --import tsx benchmarks/internal/pty-drive.ts",
96
+ "live:home": "node --import tsx benchmarks/internal/live-home.ts"
98
97
  },
99
98
  "dependencies": {
100
99
  "@anthropic-ai/claude-agent-sdk": "0.3.186",
@@ -103,7 +102,8 @@
103
102
  "@earendil-works/pi-tui": "0.84.0",
104
103
  "@silvia-odwyer/photon-node": "^0.3.4",
105
104
  "grok-mermaid": "0.2.2",
106
- "ollama": "0.6.3"
105
+ "ollama": "0.6.3",
106
+ "smol-toml": "1.8.0"
107
107
  },
108
108
  "overrides": {
109
109
  "@anthropic-ai/sdk": "0.105.0",
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: clio-test
3
3
  description: Use when writing or modifying Clio Coder's own source under src/, or verifying a change end-to-end against the real test harness. Covers the three real layers (contracts / smoke / boundaries), choosing which to run for a given change, the mock-provider and ACP-over-stdio harness, and the hot-reload dev loop for picking up latest code. Activate on any src/ edit, before declaring a change verified, or when asked whether Clio still works.
4
- version: 0.1.3
4
+ version: 0.1.4
5
5
  license: Apache-2.0
6
6
  clio:
7
7
  registry-id: iowarp/clio-coder
@@ -24,17 +24,15 @@ local-vs-contribute boundary.
24
24
  ## Commands
25
25
 
26
26
  ```bash
27
- npm run typecheck # tsc -p tsconfig.tests.json (includes tests/)
28
- npm run lint # biome check .
29
- npm run check:boundaries # import boundary rules (tsx, no build)
27
+ npm run typecheck # tsc -p tsconfig.tests.json (includes tests/ and benchmarks/internal/)
28
+ npm run lint # biome check . && scripts/check-hygiene.ts (import boundaries, skill pins, doc drift)
30
29
  npm run test:file -- 'tests/contracts/**/*.test.ts' # contract tests (tsx, import src directly, no build)
31
30
  npm run test:file -- 'tests/smoke/**/*.test.ts' # spawn dist/cli/index.js end-to-end (NEEDS a build)
32
31
  npm run test # contracts + smoke, sharded across processes (the full gate)
33
32
  npm run build # tsup -> dist/
34
33
  npm run dev # tsup --watch -> rebuilds dist/ on save
35
34
  npm run ci # typecheck && lint && skills:check && build && test && test:trace-viewer
36
- npm run test:live # local live provider smoke; requires CLIO_CODER_LIVE_SMOKE=1
37
- npm run test:live -- --delegation # adds local opencode/copilot ACP checks
35
+ npm run live:smoke -- --target <id> # one real turn against a configured target; never in CI
38
36
  ```
39
37
 
40
38
  ## Which layer catches what
@@ -44,7 +42,7 @@ npm run test:live -- --delegation # adds local opencode/copilot ACP checks
44
42
  | pure logic in `src/domains/<x>/*.ts` | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | contract tests import `src` via tsx; no build |
45
43
  | dispatch / providers / prompts / safety / config / persistence / acp behavior | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | each has a file in `tests/contracts/` |
46
44
  | skills loader / activation | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | `tests/contracts/skills.test.ts`, `skill-activation-compaction.test.ts` |
47
- | any `src/` import edit | `npm run check:boundaries` | enforces rule1/2/3 |
45
+ | any `src/` import edit | `npm run lint` | the hygiene check enforces rule1/2/3 |
48
46
  | `src/cli/*` or `src/entry/*` user-facing flow | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke spawns the real `dist/cli/index.js` |
49
47
  | ACP surface (`src/cli/acp.ts`, engine ACP) | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke drives `clio-coder acp` over JSON-RPC/stdio |
50
48
 
@@ -54,8 +52,9 @@ a single file.
54
52
  ## Boundary rules you must not break
55
53
 
56
54
  `tests/boundaries/check-boundaries.ts` enforces three rules (also the Hard
57
- Invariants in `CLIO-CODER.md`). If `npm run check:boundaries` reports a
58
- violation, fix the import — never silence the check:
55
+ Invariants in `CLIO-CODER.md`), run by `scripts/check-hygiene.ts` under
56
+ `npm run lint`. If it reports a violation, fix the import — never silence the
57
+ check:
59
58
 
60
59
  - **rule1**: only `src/engine/**` may value-import `@earendil-works/pi-*`. Outside
61
60
  engine, use Clio contracts or type-only imports that erase at compile time.
@@ -70,9 +69,9 @@ There are two independent reload mechanisms; know which applies.
70
69
 
71
70
  **Source reload for tests.** This is the "pick up latest code" loop:
72
71
 
73
- - **Fast loop — no build.** The contracts glob and `check:boundaries` run
74
- `node --import tsx --test` and import `src/**` directly, so they always run the
75
- latest source with zero build step. Iterate here whenever the change is pure
72
+ - **Fast loop — no build.** The contracts glob runs `node --import tsx --test`
73
+ and imports `src/**` directly, and the hygiene lint reads source statically,
74
+ so both always see the latest source with zero build step. Iterate here whenever the change is pure
76
75
  logic or a contract.
77
76
  - **Full loop — needs `dist/`.** The smoke glob spawns `dist/cli/index.js`, so it
78
77
  only sees code that has been built. Keep `npm run dev` (`tsup --watch`) running
@@ -98,7 +97,7 @@ restart the process (against a freshly built `dist/`).
98
97
  1. Write the change.
99
98
  2. `npm run typecheck` and `npm run lint`.
100
99
  3. Run the narrowest layer from the table above.
101
- 4. `npm run check:boundaries` if you touched imports.
100
+ 4. `npm run lint` if you touched imports.
102
101
  5. If you touched CLI/entry/ACP: `npm run build` (or rely on `dev` watch), then
103
102
  `npm run test:file -- 'tests/smoke/**/*.test.ts'`.
104
103
  6. `npm run ci` before calling it done. Report exactly what ran and what is
@@ -113,8 +112,10 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
113
112
 
114
113
  ## What NOT to do
115
114
 
116
- - Don't reintroduce `tests/unit|integration|e2e/` or a pty harness — that
117
- taxonomy was deliberately removed.
115
+ - Don't reintroduce `tests/unit|integration|e2e/`; that taxonomy was
116
+ deliberately removed. Don't add a second pseudo-terminal: `tests/harness/pty.ts`
117
+ is the one PTY, used by the three `*-pty`/`tui-width-matrix` smoke suites and
118
+ by `benchmarks/internal/pty-drive.ts`.
118
119
  - Don't add `scripts/diag-*.ts` or `scripts/verify-*.ts`. A test belongs in
119
120
  `tests/`; a one-off probe belongs in `/tmp` and gets deleted (see
120
121
  `references/harness.md`).
@@ -126,5 +127,7 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
126
127
 
127
128
  ## Harness reference
128
129
 
129
- Driving the real CLI, the mock provider, and ACP over stdio, plus the throwaway
130
- probe pattern: **see `references/harness.md`**.
130
+ Driving the real CLI, the mock provider, ACP over stdio, and the PTY, plus the
131
+ throwaway probe pattern: **see `references/harness.md`**. Driving the real
132
+ binary against a real model (headless, PTY, tmux, herdr) is a different claim
133
+ and lives in `benchmarks/internal/SKILL.md`.
@@ -7,7 +7,7 @@ the gap (it cites the dead unit/integration/e2e taxonomy), then WITH it.
7
7
  Prompt: "I changed pure logic in `src/domains/dispatch/validation.ts`. What do I
8
8
  run and why?"
9
9
  Expected:
10
- - `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `check:boundaries` if imports changed).
10
+ - `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `npm run lint` if imports changed).
11
11
  - Explains contracts import `src` via tsx, so no build is needed.
12
12
  - Does NOT suggest `test:unit` / `test:e2e` (those don't exist).
13
13
 
@@ -26,14 +26,14 @@ Expected:
26
26
  interactive testing. Distinguishes this from config hot-reload (classify.ts).
27
27
 
28
28
  ## T4 — boundary violation
29
- Prompt: "`check:boundaries` says a domain imports another domain's extension.ts.
29
+ Prompt: "`npm run lint` says a domain imports another domain's extension.ts.
30
30
  Quickest fix?"
31
31
  Expected:
32
32
  - Route through the target domain's `index.ts` contract (rule3). Does NOT
33
33
  suggest a `biome-ignore` or exclude.
34
34
 
35
35
  ## Baseline failure modes to watch for (RED)
36
- - Cites `test:unit`/`test:integration`/`test:e2e` or a pty harness.
36
+ - Cites `test:unit`/`test:integration`/`test:e2e`, or a PTY other than `tests/harness/pty.ts`.
37
37
  - Claims smoke tests run against source (they run against `dist/`).
38
38
  - Invents a hot-reload feature that reloads a running session's code.
39
39
 
@@ -1,18 +1,20 @@
1
1
  # Clio test harness reference
2
2
 
3
- How to drive the real Clio binary, a mock provider, and the ACP surface in
4
- tests. All of this is non-interactive there is no pty harness in v0.2.2.
3
+ How to drive the real Clio binary, a mock provider, the ACP surface, and a
4
+ real pseudo-terminal in tests. Every model here is a stub; these are machinery
5
+ tests. A run against a real model is `benchmarks/internal/SKILL.md`.
5
6
 
6
7
  ## Contents
7
8
  - The spawn harness (`runCli`, `makeScratchHome`)
8
9
  - Mocking a provider (OpenAI-compatible SSE fixture)
9
10
  - ACP over JSON-RPC/stdio
11
+ - The PTY (`openPty`, `runInPty`)
10
12
  - One-off probes (no test file)
11
13
 
12
14
  ## The spawn harness
13
15
 
14
- `tests/harness/spawn.ts` is the only harness. It spawns `node dist/cli/index.js`,
15
- so **build first** (or keep `npm run dev` running) before `test:smoke`.
16
+ `tests/harness/spawn.ts` spawns `node dist/cli/index.js` with piped stdio, so
17
+ **build first** (or keep `npm run dev` running) before running smoke.
16
18
 
17
19
  ```ts
18
20
  import { makeScratchHome, runCli } from "../harness/spawn.js";
@@ -76,11 +78,38 @@ client (see `createJsonRpcProcessClient` in the smoke test): `initialize` →
76
78
  non-spec discriminator breaks strict clients like Zed, so the smoke test asserts
77
79
  every emitted variant is in the v1 set.
78
80
 
81
+ ## The PTY
82
+
83
+ Piped stdio reports no terminal width and no TTY, so the TUI refuses to start
84
+ and every width-sensitive path collapses to 80 columns. `tests/harness/pty.ts`
85
+ opens a real pseudo-terminal through `node-pty` (a devDependency, never
86
+ shipped):
87
+
88
+ ```ts
89
+ import { openPty, runInPty, stripAnsi, visibleLines } from "../harness/pty.js";
90
+
91
+ // Scripted: type on a schedule, stop when the output matches, bounded by a timeout.
92
+ const run = await runInPty(process.execPath, [CLI], { cols: 120, rows: 40, cwd, env,
93
+ readyWhen: /ctx /, input: [{ afterMs: 200, data: "/quit\r" }], until: /bye/, timeoutMs: 20_000 });
94
+
95
+ // Controllable: write, resize, pause output, wait for a matcher, wait for exit.
96
+ const session = await openPty(process.execPath, [CLI], { cols: 140, rows: 44, cwd, env });
97
+ await session.waitForOutput((out) => /ctx /.test(stripAnsi(out)), 30_000);
98
+ session.write("/quit\r");
99
+ await session.waitForExit(10_000);
100
+ ```
101
+
102
+ Use it only for what a pipe cannot show: width, raw mode, SIGINT through a
103
+ terminal, the alternate-screen and keyboard-protocol teardown. The three
104
+ suites that need it are `tests/smoke/tui-width-matrix.test.ts`,
105
+ `instant-shell-pty.test.ts`, and `render-trace-pty.test.ts`. Anything else
106
+ belongs on `runCli`.
107
+
79
108
  ## One-off probes (no test file)
80
109
 
81
110
  To poke at Clio without writing a permanent test, drop a throwaway script in
82
- `/tmp` and run it with tsx. Delete it when done — never leave probes under
83
- `tests/` or `scripts/`.
111
+ your scratch directory and run it with tsx. Delete it when done — never leave
112
+ probes under `tests/`, `scripts/`, or `benchmarks/`.
84
113
 
85
114
  ```ts
86
115
  // /tmp/probe.ts
@@ -1,4 +1,4 @@
1
- # Where Clio's tests live (v0.2.2)
1
+ # Where Clio's tests live
2
2
 
3
3
  Three layers under `tests/`. Add a new test next to the closest existing file;
4
4
  create a new file only for a genuinely new domain cluster.
@@ -7,10 +7,17 @@ create a new file only for a genuinely new domain cluster.
7
7
 
8
8
  | Layer | Path | Runner | Build needed |
9
9
  |---|---|---|---|
10
- | contracts | `tests/contracts/*.test.ts` | `node --import tsx --test` | no (imports `src`) |
11
- | smoke | `tests/smoke/*.test.ts` | `node --import tsx --test` | **yes** (spawns `dist/`) |
12
- | boundaries | `tests/boundaries/*.test.ts` | `node --import tsx --test` | no |
13
- | harness (not a test) | `tests/harness/spawn.ts` | imported by smoke | — |
10
+ | contracts | `tests/contracts/*.test.ts` | `npm run test:file -- <glob>` (tsx + scratch root) | no (imports `src`) |
11
+ | smoke | `tests/smoke/*.test.ts` | `npm run test:file -- <glob>` | **yes** (spawns `dist/`) |
12
+ | boundaries | `tests/boundaries/check-boundaries.ts` | `npm run lint` (hygiene) | no |
13
+ | harness (not tests) | `tests/harness/*.ts` | imported by contracts and smoke | — |
14
+
15
+ The harness modules: `spawn.ts` (run the built CLI with pipes), `scratch-env.ts`
16
+ (isolated Clio home), `pty.ts` (a real pseudo-terminal), `openai-compat-fixture.ts`
17
+ and `fake-lmstudio-server.ts` (stub providers), `fake-ssh.ts` (stub fleet node),
18
+ `clock.ts` (steppable clock), plus dispatch, receipt, and module-graph helpers.
19
+ Everything under `tests/` stubs the model. Real-model runs are
20
+ `benchmarks/internal/` and never run under `npm test`.
14
21
 
15
22
  ## Contract test files
16
23
 
@@ -33,18 +40,21 @@ create a new file only for a genuinely new domain cluster.
33
40
  | Area | File |
34
41
  |---|---|
35
42
  | non-interactive CLI + ACP-over-stdio end-to-end | `tests/smoke/cli.test.ts` |
36
- | import boundary rules (rule1/2/3) | `tests/boundaries/boundaries.test.ts` |
37
- | boundary checker implementation | `tests/boundaries/check-boundaries.ts` |
43
+ | the package as installed from `npm pack` | `tests/smoke/pack-install.test.ts` |
44
+ | TUI at real terminal sizes, NO_COLOR, Ctrl-C teardown (PTY) | `tests/smoke/tui-width-matrix.test.ts` |
45
+ | instant shell before hydration, SIGTERM through the lease (PTY) | `tests/smoke/instant-shell-pty.test.ts` |
46
+ | committed-frame render trace under PTY backpressure (PTY) | `tests/smoke/render-trace-pty.test.ts` |
47
+ | import boundary rules (rule1/2/3), run under `npm run lint` | `tests/boundaries/check-boundaries.ts` |
38
48
 
39
49
  ## Running a subset
40
50
 
41
51
  ```bash
42
52
  # all contracts
43
- node --import tsx --test 'tests/contracts/**/*.test.ts'
53
+ npm run test:file -- 'tests/contracts/**/*.test.ts'
44
54
  # one file
45
- node --import tsx --test tests/contracts/skills.test.ts
55
+ npm run test:file -- tests/contracts/skills.test.ts
46
56
  # only it.only / describe.only within a file
47
- node --import tsx --test --test-only tests/contracts/skills.test.ts
57
+ npm run test:file -- --test-only tests/contracts/skills.test.ts
48
58
  ```
49
59
 
50
60
  ## Writing tests