@iowarp/clio-coder 0.3.3 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/CHANGELOG.md +39 -0
  2. package/CONTRIBUTING.md +1 -1
  3. package/README.md +3 -3
  4. package/dist/{acp-P2AQILE2.js → acp-S5R4RR5B.js} +7 -6
  5. package/dist/{agents-72W3BI7I.js → agents-P6DMMVZY.js} +24 -21
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5TWEIYDN.js → auth-2XCZLPKS.js} +12 -8
  8. package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
  9. package/dist/{chunk-2DJ2KNFG.js → chunk-2LZI5CAG.js} +133 -13
  10. package/dist/{chunk-OAO4GE4M.js → chunk-2TZWSW76.js} +2 -2
  11. package/dist/{chunk-OOJYHWRB.js → chunk-34475P3I.js} +2 -2
  12. package/dist/{chunk-4XUGQOHA.js → chunk-35MKKU5R.js} +4 -4
  13. package/dist/{chunk-LM5TQCJZ.js → chunk-3HZ5RWN2.js} +5 -5
  14. package/dist/{chunk-AGYYIBLL.js → chunk-3JLKSKD7.js} +2 -2
  15. package/dist/{chunk-KZWTDYJF.js → chunk-4JUF2NNX.js} +7 -7
  16. package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
  17. package/dist/chunk-5M54SPOL.js +926 -0
  18. package/dist/{chunk-STBPMHSX.js → chunk-7RXG6QRZ.js} +51 -11
  19. package/dist/{chunk-X6IAEBZR.js → chunk-A2GZF7DC.js} +5 -5
  20. package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
  21. package/dist/chunk-AOCYTWAV.js +449 -0
  22. package/dist/chunk-BEY543CS.js +258 -0
  23. package/dist/{chunk-6N5PTWMY.js → chunk-BP4OYD6A.js} +32 -13
  24. package/dist/chunk-BPGS2WCQ.js +612 -0
  25. package/dist/{chunk-V6RTAOC2.js → chunk-BRXQQJFP.js} +8 -8
  26. package/dist/chunk-CFGTUFWB.js +67 -0
  27. package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
  28. package/dist/{chunk-DUYJ5IO6.js → chunk-EDRHSCIE.js} +4 -4
  29. package/dist/{chunk-XBXAASKX.js → chunk-EFADSJET.js} +2 -2
  30. package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
  31. package/dist/chunk-FYYLNIL5.js +313 -0
  32. package/dist/{chunk-FNTMWMX5.js → chunk-HV5X7OR2.js} +14 -12
  33. package/dist/{chunk-TZTZS7QK.js → chunk-HXG4IURW.js} +5 -3
  34. package/dist/{chunk-5UFT4SUX.js → chunk-K6WL7QZT.js} +3 -3
  35. package/dist/chunk-K7VKOLQQ.js +15 -0
  36. package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
  37. package/dist/{chunk-OC7FIQPC.js → chunk-KRPY7NTG.js} +10 -7
  38. package/dist/chunk-LL4KHSZI.js +22 -0
  39. package/dist/{chunk-DSELYM6W.js → chunk-MEQ45TQ4.js} +15 -9
  40. package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
  41. package/dist/{chunk-6SGHMWE3.js → chunk-N4CZJQRK.js} +5 -5
  42. package/dist/{chunk-TZK7PACC.js → chunk-NILBFAPG.js} +14 -8
  43. package/dist/chunk-OZNBF4L3.js +23 -0
  44. package/dist/{verify-375KUB3Y.js → chunk-PCZJO5TI.js} +127 -42
  45. package/dist/chunk-QQK64KLB.js +1360 -0
  46. package/dist/{chunk-SRDMMSEP.js → chunk-QQL5RT5M.js} +979 -1619
  47. package/dist/{chunk-LZSJBIVT.js → chunk-QWU7ZBO7.js} +70 -720
  48. package/dist/{chunk-2TLUCQVG.js → chunk-RD5U66HV.js} +3 -3
  49. package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
  50. package/dist/{chunk-OQ33BKR3.js → chunk-TTNYS3EA.js} +3 -60
  51. package/dist/chunk-TW3WDMVS.js +677 -0
  52. package/dist/chunk-TZSKNMZG.js +434 -0
  53. package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
  54. package/dist/{chunk-PIWWS5BL.js → chunk-UZHIZC5S.js} +7 -7
  55. package/dist/{chunk-UFIIWP2H.js → chunk-VAWWTKDP.js} +8 -8
  56. package/dist/{chunk-COU2UHX6.js → chunk-VEZEGCGW.js} +170 -2
  57. package/dist/{chunk-LW6DSM3M.js → chunk-VMNQ6OZA.js} +98 -202
  58. package/dist/chunk-VSNATDE6.js +122 -0
  59. package/dist/chunk-W6GROXXM.js +69 -0
  60. package/dist/chunk-WPQLXFOZ.js +375 -0
  61. package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
  62. package/dist/{chunk-PAJK6MAQ.js → chunk-X6COSD2O.js} +5 -5
  63. package/dist/chunk-ZGVHUX3M.js +66 -0
  64. package/dist/{chunk-LWLEKMDQ.js → chunk-ZYKPLLNQ.js} +510 -547
  65. package/dist/cli/index.js +27 -23
  66. package/dist/{clio-JOU4FXVA.js → clio-J5JIOIDS.js} +7 -6
  67. package/dist/{code-nav-7AX6FYE6.js → code-nav-AXCXSBHX.js} +5 -3
  68. package/dist/{config-XCDVKR23.js → config-OEBMIN2U.js} +37 -27
  69. package/dist/{configure-4GAP54ZW.js → configure-PUQOSIXQ.js} +16 -13
  70. package/dist/{context-5VKGUVJJ.js → context-EKDCKUUZ.js} +82 -7
  71. package/dist/{context-4UOGGLQ5.js → context-MGSE4Z2T.js} +33 -23
  72. package/dist/{context-77FM5DV5.js → context-URSXPBCK.js} +17 -9
  73. package/dist/{context-clear-XXJRLCJJ.js → context-clear-KDAJRNUK.js} +33 -23
  74. package/dist/context-working-set-SBKMPPI2.js +1552 -0
  75. package/dist/{dispatch-runner-QPRDDBDX.js → dispatch-runner-MSWN72NK.js} +43 -29
  76. package/dist/{doctor-HR46URBJ.js → doctor-7BSE27PJ.js} +10 -10
  77. package/dist/{eval-XSSNATB4.js → eval-IZGDOO4H.js} +9 -8
  78. package/dist/{evidence-6HG2PY2B.js → evidence-SR7WXB5B.js} +51 -23
  79. package/dist/{evolve-K7YU3NCY.js → evolve-K7VE2CBX.js} +30 -20
  80. package/dist/{fleet-VY3HHKN6.js → fleet-7XMJNQNF.js} +48 -38
  81. package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-AQNAH644.js} +3 -3
  82. package/dist/{init-JYGXI3FK.js → init-JGNPAYXT.js} +41 -31
  83. package/dist/{memory-WFZMGYHX.js → memory-4ALKDJ4Q.js} +32 -22
  84. package/dist/{models-I5QWSEOM.js → models-ZMMLFJNN.js} +22 -19
  85. package/dist/{monitor-GE4ID3IA.js → monitor-2F3T5KHP.js} +55 -43
  86. package/dist/{orchestrator-EM5MC3HM.js → orchestrator-ORHT43JB.js} +572 -388
  87. package/dist/{reset-L2FQEE3E.js → reset-NXGTYNUO.js} +4 -3
  88. package/dist/{run-ZU3QMZPZ.js → run-RF4WJGMT.js} +51 -41
  89. package/dist/{share-S5BZQC5I.js → share-UT3W6E4M.js} +5 -4
  90. package/dist/{skills-X5VXCRNQ.js → skills-PSACKC5Q.js} +2 -2
  91. package/dist/{skills-eval-WKIHWTHR.js → skills-eval-WJSI55RZ.js} +34 -24
  92. package/dist/{targets-SNCPI2NR.js → targets-PIIRAOYS.js} +23 -20
  93. package/dist/{terminal-lease-BNAHVHBS.js → terminal-lease-ULWXWNVY.js} +4 -3
  94. package/dist/{upgrade-JQHHPQ4K.js → upgrade-346TZ6AV.js} +18 -17
  95. package/dist/{usage-OR4O5SMZ.js → usage-6KKXR32N.js} +34 -24
  96. package/dist/verifiers-4UUM6TEE.js +1214 -0
  97. package/dist/verify-X5HDROLA.js +25 -0
  98. package/dist/{wiki-generate-UEXP2ARI.js → wiki-generate-7STOCIFZ.js} +42 -31
  99. package/dist/worker/entry.js +33 -24
  100. package/docs/README.md +7 -6
  101. package/docs/acp.md +1 -1
  102. package/docs/alcf-provider.md +1 -1
  103. package/docs/architecture.md +2 -2
  104. package/docs/artifact-versions.md +1 -1
  105. package/docs/built-in-agents.md +1 -1
  106. package/docs/capacity-and-scheduling.md +1 -1
  107. package/docs/commands-and-modes.md +53 -21
  108. package/docs/config-knobs-audit.md +1 -2
  109. package/docs/configuration-and-targets.md +15 -1
  110. package/docs/context-engine.md +64 -12
  111. package/docs/context-working-set.md +194 -0
  112. package/docs/development-pipeline.md +1 -1
  113. package/docs/documentation-coverage.md +5 -5
  114. package/docs/documentation-guide.md +6 -5
  115. package/docs/environment-variables.md +2 -1
  116. package/docs/eval-runner.md +1 -1
  117. package/docs/evals-internal.md +14 -1
  118. package/docs/evidence-and-memory.md +74 -2
  119. package/docs/evolution.md +1 -1
  120. package/docs/exit-codes-and-output.md +1 -1
  121. package/docs/extensions-and-sharing.md +2 -2
  122. package/docs/fleet-dispatch.md +22 -7
  123. package/docs/glossary.md +21 -1
  124. package/docs/installation-and-lifecycle.md +2 -2
  125. package/docs/middleware-and-components.md +1 -1
  126. package/docs/model-catalog.md +7 -9
  127. package/docs/observability.md +4 -4
  128. package/docs/proactive-memory.md +1 -1
  129. package/docs/prompt-envelope-and-tools.md +4 -4
  130. package/docs/provider-adapter-cookbook.md +1 -1
  131. package/docs/release-cut-checklist.md +31 -31
  132. package/docs/safety-model.md +23 -4
  133. package/docs/scientific-validation.md +21 -3
  134. package/docs/session-lifecycle.md +3 -3
  135. package/docs/skills-marketplace.md +1 -1
  136. package/docs/tool-usage.md +79 -12
  137. package/docs/trace-store.md +1 -1
  138. package/docs/troubleshooting.md +1 -1
  139. package/docs/tui-design.md +1 -1
  140. package/docs/worker-dispatch-mechanics.md +11 -1
  141. package/package.json +8 -11
  142. package/skills/meta/clio-test/SKILL.md +20 -17
  143. package/skills/meta/clio-test/evals.md +3 -3
  144. package/skills/meta/clio-test/references/harness.md +35 -6
  145. package/skills/meta/clio-test/references/test-map.md +20 -10
  146. package/skills/registry.yaml +2 -2
  147. package/skills/skill-marketplace.json +1 -1
  148. package/src/cli/context-working-set.ts +513 -0
  149. package/src/cli/context.ts +8 -0
  150. package/src/cli/evidence.ts +20 -2
  151. package/src/cli/index.ts +4 -0
  152. package/src/cli/verifiers.ts +325 -0
  153. package/src/core/bash-exec.ts +39 -14
  154. package/src/core/bus-events.ts +19 -4
  155. package/src/core/config.ts +54 -0
  156. package/src/core/defaults.ts +50 -3
  157. package/src/core/verification-scripts.ts +6 -0
  158. package/src/domains/agents/builtins/verifier.md +3 -0
  159. package/src/domains/config/classify.ts +1 -0
  160. package/src/domains/context/working-set/contract.ts +161 -0
  161. package/src/domains/context/working-set/defaults.ts +28 -0
  162. package/src/domains/context/working-set/engine.ts +203 -0
  163. package/src/domains/context/working-set/fold.ts +62 -0
  164. package/src/domains/context/working-set/horizon.ts +38 -0
  165. package/src/domains/context/working-set/marker.ts +103 -0
  166. package/src/domains/context/working-set/path-index.ts +436 -0
  167. package/src/domains/context/working-set/payload.ts +152 -0
  168. package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
  169. package/src/domains/context/working-set/policies/index.ts +21 -0
  170. package/src/domains/context/working-set/policies/structural.ts +160 -0
  171. package/src/domains/context/working-set/project.ts +132 -0
  172. package/src/domains/context/working-set/protect.ts +109 -0
  173. package/src/domains/context/working-set/recall.ts +177 -0
  174. package/src/domains/context/working-set/replay/controls.ts +112 -0
  175. package/src/domains/context/working-set/replay/load-clio.ts +199 -0
  176. package/src/domains/context/working-set/replay/metrics.ts +185 -0
  177. package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
  178. package/src/domains/context/working-set/replay/report.ts +139 -0
  179. package/src/domains/context/working-set/replay/runner.ts +325 -0
  180. package/src/domains/context/working-set/replay/synthetic.ts +422 -0
  181. package/src/domains/context/working-set/replay/trace.ts +21 -0
  182. package/src/domains/context/working-set/visible.ts +54 -0
  183. package/src/domains/evidence/build.ts +112 -45
  184. package/src/domains/evidence/eval.ts +24 -7
  185. package/src/domains/evidence/index.ts +53 -0
  186. package/src/domains/evidence/ordering.ts +12 -0
  187. package/src/domains/evidence/run-trust.ts +221 -0
  188. package/src/domains/evidence/store.ts +46 -6
  189. package/src/domains/evidence/trust-status.ts +854 -0
  190. package/src/domains/evidence/types.ts +26 -0
  191. package/src/domains/middleware/memory-intervention.ts +3 -0
  192. package/src/domains/middleware/stalled-turn.ts +165 -4
  193. package/src/domains/safety/autonomy.ts +1 -1
  194. package/src/domains/safety/default-path-policy.ts +8 -0
  195. package/src/domains/safety/finish-contract.ts +4 -3
  196. package/src/domains/safety/policy-engine.ts +48 -6
  197. package/src/domains/session/compaction/compact.ts +23 -1
  198. package/src/domains/session/compaction/cut-point.ts +2 -0
  199. package/src/domains/session/compaction/tokens.ts +16 -1
  200. package/src/domains/session/context-ledger.ts +2 -0
  201. package/src/domains/session/entries.ts +107 -1
  202. package/src/domains/session/manager.ts +9 -2
  203. package/src/domains/session/migrations/index.ts +22 -3
  204. package/src/engine/acp/server.ts +3 -0
  205. package/src/engine/agent.ts +18 -1
  206. package/src/engine/session.ts +9 -3
  207. package/src/entry/orchestrator.ts +16 -4
  208. package/src/interactive/chat-loop-messages.ts +18 -6
  209. package/src/interactive/chat-panel.ts +17 -1
  210. package/src/interactive/chat-renderer.ts +30 -21
  211. package/src/interactive/context-meter.ts +10 -0
  212. package/src/interactive/context-overlay.ts +81 -6
  213. package/src/interactive/context-recall-command.ts +110 -0
  214. package/src/interactive/interactive-slash-runtime.ts +37 -1
  215. package/src/interactive/model-session-replay.ts +21 -0
  216. package/src/interactive/overlay-general-openers.ts +6 -0
  217. package/src/interactive/overlay-session-lifecycle.ts +8 -4
  218. package/src/interactive/renderers/tool-execution.ts +18 -2
  219. package/src/interactive/session-transcript.ts +2 -2
  220. package/src/interactive/slash-commands.ts +29 -2
  221. package/src/interactive/turn-context.ts +238 -88
  222. package/src/interactive/turn-middleware.ts +6 -6
  223. package/src/tools/agent-tools.ts +11 -4
  224. package/src/tools/bash.ts +144 -82
  225. package/src/tools/builtin-tool-catalog.ts +11 -5
  226. package/src/tools/context/index.ts +105 -3
  227. package/src/tools/context/surface.ts +3 -2
  228. package/src/tools/core-bootstrap.ts +21 -0
  229. package/src/tools/dispatch-runner.ts +9 -7
  230. package/src/tools/monitor.ts +28 -20
  231. package/src/tools/registry.ts +59 -7
  232. package/src/tools/result-disposition.ts +550 -0
  233. package/src/tools/result-shaping.ts +262 -19
  234. package/src/tools/safe-exec.ts +2 -0
  235. package/src/tools/verify/authoring.ts +1119 -0
  236. package/src/tools/verify/catalog.ts +346 -0
  237. package/src/tools/verify/index.ts +13 -3
  238. package/src/tools/verify/scripts.ts +135 -37
  239. package/src/tools/verify/surface.ts +9 -5
  240. package/src/tools/worker-evidence.ts +35 -12
  241. package/dist/chunk-J7CWMCQD.js +0 -255
  242. package/dist/chunk-T6YILFSB.js +0 -80
  243. package/dist/chunk-VAKQQHWR.js +0 -434
  244. package/dist/chunk-VPAYEGVX.js +0 -184
@@ -1,11 +1,15 @@
1
1
  # Clio Coder Scientific Validation Contracts
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  Scientific software development cannot treat simple file presence as proof of correctness. A simulation script that crashes on rank 48, or writes out NetCDF arrays filled with `NaN`s, may still successfully write a file to the disk.
7
7
 
8
- Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.3.3, core Clio does not parse or enforce a scientific contract schema. The presence of `.clio-coder/validation.yaml`, `.clio-coder/validation.yml`, `validation.yaml`, `validation.yml`, or `VALIDATION.md` at the workspace root raises the default rigor level to `high`; the file contents are advisory material for developers, project agents, and external validators.
8
+ Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.3.4, the session rigor resolver does not parse or enforce a scientific contract schema. The presence of `.clio-coder/validation.yaml`, `.clio-coder/validation.yml`, `validation.yaml`, `validation.yml`, or `VALIDATION.md` at the workspace root raises the default rigor level to `high`; the file contents are advisory material for developers, project agents, and external validators.
9
+
10
+ This advisory convention is separate from the executable project verifier catalog at `.clio-coder/verifiers.yaml`. The verifier catalog has a strict version-1 schema and admits exact argv vectors to the `verify` tool. Scientific validation contracts and handbook expectations do not grant command authority: prose such as `validators: ["python tools/check_grid.py"]` remains guidance until the project owner confirms the equivalent argv, cwd, timeout, and tags in `verifiers.yaml`. The executable catalog does not interpret numerical tolerances or artifact expectations; it only runs the explicitly declared process vector through safe-exec.
11
+
12
+ `clio-coder verifiers author` can inspect top-level `validators` entries in the YAML contract filenames above and propose catalog checks. It labels those vectors as project-declared and shows their source index, exact argv, cwd, timeout, tags, catalog path, and resulting execution authority. This inspection is read-only. A command string with sound quoting and no shell operator can be represented as argv for review; shell expansion, pipes, redirection, environment assignments, incomplete quoting, and Markdown prose receive a manual JSON-argv diagnostic. Nothing becomes executable and nothing is dry-run until the operator confirms the catalog write with `--yes`.
9
13
 
10
14
  The convention below is a recommended shape for scientific projects that need to document expected dimensions, attributes, numerical tolerances, scheduler context, and verification commands for scientific artifacts. Developed at the [Gnosis Research Center (GRC)](https://grc.iit.edu) at Illinois Tech as part of the NSF-funded scientific-software context (NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318)), this convention links execution metadata with physical output checks without claiming that the current harness executes those checks automatically.
11
15
 
@@ -51,6 +55,20 @@ notes: |
51
55
  Re-run check_grid.py after job completion is observed.
52
56
  ```
53
57
 
58
+ The `validators` values above are intentionally advisory shell-like prose. Preview the exact catalog proposal with `clio-coder verifiers author`, or declare the Python validator manually without granting free-form shell interpretation:
59
+
60
+ ```yaml
61
+ # .clio-coder/verifiers.yaml
62
+ version: 1
63
+ checks:
64
+ - id: validate-grid
65
+ description: Validate the generated regional grid
66
+ command: [python, tools/check_grid.py, out/region_west.nc]
67
+ cwd: .
68
+ timeoutMs: 120000
69
+ tags: [scientific, netcdf]
70
+ ```
71
+
54
72
  ### Suggested Fields:
55
73
  1. **`version`:** Set to `1` for project-local compatibility.
56
74
  2. **`runtime.kind`:** Document execution mode (`local`, `slurm`, `mpi`, or `other`).
@@ -77,7 +95,7 @@ Comparing floating-point values in scientific computations must accommodate roun
77
95
 
78
96
  ## Common Scientific Artifact Families
79
97
 
80
- The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.3.3:
98
+ The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.3.4:
81
99
 
82
100
  - **`HDF5` / `NetCDF` / `Zarr`:** Multi-dimensional scientific array files.
83
101
  - **`FITS`:** Flexible Image Transport System (used in astrophysics).
@@ -1,6 +1,6 @@
1
1
  # Session Lifecycle
2
2
 
3
- This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.3.3`.
3
+ This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.3.4`.
4
4
 
5
5
  Source implementations: `src/engine/session.ts` and `src/domains/session/`.
6
6
 
@@ -39,11 +39,11 @@ export interface ClioSessionMeta {
39
39
  piMonoVersion: string;
40
40
  platform: string;
41
41
  nodeVersion: string;
42
- sessionFormatVersion?: number; // CURRENT_SESSION_FORMAT_VERSION = 3
42
+ sessionFormatVersion?: number; // CURRENT_SESSION_FORMAT_VERSION = 4
43
43
  }
44
44
  ```
45
45
 
46
- Format version `CURRENT_SESSION_FORMAT_VERSION = 3` (`src/engine/session.ts:66`) is stamped on all sessions created in `v0.3.3`. Sessions with missing or earlier format versions trigger schema migrations in `src/domains/session/migrations/` on `/resume`.
46
+ Format version `CURRENT_SESSION_FORMAT_VERSION = 4` (`src/engine/session.ts`) is stamped on all sessions created since the working-set layer landed. Version 4 adds the `contextEviction` and `contextRecall` ledger kinds. `runMigrations` in `src/domains/session/migrations/` rejects both directions on `/resume`: a missing or earlier version names the remedy (remove the session directory), and a version from the future says the session was written by a newer Clio and must not be read by this build.
47
47
 
48
48
  ---
49
49
 
@@ -1,7 +1,7 @@
1
1
  # Skills Marketplace
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  The Skills Hub (`/skill`) shows project skills, user skills, and the marketplace. Every marketplace row comes from the same local lookup that `clio-coder skills install <name>` and `/skill <name>` resolve through, so the hub lists nothing it cannot install.
7
7
 
@@ -1,11 +1,11 @@
1
1
  # Tool Usage Reference
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
7
7
 
8
- In Clio Coder v0.3.3, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
8
+ In Clio Coder v0.3.4, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
9
9
 
10
10
  ## Observation envelope: truncation notices, offload, next hints, and the turn budget
11
11
 
@@ -21,7 +21,7 @@ Truncated text results append exactly one notice line:
21
21
 
22
22
  Segments that do not apply are omitted. `<total>` renders as `N+` when the search stopped early at its item limit, so the true total was never counted. `next:` is an exact argument fragment (for example `limit=200` or `offset=451`); re-issue the same call with that argument changed to continue.
23
23
 
24
- Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<toolCallId>.txt` and the notice's `full:` segment names the path. Read it with `read` using offset/limit. Tools offload only when the byte cap cut collected content; a bare item-limit truncation continues via `next` and does not offload. `read` never offloads, because the source file is directly re-addressable via `offset`.
24
+ Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<sha256 of the captured text>.txt` and the notice's `full:` segment names the path. Read it with `read` using offset/limit. Tools offload only when the byte cap cut collected content; a bare item-limit truncation continues via `next` and does not offload. `read` never offloads, because the source file is directly re-addressable via `offset`.
25
25
 
26
26
  JSON-format results (code_nav, context scope=docs/workspace) never get an appended notice. An oversize JSON payload is replaced whole by the parseable stub `{"error":"result exceeded <cap>","offloadPath":"...","next":"..."}` so the model never receives JSON cut mid-document. Empty results are also valid JSON with empty arrays and `next` populated.
27
27
 
@@ -102,17 +102,24 @@ Arguments:
102
102
  - `command` (required).
103
103
  - `cwd` (optional). Working directory; resolved against the session workspace and rejected when it escapes it. The safety net blocks an escaping cwd at admission, and the tool enforces the same rule itself.
104
104
  - `timeout_ms` (optional). Default 300000 (5 minutes).
105
+ - `output_policy` (optional). Canonical model-context disposition: `full`, `bounded`, `summary`, or `metadata-only`. Omission is exactly `bounded`.
105
106
 
106
107
  Workspace containment: commands whose filesystem targets resolve outside the session workspace escalate to `system_modify` and ask for one-shot confirmation at every autonomy level (headless runs deny asks). Recognized targets are shell redirects, `tee`/`mkdir`/`touch` path operands, `cp`/`mv`/`ln` destinations, in-place `sed -i` operands, and any `cd`/`pushd` whose directory leaves the workspace, since a `cd` outside re-bases every relative path that follows it. Inside-workspace equivalents stay plain `execute` with no new prompts.
107
108
 
108
- Output shaping is tail-biased: the display keeps the LAST 16KB / 2000 lines, because the failing assertion, compiler error, and exit summary live at the end. Before truncating, the full output is spilled to the per-session scratch file and the appended note names the path; read it with offset/limit. A command producing more than 16MB of output is stopped with an error. A timeout or nonzero exit returns the shaped output plus a status line (`bash: command timed out after <ms>ms`, `bash: command failed (exit N)`).
109
+ The default `bounded` policy keeps a tail-biased model excerpt under the 16KB result budget, because the failing assertion, compiler error, and exit summary usually live at the end. `summary` is useful for noisy builds and test runs: code deterministically selects a bounded head, tail, and error-like lines, applies Clio's repository secret redactor, and records the source hash and algorithm in summary provenance. `metadata-only` is appropriate when the model needs only outcome and termination facts; stdout and stderr stay out of model context while the operator presentation, retained byte size, and retrieval path remain available. `full` is for output known to be small. It is admitted only when the complete captured result and its facts fit the bounded result/context budget; otherwise the result explicitly records a typed downgrade to tail-biased `bounded` and provides retrieval. Do not use `full` as the routine default.
109
110
 
110
- Reach for bash for builds, git, package managers, and anything without a dedicated tool. Prefer the dedicated tools over their shell equivalents: grep/find/read/ls get envelope truncation, exact continuation hints, and the shared ignore policy that `cat`, shell `grep`, and shell `find` do not. Prefer `verify` over bash for declared package.json verification scripts, since verify produces typed evidence.
111
+ Presentation is independent from model context. The operator-facing display remains folded and tail-biased under every policy. When the display or selected context omits captured content, the terminal result writes one per-session scratch artifact and names it in the result. Live updates use the selected policy, remain bounded, and never write per-update artifacts. Every terminal result records requested and applied context modes, captured/displayed/context bytes, truncation or downgrade state, and any offload path. Exit code, signal, timeout, abort, and output-cap facts survive every policy. Scratch retrieval may contain the raw retained output; the deterministic `summary` projection is the redacted surface.
112
+
113
+ A command producing more than 16MB of combined output is stopped with an error. UTF-8 decoding spans process chunks, and a code point split by the hard byte cap is discarded rather than replaced with an invalid character. Raw NUL bytes are removed from model context under every policy, which leaves multi-byte code points and ANSI escape sequences whole; the operator presentation and the scratch artifact keep the captured bytes, and the result still records the omission and its retrieval path. A timeout, abort, output cap, or nonzero exit preserves captured diagnostics and appends a status line such as `bash: command timed out after <ms>ms` or `bash: command failed (exit N)` before canonical shaping.
114
+
115
+ Reach for bash for builds, git, package managers, and anything without a dedicated tool. Prefer the dedicated tools over their shell equivalents: grep/find/read/ls get envelope truncation, exact continuation hints, and the shared ignore policy that `cat`, shell `grep`, and shell `find` do not. Prefer `verify` over bash for declared package scripts and project-catalog entries, since verify produces typed evidence.
111
116
 
112
117
  ```text
113
118
  bash(command="git status --short")
114
119
  bash(command="git log --oneline -10")
115
120
  bash(command="npm run build", timeout_ms=600000)
121
+ bash(command="npm run test", timeout_ms=600000, output_policy="summary")
122
+ bash(command="make artifact", output_policy="metadata-only")
116
123
  ```
117
124
 
118
125
  ## grep: search file contents with ripgrep
@@ -266,27 +273,87 @@ dispatch(tasks=["Refactor step 1", "Refactor step 2"], mode="sequential", timeou
266
273
 
267
274
  ## verify: run declared verification checks
268
275
 
269
- One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/frontend.ts`.
276
+ One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/catalog.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/authoring.ts`, `src/tools/verify/frontend.ts`.
270
277
 
271
278
  Arguments:
272
279
 
273
- - `check` (optional). A declared package.json script name or `"frontend"`. Omit to list available checks.
280
+ - `check` (optional). A declared project-catalog ID, package.json script name, or `"frontend"`. Omit to list available checks.
274
281
  - `path` (check=frontend). Artifact file under the workspace root.
275
- - `args` (optional). Extra arguments passed to the script after `--`. A JSON-string array is tolerated and parsed.
282
+ - `args` (package scripts only). Extra arguments passed after `--`. A JSON-string array is tolerated and parsed. Project-catalog checks ignore this field.
276
283
  - `browser` (check=frontend). `auto` (default), `required`, or `off`.
277
- - `cwd` (optional). Working directory.
278
- - `timeout_ms` (optional). Default 120000.
284
+ - `cwd` (package scripts only). Package working directory. Project catalogs are always discovered at the session workspace root, and a project check uses its declared `cwd`.
285
+ - `timeout_ms` (package scripts and frontend only). Default 120000. A project check uses its declared `timeoutMs`.
286
+ - `max_output_bytes` (package scripts and frontend only). Default 600000. Project checks retain the safe-exec default cap.
287
+
288
+ `verify()` lists checks grouped as `package.json` and `.clio-coder/verifiers.yaml`. Both providers project through the same canonical metadata: `{id, description, command, cwd, timeoutMs, tags, source}`. Package scripts must match the verification family `test*/lint*/build*/typecheck*/check*/format*/ci*` (a family prefix, optionally followed by `:`, `.`, or `-` and a suffix, e.g. `test:unit`). `verify(check="typecheck")` runs `npm run typecheck` through the safe-exec spine with no shell. A package script name outside the family is rejected with a pointer to run it through bash.
289
+
290
+ ### Project verifier catalog
291
+
292
+ Projects may commit a versioned executable catalog at `.clio-coder/verifiers.yaml`:
293
+
294
+ ```yaml
295
+ version: 1
296
+ checks:
297
+ - id: rust-workspace
298
+ description: Run the Rust workspace tests
299
+ command: [cargo, test, --workspace]
300
+ cwd: .
301
+ timeoutMs: 600000
302
+ tags: [rust, test]
303
+ ```
304
+
305
+ Version 1 is strict. Every root and check field shown above is required, unknown fields fail, and duplicate IDs fail. A project ID uses lowercase letters, digits, `.`, `_`, `:`, or `-`, begins with a letter or digit, and is at most 64 UTF-8 bytes. `frontend` is reserved. Descriptions are trimmed single-line text capped at 512 bytes. `command` is a nonempty argv array with at most 64 entries and 4096 bytes per entry. A shell command string is invalid, and explicit shell executables such as `sh`, `bash`, `pwsh`, and `cmd` are rejected. `cwd` is a repository-relative existing directory capped at 512 bytes; absolute paths, `..` escapes, and symbolic-link escapes fail. `timeoutMs` is a positive integer capped at 900000. A check may carry at most 16 distinct lowercase tags of at most 32 bytes each. The whole file is capped at 262144 bytes and may contain at most 128 checks. YAML aliases are disabled.
306
+
307
+ Provider IDs share one namespace. If a catalog ID collides with a discovered package script, listing and execution fail and identify both source files. Catalog parsing also fails closed before any package or project check runs.
308
+
309
+ `verify(check="rust-workspace")` spawns exactly `cargo` with `test` and `--workspace`; it does not interpolate model text or invoke a shell. Model-supplied `args`, `cwd`, `timeout_ms`, `max_output_bytes`, or undeclared environment fields cannot widen or replace the catalog entry. Safe execution passes only Clio's small environment allowlist, applies cancellation and the declared timeout, and shapes output at the standard 600000-byte cap. Execution details retain the compatible command string plus exact `argv`, `cwd`, `exitCode`, `durationMs`, `aborted`, `timedOut`, and `outputCapped` evidence, along with the check's declared source, command, cwd, timeout, description, and tags.
310
+
311
+ ### Guided catalog authoring
312
+
313
+ An empty `verify()` result points to `clio-coder verifiers author`. The authoring command inspects only command-bearing files at the workspace root:
314
+
315
+ - verification-family package scripts, projected exactly as `npm run <script>` and shown as already active rather than duplicated into the catalog;
316
+ - `Cargo.toml`, projected to Cargo's package or workspace test vector;
317
+ - visible build and test entries in `CMakePresets.json`, projected to the corresponding `cmake --build --preset` or `ctest --preset` vector;
318
+ - declared Python runners in `pyproject.toml`, `pytest.ini`, `tox.ini`, `noxfile.py`, or the pytest section of `setup.cfg`;
319
+ - a module directive in `go.mod`, projected to `go test ./...`;
320
+ - top-level `validators` entries in the documented YAML scientific-validation files.
321
+
322
+ Every proposal records its source path and location and labels the command origin as `project-declared` or `toolchain-defined`. Project-declared examples include a package script, a Python entry point, and an exact validation-contract command. Toolchain-defined examples include Cargo's test command, a named CMake preset invocation, a configured Python runner, and Go's module test command. Validation command strings are converted to argv only when their quoting is complete and they contain no shell operators, expansion, redirection, or environment assignment. Ambiguous entries and `VALIDATION.md` prose receive a manual-entry diagnostic. Directory names such as `build`, `tests`, `python`, or `cargo` never imply a command.
323
+
324
+ `discover` and every mutating command first print an authority preview. Each check shows the destination or active source path, source provenance, exact JSON argv vector, repository-relative cwd, timeout, tags, and effective execution authority. Preview and discovery do not create `.clio-coder`, write a file, or run a check. A mutating command without `--yes` ends after the preview. Repeating the reviewed command with `--yes` is the explicit write decision; the serialized YAML must pass the production catalog parser before the atomic write is reachable.
325
+
326
+ ```text
327
+ clio-coder verifiers discover
328
+ clio-coder verifiers author
329
+ clio-coder verifiers author --exclude cmake-build-debug --rename go-test=go-suite
330
+ clio-coder verifiers author --dry-run go-suite --yes
331
+ clio-coder verifiers validate
332
+ ```
333
+
334
+ `validate` reads the committed file with the same parser used by `verify()`. `dry-run <id>` is an explicit request to execute one admitted check through the production `verify` path. `author --dry-run <id> --yes` writes only after confirmation and starts the selected dry run only after the write is accepted by production discovery.
335
+
336
+ Later changes use the same preview and confirmation boundary. `edit` preserves the ID unless `rename` is requested. Renames and additions reject collisions with catalog IDs and active package-script IDs. Removals state that the deleted command will no longer be executable through catalog authority. Generated IDs are stable for a stable ordered signal set; a collision receives the first available deterministic `-2`, `-3`, and later suffix.
337
+
338
+ ```text
339
+ clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]'
340
+ clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]' --tags scientific,netcdf --yes
341
+ clio-coder verifiers edit validate-grid --timeout-ms 300000
342
+ clio-coder verifiers rename validate-grid validate-regional-grid --yes
343
+ clio-coder verifiers remove validate-regional-grid --yes
344
+ ```
279
345
 
280
- `verify()` with no check lists declared checks grouped by source; today the only source is package.json scripts whose names match the verification family `test*/lint*/build*/typecheck*/check*/format*/ci*` (a family prefix, optionally followed by `:`, `.`, or `-` and a suffix, e.g. `test:unit`). `verify(check="typecheck")` runs `npm run typecheck` through the safe-exec spine with no shell; output is capped at 600000 bytes and `details = {command, cwd, exitCode, durationMs, timedOut, outputCapped}`. A script name outside the family is rejected with a pointer to run it through bash.
346
+ The `add` command is the explicit path for an unsupported or ambiguous project. `--command` must be a JSON argv array, so manual entry still cannot turn a shell command string into executable catalog authority.
281
347
 
282
348
  `verify(check="frontend", path=<file>)` validates an HTML, CSS, or JavaScript artifact without shell access. The path must stay inside the workspace root and end in `.html`, `.htm`, `.css`, `.js`, `.mjs`, or `.cjs`. Checks per type: HTML tag balance (comment-aware, HTML5 optional end tags honored), inline and referenced script syntax (classic scripts parsed in-process, modules via `node --check`), inline and linked CSS brace/string/comment balance, local script and stylesheet references resolved and existence-checked (external and root-relative references are skipped), and an optional headless browser load. `browser="auto"` warns when no chromium/chrome/edge executable is on PATH, `"required"` fails, `"off"` skips. Each check reports pass, warn, fail, or skip; any fail makes the whole result an error. `details = {action: "verify", check: "frontend", path, browserMode, status, checks}`.
283
349
 
284
- Prefer verify over bash for the verification family: the typed result feeds the finish contract as validation evidence.
350
+ Prefer verify over bash for the verification family and project catalog: the typed result feeds the finish contract as validation evidence.
285
351
 
286
352
  ```text
287
353
  verify()
288
354
  verify(check="typecheck")
289
355
  verify(check="test", args=["tests/contracts/dispatch.test.ts"])
356
+ verify(check="rust-workspace")
290
357
  verify(check="frontend", path="site/index.html", browser="off")
291
358
  ```
292
359
 
@@ -1,7 +1,7 @@
1
1
  # Trace store contract
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  Clio's trace database is a rebuildable, queryable mirror. Receipts, session
7
7
  ledgers, gate artifacts, and evidence remain the source of truth. Removing
@@ -1,6 +1,6 @@
1
1
  # Troubleshooting & Error Remediation
2
2
 
3
- This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.3`.
3
+ This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.4`.
4
4
 
5
5
  ---
6
6
 
@@ -1,7 +1,7 @@
1
1
  # Clio TUI Design System
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  This document is the reference specification for the Clio Coder TUI visual layout, styling, and behavior. It describes color semantics, the glyph vocabulary, structural recipes, and state choreography for all surfaces under [src/interactive/](../src/interactive/).
7
7
 
@@ -1,7 +1,7 @@
1
1
  # Worker Dispatch Mechanics
2
2
 
3
3
  > [!TIP]
4
- > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.3).
4
+ > **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.4).
5
5
 
6
6
  This document describes the design and lifecycle of Clio Coder dispatched workers, focusing on the spawning sequence, execution isolation, the standard input/output NDJSON communication loop, and permission escalation routing.
7
7
 
@@ -213,6 +213,16 @@ Receipts carry exactly one integrity version (`RUN_RECEIPT_INTEGRITY_VERSION = 1
213
213
  - **Strict Primitive Handling**: `undefined` object properties are omitted; non-finite numbers (`NaN`, `Infinity`) or `bigint` throw an explicit serialization error.
214
214
  - **Coverage**: Includes every current receipt field and reconstructible ledger field, including route intent/decision/quality, execution role, worker identity, result-contract conformance, node/reroute/gate/plan provenance, briefing, steering, and `outcomeCode`.
215
215
 
216
+ Integrity is only the artifact-integrity axis of the canonical trust status.
217
+ The other axes are validation grounding, independent review, context
218
+ provenance, autonomy enforcement, and completion evidence. Sealing proves that
219
+ the receipt matches its covered ledger facts; it does not verify correctness,
220
+ establish context authorship, turn a correlated review into an independent
221
+ one, or prove completion. Every non-absent canonical fact retains a named
222
+ source and authority plus bounded references to detailed artifacts. The full
223
+ state vocabulary and compatibility map are documented in
224
+ [`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
225
+
216
226
  ### 5.3 Acceptance Coverage
217
227
 
218
228
  The assignment contract's acceptance scenarios map to deterministic contract
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iowarp/clio-coder",
3
- "version": "0.3.3",
3
+ "version": "0.3.4",
4
4
  "description": "Coding agent for HPC and scientific-software developers, part of IOWarp's CLIO ecosystem of agentic science.",
5
5
  "keywords": [
6
6
  "ai",
@@ -77,7 +77,7 @@
77
77
  "pretest": "test -f dist/assets/codewiki.json && [ -z \"$(find src -newer dist/assets/codewiki.json -type f -print -quit)\" ] || npm run build",
78
78
  "test": "node scripts/shard-tests.mjs",
79
79
  "test:coverage": "node scripts/test-coverage.mjs --experimental-test-coverage --test-coverage-include='src/**/*.ts' --test-coverage-exclude='src/**/*.d.ts' 'tests/contracts/**/*.test.ts' 'tests/smoke/**/*.test.ts'",
80
- "test:repeat": "node tests/harness/repeat-tests.mjs",
80
+ "test:repeat": "node scripts/repeat-tests.mjs",
81
81
  "test:trace-viewer": "npm --prefix apps/trace-viewer test",
82
82
  "trace:ui": "node apps/trace-viewer/server.mjs",
83
83
  "ci": "npm run typecheck && npm run lint && npm run skills:check && npm run build && npm run test && npm run test:trace-viewer",
@@ -86,15 +86,12 @@
86
86
  "prepublishOnly": "npm run ci:release",
87
87
  "skills:pin": "node --import tsx scripts/pin-skills.ts",
88
88
  "skills:check": "node --import tsx scripts/pin-skills.ts --check",
89
- "//": "below here: real providers or a live model target, cost money and/or time, never run in CI",
90
- "test:live": "node scripts/live-smoke.mjs",
91
- "test:live-eval": "node scripts/live-eval-recon.mjs",
92
- "test:live-eval:fleet-dispatch": "node scripts/live-eval-fleet-dispatch.mjs",
93
- "test:live-verify:dispatch-routing": "node scripts/live-verify-dispatch-routing.mjs",
94
- "test:lifecycle": "node --import tsx scripts/lifecycle-matrix.mjs",
95
- "bench:swe": "python3 benchmarks/community/swe-bench-lite/swebench_clio.py",
96
- "bench:scicode": "python3 benchmarks/community/scicode/scicode_clio.py",
97
- "bench:tb": "python3 benchmarks/community/clio_fleet.py"
89
+ "//": "below here: a real model target, chosen with --target <id>; costs money and/or GPU time, never run in CI",
90
+ "live:smoke": "node --import tsx benchmarks/internal/live-smoke.ts",
91
+ "live:recon": "node --import tsx benchmarks/internal/live-recon.ts",
92
+ "live:fleet-dispatch": "node --import tsx benchmarks/internal/live-fleet-dispatch.ts",
93
+ "live:tui": "node --import tsx benchmarks/internal/pty-drive.ts",
94
+ "live:home": "node --import tsx benchmarks/internal/live-home.ts"
98
95
  },
99
96
  "dependencies": {
100
97
  "@anthropic-ai/claude-agent-sdk": "0.3.186",
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: clio-test
3
3
  description: Use when writing or modifying Clio Coder's own source under src/, or verifying a change end-to-end against the real test harness. Covers the three real layers (contracts / smoke / boundaries), choosing which to run for a given change, the mock-provider and ACP-over-stdio harness, and the hot-reload dev loop for picking up latest code. Activate on any src/ edit, before declaring a change verified, or when asked whether Clio still works.
4
- version: 0.1.3
4
+ version: 0.1.4
5
5
  license: Apache-2.0
6
6
  clio:
7
7
  registry-id: iowarp/clio-coder
@@ -24,17 +24,15 @@ local-vs-contribute boundary.
24
24
  ## Commands
25
25
 
26
26
  ```bash
27
- npm run typecheck # tsc -p tsconfig.tests.json (includes tests/)
28
- npm run lint # biome check .
29
- npm run check:boundaries # import boundary rules (tsx, no build)
27
+ npm run typecheck # tsc -p tsconfig.tests.json (includes tests/ and benchmarks/internal/)
28
+ npm run lint # biome check . && scripts/check-hygiene.ts (import boundaries, skill pins, doc drift)
30
29
  npm run test:file -- 'tests/contracts/**/*.test.ts' # contract tests (tsx, import src directly, no build)
31
30
  npm run test:file -- 'tests/smoke/**/*.test.ts' # spawn dist/cli/index.js end-to-end (NEEDS a build)
32
31
  npm run test # contracts + smoke, sharded across processes (the full gate)
33
32
  npm run build # tsup -> dist/
34
33
  npm run dev # tsup --watch -> rebuilds dist/ on save
35
34
  npm run ci # typecheck && lint && skills:check && build && test && test:trace-viewer
36
- npm run test:live # local live provider smoke; requires CLIO_CODER_LIVE_SMOKE=1
37
- npm run test:live -- --delegation # adds local opencode/copilot ACP checks
35
+ npm run live:smoke -- --target <id> # one real turn against a configured target; never in CI
38
36
  ```
39
37
 
40
38
  ## Which layer catches what
@@ -44,7 +42,7 @@ npm run test:live -- --delegation # adds local opencode/copilot ACP checks
44
42
  | pure logic in `src/domains/<x>/*.ts` | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | contract tests import `src` via tsx; no build |
45
43
  | dispatch / providers / prompts / safety / config / persistence / acp behavior | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | each has a file in `tests/contracts/` |
46
44
  | skills loader / activation | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | `tests/contracts/skills.test.ts`, `skill-activation-compaction.test.ts` |
47
- | any `src/` import edit | `npm run check:boundaries` | enforces rule1/2/3 |
45
+ | any `src/` import edit | `npm run lint` | the hygiene check enforces rule1/2/3 |
48
46
  | `src/cli/*` or `src/entry/*` user-facing flow | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke spawns the real `dist/cli/index.js` |
49
47
  | ACP surface (`src/cli/acp.ts`, engine ACP) | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke drives `clio-coder acp` over JSON-RPC/stdio |
50
48
 
@@ -54,8 +52,9 @@ a single file.
54
52
  ## Boundary rules you must not break
55
53
 
56
54
  `tests/boundaries/check-boundaries.ts` enforces three rules (also the Hard
57
- Invariants in `CLIO-CODER.md`). If `npm run check:boundaries` reports a
58
- violation, fix the import — never silence the check:
55
+ Invariants in `CLIO-CODER.md`), run by `scripts/check-hygiene.ts` under
56
+ `npm run lint`. If it reports a violation, fix the import — never silence the
57
+ check:
59
58
 
60
59
  - **rule1**: only `src/engine/**` may value-import `@earendil-works/pi-*`. Outside
61
60
  engine, use Clio contracts or type-only imports that erase at compile time.
@@ -70,9 +69,9 @@ There are two independent reload mechanisms; know which applies.
70
69
 
71
70
  **Source reload for tests.** This is the "pick up latest code" loop:
72
71
 
73
- - **Fast loop — no build.** The contracts glob and `check:boundaries` run
74
- `node --import tsx --test` and import `src/**` directly, so they always run the
75
- latest source with zero build step. Iterate here whenever the change is pure
72
+ - **Fast loop — no build.** The contracts glob runs `node --import tsx --test`
73
+ and imports `src/**` directly, and the hygiene lint reads source statically,
74
+ so both always see the latest source with zero build step. Iterate here whenever the change is pure
76
75
  logic or a contract.
77
76
  - **Full loop — needs `dist/`.** The smoke glob spawns `dist/cli/index.js`, so it
78
77
  only sees code that has been built. Keep `npm run dev` (`tsup --watch`) running
@@ -98,7 +97,7 @@ restart the process (against a freshly built `dist/`).
98
97
  1. Write the change.
99
98
  2. `npm run typecheck` and `npm run lint`.
100
99
  3. Run the narrowest layer from the table above.
101
- 4. `npm run check:boundaries` if you touched imports.
100
+ 4. `npm run lint` if you touched imports.
102
101
  5. If you touched CLI/entry/ACP: `npm run build` (or rely on `dev` watch), then
103
102
  `npm run test:file -- 'tests/smoke/**/*.test.ts'`.
104
103
  6. `npm run ci` before calling it done. Report exactly what ran and what is
@@ -113,8 +112,10 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
113
112
 
114
113
  ## What NOT to do
115
114
 
116
- - Don't reintroduce `tests/unit|integration|e2e/` or a pty harness — that
117
- taxonomy was deliberately removed.
115
+ - Don't reintroduce `tests/unit|integration|e2e/`; that taxonomy was
116
+ deliberately removed. Don't add a second pseudo-terminal: `tests/harness/pty.ts`
117
+ is the one PTY, used by the three `*-pty`/`tui-width-matrix` smoke suites and
118
+ by `benchmarks/internal/pty-drive.ts`.
118
119
  - Don't add `scripts/diag-*.ts` or `scripts/verify-*.ts`. A test belongs in
119
120
  `tests/`; a one-off probe belongs in `/tmp` and gets deleted (see
120
121
  `references/harness.md`).
@@ -126,5 +127,7 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
126
127
 
127
128
  ## Harness reference
128
129
 
129
- Driving the real CLI, the mock provider, and ACP over stdio, plus the throwaway
130
- probe pattern: **see `references/harness.md`**.
130
+ Driving the real CLI, the mock provider, ACP over stdio, and the PTY, plus the
131
+ throwaway probe pattern: **see `references/harness.md`**. Driving the real
132
+ binary against a real model (headless, PTY, tmux, herdr) is a different claim
133
+ and lives in `benchmarks/internal/SKILL.md`.
@@ -7,7 +7,7 @@ the gap (it cites the dead unit/integration/e2e taxonomy), then WITH it.
7
7
  Prompt: "I changed pure logic in `src/domains/dispatch/validation.ts`. What do I
8
8
  run and why?"
9
9
  Expected:
10
- - `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `check:boundaries` if imports changed).
10
+ - `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `npm run lint` if imports changed).
11
11
  - Explains contracts import `src` via tsx, so no build is needed.
12
12
  - Does NOT suggest `test:unit` / `test:e2e` (those don't exist).
13
13
 
@@ -26,14 +26,14 @@ Expected:
26
26
  interactive testing. Distinguishes this from config hot-reload (classify.ts).
27
27
 
28
28
  ## T4 — boundary violation
29
- Prompt: "`check:boundaries` says a domain imports another domain's extension.ts.
29
+ Prompt: "`npm run lint` says a domain imports another domain's extension.ts.
30
30
  Quickest fix?"
31
31
  Expected:
32
32
  - Route through the target domain's `index.ts` contract (rule3). Does NOT
33
33
  suggest a `biome-ignore` or exclude.
34
34
 
35
35
  ## Baseline failure modes to watch for (RED)
36
- - Cites `test:unit`/`test:integration`/`test:e2e` or a pty harness.
36
+ - Cites `test:unit`/`test:integration`/`test:e2e`, or a PTY other than `tests/harness/pty.ts`.
37
37
  - Claims smoke tests run against source (they run against `dist/`).
38
38
  - Invents a hot-reload feature that reloads a running session's code.
39
39
 
@@ -1,18 +1,20 @@
1
1
  # Clio test harness reference
2
2
 
3
- How to drive the real Clio binary, a mock provider, and the ACP surface in
4
- tests. All of this is non-interactive there is no pty harness in v0.2.2.
3
+ How to drive the real Clio binary, a mock provider, the ACP surface, and a
4
+ real pseudo-terminal in tests. Every model here is a stub; these are machinery
5
+ tests. A run against a real model is `benchmarks/internal/SKILL.md`.
5
6
 
6
7
  ## Contents
7
8
  - The spawn harness (`runCli`, `makeScratchHome`)
8
9
  - Mocking a provider (OpenAI-compatible SSE fixture)
9
10
  - ACP over JSON-RPC/stdio
11
+ - The PTY (`openPty`, `runInPty`)
10
12
  - One-off probes (no test file)
11
13
 
12
14
  ## The spawn harness
13
15
 
14
- `tests/harness/spawn.ts` is the only harness. It spawns `node dist/cli/index.js`,
15
- so **build first** (or keep `npm run dev` running) before `test:smoke`.
16
+ `tests/harness/spawn.ts` spawns `node dist/cli/index.js` with piped stdio, so
17
+ **build first** (or keep `npm run dev` running) before running smoke.
16
18
 
17
19
  ```ts
18
20
  import { makeScratchHome, runCli } from "../harness/spawn.js";
@@ -76,11 +78,38 @@ client (see `createJsonRpcProcessClient` in the smoke test): `initialize` →
76
78
  non-spec discriminator breaks strict clients like Zed, so the smoke test asserts
77
79
  every emitted variant is in the v1 set.
78
80
 
81
+ ## The PTY
82
+
83
+ Piped stdio reports no terminal width and no TTY, so the TUI refuses to start
84
+ and every width-sensitive path collapses to 80 columns. `tests/harness/pty.ts`
85
+ opens a real pseudo-terminal through `node-pty` (a devDependency, never
86
+ shipped):
87
+
88
+ ```ts
89
+ import { openPty, runInPty, stripAnsi, visibleLines } from "../harness/pty.js";
90
+
91
+ // Scripted: type on a schedule, stop when the output matches, bounded by a timeout.
92
+ const run = await runInPty(process.execPath, [CLI], { cols: 120, rows: 40, cwd, env,
93
+ readyWhen: /ctx /, input: [{ afterMs: 200, data: "/quit\r" }], until: /bye/, timeoutMs: 20_000 });
94
+
95
+ // Controllable: write, resize, pause output, wait for a matcher, wait for exit.
96
+ const session = await openPty(process.execPath, [CLI], { cols: 140, rows: 44, cwd, env });
97
+ await session.waitForOutput((out) => /ctx /.test(stripAnsi(out)), 30_000);
98
+ session.write("/quit\r");
99
+ await session.waitForExit(10_000);
100
+ ```
101
+
102
+ Use it only for what a pipe cannot show: width, raw mode, SIGINT through a
103
+ terminal, the alternate-screen and keyboard-protocol teardown. The three
104
+ suites that need it are `tests/smoke/tui-width-matrix.test.ts`,
105
+ `instant-shell-pty.test.ts`, and `render-trace-pty.test.ts`. Anything else
106
+ belongs on `runCli`.
107
+
79
108
  ## One-off probes (no test file)
80
109
 
81
110
  To poke at Clio without writing a permanent test, drop a throwaway script in
82
- `/tmp` and run it with tsx. Delete it when done — never leave probes under
83
- `tests/` or `scripts/`.
111
+ your scratch directory and run it with tsx. Delete it when done — never leave
112
+ probes under `tests/`, `scripts/`, or `benchmarks/`.
84
113
 
85
114
  ```ts
86
115
  // /tmp/probe.ts
@@ -1,4 +1,4 @@
1
- # Where Clio's tests live (v0.2.2)
1
+ # Where Clio's tests live
2
2
 
3
3
  Three layers under `tests/`. Add a new test next to the closest existing file;
4
4
  create a new file only for a genuinely new domain cluster.
@@ -7,10 +7,17 @@ create a new file only for a genuinely new domain cluster.
7
7
 
8
8
  | Layer | Path | Runner | Build needed |
9
9
  |---|---|---|---|
10
- | contracts | `tests/contracts/*.test.ts` | `node --import tsx --test` | no (imports `src`) |
11
- | smoke | `tests/smoke/*.test.ts` | `node --import tsx --test` | **yes** (spawns `dist/`) |
12
- | boundaries | `tests/boundaries/*.test.ts` | `node --import tsx --test` | no |
13
- | harness (not a test) | `tests/harness/spawn.ts` | imported by smoke | — |
10
+ | contracts | `tests/contracts/*.test.ts` | `npm run test:file -- <glob>` (tsx + scratch root) | no (imports `src`) |
11
+ | smoke | `tests/smoke/*.test.ts` | `npm run test:file -- <glob>` | **yes** (spawns `dist/`) |
12
+ | boundaries | `tests/boundaries/check-boundaries.ts` | `npm run lint` (hygiene) | no |
13
+ | harness (not tests) | `tests/harness/*.ts` | imported by contracts and smoke | — |
14
+
15
+ The harness modules: `spawn.ts` (run the built CLI with pipes), `scratch-env.ts`
16
+ (isolated Clio home), `pty.ts` (a real pseudo-terminal), `openai-compat-fixture.ts`
17
+ and `fake-lmstudio-server.ts` (stub providers), `fake-ssh.ts` (stub fleet node),
18
+ `clock.ts` (steppable clock), plus dispatch, receipt, and module-graph helpers.
19
+ Everything under `tests/` stubs the model. Real-model runs are
20
+ `benchmarks/internal/` and never run under `npm test`.
14
21
 
15
22
  ## Contract test files
16
23
 
@@ -33,18 +40,21 @@ create a new file only for a genuinely new domain cluster.
33
40
  | Area | File |
34
41
  |---|---|
35
42
  | non-interactive CLI + ACP-over-stdio end-to-end | `tests/smoke/cli.test.ts` |
36
- | import boundary rules (rule1/2/3) | `tests/boundaries/boundaries.test.ts` |
37
- | boundary checker implementation | `tests/boundaries/check-boundaries.ts` |
43
+ | the package as installed from `npm pack` | `tests/smoke/pack-install.test.ts` |
44
+ | TUI at real terminal sizes, NO_COLOR, Ctrl-C teardown (PTY) | `tests/smoke/tui-width-matrix.test.ts` |
45
+ | instant shell before hydration, SIGTERM through the lease (PTY) | `tests/smoke/instant-shell-pty.test.ts` |
46
+ | committed-frame render trace under PTY backpressure (PTY) | `tests/smoke/render-trace-pty.test.ts` |
47
+ | import boundary rules (rule1/2/3), run under `npm run lint` | `tests/boundaries/check-boundaries.ts` |
38
48
 
39
49
  ## Running a subset
40
50
 
41
51
  ```bash
42
52
  # all contracts
43
- node --import tsx --test 'tests/contracts/**/*.test.ts'
53
+ npm run test:file -- 'tests/contracts/**/*.test.ts'
44
54
  # one file
45
- node --import tsx --test tests/contracts/skills.test.ts
55
+ npm run test:file -- tests/contracts/skills.test.ts
46
56
  # only it.only / describe.only within a file
47
- node --import tsx --test --test-only tests/contracts/skills.test.ts
57
+ npm run test:file -- --test-only tests/contracts/skills.test.ts
48
58
  ```
49
59
 
50
60
  ## Writing tests
@@ -61,8 +61,8 @@ skills:
61
61
  sha256: 386ebfdac9379ebf5630fa7a9af259bb45f473f7d3d679b0a7ef4164f5fc4dea
62
62
  - name: clio-test
63
63
  path: meta/clio-test
64
- version: 0.1.3
65
- sha256: 99683ea7d7bf5bf57788e58ee6c66e01a4e30ec8d3d7d0a1f8b2b85caff4ae5a
64
+ version: 0.1.4
65
+ sha256: 90ca1fe7660cd8c3d835801883ebd55d2d17cb652b8585b28ddb82aec642336e
66
66
  - name: credentials
67
67
  path: meta/credentials
68
68
  version: 0.1.2
@@ -109,7 +109,7 @@
109
109
  "name": "clio-test",
110
110
  "description": "Use when writing or modifying Clio Coder's own source under src/, or verifying a change end-to-end against the real test harness. Covers the three real layers (contracts / smoke / boundaries), choosing which to run for a given change, the mock-provider and ACP-over-stdio harness, and the hot-reload dev loop for picking up latest code. Activate on any src/ edit, before declaring a change verified, or when asked whether Clio still works.",
111
111
  "sourceUrl": "https://github.com/iowarp/clio-coder/tree/main/skills/meta/clio-test",
112
- "version": "0.1.3",
112
+ "version": "0.1.4",
113
113
  "audit": "pass",
114
114
  "category": "meta"
115
115
  },