@iowarp/clio-coder 0.3.2 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +269 -458
- package/CONTRIBUTING.md +1 -1
- package/README.md +3 -3
- package/dist/{acp-BIYHVZIM.js → acp-S5R4RR5B.js} +7 -6
- package/dist/{agents-YT6SSRIT.js → agents-P6DMMVZY.js} +24 -21
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5TWEIYDN.js → auth-2XCZLPKS.js} +12 -8
- package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
- package/dist/{chunk-WMSVI4G2.js → chunk-2LZI5CAG.js} +133 -13
- package/dist/{chunk-OAO4GE4M.js → chunk-2TZWSW76.js} +2 -2
- package/dist/{chunk-OOJYHWRB.js → chunk-34475P3I.js} +2 -2
- package/dist/{chunk-WVO7V2QY.js → chunk-35MKKU5R.js} +4 -4
- package/dist/{chunk-LBNRH5WM.js → chunk-3HZ5RWN2.js} +5 -5
- package/dist/{chunk-AGYYIBLL.js → chunk-3JLKSKD7.js} +2 -2
- package/dist/{chunk-MBS4V7ZP.js → chunk-4JUF2NNX.js} +7 -7
- package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
- package/dist/chunk-5M54SPOL.js +926 -0
- package/dist/{chunk-STBPMHSX.js → chunk-7RXG6QRZ.js} +51 -11
- package/dist/{chunk-77VKQEHF.js → chunk-A2GZF7DC.js} +5 -5
- package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
- package/dist/chunk-AOCYTWAV.js +449 -0
- package/dist/chunk-BEY543CS.js +258 -0
- package/dist/{chunk-6N5PTWMY.js → chunk-BP4OYD6A.js} +32 -13
- package/dist/chunk-BPGS2WCQ.js +612 -0
- package/dist/{chunk-J5HN4RYU.js → chunk-BRXQQJFP.js} +8 -8
- package/dist/chunk-CFGTUFWB.js +67 -0
- package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
- package/dist/{chunk-G2DE3C7R.js → chunk-EDRHSCIE.js} +4 -4
- package/dist/{chunk-4KLWL3UC.js → chunk-EFADSJET.js} +2 -2
- package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
- package/dist/chunk-FYYLNIL5.js +313 -0
- package/dist/{chunk-EPVUXGXG.js → chunk-HV5X7OR2.js} +14 -12
- package/dist/{chunk-TZTZS7QK.js → chunk-HXG4IURW.js} +5 -3
- package/dist/{chunk-IGLFWIYI.js → chunk-K6WL7QZT.js} +3 -3
- package/dist/chunk-K7VKOLQQ.js +15 -0
- package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
- package/dist/{chunk-V4RXGQ5Q.js → chunk-KRPY7NTG.js} +10 -7
- package/dist/chunk-LL4KHSZI.js +22 -0
- package/dist/{chunk-KJ5LWLOE.js → chunk-MEQ45TQ4.js} +15 -9
- package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
- package/dist/{chunk-AO4RKG4M.js → chunk-N4CZJQRK.js} +5 -5
- package/dist/{chunk-ARBGF5F7.js → chunk-NILBFAPG.js} +14 -8
- package/dist/chunk-OZNBF4L3.js +23 -0
- package/dist/{verify-G6V4D2G7.js → chunk-PCZJO5TI.js} +127 -42
- package/dist/chunk-QQK64KLB.js +1360 -0
- package/dist/{chunk-6EJV5X2W.js → chunk-QQL5RT5M.js} +979 -1619
- package/dist/{chunk-LZSJBIVT.js → chunk-QWU7ZBO7.js} +70 -720
- package/dist/{chunk-2EHAIA3X.js → chunk-RD5U66HV.js} +3 -3
- package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
- package/dist/{chunk-OQ33BKR3.js → chunk-TTNYS3EA.js} +3 -60
- package/dist/chunk-TW3WDMVS.js +677 -0
- package/dist/chunk-TZSKNMZG.js +434 -0
- package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
- package/dist/{chunk-7EYHLWU7.js → chunk-UZHIZC5S.js} +7 -7
- package/dist/{chunk-QTYWRVRA.js → chunk-VAWWTKDP.js} +8 -8
- package/dist/{chunk-X75S7HFS.js → chunk-VEZEGCGW.js} +214 -20
- package/dist/{chunk-OHHN2SO4.js → chunk-VMNQ6OZA.js} +98 -202
- package/dist/chunk-VSNATDE6.js +122 -0
- package/dist/chunk-W6GROXXM.js +69 -0
- package/dist/chunk-WPQLXFOZ.js +375 -0
- package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
- package/dist/{chunk-3ZXDFGR5.js → chunk-X6COSD2O.js} +5 -5
- package/dist/chunk-ZGVHUX3M.js +66 -0
- package/dist/{chunk-MAW544W2.js → chunk-ZWMF7253.js} +4 -4
- package/dist/{chunk-MQSRRFWA.js → chunk-ZYKPLLNQ.js} +563 -546
- package/dist/cli/index.js +27 -23
- package/dist/{clio-4LY5K2AC.js → clio-J5JIOIDS.js} +7 -6
- package/dist/{code-nav-7AX6FYE6.js → code-nav-AXCXSBHX.js} +5 -3
- package/dist/{config-GTLUW2PR.js → config-OEBMIN2U.js} +37 -27
- package/dist/{configure-R6A64DHX.js → configure-PUQOSIXQ.js} +16 -13
- package/dist/{context-5VKGUVJJ.js → context-EKDCKUUZ.js} +82 -7
- package/dist/{context-RW5HC47S.js → context-MGSE4Z2T.js} +33 -23
- package/dist/{context-JFZEJ7W5.js → context-URSXPBCK.js} +17 -9
- package/dist/{context-clear-6ZHBAZZT.js → context-clear-KDAJRNUK.js} +33 -23
- package/dist/context-working-set-SBKMPPI2.js +1552 -0
- package/dist/{dispatch-runner-VKBRCWQC.js → dispatch-runner-MSWN72NK.js} +43 -29
- package/dist/{doctor-KI767GSN.js → doctor-7BSE27PJ.js} +10 -10
- package/dist/{eval-XSSNATB4.js → eval-IZGDOO4H.js} +9 -8
- package/dist/{evidence-UA6AWDQQ.js → evidence-SR7WXB5B.js} +51 -23
- package/dist/{evolve-QNTFGV6Z.js → evolve-K7VE2CBX.js} +30 -20
- package/dist/{fleet-Q7UOMUSG.js → fleet-7XMJNQNF.js} +48 -38
- package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-AQNAH644.js} +3 -3
- package/dist/{init-WBB65ZHQ.js → init-JGNPAYXT.js} +41 -31
- package/dist/{memory-MD3O64RI.js → memory-4ALKDJ4Q.js} +32 -22
- package/dist/{models-BZU34YWD.js → models-ZMMLFJNN.js} +22 -19
- package/dist/{monitor-MEQA5C3I.js → monitor-2F3T5KHP.js} +55 -43
- package/dist/{orchestrator-CGFKEP27.js → orchestrator-ORHT43JB.js} +2507 -1896
- package/dist/{reset-L2FQEE3E.js → reset-NXGTYNUO.js} +4 -3
- package/dist/{run-IV4Q6RLN.js → run-RF4WJGMT.js} +51 -41
- package/dist/{share-S5BZQC5I.js → share-UT3W6E4M.js} +5 -4
- package/dist/{skills-LQEKRDTN.js → skills-PSACKC5Q.js} +2 -2
- package/dist/{skills-eval-3DC4HEWS.js → skills-eval-WJSI55RZ.js} +34 -24
- package/dist/{targets-C4SSGQOB.js → targets-PIIRAOYS.js} +23 -20
- package/dist/{terminal-lease-IT5JW2NR.js → terminal-lease-ULWXWNVY.js} +5 -4
- package/dist/{upgrade-7TT7SQ3G.js → upgrade-346TZ6AV.js} +18 -17
- package/dist/{usage-GV4PKT3M.js → usage-6KKXR32N.js} +34 -24
- package/dist/verifiers-4UUM6TEE.js +1214 -0
- package/dist/verify-X5HDROLA.js +25 -0
- package/dist/{wiki-generate-DQF6Z66B.js → wiki-generate-7STOCIFZ.js} +42 -31
- package/dist/worker/entry.js +33 -24
- package/docs/README.md +8 -7
- package/docs/acp.md +1 -1
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +53 -21
- package/docs/config-knobs-audit.md +1 -2
- package/docs/configuration-and-targets.md +15 -1
- package/docs/context-engine.md +64 -12
- package/docs/context-working-set.md +194 -0
- package/docs/development-pipeline.md +1 -1
- package/docs/documentation-coverage.md +5 -5
- package/docs/documentation-guide.md +6 -5
- package/docs/environment-variables.md +2 -1
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +14 -1
- package/docs/evidence-and-memory.md +74 -2
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +2 -2
- package/docs/fleet-dispatch.md +22 -7
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +6 -6
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +7 -9
- package/docs/observability.md +4 -4
- package/docs/performance-methodology.md +2 -2
- package/docs/proactive-memory.md +1 -1
- package/docs/prompt-envelope-and-tools.md +4 -4
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +35 -35
- package/docs/safety-model.md +23 -4
- package/docs/scientific-validation.md +21 -3
- package/docs/session-lifecycle.md +3 -3
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +79 -12
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +1 -1
- package/docs/tui-design.md +2 -2
- package/docs/worker-dispatch-mechanics.md +11 -1
- package/package.json +8 -11
- package/skills/meta/clio-test/SKILL.md +20 -17
- package/skills/meta/clio-test/evals.md +3 -3
- package/skills/meta/clio-test/references/harness.md +35 -6
- package/skills/meta/clio-test/references/test-map.md +20 -10
- package/skills/registry.yaml +2 -2
- package/skills/skill-marketplace.json +1 -1
- package/src/cli/context-working-set.ts +513 -0
- package/src/cli/context.ts +8 -0
- package/src/cli/evidence.ts +20 -2
- package/src/cli/index.ts +4 -0
- package/src/cli/verifiers.ts +325 -0
- package/src/core/bash-exec.ts +39 -14
- package/src/core/bus-events.ts +19 -4
- package/src/core/config.ts +54 -0
- package/src/core/defaults.ts +50 -3
- package/src/core/git-commit-attribution.ts +46 -21
- package/src/core/verification-scripts.ts +6 -0
- package/src/domains/agents/builtins/verifier.md +3 -0
- package/src/domains/config/classify.ts +1 -0
- package/src/domains/config/keybindings.ts +3 -3
- package/src/domains/context/working-set/contract.ts +161 -0
- package/src/domains/context/working-set/defaults.ts +28 -0
- package/src/domains/context/working-set/engine.ts +203 -0
- package/src/domains/context/working-set/fold.ts +62 -0
- package/src/domains/context/working-set/horizon.ts +38 -0
- package/src/domains/context/working-set/marker.ts +103 -0
- package/src/domains/context/working-set/path-index.ts +436 -0
- package/src/domains/context/working-set/payload.ts +152 -0
- package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
- package/src/domains/context/working-set/policies/index.ts +21 -0
- package/src/domains/context/working-set/policies/structural.ts +160 -0
- package/src/domains/context/working-set/project.ts +132 -0
- package/src/domains/context/working-set/protect.ts +109 -0
- package/src/domains/context/working-set/recall.ts +177 -0
- package/src/domains/context/working-set/replay/controls.ts +112 -0
- package/src/domains/context/working-set/replay/load-clio.ts +199 -0
- package/src/domains/context/working-set/replay/metrics.ts +185 -0
- package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
- package/src/domains/context/working-set/replay/report.ts +139 -0
- package/src/domains/context/working-set/replay/runner.ts +325 -0
- package/src/domains/context/working-set/replay/synthetic.ts +422 -0
- package/src/domains/context/working-set/replay/trace.ts +21 -0
- package/src/domains/context/working-set/visible.ts +54 -0
- package/src/domains/evidence/build.ts +112 -45
- package/src/domains/evidence/eval.ts +24 -7
- package/src/domains/evidence/index.ts +53 -0
- package/src/domains/evidence/ordering.ts +12 -0
- package/src/domains/evidence/run-trust.ts +221 -0
- package/src/domains/evidence/store.ts +46 -6
- package/src/domains/evidence/trust-status.ts +854 -0
- package/src/domains/evidence/types.ts +26 -0
- package/src/domains/middleware/memory-intervention.ts +3 -0
- package/src/domains/middleware/stalled-turn.ts +165 -4
- package/src/domains/safety/autonomy.ts +1 -1
- package/src/domains/safety/default-path-policy.ts +8 -0
- package/src/domains/safety/finish-contract.ts +4 -3
- package/src/domains/safety/policy-engine.ts +48 -6
- package/src/domains/session/compaction/compact.ts +23 -1
- package/src/domains/session/compaction/cut-point.ts +2 -0
- package/src/domains/session/compaction/tokens.ts +16 -1
- package/src/domains/session/context-ledger.ts +2 -0
- package/src/domains/session/entries.ts +107 -1
- package/src/domains/session/manager.ts +9 -2
- package/src/domains/session/migrations/index.ts +22 -3
- package/src/engine/acp/server.ts +3 -0
- package/src/engine/agent.ts +18 -1
- package/src/engine/session.ts +9 -3
- package/src/entry/orchestrator.ts +16 -4
- package/src/interactive/chat-loop-messages.ts +18 -6
- package/src/interactive/chat-panel.ts +571 -244
- package/src/interactive/chat-renderer.ts +79 -39
- package/src/interactive/context-meter.ts +10 -0
- package/src/interactive/context-overlay.ts +81 -6
- package/src/interactive/context-recall-command.ts +110 -0
- package/src/interactive/editor-submit.ts +26 -1
- package/src/interactive/footer/widgets.ts +22 -20
- package/src/interactive/footer-panel.ts +6 -1
- package/src/interactive/interactive-application.ts +2 -0
- package/src/interactive/interactive-event-projection.ts +12 -0
- package/src/interactive/interactive-slash-runtime.ts +49 -8
- package/src/interactive/model-session-replay.ts +21 -0
- package/src/interactive/overlay-general-openers.ts +6 -0
- package/src/interactive/overlay-session-lifecycle.ts +8 -4
- package/src/interactive/overlays/ask-user.ts +146 -24
- package/src/interactive/renderers/tool-execution.ts +167 -56
- package/src/interactive/session-transcript.ts +2 -2
- package/src/interactive/slash-commands.ts +29 -2
- package/src/interactive/status/index.ts +12 -1
- package/src/interactive/status/reasoning.ts +87 -0
- package/src/interactive/status/summary.ts +13 -2
- package/src/interactive/transcript-detail.ts +120 -0
- package/src/interactive/turn-context.ts +238 -88
- package/src/interactive/turn-middleware.ts +6 -6
- package/src/tools/agent-tools.ts +11 -4
- package/src/tools/bash.ts +144 -82
- package/src/tools/builtin-tool-catalog.ts +18 -6
- package/src/tools/context/index.ts +105 -3
- package/src/tools/context/surface.ts +3 -2
- package/src/tools/core-bootstrap.ts +21 -0
- package/src/tools/dispatch-runner.ts +9 -7
- package/src/tools/monitor.ts +28 -20
- package/src/tools/presentation.ts +107 -0
- package/src/tools/registry.ts +65 -7
- package/src/tools/result-disposition.ts +550 -0
- package/src/tools/result-shaping.ts +262 -19
- package/src/tools/safe-exec.ts +2 -0
- package/src/tools/verify/authoring.ts +1119 -0
- package/src/tools/verify/catalog.ts +346 -0
- package/src/tools/verify/index.ts +13 -3
- package/src/tools/verify/scripts.ts +135 -37
- package/src/tools/verify/surface.ts +9 -5
- package/src/tools/worker-evidence.ts +35 -12
- package/dist/chunk-MNA4JGU4.js +0 -255
- package/dist/chunk-SRF2PJNW.js +0 -184
- package/dist/chunk-T6YILFSB.js +0 -80
- package/dist/chunk-VAKQQHWR.js +0 -434
package/docs/safety-model.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio Coder Safety Model
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder's safety posture is code-enforced, not prompt-only. As the orchestrator coding agent in the [IOWarp](https://iowarp.ai) ecosystem developed by the [Gnosis Research Center](https://grc.iit.edu) at Illinois Tech under NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318), Clio gates execution by target capabilities, the tool registry, the safety policy engine, project policies, protected-artifact checks, and audit receipts.
|
|
7
7
|
|
|
@@ -13,7 +13,7 @@ Source of truth: `src/domains/safety/**`, `src/tools/registry.ts`, `src/tools/bo
|
|
|
13
13
|
|
|
14
14
|
The `autonomy` setting (`read-only` | `suggest` | `auto-edit` | `full-auto`) is an enforced dial. It controls exactly one thing: which action classes run immediately, which park for operator approval, and which are auto-denied. The safety net (damage-control rules, path policy, protected artifacts, loop guard, dispatch scope admission) is independent of the dial and identical at every level. When a `[safety-net]` notice appears at full-auto, that is the always-on net working as designed, not a contradiction of the level.
|
|
15
15
|
|
|
16
|
-
In Clio Coder v0.3.
|
|
16
|
+
In Clio Coder v0.3.4, effective autonomy resolution is strictly centralized in `src/entry/orchestrator.ts` through `resolveEffectiveAutonomy` and `resolveBaselineAutonomy`. Every admission surface (tool registry admission, dispatch plan provenance, and ACP session snapshots) delegates to this pair of functions so that fallback paths cannot diverge across execution contexts. `resolveBaselineAutonomy` evaluates dispatch settings overrides, headless CLI options, and configuration settings before applying the default `auto-edit` level. `resolveEffectiveAutonomy` combines any active ACP session autonomy level with the baseline resolution.
|
|
17
17
|
|
|
18
18
|
### Autonomy levels
|
|
19
19
|
|
|
@@ -132,6 +132,8 @@ Shell operators split two ways. Unrecognized sequencing and redirection (`||`, `
|
|
|
132
132
|
|
|
133
133
|
Bash `cwd` is resolved under the workspace root. Escaping the workspace is blocked unless a reviewed project policy permits the exact command/cwd combination.
|
|
134
134
|
|
|
135
|
+
Bash `output_policy` changes only the canonical model-context disposition after execution; it never changes command classification, autonomy, cwd containment, environment filtering, process-group termination, or the 16 MiB capture ceiling. Omitted/`bounded` retains the diagnostic tail. `summary` uses a deterministic code path with repository secret redaction and bounded head, tail, and error-like lines. `metadata-only` retains success/failure, exit, signal, timeout, abort, output-cap, byte-size, and retrieval facts without stdout or stderr in model context. `full` is appropriate only for known-small results and records a typed downgrade plus retrieval when the hard result budget cannot admit it. Operator presentation remains independently tail-biased, and only the terminal result may write its single retained scratch artifact.
|
|
136
|
+
|
|
135
137
|
---
|
|
136
138
|
|
|
137
139
|
## Policy Engine Evaluation Order
|
|
@@ -230,9 +232,17 @@ Command entry notes:
|
|
|
230
232
|
Prefer typed tools over Bash:
|
|
231
233
|
|
|
232
234
|
- `git` (op=status/diff/log) uses fixed command vectors.
|
|
233
|
-
- `verify(check="<
|
|
235
|
+
- `verify(check="<id>")` runs either a declared package.json verification script (the `test*/lint*/build*/typecheck*/check*/format*/ci*` family) or an exact version-1 `.clio-coder/verifiers.yaml` argv vector through bounded execution helpers with no shell; `verify()` lists both sources through one canonical check projection.
|
|
234
236
|
- `verify(check="frontend", path=...)` validates frontend artifacts without granting arbitrary shell access.
|
|
235
237
|
|
|
238
|
+
A package-script check and the frontend validator are in the no-prompt set at `auto-edit`: both are bounded by the verification-script family and a fixed argv shape. A project-catalog check is not. The engine resolves the check id against `.clio-coder/verifiers.yaml` on every call and treats the declared argv exactly like a bash command string: the damage-control rules and the zero-access read guard scan it, and it is tagged unrecognized, so `auto-edit` parks it for one confirmation that shows the argv and `full-auto` runs it. `.clio-coder/verifiers.yaml` and `.clio-coder/safety.yaml` are read-only to the model's `write`, `edit`, and bash redirect paths through the default path policy: both files are operator authority, and a model that could author either one could widen its own permissions in two tool calls.
|
|
239
|
+
|
|
240
|
+
The project verifier catalog is an executable authority supplied by the repository, not by model prose. Its schema rejects unknown fields, shell strings and shell executables, invalid or duplicate IDs, oversized values, absolute or escaping working directories, unsupported versions, and collisions with package-provider IDs. A catalog entry fixes argv, repository-relative cwd, and timeout. Tool-call `args`, `cwd`, timeout, output-cap, or environment-shaped fields cannot widen it. Safe-exec uses `spawn` without a shell, filters the child environment to the Clio allowlist, honors cancellation, and reports exact argv and termination evidence.
|
|
241
|
+
|
|
242
|
+
`clio-coder verifiers discover` and `clio-coder verifiers author` do not grant authority during inspection. They read only declared package, Cargo, CMake preset, Python runner, Go module, and YAML validation-command signals and render exact argv vectors with provenance. The preview names the catalog path, cwd, timeout, tags, and authority consequence for every check. Toolchain conventions are labeled separately from literal project declarations. Ambiguous validation prose and directory-only hints are rejected with a JSON argv manual-entry path.
|
|
243
|
+
|
|
244
|
+
Authoring validation serializes the proposed catalog in memory and passes it to the production catalog parser. Discovery, revision, and preview cannot reach the filesystem writer or verifier executor. A reviewed mutating CLI invocation must be repeated with `--yes` before an atomic catalog write is reachable. An optional authoring dry run begins only after that decision and goes through the production `verify` execution path. Editing, renaming, and removal use the same preview boundary; collisions fail before writing, and removal explicitly revokes that catalog command's execution authority.
|
|
245
|
+
|
|
236
246
|
The frontend check accepts `.html`, `.htm`, `.css`, `.js`, `.mjs`, and `.cjs` under the workspace root. It checks HTML tag balance, local script/style references, JavaScript syntax, CSS brace/comment/string balance, and optionally loads HTML with an available headless Chromium/Chrome/Edge executable (`browser: auto|required|off`).
|
|
237
247
|
|
|
238
248
|
The `edit` tool also carries conservative matching rules. It preserves
|
|
@@ -344,10 +354,19 @@ On every settled `turn_end`, the finish-contract assessor scans entries since th
|
|
|
344
354
|
The assessor decision order is:
|
|
345
355
|
|
|
346
356
|
1. If the window has no successful mutating receipt or settled mutating `!` bash execution, the contract passes with `no_mutation`.
|
|
347
|
-
2. If the window has validation evidence, the contract passes with `validation_evidence`. Evidence includes successful validation commands, `verify` checks (declared
|
|
357
|
+
2. If the window has validation evidence, the contract passes with `validation_evidence`. Evidence includes successful validation commands, `verify` checks (declared package scripts, admitted project-catalog entries, and the frontend check), passed dispatch receipts, and protected-artifact validation records.
|
|
348
358
|
3. If the assistant explicitly states what could not be verified and why, the contract passes with `explicit_limitation`.
|
|
349
359
|
4. Otherwise, the contract engages with `unvalidated_mutation`.
|
|
350
360
|
|
|
361
|
+
The finish assessment projects only onto the canonical completion-evidence
|
|
362
|
+
axis. `validation_evidence` becomes `evidenced`, `unvalidated_mutation` becomes
|
|
363
|
+
`incomplete`, `explicit_limitation` becomes `limited`, and `no_mutation`
|
|
364
|
+
becomes `not_applicable`. The worker receipt's autonomy grade belongs to the
|
|
365
|
+
separate autonomy-enforcement axis. Even `enforced` autonomy cannot promote
|
|
366
|
+
completion evidence. The canonical state vocabulary, attribution rules, and
|
|
367
|
+
persisted-format compatibility table are documented in
|
|
368
|
+
[`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
|
|
369
|
+
|
|
351
370
|
- **Normal Rigor**: Clio issues a soft advisory warning (`FINISH_CONTRACT_ADVISORY_MESSAGE`) injected as a reminder for the next turn, but permits the turn to settle.
|
|
352
371
|
- **High Rigor**: Clio withholds completion. The assessor emits `request_continuation` and a warning `inject_reminder` carrying `HIGH_RIGOR_REVALIDATION_MESSAGE`, instructing the model to run a verification-family command (e.g. `npm test`, `npm run build`) or explicitly declare a limitation before ending.
|
|
353
372
|
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
# Clio Coder Scientific Validation Contracts
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Scientific software development cannot treat simple file presence as proof of correctness. A simulation script that crashes on rank 48, or writes out NetCDF arrays filled with `NaN`s, may still successfully write a file to the disk.
|
|
7
7
|
|
|
8
|
-
Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.3.
|
|
8
|
+
Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.3.4, the session rigor resolver does not parse or enforce a scientific contract schema. The presence of `.clio-coder/validation.yaml`, `.clio-coder/validation.yml`, `validation.yaml`, `validation.yml`, or `VALIDATION.md` at the workspace root raises the default rigor level to `high`; the file contents are advisory material for developers, project agents, and external validators.
|
|
9
|
+
|
|
10
|
+
This advisory convention is separate from the executable project verifier catalog at `.clio-coder/verifiers.yaml`. The verifier catalog has a strict version-1 schema and admits exact argv vectors to the `verify` tool. Scientific validation contracts and handbook expectations do not grant command authority: prose such as `validators: ["python tools/check_grid.py"]` remains guidance until the project owner confirms the equivalent argv, cwd, timeout, and tags in `verifiers.yaml`. The executable catalog does not interpret numerical tolerances or artifact expectations; it only runs the explicitly declared process vector through safe-exec.
|
|
11
|
+
|
|
12
|
+
`clio-coder verifiers author` can inspect top-level `validators` entries in the YAML contract filenames above and propose catalog checks. It labels those vectors as project-declared and shows their source index, exact argv, cwd, timeout, tags, catalog path, and resulting execution authority. This inspection is read-only. A command string with sound quoting and no shell operator can be represented as argv for review; shell expansion, pipes, redirection, environment assignments, incomplete quoting, and Markdown prose receive a manual JSON-argv diagnostic. Nothing becomes executable and nothing is dry-run until the operator confirms the catalog write with `--yes`.
|
|
9
13
|
|
|
10
14
|
The convention below is a recommended shape for scientific projects that need to document expected dimensions, attributes, numerical tolerances, scheduler context, and verification commands for scientific artifacts. Developed at the [Gnosis Research Center (GRC)](https://grc.iit.edu) at Illinois Tech as part of the NSF-funded scientific-software context (NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318)), this convention links execution metadata with physical output checks without claiming that the current harness executes those checks automatically.
|
|
11
15
|
|
|
@@ -51,6 +55,20 @@ notes: |
|
|
|
51
55
|
Re-run check_grid.py after job completion is observed.
|
|
52
56
|
```
|
|
53
57
|
|
|
58
|
+
The `validators` values above are intentionally advisory shell-like prose. Preview the exact catalog proposal with `clio-coder verifiers author`, or declare the Python validator manually without granting free-form shell interpretation:
|
|
59
|
+
|
|
60
|
+
```yaml
|
|
61
|
+
# .clio-coder/verifiers.yaml
|
|
62
|
+
version: 1
|
|
63
|
+
checks:
|
|
64
|
+
- id: validate-grid
|
|
65
|
+
description: Validate the generated regional grid
|
|
66
|
+
command: [python, tools/check_grid.py, out/region_west.nc]
|
|
67
|
+
cwd: .
|
|
68
|
+
timeoutMs: 120000
|
|
69
|
+
tags: [scientific, netcdf]
|
|
70
|
+
```
|
|
71
|
+
|
|
54
72
|
### Suggested Fields:
|
|
55
73
|
1. **`version`:** Set to `1` for project-local compatibility.
|
|
56
74
|
2. **`runtime.kind`:** Document execution mode (`local`, `slurm`, `mpi`, or `other`).
|
|
@@ -77,7 +95,7 @@ Comparing floating-point values in scientific computations must accommodate roun
|
|
|
77
95
|
|
|
78
96
|
## Common Scientific Artifact Families
|
|
79
97
|
|
|
80
|
-
The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.3.
|
|
98
|
+
The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.3.4:
|
|
81
99
|
|
|
82
100
|
- **`HDF5` / `NetCDF` / `Zarr`:** Multi-dimensional scientific array files.
|
|
83
101
|
- **`FITS`:** Flexible Image Transport System (used in astrophysics).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Session Lifecycle
|
|
2
2
|
|
|
3
|
-
This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.3.
|
|
3
|
+
This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.3.4`.
|
|
4
4
|
|
|
5
5
|
Source implementations: `src/engine/session.ts` and `src/domains/session/`.
|
|
6
6
|
|
|
@@ -39,11 +39,11 @@ export interface ClioSessionMeta {
|
|
|
39
39
|
piMonoVersion: string;
|
|
40
40
|
platform: string;
|
|
41
41
|
nodeVersion: string;
|
|
42
|
-
sessionFormatVersion?: number; // CURRENT_SESSION_FORMAT_VERSION =
|
|
42
|
+
sessionFormatVersion?: number; // CURRENT_SESSION_FORMAT_VERSION = 4
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
46
|
-
Format version `CURRENT_SESSION_FORMAT_VERSION =
|
|
46
|
+
Format version `CURRENT_SESSION_FORMAT_VERSION = 4` (`src/engine/session.ts`) is stamped on all sessions created since the working-set layer landed. Version 4 adds the `contextEviction` and `contextRecall` ledger kinds. `runMigrations` in `src/domains/session/migrations/` rejects both directions on `/resume`: a missing or earlier version names the remedy (remove the session directory), and a version from the future says the session was written by a newer Clio and must not be read by this build.
|
|
47
47
|
|
|
48
48
|
---
|
|
49
49
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Skills Marketplace
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
The Skills Hub (`/skill`) shows project skills, user skills, and the marketplace. Every marketplace row comes from the same local lookup that `clio-coder skills install <name>` and `/skill <name>` resolve through, so the hub lists nothing it cannot install.
|
|
7
7
|
|
package/docs/tool-usage.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
# Tool Usage Reference
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
|
|
7
7
|
|
|
8
|
-
In Clio Coder v0.3.
|
|
8
|
+
In Clio Coder v0.3.4, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
|
|
9
9
|
|
|
10
10
|
## Observation envelope: truncation notices, offload, next hints, and the turn budget
|
|
11
11
|
|
|
@@ -21,7 +21,7 @@ Truncated text results append exactly one notice line:
|
|
|
21
21
|
|
|
22
22
|
Segments that do not apply are omitted. `<total>` renders as `N+` when the search stopped early at its item limit, so the true total was never counted. `next:` is an exact argument fragment (for example `limit=200` or `offset=451`); re-issue the same call with that argument changed to continue.
|
|
23
23
|
|
|
24
|
-
Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<
|
|
24
|
+
Offload: when the byte cap cut content that was already collected, the complete rendering is written to `<clio-coder state dir>/scratch/<sessionId>/<sha256 of the captured text>.txt` and the notice's `full:` segment names the path. Read it with `read` using offset/limit. Tools offload only when the byte cap cut collected content; a bare item-limit truncation continues via `next` and does not offload. `read` never offloads, because the source file is directly re-addressable via `offset`.
|
|
25
25
|
|
|
26
26
|
JSON-format results (code_nav, context scope=docs/workspace) never get an appended notice. An oversize JSON payload is replaced whole by the parseable stub `{"error":"result exceeded <cap>","offloadPath":"...","next":"..."}` so the model never receives JSON cut mid-document. Empty results are also valid JSON with empty arrays and `next` populated.
|
|
27
27
|
|
|
@@ -102,17 +102,24 @@ Arguments:
|
|
|
102
102
|
- `command` (required).
|
|
103
103
|
- `cwd` (optional). Working directory; resolved against the session workspace and rejected when it escapes it. The safety net blocks an escaping cwd at admission, and the tool enforces the same rule itself.
|
|
104
104
|
- `timeout_ms` (optional). Default 300000 (5 minutes).
|
|
105
|
+
- `output_policy` (optional). Canonical model-context disposition: `full`, `bounded`, `summary`, or `metadata-only`. Omission is exactly `bounded`.
|
|
105
106
|
|
|
106
107
|
Workspace containment: commands whose filesystem targets resolve outside the session workspace escalate to `system_modify` and ask for one-shot confirmation at every autonomy level (headless runs deny asks). Recognized targets are shell redirects, `tee`/`mkdir`/`touch` path operands, `cp`/`mv`/`ln` destinations, in-place `sed -i` operands, and any `cd`/`pushd` whose directory leaves the workspace, since a `cd` outside re-bases every relative path that follows it. Inside-workspace equivalents stay plain `execute` with no new prompts.
|
|
107
108
|
|
|
108
|
-
|
|
109
|
+
The default `bounded` policy keeps a tail-biased model excerpt under the 16KB result budget, because the failing assertion, compiler error, and exit summary usually live at the end. `summary` is useful for noisy builds and test runs: code deterministically selects a bounded head, tail, and error-like lines, applies Clio's repository secret redactor, and records the source hash and algorithm in summary provenance. `metadata-only` is appropriate when the model needs only outcome and termination facts; stdout and stderr stay out of model context while the operator presentation, retained byte size, and retrieval path remain available. `full` is for output known to be small. It is admitted only when the complete captured result and its facts fit the bounded result/context budget; otherwise the result explicitly records a typed downgrade to tail-biased `bounded` and provides retrieval. Do not use `full` as the routine default.
|
|
109
110
|
|
|
110
|
-
|
|
111
|
+
Presentation is independent from model context. The operator-facing display remains folded and tail-biased under every policy. When the display or selected context omits captured content, the terminal result writes one per-session scratch artifact and names it in the result. Live updates use the selected policy, remain bounded, and never write per-update artifacts. Every terminal result records requested and applied context modes, captured/displayed/context bytes, truncation or downgrade state, and any offload path. Exit code, signal, timeout, abort, and output-cap facts survive every policy. Scratch retrieval may contain the raw retained output; the deterministic `summary` projection is the redacted surface.
|
|
112
|
+
|
|
113
|
+
A command producing more than 16MB of combined output is stopped with an error. UTF-8 decoding spans process chunks, and a code point split by the hard byte cap is discarded rather than replaced with an invalid character. Raw NUL bytes are removed from model context under every policy, which leaves multi-byte code points and ANSI escape sequences whole; the operator presentation and the scratch artifact keep the captured bytes, and the result still records the omission and its retrieval path. A timeout, abort, output cap, or nonzero exit preserves captured diagnostics and appends a status line such as `bash: command timed out after <ms>ms` or `bash: command failed (exit N)` before canonical shaping.
|
|
114
|
+
|
|
115
|
+
Reach for bash for builds, git, package managers, and anything without a dedicated tool. Prefer the dedicated tools over their shell equivalents: grep/find/read/ls get envelope truncation, exact continuation hints, and the shared ignore policy that `cat`, shell `grep`, and shell `find` do not. Prefer `verify` over bash for declared package scripts and project-catalog entries, since verify produces typed evidence.
|
|
111
116
|
|
|
112
117
|
```text
|
|
113
118
|
bash(command="git status --short")
|
|
114
119
|
bash(command="git log --oneline -10")
|
|
115
120
|
bash(command="npm run build", timeout_ms=600000)
|
|
121
|
+
bash(command="npm run test", timeout_ms=600000, output_policy="summary")
|
|
122
|
+
bash(command="make artifact", output_policy="metadata-only")
|
|
116
123
|
```
|
|
117
124
|
|
|
118
125
|
## grep: search file contents with ripgrep
|
|
@@ -266,27 +273,87 @@ dispatch(tasks=["Refactor step 1", "Refactor step 2"], mode="sequential", timeou
|
|
|
266
273
|
|
|
267
274
|
## verify: run declared verification checks
|
|
268
275
|
|
|
269
|
-
One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/frontend.ts`.
|
|
276
|
+
One EXECUTE entry point for declared verification. Sources: `src/tools/verify/index.ts`, `src/tools/verify/catalog.ts`, `src/tools/verify/scripts.ts`, `src/tools/verify/authoring.ts`, `src/tools/verify/frontend.ts`.
|
|
270
277
|
|
|
271
278
|
Arguments:
|
|
272
279
|
|
|
273
|
-
- `check` (optional). A declared package.json script name or `"frontend"`. Omit to list available checks.
|
|
280
|
+
- `check` (optional). A declared project-catalog ID, package.json script name, or `"frontend"`. Omit to list available checks.
|
|
274
281
|
- `path` (check=frontend). Artifact file under the workspace root.
|
|
275
|
-
- `args` (
|
|
282
|
+
- `args` (package scripts only). Extra arguments passed after `--`. A JSON-string array is tolerated and parsed. Project-catalog checks ignore this field.
|
|
276
283
|
- `browser` (check=frontend). `auto` (default), `required`, or `off`.
|
|
277
|
-
- `cwd` (
|
|
278
|
-
- `timeout_ms` (
|
|
284
|
+
- `cwd` (package scripts only). Package working directory. Project catalogs are always discovered at the session workspace root, and a project check uses its declared `cwd`.
|
|
285
|
+
- `timeout_ms` (package scripts and frontend only). Default 120000. A project check uses its declared `timeoutMs`.
|
|
286
|
+
- `max_output_bytes` (package scripts and frontend only). Default 600000. Project checks retain the safe-exec default cap.
|
|
287
|
+
|
|
288
|
+
`verify()` lists checks grouped as `package.json` and `.clio-coder/verifiers.yaml`. Both providers project through the same canonical metadata: `{id, description, command, cwd, timeoutMs, tags, source}`. Package scripts must match the verification family `test*/lint*/build*/typecheck*/check*/format*/ci*` (a family prefix, optionally followed by `:`, `.`, or `-` and a suffix, e.g. `test:unit`). `verify(check="typecheck")` runs `npm run typecheck` through the safe-exec spine with no shell. A package script name outside the family is rejected with a pointer to run it through bash.
|
|
289
|
+
|
|
290
|
+
### Project verifier catalog
|
|
291
|
+
|
|
292
|
+
Projects may commit a versioned executable catalog at `.clio-coder/verifiers.yaml`:
|
|
293
|
+
|
|
294
|
+
```yaml
|
|
295
|
+
version: 1
|
|
296
|
+
checks:
|
|
297
|
+
- id: rust-workspace
|
|
298
|
+
description: Run the Rust workspace tests
|
|
299
|
+
command: [cargo, test, --workspace]
|
|
300
|
+
cwd: .
|
|
301
|
+
timeoutMs: 600000
|
|
302
|
+
tags: [rust, test]
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
Version 1 is strict. Every root and check field shown above is required, unknown fields fail, and duplicate IDs fail. A project ID uses lowercase letters, digits, `.`, `_`, `:`, or `-`, begins with a letter or digit, and is at most 64 UTF-8 bytes. `frontend` is reserved. Descriptions are trimmed single-line text capped at 512 bytes. `command` is a nonempty argv array with at most 64 entries and 4096 bytes per entry. A shell command string is invalid, and explicit shell executables such as `sh`, `bash`, `pwsh`, and `cmd` are rejected. `cwd` is a repository-relative existing directory capped at 512 bytes; absolute paths, `..` escapes, and symbolic-link escapes fail. `timeoutMs` is a positive integer capped at 900000. A check may carry at most 16 distinct lowercase tags of at most 32 bytes each. The whole file is capped at 262144 bytes and may contain at most 128 checks. YAML aliases are disabled.
|
|
306
|
+
|
|
307
|
+
Provider IDs share one namespace. If a catalog ID collides with a discovered package script, listing and execution fail and identify both source files. Catalog parsing also fails closed before any package or project check runs.
|
|
308
|
+
|
|
309
|
+
`verify(check="rust-workspace")` spawns exactly `cargo` with `test` and `--workspace`; it does not interpolate model text or invoke a shell. Model-supplied `args`, `cwd`, `timeout_ms`, `max_output_bytes`, or undeclared environment fields cannot widen or replace the catalog entry. Safe execution passes only Clio's small environment allowlist, applies cancellation and the declared timeout, and shapes output at the standard 600000-byte cap. Execution details retain the compatible command string plus exact `argv`, `cwd`, `exitCode`, `durationMs`, `aborted`, `timedOut`, and `outputCapped` evidence, along with the check's declared source, command, cwd, timeout, description, and tags.
|
|
310
|
+
|
|
311
|
+
### Guided catalog authoring
|
|
312
|
+
|
|
313
|
+
An empty `verify()` result points to `clio-coder verifiers author`. The authoring command inspects only command-bearing files at the workspace root:
|
|
314
|
+
|
|
315
|
+
- verification-family package scripts, projected exactly as `npm run <script>` and shown as already active rather than duplicated into the catalog;
|
|
316
|
+
- `Cargo.toml`, projected to Cargo's package or workspace test vector;
|
|
317
|
+
- visible build and test entries in `CMakePresets.json`, projected to the corresponding `cmake --build --preset` or `ctest --preset` vector;
|
|
318
|
+
- declared Python runners in `pyproject.toml`, `pytest.ini`, `tox.ini`, `noxfile.py`, or the pytest section of `setup.cfg`;
|
|
319
|
+
- a module directive in `go.mod`, projected to `go test ./...`;
|
|
320
|
+
- top-level `validators` entries in the documented YAML scientific-validation files.
|
|
321
|
+
|
|
322
|
+
Every proposal records its source path and location and labels the command origin as `project-declared` or `toolchain-defined`. Project-declared examples include a package script, a Python entry point, and an exact validation-contract command. Toolchain-defined examples include Cargo's test command, a named CMake preset invocation, a configured Python runner, and Go's module test command. Validation command strings are converted to argv only when their quoting is complete and they contain no shell operators, expansion, redirection, or environment assignment. Ambiguous entries and `VALIDATION.md` prose receive a manual-entry diagnostic. Directory names such as `build`, `tests`, `python`, or `cargo` never imply a command.
|
|
323
|
+
|
|
324
|
+
`discover` and every mutating command first print an authority preview. Each check shows the destination or active source path, source provenance, exact JSON argv vector, repository-relative cwd, timeout, tags, and effective execution authority. Preview and discovery do not create `.clio-coder`, write a file, or run a check. A mutating command without `--yes` ends after the preview. Repeating the reviewed command with `--yes` is the explicit write decision; the serialized YAML must pass the production catalog parser before the atomic write is reachable.
|
|
325
|
+
|
|
326
|
+
```text
|
|
327
|
+
clio-coder verifiers discover
|
|
328
|
+
clio-coder verifiers author
|
|
329
|
+
clio-coder verifiers author --exclude cmake-build-debug --rename go-test=go-suite
|
|
330
|
+
clio-coder verifiers author --dry-run go-suite --yes
|
|
331
|
+
clio-coder verifiers validate
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
`validate` reads the committed file with the same parser used by `verify()`. `dry-run <id>` is an explicit request to execute one admitted check through the production `verify` path. `author --dry-run <id> --yes` writes only after confirmation and starts the selected dry run only after the write is accepted by production discovery.
|
|
335
|
+
|
|
336
|
+
Later changes use the same preview and confirmation boundary. `edit` preserves the ID unless `rename` is requested. Renames and additions reject collisions with catalog IDs and active package-script IDs. Removals state that the deleted command will no longer be executable through catalog authority. Generated IDs are stable for a stable ordered signal set; a collision receives the first available deterministic `-2`, `-3`, and later suffix.
|
|
337
|
+
|
|
338
|
+
```text
|
|
339
|
+
clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]'
|
|
340
|
+
clio-coder verifiers add --id validate-grid --description "Validate the regional grid" --command '["python","tools/check_grid.py","out/region_west.nc"]' --tags scientific,netcdf --yes
|
|
341
|
+
clio-coder verifiers edit validate-grid --timeout-ms 300000
|
|
342
|
+
clio-coder verifiers rename validate-grid validate-regional-grid --yes
|
|
343
|
+
clio-coder verifiers remove validate-regional-grid --yes
|
|
344
|
+
```
|
|
279
345
|
|
|
280
|
-
`
|
|
346
|
+
The `add` command is the explicit path for an unsupported or ambiguous project. `--command` must be a JSON argv array, so manual entry still cannot turn a shell command string into executable catalog authority.
|
|
281
347
|
|
|
282
348
|
`verify(check="frontend", path=<file>)` validates an HTML, CSS, or JavaScript artifact without shell access. The path must stay inside the workspace root and end in `.html`, `.htm`, `.css`, `.js`, `.mjs`, or `.cjs`. Checks per type: HTML tag balance (comment-aware, HTML5 optional end tags honored), inline and referenced script syntax (classic scripts parsed in-process, modules via `node --check`), inline and linked CSS brace/string/comment balance, local script and stylesheet references resolved and existence-checked (external and root-relative references are skipped), and an optional headless browser load. `browser="auto"` warns when no chromium/chrome/edge executable is on PATH, `"required"` fails, `"off"` skips. Each check reports pass, warn, fail, or skip; any fail makes the whole result an error. `details = {action: "verify", check: "frontend", path, browserMode, status, checks}`.
|
|
283
349
|
|
|
284
|
-
Prefer verify over bash for the verification family: the typed result feeds the finish contract as validation evidence.
|
|
350
|
+
Prefer verify over bash for the verification family and project catalog: the typed result feeds the finish contract as validation evidence.
|
|
285
351
|
|
|
286
352
|
```text
|
|
287
353
|
verify()
|
|
288
354
|
verify(check="typecheck")
|
|
289
355
|
verify(check="test", args=["tests/contracts/dispatch.test.ts"])
|
|
356
|
+
verify(check="rust-workspace")
|
|
290
357
|
verify(check="frontend", path="site/index.html", browser="off")
|
|
291
358
|
```
|
|
292
359
|
|
package/docs/trace-store.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Trace store contract
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio's trace database is a rebuildable, queryable mirror. Receipts, session
|
|
7
7
|
ledgers, gate artifacts, and evidence remain the source of truth. Removing
|
package/docs/troubleshooting.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Troubleshooting & Error Remediation
|
|
2
2
|
|
|
3
|
-
This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.
|
|
3
|
+
This guide provides concrete, actionable remediation procedures for operational errors, permission denials, target connection failures, and system diagnostics in Clio Coder `v0.3.4`.
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
package/docs/tui-design.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio TUI Design System
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
This document is the reference specification for the Clio Coder TUI visual layout, styling, and behavior. It describes color semantics, the glyph vocabulary, structural recipes, and state choreography for all surfaces under [src/interactive/](../src/interactive/).
|
|
7
7
|
|
|
@@ -152,7 +152,7 @@ The Clio screen maintains a responsive, four-zone structure: the launchpad / ses
|
|
|
152
152
|
|
|
153
153
|
In fullscreen mode, `PageUp` and `PageDown` scroll one viewport, `Home` and `End` jump to its bounds, `Ctrl+Shift+Up` and `Ctrl+Shift+Down` jump between semantic prompts, and the mouse wheel scrolls the transcript. Dragging the scrollbar thumb moves the viewport directly. `terminal.fullscreenScrollbar` is `hidden`, `auto` (visible during interaction), or `always`. Manual scrolling suspends follow-end so new output does not steal the operator's position; returning to the bottom resumes it. Both fullscreen settings are restart-scoped because Clio constructs its terminal renderer and component graph once at startup.
|
|
154
154
|
|
|
155
|
-
`terminal.smoothStreaming` controls presentation-only pacing of derived assistant text and thinking. `off`, the 0.3.
|
|
155
|
+
`terminal.smoothStreaming` controls presentation-only pacing of derived assistant text and thinking. `off`, the 0.3.3 release default, is the existing immediate 16 ms coalescer. `auto` paces only on a capable local TTY and bypasses pacing for non-TTY, SSH, multiplexers, CI, screen-reader/reduced-motion markers, or observed stdout backpressure. `on` explicitly requests grapheme-safe pacing, while still stopping frame production behind stdout backpressure. Raw provider wrappers never enter the panel, canonical events and persistence remain synchronous, and tool/message/turn/abort/retry/submit/teardown boundaries drain visible state before they continue. `CLIO_CODER_SMOOTH_STREAM` is the one-process escape hatch and takes precedence over settings; invalid values resolve to `off`.
|
|
156
156
|
|
|
157
157
|
Interactive startup uses one terminal lease across both boot stages. Stage 0 owns the terminal, renderer, root host, exact editor instance, input decoder, raw mode, resize subscription, protocol queries, signals, and stop lifecycle, and commits a measured minimal frame while services hydrate. Hydration synchronously swaps the root and input/signal delegates without reconstructing the editor or initializing terminal protocols again. Early Enter submissions become immutable, visibly queued admissions and drain once through the ordinary command pipeline; a later draft and cursor stay in the same editor. Boot failure or an early signal closes the lease exactly once, restores the terminal, and prints recoverable queued input and draft text. `CLIO_CODER_INSTANT_SHELL=0` selects the legacy fully hydrated first frame; ACP, headless, ordinary non-TTY invocation, and subcommand execution never acquire the lease. An explicit `CLIO_CODER_INTERACTIVE=1` retains its established force-interactive behavior on a non-TTY stream.
|
|
158
158
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Worker Dispatch Mechanics
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
This document describes the design and lifecycle of Clio Coder dispatched workers, focusing on the spawning sequence, execution isolation, the standard input/output NDJSON communication loop, and permission escalation routing.
|
|
7
7
|
|
|
@@ -213,6 +213,16 @@ Receipts carry exactly one integrity version (`RUN_RECEIPT_INTEGRITY_VERSION = 1
|
|
|
213
213
|
- **Strict Primitive Handling**: `undefined` object properties are omitted; non-finite numbers (`NaN`, `Infinity`) or `bigint` throw an explicit serialization error.
|
|
214
214
|
- **Coverage**: Includes every current receipt field and reconstructible ledger field, including route intent/decision/quality, execution role, worker identity, result-contract conformance, node/reroute/gate/plan provenance, briefing, steering, and `outcomeCode`.
|
|
215
215
|
|
|
216
|
+
Integrity is only the artifact-integrity axis of the canonical trust status.
|
|
217
|
+
The other axes are validation grounding, independent review, context
|
|
218
|
+
provenance, autonomy enforcement, and completion evidence. Sealing proves that
|
|
219
|
+
the receipt matches its covered ledger facts; it does not verify correctness,
|
|
220
|
+
establish context authorship, turn a correlated review into an independent
|
|
221
|
+
one, or prove completion. Every non-absent canonical fact retains a named
|
|
222
|
+
source and authority plus bounded references to detailed artifacts. The full
|
|
223
|
+
state vocabulary and compatibility map are documented in
|
|
224
|
+
[`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
|
|
225
|
+
|
|
216
226
|
### 5.3 Acceptance Coverage
|
|
217
227
|
|
|
218
228
|
The assignment contract's acceptance scenarios map to deterministic contract
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@iowarp/clio-coder",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.4",
|
|
4
4
|
"description": "Coding agent for HPC and scientific-software developers, part of IOWarp's CLIO ecosystem of agentic science.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -77,7 +77,7 @@
|
|
|
77
77
|
"pretest": "test -f dist/assets/codewiki.json && [ -z \"$(find src -newer dist/assets/codewiki.json -type f -print -quit)\" ] || npm run build",
|
|
78
78
|
"test": "node scripts/shard-tests.mjs",
|
|
79
79
|
"test:coverage": "node scripts/test-coverage.mjs --experimental-test-coverage --test-coverage-include='src/**/*.ts' --test-coverage-exclude='src/**/*.d.ts' 'tests/contracts/**/*.test.ts' 'tests/smoke/**/*.test.ts'",
|
|
80
|
-
"test:repeat": "node
|
|
80
|
+
"test:repeat": "node scripts/repeat-tests.mjs",
|
|
81
81
|
"test:trace-viewer": "npm --prefix apps/trace-viewer test",
|
|
82
82
|
"trace:ui": "node apps/trace-viewer/server.mjs",
|
|
83
83
|
"ci": "npm run typecheck && npm run lint && npm run skills:check && npm run build && npm run test && npm run test:trace-viewer",
|
|
@@ -86,15 +86,12 @@
|
|
|
86
86
|
"prepublishOnly": "npm run ci:release",
|
|
87
87
|
"skills:pin": "node --import tsx scripts/pin-skills.ts",
|
|
88
88
|
"skills:check": "node --import tsx scripts/pin-skills.ts --check",
|
|
89
|
-
"//": "below here:
|
|
90
|
-
"
|
|
91
|
-
"
|
|
92
|
-
"
|
|
93
|
-
"
|
|
94
|
-
"
|
|
95
|
-
"bench:swe": "python3 benchmarks/community/swe-bench-lite/swebench_clio.py",
|
|
96
|
-
"bench:scicode": "python3 benchmarks/community/scicode/scicode_clio.py",
|
|
97
|
-
"bench:tb": "python3 benchmarks/community/clio_fleet.py"
|
|
89
|
+
"//": "below here: a real model target, chosen with --target <id>; costs money and/or GPU time, never run in CI",
|
|
90
|
+
"live:smoke": "node --import tsx benchmarks/internal/live-smoke.ts",
|
|
91
|
+
"live:recon": "node --import tsx benchmarks/internal/live-recon.ts",
|
|
92
|
+
"live:fleet-dispatch": "node --import tsx benchmarks/internal/live-fleet-dispatch.ts",
|
|
93
|
+
"live:tui": "node --import tsx benchmarks/internal/pty-drive.ts",
|
|
94
|
+
"live:home": "node --import tsx benchmarks/internal/live-home.ts"
|
|
98
95
|
},
|
|
99
96
|
"dependencies": {
|
|
100
97
|
"@anthropic-ai/claude-agent-sdk": "0.3.186",
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: clio-test
|
|
3
3
|
description: Use when writing or modifying Clio Coder's own source under src/, or verifying a change end-to-end against the real test harness. Covers the three real layers (contracts / smoke / boundaries), choosing which to run for a given change, the mock-provider and ACP-over-stdio harness, and the hot-reload dev loop for picking up latest code. Activate on any src/ edit, before declaring a change verified, or when asked whether Clio still works.
|
|
4
|
-
version: 0.1.
|
|
4
|
+
version: 0.1.4
|
|
5
5
|
license: Apache-2.0
|
|
6
6
|
clio:
|
|
7
7
|
registry-id: iowarp/clio-coder
|
|
@@ -24,17 +24,15 @@ local-vs-contribute boundary.
|
|
|
24
24
|
## Commands
|
|
25
25
|
|
|
26
26
|
```bash
|
|
27
|
-
npm run typecheck # tsc -p tsconfig.tests.json (includes tests/)
|
|
28
|
-
npm run lint # biome check .
|
|
29
|
-
npm run check:boundaries # import boundary rules (tsx, no build)
|
|
27
|
+
npm run typecheck # tsc -p tsconfig.tests.json (includes tests/ and benchmarks/internal/)
|
|
28
|
+
npm run lint # biome check . && scripts/check-hygiene.ts (import boundaries, skill pins, doc drift)
|
|
30
29
|
npm run test:file -- 'tests/contracts/**/*.test.ts' # contract tests (tsx, import src directly, no build)
|
|
31
30
|
npm run test:file -- 'tests/smoke/**/*.test.ts' # spawn dist/cli/index.js end-to-end (NEEDS a build)
|
|
32
31
|
npm run test # contracts + smoke, sharded across processes (the full gate)
|
|
33
32
|
npm run build # tsup -> dist/
|
|
34
33
|
npm run dev # tsup --watch -> rebuilds dist/ on save
|
|
35
34
|
npm run ci # typecheck && lint && skills:check && build && test && test:trace-viewer
|
|
36
|
-
npm run
|
|
37
|
-
npm run test:live -- --delegation # adds local opencode/copilot ACP checks
|
|
35
|
+
npm run live:smoke -- --target <id> # one real turn against a configured target; never in CI
|
|
38
36
|
```
|
|
39
37
|
|
|
40
38
|
## Which layer catches what
|
|
@@ -44,7 +42,7 @@ npm run test:live -- --delegation # adds local opencode/copilot ACP checks
|
|
|
44
42
|
| pure logic in `src/domains/<x>/*.ts` | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | contract tests import `src` via tsx; no build |
|
|
45
43
|
| dispatch / providers / prompts / safety / config / persistence / acp behavior | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | each has a file in `tests/contracts/` |
|
|
46
44
|
| skills loader / activation | `npm run test:file -- 'tests/contracts/**/*.test.ts'` | `tests/contracts/skills.test.ts`, `skill-activation-compaction.test.ts` |
|
|
47
|
-
| any `src/` import edit | `npm run
|
|
45
|
+
| any `src/` import edit | `npm run lint` | the hygiene check enforces rule1/2/3 |
|
|
48
46
|
| `src/cli/*` or `src/entry/*` user-facing flow | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke spawns the real `dist/cli/index.js` |
|
|
49
47
|
| ACP surface (`src/cli/acp.ts`, engine ACP) | build, then `npm run test:file -- 'tests/smoke/**/*.test.ts'` | smoke drives `clio-coder acp` over JSON-RPC/stdio |
|
|
50
48
|
|
|
@@ -54,8 +52,9 @@ a single file.
|
|
|
54
52
|
## Boundary rules you must not break
|
|
55
53
|
|
|
56
54
|
`tests/boundaries/check-boundaries.ts` enforces three rules (also the Hard
|
|
57
|
-
Invariants in `CLIO-CODER.md`)
|
|
58
|
-
violation, fix the import — never silence the
|
|
55
|
+
Invariants in `CLIO-CODER.md`), run by `scripts/check-hygiene.ts` under
|
|
56
|
+
`npm run lint`. If it reports a violation, fix the import — never silence the
|
|
57
|
+
check:
|
|
59
58
|
|
|
60
59
|
- **rule1**: only `src/engine/**` may value-import `@earendil-works/pi-*`. Outside
|
|
61
60
|
engine, use Clio contracts or type-only imports that erase at compile time.
|
|
@@ -70,9 +69,9 @@ There are two independent reload mechanisms; know which applies.
|
|
|
70
69
|
|
|
71
70
|
**Source reload for tests.** This is the "pick up latest code" loop:
|
|
72
71
|
|
|
73
|
-
- **Fast loop — no build.** The contracts glob
|
|
74
|
-
|
|
75
|
-
latest source with zero build step. Iterate here whenever the change is pure
|
|
72
|
+
- **Fast loop — no build.** The contracts glob runs `node --import tsx --test`
|
|
73
|
+
and imports `src/**` directly, and the hygiene lint reads source statically,
|
|
74
|
+
so both always see the latest source with zero build step. Iterate here whenever the change is pure
|
|
76
75
|
logic or a contract.
|
|
77
76
|
- **Full loop — needs `dist/`.** The smoke glob spawns `dist/cli/index.js`, so it
|
|
78
77
|
only sees code that has been built. Keep `npm run dev` (`tsup --watch`) running
|
|
@@ -98,7 +97,7 @@ restart the process (against a freshly built `dist/`).
|
|
|
98
97
|
1. Write the change.
|
|
99
98
|
2. `npm run typecheck` and `npm run lint`.
|
|
100
99
|
3. Run the narrowest layer from the table above.
|
|
101
|
-
4. `npm run
|
|
100
|
+
4. `npm run lint` if you touched imports.
|
|
102
101
|
5. If you touched CLI/entry/ACP: `npm run build` (or rely on `dev` watch), then
|
|
103
102
|
`npm run test:file -- 'tests/smoke/**/*.test.ts'`.
|
|
104
103
|
6. `npm run ci` before calling it done. Report exactly what ran and what is
|
|
@@ -113,8 +112,10 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
|
|
|
113
112
|
|
|
114
113
|
## What NOT to do
|
|
115
114
|
|
|
116
|
-
- Don't reintroduce `tests/unit|integration|e2e
|
|
117
|
-
|
|
115
|
+
- Don't reintroduce `tests/unit|integration|e2e/`; that taxonomy was
|
|
116
|
+
deliberately removed. Don't add a second pseudo-terminal: `tests/harness/pty.ts`
|
|
117
|
+
is the one PTY, used by the three `*-pty`/`tui-width-matrix` smoke suites and
|
|
118
|
+
by `benchmarks/internal/pty-drive.ts`.
|
|
118
119
|
- Don't add `scripts/diag-*.ts` or `scripts/verify-*.ts`. A test belongs in
|
|
119
120
|
`tests/`; a one-off probe belongs in `/tmp` and gets deleted (see
|
|
120
121
|
`references/harness.md`).
|
|
@@ -126,5 +127,7 @@ node --import tsx --test --test-only tests/contracts/<file>.test.ts # it.only
|
|
|
126
127
|
|
|
127
128
|
## Harness reference
|
|
128
129
|
|
|
129
|
-
Driving the real CLI, the mock provider,
|
|
130
|
-
probe pattern: **see `references/harness.md`**.
|
|
130
|
+
Driving the real CLI, the mock provider, ACP over stdio, and the PTY, plus the
|
|
131
|
+
throwaway probe pattern: **see `references/harness.md`**. Driving the real
|
|
132
|
+
binary against a real model (headless, PTY, tmux, herdr) is a different claim
|
|
133
|
+
and lives in `benchmarks/internal/SKILL.md`.
|
|
@@ -7,7 +7,7 @@ the gap (it cites the dead unit/integration/e2e taxonomy), then WITH it.
|
|
|
7
7
|
Prompt: "I changed pure logic in `src/domains/dispatch/validation.ts`. What do I
|
|
8
8
|
run and why?"
|
|
9
9
|
Expected:
|
|
10
|
-
- `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `
|
|
10
|
+
- `npm run test:file -- 'tests/contracts/**/*.test.ts'` (and `npm run lint` if imports changed).
|
|
11
11
|
- Explains contracts import `src` via tsx, so no build is needed.
|
|
12
12
|
- Does NOT suggest `test:unit` / `test:e2e` (those don't exist).
|
|
13
13
|
|
|
@@ -26,14 +26,14 @@ Expected:
|
|
|
26
26
|
interactive testing. Distinguishes this from config hot-reload (classify.ts).
|
|
27
27
|
|
|
28
28
|
## T4 — boundary violation
|
|
29
|
-
Prompt: "`
|
|
29
|
+
Prompt: "`npm run lint` says a domain imports another domain's extension.ts.
|
|
30
30
|
Quickest fix?"
|
|
31
31
|
Expected:
|
|
32
32
|
- Route through the target domain's `index.ts` contract (rule3). Does NOT
|
|
33
33
|
suggest a `biome-ignore` or exclude.
|
|
34
34
|
|
|
35
35
|
## Baseline failure modes to watch for (RED)
|
|
36
|
-
- Cites `test:unit`/`test:integration`/`test:e2e
|
|
36
|
+
- Cites `test:unit`/`test:integration`/`test:e2e`, or a PTY other than `tests/harness/pty.ts`.
|
|
37
37
|
- Claims smoke tests run against source (they run against `dist/`).
|
|
38
38
|
- Invents a hot-reload feature that reloads a running session's code.
|
|
39
39
|
|