@iowarp/clio-coder 0.3.2 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +269 -458
- package/CONTRIBUTING.md +1 -1
- package/README.md +3 -3
- package/dist/{acp-BIYHVZIM.js → acp-S5R4RR5B.js} +7 -6
- package/dist/{agents-YT6SSRIT.js → agents-P6DMMVZY.js} +24 -21
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5TWEIYDN.js → auth-2XCZLPKS.js} +12 -8
- package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
- package/dist/{chunk-WMSVI4G2.js → chunk-2LZI5CAG.js} +133 -13
- package/dist/{chunk-OAO4GE4M.js → chunk-2TZWSW76.js} +2 -2
- package/dist/{chunk-OOJYHWRB.js → chunk-34475P3I.js} +2 -2
- package/dist/{chunk-WVO7V2QY.js → chunk-35MKKU5R.js} +4 -4
- package/dist/{chunk-LBNRH5WM.js → chunk-3HZ5RWN2.js} +5 -5
- package/dist/{chunk-AGYYIBLL.js → chunk-3JLKSKD7.js} +2 -2
- package/dist/{chunk-MBS4V7ZP.js → chunk-4JUF2NNX.js} +7 -7
- package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
- package/dist/chunk-5M54SPOL.js +926 -0
- package/dist/{chunk-STBPMHSX.js → chunk-7RXG6QRZ.js} +51 -11
- package/dist/{chunk-77VKQEHF.js → chunk-A2GZF7DC.js} +5 -5
- package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
- package/dist/chunk-AOCYTWAV.js +449 -0
- package/dist/chunk-BEY543CS.js +258 -0
- package/dist/{chunk-6N5PTWMY.js → chunk-BP4OYD6A.js} +32 -13
- package/dist/chunk-BPGS2WCQ.js +612 -0
- package/dist/{chunk-J5HN4RYU.js → chunk-BRXQQJFP.js} +8 -8
- package/dist/chunk-CFGTUFWB.js +67 -0
- package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
- package/dist/{chunk-G2DE3C7R.js → chunk-EDRHSCIE.js} +4 -4
- package/dist/{chunk-4KLWL3UC.js → chunk-EFADSJET.js} +2 -2
- package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
- package/dist/chunk-FYYLNIL5.js +313 -0
- package/dist/{chunk-EPVUXGXG.js → chunk-HV5X7OR2.js} +14 -12
- package/dist/{chunk-TZTZS7QK.js → chunk-HXG4IURW.js} +5 -3
- package/dist/{chunk-IGLFWIYI.js → chunk-K6WL7QZT.js} +3 -3
- package/dist/chunk-K7VKOLQQ.js +15 -0
- package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
- package/dist/{chunk-V4RXGQ5Q.js → chunk-KRPY7NTG.js} +10 -7
- package/dist/chunk-LL4KHSZI.js +22 -0
- package/dist/{chunk-KJ5LWLOE.js → chunk-MEQ45TQ4.js} +15 -9
- package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
- package/dist/{chunk-AO4RKG4M.js → chunk-N4CZJQRK.js} +5 -5
- package/dist/{chunk-ARBGF5F7.js → chunk-NILBFAPG.js} +14 -8
- package/dist/chunk-OZNBF4L3.js +23 -0
- package/dist/{verify-G6V4D2G7.js → chunk-PCZJO5TI.js} +127 -42
- package/dist/chunk-QQK64KLB.js +1360 -0
- package/dist/{chunk-6EJV5X2W.js → chunk-QQL5RT5M.js} +979 -1619
- package/dist/{chunk-LZSJBIVT.js → chunk-QWU7ZBO7.js} +70 -720
- package/dist/{chunk-2EHAIA3X.js → chunk-RD5U66HV.js} +3 -3
- package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
- package/dist/{chunk-OQ33BKR3.js → chunk-TTNYS3EA.js} +3 -60
- package/dist/chunk-TW3WDMVS.js +677 -0
- package/dist/chunk-TZSKNMZG.js +434 -0
- package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
- package/dist/{chunk-7EYHLWU7.js → chunk-UZHIZC5S.js} +7 -7
- package/dist/{chunk-QTYWRVRA.js → chunk-VAWWTKDP.js} +8 -8
- package/dist/{chunk-X75S7HFS.js → chunk-VEZEGCGW.js} +214 -20
- package/dist/{chunk-OHHN2SO4.js → chunk-VMNQ6OZA.js} +98 -202
- package/dist/chunk-VSNATDE6.js +122 -0
- package/dist/chunk-W6GROXXM.js +69 -0
- package/dist/chunk-WPQLXFOZ.js +375 -0
- package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
- package/dist/{chunk-3ZXDFGR5.js → chunk-X6COSD2O.js} +5 -5
- package/dist/chunk-ZGVHUX3M.js +66 -0
- package/dist/{chunk-MAW544W2.js → chunk-ZWMF7253.js} +4 -4
- package/dist/{chunk-MQSRRFWA.js → chunk-ZYKPLLNQ.js} +563 -546
- package/dist/cli/index.js +27 -23
- package/dist/{clio-4LY5K2AC.js → clio-J5JIOIDS.js} +7 -6
- package/dist/{code-nav-7AX6FYE6.js → code-nav-AXCXSBHX.js} +5 -3
- package/dist/{config-GTLUW2PR.js → config-OEBMIN2U.js} +37 -27
- package/dist/{configure-R6A64DHX.js → configure-PUQOSIXQ.js} +16 -13
- package/dist/{context-5VKGUVJJ.js → context-EKDCKUUZ.js} +82 -7
- package/dist/{context-RW5HC47S.js → context-MGSE4Z2T.js} +33 -23
- package/dist/{context-JFZEJ7W5.js → context-URSXPBCK.js} +17 -9
- package/dist/{context-clear-6ZHBAZZT.js → context-clear-KDAJRNUK.js} +33 -23
- package/dist/context-working-set-SBKMPPI2.js +1552 -0
- package/dist/{dispatch-runner-VKBRCWQC.js → dispatch-runner-MSWN72NK.js} +43 -29
- package/dist/{doctor-KI767GSN.js → doctor-7BSE27PJ.js} +10 -10
- package/dist/{eval-XSSNATB4.js → eval-IZGDOO4H.js} +9 -8
- package/dist/{evidence-UA6AWDQQ.js → evidence-SR7WXB5B.js} +51 -23
- package/dist/{evolve-QNTFGV6Z.js → evolve-K7VE2CBX.js} +30 -20
- package/dist/{fleet-Q7UOMUSG.js → fleet-7XMJNQNF.js} +48 -38
- package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-AQNAH644.js} +3 -3
- package/dist/{init-WBB65ZHQ.js → init-JGNPAYXT.js} +41 -31
- package/dist/{memory-MD3O64RI.js → memory-4ALKDJ4Q.js} +32 -22
- package/dist/{models-BZU34YWD.js → models-ZMMLFJNN.js} +22 -19
- package/dist/{monitor-MEQA5C3I.js → monitor-2F3T5KHP.js} +55 -43
- package/dist/{orchestrator-CGFKEP27.js → orchestrator-ORHT43JB.js} +2507 -1896
- package/dist/{reset-L2FQEE3E.js → reset-NXGTYNUO.js} +4 -3
- package/dist/{run-IV4Q6RLN.js → run-RF4WJGMT.js} +51 -41
- package/dist/{share-S5BZQC5I.js → share-UT3W6E4M.js} +5 -4
- package/dist/{skills-LQEKRDTN.js → skills-PSACKC5Q.js} +2 -2
- package/dist/{skills-eval-3DC4HEWS.js → skills-eval-WJSI55RZ.js} +34 -24
- package/dist/{targets-C4SSGQOB.js → targets-PIIRAOYS.js} +23 -20
- package/dist/{terminal-lease-IT5JW2NR.js → terminal-lease-ULWXWNVY.js} +5 -4
- package/dist/{upgrade-7TT7SQ3G.js → upgrade-346TZ6AV.js} +18 -17
- package/dist/{usage-GV4PKT3M.js → usage-6KKXR32N.js} +34 -24
- package/dist/verifiers-4UUM6TEE.js +1214 -0
- package/dist/verify-X5HDROLA.js +25 -0
- package/dist/{wiki-generate-DQF6Z66B.js → wiki-generate-7STOCIFZ.js} +42 -31
- package/dist/worker/entry.js +33 -24
- package/docs/README.md +8 -7
- package/docs/acp.md +1 -1
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +53 -21
- package/docs/config-knobs-audit.md +1 -2
- package/docs/configuration-and-targets.md +15 -1
- package/docs/context-engine.md +64 -12
- package/docs/context-working-set.md +194 -0
- package/docs/development-pipeline.md +1 -1
- package/docs/documentation-coverage.md +5 -5
- package/docs/documentation-guide.md +6 -5
- package/docs/environment-variables.md +2 -1
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +14 -1
- package/docs/evidence-and-memory.md +74 -2
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +2 -2
- package/docs/fleet-dispatch.md +22 -7
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +6 -6
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +7 -9
- package/docs/observability.md +4 -4
- package/docs/performance-methodology.md +2 -2
- package/docs/proactive-memory.md +1 -1
- package/docs/prompt-envelope-and-tools.md +4 -4
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +35 -35
- package/docs/safety-model.md +23 -4
- package/docs/scientific-validation.md +21 -3
- package/docs/session-lifecycle.md +3 -3
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +79 -12
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +1 -1
- package/docs/tui-design.md +2 -2
- package/docs/worker-dispatch-mechanics.md +11 -1
- package/package.json +8 -11
- package/skills/meta/clio-test/SKILL.md +20 -17
- package/skills/meta/clio-test/evals.md +3 -3
- package/skills/meta/clio-test/references/harness.md +35 -6
- package/skills/meta/clio-test/references/test-map.md +20 -10
- package/skills/registry.yaml +2 -2
- package/skills/skill-marketplace.json +1 -1
- package/src/cli/context-working-set.ts +513 -0
- package/src/cli/context.ts +8 -0
- package/src/cli/evidence.ts +20 -2
- package/src/cli/index.ts +4 -0
- package/src/cli/verifiers.ts +325 -0
- package/src/core/bash-exec.ts +39 -14
- package/src/core/bus-events.ts +19 -4
- package/src/core/config.ts +54 -0
- package/src/core/defaults.ts +50 -3
- package/src/core/git-commit-attribution.ts +46 -21
- package/src/core/verification-scripts.ts +6 -0
- package/src/domains/agents/builtins/verifier.md +3 -0
- package/src/domains/config/classify.ts +1 -0
- package/src/domains/config/keybindings.ts +3 -3
- package/src/domains/context/working-set/contract.ts +161 -0
- package/src/domains/context/working-set/defaults.ts +28 -0
- package/src/domains/context/working-set/engine.ts +203 -0
- package/src/domains/context/working-set/fold.ts +62 -0
- package/src/domains/context/working-set/horizon.ts +38 -0
- package/src/domains/context/working-set/marker.ts +103 -0
- package/src/domains/context/working-set/path-index.ts +436 -0
- package/src/domains/context/working-set/payload.ts +152 -0
- package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
- package/src/domains/context/working-set/policies/index.ts +21 -0
- package/src/domains/context/working-set/policies/structural.ts +160 -0
- package/src/domains/context/working-set/project.ts +132 -0
- package/src/domains/context/working-set/protect.ts +109 -0
- package/src/domains/context/working-set/recall.ts +177 -0
- package/src/domains/context/working-set/replay/controls.ts +112 -0
- package/src/domains/context/working-set/replay/load-clio.ts +199 -0
- package/src/domains/context/working-set/replay/metrics.ts +185 -0
- package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
- package/src/domains/context/working-set/replay/report.ts +139 -0
- package/src/domains/context/working-set/replay/runner.ts +325 -0
- package/src/domains/context/working-set/replay/synthetic.ts +422 -0
- package/src/domains/context/working-set/replay/trace.ts +21 -0
- package/src/domains/context/working-set/visible.ts +54 -0
- package/src/domains/evidence/build.ts +112 -45
- package/src/domains/evidence/eval.ts +24 -7
- package/src/domains/evidence/index.ts +53 -0
- package/src/domains/evidence/ordering.ts +12 -0
- package/src/domains/evidence/run-trust.ts +221 -0
- package/src/domains/evidence/store.ts +46 -6
- package/src/domains/evidence/trust-status.ts +854 -0
- package/src/domains/evidence/types.ts +26 -0
- package/src/domains/middleware/memory-intervention.ts +3 -0
- package/src/domains/middleware/stalled-turn.ts +165 -4
- package/src/domains/safety/autonomy.ts +1 -1
- package/src/domains/safety/default-path-policy.ts +8 -0
- package/src/domains/safety/finish-contract.ts +4 -3
- package/src/domains/safety/policy-engine.ts +48 -6
- package/src/domains/session/compaction/compact.ts +23 -1
- package/src/domains/session/compaction/cut-point.ts +2 -0
- package/src/domains/session/compaction/tokens.ts +16 -1
- package/src/domains/session/context-ledger.ts +2 -0
- package/src/domains/session/entries.ts +107 -1
- package/src/domains/session/manager.ts +9 -2
- package/src/domains/session/migrations/index.ts +22 -3
- package/src/engine/acp/server.ts +3 -0
- package/src/engine/agent.ts +18 -1
- package/src/engine/session.ts +9 -3
- package/src/entry/orchestrator.ts +16 -4
- package/src/interactive/chat-loop-messages.ts +18 -6
- package/src/interactive/chat-panel.ts +571 -244
- package/src/interactive/chat-renderer.ts +79 -39
- package/src/interactive/context-meter.ts +10 -0
- package/src/interactive/context-overlay.ts +81 -6
- package/src/interactive/context-recall-command.ts +110 -0
- package/src/interactive/editor-submit.ts +26 -1
- package/src/interactive/footer/widgets.ts +22 -20
- package/src/interactive/footer-panel.ts +6 -1
- package/src/interactive/interactive-application.ts +2 -0
- package/src/interactive/interactive-event-projection.ts +12 -0
- package/src/interactive/interactive-slash-runtime.ts +49 -8
- package/src/interactive/model-session-replay.ts +21 -0
- package/src/interactive/overlay-general-openers.ts +6 -0
- package/src/interactive/overlay-session-lifecycle.ts +8 -4
- package/src/interactive/overlays/ask-user.ts +146 -24
- package/src/interactive/renderers/tool-execution.ts +167 -56
- package/src/interactive/session-transcript.ts +2 -2
- package/src/interactive/slash-commands.ts +29 -2
- package/src/interactive/status/index.ts +12 -1
- package/src/interactive/status/reasoning.ts +87 -0
- package/src/interactive/status/summary.ts +13 -2
- package/src/interactive/transcript-detail.ts +120 -0
- package/src/interactive/turn-context.ts +238 -88
- package/src/interactive/turn-middleware.ts +6 -6
- package/src/tools/agent-tools.ts +11 -4
- package/src/tools/bash.ts +144 -82
- package/src/tools/builtin-tool-catalog.ts +18 -6
- package/src/tools/context/index.ts +105 -3
- package/src/tools/context/surface.ts +3 -2
- package/src/tools/core-bootstrap.ts +21 -0
- package/src/tools/dispatch-runner.ts +9 -7
- package/src/tools/monitor.ts +28 -20
- package/src/tools/presentation.ts +107 -0
- package/src/tools/registry.ts +65 -7
- package/src/tools/result-disposition.ts +550 -0
- package/src/tools/result-shaping.ts +262 -19
- package/src/tools/safe-exec.ts +2 -0
- package/src/tools/verify/authoring.ts +1119 -0
- package/src/tools/verify/catalog.ts +346 -0
- package/src/tools/verify/index.ts +13 -3
- package/src/tools/verify/scripts.ts +135 -37
- package/src/tools/verify/surface.ts +9 -5
- package/src/tools/worker-evidence.ts +35 -12
- package/dist/chunk-MNA4JGU4.js +0 -255
- package/dist/chunk-SRF2PJNW.js +0 -184
- package/dist/chunk-T6YILFSB.js +0 -80
- package/dist/chunk-VAKQQHWR.js +0 -434
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
# Evidence Corpus and Long-Term Memory
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.4). Use it to design, validate, and simulate memory proposals, approval loops, pruning rules, and token budgets.
|
|
5
5
|
|
|
6
|
-
Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.
|
|
6
|
+
Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.4, forensic evidence auto-builds on dispatch run completion: when a run finalizes, the observability domain automatically compiles the evidence bundle under `<dataDir>/evidence/run-<id>/` and updates a compact sidecar index row in `<stateDir>/evidence-index.json`. Long-term memory records are local, evidence-linked, and only injected after explicit approval. Use the TUI [`/view`](observability.md) command for interactive inspection of receipts, dispatch output, durable tool output, compaction summaries, and session accountability before building or citing evidence.
|
|
7
7
|
|
|
8
8
|
Source of truth: `src/domains/evidence/**`, `src/domains/memory/**`, `src/cli/evidence.ts`, and `src/cli/memory.ts`.
|
|
9
9
|
|
|
@@ -48,6 +48,7 @@ Run/session evidence files:
|
|
|
48
48
|
├── audit-linked.jsonl
|
|
49
49
|
├── receipt.json
|
|
50
50
|
├── gate-decisions.json
|
|
51
|
+
├── trust-status.json
|
|
51
52
|
├── protected-artifacts.json
|
|
52
53
|
├── findings.json
|
|
53
54
|
└── findings.md
|
|
@@ -67,6 +68,7 @@ Eval evidence adds `eval-result.json` and uses empty receipt/protected-artifact
|
|
|
67
68
|
| `audit-linked.jsonl` | Audit rows linked to run/session context when available. |
|
|
68
69
|
| `receipt.json` | Receipt bundle (`{ version: 1, receipts: [...] }`); only receipts that pass integrity verification contribute verified fields. |
|
|
69
70
|
| `gate-decisions.json` | Integrity-verified review verdicts, compete winner selections, and winner confirmations discovered from linked receipt ids. |
|
|
71
|
+
| `trust-status.json` | Canonical per-run six-axis trust projections derived from authenticated receipts, gate decisions, grounded validation artifacts, and exact finish-contract audit rows. |
|
|
70
72
|
| `protected-artifacts.json` | Protected artifact state/events. |
|
|
71
73
|
| `findings.json` / `findings.md` | Structured and readable findings. |
|
|
72
74
|
|
|
@@ -159,6 +161,76 @@ whether applicable validation evidence was observed. Briefing provenance is
|
|
|
159
161
|
also distinct from bounded project-context provenance: both can be absent or
|
|
160
162
|
present independently, and neither hash is evidence for the other.
|
|
161
163
|
|
|
164
|
+
### Canonical trust status
|
|
165
|
+
|
|
166
|
+
`src/domains/evidence/trust-status.ts` defines the version 1 canonical trust
|
|
167
|
+
status. It is a six-axis algebra, not an overall trust verdict, confidence
|
|
168
|
+
percentage, or pass/fail score. Consumers project only the axes needed for a
|
|
169
|
+
decision and preserve every other axis unchanged.
|
|
170
|
+
|
|
171
|
+
| Axis | Closed states | Question answered |
|
|
172
|
+
|---|---|---|
|
|
173
|
+
| Artifact integrity | `verified`, `failed`, `absent`, `unknown`, `not_applicable` | Did the integrity verifier authenticate the referenced artifact? |
|
|
174
|
+
| Validation grounding | `validated`, `failed`, `ungrounded`, `absent`, `unknown`, `not_applicable` | What correctness-bearing validation was observed and grounded? |
|
|
175
|
+
| Independent review | `passed`, `failed`, `inconclusive`, `not_independent`, `absent`, `unknown`, `not_applicable` | What outcome did an authenticated independent reviewer or judge record? |
|
|
176
|
+
| Context provenance | `recorded`, `invalid`, `absent`, `unknown`, `not_applicable` | Is the origin of briefing, project context, or linked evidence recorded consistently? |
|
|
177
|
+
| Autonomy enforcement | `enforced`, `approximated`, `bypassed`, `absent`, `unknown`, `not_applicable` | How faithfully did the runtime enforce the selected authority? |
|
|
178
|
+
| Completion evidence | `evidenced`, `incomplete`, `limited`, `absent`, `unknown`, `not_applicable` | What did the finish contract observe at the completion boundary? |
|
|
179
|
+
|
|
180
|
+
`absent` means no fact was recorded and carries a reason but no invented
|
|
181
|
+
attribution. `unknown` means a named source exists but cannot establish the
|
|
182
|
+
answer. `not_applicable` means a named authority determined that the axis does
|
|
183
|
+
not apply. Every non-absent state names both its source and its authority.
|
|
184
|
+
Sources may retain up to 16 typed artifact references. References contain an
|
|
185
|
+
artifact kind, identifier, and optional SHA-256 digest; they never embed the
|
|
186
|
+
artifact body. Normalization sorts the references and rejects duplicates,
|
|
187
|
+
unbounded lists, unknown fields, invalid identifiers, and sources that are not
|
|
188
|
+
permitted to speak for an axis.
|
|
189
|
+
|
|
190
|
+
The composition rules prohibit cross-axis promotion:
|
|
191
|
+
|
|
192
|
+
- Verified artifact integrity never promotes validation grounding.
|
|
193
|
+
- Recorded context provenance never promotes validation or correctness.
|
|
194
|
+
- A passing review never establishes authorship or context origin.
|
|
195
|
+
- Enforced autonomy never promotes completion evidence.
|
|
196
|
+
- A completion self-report never promotes validation grounding. The linked
|
|
197
|
+
`completion_contract` audit row is the run's own report of what it did, so it
|
|
198
|
+
reaches completion evidence and no other axis. Validation grounding is filled
|
|
199
|
+
only by independently observed executions the session ledger recorded.
|
|
200
|
+
|
|
201
|
+
The current adapters apply the following persisted-format compatibility rules.
|
|
202
|
+
They do not mutate receipt, gate-decision, evidence-bundle, or session formats.
|
|
203
|
+
|
|
204
|
+
| Existing persisted fact | Canonical mapping |
|
|
205
|
+
|---|---|
|
|
206
|
+
| Missing receipt | Every receipt-owned axis is `absent` with `artifact_missing`. |
|
|
207
|
+
| Current receipt present but integrity not checked | Artifact integrity is `unknown`; the receipt's own digest never authenticates itself. The other receipt-owned axes are `absent` with `not_observed` until authentication succeeds. |
|
|
208
|
+
| Historical receipt missing its integrity block | Receipt-owned axes are `unknown` through the compatibility source, even if a caller presents a contradictory positive verification result. |
|
|
209
|
+
| Integrity verification succeeds or fails | Artifact integrity is `verified` or `failed`. A failure leaves the receipt-owned validation grounding, context provenance, and autonomy enforcement `absent`; no untrusted receipt claim contributes a positive state. Validation the session ledger observed on its own (a validation command that ran and exited 0) still grounds the run, so a tampered run can read `artifactIntegrity: failed` beside `validationGrounding: validated`. The two axes name different artifacts and different authorities, and the bundle's `receipt-integrity` finding is what flags the pairing. |
|
|
210
|
+
| Receipt `verification.state: verified` | Validation grounding is `validated` unless a stronger typed failure or ungrounded claim is present. |
|
|
211
|
+
| Receipt `verification.state: unverified` | Validation grounding is `absent` with `not_observed`; lack of a validation tool is not a failed validation. |
|
|
212
|
+
| Receipt verification `unknown` or `not_applicable` | Validation grounding preserves `unknown` or `not_applicable`. A missing historical verification field maps to `unknown`. |
|
|
213
|
+
| Typed receipt validation or result-contract quality | A passing correctness-bearing fact maps to `validated`; a failing fact maps to `failed`; an ungrounded passing claim maps to `ungrounded`. |
|
|
214
|
+
| Valid bounded project context or valid briefing hash | Context provenance is `recorded`. Explicit project-context tier `none` with no briefing is `not_applicable`; a missing historical field is `unknown`; a contradictory block is `invalid`. |
|
|
215
|
+
| Gate decision | An authenticated independent pass or fail maps to `passed` or `failed`. Correlated review maps to `not_independent`. Unauthenticated artifacts map to `unknown`; operator or full-auto confirmation alone is `not_applicable` to independent review. |
|
|
216
|
+
| Receipt autonomy grade | `mediated`, `approximated`, and `bypassed` map to `enforced`, `approximated`, and `bypassed`. A dangerous-bypass flag always normalizes to `bypassed`; a missing historical block is `unknown`. |
|
|
217
|
+
| Finish-contract assessment | `validation_evidence`, `unvalidated_mutation`, `explicit_limitation`, and `no_mutation` map to `evidenced`, `incomplete`, `limited`, and `not_applicable`. A run whose receipt was presented and rejected downgrades `evidenced` to `unknown`: the row still points at its own record, but a rejected receipt authenticates nothing about the run it names. |
|
|
218
|
+
| Malformed audit row identifier | A blank or whitespace-only optional identifier is treated as absent. The row falls back to its derived correlation id or drops out of the trust projection; it never aborts the bundle. |
|
|
219
|
+
| Bundle without `trust-status.json` | Inspection reports `projection: historical_format` with no canonical run projections. It never reconstructs positive states from older summary tags. |
|
|
220
|
+
|
|
221
|
+
Receipt inspection, worker output, monitor details, and evidence rebuilding all
|
|
222
|
+
use the same authenticated receipt projection boundary. Evidence rebuilding
|
|
223
|
+
then composes independently authenticated gate decisions and exact
|
|
224
|
+
finish-contract records without changing receipt-owned axes. Findings such as
|
|
225
|
+
`no-validation`, `proxy-validation`, `external-approximation`, and
|
|
226
|
+
`external-bypass` are selected from the canonical states, while their detailed
|
|
227
|
+
domain artifacts remain in the receipt, gate, audit, and trace files.
|
|
228
|
+
|
|
229
|
+
The canonical aggregate is an additive projection for downstream work. Receipt
|
|
230
|
+
integrity remains version 15, evidence bundles remain version 1, gate decisions
|
|
231
|
+
remain version 2, and no persisted receipt field or cryptographic algorithm
|
|
232
|
+
changes.
|
|
233
|
+
|
|
162
234
|
### Mutation-Report Grounding
|
|
163
235
|
|
|
164
236
|
Mutation-report receipts are grounded directly against observed tool events recorded in the run ledger:
|
package/docs/evolution.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Evolution and Change Manifests
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder uses change manifests to make harness changes reviewable, falsifiable, and rollback-friendly. CLIO stands for Context Layer for Input/Output, named for the Greek muse of history. A manifest is JSON, generated or checked with `clio-coder evolve manifest`, and should describe what changed, why, what evidence supports it, what could regress, how to validate it, and how to roll it back.
|
|
7
7
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Exit Codes & Machine-Readable Output Contracts
|
|
2
2
|
|
|
3
|
-
This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.
|
|
3
|
+
This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.4`.
|
|
4
4
|
|
|
5
5
|
Source implementations: `src/cli/` and `src/entry/`.
|
|
6
6
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Extensions, Prompt Templates, Skills, and Share Archives
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder has lightweight community-oriented resource packaging. Extensions are filesystem bundles that contribute prompts and skills. Share archives are portable JSON files for moving project/user Clio resources between machines or collaborators. Themes are built into the engine and are no longer loaded from extensions.
|
|
7
7
|
|
|
@@ -251,7 +251,7 @@ Share archives are single JSON files:
|
|
|
251
251
|
"formatVersion": 1,
|
|
252
252
|
"manifest": {
|
|
253
253
|
"format": "clio.share.v1",
|
|
254
|
-
"clioVersion": "0.3.
|
|
254
|
+
"clioVersion": "0.3.4",
|
|
255
255
|
"createdAt": "...",
|
|
256
256
|
"files": []
|
|
257
257
|
},
|
package/docs/fleet-dispatch.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fleet Dispatch
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.
|
|
3
|
+
> **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.4).
|
|
4
4
|
|
|
5
5
|
Clio Coder dispatches bounded worker agents. With a fleet configured, those
|
|
6
6
|
workers run on remote machines over SSH while the orchestrator keeps every
|
|
@@ -579,6 +579,19 @@ basis unknown/not applicable). A read-only Scout can therefore report `receipt_i
|
|
|
579
579
|
bounded `project_context` provenance are also rendered independently; neither
|
|
580
580
|
hash substitutes for the other.
|
|
581
581
|
|
|
582
|
+
The canonical terminology for these facts is the six-axis trust status in
|
|
583
|
+
[`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
|
|
584
|
+
Receipt integrity projects onto artifact integrity; receipt verification,
|
|
585
|
+
typed quality, and validation grounding project onto validation grounding;
|
|
586
|
+
gate decisions project onto independent review; briefing and project context
|
|
587
|
+
project onto context provenance; and `autonomyEnforcement` projects onto
|
|
588
|
+
autonomy enforcement. A receipt does not contain independent-review or
|
|
589
|
+
completion-evidence outcomes merely because it is sealed. Those axes remain
|
|
590
|
+
`absent` until an authenticated gate artifact or finish assessment is composed.
|
|
591
|
+
In particular, verified integrity cannot validate claims, known provenance
|
|
592
|
+
cannot establish correctness, and a review verdict cannot establish
|
|
593
|
+
authorship.
|
|
594
|
+
|
|
582
595
|
Gate references point backward: a reviewer references the builder it
|
|
583
596
|
reviewed, a revise builder references the reviewer whose findings it
|
|
584
597
|
received, and a judge references every candidate. Because a worker receipt
|
|
@@ -682,11 +695,13 @@ After `npm run build`, an operator with a configured model target can run the
|
|
|
682
695
|
single-turn, read-only fleet lifecycle check explicitly:
|
|
683
696
|
|
|
684
697
|
```bash
|
|
685
|
-
|
|
698
|
+
npm run live:fleet-dispatch -- --target <id> [--model <wireId>] [--thinking medium]
|
|
686
699
|
```
|
|
687
700
|
|
|
688
|
-
It is not part of deterministic CI. The
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
701
|
+
It is not part of deterministic CI. The driver
|
|
702
|
+
(`benchmarks/internal/live-fleet-dispatch.ts`) copies the repository into a
|
|
703
|
+
committed temporary workspace, sandboxes all Clio config, state, data, and
|
|
704
|
+
cache under a scratch home holding only the chosen target, exercises Scout,
|
|
705
|
+
bounded spot-checking, detached Debugger briefing, steering, wait, and
|
|
706
|
+
collect, and fails if any workspace content changes. A failed run retains its
|
|
707
|
+
scratch tree for diagnosis.
|
package/docs/glossary.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Clio Coder Glossary
|
|
2
2
|
|
|
3
|
-
This document defines core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
|
|
3
|
+
This document defines the 45 core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
@@ -165,3 +165,23 @@ This document defines core architectural concepts and terminology used throughou
|
|
|
165
165
|
### 40. Delegate
|
|
166
166
|
- **Definition**: Another coding agent Clio drives over ACP stdio as if it were a worker, configured under `delegation.agents` and invoked with `/delegate`. A delegate is a foreign harness, not a model target.
|
|
167
167
|
- **Owning Type**: `DelegationAgentConfig` in `src/core/defaults.ts`.
|
|
168
|
+
|
|
169
|
+
### 41. Working Set
|
|
170
|
+
- **Definition**: The part of the session ledger the model receives on the next request. It is the ledger with the current eviction projection applied, and it exists only in memory; the ledger itself is never narrowed. Not to be confused with the context ledger, which is the accounting of how the window is spent.
|
|
171
|
+
- **Owning Type**: `WorkingSetView` in `src/domains/context/working-set/contract.ts`.
|
|
172
|
+
|
|
173
|
+
### 42. Projection
|
|
174
|
+
- **Definition**: The pure, idempotent transform from ledger entries to the entries the replay builder hands the model. It substitutes markers for evicted bodies and drops thinking from closed turns, returning unaffected entries by reference. Nothing about it is persisted.
|
|
175
|
+
- **Owning Type**: `projectWorkingSet` in `src/domains/context/working-set/project.ts`.
|
|
176
|
+
|
|
177
|
+
### 43. Eviction
|
|
178
|
+
- **Definition**: The decision that a tool-result body or an assistant turn's thinking leaves the working set, recorded as an append-only ledger entry with a typed reason. It removes nothing: the original entry stays in the ledger and stays visible in the transcript, `/resume`, `/fork`, and the HTML export.
|
|
179
|
+
- **Owning Type**: `ContextEvictionEntry` in `src/domains/session/entries.ts`.
|
|
180
|
+
|
|
181
|
+
### 44. Recall
|
|
182
|
+
- **Definition**: Readmitting an evicted body by ref, through `context(scope="recall", ref=...)` for the model or `/context recall <ref>` for the operator. A recall does not un-evict: the marker stays where it was so the provider prefix cache is untouched, and repeated recalls of one ref are the churn signal.
|
|
183
|
+
- **Owning Type**: `ContextRecallEntry` in `src/domains/session/entries.ts`; resolution in `resolveRecall` in `src/domains/context/working-set/recall.ts`.
|
|
184
|
+
|
|
185
|
+
### 45. Marker
|
|
186
|
+
- **Definition**: The byte-stable one-line stub the projection renders in place of an evicted body, naming the ref, the reason, the tool, the size, and the exact recall call. It carries no timestamp and no counter, because a marker whose bytes drifted between renders would cold-start the prefix cache on a turn that evicted nothing new.
|
|
187
|
+
- **Owning Type**: `renderMarker` in `src/domains/context/working-set/marker.ts`.
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Clio Coder is designed to be self-contained and platform-compliant. This document outlines the default directory paths, file purposes, permission levels, and lifecycle commands (`install`, `reset`, `upgrade`, and `uninstall`). Clio Coder installs from npm as `@iowarp/clio-coder` (`npm install -g @iowarp/clio-coder`, published since v0.3.0) or from a source checkout with a deterministic local symlink; the CLI classifies both install kinds and `clio-coder upgrade` handles each.
|
|
4
4
|
|
|
5
5
|
> [!TIP]
|
|
6
|
-
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.
|
|
6
|
+
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.4). You can open it directly in any web browser to view details dynamically.
|
|
7
7
|
|
|
8
8
|
---
|
|
9
9
|
|
|
@@ -221,25 +221,25 @@ next `clio-coder` launch refreshes it. `install.json` then reads
|
|
|
221
221
|
`upgradedFrom: "0.3.0"`; doctor's row becomes
|
|
222
222
|
`0.3.1 (installed ..., upgraded ... from 0.3.0)`.
|
|
223
223
|
|
|
224
|
-
#### Upgrading to 0.3.
|
|
224
|
+
#### Upgrading to 0.3.3
|
|
225
225
|
|
|
226
|
-
Upgrading from 0.3.1 to 0.3.
|
|
226
|
+
Upgrading from 0.3.1 to 0.3.3 is automated:
|
|
227
227
|
|
|
228
228
|
```bash
|
|
229
229
|
clio-coder upgrade
|
|
230
230
|
```
|
|
231
231
|
|
|
232
|
-
Key lifecycle and operational updates in v0.3.
|
|
232
|
+
Key lifecycle and operational updates in v0.3.4:
|
|
233
233
|
- Upgraded the underlying engine SDK libraries to 0.84.0 with signal-aware OAuth cancellation.
|
|
234
234
|
- Hardened migration resilience: damaged `credentials.yaml` files no longer block upgrades when no renames are needed (#121); `--skip-migrations` is available as a recovery override.
|
|
235
|
-
- Fullscreen TUI mode (`terminal.tuiMode`, `terminal.fullscreenScrollbar`) is available via Settings → Terminal (restart required). Adaptive presentation pacing is the live `terminal.smoothStreaming` setting; 0.3.
|
|
235
|
+
- Fullscreen TUI mode (`terminal.tuiMode`, `terminal.fullscreenScrollbar`) is available via Settings → Terminal (restart required). Adaptive presentation pacing is the live `terminal.smoothStreaming` setting; 0.3.3 defaults it to `off`, with conservative `auto` and explicit `on` available from the same section.
|
|
236
236
|
- Interactive launch paints a measured Stage 0 shell on the same terminal and editor that Stage 1 hydrates. Typing, queued submits, resize, and Ctrl+C remain live during hydration; set `CLIO_CODER_INSTANT_SHELL=0` for the legacy fully hydrated first-frame path.
|
|
237
237
|
- Turn settlement is enforced on `/new`, `/resume`, `/tree`, and `/fork` to cleanly commit in-flight streams before session writer replacement (#114).
|
|
238
238
|
- Resumed and forked session entry replays standardize message prefixes through `src/engine/messages.ts`.
|
|
239
239
|
- `AI_AGENT=clio-coder` is set on all child processes for system attribution.
|
|
240
240
|
|
|
241
241
|
The first interactive launch after upgrading shows the version notice:
|
|
242
|
-
`clio: upgraded 0.3.1 → 0.3.
|
|
242
|
+
`clio: upgraded 0.3.1 → 0.3.3. What changed at the keyboard: ...`
|
|
243
243
|
Recorded once per version in `install.json` as `noticedVersion`.
|
|
244
244
|
|
|
245
245
|
### C. System Resets (`clio-coder reset`)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Middleware and Component Registry
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder has two related but separate surfaces:
|
|
7
7
|
|
package/docs/model-catalog.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Model Catalog, Runtime Refresh, and Field Notes
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder treats a selectable model as the intersection of three sources:
|
|
7
7
|
|
|
@@ -39,14 +39,12 @@ worker spec and receipt are written.
|
|
|
39
39
|
|
|
40
40
|
## Benchmarking Models
|
|
41
41
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
The benchmarks record context-window, thinking, sampling, weight quantization, and KV-cache settings so sweeps can be compared consistently.
|
|
42
|
+
The public benchmark adapters under [benchmarks/community/](../benchmarks/community/)
|
|
43
|
+
(SWE-bench Lite, Terminal-Bench, SciCode, HumanEval) drive Clio through
|
|
44
|
+
`clio-coder run --json` or `clio-coder eval run`, each taking `--target` and
|
|
45
|
+
`--model` from the configured targets. `benchmarks/README.md` has the
|
|
46
|
+
commands. The run manifests record the target profile (runtime, model,
|
|
47
|
+
thinking level) so sweeps can be compared consistently.
|
|
50
48
|
|
|
51
49
|
## What "sanctioned" means
|
|
52
50
|
|
package/docs/observability.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Observability Viewer
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
`/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
|
|
7
7
|
|
|
@@ -69,7 +69,7 @@ Clio resolves directories under platform-specific XDG defaults (on Linux, these
|
|
|
69
69
|
| **Dispatch outputs** | Logs and ledger records detailing worker execution. | `<stateDir>/runs.json` and `<stateDir>/receipts/<runId>.json` |
|
|
70
70
|
| **Task ledgers** | Per-turn task-board goals, active runs, required validation evidence, and operator-task provenance when present. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
|
|
71
71
|
| **Workspace outputs** | Latest successful `artifact`, `write`, or `edit` result for each normalized path on the active session branch. Missing files remain visible as durable recorded facts. | Recorded path beneath the session metadata `cwd` |
|
|
72
|
-
| **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<
|
|
72
|
+
| **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt` |
|
|
73
73
|
| **Protected artifacts** | Validation-protected artifact metadata and its absolute artifact path when available. | Session ledger record plus the protected workspace path |
|
|
74
74
|
| **Compaction** | Summaries of compacted history sessions. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
|
|
75
75
|
| **Prompt manifests** | One validated record per prompt compile: `systemPromptHash`, previous hash, token estimate, thinking dial at compile time, per-section token estimates, and per-fragment content hashes. Identifies the exact compiled prompt and supports hash diffs without storing prompt text. Malformed records appear as an explicit read-error artifact. | `<stateDir>/sessions/<cwdHash>/<sessionId>/prompt-manifest.jsonl` |
|
|
@@ -159,8 +159,8 @@ The base provenance sets, steering, routing, quality, worker identity, and resul
|
|
|
159
159
|
| `safety.toolTelemetry.ingestionErrors` | `number` | Current dispatch receipts | Malformed or lost frames, event-fold/source errors, and drain timeouts that make otherwise mediated telemetry incomplete | experimental |
|
|
160
160
|
| `safety.toolTelemetry.unfinished` | `{ tool, count }[]` | Current dispatch receipts | Tool starts that had no matching finish when the receipt sealed | experimental |
|
|
161
161
|
| `safety.toolTelemetry.workspaceMutationPossible` | `boolean` | Current dispatch receipts | Whether incomplete or unavailable telemetry could conceal a shared-workspace mutation; retry admission fails closed when true | experimental |
|
|
162
|
-
| `autonomyEnforcement.grade` | `string` | Always in v0.3.
|
|
163
|
-
| `autonomyEnforcement.autonomy` | `string` | Always in v0.3.
|
|
162
|
+
| `autonomyEnforcement.grade` | `string` | Always in v0.3.4 | The autonomy grade level enforced for the run | experimental |
|
|
163
|
+
| `autonomyEnforcement.autonomy` | `string` | Always in v0.3.4 | The effective autonomy level name (e.g. auto-edit, suggest, read-only, full-auto) | experimental |
|
|
164
164
|
| `autonomyEnforcement.externalMode` | `string` | When running external worker | The execution mode of the external worker runtime | experimental |
|
|
165
165
|
| `autonomyEnforcement.dangerousBypass` | `boolean` | When running external worker | Whether a safety bypass was explicitly activated | experimental |
|
|
166
166
|
| `validationGrounding.claimed` | `number` | Validation grounding evaluated | Count of validations claimed by worker | experimental |
|
|
@@ -101,7 +101,7 @@ chunk name. Every lazy-graph contract must prove the heavyweight marker absent
|
|
|
101
101
|
before first use and present after a real invocation, then repeat from a packed
|
|
102
102
|
installation in a foreign working directory.
|
|
103
103
|
|
|
104
|
-
## Corrected 0.3.
|
|
104
|
+
## Corrected 0.3.3 baseline
|
|
105
105
|
|
|
106
106
|
These observations were recorded on 2026-08-19 in WSL2 Linux
|
|
107
107
|
`6.18.33.2-microsoft-standard-WSL2`, x86-64, with an 80x24 `xterm-256color`
|
|
@@ -384,7 +384,7 @@ and the dispatch reservation, approval, gate, detach, monitor, and steer suites.
|
|
|
384
384
|
## Adaptive stream-pacer observations
|
|
385
385
|
|
|
386
386
|
`terminal.smoothStreaming` is presentation-only. `off` is the exact existing
|
|
387
|
-
16 ms coalescer and remains the 0.3.
|
|
387
|
+
16 ms coalescer and remains the 0.3.3 default. `auto` uses the pacer only on a
|
|
388
388
|
capable local TTY with no accessibility, remote/multiplexer, CI, or observed
|
|
389
389
|
backpressure signal. `on` requests pacing, but frame construction still stops
|
|
390
390
|
behind stdout backpressure. The pacer never republishes slices on the public
|
package/docs/proactive-memory.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Proactive task memory
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.
|
|
3
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.4).
|
|
4
4
|
|
|
5
5
|
Clio's proactive task memory protects long-running work from behavioral state
|
|
6
6
|
decay: a requirement, environment fact, failed attempt, or diagnosis can still
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Prompt Envelope and Tools
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
Clio Coder keeps the model-facing envelope stable and moves enforcement into the runtime registry and safety policy.
|
|
7
7
|
|
|
@@ -85,7 +85,7 @@ Several tools absorb what used to be separate tools:
|
|
|
85
85
|
- `find(pattern, path?, order?, limit?, include_ignored?)` locates paths by glob pattern (`*`, `**`, `?`, `[abc]`), default limit 500. `order="path"` (default) returns fd's native order; `order="mtime"` returns newest first from a bounded candidate set instead of statting the whole tree, and reports `details.candidates` when the candidate cap made the ordering approximate.
|
|
86
86
|
- `grep(pattern, path?, mode?, glob?, ignore_case?, literal?, context?, limit?, include_ignored?)` searches file contents with ripgrep, degrading to a bounded pure-Node search when rg is absent. `mode=content` (default) returns line-referenced matches, `mode=files` returns matching paths, `mode=count` returns per-file counts. Context lines are consumed from rg's `--json` stream.
|
|
87
87
|
- `context(scope="workspace"|"docs"|"skills")` is the one OBSERVE entry point for material about the working environment: the session workspace snapshot, retrieval over Clio's bundled documentation (`query` required), and skill listing or loading (`name` optional, `include_tree` for the skill's resource files).
|
|
88
|
-
- `verify(check?, path?, args?, browser?, cwd?, timeout_ms?)` runs declared verification. `verify()`
|
|
88
|
+
- `verify(check?, path?, args?, browser?, cwd?, timeout_ms?)` runs declared verification. `verify()` lists package.json verification scripts and strict version-1 `.clio-coder/verifiers.yaml` entries through the same `{id, description, command, cwd, timeoutMs, tags, source}` projection. `verify(check="<id>")` runs a package script or the catalog's exact argv/cwd/timeout through safe-exec with no shell. Model `args`, cwd, timeout, output-cap, and environment fields cannot mutate a project entry. `verify(check="frontend", path=...)` validates an HTML/CSS/JS artifact without granting shell access.
|
|
89
89
|
- `artifact(kind="plan"|"review"|"report", content, ...)` writes named artifacts behind one surface: Markdown documents (default `.clio-coder/artifacts/PLAN.md`/`REVIEW.md`/`REPORT.md`; `path` may override inside the workspace) that terminate the turn, because writing the artifact is the answer. Skills are not artifacts; a `SKILL.md` is written with the ordinary write tool and validated by the skills loader.
|
|
90
90
|
- `dispatch(task?, tasks?, mode?, ...)` supports a first-class singular assignment (`task`) and a batch (`tasks`), never both. `task` is worker instructions; `briefing` is optional bounded parent context/data and cannot replace it. Briefing stays a separate dynamic message and receipt provenance, never part of the receipt task. A shared top-level briefing applies to strings and objects without an override; an object-level briefing wins. Blank values are omitted, the cap is 12,000 UTF-8 bytes, and approval pins the exact canonical value. Ordinary handles enter one registered event consumer immediately. Synchronous calls auto-wait for stream-and-receipt completion; `detach:true` returns ids after durable batch registration while the same consumer continues. Review and compete retain gate-sensitive direct drains. Task objects may include `persona` and `tool_profile`. Pipeline output is threaded as bounded data. A successful native or ACP run requires a nonempty receipt-sealed final output; exit zero without one fails as `worker_final_output_missing`, with unfinished text retained only as partial diagnostics. `dispatch(list=true)` renders the catalog.
|
|
91
91
|
- `monitor(run_id?, mode?)` is read-only visibility into known synchronous and detached runs: `list` enumerates, `status` reports one, `peek` returns the in-process event tail, `receipt` exposes the stored evidence, and `wait` observes one run without collecting or canceling it. `collect` is the authoritative terminal batch operation over a detached batch or run-id list; collect before final synthesis. Completed output reports receipt integrity, evidence verification, briefing provenance, and bounded project-context provenance as different fields.
|
|
@@ -107,7 +107,7 @@ The six content-returning OBSERVE tools (`read`, `grep`, `find`, `ls`, `code_nav
|
|
|
107
107
|
|
|
108
108
|
Unknown segments are omitted. `<total>` renders as `N+` when the search was killed early at its limit, meaning matches beyond it exist but were never counted. `next` is always an exact continuation call fragment such as `limit=200` or `offset=451`, never prose. Untruncated results get no notice. Empty results are standardized: `grep` returns `No matches found`, `find` returns `No files found matching pattern`, `ls` returns `(empty directory)`, and the JSON-format tools return valid JSON with empty arrays and `next` populated.
|
|
109
109
|
|
|
110
|
-
**Offload on truncation.** When a byte cap cuts collected content, the tool spills its full rendering to the per-session scratch file (`<stateDir>/scratch/<sessionId>/<
|
|
110
|
+
**Offload on truncation.** When a byte cap cuts collected content, the tool spills its full rendering to the per-session scratch file (`<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt`) and reports the path in the notice, so no collected match, path, or line is ever unrecoverable. Two deliberate exceptions exist: `read` never offloads because the source file is directly re-addressable via `next: offset=N`, and a bare item-limit truncation without a byte cut continues via `next` alone, since an offload would only duplicate the body.
|
|
111
111
|
|
|
112
112
|
**Always-valid JSON.** `code_nav` and the JSON scopes of `context` declare `format: "json"`. A JSON payload must parse or be replaced whole; it is never cut mid-document. An oversize payload is offloaded and the body is replaced by the parseable stub:
|
|
113
113
|
|
|
@@ -133,7 +133,7 @@ Tool descriptions are tiered by how much a wrong call costs. The hot tools the m
|
|
|
133
133
|
|
|
134
134
|
Clio uses two context-protection mechanisms.
|
|
135
135
|
|
|
136
|
-
1. Tool results are capped at the source and again at the registry boundary. OBSERVE tools use the envelope caps above. Exact mutation tools (`write`, `edit`, `artifact`) use 8KB; `steer` and `credential_present` use 4KB; `ask_user` has a 20KB policy. Summary-kind tools (`bash`, `git`, `verify`, `dispatch`, `monitor`) use 16KB at the registry boundary. `web_fetch` is bounded at 16KB after shaping and may read more before it: its `max_bytes` argument defaults to 600KB and is hard-capped at 5MB. Tools without an explicit result-size policy use an approximately 18KB generic backstop. Over-cap generic results are shown briefly and, when possible, saved under `<stateDir>/scratch/<sessionId>/<
|
|
136
|
+
1. Tool results are capped at the source and again at the registry boundary. OBSERVE tools use the envelope caps above. Exact mutation tools (`write`, `edit`, `artifact`) use 8KB; `steer` and `credential_present` use 4KB; `ask_user` has a 20KB policy. Summary-kind tools (`bash`, `git`, `verify`, `dispatch`, `monitor`) use 16KB at the registry boundary. Bash also exposes the canonical per-call `output_policy`: omitted/`bounded` keeps its diagnostic tail, `summary` selects stable redacted head/error/tail evidence, `metadata-only` keeps facts and retrieval without stdout/stderr context, and `full` succeeds only inside the same hard result budget or records a typed downgrade. This model-context choice does not change the folded tail-biased operator presentation. `web_fetch` is bounded at 16KB after shaping and may read more before it: its `max_bytes` argument defaults to 600KB and is hard-capped at 5MB. Tools without an explicit result-size policy use an approximately 18KB generic backstop. Over-cap generic results are shown briefly and, when possible, saved under `<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt` with an `offloadPath` detail and a 10MB scratch-file cap.
|
|
137
137
|
2. Auto-compaction uses one pressure threshold. The default threshold is 0.8. When pressure crosses the threshold, Clio first masks stale tool observations and stale thinking older than `excludeLastTurns`. If pressure remains above the threshold, it runs the LLM summary compaction path and replays from the compacted session view.
|
|
138
138
|
|
|
139
139
|
Manual `/context compact`, `CLIO_CODER_FORCE_COMPACT=1`, and overflow recovery force the LLM summary path directly.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Provider Adapter Cookbook
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.4).
|
|
5
5
|
|
|
6
6
|
This cookbook guides developers through implementing custom model runtimes and inference server integrations within Clio Coder. It explains the runtime descriptor interfaces, probing protocols, model synthesis, and how to configure reasoning and thinking behaviors.
|
|
7
7
|
|
|
@@ -1,23 +1,23 @@
|
|
|
1
|
-
# v0.3.
|
|
1
|
+
# v0.3.4 Release-Cut Checklist
|
|
2
2
|
|
|
3
|
-
The ordered steps that turn the prepared `v0.3.
|
|
3
|
+
The ordered steps that turn the prepared `v0.3.4` branch into a published
|
|
4
4
|
release. Everything above the line marked **AUTHORIZATION BOUNDARY** is
|
|
5
5
|
repeatable and reversible and is run locally before the cut. Everything below
|
|
6
6
|
it is external or irreversible and needs an explicit decision from the
|
|
7
|
-
operator.
|
|
8
|
-
every step
|
|
7
|
+
operator. This page is the procedure and the release report carries the live
|
|
8
|
+
state of every step.
|
|
9
9
|
|
|
10
10
|
## Status of the prepared tree
|
|
11
11
|
|
|
12
12
|
| Item | State |
|
|
13
13
|
| --- | --- |
|
|
14
|
-
| Branch | `v0.3.
|
|
15
|
-
| `package.json` version | `0.3.
|
|
16
|
-
| `main` |
|
|
17
|
-
| `origin/main` | `
|
|
18
|
-
| Tags | none for 0.3.
|
|
19
|
-
| GitHub Release | none for 0.3.
|
|
20
|
-
| npm registry | `@iowarp/clio-coder@0.3.
|
|
14
|
+
| Branch | `v0.3.4`; `origin/v0.3.4` exists and is pushed to the reviewed tip before the cut |
|
|
15
|
+
| `package.json` version | `0.3.4`; the top `CHANGELOG.md` heading is `## 0.3.4 - 2026-08-22` |
|
|
16
|
+
| `main` | `8a1c8304`, the published `v0.3.3` commit; it is an ancestor of `v0.3.4` and moves only at Part 4. |
|
|
17
|
+
| `origin/main` | `8a1c8304`, matching the published `v0.3.3` commit |
|
|
18
|
+
| Tags | none for 0.3.4, local or remote |
|
|
19
|
+
| GitHub Release | none for 0.3.4 |
|
|
20
|
+
| npm registry | `@iowarp/clio-coder@0.3.4` absent; `latest` is `0.3.3` |
|
|
21
21
|
| Commit provenance identity | Post-release maintainer follow-up, not a gate: verifying `clio-coder@iowarp.ai` on IOWarp-controlled GitHub and GitLab identities (such as `clio-coder-bot` or `iowarp-clio`, with `assets/clio-coder-avatar-512.png` as the avatar) only changes how those platforms render the trailers. |
|
|
22
22
|
|
|
23
23
|
---
|
|
@@ -40,11 +40,11 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
40
40
|
resources, and the tarball and unpacked size budgets)
|
|
41
41
|
9. Step 8 again under the other supported Node major. Both Node 22 and
|
|
42
42
|
Node 24 must be green; the repo is developed against 22.22.3 and 24.9.0.
|
|
43
|
-
10. `npm run
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
43
|
+
10. `npm run live:smoke -- --target <id>` for one real headless turn through
|
|
44
|
+
the built binary against a configured target, which is the one release
|
|
45
|
+
check a deterministic suite cannot give. The packaged-install lifecycle
|
|
46
|
+
(pack, install into a clean prefix, run the installed launcher) is
|
|
47
|
+
`tests/smoke/pack-install.test.ts` and already ran under step 5.
|
|
48
48
|
11. `npm pack --dry-run`, then a real `npm pack` into a temporary directory.
|
|
49
49
|
Inspect the complete file list: `skills/`, `docs/*.md`, the builtin
|
|
50
50
|
agents, the model catalogs, and `damage-control-rules.yaml` are present;
|
|
@@ -58,22 +58,22 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
58
58
|
## Part 2: version and notes (repeatable)
|
|
59
59
|
|
|
60
60
|
13. Files carrying a version reference, to update together if the number
|
|
61
|
-
changes: `package.json` and `package-lock.json`, the `## 0.3.
|
|
62
|
-
heading in `CHANGELOG.md`, the `(Version: 0.3.
|
|
63
|
-
the `Blueprint (v0.3.
|
|
61
|
+
changes: `package.json` and `package-lock.json`, the `## 0.3.4 - <date>`
|
|
62
|
+
heading in `CHANGELOG.md`, the `(Version: 0.3.4)` markers in `docs/*.md`,
|
|
63
|
+
the `Blueprint (v0.3.4)` titles in `docs/html/*.html`, the `--branch`
|
|
64
64
|
pin in the README install block (the hygiene lint checks it), and the
|
|
65
65
|
measured-at figures in `scripts/check-release.mjs` if the package size
|
|
66
66
|
moved materially.
|
|
67
|
-
14. Confirm the `## 0.3.
|
|
67
|
+
14. Confirm the `## 0.3.4` section of `CHANGELOG.md` describes every
|
|
68
68
|
user-visible behavior change, including the ones that alter existing
|
|
69
69
|
behavior, and carries no Workbench release narrative. The release workflow
|
|
70
70
|
uses this section verbatim as the GitHub Release body.
|
|
71
71
|
15. Re-run `npm run ci:release` after any version edit and commit as one
|
|
72
|
-
commit on `v0.3.
|
|
72
|
+
commit on `v0.3.4`.
|
|
73
73
|
|
|
74
74
|
## Part 3: present the gate
|
|
75
75
|
|
|
76
|
-
16. Report to the operator before touching `main`: the exact final `v0.3.
|
|
76
|
+
16. Report to the operator before touching `main`: the exact final `v0.3.4`
|
|
77
77
|
SHA and clean status, the commits added since the handoff SHA, the gate
|
|
78
78
|
commands with pass/fail totals for both Node majors, the package version
|
|
79
79
|
and changelog heading, the tarball audit, the clean-install results and any
|
|
@@ -92,9 +92,9 @@ confirming the exact SHA and the commands.
|
|
|
92
92
|
## Part 4: fast-forward `main`
|
|
93
93
|
|
|
94
94
|
17. `git fetch origin` immediately before integrating; require `origin/main`
|
|
95
|
-
to be an ancestor of the reviewed `v0.3.
|
|
95
|
+
to be an ancestor of the reviewed `v0.3.4` tip and confirm no other
|
|
96
96
|
worktree has `main` checked out.
|
|
97
|
-
18. `git checkout main && git merge --ff-only v0.3.
|
|
97
|
+
18. `git checkout main && git merge --ff-only v0.3.4`. No merge commit, no
|
|
98
98
|
rebase, no reset. Verify `main` equals the reviewed SHA and is clean.
|
|
99
99
|
19. `git fetch origin` once more; stop on any unexpected remote movement. Then
|
|
100
100
|
`git push origin main`. Never `--force` or `--force-with-lease`.
|
|
@@ -105,12 +105,12 @@ confirming the exact SHA and the commands.
|
|
|
105
105
|
Node 24 jobs must succeed on the exact release SHA. A red or pending run
|
|
106
106
|
blocks the tag; a flake is rerun only with concrete evidence, never
|
|
107
107
|
silenced with an unrelated change.
|
|
108
|
-
21. Reconfirm that tag `v0.3.
|
|
109
|
-
`git tag -a v0.3.
|
|
110
|
-
`git push origin v0.3.
|
|
108
|
+
21. Reconfirm that tag `v0.3.4` and the GitHub Release do not exist, then
|
|
109
|
+
`git tag -a v0.3.4 -m "Clio Coder 0.3.4"` on the green SHA and
|
|
110
|
+
`git push origin v0.3.4`.
|
|
111
111
|
22. The tag push triggers `.github/workflows/release.yml`, which requires a
|
|
112
112
|
successful `ci` run for the tagged SHA, verifies the tag matches
|
|
113
|
-
`package.json`, builds and audits the artifact, extracts the `## 0.3.
|
|
113
|
+
`package.json`, builds and audits the artifact, extracts the `## 0.3.4`
|
|
114
114
|
section of `CHANGELOG.md` as the release body, and attaches the tarball.
|
|
115
115
|
Do not create a release by hand. Verify the run's SHA, the notes, the
|
|
116
116
|
attached tarball, and the URL.
|
|
@@ -118,9 +118,9 @@ confirming the exact SHA and the commands.
|
|
|
118
118
|
## Part 6: npm publication (irreversible)
|
|
119
119
|
|
|
120
120
|
23. `npm whoami` and confirm the registry and account; reconfirm
|
|
121
|
-
`@iowarp/clio-coder@0.3.
|
|
121
|
+
`@iowarp/clio-coder@0.3.4` is still absent.
|
|
122
122
|
24. Obtain the operator's explicit dist-tag decision. `latest` makes this the
|
|
123
|
-
default install for every user; `--tag next` keeps `0.3.
|
|
123
|
+
default install for every user; `--tag next` keeps `0.3.3` as the default.
|
|
124
124
|
25. Run `npm publish` (or `npm publish --tag next`) once. `prepublishOnly`
|
|
125
125
|
re-runs `ci:release` as a safety net; it is not a substitute for Part 1.
|
|
126
126
|
26. A published version cannot be replaced. `npm unpublish` is restricted and
|
|
@@ -128,15 +128,15 @@ confirming the exact SHA and the commands.
|
|
|
128
128
|
|
|
129
129
|
## Part 7: post-publish verification and follow-ups
|
|
130
130
|
|
|
131
|
-
27. `npm view @iowarp/clio-coder@0.3.
|
|
131
|
+
27. `npm view @iowarp/clio-coder@0.3.4` and the selected dist-tag.
|
|
132
132
|
28. On a clean machine, `npm install -g @iowarp/clio-coder` from the registry
|
|
133
133
|
rather than from a local tarball, then repeat step 12 against it, plus
|
|
134
134
|
`configure` to a real target and one real turn when one is authorized.
|
|
135
135
|
This is the only step that tests what users actually receive.
|
|
136
|
-
29. From an installation of 0.3.
|
|
137
|
-
applies 0.3.
|
|
138
|
-
30.
|
|
139
|
-
|
|
136
|
+
29. From an installation of 0.3.3, verify `clio-coder upgrade` finds and
|
|
137
|
+
applies 0.3.4.
|
|
138
|
+
30. Record the SHA, CI URL, tag, GitHub Release URL, npm version and dist-tag,
|
|
139
|
+
tarball evidence, and the post-publish verification in the release report.
|
|
140
140
|
31. Maintainer follow-up, independent of the release: verify the commit
|
|
141
141
|
provenance email `clio-coder@iowarp.ai` on IOWarp-controlled GitHub and
|
|
142
142
|
GitLab identities such as `clio-coder-bot` or `iowarp-clio`, and upload
|