@iowarp/clio-coder 0.3.3 → 0.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +74 -0
- package/CONTRIBUTING.md +7 -7
- package/README.md +3 -3
- package/dist/{acp-P2AQILE2.js → acp-2BEHC4DL.js} +9 -8
- package/dist/{agents-72W3BI7I.js → agents-LNNFTM53.js} +29 -24
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5TWEIYDN.js → auth-KXXFI2VS.js} +14 -10
- package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
- package/dist/{chunk-YCWGATWI.js → chunk-24I7BN55.js} +2 -2
- package/dist/{chunk-EKMEHE4H.js → chunk-33YXPOE3.js} +2 -3
- package/dist/chunk-3BPUFZDL.js +37 -0
- package/dist/chunk-43AOLP7E.js +375 -0
- package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
- package/dist/{chunk-6SGHMWE3.js → chunk-5JGRAMKL.js} +5 -5
- package/dist/{chunk-V6RTAOC2.js → chunk-6US73PDB.js} +572 -51
- package/dist/{chunk-5UFT4SUX.js → chunk-6XXKFVSN.js} +3 -3
- package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
- package/dist/chunk-AOCYTWAV.js +449 -0
- package/dist/chunk-CFGTUFWB.js +67 -0
- package/dist/chunk-CJUB2JJ2.js +1478 -0
- package/dist/{chunk-FNTMWMX5.js → chunk-CKXWIANG.js} +14 -12
- package/dist/{chunk-PIWWS5BL.js → chunk-CYQKWTG3.js} +63 -78
- package/dist/{chunk-LZSJBIVT.js → chunk-DJVECN66.js} +271 -762
- package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
- package/dist/{chunk-ZWMF7253.js → chunk-E2ER4LJF.js} +304 -9
- package/dist/{chunk-STBPMHSX.js → chunk-EKY57CSP.js} +51 -84
- package/dist/{chunk-DUYJ5IO6.js → chunk-EYPA3EGJ.js} +12 -4
- package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
- package/dist/chunk-FYYLNIL5.js +313 -0
- package/dist/{chunk-OQ33BKR3.js → chunk-G7MUEIGA.js} +3 -60
- package/dist/chunk-GEYXPTRF.js +613 -0
- package/dist/chunk-GOXNB3AO.js +261 -0
- package/dist/{chunk-G4BMMOKF.js → chunk-HVDIIIQW.js} +2 -2
- package/dist/chunk-HWUFFB6L.js +83 -0
- package/dist/{chunk-4XUGQOHA.js → chunk-K7T3E2SR.js} +15 -8
- package/dist/chunk-K7VKOLQQ.js +15 -0
- package/dist/{chunk-UFIIWP2H.js → chunk-KHSFENX2.js} +8 -8
- package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
- package/dist/chunk-LCGCVYZ4.js +57 -0
- package/dist/chunk-LL4KHSZI.js +22 -0
- package/dist/{chunk-PAJK6MAQ.js → chunk-LYF7OHWH.js} +42 -15
- package/dist/{chunk-POHLU5DW.js → chunk-M6L6IDJG.js} +3 -3
- package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
- package/dist/{chunk-X4RCMKVQ.js → chunk-NDINPTJ4.js} +2 -2
- package/dist/{chunk-TZK7PACC.js → chunk-NILBFAPG.js} +14 -8
- package/dist/chunk-ODFEOB4F.js +1082 -0
- package/dist/{chunk-AGYYIBLL.js → chunk-OH3TOQTB.js} +6 -2
- package/dist/chunk-OZNBF4L3.js +23 -0
- package/dist/{chunk-DSELYM6W.js → chunk-PBTHKCPN.js} +30 -10
- package/dist/{verify-375KUB3Y.js → chunk-PCZJO5TI.js} +127 -42
- package/dist/{chunk-ED4KHGC3.js → chunk-PPAMZ32Z.js} +9 -2
- package/dist/{chunk-SRDMMSEP.js → chunk-QM3F2GKX.js} +1063 -1645
- package/dist/{chunk-X6IAEBZR.js → chunk-QNQHSOLF.js} +7 -7
- package/dist/{chunk-OC7FIQPC.js → chunk-R46L2BIR.js} +10 -7
- package/dist/{chunk-2TLUCQVG.js → chunk-RD5U66HV.js} +3 -3
- package/dist/{chunk-6N5PTWMY.js → chunk-RY3LY4J5.js} +50 -13
- package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
- package/dist/{chunk-OOJYHWRB.js → chunk-TSHXZTOQ.js} +6 -5
- package/dist/chunk-TZSKNMZG.js +434 -0
- package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
- package/dist/{chunk-VJWL6YS5.js → chunk-UUVG37B4.js} +2 -2
- package/dist/{chunk-COU2UHX6.js → chunk-VEZEGCGW.js} +170 -2
- package/dist/chunk-W6GROXXM.js +69 -0
- package/dist/{chunk-OAO4GE4M.js → chunk-WHGPSPT5.js} +2 -2
- package/dist/chunk-WHJYKASB.js +677 -0
- package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
- package/dist/{chunk-YHZX5GEU.js → chunk-XAKHZX5N.js} +2 -2
- package/dist/{chunk-TZTZS7QK.js → chunk-XE2VEJHX.js} +5 -3
- package/dist/{chunk-LM5TQCJZ.js → chunk-XF5N4U5A.js} +8 -7
- package/dist/{chunk-LWLEKMDQ.js → chunk-XXQNGV4M.js} +1073 -552
- package/dist/{chunk-KZWTDYJF.js → chunk-XYDYPRZI.js} +7 -7
- package/dist/chunk-ZGVHUX3M.js +66 -0
- package/dist/{chunk-LW6DSM3M.js → chunk-ZRGEBJ4T.js} +1192 -1119
- package/dist/{chunk-2DJ2KNFG.js → chunk-ZXF4XRKW.js} +202 -40
- package/dist/chunk-ZZMN5OM4.js +122 -0
- package/dist/cli/index.js +34 -30
- package/dist/{clio-JOU4FXVA.js → clio-M2KGYUFZ.js} +7 -6
- package/dist/{code-nav-7AX6FYE6.js → code-nav-GQNL7XA6.js} +8 -6
- package/dist/codewiki/build-worker.js +4 -4
- package/dist/{components-KELWS457.js → components-5TTYYX6G.js} +3 -3
- package/dist/{config-XCDVKR23.js → config-XUUYQIWO.js} +47 -35
- package/dist/{configure-4GAP54ZW.js → configure-IHJ7YOMV.js} +18 -15
- package/dist/{context-77FM5DV5.js → context-74JLXAWD.js} +18 -10
- package/dist/{context-4UOGGLQ5.js → context-75MIWW3U.js} +41 -29
- package/dist/{context-5VKGUVJJ.js → context-ZQ7SIFJV.js} +85 -9
- package/dist/{context-clear-XXJRLCJJ.js → context-clear-GYKWNUML.js} +41 -29
- package/dist/{context-index-BZ4UYMTC.js → context-index-SSR5ECNE.js} +3 -3
- package/dist/context-working-set-UX5KEP4J.js +1553 -0
- package/dist/{dispatch-runner-QPRDDBDX.js → dispatch-runner-GIJBHNFL.js} +47 -32
- package/dist/{docs-2C2LTVT2.js → docs-6FZSCG5B.js} +3 -3
- package/dist/{doctor-HR46URBJ.js → doctor-SVJ5BZCW.js} +12 -12
- package/dist/{eval-XSSNATB4.js → eval-CG6LLBLD.js} +54 -238
- package/dist/{evidence-6HG2PY2B.js → evidence-ZYFIEN42.js} +57 -28
- package/dist/{evolve-K7YU3NCY.js → evolve-QGEXEMDW.js} +36 -25
- package/dist/{extensions-QVDOHDGJ.js → extensions-ADGNCJJD.js} +3 -3
- package/dist/{fleet-VY3HHKN6.js → fleet-S5R4ZOQY.js} +73 -44
- package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-BHSNPBMH.js} +3 -3
- package/dist/{init-JYGXI3FK.js → init-5DRU55YR.js} +49 -37
- package/dist/memory-7YKKR6UC.js +467 -0
- package/dist/{models-I5QWSEOM.js → models-ZPOLRU2C.js} +24 -21
- package/dist/{monitor-GE4ID3IA.js → monitor-US5F5YGZ.js} +73 -46
- package/dist/{orchestrator-EM5MC3HM.js → orchestrator-E2AL4T5N.js} +1624 -1007
- package/dist/{paths-UXLN5YYZ.js → paths-E7KYAQWE.js} +3 -3
- package/dist/{reset-L2FQEE3E.js → reset-KZ652EK6.js} +6 -5
- package/dist/{run-ZU3QMZPZ.js → run-SRNBKDWD.js} +76 -54
- package/dist/{share-S5BZQC5I.js → share-CGZE33UP.js} +7 -6
- package/dist/{skills-X5VXCRNQ.js → skills-S2X4DLY5.js} +4 -4
- package/dist/{skills-eval-WKIHWTHR.js → skills-eval-W2GGIC4R.js} +40 -29
- package/dist/{targets-SNCPI2NR.js → targets-54SWINWB.js} +28 -23
- package/dist/{terminal-lease-BNAHVHBS.js → terminal-lease-SAIF2OGY.js} +6 -4
- package/dist/{uninstall-FZCQCDKC.js → uninstall-BVLWXKBT.js} +3 -3
- package/dist/{upgrade-JQHHPQ4K.js → upgrade-JKAR27XC.js} +20 -19
- package/dist/{usage-OR4O5SMZ.js → usage-MSAWCLX4.js} +79 -36
- package/dist/verifiers-NCBTHHN2.js +1220 -0
- package/dist/verify-X5HDROLA.js +25 -0
- package/dist/{wiki-generate-UEXP2ARI.js → wiki-generate-GUSOQ6ZP.js} +50 -37
- package/dist/worker/entry.js +90 -70
- package/dist/{workspace-G4ZWUIPR.js → workspace-ZJ6BFM3Q.js} +4 -4
- package/docs/README.md +8 -7
- package/docs/acp.md +1 -1
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-placement.md +1 -2
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +60 -26
- package/docs/config-knobs-audit.md +1 -2
- package/docs/configuration-and-targets.md +26 -1
- package/docs/context-engine.md +67 -13
- package/docs/context-working-set.md +194 -0
- package/docs/development-pipeline.md +1 -1
- package/docs/documentation-coverage.md +6 -6
- package/docs/documentation-guide.md +7 -6
- package/docs/environment-variables.md +2 -1
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +4 -32
- package/docs/evidence-and-memory.md +139 -7
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +2 -2
- package/docs/fleet-dispatch.md +49 -8
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +2 -2
- package/docs/middleware-and-components.md +19 -2
- package/docs/model-catalog.md +7 -9
- package/docs/observability.md +4 -4
- package/docs/proactive-memory.md +26 -16
- package/docs/prompt-envelope-and-tools.md +7 -5
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +43 -40
- package/docs/safety-model.md +49 -8
- package/docs/scientific-validation.md +21 -3
- package/docs/session-lifecycle.md +3 -3
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +79 -12
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +1 -1
- package/docs/tui-design.md +38 -4
- package/docs/worker-dispatch-mechanics.md +11 -1
- package/package.json +13 -13
- package/skills/meta/clio-test/SKILL.md +20 -17
- package/skills/meta/clio-test/evals.md +3 -3
- package/skills/meta/clio-test/references/harness.md +35 -6
- package/skills/meta/clio-test/references/test-map.md +20 -10
- package/skills/registry.yaml +2 -2
- package/skills/skill-marketplace.json +1 -1
- package/src/cli/agents.ts +2 -3
- package/src/cli/argv.ts +14 -1
- package/src/cli/context-working-set.ts +513 -0
- package/src/cli/context.ts +8 -0
- package/src/cli/evidence.ts +20 -2
- package/src/cli/fleet.ts +15 -0
- package/src/cli/index.ts +5 -1
- package/src/cli/memory.ts +272 -10
- package/src/cli/modes/json-stream.ts +2 -2
- package/src/cli/modes/print.ts +12 -1
- package/src/cli/run.ts +22 -2
- package/src/cli/targets.ts +12 -3
- package/src/cli/usage.ts +55 -7
- package/src/cli/verifiers.ts +325 -0
- package/src/core/bash-exec.ts +39 -14
- package/src/core/bus-events.ts +22 -4
- package/src/core/config.ts +54 -0
- package/src/core/defaults.ts +50 -3
- package/src/core/response-model-id.ts +134 -0
- package/src/core/toml.ts +62 -0
- package/src/core/verification-scripts.ts +6 -0
- package/src/core/workspace-files.ts +0 -1
- package/src/domains/agents/builtins/architect.md +1 -1
- package/src/domains/agents/builtins/verifier.md +3 -0
- package/src/domains/agents/catalog.ts +5 -4
- package/src/domains/agents/recipe.ts +54 -14
- package/src/domains/agents/result-contract.ts +7 -4
- package/src/domains/config/classify.ts +1 -0
- package/src/domains/context/bootstrap.ts +36 -27
- package/src/domains/context/project-metadata.ts +19 -63
- package/src/domains/context/prompt-context.ts +8 -0
- package/src/domains/context/working-set/contract.ts +161 -0
- package/src/domains/context/working-set/defaults.ts +28 -0
- package/src/domains/context/working-set/engine.ts +203 -0
- package/src/domains/context/working-set/fold.ts +62 -0
- package/src/domains/context/working-set/horizon.ts +38 -0
- package/src/domains/context/working-set/marker.ts +103 -0
- package/src/domains/context/working-set/path-index.ts +436 -0
- package/src/domains/context/working-set/payload.ts +152 -0
- package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
- package/src/domains/context/working-set/policies/index.ts +20 -0
- package/src/domains/context/working-set/policies/structural.ts +160 -0
- package/src/domains/context/working-set/project.ts +132 -0
- package/src/domains/context/working-set/protect.ts +109 -0
- package/src/domains/context/working-set/recall.ts +177 -0
- package/src/domains/context/working-set/replay/controls.ts +112 -0
- package/src/domains/context/working-set/replay/load-clio.ts +199 -0
- package/src/domains/context/working-set/replay/metrics.ts +185 -0
- package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
- package/src/domains/context/working-set/replay/report.ts +139 -0
- package/src/domains/context/working-set/replay/runner.ts +325 -0
- package/src/domains/context/working-set/replay/synthetic.ts +422 -0
- package/src/domains/context/working-set/replay/trace.ts +21 -0
- package/src/domains/context/working-set/visible.ts +54 -0
- package/src/domains/dispatch/budget-envelope.ts +396 -0
- package/src/domains/dispatch/contract.ts +2 -0
- package/src/domains/dispatch/extension.ts +81 -27
- package/src/domains/dispatch/orphan-recovery.ts +1 -0
- package/src/domains/dispatch/receipt-integrity.ts +4 -0
- package/src/domains/dispatch/state.ts +1 -0
- package/src/domains/dispatch/types.ts +10 -3
- package/src/domains/dispatch/validation.ts +14 -0
- package/src/domains/dispatch/worker-spawn.ts +14 -3
- package/src/domains/eval/metrics/evidence.ts +0 -116
- package/src/domains/eval/metrics/invariants.ts +1 -1
- package/src/domains/eval/runners/clio-run.ts +1 -10
- package/src/domains/eval/runners/external-command.ts +2 -29
- package/src/domains/eval/schema/suite.ts +0 -7
- package/src/domains/eval/suites/run.ts +1 -7
- package/src/domains/evidence/build.ts +112 -45
- package/src/domains/evidence/eval.ts +24 -7
- package/src/domains/evidence/index.ts +53 -0
- package/src/domains/evidence/ordering.ts +12 -0
- package/src/domains/evidence/run-trust.ts +221 -0
- package/src/domains/evidence/store.ts +46 -6
- package/src/domains/evidence/trust-status.ts +854 -0
- package/src/domains/evidence/types.ts +26 -0
- package/src/domains/memory/index.ts +22 -0
- package/src/domains/memory/operations.ts +58 -1
- package/src/domains/memory/promotion.ts +281 -0
- package/src/domains/memory/prompt-section.ts +25 -5
- package/src/domains/memory/proposal.ts +51 -7
- package/src/domains/memory/task-bank.ts +3 -2
- package/src/domains/memory/task-memory-handoff.ts +181 -24
- package/src/domains/memory/task-memory-policy.ts +3 -1
- package/src/domains/memory/types.ts +37 -0
- package/src/domains/memory/validate.ts +178 -0
- package/src/domains/middleware/memory-intervention.ts +38 -25
- package/src/domains/middleware/runtime.ts +6 -0
- package/src/domains/middleware/skills-reminder.ts +19 -4
- package/src/domains/middleware/stalled-turn.ts +208 -5
- package/src/domains/middleware/types.ts +10 -0
- package/src/domains/observability/contract.ts +6 -1
- package/src/domains/observability/cost.ts +20 -4
- package/src/domains/observability/extension.ts +2 -2
- package/src/domains/providers/index.ts +3 -0
- package/src/domains/providers/model-discovery.ts +9 -0
- package/src/domains/providers/runtime-resolution.ts +38 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +97 -16
- package/src/domains/providers/types/context-window-slots.ts +18 -0
- package/src/domains/providers/types/runtime-descriptor.ts +3 -1
- package/src/domains/safety/autonomy.ts +1 -1
- package/src/domains/safety/call-target.ts +211 -14
- package/src/domains/safety/decision-presentation.ts +268 -0
- package/src/domains/safety/default-path-policy.ts +8 -0
- package/src/domains/safety/finish-contract.ts +4 -3
- package/src/domains/safety/policy-engine.ts +48 -6
- package/src/domains/safety/redaction.ts +73 -0
- package/src/domains/session/compaction/compact.ts +23 -1
- package/src/domains/session/compaction/cut-point.ts +2 -0
- package/src/domains/session/compaction/tokens.ts +16 -1
- package/src/domains/session/context-ledger.ts +12 -1
- package/src/domains/session/decision-board.ts +4 -0
- package/src/domains/session/entries.ts +110 -1
- package/src/domains/session/history.ts +68 -19
- package/src/domains/session/manager.ts +9 -2
- package/src/domains/session/migrations/index.ts +22 -3
- package/src/domains/session/usage.ts +24 -7
- package/src/engine/acp/event-mapper.ts +7 -0
- package/src/engine/acp/server.ts +32 -2
- package/src/engine/agent.ts +18 -1
- package/src/engine/apis/lmstudio.ts +25 -4
- package/src/engine/apis/openai-completions.ts +147 -22
- package/src/engine/claude/sdk-runtime.ts +8 -2
- package/src/engine/claude/tool-safety.ts +13 -0
- package/src/engine/loop-guard.ts +27 -3
- package/src/engine/session.ts +9 -3
- package/src/engine/worker-events.ts +4 -3
- package/src/engine/worker-runtime.ts +59 -54
- package/src/entry/orchestrator.ts +34 -5
- package/src/interactive/chat-loop-messages.ts +40 -6
- package/src/interactive/chat-loop.ts +13 -0
- package/src/interactive/chat-panel.ts +17 -1
- package/src/interactive/chat-renderer.ts +49 -24
- package/src/interactive/clio-editor.ts +44 -7
- package/src/interactive/context-meter.ts +10 -0
- package/src/interactive/context-overlay.ts +120 -7
- package/src/interactive/context-recall-command.ts +110 -0
- package/src/interactive/cost-overlay.ts +39 -8
- package/src/interactive/dispatch-board.ts +212 -35
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +6 -1
- package/src/interactive/interactive-input-runtime.ts +11 -1
- package/src/interactive/interactive-presentation.ts +11 -1
- package/src/interactive/interactive-slash-runtime.ts +37 -1
- package/src/interactive/memory-overlay.ts +89 -4
- package/src/interactive/model-session-replay.ts +21 -0
- package/src/interactive/overlay-ask-user-lifecycle.ts +1 -1
- package/src/interactive/overlay-frame.ts +5 -2
- package/src/interactive/overlay-general-openers.ts +46 -1
- package/src/interactive/overlay-key-routing.ts +41 -1
- package/src/interactive/overlay-lifecycle.ts +11 -4
- package/src/interactive/overlay-permission-lifecycle.ts +23 -8
- package/src/interactive/overlay-session-lifecycle.ts +8 -4
- package/src/interactive/overlay-transitions.ts +11 -0
- package/src/interactive/overlays/ask-user.ts +74 -30
- package/src/interactive/overlays/decisions.ts +3 -1
- package/src/interactive/permission-hint.ts +35 -0
- package/src/interactive/permission-overlay.ts +95 -45
- package/src/interactive/renderers/tool-execution.ts +37 -51
- package/src/interactive/session-last-turn.ts +8 -1
- package/src/interactive/session-transcript.ts +2 -2
- package/src/interactive/session-usage-reseed.ts +36 -10
- package/src/interactive/slash-commands.ts +31 -4
- package/src/interactive/status/summary.ts +5 -0
- package/src/interactive/status/types.ts +5 -0
- package/src/interactive/terminal-lease.ts +1 -0
- package/src/interactive/turn-context.ts +333 -110
- package/src/interactive/turn-middleware.ts +7 -6
- package/src/interactive/turn-runtime.ts +37 -8
- package/src/interactive/turn-state.ts +3 -0
- package/src/interactive/worker-progress.ts +440 -0
- package/src/interactive/worker-stream.ts +51 -110
- package/src/tools/agent-tools.ts +39 -7
- package/src/tools/ask-user.ts +21 -1
- package/src/tools/bash.ts +144 -82
- package/src/tools/builtin-tool-catalog.ts +11 -5
- package/src/tools/context/index.ts +107 -5
- package/src/tools/context/surface.ts +3 -2
- package/src/tools/core-bootstrap.ts +21 -0
- package/src/tools/dispatch-arguments.ts +8 -0
- package/src/tools/dispatch-event-text.ts +19 -0
- package/src/tools/dispatch-runner.ts +9 -7
- package/src/tools/dispatch.ts +24 -1
- package/src/tools/monitor.ts +43 -20
- package/src/tools/registry.ts +72 -10
- package/src/tools/result-disposition.ts +706 -0
- package/src/tools/result-shaping.ts +321 -20
- package/src/tools/safe-exec.ts +2 -0
- package/src/tools/verify/authoring.ts +1120 -0
- package/src/tools/verify/catalog.ts +346 -0
- package/src/tools/verify/index.ts +13 -3
- package/src/tools/verify/scripts.ts +135 -37
- package/src/tools/verify/surface.ts +9 -5
- package/src/tools/worker-evidence.ts +54 -12
- package/src/worker/spec-contract.ts +43 -3
- package/dist/chunk-J7CWMCQD.js +0 -255
- package/dist/chunk-T6YILFSB.js +0 -80
- package/dist/chunk-VAKQQHWR.js +0 -434
- package/dist/chunk-VPAYEGVX.js +0 -184
- package/dist/chunk-XBXAASKX.js +0 -18
- package/dist/memory-WFZMGYHX.js +0 -236
- package/src/domains/eval/metrics/chaos-stream.ts +0 -93
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
# Evidence Corpus and Long-Term Memory
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.6). Use it to design, validate, and simulate memory proposals, approval loops, pruning rules, and token budgets.
|
|
5
5
|
|
|
6
|
-
Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.
|
|
6
|
+
Clio Coder treats run claims and agent lessons as structured artifacts to support reproducibility and scientific provenance. In evaluations such as [SWE-bench](https://www.swebench.com), capturing granular execution evidence is essential for validating agent claims. Evidence corpora are deterministic directories built from run ledgers, receipts, sessions, audits, and eval artifacts. In v0.3.6, forensic evidence auto-builds on dispatch run completion: when a run finalizes, the observability domain automatically compiles the evidence bundle under `<dataDir>/evidence/run-<id>/` and updates a compact sidecar index row in `<stateDir>/evidence-index.json`. Long-term memory records are local, evidence-linked, and only injected after explicit approval. Use the TUI [`/view`](observability.md) command for interactive inspection of receipts, dispatch output, durable tool output, compaction summaries, and session accountability before building or citing evidence.
|
|
7
7
|
|
|
8
8
|
Source of truth: `src/domains/evidence/**`, `src/domains/memory/**`, `src/cli/evidence.ts`, and `src/cli/memory.ts`.
|
|
9
9
|
|
|
@@ -48,6 +48,7 @@ Run/session evidence files:
|
|
|
48
48
|
├── audit-linked.jsonl
|
|
49
49
|
├── receipt.json
|
|
50
50
|
├── gate-decisions.json
|
|
51
|
+
├── trust-status.json
|
|
51
52
|
├── protected-artifacts.json
|
|
52
53
|
├── findings.json
|
|
53
54
|
└── findings.md
|
|
@@ -67,6 +68,7 @@ Eval evidence adds `eval-result.json` and uses empty receipt/protected-artifact
|
|
|
67
68
|
| `audit-linked.jsonl` | Audit rows linked to run/session context when available. |
|
|
68
69
|
| `receipt.json` | Receipt bundle (`{ version: 1, receipts: [...] }`); only receipts that pass integrity verification contribute verified fields. |
|
|
69
70
|
| `gate-decisions.json` | Integrity-verified review verdicts, compete winner selections, and winner confirmations discovered from linked receipt ids. |
|
|
71
|
+
| `trust-status.json` | Canonical per-run six-axis trust projections derived from authenticated receipts, gate decisions, grounded validation artifacts, and exact finish-contract audit rows. |
|
|
70
72
|
| `protected-artifacts.json` | Protected artifact state/events. |
|
|
71
73
|
| `findings.json` / `findings.md` | Structured and readable findings. |
|
|
72
74
|
|
|
@@ -159,6 +161,76 @@ whether applicable validation evidence was observed. Briefing provenance is
|
|
|
159
161
|
also distinct from bounded project-context provenance: both can be absent or
|
|
160
162
|
present independently, and neither hash is evidence for the other.
|
|
161
163
|
|
|
164
|
+
### Canonical trust status
|
|
165
|
+
|
|
166
|
+
`src/domains/evidence/trust-status.ts` defines the version 1 canonical trust
|
|
167
|
+
status. It is a six-axis algebra, not an overall trust verdict, confidence
|
|
168
|
+
percentage, or pass/fail score. Consumers project only the axes needed for a
|
|
169
|
+
decision and preserve every other axis unchanged.
|
|
170
|
+
|
|
171
|
+
| Axis | Closed states | Question answered |
|
|
172
|
+
|---|---|---|
|
|
173
|
+
| Artifact integrity | `verified`, `failed`, `absent`, `unknown`, `not_applicable` | Did the integrity verifier authenticate the referenced artifact? |
|
|
174
|
+
| Validation grounding | `validated`, `failed`, `ungrounded`, `absent`, `unknown`, `not_applicable` | What correctness-bearing validation was observed and grounded? |
|
|
175
|
+
| Independent review | `passed`, `failed`, `inconclusive`, `not_independent`, `absent`, `unknown`, `not_applicable` | What outcome did an authenticated independent reviewer or judge record? |
|
|
176
|
+
| Context provenance | `recorded`, `invalid`, `absent`, `unknown`, `not_applicable` | Is the origin of briefing, project context, or linked evidence recorded consistently? |
|
|
177
|
+
| Autonomy enforcement | `enforced`, `approximated`, `bypassed`, `absent`, `unknown`, `not_applicable` | How faithfully did the runtime enforce the selected authority? |
|
|
178
|
+
| Completion evidence | `evidenced`, `incomplete`, `limited`, `absent`, `unknown`, `not_applicable` | What did the finish contract observe at the completion boundary? |
|
|
179
|
+
|
|
180
|
+
`absent` means no fact was recorded and carries a reason but no invented
|
|
181
|
+
attribution. `unknown` means a named source exists but cannot establish the
|
|
182
|
+
answer. `not_applicable` means a named authority determined that the axis does
|
|
183
|
+
not apply. Every non-absent state names both its source and its authority.
|
|
184
|
+
Sources may retain up to 16 typed artifact references. References contain an
|
|
185
|
+
artifact kind, identifier, and optional SHA-256 digest; they never embed the
|
|
186
|
+
artifact body. Normalization sorts the references and rejects duplicates,
|
|
187
|
+
unbounded lists, unknown fields, invalid identifiers, and sources that are not
|
|
188
|
+
permitted to speak for an axis.
|
|
189
|
+
|
|
190
|
+
The composition rules prohibit cross-axis promotion:
|
|
191
|
+
|
|
192
|
+
- Verified artifact integrity never promotes validation grounding.
|
|
193
|
+
- Recorded context provenance never promotes validation or correctness.
|
|
194
|
+
- A passing review never establishes authorship or context origin.
|
|
195
|
+
- Enforced autonomy never promotes completion evidence.
|
|
196
|
+
- A completion self-report never promotes validation grounding. The linked
|
|
197
|
+
`completion_contract` audit row is the run's own report of what it did, so it
|
|
198
|
+
reaches completion evidence and no other axis. Validation grounding is filled
|
|
199
|
+
only by independently observed executions the session ledger recorded.
|
|
200
|
+
|
|
201
|
+
The current adapters apply the following persisted-format compatibility rules.
|
|
202
|
+
They do not mutate receipt, gate-decision, evidence-bundle, or session formats.
|
|
203
|
+
|
|
204
|
+
| Existing persisted fact | Canonical mapping |
|
|
205
|
+
|---|---|
|
|
206
|
+
| Missing receipt | Every receipt-owned axis is `absent` with `artifact_missing`. |
|
|
207
|
+
| Current receipt present but integrity not checked | Artifact integrity is `unknown`; the receipt's own digest never authenticates itself. The other receipt-owned axes are `absent` with `not_observed` until authentication succeeds. |
|
|
208
|
+
| Historical receipt missing its integrity block | Receipt-owned axes are `unknown` through the compatibility source, even if a caller presents a contradictory positive verification result. |
|
|
209
|
+
| Integrity verification succeeds or fails | Artifact integrity is `verified` or `failed`. A failure leaves the receipt-owned validation grounding, context provenance, and autonomy enforcement `absent`; no untrusted receipt claim contributes a positive state. Validation the session ledger observed on its own (a validation command that ran and exited 0) still grounds the run, so a tampered run can read `artifactIntegrity: failed` beside `validationGrounding: validated`. The two axes name different artifacts and different authorities, and the bundle's `receipt-integrity` finding is what flags the pairing. |
|
|
210
|
+
| Receipt `verification.state: verified` | Validation grounding is `validated` unless a stronger typed failure or ungrounded claim is present. |
|
|
211
|
+
| Receipt `verification.state: unverified` | Validation grounding is `absent` with `not_observed`; lack of a validation tool is not a failed validation. |
|
|
212
|
+
| Receipt verification `unknown` or `not_applicable` | Validation grounding preserves `unknown` or `not_applicable`. A missing historical verification field maps to `unknown`. |
|
|
213
|
+
| Typed receipt validation or result-contract quality | A passing correctness-bearing fact maps to `validated`; a failing fact maps to `failed`; an ungrounded passing claim maps to `ungrounded`. |
|
|
214
|
+
| Valid bounded project context or valid briefing hash | Context provenance is `recorded`. Explicit project-context tier `none` with no briefing is `not_applicable`; a missing historical field is `unknown`; a contradictory block is `invalid`. |
|
|
215
|
+
| Gate decision | An authenticated independent pass or fail maps to `passed` or `failed`. Correlated review maps to `not_independent`. Unauthenticated artifacts map to `unknown`; operator or full-auto confirmation alone is `not_applicable` to independent review. |
|
|
216
|
+
| Receipt autonomy grade | `mediated`, `approximated`, and `bypassed` map to `enforced`, `approximated`, and `bypassed`. A dangerous-bypass flag always normalizes to `bypassed`; a missing historical block is `unknown`. |
|
|
217
|
+
| Finish-contract assessment | `validation_evidence`, `unvalidated_mutation`, `explicit_limitation`, and `no_mutation` map to `evidenced`, `incomplete`, `limited`, and `not_applicable`. A run whose receipt was presented and rejected downgrades `evidenced` to `unknown`: the row still points at its own record, but a rejected receipt authenticates nothing about the run it names. |
|
|
218
|
+
| Malformed audit row identifier | A blank or whitespace-only optional identifier is treated as absent. The row falls back to its derived correlation id or drops out of the trust projection; it never aborts the bundle. |
|
|
219
|
+
| Bundle without `trust-status.json` | Inspection reports `projection: historical_format` with no canonical run projections. It never reconstructs positive states from older summary tags. |
|
|
220
|
+
|
|
221
|
+
Receipt inspection, worker output, monitor details, and evidence rebuilding all
|
|
222
|
+
use the same authenticated receipt projection boundary. Evidence rebuilding
|
|
223
|
+
then composes independently authenticated gate decisions and exact
|
|
224
|
+
finish-contract records without changing receipt-owned axes. Findings such as
|
|
225
|
+
`no-validation`, `proxy-validation`, `external-approximation`, and
|
|
226
|
+
`external-bypass` are selected from the canonical states, while their detailed
|
|
227
|
+
domain artifacts remain in the receipt, gate, audit, and trace files.
|
|
228
|
+
|
|
229
|
+
The canonical aggregate is an additive projection for downstream work. Receipt
|
|
230
|
+
integrity remains version 15, evidence bundles remain version 1, gate decisions
|
|
231
|
+
remain version 2, and no persisted receipt field or cryptographic algorithm
|
|
232
|
+
changes.
|
|
233
|
+
|
|
162
234
|
### Mutation-Report Grounding
|
|
163
235
|
|
|
164
236
|
Mutation-report receipts are grounded directly against observed tool events recorded in the run ledger:
|
|
@@ -174,7 +246,8 @@ Mutation-report receipts are grounded directly against observed tool events reco
|
|
|
174
246
|
|
|
175
247
|
```bash
|
|
176
248
|
clio-coder memory list
|
|
177
|
-
clio-coder memory propose --from-evidence <evidenceId>
|
|
249
|
+
clio-coder memory propose --from-evidence <evidenceId> [scope options]
|
|
250
|
+
clio-coder memory promote --from-handoff <path> [--entry <id>...] --scope <scope> [scope options]
|
|
178
251
|
clio-coder memory approve <memoryId>
|
|
179
252
|
clio-coder memory reject <memoryId>
|
|
180
253
|
clio-coder memory prune --stale
|
|
@@ -195,6 +268,8 @@ The store is capped at `500` records and is sorted by scope, key, creation time,
|
|
|
195
268
|
```mermaid
|
|
196
269
|
stateDiagram-v2
|
|
197
270
|
evidence --> proposed: propose --from-evidence
|
|
271
|
+
taskBank --> proposed: /memory selected-entry action
|
|
272
|
+
redactedHandoff --> proposed: promote --from-handoff
|
|
198
273
|
proposed --> approved: approve <id>
|
|
199
274
|
proposed --> rejected: reject <id>
|
|
200
275
|
approved --> rejected: reject <id>
|
|
@@ -205,6 +280,45 @@ stateDiagram-v2
|
|
|
205
280
|
|
|
206
281
|
Records must cite at least one evidence ID to be considered for prompt injection. Rejected records remain in the store until stale pruning so the same bad lesson is not immediately re-proposed from the same evidence.
|
|
207
282
|
|
|
283
|
+
Task-bank promotion is a reviewed export from transient execution memory. The
|
|
284
|
+
`/memory` overlay offers repo and global proposal actions only on selected
|
|
285
|
+
knowledge and procedural rows. Status remains private and cannot enter the
|
|
286
|
+
promotion service. The first global action arms a warning, and the second
|
|
287
|
+
action acknowledges the broader applicability. A successful action writes an
|
|
288
|
+
unapproved record and names the separate `memory approve` command required to
|
|
289
|
+
make it injectable.
|
|
290
|
+
|
|
291
|
+
The CLI consumes a version 2 `clio-task-memory` handoff snapshot. Omitting
|
|
292
|
+
`--entry` proposes every knowledge and procedural entry; repeating `--entry`
|
|
293
|
+
selects exact entry IDs. Version 2 snapshots carry source session, evidence,
|
|
294
|
+
runtime, agent, timestamps, and export-redaction facts. Version 1 snapshots
|
|
295
|
+
remain seedable but cannot be promoted because they do not carry source
|
|
296
|
+
session or evidence provenance.
|
|
297
|
+
|
|
298
|
+
Every promotion redacts secret-shaped values before `records.json` is written.
|
|
299
|
+
The durable provenance block records the source kind, session, selected entry,
|
|
300
|
+
entry class and timestamps, plus the replacement count and source field paths.
|
|
301
|
+
Promotion never approves its own output.
|
|
302
|
+
|
|
303
|
+
### Explicit scope selection
|
|
304
|
+
|
|
305
|
+
Reviewed scope options are closed to four choices:
|
|
306
|
+
|
|
307
|
+
| Scope | Required selection | Validation |
|
|
308
|
+
| --- | --- | --- |
|
|
309
|
+
| `repo` | `--repository <canonical-absolute-path>` | The path must exist and already equal its canonical absolute identity. Symlink aliases and paths containing unresolved segments are rejected. |
|
|
310
|
+
| `global` | `--acknowledge-global` | The acknowledgement is separate from `--scope global`. |
|
|
311
|
+
| `runtime` | `--runtime <id>` | The ID must be valid and must occur in the source provenance. |
|
|
312
|
+
| `agent` | `--agent <id>` | The ID must be valid and must occur in the source provenance. |
|
|
313
|
+
|
|
314
|
+
The same options may be added to `memory propose --from-evidence`. With no
|
|
315
|
+
scope option, evidence proposals keep the existing inference order. An
|
|
316
|
+
explicit repository may differ from the repository that produced the
|
|
317
|
+
evidence, which supports a reviewed lesson about repository A learned while
|
|
318
|
+
working in repository B. Runtime and agent overrides may only select an exact
|
|
319
|
+
identity already recorded by the evidence. Global scope always requires its
|
|
320
|
+
own acknowledgement. No inference path widens an explicit choice.
|
|
321
|
+
|
|
208
322
|
---
|
|
209
323
|
|
|
210
324
|
## Prompt injection rules
|
|
@@ -215,7 +329,7 @@ Defaults:
|
|
|
215
329
|
|
|
216
330
|
| Constraint | Default |
|
|
217
331
|
| --- | --- |
|
|
218
|
-
|
|
|
332
|
+
| Base scopes | `global`, `repo` |
|
|
219
333
|
| Token budget | `400` estimated tokens |
|
|
220
334
|
| Max records | `5` |
|
|
221
335
|
| Required status | `approved: true` |
|
|
@@ -224,6 +338,12 @@ Defaults:
|
|
|
224
338
|
|
|
225
339
|
Rendered memory lines always cite record ID, scope, lesson, and evidence IDs. The prompt tells the model not to extrapolate beyond cited findings.
|
|
226
340
|
|
|
341
|
+
Interactive main-agent sessions additionally admit records for the exact
|
|
342
|
+
active runtime. `clio-coder run --agent` admits records for the exact resolved
|
|
343
|
+
runtime and selected agent. Runtime and agent records use structured identity
|
|
344
|
+
fields; `appliesWhen` text cannot grant either applicability. Missing,
|
|
345
|
+
malformed, or different active identities exclude those records.
|
|
346
|
+
|
|
227
347
|
### Repository-scoped identity
|
|
228
348
|
|
|
229
349
|
Repository memory is selected by an exact canonical absolute-path identity. The interactive orchestrator and `clio-coder run --agent` compute that identity from the active working directory; symlink aliases collapse to the same key. A repository move, a different Git worktree path, a subdirectory launch, a malformed identity, or a missing identity does not inherit another repository's memory. Global records are unaffected.
|
|
@@ -236,15 +356,27 @@ Every `scope: "repo"` record must carry:
|
|
|
236
356
|
|
|
237
357
|
The structured `repository` field is the only applicability mechanism: store validation rejects repo records without it, and `appliesWhen` tokens never grant repository applicability. There is intentionally no automatic path rewrite for moved repositories or worktrees: a filesystem move produces a different identity and the record simply stops applying until it is re-scoped with new evidence.
|
|
238
358
|
|
|
359
|
+
Runtime and agent records follow the same fail-closed shape:
|
|
360
|
+
|
|
361
|
+
```json
|
|
362
|
+
{ "runtime": { "kind": "runtime", "key": "openai" } }
|
|
363
|
+
```
|
|
364
|
+
|
|
365
|
+
```json
|
|
366
|
+
{ "agent": { "kind": "agent", "key": "coder" } }
|
|
367
|
+
```
|
|
368
|
+
|
|
369
|
+
Only the field matching the record scope is present.
|
|
370
|
+
|
|
239
371
|
---
|
|
240
372
|
|
|
241
373
|
## Recommended workflow
|
|
242
374
|
|
|
243
375
|
1. Build evidence from the run/session/eval that taught the lesson.
|
|
244
376
|
2. Inspect the evidence and findings.
|
|
245
|
-
3. Propose memory from the evidence.
|
|
246
|
-
4. Review the proposed lesson
|
|
247
|
-
5. Approve only if it is durable and useful.
|
|
377
|
+
3. Propose memory from the evidence, or promote selected public task memory from `/memory` or a redacted handoff.
|
|
378
|
+
4. Review the proposed lesson, source provenance, redaction facts, and exact scope.
|
|
379
|
+
5. Approve only if it is durable and useful under that scope.
|
|
248
380
|
6. Reject incorrect or overbroad records.
|
|
249
381
|
7. Prune stale records periodically.
|
|
250
382
|
|
package/docs/evolution.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Evolution and Change Manifests
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive change manifest editor, authority risk assessor, and checklist workspace is located at [docs/html/evolution_blueprint.html](html/evolution_blueprint.html) (Version: 0.3.6).
|
|
5
5
|
|
|
6
6
|
Clio Coder uses change manifests to make harness changes reviewable, falsifiable, and rollback-friendly. CLIO stands for Context Layer for Input/Output, named for the Greek muse of history. A manifest is JSON, generated or checked with `clio-coder evolve manifest`, and should describe what changed, why, what evidence supports it, what could regress, how to validate it, and how to roll it back.
|
|
7
7
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Exit Codes & Machine-Readable Output Contracts
|
|
2
2
|
|
|
3
|
-
This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.
|
|
3
|
+
This document specifies the process exit codes, machine-readable JSON streaming formats, standard I/O separation rules, and `--help` conventions across all Clio Coder CLI commands in `v0.3.6`.
|
|
4
4
|
|
|
5
5
|
Source implementations: `src/cli/` and `src/entry/`.
|
|
6
6
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Extensions, Prompt Templates, Skills, and Share Archives
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/extensions_blueprint.html](html/extensions_blueprint.html) (Version: 0.3.6).
|
|
5
5
|
|
|
6
6
|
Clio Coder has lightweight community-oriented resource packaging. Extensions are filesystem bundles that contribute prompts and skills. Share archives are portable JSON files for moving project/user Clio resources between machines or collaborators. Themes are built into the engine and are no longer loaded from extensions.
|
|
7
7
|
|
|
@@ -251,7 +251,7 @@ Share archives are single JSON files:
|
|
|
251
251
|
"formatVersion": 1,
|
|
252
252
|
"manifest": {
|
|
253
253
|
"format": "clio.share.v1",
|
|
254
|
-
"clioVersion": "0.3.
|
|
254
|
+
"clioVersion": "0.3.6",
|
|
255
255
|
"createdAt": "...",
|
|
256
256
|
"files": []
|
|
257
257
|
},
|
package/docs/fleet-dispatch.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fleet Dispatch
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.
|
|
3
|
+
> **Interactive Spec Available:** An interactive fleet node topology planner, scout router, receipt verifier, and failure taxonomy simulator is located at [docs/html/fleet_dispatch_blueprint.html](html/fleet_dispatch_blueprint.html) (Version: 0.3.6).
|
|
4
4
|
|
|
5
5
|
Clio Coder dispatches bounded worker agents. With a fleet configured, those
|
|
6
6
|
workers run on remote machines over SSH while the orchestrator keeps every
|
|
@@ -77,7 +77,11 @@ briefing wins. Supplying both `task` and `tasks` fails instead of choosing one.
|
|
|
77
77
|
After approval, execution consumes only the registry-owned resolved plan, so
|
|
78
78
|
later mutation of raw arguments cannot change either field.
|
|
79
79
|
|
|
80
|
-
Recipes
|
|
80
|
+
Recipes declare a default with `budget: {toolCalls, readReserve, synthesis}`. They may also declare `maximum: {toolCalls, readReserve}` inside that object. A recipe without `maximum` is an exact pin, which preserves the fixed behavior of existing recipes. A ranged recipe admits the optional dispatch request `budget: {toolCalls, readReserve, retryRevision?}` only when the request is inside its maximum. `retryRevision` has the same two integer fields and preauthorizes the ceiling that a later automatic retry, bounded result-contract revision, or review revision may select. The loop guard raises a result-contract revision boundary only in this case. A phase without that ceiling cannot grow and retains the existing text-only repair behavior.
|
|
81
|
+
|
|
82
|
+
`toolCalls` is the admitted-call phase boundary. The final `readReserve` slots accept canonical `read` plus the agent's granted mutation tools, so a writer can still deliver inside its own reserve. Admission requires integers and `0 <= readReserve < toolCalls` for every declared phase. `synthesis: true` forces a text-only final round, while `false` stops after the admitted phase. `guardrails.workerToolCallCap` remains the operator-controlled lifetime ceiling and always wins when lower. A default may be clamped by a lower operator cap so default callers retain their prior behavior; an explicit request outside the operator cap is denied.
|
|
83
|
+
|
|
84
|
+
Admission computes one immutable envelope with the recipe policy, invocation request, effective worker budget, and every clamp or escalation reason. Native workers and Claude SDK enforce the effective budget. Claude Code, Antigravity, and ACP delegation reject invocation envelopes because their black-box loops cannot provide equivalent per-call mediation. Before launch, every admitted WorkerSpec v3 still contains one concrete effective budget and a settings fingerprint. The envelope provenance is sealed in the run ledger and receipt and appears in monitor, fleet status, and the live fleet card.
|
|
81
85
|
|
|
82
86
|
## Node setup
|
|
83
87
|
|
|
@@ -579,6 +583,19 @@ basis unknown/not applicable). A read-only Scout can therefore report `receipt_i
|
|
|
579
583
|
bounded `project_context` provenance are also rendered independently; neither
|
|
580
584
|
hash substitutes for the other.
|
|
581
585
|
|
|
586
|
+
The canonical terminology for these facts is the six-axis trust status in
|
|
587
|
+
[`evidence-and-memory.md`](evidence-and-memory.md#canonical-trust-status).
|
|
588
|
+
Receipt integrity projects onto artifact integrity; receipt verification,
|
|
589
|
+
typed quality, and validation grounding project onto validation grounding;
|
|
590
|
+
gate decisions project onto independent review; briefing and project context
|
|
591
|
+
project onto context provenance; and `autonomyEnforcement` projects onto
|
|
592
|
+
autonomy enforcement. A receipt does not contain independent-review or
|
|
593
|
+
completion-evidence outcomes merely because it is sealed. Those axes remain
|
|
594
|
+
`absent` until an authenticated gate artifact or finish assessment is composed.
|
|
595
|
+
In particular, verified integrity cannot validate claims, known provenance
|
|
596
|
+
cannot establish correctness, and a review verdict cannot establish
|
|
597
|
+
authorship.
|
|
598
|
+
|
|
582
599
|
Gate references point backward: a reviewer references the builder it
|
|
583
600
|
reviewed, a revise builder references the reviewer whose findings it
|
|
584
601
|
received, and a judge references every candidate. Because a worker receipt
|
|
@@ -645,6 +662,27 @@ hard block.
|
|
|
645
662
|
renders `local`), gate badges (`gate reviewer c2`), reroute badges, live
|
|
646
663
|
tool activity (names only; arguments never cross the worker stdout seam),
|
|
647
664
|
and a per-worker context meter.
|
|
665
|
+
- `Enter` on the selected Fleet Runs row opens its worker detail: the phase,
|
|
666
|
+
the running call with a redacted action descriptor (`bash running npm
|
|
667
|
+
test`), and the bounded tail of the worker's own prose. The default list
|
|
668
|
+
stays compact, so a fan-out of scouts costs one card each until an operator
|
|
669
|
+
opens one. Detail follows the cursor rather than pinning to a run.
|
|
670
|
+
- The board and the transcript worker block read one projection
|
|
671
|
+
(`src/interactive/worker-progress.ts`), so they cannot disagree about what a
|
|
672
|
+
worker is saying or touching. It keeps 40 lines and 4096 bytes of tail, 8
|
|
673
|
+
distinct tool names, 4 recent actions, and accepts 16 KB of delta bytes per
|
|
674
|
+
250 ms; what the bounds refuse is counted and named on the card beside the
|
|
675
|
+
`/view dispatch:<runId>` deep link.
|
|
676
|
+
- Action descriptors are composed where the arguments are trusted: the tool
|
|
677
|
+
registry's admission path, the Claude tool mapper, and the ACP update
|
|
678
|
+
mapper. Each reads a fixed verb vocabulary and a fixed argument-field
|
|
679
|
+
allowlist, scrubs credentials, strips escape sequences, and bounds the
|
|
680
|
+
result to 64 characters before it crosses the worker stdout seam. Raw
|
|
681
|
+
argument objects never cross at all.
|
|
682
|
+
- Reasoning content is never displayed. The detail may name a `thinking`
|
|
683
|
+
phase and the usage facts the card already carries, never the text.
|
|
684
|
+
- Settlement replaces the provisional tail with the sealed receipt's answer;
|
|
685
|
+
a run whose receipt cannot be read keeps its own last durable message.
|
|
648
686
|
- The context meter renders the worker's last-message context occupancy
|
|
649
687
|
against the model's context window: healthy below 80 percent, warn from 80,
|
|
650
688
|
critical from 95.
|
|
@@ -654,6 +692,7 @@ hard block.
|
|
|
654
692
|
- The monitor tool reports the node and reroute lineage on `status`, `list`,
|
|
655
693
|
and `collect`.
|
|
656
694
|
- `clio-coder fleet status [--json]` shows the durable ledger view cross-process.
|
|
695
|
+
- A worker permission escalation uses the `Worker escalation` consequence tier in operator presentation. The tier names the worker agent and run and describes where the one-shot answer returns. It does not approve the request, change the worker's inherited autonomy, or weaken the safety net; the existing worker escalation protocol remains the only resolution path.
|
|
657
696
|
|
|
658
697
|
## Speculation observer
|
|
659
698
|
|
|
@@ -682,11 +721,13 @@ After `npm run build`, an operator with a configured model target can run the
|
|
|
682
721
|
single-turn, read-only fleet lifecycle check explicitly:
|
|
683
722
|
|
|
684
723
|
```bash
|
|
685
|
-
|
|
724
|
+
npm run live:fleet-dispatch -- --target <id> [--model <wireId>] [--thinking medium]
|
|
686
725
|
```
|
|
687
726
|
|
|
688
|
-
It is not part of deterministic CI. The
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
727
|
+
It is not part of deterministic CI. The driver
|
|
728
|
+
(`benchmarks/internal/live-fleet-dispatch.ts`) copies the repository into a
|
|
729
|
+
committed temporary workspace, sandboxes all Clio config, state, data, and
|
|
730
|
+
cache under a scratch home holding only the chosen target, exercises Scout,
|
|
731
|
+
bounded spot-checking, detached Debugger briefing, steering, wait, and
|
|
732
|
+
collect, and fails if any workspace content changes. A failed run retains its
|
|
733
|
+
scratch tree for diagnosis.
|
package/docs/glossary.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Clio Coder Glossary
|
|
2
2
|
|
|
3
|
-
This document defines core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
|
|
3
|
+
This document defines the 45 core architectural concepts and terminology used throughout Clio Coder, mapped to their authoritative TypeScript type definitions in `src/`.
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
@@ -165,3 +165,23 @@ This document defines core architectural concepts and terminology used throughou
|
|
|
165
165
|
### 40. Delegate
|
|
166
166
|
- **Definition**: Another coding agent Clio drives over ACP stdio as if it were a worker, configured under `delegation.agents` and invoked with `/delegate`. A delegate is a foreign harness, not a model target.
|
|
167
167
|
- **Owning Type**: `DelegationAgentConfig` in `src/core/defaults.ts`.
|
|
168
|
+
|
|
169
|
+
### 41. Working Set
|
|
170
|
+
- **Definition**: The part of the session ledger the model receives on the next request. It is the ledger with the current eviction projection applied, and it exists only in memory; the ledger itself is never narrowed. Not to be confused with the context ledger, which is the accounting of how the window is spent.
|
|
171
|
+
- **Owning Type**: `WorkingSetView` in `src/domains/context/working-set/contract.ts`.
|
|
172
|
+
|
|
173
|
+
### 42. Projection
|
|
174
|
+
- **Definition**: The pure, idempotent transform from ledger entries to the entries the replay builder hands the model. It substitutes markers for evicted bodies and drops thinking from closed turns, returning unaffected entries by reference. Nothing about it is persisted.
|
|
175
|
+
- **Owning Type**: `projectWorkingSet` in `src/domains/context/working-set/project.ts`.
|
|
176
|
+
|
|
177
|
+
### 43. Eviction
|
|
178
|
+
- **Definition**: The decision that a tool-result body or an assistant turn's thinking leaves the working set, recorded as an append-only ledger entry with a typed reason. It removes nothing: the original entry stays in the ledger and stays visible in the transcript, `/resume`, `/fork`, and the HTML export.
|
|
179
|
+
- **Owning Type**: `ContextEvictionEntry` in `src/domains/session/entries.ts`.
|
|
180
|
+
|
|
181
|
+
### 44. Recall
|
|
182
|
+
- **Definition**: Readmitting an evicted body by ref, through `context(scope="recall", ref=...)` for the model or `/context recall <ref>` for the operator. A recall does not un-evict: the marker stays where it was so the provider prefix cache is untouched, and repeated recalls of one ref are the churn signal.
|
|
183
|
+
- **Owning Type**: `ContextRecallEntry` in `src/domains/session/entries.ts`; resolution in `resolveRecall` in `src/domains/context/working-set/recall.ts`.
|
|
184
|
+
|
|
185
|
+
### 45. Marker
|
|
186
|
+
- **Definition**: The byte-stable one-line stub the projection renders in place of an evicted body, naming the ref, the reason, the tool, the size, and the exact recall call. It carries no timestamp and no counter, because a marker whose bytes drifted between renders would cold-start the prefix cache on a turn that evicted nothing new.
|
|
187
|
+
- **Owning Type**: `renderMarker` in `src/domains/context/working-set/marker.ts`.
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Clio Coder is designed to be self-contained and platform-compliant. This document outlines the default directory paths, file purposes, permission levels, and lifecycle commands (`install`, `reset`, `upgrade`, and `uninstall`). Clio Coder installs from npm as `@iowarp/clio-coder` (`npm install -g @iowarp/clio-coder`, published since v0.3.0) or from a source checkout with a deterministic local symlink; the CLI classifies both install kinds and `clio-coder upgrade` handles each.
|
|
4
4
|
|
|
5
5
|
> [!TIP]
|
|
6
|
-
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.
|
|
6
|
+
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.3.6). You can open it directly in any web browser to view details dynamically.
|
|
7
7
|
|
|
8
8
|
---
|
|
9
9
|
|
|
@@ -229,7 +229,7 @@ Upgrading from 0.3.1 to 0.3.3 is automated:
|
|
|
229
229
|
clio-coder upgrade
|
|
230
230
|
```
|
|
231
231
|
|
|
232
|
-
Key lifecycle and operational updates in v0.3.
|
|
232
|
+
Key lifecycle and operational updates in v0.3.6:
|
|
233
233
|
- Upgraded the underlying engine SDK libraries to 0.84.0 with signal-aware OAuth cancellation.
|
|
234
234
|
- Hardened migration resilience: damaged `credentials.yaml` files no longer block upgrades when no renames are needed (#121); `--skip-migrations` is available as a recovery override.
|
|
235
235
|
- Fullscreen TUI mode (`terminal.tuiMode`, `terminal.fullscreenScrollbar`) is available via Settings → Terminal (restart required). Adaptive presentation pacing is the live `terminal.smoothStreaming` setting; 0.3.3 defaults it to `off`, with conservative `auto` and explicit `on` available from the same section.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Middleware and Component Registry
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.3.6).
|
|
5
5
|
|
|
6
6
|
Clio Coder has two related but separate surfaces:
|
|
7
7
|
|
|
@@ -114,7 +114,24 @@ Middleware hook budgets are phase-aware through `DEFAULT_MIDDLEWARE_HOOK_BUDGETS
|
|
|
114
114
|
|
|
115
115
|
Per-phase budgets can be overridden via `CLIO_CODER_HOOK_BUDGET_<PHASE>_MS` or global `CLIO_CODER_HOOK_BUDGET_MS`. Warmup grace exempts initial calls (`DEFAULT_HOOK_BUDGET_WARMUP_CALLS = 1`), and steady-state warnings trigger when at least 3 of the last 5 post-warmup calls exceed budget (`DEFAULT_HOOK_BUDGET_WINDOW = 5`, `DEFAULT_HOOK_BUDGET_THRESHOLD = 3`). Overruns are reported but do not abort the turn. The orchestrator and workers share the middleware contract, but worker guard state is process-local.
|
|
116
116
|
|
|
117
|
-
Middleware reminders are visible request text, not hidden prompt state. `turn_start` reminders flush into the same accepted request; `turn_end` reminders flush once on the next request.
|
|
117
|
+
Middleware reminders are visible request text, not hidden prompt state. `turn_start` reminders flush into the same accepted request; `turn_end` reminders flush once on the next request. A `request_continuation` from any producer is capped at one automatic continuation per user prompt; a second producer in the same prompt gets a footer notice that the nudge is spent, and the turn is handed back to the operator rather than looped.
|
|
118
|
+
|
|
119
|
+
### Built-in registrations
|
|
120
|
+
|
|
121
|
+
These ship in every interactive session. Each is one bounded behavior with a visible reminder; none changes a tool policy or a safety verdict.
|
|
122
|
+
|
|
123
|
+
| Id | Hooks | What it does |
|
|
124
|
+
| --- | --- | --- |
|
|
125
|
+
| `nudge.stalled-turn` | `turn_end` | The one declarative rule. A turn that called no tools and ended on an announced action ("Next I will inspect `src/cli/index.ts`") is continued once with a reminder to perform it or say plainly that it is finished. Questions, "let me know", conditional offers ("if you want me to"), and completion statements are not announcements. |
|
|
126
|
+
| `observer.skills-reminder` | `turn_start`, `turn_end` | Once per session, on the first substantive turn, when installed or installable skills exist, injects one line teaching the suggestion protocol: list with `context(scope="skills")`, open the reply with `Suggested skill: /skill <name>` when one matches, then continue the task in the same turn. Only the operator loads a skill. At `turn_end`, a reply that made the suggestion and stopped with only listing calls behind it is continued once (#184): the suggestion is not the task. Greetings do not spend the session's one reminder; a resumed or forked session never gets one. |
|
|
127
|
+
| `observer.task-board-reminder` | `turn_start` | Once per session, when the operator's text literally enumerates three or more steps (`1)`, `2.`, `step 3:`, or three bulleted lines), injects one line asking for `tasks action="plan"` before the first edit. Prose that merely mentions numbers never counts. |
|
|
128
|
+
| `nudge.open-tasks` | `turn_end` | A settled work turn (one that called tools) that ends while the session task board still has pending or active tasks is continued once with the open list. Pure conversation turns, aborted or errored turns, and boards where every remaining task is blocked do not trigger. |
|
|
129
|
+
| `nudge.detached-dispatch` | `turn_end` | A settled turn that ends while a detached dispatch batch has every run terminal and uncollected is continued once, naming the ready batches; `monitor mode="collect"` clears it, including across resume. Batches with runs still in flight, and surfaces without `monitor`, do not trigger. |
|
|
130
|
+
| `nudge.read-only-exploration` | `after_tool`, `turn_end` | After nine or more read-only calls (`read`, `grep`, `find`, `ls`, `code_nav`, read-only shell) in one user turn without a successful Scout dispatch, injects one advisory to delegate broad reconnaissance to Scout. One advisory per user turn, and only on surfaces that have `dispatch`. |
|
|
131
|
+
| `rail.unbacked-worker-claim` | `after_tool`, `turn_end` | A reply that reports worker or Scout results in a turn with no `dispatch` call gets one warning that the claim is not backed by a receipt. A `[worker result]` note the operator shared is receipt-backed and exempt. No continuation: the operator decides. |
|
|
132
|
+
| `observer.memory-intervention` | `after_tool` | Every `memory.intervention.everyNTools` tool calls, asks a background model for a bounded reflection over the recent window and injects it as a reminder when it arrives. Governed by the `memory.intervention` settings block. |
|
|
133
|
+
|
|
134
|
+
Two coded controls sit beside the registrations rather than among them. `tool-choice-control` turns `require_tool` and `lock_tools` effects into the provider's tool-choice field for the next round: a required tool clears when that tool starts, a lock lasts until the next submitted turn and outranks later requirements. `hook-receipts` is the durable ring (200 entries, throttled to one write per two seconds) of user-defined hook executions that `clio-coder config inspect` reads.
|
|
118
135
|
|
|
119
136
|
User-defined hook declarations load from three places: `<extensionRoot>/hooks.yaml`, `.clio-coder/hooks.yaml`, and `.clio-coder/hooks.local.yaml`. A hook can be `prompt`, `effect`, or `command`. Command hooks run an argv array without a shell, under the workspace with a timeout and bounded output, and every hook execution emits a receipt.
|
|
120
137
|
|
package/docs/model-catalog.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Model Catalog, Runtime Refresh, and Field Notes
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.3.6).
|
|
5
5
|
|
|
6
6
|
Clio Coder treats a selectable model as the intersection of three sources:
|
|
7
7
|
|
|
@@ -39,14 +39,12 @@ worker spec and receipt are written.
|
|
|
39
39
|
|
|
40
40
|
## Benchmarking Models
|
|
41
41
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
The benchmarks record context-window, thinking, sampling, weight quantization, and KV-cache settings so sweeps can be compared consistently.
|
|
42
|
+
The public benchmark adapters under [benchmarks/community/](../benchmarks/community/)
|
|
43
|
+
(SWE-bench Lite, Terminal-Bench, SciCode, HumanEval) drive Clio through
|
|
44
|
+
`clio-coder run --json` or `clio-coder eval run`, each taking `--target` and
|
|
45
|
+
`--model` from the configured targets. `benchmarks/README.md` has the
|
|
46
|
+
commands. The run manifests record the target profile (runtime, model,
|
|
47
|
+
thinking level) so sweeps can be compared consistently.
|
|
50
48
|
|
|
51
49
|
## What "sanctioned" means
|
|
52
50
|
|
package/docs/observability.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Observability Viewer
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.6).
|
|
5
5
|
|
|
6
6
|
`/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
|
|
7
7
|
|
|
@@ -69,7 +69,7 @@ Clio resolves directories under platform-specific XDG defaults (on Linux, these
|
|
|
69
69
|
| **Dispatch outputs** | Logs and ledger records detailing worker execution. | `<stateDir>/runs.json` and `<stateDir>/receipts/<runId>.json` |
|
|
70
70
|
| **Task ledgers** | Per-turn task-board goals, active runs, required validation evidence, and operator-task provenance when present. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
|
|
71
71
|
| **Workspace outputs** | Latest successful `artifact`, `write`, or `edit` result for each normalized path on the active session branch. Missing files remain visible as durable recorded facts. | Recorded path beneath the session metadata `cwd` |
|
|
72
|
-
| **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<
|
|
72
|
+
| **Tool outputs** | Offloaded large outputs or execution logs. | `<stateDir>/scratch/<sessionId>/<sha256 of the captured text>.txt` |
|
|
73
73
|
| **Protected artifacts** | Validation-protected artifact metadata and its absolute artifact path when available. | Session ledger record plus the protected workspace path |
|
|
74
74
|
| **Compaction** | Summaries of compacted history sessions. | `<stateDir>/sessions/<cwdHash>/<sessionId>/current.jsonl` |
|
|
75
75
|
| **Prompt manifests** | One validated record per prompt compile: `systemPromptHash`, previous hash, token estimate, thinking dial at compile time, per-section token estimates, and per-fragment content hashes. Identifies the exact compiled prompt and supports hash diffs without storing prompt text. Malformed records appear as an explicit read-error artifact. | `<stateDir>/sessions/<cwdHash>/<sessionId>/prompt-manifest.jsonl` |
|
|
@@ -159,8 +159,8 @@ The base provenance sets, steering, routing, quality, worker identity, and resul
|
|
|
159
159
|
| `safety.toolTelemetry.ingestionErrors` | `number` | Current dispatch receipts | Malformed or lost frames, event-fold/source errors, and drain timeouts that make otherwise mediated telemetry incomplete | experimental |
|
|
160
160
|
| `safety.toolTelemetry.unfinished` | `{ tool, count }[]` | Current dispatch receipts | Tool starts that had no matching finish when the receipt sealed | experimental |
|
|
161
161
|
| `safety.toolTelemetry.workspaceMutationPossible` | `boolean` | Current dispatch receipts | Whether incomplete or unavailable telemetry could conceal a shared-workspace mutation; retry admission fails closed when true | experimental |
|
|
162
|
-
| `autonomyEnforcement.grade` | `string` | Always in v0.3.
|
|
163
|
-
| `autonomyEnforcement.autonomy` | `string` | Always in v0.3.
|
|
162
|
+
| `autonomyEnforcement.grade` | `string` | Always in v0.3.6 | The autonomy grade level enforced for the run | experimental |
|
|
163
|
+
| `autonomyEnforcement.autonomy` | `string` | Always in v0.3.6 | The effective autonomy level name (e.g. auto-edit, suggest, read-only, full-auto) | experimental |
|
|
164
164
|
| `autonomyEnforcement.externalMode` | `string` | When running external worker | The execution mode of the external worker runtime | experimental |
|
|
165
165
|
| `autonomyEnforcement.dangerousBypass` | `boolean` | When running external worker | Whether a safety bypass was explicitly activated | experimental |
|
|
166
166
|
| `validationGrounding.claimed` | `number` | Validation grounding evaluated | Count of validations claimed by worker | experimental |
|
package/docs/proactive-memory.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Proactive task memory
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.
|
|
3
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.6).
|
|
4
4
|
|
|
5
5
|
Clio's proactive task memory protects long-running work from behavioral state
|
|
6
6
|
decay: a requirement, environment fact, failed attempt, or diagnosis can still
|
|
@@ -133,7 +133,7 @@ it runs detached from it.
|
|
|
133
133
|
| Interval | After `memory.intervention.everyNTools` completed tools since the last prompted step; default 10. This is the nondeterministic/citation-gated path. |
|
|
134
134
|
| Tool-error streak | Two consecutive error outcomes. A successful tool resets the streak. |
|
|
135
135
|
| Loop signal | Reuses the orchestrator loop guard's verdict; it does not infer a second competing loop detector. |
|
|
136
|
-
| Repeated failure | The rules tier records failed
|
|
136
|
+
| Repeated failure | The rules tier records failed operation fingerprints and annotates the failing tool result once the same failure appears twice in the bounded trajectory. |
|
|
137
137
|
| Post-compaction | The first turn start after compaction restores status and knowledge once, without a model call, because compaction is precisely where execution facts leave the active window. |
|
|
138
138
|
|
|
139
139
|
### Two delivery channels
|
|
@@ -145,11 +145,13 @@ repeated failure uses exactly one of them:
|
|
|
145
145
|
|
|
146
146
|
- **Mid-turn annotation.** The second identical failure appends one cited
|
|
147
147
|
`Memory:` advisory to that tool's own result, through the existing
|
|
148
|
-
`annotate_tool_result` effect the loop guard already uses. The advisory
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
the
|
|
148
|
+
`annotate_tool_result` effect the loop guard already uses. The advisory uses
|
|
149
|
+
the canonical result-disposition digest when one is available. Older hook
|
|
150
|
+
producers fall back to the first tool-error line that names a problem. Every
|
|
151
|
+
digest is redacted and byte-capped before it reaches the task bank. The model
|
|
152
|
+
reads the advisory on its very next round. This is spent once per operation
|
|
153
|
+
fingerprint per turn and re-earned in a later turn, because the same command
|
|
154
|
+
failing again after an operator turn is news again.
|
|
153
155
|
- **Next-turn reminder.** Post-compaction reactivation and any background-model
|
|
154
156
|
reminder ride the `inject_reminder` buffer into the next submitted turn, inside
|
|
155
157
|
the visible `<system-reminder>` block, and persist in the session ledger.
|
|
@@ -319,15 +321,23 @@ operation while leaving deterministic protection active.
|
|
|
319
321
|
Measured on the shipped prompt against `google/gemma-4-26b-a4b-qat`, across ten
|
|
320
322
|
live steps and forty controlled runs on the same route.
|
|
321
323
|
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
324
|
+
Earlier measurements found that the tier wrote `update_status` reliably and
|
|
325
|
+
`save_knowledge` rarely. At that time a successful trajectory step carried an
|
|
326
|
+
opaque result fingerprint, so a window of successful reads told the model which
|
|
327
|
+
files were touched and nothing about what was in them.
|
|
328
|
+
|
|
329
|
+
A current trajectory step keeps two fields with different jobs. The operation
|
|
330
|
+
fingerprint identifies repeated calls and remains derived only from the tool name
|
|
331
|
+
and arguments. The result digest is human-readable diagnostic content from the
|
|
332
|
+
canonical result-disposition projection, with explicit source provenance. Secret
|
|
333
|
+
redaction and a 240-byte cap apply before the digest reaches the task bank or the
|
|
334
|
+
background request. A metadata-only disposition contributes outcome facts and no
|
|
335
|
+
captured body. Results without a canonical disposition use a redacted deterministic
|
|
336
|
+
fallback, so older tool producers remain useful without gaining a second model
|
|
337
|
+
summarizer.
|
|
338
|
+
|
|
339
|
+
Three candidate causes were ruled out in the earlier implementation by
|
|
340
|
+
controlled runs that changed one variable at a time:
|
|
331
341
|
|
|
332
342
|
- rewriting the prompt's second worked example to carry a `save_knowledge` moved
|
|
333
343
|
nothing, and made the model emit no operations at all in four of five runs;
|