@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
package/docs/observability.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Observability Viewer
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
`/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
|
|
7
7
|
|
|
@@ -13,6 +13,35 @@
|
|
|
13
13
|
|
|
14
14
|
`/view` opens a full-screen split viewer. The left pane groups artifacts by category and supports type-to-filter. The right pane renders the selected artifact with pager controls. `Tab` or `Shift+Tab` switches between the artifact list and details. `Left` and `Right` jump to the previous or next non-empty category from either pane; `Up` and `Down` select artifacts in the list or scroll details in the content pane. Category jumps honor the active filter and wrap at the ends. `v` verifies a selected receipt. `o` shows the absolute backing path through the notice channel when the selected artifact has one; pathless artifacts produce a warning notice instead. In the list pane, `Esc` clears a non-empty filter before a second `Esc` closes the viewer.
|
|
15
15
|
|
|
16
|
+
## Trace retention and state usage
|
|
17
|
+
|
|
18
|
+
The SQLite trace mirror at `<state-dir>/trace.sqlite` is rebuildable and bounded. By default Clio retains terminal runs for 30 days and limits the allocated database to 128 MiB (134,217,728 bytes), whichever limit is reached first. The policy runs after each dispatch or interactive turn becomes terminal. It deletes a run as one unit across `runs`, `phases`, `events`, `envelopes`, `gate_results`, `agent_sessions`, and `processes`. A `queued` or `running` run is never a candidate, even when its start time is older than the age cutoff or its rows put the store over the byte limit.
|
|
19
|
+
|
|
20
|
+
Two environment variables configure the automatic policy:
|
|
21
|
+
|
|
22
|
+
| Variable | Default | Valid values |
|
|
23
|
+
| --- | ---: | --- |
|
|
24
|
+
| `CLIO_CODER_TRACE_RETENTION_DAYS` | `30` | An integer of at least 1. |
|
|
25
|
+
| `CLIO_CODER_TRACE_MAX_BYTES` | `134217728` | An integer of at least 1,048,576. |
|
|
26
|
+
|
|
27
|
+
An operator can apply the current policy immediately or supply one-command overrides:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
clio-coder trace prune
|
|
31
|
+
clio-coder trace prune --max-age-days 14 --max-bytes 67108864
|
|
32
|
+
clio-coder trace prune --json
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
The command reports terminal runs removed, total rows removed across the seven run-owned tables, physical bytes reclaimed from `trace.sqlite` and its WAL sidecars, whether `VACUUM` ran, and how many live runs were protected. Age pruning uses a terminal run's `ended_at`. Size pruning removes the oldest terminal runs until the live database pages fit or no terminal candidate remains.
|
|
36
|
+
|
|
37
|
+
Deleting SQLite rows creates reusable pages but does not normally reduce the file. Clio runs `VACUUM` when at least 20 percent of allocated pages are reclaimable, or whenever reclaiming deleted pages is necessary to enforce the 128 MiB bound. It then truncates the WAL. Smaller deletions remain available for SQLite to reuse and avoid rewriting the whole database on every completed run.
|
|
38
|
+
|
|
39
|
+
`clio-coder doctor` includes a `state storage` row with the recursive byte total for the state directory and the largest top-level contributor. For example:
|
|
40
|
+
|
|
41
|
+
```text
|
|
42
|
+
OK state storage 96.4 MiB (101,082,624 bytes); largest contributor trace.sqlite at 89.9 MiB (94,248,960 bytes)
|
|
43
|
+
```
|
|
44
|
+
|
|
16
45
|
---
|
|
17
46
|
|
|
18
47
|
## The Evidence Spine End-to-End
|
|
@@ -69,8 +98,28 @@ An `EvidenceIndexRow` has the following schema:
|
|
|
69
98
|
| `turns` | `turns in window` and the `tokens` fact | Folded calls that were turns, so labelled calls are subtracted exactly as `/cost` subtracts them. |
|
|
70
99
|
| `sideQuestions` | `side questions in window` and the `tokens` fact | `/btw` rounds in the window. |
|
|
71
100
|
| `handoffs` | `handoffs in window` and the `tokens` fact | `/handoff` extraction rounds in the window. |
|
|
101
|
+
| `prewarms` | `pre-warms in window` and the `tokens` fact | Prompt pre-warm rounds in the window. |
|
|
102
|
+
| `backgroundMemorySteps` | `background memory steps in window` and the `tokens` fact | Proactive-memory model steps in the window. |
|
|
103
|
+
|
|
104
|
+
The last five fields appear only when at least one labelled call falls in the window, and each individual line is printed only when its own count is above zero. An archive with no labelled call in it renders exactly as it did before those fields existed, so their presence is itself the signal that money was spent beside a session. All four labelled kinds are subtracted from `turns` the same way, so a session's turn count never includes a round the operator did not take.
|
|
105
|
+
|
|
106
|
+
The report also prints a prompt-cache block, one row per session that recorded any cache telemetry:
|
|
107
|
+
|
|
108
|
+
```text
|
|
109
|
+
prompt cache by session (from backend timings and persisted verdicts)
|
|
110
|
+
session uncached prefill hot/partial/cold/small
|
|
111
|
+
3vpu6z19ee7t 130353 4/3/2/0
|
|
112
|
+
```
|
|
72
113
|
|
|
73
|
-
|
|
114
|
+
`uncached prefill` is the sum of the backend's own newly evaluated prompt tokens across every persisted call in that session, and reads `n/a` rather than `0` when the server reported no cache figure to subtract. The four counts are the per-call verdicts. Both facts are also in `--json` under a `session-cache` fact per session.
|
|
115
|
+
|
|
116
|
+
`clio-coder doctor` reports the same evidence for the latest session only, as one row, so a cache problem is visible without opening the TUI or a report:
|
|
117
|
+
|
|
118
|
+
```text
|
|
119
|
+
OK cache telemetry last session 3vpu6z19ee7t: hot 4 · partial 3 · cold 2 · small 0; top expected reason dispatch (3)
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The row reads `top expected reason none` when the session recorded verdicts but no expected-cold reason, and it degrades to a warning saying `no prompt-cache telemetry recorded` when the latest session has none at all, which is the honest answer for a target whose backend reports nothing rather than a claim of a perfect cache. "Latest" selects the most recent `current.jsonl` by its newest entry timestamp, falling back to the file's mtime, so the other diagnostic JSONL files in a session directory cannot be mistaken for the conversation.
|
|
74
123
|
|
|
75
124
|
### The Out-of-Turn Usage Store
|
|
76
125
|
|
|
@@ -102,6 +151,8 @@ A row has the following schema:
|
|
|
102
151
|
|
|
103
152
|
`repoIdentity` is the same cwd hash the session ledger is filed under, which is what lets `usage report --repo <path>` select these rows with the hash it already computes for the ledgers.
|
|
104
153
|
|
|
154
|
+
`label` is one of `side-question`, `handoff`, `prewarm`, or `background-memory`. The last two joined for the same reason as the first two: a prompt pre-warm and a proactive-memory step are provider calls the operator did not ask for and would otherwise never see, and neither appends anything to the session JSONL. A row may also carry `timing { durationMs }` and a `promptCache` block built from the backend's own prefill facts when the server reported them; a backend that reports no timings simply omits the block, as LM Studio's OpenAI-compatible port does.
|
|
155
|
+
|
|
105
156
|
---
|
|
106
157
|
|
|
107
158
|
## Artifact Categories and Path Layouts
|
package/docs/proactive-memory.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Proactive task memory
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.
|
|
3
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.3.9).
|
|
4
4
|
|
|
5
5
|
Clio's proactive task memory protects long-running work from behavioral state
|
|
6
6
|
decay: a requirement, environment fact, failed attempt, or diagnosis can still
|
|
@@ -120,7 +120,7 @@ malformed response, or telemetry failure is silent and never blocks a tool.
|
|
|
120
120
|
- `memory.intervention.everyNTools` (default `10`): Minimum completed-tool interval between background interventions.
|
|
121
121
|
- `memory.intervention.windowSteps` (default `8`): Completed tool-trajectory window analyzed during background evaluation.
|
|
122
122
|
- `memory.intervention.maxTokens` (default `400`): Bounds the rendered memory-bank and reminder context budget; the policy model output cap is a separate fixed `4,000`-token contract in `task-memory-policy.ts`, sized so that a model which reasons anyway still reaches its envelope.
|
|
123
|
-
- `memory.intervention.timeoutMs` (default `
|
|
123
|
+
- `memory.intervention.timeoutMs` (default `30000`): Wall-clock limit for one background memory-policy request. The step is detached, so this deadline never delays a turn, but it does hold a request slot on a real inference endpoint that your own turns and your dispatched workers queue against. The default is what a turn boundary can wait for rather than what a long-tailed route eventually answers in: on the reference route below, 23 of 60 steps ran past 30 seconds and 531 of the measured 1,666 seconds were spent beyond that mark. Raise it only if you have measured that your route's slow steps are the ones producing reminders, and read the trade in "Cost and the default decision" first.
|
|
124
124
|
|
|
125
125
|
## Trigger semantics
|
|
126
126
|
|
|
@@ -203,6 +203,88 @@ Thus `last` remains `injected` across such continuations until a later
|
|
|
203
203
|
tool-bearing or explicitly triggered memory step produces a new outcome (e.g.,
|
|
204
204
|
a healthy tool leading to `silent`).
|
|
205
205
|
|
|
206
|
+
## Cost and the default decision
|
|
207
|
+
|
|
208
|
+
The LLM tier costs real tokens, real seconds of model time, and a request slot on
|
|
209
|
+
a server that is usually the same machine the operator's own turns run on. Every
|
|
210
|
+
step is therefore accounted for the way a `/btw` side question is: one cost entry
|
|
211
|
+
under the `background-memory` label, which `/cost` shows as its own `memory steps`
|
|
212
|
+
row, and one durable row in `<stateDir>/usage/out-of-turn.jsonl` carrying the
|
|
213
|
+
usage, the call's duration, and the backend's prefill facts, which
|
|
214
|
+
`clio-coder usage report` folds after the process exits. `/memory` shows the
|
|
215
|
+
lifetime figures folded from `steps.jsonl`: steps, tokens, model time, and the
|
|
216
|
+
hit rate.
|
|
217
|
+
|
|
218
|
+
### The measurement
|
|
219
|
+
|
|
220
|
+
From one operator's `steps.jsonl`, 274 rows spanning 2026-08-14 to 2026-08-29 on
|
|
221
|
+
a small local background route:
|
|
222
|
+
|
|
223
|
+
| Figure | Value |
|
|
224
|
+
| --- | --- |
|
|
225
|
+
| Model-tier steps | 60 |
|
|
226
|
+
| Tokens | 137,205 |
|
|
227
|
+
| Model time | 1,666.6 s |
|
|
228
|
+
| Step latency | median 18.7 s, p90 70.2 s, max 102.5 s |
|
|
229
|
+
| Injections produced by the model tier | 6 |
|
|
230
|
+
| Hit rate | 10.0 percent |
|
|
231
|
+
| Cost per injection | 22,868 tokens and 278 s of model time |
|
|
232
|
+
| Model-tier injections in the last 5 days | 0 of 4 steps |
|
|
233
|
+
|
|
234
|
+
Four further injections in the same window came from the free rules tier, so the
|
|
235
|
+
lifetime total of 10 injections is not the model tier's score. Rules-tier
|
|
236
|
+
injections cost nothing.
|
|
237
|
+
|
|
238
|
+
### The decision
|
|
239
|
+
|
|
240
|
+
The default does not change, and it is a deliberate default rather than an
|
|
241
|
+
unexamined one:
|
|
242
|
+
|
|
243
|
+
- `memory.intervention.enabled` stays `true`. It runs the rules tier, which makes
|
|
244
|
+
no model calls, spends no tokens, and produced 4 of the 10 injections.
|
|
245
|
+
- The LLM tier stays opt-in through `background.target` and `background.model`,
|
|
246
|
+
which is already the case: an unset background role never resolves a client.
|
|
247
|
+
A 10 percent hit rate at 22,868 tokens per injection does not earn a default-on
|
|
248
|
+
position, and it is not so poor that it earns removal from an operator who has
|
|
249
|
+
measured their own route and wants it.
|
|
250
|
+
- The step deadline drops from 180 s to 30 s. This is the one behavioral change,
|
|
251
|
+
and it is a genuine trade: at 30 s, two of the six observed injections, at
|
|
252
|
+
53.6 s and 57.7 s, would have been cut, while 531 s of the 1,666 s spent would
|
|
253
|
+
not have been spent at all. The deadline is the bound on what one optional call
|
|
254
|
+
may hold a shared local server for, not a prediction of when a route answers.
|
|
255
|
+
- A step that would run on the endpoint the chat target is streaming against is
|
|
256
|
+
skipped with reason `endpoint_busy`, and the skip is recorded. On a single-slot
|
|
257
|
+
llama.cpp router the alternative is queueing behind the operator's own decoding
|
|
258
|
+
or evicting the resident model, and neither is a cost an optional call may
|
|
259
|
+
impose.
|
|
260
|
+
|
|
261
|
+
### What a background target costs on a shared local server
|
|
262
|
+
|
|
263
|
+
If `background.target` names the same server as `orchestrator.target`, that
|
|
264
|
+
server's slots are shared. On a llama.cpp router started with `--parallel 1`
|
|
265
|
+
there is exactly one, and the memory step and the operator's turn contend for it.
|
|
266
|
+
|
|
267
|
+
The consequence is worth stating plainly: a shared endpoint suppresses the model
|
|
268
|
+
tier rather than merely delaying it. A step is started from the `turn_end` hook,
|
|
269
|
+
which fires inside the streaming run at `agent_end`
|
|
270
|
+
(`src/interactive/turn-runtime.ts`), while the chat loop still holds its
|
|
271
|
+
foreground registration on that endpoint; the loop releases the hold afterwards,
|
|
272
|
+
in the `finally` around the run (`src/interactive/chat-loop.ts`). Every boundary
|
|
273
|
+
therefore finds the endpoint busy and records `dropped`/`endpoint_busy`. That is
|
|
274
|
+
the intended trade: an optional call may not take the one slot the operator's own
|
|
275
|
+
turn is using, and it may not make the server swap the resident model out. The
|
|
276
|
+
`/memory` step list and `steps.jsonl` say so on every boundary, so the tier is
|
|
277
|
+
visibly declining rather than quietly idle.
|
|
278
|
+
|
|
279
|
+
The second mechanism is an `expected cold` stamp, for the case where a step did
|
|
280
|
+
run on the chat endpoint. Its prompt is a trajectory rather than the chat prefix,
|
|
281
|
+
so the next turn's prefill is expected to be cold; `/context` names
|
|
282
|
+
`background_memory` as the reason instead of reporting an unexplained cold
|
|
283
|
+
prefix.
|
|
284
|
+
|
|
285
|
+
Pointing the background role at a second machine avoids both effects and is the
|
|
286
|
+
arrangement the tier is designed for.
|
|
287
|
+
|
|
206
288
|
## Choosing a background model
|
|
207
289
|
|
|
208
290
|
Memory reads a trajectory and writes a fixed envelope. It does not plan, and it
|
|
@@ -257,7 +339,7 @@ memory:
|
|
|
257
339
|
everyNTools: 10
|
|
258
340
|
windowSteps: 8
|
|
259
341
|
maxTokens: 400
|
|
260
|
-
timeoutMs:
|
|
342
|
+
timeoutMs: 30000
|
|
261
343
|
```
|
|
262
344
|
|
|
263
345
|
With `background.target` and `background.model` unset, Clio stays in the
|
|
@@ -289,12 +371,12 @@ percentile of 79.9, and a 95th of 131.6. Capability is not the constraint;
|
|
|
289
371
|
latency is, its spread is wide, and the detached step above is what makes the
|
|
290
372
|
tier usable anyway.
|
|
291
373
|
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
374
|
+
The deadline is a bound on what an optional call may hold that server for, not a
|
|
375
|
+
figure sized to capture the tail. The shipped 30000 sits above the median and
|
|
376
|
+
below the tail deliberately, and a step that exceeds it records `timeout` with
|
|
377
|
+
its work discarded. A route whose steps mostly record `timeout` is a
|
|
378
|
+
misconfigured deadline before it is a slow model, so read the ledger before
|
|
379
|
+
raising it: `/memory` shows the hit rate the raise would be buying.
|
|
298
380
|
|
|
299
381
|
The target ID is not hard-coded. Any configured orchestrator-eligible local
|
|
300
382
|
target and wire model can fill the background role. Before enabling it, use the
|
|
@@ -316,6 +398,25 @@ For an immediate kill switch, set `memory.intervention.enabled` to `false` in
|
|
|
316
398
|
`/settings`. Removing the background target instead returns to rules-only
|
|
317
399
|
operation while leaving deterministic protection active.
|
|
318
400
|
|
|
401
|
+
## Where what the tier writes ends up
|
|
402
|
+
|
|
403
|
+
A bank entry lives and dies with its session. When a reminder actually reaches
|
|
404
|
+
the operator, the entries it cited are also proposed into the durable store at
|
|
405
|
+
`<dataDir>/memory/records.json`, unapproved, scoped to the repository the session
|
|
406
|
+
is working in, with provenance naming the session and the source entry. That is
|
|
407
|
+
the one automatic writer of that file; everything else about it is unchanged.
|
|
408
|
+
`/memory` and `clio-coder memory list` show the proposal, and
|
|
409
|
+
`clio-coder memory approve <id>` is still a separate operator action, so nothing
|
|
410
|
+
the background plane produced reaches a system prompt without review. A step with
|
|
411
|
+
no session, or one running outside a canonical repository, proposes nothing:
|
|
412
|
+
global scope broadens applicability to every future session and is not a claim a
|
|
413
|
+
background step may make on the operator's behalf.
|
|
414
|
+
|
|
415
|
+
Rules-tier reminders are not proposed. Their entries are this middleware's own
|
|
416
|
+
one-line records of a repeated tool failure, and filing each one as a durable
|
|
417
|
+
lesson would fill the review queue with rows nobody asked for. They remain
|
|
418
|
+
promotable by hand from `/memory`.
|
|
419
|
+
|
|
319
420
|
## What the LLM tier actually writes
|
|
320
421
|
|
|
321
422
|
Measured on the shipped prompt against `google/gemma-4-26b-a4b-qat`, across ten
|
|
@@ -392,11 +493,23 @@ keeps one previous generation as `steps.jsonl.1`. Every exact-schema record has:
|
|
|
392
493
|
- `silent`, `injected`, `gated`, `timeout`, `malformed`, or `dropped` decision;
|
|
393
494
|
- count of cited entries, input/output/total memory-model tokens, and latency.
|
|
394
495
|
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
496
|
+
The same steps are also billed. See "Cost and the default decision" for the
|
|
497
|
+
`/cost` row, the durable out-of-turn usage row, and the lifetime figures `/memory`
|
|
498
|
+
folds out of this file.
|
|
499
|
+
|
|
500
|
+
`dropped` is the one outcome that ran no step. It has two causes, separated by
|
|
501
|
+
the row's reason: `step_in_flight` means the boundary triggered while an earlier
|
|
502
|
+
step still held the single in-flight slot, and `endpoint_busy` means the step
|
|
503
|
+
would have called the endpoint the chat target was streaming against. Both cost
|
|
504
|
+
no tokens and no latency, both leave their triggers pending for the next free
|
|
505
|
+
boundary, and neither replaces the operator-visible last decision. Counting
|
|
506
|
+
`dropped` rows against `llm` rows over a session is how a starved cadence becomes
|
|
507
|
+
visible.
|
|
508
|
+
|
|
509
|
+
A step that exceeds the deadline records `timeout`, never `silent`: reason
|
|
510
|
+
`deadline` when the policy's own race fired first, and `timed_out` when the
|
|
511
|
+
transport aborted at its deadline. Both are distinct from `client_error`, which
|
|
512
|
+
is a route that refused rather than a route that was slow.
|
|
400
513
|
|
|
401
514
|
The log contains no task, trajectory, bank, error, or reminder text. File creation,
|
|
402
515
|
rotation, serialization, and injected sinks are all best effort; a read-only
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Prompt Envelope and Tools
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Clio Coder keeps the model-facing envelope stable and moves enforcement into the runtime registry and safety policy.
|
|
7
7
|
|
|
@@ -13,6 +13,24 @@ The chat loop compiles one provider-facing system prompt for a session. The comp
|
|
|
13
13
|
|
|
14
14
|
The compiled prompt is reused byte-for-byte on ordinary submits. It recompiles only when that key changes or when config hot-reload invalidates the prompt cache. Path-scoped project rules can therefore recompile the prompt when a matching file enters working context. When recompilation changes the text, the session ledger records a `promptRecompiled` entry with the previous hash, new hash, and token estimate.
|
|
15
15
|
|
|
16
|
+
## Section order: stable prefix first
|
|
17
|
+
|
|
18
|
+
The compiled prompt lays its sections down in `SESSION_PROMPT_SECTION_ORDER` (`src/domains/prompts/compiler.ts`): identity, operating contract, delegation, skills, safety, tool contract, fleet, retrieval hints, project context, memory, runtime, then the operator-editable tail fragments (workspace root, Clio repo awareness, project rules, operator profile) in their own order.
|
|
19
|
+
|
|
20
|
+
One rule fixes that list. A section goes as late as its volatility, and anything that reads a clock, a probe, or a mutable store goes after everything that does not. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so a section that can change between two turns must not sit ahead of sections that cannot. The runtime block is last of the compiled sections because its `Context window: N` moves when the backend reloads a model or a co-residency clamp lands; memory sits just ahead of it because an approved memory record rewrites that section mid-session; project rules are dead last because path-scoped rules join the prompt when a matching file enters working context.
|
|
21
|
+
|
|
22
|
+
`Context window: N` is the window the backend will actually serve. A recorded loaded window outranks a probe, which reports a figure the target advertises without saying it is what is open, so a resumed session states the window its ledger measured rather than a re-probed server-wide number. Each prompt-manifest record carries that window and the layer that answered it (`contextWindow`, `contextWindowSource`) alongside a `version` for the prompt layout itself, so a recompile whose only cause was the window moving is explained by the record rather than inferred.
|
|
23
|
+
|
|
24
|
+
`PROMPT_MANIFEST_VERSION` (`src/domains/session/prompt-manifest.ts`) is `2` as of this release, and the reordering above is what moved it. The field is additive: a record written by 0.3.8 carries no `version` and reads back as version 1, so a `prompt-manifest.jsonl` from an older session still parses. The rule for the field is that it tracks the layout rather than the inputs. Bump it when the compiled text moves for a reason other than a changed fragment, a changed tool surface, or a changed setting, so that a resumed session has the version in hand to explain the single `promptRecompiled` entry its first compile writes.
|
|
25
|
+
|
|
26
|
+
### What not to add to the prefix
|
|
27
|
+
|
|
28
|
+
Two additions look free and are not.
|
|
29
|
+
|
|
30
|
+
The first is a terseness rule. It is tempting to cap the prose a model emits between tool calls, because that text is generated tokens on every hop of a long turn. Anthropic measured that exact change on Claude Code and reported a 3 percent quality regression, so a word-count or verbosity limit on inter-tool text is a bad trade: the tokens it saves are the cheapest ones in the turn, and the model's own narration of what it is about to do is load-bearing for what it then does. Bound tool results instead, where a single `grep` can cost thousands of tokens and the envelope caps already do the work.
|
|
31
|
+
|
|
32
|
+
The second is anything that varies with the wall clock or the working tree. No timestamp, no `git status`, no branch name, no session id, and no run id belongs anywhere in the compiled prefix. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so one such field turns the whole prompt into a cache miss on every turn for no information the model could not have asked a tool for. On the sprint's measurement server that is a whole 2,778-token prompt re-prefilled at 2.6 s where the same change behind the stable sections cost 516 tokens and 0.72 s. Volatile facts belong in the user message, in a tool result, or in the runtime block, which is last for this reason.
|
|
33
|
+
|
|
16
34
|
The disk fragments under `src/domains/prompts/fragments/` are layered by who reads them. `identity.clio` and `operating.contract` are constitutional: they render for every reader, name no tool, and state what is always true about Clio and her harness. `operating.delegation` (fleet coordination, receipts, spot-checks, shared `[worker result]` notes) renders only when `dispatch` is on the session's tool surface, and `operating.skills` (skill-shaped tasks, `/skill <name>` suggestions) only when `context` is; a fragment that teaches a tool is absent when the tool is, the same rule the Fleet block follows. `identity.docs-routing`, the directive to call `context(scope="docs")` before answering a question about Clio herself, follows the `context` gate too, while `identity.self-awareness` (installed paths, code outranks docs, configuration locations) names no tool and is unconditional. `operating.worker` (the assigned-task contract) renders only for dispatched workers, which never see the coordinator fragments. `safety.<level>` states what runs, what is approval-required, and what is blocked at the effective autonomy, in the safety net's action-class vocabulary (read, write, command, `system_modify`, `git_destructive`) and never by tool name, so the same body is true on every surface; the session and every worker read that one body, and what "approval-required" resolves to is the only role text (one operator confirmation for the session, the worker's `onPermission` routing for a worker).
|
|
17
35
|
|
|
18
36
|
Prompt extensions can add dynamic fragments for project rules, the operator profile, and Clio source-tree awareness. Pending skill requests and middleware reminders are visible text in the user message, not hidden prompt machinery.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Provider Adapter Cookbook
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
This cookbook guides developers through implementing custom model runtimes and inference server integrations within Clio Coder. It explains the runtime descriptor interfaces, probing protocols, model synthesis, and how to configure reasoning and thinking behaviors.
|
|
7
7
|
|
|
@@ -39,8 +39,14 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
39
39
|
7. `npm run ci` (runs 1 through 6)
|
|
40
40
|
8. `npm run ci:release` (7 plus `scripts/check-release.mjs`: dist shebang
|
|
41
41
|
integrity, version coherence between `package.json` and the top
|
|
42
|
-
`CHANGELOG.md` heading, the
|
|
43
|
-
|
|
42
|
+
`CHANGELOG.md` heading, the deterministic 26-scenario behavioral machinery
|
|
43
|
+
corpus against its checked baseline, the forbidden-file list, the required
|
|
44
|
+
runtime resources, and the tarball and unpacked size budgets). A baseline
|
|
45
|
+
mismatch prints reviewable evidence and names prompt- or recipe-affected
|
|
46
|
+
corpus results. For an intentional change, inspect that diff, run
|
|
47
|
+
`node benchmarks/eval/check-behavioral-release.mjs --update` (with `TMPDIR` on a disk-backed path if `/tmp` is a small tmpfs),
|
|
48
|
+
review `benchmarks/eval/behavioral-machinery-baseline.json`, and commit it
|
|
49
|
+
with the change.
|
|
44
50
|
9. Optional: step 8 again under Node 24. Hosted CI gates on Node 22 alone,
|
|
45
51
|
the `engines` floor; the weekly `flake-hunt` workflow carries Node 24.
|
|
46
52
|
Repeat locally only when the cut touches runtime-sensitive code.
|
|
@@ -58,7 +64,17 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
58
64
|
12. Install that tarball into a clean temporary prefix with an empty
|
|
59
65
|
`CLIO_CODER_HOME` and verify `--version`, `--help`, an empty-state non-TTY
|
|
60
66
|
launch, `doctor`, and `uninstall --dry-run` without developer-local state.
|
|
61
|
-
13.
|
|
67
|
+
13. Before interactive release testing, run the model-required public
|
|
68
|
+
behavioral corpus manually against the release target and built CLI:
|
|
69
|
+
`node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model.yaml --target mini --clio-coder-entry dist/cli/index.js`
|
|
70
|
+
and
|
|
71
|
+
`node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model-negative-control.yaml --target mini --clio-coder-entry dist/cli/index.js`.
|
|
72
|
+
Retain both Artifact v4 files as release evidence. The positive corpus must
|
|
73
|
+
report its scenario and role rows without an undeclared envelope mismatch;
|
|
74
|
+
the negative control must still record violated exploration and safety
|
|
75
|
+
labels. These model-dependent runs are manual and are never required by
|
|
76
|
+
ordinary deterministic CI. Continue with interactive release testing,
|
|
77
|
+
which this cut added because the release is
|
|
62
78
|
almost entirely interactive surface: a tester agent drives the step-12
|
|
63
79
|
install through real TUI sessions in a throwaway repository, one session
|
|
64
80
|
per shipped feature, against local targets for the main session and a
|
package/docs/safety-model.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio Coder Safety Model
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Clio Coder's safety posture is code-enforced, not prompt-only. As the orchestrator coding agent in the [IOWarp](https://iowarp.ai) ecosystem developed by the [Gnosis Research Center](https://grc.iit.edu) at Illinois Tech under NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318), Clio gates execution by target capabilities, the tool registry, the safety policy engine, project policies, protected-artifact checks, and audit receipts.
|
|
7
7
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio Coder Scientific Validation Contracts
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Scientific software development cannot treat simple file presence as proof of correctness. A simulation script that crashes on rank 48, or writes out NetCDF arrays filled with `NaN`s, may still successfully write a file to the disk.
|
|
7
7
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Skills Marketplace
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
The Skills Hub (`/skill`) shows project skills, user skills, and the marketplace. Every marketplace row comes from the same local lookup that `clio-coder skills install <name>` and `/skill <name>` resolve through, so the hub lists nothing it cannot install.
|
|
7
7
|
|
package/docs/tool-usage.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Tool Usage Reference
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
|
|
7
7
|
|
package/docs/trace-store.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Trace store contract
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Clio's trace database is a rebuildable, queryable mirror. Receipts, session
|
|
7
7
|
ledgers, gate artifacts, and evidence remain the source of truth. Removing
|
package/docs/troubleshooting.md
CHANGED
|
@@ -26,6 +26,93 @@ This guide provides concrete, actionable remediation procedures for operational
|
|
|
26
26
|
|
|
27
27
|
---
|
|
28
28
|
|
|
29
|
+
## Reading a cold cache
|
|
30
|
+
|
|
31
|
+
On a local server prefill is most of what a turn costs, so a cold prefix cache is the difference between a first token in under a second and one in fifty. This is how to find out why a turn went cold, starting from what the TUI shows.
|
|
32
|
+
|
|
33
|
+
**1. Read the `/context` cache lines.** Two lines answer different questions. The prompt-cache line says what the provider reported and whether the compiled prompt shell was reused. The prefill line says what the server itself did:
|
|
34
|
+
|
|
35
|
+
```text
|
|
36
|
+
prefill: 34,951 uncached · 0 cached · 48,617 ms
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Those are the server's own numbers, not Clio's estimate. `server does not report cache reads` in place of the cached figure means the backend gave no `cache_n` at all, which is LM Studio 2.29.0's OpenAI-compatible port today; on that target the verdict comes from the provider's `cached_tokens` instead and the prefill line reports only total prompt work and milliseconds.
|
|
40
|
+
|
|
41
|
+
**2. Look for the expected-cold line.** When the last settled run came back `cold` and Clio had recorded a cause, `/context` names it rather than warning:
|
|
42
|
+
|
|
43
|
+
```text
|
|
44
|
+
last cold turn: working-set eviction (expected)
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
The eight causes and what stamps each one are in [context-engine.md](context-engine.md#cache-divergence-honesty). `background_memory` renders in prose as `last cold turn: background memory step (expected)`.
|
|
48
|
+
|
|
49
|
+
**3. Confirm it in the ledger.** The reasons are durable, so a finished session answers the same question without the TUI. Open `current.jsonl` under the session directory `clio-coder paths` reports and read the run's first assistant entry:
|
|
50
|
+
|
|
51
|
+
```json
|
|
52
|
+
{
|
|
53
|
+
"timing": { "ttftMs": 53194, "apiMs": 56770 },
|
|
54
|
+
"promptCache": {
|
|
55
|
+
"input": 34951, "cacheRead": 0, "cacheWrite": 0,
|
|
56
|
+
"backendVerdict": "cold",
|
|
57
|
+
"expectedColdReasons": ["dispatch", "residency"],
|
|
58
|
+
"backend": {
|
|
59
|
+
"promptTokens": 34951, "cachedTokens": 0, "predictedTokens": 24,
|
|
60
|
+
"promptMs": 48617, "predictedMs": 373, "source": "llamacpp-timings"
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
`expectedColdReasons` is stamped once per run, on its first persisted call, so a turn with several model calls carries it on the first one only. `clio-coder doctor` folds the latest session for you and prints the verdict counts plus the most frequent reason, and `clio-coder usage report` gives per-session uncached prefill and verdict counts across the window.
|
|
67
|
+
|
|
68
|
+
**4. When there is no reason, the warning is the finding.** A cold backend with a reused prompt shell and no recorded reason is a real disagreement: Clio kept the bytes stable and the server re-prefilled anyway. `/context` leaves the warning in place for exactly that case. Four causes are worth checking in order, and none of them is a Clio bug:
|
|
69
|
+
|
|
70
|
+
- **The server slept.** A llama.cpp router started with `--sleep-idle-seconds N` drops the slot's prefix cache when it sleeps, and `--cache-ram` does not reliably restore a large state. A gap longer than that setting between two turns explains a cold turn completely. Raise the flag, or accept that a session left idle pays for its first turn back.
|
|
71
|
+
- **Something else used the endpoint.** A worker, a second Clio session, or another client on the same server evicts the slot. Clio stamps `dispatch`, `residency`, and `background_memory` only for work it can attribute to itself on that endpoint; a foreign process leaves no stamp. `clio-coder targets --probe` reports the endpoint's slot count, and `/fleet` settings show active slots per endpoint.
|
|
72
|
+
- **The model was swapped.** A router serving one model at a time reloads on a residency change, and everything the previous model had cached is gone. This normally does stamp `residency`, but only when the mutation went through Clio.
|
|
73
|
+
- **The prompt moved for a reason Clio did not classify.** Compare the run's `promptHash` and `toolSignature` in `context-snapshots.jsonl` against the previous run's. Equal hashes with a cold backend point at the server; different hashes with no `prompt_recompiled` or `tool_surface_change` stamp is worth an issue.
|
|
74
|
+
|
|
75
|
+
One case is expected on hybrid architectures and looks like a bug. Qwen3.8 keeps recurrent state that llama.cpp cannot roll back to an arbitrary token, so a change anywhere inside a cached prefix re-prefills from the last context checkpoint rather than from the changed byte. A small edit to old history can therefore cost thousands of tokens of prefill with the prompt hash otherwise stable. The server's checkpoint count and its `--checkpoint-min-step` are the levers; see the `qwen3.8-27b` family's `serving` and `measuredUnder` notes in `src/domains/providers/models/local-models/clio-local-coding-targets.yaml` for the measured figures and the exact argv they were taken under.
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## A TUI that stops answering the keyboard
|
|
80
|
+
|
|
81
|
+
When an interactive session stops responding to typing, the question worth
|
|
82
|
+
answering before anything else is which half of the input pipeline stopped: the
|
|
83
|
+
stdin reader that hands bytes to the application, or the renderer that turns
|
|
84
|
+
them into a frame on stdout. Clio keeps that evidence without being asked. Every
|
|
85
|
+
interactive process holds a bounded in-memory ring of the last 256 input-ingress
|
|
86
|
+
records and the last 256 committed frames, and writes it out when the process
|
|
87
|
+
receives `SIGTERM`, which is the signal a `kill` of the stuck pane sends.
|
|
88
|
+
|
|
89
|
+
The dump lands in the state directory `clio-coder paths` reports:
|
|
90
|
+
|
|
91
|
+
```text
|
|
92
|
+
<stateDir>/input-wedge/<ISO timestamp>-<pid>.json
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The five newest dumps are kept and older ones are removed as new ones land.
|
|
96
|
+
Read `classification` first:
|
|
97
|
+
|
|
98
|
+
| `classification` | What it means |
|
|
99
|
+
| :--- | :--- |
|
|
100
|
+
| `input-not-committed` | Bytes reached the application and no frame carrying them ever reached stdout. The renderer is the stuck half. |
|
|
101
|
+
| `no-input-recorded` | Nothing was delivered at all. If the operator was typing, the stdin reader is the stuck half. |
|
|
102
|
+
| `input-committed` | Both halves were moving. Whatever the session was doing, it was not this pipeline. |
|
|
103
|
+
|
|
104
|
+
`msSinceLastInputIngress` and `msSinceLastCommittedFrame` say how long each half
|
|
105
|
+
had been quiet when the signal arrived, and the `inputIngress` and `frames`
|
|
106
|
+
arrays carry the records themselves. Frames are kept only when they reached
|
|
107
|
+
stdout, so an empty `frames` array is itself a finding.
|
|
108
|
+
|
|
109
|
+
For a full session trace rather than the tail, set `CLIO_CODER_RENDER_TRACE` to
|
|
110
|
+
a file path before starting the session. That writes every record, including
|
|
111
|
+
provider deltas and terminal writes, as JSONL. The ring is the always-on subset
|
|
112
|
+
of the same records, for the case where nobody armed the trace first.
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
29
116
|
## Diagnostic Commands
|
|
30
117
|
|
|
31
118
|
When encountering unexpected system behavior:
|
package/docs/tui-design.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio TUI Design System
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive color/glyph token laboratory and terminal transcript preview renderer is located at [docs/html/tui_design_blueprint.html](html/tui_design_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
This document is the reference specification for the Clio Coder TUI visual layout, styling, and behavior. It describes color semantics, the glyph vocabulary, structural recipes, and state choreography for all surfaces under [src/interactive/](../src/interactive/).
|
|
7
7
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Worker Dispatch Mechanics
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive NDJSON protocol timeline stream and heartbeat watchdog simulator is located at [docs/html/worker_dispatch_blueprint.html](html/worker_dispatch_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
This document describes the design and lifecycle of Clio Coder dispatched workers, focusing on the spawning sequence, execution isolation, the standard input/output NDJSON communication loop, and permission escalation routing.
|
|
7
7
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@iowarp/clio-coder",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.9",
|
|
4
4
|
"description": "Coding agent for HPC and scientific-software developers, part of IOWarp's CLIO ecosystem of agentic science.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -92,6 +92,7 @@
|
|
|
92
92
|
"//": "below here: a real model target, chosen with --target <id>; costs money and/or GPU time, never run in CI",
|
|
93
93
|
"live:smoke": "node --import tsx benchmarks/internal/live-smoke.ts",
|
|
94
94
|
"live:fleet-dispatch": "node --import tsx benchmarks/internal/live-fleet-dispatch.ts",
|
|
95
|
+
"live:release-residue": "node --import tsx tests/live/release-residue.ts",
|
|
95
96
|
"live:tui": "node --import tsx benchmarks/internal/pty-drive.ts",
|
|
96
97
|
"live:home": "node --import tsx benchmarks/internal/live-home.ts"
|
|
97
98
|
},
|
package/src/cli/agents.ts
CHANGED
|
@@ -9,7 +9,7 @@ import { printError } from "./shared.js";
|
|
|
9
9
|
|
|
10
10
|
const HELP = `clio-coder agents [--json] [--all]
|
|
11
11
|
|
|
12
|
-
List user-facing agent specs from built-in, user, and project recipes.
|
|
12
|
+
List user-facing agent specs from built-in, extension, user, and project recipes.
|
|
13
13
|
|
|
14
14
|
Flags:
|
|
15
15
|
--json emit specs as JSON instead of the formatted table
|