@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,55 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to Clio Coder are documented in this file. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and versions follow Semantic Versioning; pre-1.0 minor releases may include incompatible changes.
|
|
4
4
|
|
|
5
|
+
## 0.3.9 - 2026-08-30
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
- Behavioral results now carry an additive, strictly parsed execution envelope that binds prompt fragment ids, versions, content hashes and composition hash to recipe, route, tool surface, autonomy, policy hashes, project-context provenance, and corpus version (#164). Comparisons mark scenario/role rows incomparable when any undeclared envelope field drifts, summarize mean and variance changes independently per scenario and role, and name every prompt- or recipe-affected corpus result. The release check runs all 26 public machinery-only scenarios against a reviewable checked baseline in ordinary CI; intentional updates require an explicit baseline regeneration and diff review, while the live model corpus and its negative control remain separate manual release evidence.
|
|
9
|
+
- The rebuildable SQLite trace mirror is now bounded by a terminal-run retention policy and can be reclaimed explicitly with `clio-coder trace prune` (#226). The default keeps 30 days and at most 128 MiB, configurable with `CLIO_CODER_TRACE_RETENTION_DAYS` and `CLIO_CODER_TRACE_MAX_BYTES`; age and size pruning delete every dependent trace row as one unit while excluding all queued and running runs. The command reports runs, rows, physical bytes reclaimed, protected live runs, and whether `VACUUM` ran. A vacuum rewrites the database once at least 20 percent of its pages are reclaimable or the byte bound requires it, then truncates the WAL, so deleting history returns disk space instead of only filling SQLite's freelist. `clio-coder doctor` now reports the recursive state-directory size and its largest top-level contributor, making a growing `trace.sqlite` visible before it reaches a home-directory quota.
|
|
10
|
+
- An interactive session now keeps an always-on record of its input pipeline and writes it out on `SIGTERM` (#224). A pane that stopped answering the keyboard previously left nothing behind unless `CLIO_CODER_RENDER_TRACE` had been armed before the session started, which nobody does before a bug they have not seen yet. The render tracer runs in every interactive process now, keeping the last 256 `input_ingress` records and the last 256 committed frames in a bounded in-memory ring and writing the JSONL file only when the environment variable names a path. The `SIGTERM` that recovers a stuck pane dumps that ring to `<stateDir>/input-wedge/<timestamp>-<pid>.json`, classified as `input-not-committed` when bytes reached the application and no frame carrying them reached stdout, `no-input-recorded` when the reader delivered nothing, and `input-committed` when both halves moved; the five newest dumps are kept. A PTY smoke test covers `/share` of a worker run and of a council member run against the OpenAI-compatible fixture, asserting a new `input_ingress` record and a committed frame whose `inputHighWater` covers it, in about 5 seconds.
|
|
11
|
+
- Behavioral eval comparison now reports correctness, safety, label violations, tool-call efficiency, unnecessary exploration, delegation quality, unsupported claims, tokens, latency, cost, and repeat variability as separate sourced metric families (#161). Each behavioral result carries an additive role and target/model-bound projection whose unmeasured observations stay null; distributions include coverage, min/max, p90, population variance, and standard deviation. `eval compare` classifies every shared row as improved, regressed, unchanged, or incomparable and supports equivalent text, JSON, Markdown, and JUnit-compatible output. Correctness and safety regressions, including a lost measurement that existed in the baseline, fail independently of pass-rate or cost gains, while threshold files keep release-blocking `fail` assertions separate from non-blocking `informational` budgets.
|
|
12
|
+
- A versioned public behavioral corpus now exercises the shipped agent system without private inputs (#160). Corpus `public-built-in-behavior` 1.0.0 pairs positive and adversarial machinery cases for all 13 built-in worker recipes; every case loads the production recipe catalog, follows real dispatch admission through a scripted worker, and verifies the sealed receipt and result-contract outcome instead of grepping frontmatter. Four isolated `mini` scenarios cover tool choice, exploration, delegation, safety comprehension, claim grounding, denied-tool recovery, completion behavior, and task correctness using per-tool calls and blocks, distinct/allowlisted read counts, decoy hits, and grader-emitted unsupported-claim and completion facts. A live decoy negative control must produce violated exploration and safety labels, and corpus contracts prevent private endpoints, credential values, and mutable external inputs from entering the suites.
|
|
13
|
+
- Behavioral eval scenarios have a versioned, fail-closed contract (#156). Suite v2 tasks can declare bounded expected and forbidden rules across tool choice, exploration, delegation, safety comprehension, claim grounding, denied-tool recovery, completion behavior, and task correctness; deterministic judge inputs are canonicalized from observable transcript, tool, receipt, and grader facts. Artifact v4 stores the result as an additive `clio.eval.behavior.v1` sibling that references the unchanged `clio.eval.verdict.v1` identity, preserving existing readers and the tracked-metrics baseline. `unknown`, `unmeasured`, `behavioral_failure`, and `infrastructure_failure` remain distinct, and malformed, partial, contradictory, or cross-linked verdicts cannot parse as passes.
|
|
14
|
+
- Clio pre-warms the prompt prefix on a local-native target, so a first turn no longer pays for prefill the operator was never going to avoid (#253). After the session prompt compiles at session start, after a resume rebuilds the message array, and after a compaction settles, Clio sends the exact request the next turn would send minus the operator's text: the same system prompt, the same tool schemas, the same replayed messages, the same thinking level, and the same `cache_prompt`, with one single-character user message so the chat template renders the prefix up to the user turn, and `max_tokens: 1`. The payload is built through the same engine dispatcher a turn runs on rather than hand-assembled, because a single differing byte ahead of the user turn re-prefills everything after it. Measured on a llama.cpp router (build `b226-2115b73d8`, Qwen3.8-27B), a resumed 34,951-token session's first turn re-prefilled cold at `promptMs 48617` and 53.2 s to first token; with the prefix already resident the same turn reads it from cache. The round is refused rather than queued whenever it would compete with real work: it never runs off the `local-native` tier whatever `prewarm.enabled` says, never while a turn is in flight or any dispatch is outstanding, and never on a worker or in headless `run`. The round claims one slot on its endpoint while its request is out and releases it in a `finally`, so endpoint capacity counts a pre-warm the way it counts the orchestrator's streaming turn. Pressing Enter lets go of an in-flight pre-warm at the keystroke. Whether it also aborts the request is gated on the backend, because the measured one ignores a cancellation: aborting 1.5 s into a 47,620-token prefill did not stop the server, which finished prefilling, so the prefix survived and the next request read 47,596 of 47,620 tokens from cache at `prompt_ms 927` while waiting 89.5 s of wall clock for the abandoned request to leave the single slot. Letting the pre-warm complete cost the same wall clock and kept the usage record, so a submit detaches the round rather than aborting it, withholds its `/context` line, and still records the prefill the server performed. Each round appends one `prewarm` ledger entry with its `timing` and `promptCache`, contributes zero tokens to the context estimate, is never a model message, and is never an expected-cold reason. `/context` reports `prewarmed: N tokens in X ms` until the next settled run, and `/cost` and `clio-coder usage report` carry pre-warm calls as their own row beside side questions and handoffs. `prewarm.enabled` defaults to `true` and is classified next-turn.
|
|
15
|
+
- Eval artifacts carry a fail-closed `clio.eval.verdict.v1` envelope with ledger and receipt sourced performance metrics, per-scenario pass and distribution aggregates, and serving-configuration provenance; `eval compare` can filter tracked metrics, refuses configuration drift unless explicitly allowed, and `eval run --trials N` isolates every trial in a fresh workspace (#252).
|
|
16
|
+
- Every model call on a llama.cpp or LM Studio target persists the server's own prefill facts (`prompt_n`, `cache_n`, `predicted_n`, `prompt_ms`, `predicted_ms`) as a `backend` object on the assistant entry's `promptCache`, and the cache verdict is derived from `cache_n` when pi-ai reports no cache reads (#247). `residency`, `thinking_change`, `tool_surface_change`, and `prompt_recompiled` join the expected-cold reasons, so the `/context` "shell reused, backend cold" warning no longer fires for a disturbance Clio caused. `/context` shows `prefill: N uncached · M cached · X ms` for the last call, `/cost` and `clio-coder usage report` carry per-session uncached prefill totals and verdict counts, and `clio-coder doctor` summarizes the last session's verdicts.
|
|
17
|
+
- The `/handoff` extraction round binds its JSON schema on the wire and gets one bounded repair attempt (#223). The round embedded `HANDOFF_RESPONSE_SCHEMA` in the system prompt and bound nothing, and a refused parse ended the command, so the 0.3.7 release test recorded `/handoff` as BLOCKED on a local target after two attempts that both ended "the extraction round returned no JSON object"; the dropped-path listing, the `e` editor, and accept-mints-a-session were all unreachable and the operator paid for the round either way. The out-of-turn seam now looks up the runtime's own spelling of a JSON-schema response constraint and sends it: `response_format: { type: "json_object", schema }` for llamacpp, the standard `{ type: "json_schema", json_schema: { name, strict, schema } }` for lmstudio, and nothing at all for a runtime with no known dialect, which still gets the prose instruction. The table is keyed by runtime id rather than by capability flag because a generic OpenAI-compatible gateway answers HTTP 200 to a spelling it does not implement and returns unconstrained JSON, which would turn a known non-enforcement into a silent one. Native enforcement is an optimization here, never a precondition: a server that refuses the constrained request with the 400 the worker seam already recognizes gets one unconstrained retry, and anything else rejects. A parse the extractor refuses now gets exactly one repair round that quotes the parser's complaint and the first answer verbatim, billed through the same out-of-turn usage store, and a terminal refusal names what was asked for, what each round said, and what round 2 returned. Verified on `mini` (llama.cpp, `muse-30b-dense`) against a ten-turn session: the constrained request carried all five schema properties and the extraction was accepted, and a deliberately truncated first round refused with the ticket's own message and the repair round was accepted.
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
- The pre-warm scheduler's zero-delay timers are ref'd, so awaiting `whenPrewarmSettled` can no longer deadlock a process whose event loop holds nothing else. Node 22, the engines floor, drains the loop past a due unref'd timer, which cancelled entire hosted-CI test lanes with `Promise resolution is still pending but the event loop has already resolved`; Node 24 fires the due timer first, which is why no development machine ever reproduced it. A due zero-delay timer holds the loop for one tick at most, and production sessions always hold live handles, so no operator-visible behavior changes.
|
|
21
|
+
- Fleet steps now transfer their held whole-plan reservation into the worker lease even when the request carries fleet lineage, so a one-step fleet can run on a one-slot endpoint without counting its own reservation twice. Endpoint-capacity refusals now identify active leases, held reservations, and foreground streams instead of always blaming the orchestrator's turn, and reservation members record when admission consumes them.
|
|
22
|
+
- A consumed prompt now reaches one committed pending frame before turn admission can make it durable (#251). The shipped state machinery was live during a long forced auto-compaction, but the real editor path only queued a render before starting the capability probe, prompt compile, and overflow preflight, so release-test windows lasting 17 to 260 ms opened and closed between renderer ticks and showed `MESSAGE`, no user row, and the previous turn's receipt throughout. The chat loop now opens its reference-counted preparation window and awaits an internal presentation barrier that flushes `· preparing`, the `PREPARING` rail, and the preparing footer before admission work starts; the existing `compactionSummary` then user-turn ordering is unchanged. Manual `/context compact` remains a separate path and its live island now reports a single `compact` phase at 0% instead of publishing `done` and rendering the five `/context init` stages at 100% from its first frame; only its terminal event renders `done` and 100%.
|
|
23
|
+
- The v0.3.7 release-test residue now has one manually gated live driver that uses the built CLI, a real PTY, native workers, the builtin recipes, and a scrubbed isolated home against an operator-selected target (#222). On `mini` with `muse-30b-dense`, it proved the in-flight `/oracle` refusal with zero new receipts, scrolled a 51-line `/handoff` review until the unread path appeared under the dropped-ledger heading, proved saved and cancelled `$EDITOR` digests, rendered the compete verification refusal, and checked every discovered development command across default help, full help, and its bare entry point. The same evidence records why four dispatch paths could not produce receipts on the one-slot endpoint: proposal and gate-loop plans self-contended with their reservation, a two-member council could not admit round one, and the current admission contract intentionally accepts review verification even though the release-test row says it refuses. The proposal record also preserves its out-of-plan `requestOrigin: user` and missing plan binding rather than changing that contract here.
|
|
24
|
+
- Worker heartbeat age now comes from a monotonic stamp while the durable ledger keeps a separately derived wall-clock instant (#198). A 60-second forward wall-clock step can no longer reap a live native or ACP worker, and a backward step cannot hide a worker whose monotonic heartbeat window and grace have elapsed; restart recovery explicitly uses the persisted instant only as a human-readable last-seen bound after the host-scoped worker process is known to be gone.
|
|
25
|
+
- `targets --probe` now keeps a degraded target's health reason ahead of gateway, context-window, and residency notes, and repeats the complete reason on a wrapped detail line whenever the table row cannot hold it (#234). At both 80 and 120 columns, plain and gateway targets now say that the configured default model is not advertised by the target instead of dropping or truncating the operative clause; healthy rows and the full `health.lastError` in JSON are unchanged.
|
|
26
|
+
- `config inspect` now reports agent and fleet resource roots instead of omitting both extension-era resource kinds (#243). Each category includes the shipped builtin root plus extension, user, and project roots in their real loader order, with the numeric collision precedence retained in the JSON detail. The formerly dangling `agents` category is replaced by `agent-root` and paired with `fleet-root`, and `agents --help` now names extensions as a recipe source.
|
|
27
|
+
- Extension manifests now enforce their optional `compatibility.clio` SemVer range instead of parsing and ignoring it (#242). Malformed ranges fail manifest parsing, installation refuses a package whose range excludes the running Clio version before writing it, and installed packages are checked again at load so upgrades cannot activate incompatible resources. The diagnostic names the extension, its declared range, and the running version; incompatible packages remain visible in `extensions list` while a compatible package at another scope can still become effective. Manifests that omit the constraint retain their previous behavior.
|
|
28
|
+
- A successful write in a never-indexed project no longer strands an empty `.clio-coder/` directory (#248). The incremental codewiki refresh is contractually a no-op when the project was never indexed, but `coordinateCodewikiWrite` acquired the state-file lease before rechecking `requireExisting`, and `withStateFileLock` creates the lock's parent with `mkdirSync(..., { recursive: true })`. So every successful file-mutating tool created `.clio-coder/`, wrote and deleted `codewiki.json.lock` inside it, and left the now-empty directory behind; `inotifywait` recorded exactly that sequence on 0.3.8. The check now runs inside the workspace queue and before the lease, so nothing is materialized for a project with no codewiki. The recheck under the lock is unchanged and is still what decides the race, and an indexed project keeps serializing its incremental writes through the same queue and lease.
|
|
29
|
+
- A downgraded fleet write record now says exactly why it opened and which tool call caused it (#236). A successful opaque `bash`, `verify`, `dispatch`, `steer`, dynamic, or MCP call previously collapsed into `attributionComplete: false`, leaving the Alt+W card silent while the run was actionable and the durable verdict saying only that no complete record existed after rollback. The run-effects recorder now retains the causal tool and call id, emits a live warning when the first such call succeeds, and carries every cause through `observedRunWriteAttribution` into the write-boundary verdict's additive `attributionDowngrades` list. The live card and after-the-fact message name the tool and explain that its arguments cannot enumerate every path it may write. Incomplete telemetry, untracked code steps, and unavailable records receive their own reason codes, while a closed record keeps `attributionComplete: true`, an empty downgrade list, and the existing protection for unattributed concurrent edits.
|
|
30
|
+
- Canonical trust verdicts now stay receipt-derived across dispatch, `monitor collect`, evidence bundles and inspect output, `findings.md`, the Alt+W board, receipt verification, and eval metrics (#237). Evidence alone composed a finish-contract audit row into `completionEvidence`, so one read-only run appeared as `completion not applicable` there while every receipt surface said `completion not recorded`; evidence no longer lets that separate input override the receipt axis. Every `findings.md` now records each linked run's tier, fixed-order summary, and six axes before its diagnostic findings. Alt+W writes the tier as text and wraps every summary clause, and the receipt view wraps its versioned tier and summary onto additional header rows instead of truncating the verdict after its first clause.
|
|
31
|
+
- A prompt template that cannot load is now recognized as the command it is and refuses with its reason, instead of being dropped so the operator is told the command does not exist (#245). The safety half is unchanged: a missing or escaping `${extensionRoot}` reference still never expands and never reads outside the declaring package. What changed is that `loadPromptFile` no longer returns null for it. The template loads with an empty body and an `unavailable` reason attached, so the namespaced command is recognized and invoking it says, for example, `prompt template /wtfp:plan-section cannot run: prompt template has an unresolved or escaping extension reference: ${extensionRoot}/core/templates/missing.md (<file>)`. The reason is the same sentence the `/prompts` overlay already renders as a diagnostic, plus the file, so the two surfaces cannot drift. Every load-time reason gets the same treatment, not only the reference case: the name is now derived before the file is read, so an unreadable template also refuses with its own error rather than vanishing. A path that escaped its discovery root is still dropped, because there is no name there to offer a command under. The `/prompts` overlay marks such a template `unavailable` and shows the reason in its detail pane.
|
|
32
|
+
- Every fleet step can now use the configured transient-failure retry policy instead of only the first step (#231). Fleet steps share one root assignment, whose once-only settlement correctly becomes terminal after wave one; later retry decisions mistakenly treated that root status as the current step's liveness and silently suppressed recovery. Reserved work now reads the current plan member's held, consumed, or released state at both scheduling and timer execution, while unreserved dispatch retains the root-assignment check and the original settlement guard remains unchanged.
|
|
33
|
+
- A queued or echoed model-facing turn carries the payload byte for byte (#244, follow-up to #240). The expander was already exact, but `appendQueuedUserTurn` trimmed the engine's user text before persisting it, which broke the byte-for-byte match against the echo the loop had already written and so wrote a second, shortened row that the assistant was then parented to. One `/raw bar` submit produced a correct 5-byte `" bar"` turn and a 3-byte `"bar"` turn, and the model's parent turn was the 3-byte one, contradicting the contract in `docs/prompt-envelope-and-tools.md` that every byte after the delimiter belongs to the payload. The echo path now persists the submitted bytes and reads a trimmed copy only for the emptiness test, so the echo is recognized as the turn already in the ledger instead of being written again. The queue path is fixed the same way: a steer or a follow-up hands the engine and the queue mirror the submitted text rather than a trimmed one, because a steer is a model-facing turn too. Leading whitespace, trailing whitespace, interior runs, and tabs all survive into the turn the assistant is parented to; a message that is only whitespace is still refused.
|
|
34
|
+
- A consumed prompt no longer looks idle while Clio prepares the turn (#251). The editor is cleared and the prompt painted into the transcript before admission, and the capability probe, pre-submit auto-compaction, prompt compile, and overflow preflight all run after that and before the turn owns the stream. Nothing named that window: the composer went back to `MESSAGE` with `Ask Clio…` and the footer still reported the previous turn as done, so a run that spent 77.4 seconds in `trigger: auto` compaction was indistinguishable from a dropped Enter, and the operator pressed Enter twice more trying to submit. The chat loop now carries a turn-preparation phase from the moment `submit` is called until admission succeeds or refuses, refined to `compacting` around each pre-submit compaction. The composer rail reads `PREPARING` or `COMPACTING` with a placeholder that says which, the status machine enters the existing `preparing` phase on the same signal so the footer's verb and watchdog tick replace the previous turn's receipt, and because `preparing` is an active phase the compaction bus overlay now lands during the window instead of being dropped by an idle status. The painted transcript row is marked pending while it is not in the ledger, becomes an ordinary row at the durable append, and is marked `not sent` when the submit is refused, so a blocked preflight never leaves a committed-looking phantom turn. The phase is reference-counted across the FIFO admission gate, so a retyped prompt arriving mid-window neither narrows the state the first one is showing nor closes it early. The ordering `compactionSummary` then one user turn is unchanged and asserted, and Enter on an empty editor is still refused before any of this.
|
|
35
|
+
- `ask_user` keeps the free-text answer the operator typed instead of only the option label they chose it under (#228). A round that offered "Exact number - I'll type it" recorded that sentence and nothing else, because choosing an option committed the answer and only an option literally named Other opened a text field. The interview at `state/interviews/2026-08-25T10-14-05-319Z-...json` shows the cost: rounds 3 and 4 existed only to re-ask for the figures rounds 1 and 2 had thrown away, 35 minutes for four facts, with the operator typing the same numbers and dates three times. On any option list `t` now opens the text field for the option under the cursor without committing it, and submitting records the label and the text together as `<label>; <text>`. The answer travels as three separable facts rather than one joined string: `answer` is the one-line rendering, `options` is the labels chosen in list order, and `value` is the typed text exactly as submitted, so a label-only answer is told from a label-plus-value one by whether `value` is there rather than by parsing. All three reach the model in `latest_answers`, the interview transcript under `clioStateDir()/interviews/`, and the decision record, whose `value` carries both and whose `options` and `text` keep them separable. Choosing a plain option after typing clears the text with it, so a recorded answer never claims a figure the operator gave for a label they moved off.
|
|
36
|
+
- Dispatch plan approvals now bind every byte of every worker task with `task_bytes` and `task_sha256` while retaining the existing 255-character `task_preview` (#246). A real vote council sealed 714-byte member tasks into receipts whose shared approval artifact represented only a 258-byte preview, leaving the 456-byte ballot-directive tail outside the plan hash; changing any byte in that hidden tail now changes the approved hash without making the operator-facing plan unbounded. The other routing, identity, and path fields rendered through `safeField` now append a digest whenever sanitization or truncation changes their visible value, closing the same collision class without lengthening normal plan text.
|
|
37
|
+
- A parked `write` or `edit` can be read before it is authorized (#254). The approval card described the mutation only by size, as `content=<string 482 bytes>` or `edits=<array 3 items>`, so approving it meant approving bytes the operator had never seen; during the WTF-P NSF 25-531 UAT on 0.3.8 the only way to read them was to open the external `current.jsonl` ledger and compare payloads by hand. The card now carries a `Mutation:` line with the kind, the byte count, and a truncated SHA-256 over the exact call arguments the decision resumes, and `v` opens the complete proposed content for a write or the complete effective diff for an edit, computed by applying the edit list to the bytes on disk rather than by printing the edit array back. The mutation scrolls with the arrow and page keys in a 16-row window that names its position, and `v` closes it again; Enter, `s`, and Esc keep answering the call throughout, so the inspection never becomes a step between the operator and a denial. An edit whose file cannot be read or whose replacements do not apply says which, and still shows the requested replacements. The inspector re-derives the digest from the arguments it is about to render and refuses when it no longer matches, so a call that differs from the previewed one cannot inherit its preview. The mutation text is process-local by construction: it is never placed on the approval view, which is what reaches the transcript row, the parked notice, the desktop notification, and the worker escalation payload, all of which carry only path, size, and digest. Escape sequences and control bytes are neutralized and the neutralization is stated, tabs render as spaces, and content past 262,144 characters is cut with a line naming how much was withheld. A worker escalation has no preview because its arguments never leave the worker, and its card says exactly that instead of advertising a key it cannot honor. At 40 columns the inspect key is elided ahead of allow, stop, and deny, and the `/help` Autonomy & safety net topic documents it for the widths that cannot show it.
|
|
38
|
+
- Dispatch contract tests now load all 13 shipped builtin agent recipes through the production registry instead of substituting handcrafted `coder`, `researcher`, and `verifier` records (#232). The old researcher claimed the `base` audience and an `external-delegation` result while production seats a `shadow` researcher that must return a `research-report`, so default council tests could pass against an agent operators could never run. The shared stub now inherits each builtin's audience, result contract, and tool surface directly from its Markdown recipe; the one optional constrained-coder seam is explicit, council scripts return the real research envelope, and a catalog-wide drift contract compares every field that caused the divergence.
|
|
39
|
+
- A residency load that races the llama.cpp router's own wake no longer fails the turn. When Clio's `/v1/models` snapshot was stale, the router answered the duplicate `POST /models/load` with HTTP 400 `model is already running` or HTTP 500 while it was already loading the model, and the call was recorded as an errored assistant entry before any inference. A rejected load now re-reads `/v1/models`, treats a model that is loaded, loading, or sleeping (or a body that says it is already running) as a load in progress and waits for it, and only an absent, unloaded, or failed model preserves the rejection, whose message now carries the router's response body.
|
|
40
|
+
- The compaction summary round and a background memory step hold a slot on the endpoint they stream to for as long as the request is out, the same as a turn, a `/btw` round, and a pre-warm (#250, #229). A memory step registers only after the `endpoint_busy` admission check, so it counts on its own server for dispatch capacity without ever refusing itself.
|
|
41
|
+
- After an in-process `/resume`, the prompt manifest and the `promptRecompiled` entry follow the incoming session (#249 follow-up). The hash this process compiled was carried on the previous session's compiled prompt, so the resumed session's first entry named the abandoned session's hash and its manifest chain started from a hash that appeared nowhere in it; the compiled hash is now tracked per session and cleared on the switch. Because the backend's slot still holds the previous session's prefix, the first turn after the switch is stamped `prompt_recompiled` from the resume itself, so `/context` names the cold turn instead of warning about it. An unreachable `preferLoaded` branch in the session-prompt compile was removed; the loaded-over-probe ranking lives in runtime resolution and is proven there.
|
|
42
|
+
- A background memory step that times out or throws after its request left the process now still stamps `background_memory` on the next turn (#229). The `MemoryStepCompleted` announcement hung off the usage sink, which only runs when the call resolves, so a step cut off by the 30 s deadline, whose trajectory the server had nonetheless prefilled into its single slot, left the next turn cold with no reason and `/context` warning about a provider re-prefill Clio had caused. The announcement now wraps the client's `complete` in a `finally` (`announceMemoryStepEndpoint`), fires exactly once per request that left the process, and never for an `endpoint_busy` skip; the usage row and `/cost` accounting are unchanged.
|
|
43
|
+
- An eval result has one pass decision (#252). `verify.measure` is the scenario's code grader, and a nonzero grader exit now fails the result with `failureClass: grader_failed` while the verdict keeps `machinery: ok`, so `result.pass`, `verdict.outcome`, the per-scenario aggregates, the summary, and the exit status agree; the sprint's baseline artifact reported 26 of 30 in its summary and 23 in its aggregates for the same 30 runs, and 23 is the figure the graders produced. Every failed verdict carries a `reason`, and pre-fix artifacts are read with the reason normalized. `eval run --trials N` prepares each trial's workspace immediately before that trial and removes both the workspace and the state directory on the item's `finally` path, including when the copy itself throws; a run that died on `ENOSPC` had created all 30 workspaces up front and left every one behind. A `temp-copy` workspace inside a git checkout copies exactly the tracked and untracked-but-not-ignored set (`git ls-files --cached --others --exclude-standard`), so 1.4 GB of ignored benchmark datasets no longer ride along in every trial.
|
|
44
|
+
- A `/btw` side question and a `/handoff` extraction round now hold a slot on the endpoint they stream to for as long as the request is out, the same way the turn and the pre-warm do (#250, #229). Without it, a fleet dispatch on a one-slot server was admitted onto the endpoint the side question was occupying, and the background-memory tier read that endpoint as idle during exactly the window it was busiest.
|
|
45
|
+
- `residencyTargetKey` is total again for a configured base URL: a target written without a scheme (`mini:8080`) produced a null key through an overload typed `string`, and the residency reconcile threw `Cannot read properties of null` instead of running unlocked. The canonical form is used when the URL has one and the raw spelling otherwise, which matches what dispatch capacity does for the same target (#250 regression, found in review).
|
|
46
|
+
- The `/context` meter draws the autocompact reserve at the far end of the bar, after free space, and in the same frame token free space uses, so the held-back headroom no longer reads as a grey block wedged between the consumed categories and the outline it should match; the `▒` glyph still keeps it distinct without color, and the footer meter shares the order. `background_memory` renders in prose as `background memory step` on the `last cold turn:` line instead of falling through to the wire value (#229).
|
|
47
|
+
- Proactive memory's model tier now accounts for what it spends, is bounded by a deadline a turn boundary can wait for, and writes what it produces into the store `/memory` reads (#229). One operator's ledger held 274 steps over 14 days: 60 of them reached the background model, spending 137,205 tokens and 1,666.6 seconds of model time, with a single step holding a local server for 102.5 seconds to answer nothing, and none of it appeared in `/cost`, in `clio-coder usage report`, or anywhere else. Every model step is now billed the way a `/btw` side question is, under a `background-memory` label: `/cost` shows a `memory steps` row, `usage report` counts them in its window, and one durable row per step lands in the out-of-turn usage store carrying the provider usage, the call's duration, and the backend's prefill facts. `/memory` shows the lifetime steps, tokens, model time, and hit rate folded from `steps.jsonl`. `memory.intervention.timeoutMs` drops from 180,000 to 30,000, and a step that exceeds the deadline records `timeout` with reason `deadline` or `timed_out` rather than the `silent` a model that simply chose not to speak records; a route that refuses the connection still records `client_error`, so the three diagnoses stay distinguishable. A step whose endpoint is already serving the chat target's stream is skipped with reason `endpoint_busy` and the skip is recorded, so a single-slot local server is never made to queue the memory call behind the operator's own turn or evict the resident model to serve it; a step is started from the `turn_end` hook while the chat loop still holds its foreground registration, so a background role pointed at the chat target's own endpoint declines every boundary and records each skip rather than contending for the slot, and a step that did run on that endpoint stamps the next turn's expected-cold reason `background_memory` so the `/context` warning names its cause. The session task bank and `records.json` are connected: an injected reminder's cited entries are proposed into the durable store, unapproved and scoped to the session's repository with provenance naming the source entry, so `clio-coder memory list` and `/memory` show what the tier actually produced and approval stays a separate operator action. The default is unchanged and now recorded with its numbers in `docs/proactive-memory.md`: the free rules tier stays on, and the model tier stays opt-in behind `background.target`, which bought 6 injections from those 60 steps, a 10.0 percent hit rate at 22,868 tokens per injection.
|
|
48
|
+
- The compiled system prompt is ordered stable prefix first, so a moved context window or an approved memory record no longer re-prefills the sections behind it (#249). Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, and through 0.3.8 the runtime block sat sixth of eleven compiled sections with the memory block tenth. `Context window: N` moves whenever the backend reloads a model or a co-residency clamp lands, so one changed digit cost a fresh prefill of the tool contract, the fleet roster, the retrieval hints, memory, and the project context; on the operator's llama.cpp target a one-line change at token 500 of a 16,712-token prompt re-prefilled all 16,712 in 18.4 seconds. The order is now identity, operating contract, delegation, skills, safety, tool contract, fleet, retrieval hints, project context, memory, runtime, then the operator-editable tail fragments unchanged, under one rule: a section goes as late as its volatility, and anything that reads a clock, a probe, or a mutable store goes after everything that does not. No section's wording changed and no section was added or dropped. `Context window: N` is now read from the turn's own window resolution, where a recorded loaded window outranks a probe that reports what the server could serve rather than what the model is open at, so a resumed session states the figure its ledger measured (#227's carry-forward). Prompt-manifest records carry a layout `version` plus the window and the layer that answered it, and a resumed session's first compile names the prompt hash it replaced instead of reporting no previous prompt at all, so the one `promptRecompiled` entry the upgrade writes explains itself. One cosmetic defect went with it: the memory section carried its `# Memory` header twice, once from the memory renderer and once from the compiler.
|
|
49
|
+
- Dispatch capacity now binds independently per inference endpoint and counts the orchestrator's own active model stream (#250). Target URLs normalize to a shared scheme, host, port, and base-path key, so two target descriptors pointed at the same scheduler share leases and held reservation capacity. llama.cpp probes the selected worker's `total_slots`, with its reported `--parallel` argv as a fallback; LM Studio and Ollama use conservative local defaults, and a target may set `maxConcurrentRequests` explicitly. Global and node `budget.concurrency` semantics remain unchanged. Endpoint-aware execution plans size each wave to the available request slots, `targets --probe`, `/fleet` settings, and the Fleet Runs board expose those slots, and a saturated endpoint refuses the dispatch with its slot count plus a named collection or second-server remedy.
|
|
50
|
+
- Context accounting is reconciled against the provider's own token counts, and compaction fires on the reconciled figure (#227). A session on 2026-08-28 believed it was at 63 percent of a 131,072-token window while the backend answered `Context size has been exceeded`; the `0.9` threshold never tripped because the number it read was a chars/4 estimate. The reconciliation data was already there. `reconcileSnapshot` folded the provider's count into the `/context` overlay, the footer meter, and `context-snapshots.jsonl`, and `shouldCompact` never saw it. After each model call the attested prompt count, with cached prompt tokens folded back in, is now carried as an anchor over the live message list: the budgeted figure is that count plus a chars/4 estimate of everything appended since, and the estimate remains a floor because it prices material the attested call never saw. All three evaluation points read it: the pre-submit trigger, the post-tool continuation guard, and the preflight overflow check, so a turn that would overrun the window compacts instead of failing at the provider. A working-set projection subtracts the tokens the eviction planner priced out against that same projection and re-anchors on the projected message list, where it previously discarded the attestation entirely and fell back to pure chars/4 exactly when the accounting mattered most; a summary compaction rewrites the conversation the attestation described, so it drops the anchor and the next call re-establishes it. Every snapshot now records `estimatedTokens`, `reconciledTokens`, and `divergenceRatio`, so the divergence itself is observable in the ledger rather than inferable from an overflow.
|
|
51
|
+
- A resumed session no longer spends its first turn budgeting against a re-probed window (#227). The same session resumed reporting `contextWindowSource: "probe"` at 262,144 with a 26,214-token reserve, then corrected to `loaded` at 131,072 one turn later, a 126K swing in one turn and the entry point for the overflow above. A resume re-resolves its target before discovery has reported what the backend has open, and the probed figure is what the server could serve rather than what the model is open at. Resolution now accepts a `knownLoadedContextWindow`, and the turn context supplies the last `loaded` window the session's own snapshot ledger recorded for the same target and model. It applies only when live discovery reports no loaded window and only for that exact target and model, so a changed selection re-probes and a model reloaded at a different size corrects as soon as discovery names the live window.
|
|
52
|
+
- `/share` of a worker answer with no place to fold no longer kills the TUI (#257). The uncommitted user row's `· preparing` and `· not sent` tails from #251 were concatenated onto the last rendered line with no width budget, so a body that had already folded to the full content width came out 12 cells past the terminal, and pi-tui's `doRender` throws on an overlong line and takes the process down with it. A `research-report` answer is JSON with no space to break at, so an 80-column pane died on any shared body over 58 columns; the recorded crash was `Rendered line 16 exceeds terminal width (84 > 80)` on a 70-character body. The tail now rides on the last body line only when that line has room for it within the terminal width, and drops to its own hanging row when it does not, so it stays whole rather than breaking between the separator and the word. A contract test sweeps shared-note body lengths at 80 columns and terminal widths from 8 to 120, and the `/share` PTY regression gains an 80-column share of a spaceless body that asserts the process survives and the keyboard still reaches the editor.
|
|
53
|
+
|
|
5
54
|
## 0.3.8 - 2026-08-29
|
|
6
55
|
|
|
7
56
|
### Added
|
package/README.md
CHANGED
|
@@ -73,7 +73,11 @@ decision afterward.
|
|
|
73
73
|
compiled prompt and tool schemas byte-stable so a llama.cpp prefix cache
|
|
74
74
|
stays hot across turns and sessions, bounds every tool result so one `grep`
|
|
75
75
|
cannot blow the window, and records a per-call cache verdict in the ledger
|
|
76
|
-
so you can see when and why the cache went cold.
|
|
76
|
+
so you can see when and why the cache went cold. It also sends that prefix
|
|
77
|
+
ahead of your first keystroke on a session start, a resume, or a compaction,
|
|
78
|
+
and counts request slots per inference endpoint rather than per node, so a
|
|
79
|
+
fleet cannot admit four workers onto a one-slot server the orchestrator is
|
|
80
|
+
already streaming against.
|
|
77
81
|
- **Work goes to bounded workers, not one long context.** The orchestrator
|
|
78
82
|
dispatches focused agents with explicit tool profiles, call budgets, cost
|
|
79
83
|
ceilings, and typed result contracts. A worker that cannot produce a
|
|
@@ -279,7 +283,7 @@ dist-tag instead.
|
|
|
279
283
|
From source, pinned to this release:
|
|
280
284
|
|
|
281
285
|
```bash
|
|
282
|
-
git clone --branch v0.3.
|
|
286
|
+
git clone --branch v0.3.9 https://github.com/iowarp/clio-coder.git
|
|
283
287
|
cd clio-coder
|
|
284
288
|
npm run install:local
|
|
285
289
|
export PATH="$HOME/.local/bin:$PATH"
|
|
@@ -305,7 +309,7 @@ Full lifecycle details, including `reset` and the upgrade path, are in
|
|
|
305
309
|
|
|
306
310
|
## Status
|
|
307
311
|
|
|
308
|
-
The current release is **v0.3.
|
|
312
|
+
The current release is **v0.3.9**, installable from npm as
|
|
309
313
|
[`@iowarp/clio-coder`](https://www.npmjs.com/package/@iowarp/clio-coder) or
|
|
310
314
|
from source. Clio Coder is still experimental: we ship quickly, interfaces may
|
|
311
315
|
change between minor versions, and model-specific behavior varies by target, so
|
|
@@ -6,21 +6,21 @@ import {
|
|
|
6
6
|
} from "./chunk-2VTFPG5O.js";
|
|
7
7
|
import {
|
|
8
8
|
runClioCommand
|
|
9
|
-
} from "./chunk-
|
|
10
|
-
import "./chunk-
|
|
9
|
+
} from "./chunk-B5XRQOLB.js";
|
|
10
|
+
import "./chunk-FALJGAWU.js";
|
|
11
11
|
import "./chunk-HKIYEGME.js";
|
|
12
12
|
import "./chunk-IFBNV6H6.js";
|
|
13
|
-
import "./chunk-
|
|
13
|
+
import "./chunk-56KB5IJP.js";
|
|
14
14
|
import {
|
|
15
15
|
printError
|
|
16
16
|
} from "./chunk-XK56QHLX.js";
|
|
17
17
|
import "./chunk-5TSRNF4G.js";
|
|
18
|
-
import "./chunk-
|
|
18
|
+
import "./chunk-PNY46YEY.js";
|
|
19
19
|
import {
|
|
20
20
|
MAX_TIMER_DELAY_MS
|
|
21
21
|
} from "./chunk-FQ4SKYE4.js";
|
|
22
22
|
import "./chunk-IWHMRKLL.js";
|
|
23
|
-
import "./chunk-
|
|
23
|
+
import "./chunk-XDOQXGFO.js";
|
|
24
24
|
import "./chunk-LL4KHSZI.js";
|
|
25
25
|
import "./chunk-4ZG3XFUR.js";
|
|
26
26
|
import "./chunk-EQ63NRB7.js";
|
|
@@ -126,4 +126,4 @@ export {
|
|
|
126
126
|
resolveAcpCwd,
|
|
127
127
|
runAcpCommand
|
|
128
128
|
};
|
|
129
|
-
//# sourceMappingURL=acp-
|
|
129
|
+
//# sourceMappingURL=acp-7LOELQFP.js.map
|
|
@@ -1,44 +1,47 @@
|
|
|
1
1
|
import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
3
|
SafetyDomainModule
|
|
4
|
-
} from "./chunk-
|
|
5
|
-
import "./chunk-
|
|
4
|
+
} from "./chunk-WXCJ7VME.js";
|
|
5
|
+
import "./chunk-DG4M6ZUE.js";
|
|
6
6
|
import {
|
|
7
7
|
ensureClioState
|
|
8
|
-
} from "./chunk-
|
|
8
|
+
} from "./chunk-T3Z6VAAF.js";
|
|
9
9
|
import {
|
|
10
10
|
ConfigDomainModule,
|
|
11
11
|
loadDomains
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-465CC7FK.js";
|
|
13
13
|
import {
|
|
14
14
|
AgentsDomainModule
|
|
15
|
-
} from "./chunk-
|
|
15
|
+
} from "./chunk-YSEHGPCT.js";
|
|
16
|
+
import "./chunk-HCBCAYZU.js";
|
|
16
17
|
import "./chunk-DR52UMZW.js";
|
|
17
|
-
import "./chunk-
|
|
18
|
-
import "./chunk-
|
|
19
|
-
import "./chunk-
|
|
20
|
-
import "./chunk-
|
|
18
|
+
import "./chunk-AD7Y7STJ.js";
|
|
19
|
+
import "./chunk-AMKHQW3C.js";
|
|
20
|
+
import "./chunk-FALJGAWU.js";
|
|
21
|
+
import "./chunk-BVDVID7E.js";
|
|
21
22
|
import "./chunk-RVG5JXAL.js";
|
|
22
|
-
import "./chunk-
|
|
23
|
-
import "./chunk-
|
|
24
|
-
import "./chunk-
|
|
23
|
+
import "./chunk-3DPEIQKN.js";
|
|
24
|
+
import "./chunk-HPCTNZM2.js";
|
|
25
|
+
import "./chunk-RKKLTLYB.js";
|
|
26
|
+
import "./chunk-5QKCQQ3E.js";
|
|
27
|
+
import "./chunk-HAY4ZE2P.js";
|
|
25
28
|
import "./chunk-UOV2BYIW.js";
|
|
26
|
-
import "./chunk-
|
|
29
|
+
import "./chunk-7C6RYZGQ.js";
|
|
30
|
+
import "./chunk-N3PBVRTZ.js";
|
|
27
31
|
import "./chunk-A2NJGIB3.js";
|
|
28
|
-
import "./chunk-HWUFFB6L.js";
|
|
29
|
-
import "./chunk-TTHACPOM.js";
|
|
30
32
|
import {
|
|
31
33
|
isUserVisibleAgent
|
|
32
|
-
} from "./chunk-
|
|
34
|
+
} from "./chunk-S4COXYBG.js";
|
|
33
35
|
import "./chunk-MV3K5QF2.js";
|
|
36
|
+
import "./chunk-RAPCMZL4.js";
|
|
34
37
|
import "./chunk-UL3WSD3F.js";
|
|
35
38
|
import "./chunk-ECH6PKUQ.js";
|
|
36
39
|
import "./chunk-5B2AEOW5.js";
|
|
37
40
|
import "./chunk-CGKSTWHD.js";
|
|
38
41
|
import "./chunk-XPLRXC72.js";
|
|
39
42
|
import "./chunk-IFBNV6H6.js";
|
|
40
|
-
import "./chunk-
|
|
41
|
-
import "./chunk-
|
|
43
|
+
import "./chunk-QQ7EKM72.js";
|
|
44
|
+
import "./chunk-56KB5IJP.js";
|
|
42
45
|
import {
|
|
43
46
|
printError
|
|
44
47
|
} from "./chunk-XK56QHLX.js";
|
|
@@ -46,15 +49,14 @@ import "./chunk-5TSRNF4G.js";
|
|
|
46
49
|
import "./chunk-LU7P4LHA.js";
|
|
47
50
|
import "./chunk-IHXBNWMM.js";
|
|
48
51
|
import "./chunk-B5CSFE7B.js";
|
|
49
|
-
import "./chunk-
|
|
52
|
+
import "./chunk-PNY46YEY.js";
|
|
50
53
|
import "./chunk-FQ4SKYE4.js";
|
|
51
54
|
import "./chunk-ZGVHUX3M.js";
|
|
52
|
-
import "./chunk-
|
|
53
|
-
import "./chunk-
|
|
54
|
-
import "./chunk-R346GLFC.js";
|
|
55
|
+
import "./chunk-RKRLDWD3.js";
|
|
56
|
+
import "./chunk-KV2AOLDF.js";
|
|
55
57
|
import "./chunk-6EJMN2Y3.js";
|
|
56
58
|
import "./chunk-IWHMRKLL.js";
|
|
57
|
-
import "./chunk-
|
|
59
|
+
import "./chunk-XDOQXGFO.js";
|
|
58
60
|
import "./chunk-LL4KHSZI.js";
|
|
59
61
|
import "./chunk-4ZG3XFUR.js";
|
|
60
62
|
import "./chunk-EQ63NRB7.js";
|
|
@@ -74,7 +76,7 @@ import {
|
|
|
74
76
|
init_esm_shims();
|
|
75
77
|
var HELP = `clio-coder agents [--json] [--all]
|
|
76
78
|
|
|
77
|
-
List user-facing agent specs from built-in, user, and project recipes.
|
|
79
|
+
List user-facing agent specs from built-in, extension, user, and project recipes.
|
|
78
80
|
|
|
79
81
|
Flags:
|
|
80
82
|
--json emit specs as JSON instead of the formatted table
|
|
@@ -125,4 +127,4 @@ function renderLine(spec) {
|
|
|
125
127
|
export {
|
|
126
128
|
runAgentsCommand
|
|
127
129
|
};
|
|
128
|
-
//# sourceMappingURL=agents-
|
|
130
|
+
//# sourceMappingURL=agents-FIBG2SHA.js.map
|