@iowarp/clio-coder 0.3.8 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +102 -0
- package/NOTICE +33 -0
- package/README.md +12 -3
- package/dist/{acp-U67UHUK2.js → acp-G5WJBNCT.js} +13 -12
- package/dist/{agents-YU6SGALZ.js → agents-FMV2Q5G4.js} +41 -36
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-3IDSJEIK.js} +18 -18
- package/dist/{builtins-C6JMZVV6.js → builtins-XCZWXSC7.js} +5 -5
- package/dist/{chunk-IFBNV6H6.js → chunk-2ANTL7MR.js} +3 -3
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/{chunk-XWSF374K.js → chunk-2OQE55CK.js} +3 -3
- package/dist/{chunk-5H3GB5BO.js → chunk-32KWKNSF.js} +8 -384
- package/dist/{chunk-KTYTFRMB.js → chunk-36EJLSQQ.js} +34 -36
- package/dist/{chunk-FHJEP5SW.js → chunk-3BT2XMV4.js} +19 -13
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-GU2UIAFZ.js → chunk-3URVFKWK.js} +7 -7
- package/dist/{chunk-WEH5XRJQ.js → chunk-3XML7CDN.js} +3 -3
- package/dist/chunk-42FMPA75.js +101 -0
- package/dist/{chunk-A2NJGIB3.js → chunk-5LXZXPKX.js} +2 -2
- package/dist/{chunk-VCBR6CU7.js → chunk-5PSMVOLM.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-6DB53AJS.js} +203 -25
- package/dist/{chunk-3BINW3FP.js → chunk-76ONBSIA.js} +2 -2
- package/dist/{chunk-26LEYJZH.js → chunk-7MCTRUCE.js} +2 -2
- package/dist/{chunk-TYPGUK6W.js → chunk-AB44T6BB.js} +111 -5
- package/dist/{chunk-K4XHGFR5.js → chunk-B7OBL7PK.js} +317 -721
- package/dist/{chunk-B5CSFE7B.js → chunk-BBVJUZHB.js} +2 -2
- package/dist/{chunk-EQ63NRB7.js → chunk-BBVYXMFO.js} +2 -2
- package/dist/{chunk-ZVJ5BLO2.js → chunk-BKFJHQCA.js} +154 -16
- package/dist/chunk-BKFM6EJV.js +462 -0
- package/dist/chunk-BMS5RKQY.js +27 -0
- package/dist/{chunk-ZNLWCMVZ.js → chunk-BPKCPIL7.js} +2 -2
- package/dist/{chunk-TLQJPP24.js → chunk-BUMFYQFY.js} +1394 -1349
- package/dist/{chunk-BNAZZHFG.js → chunk-BYP5D4HI.js} +1 -1
- package/dist/chunk-C2LTL2W6.js +2447 -0
- package/dist/{chunk-ME6CCNFO.js → chunk-CODPRO7Q.js} +8 -8
- package/dist/{chunk-E77JEWSD.js → chunk-CTJ4RNAA.js} +7 -37
- package/dist/{chunk-ODFEOB4F.js → chunk-CY6FY24N.js} +26 -8
- package/dist/{chunk-7RFXX52T.js → chunk-DQITNCXG.js} +642 -172
- package/dist/{chunk-TTHACPOM.js → chunk-DYJP44XW.js} +578 -117
- package/dist/{chunk-2HFZQUHL.js → chunk-F4CKPOEQ.js} +18 -8
- package/dist/{chunk-DGSYXYMX.js → chunk-FEFIFZTL.js} +3 -3
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-MV3K5QF2.js → chunk-GCSMB2KY.js} +2 -2
- package/dist/chunk-GKF55TAZ.js +391 -0
- package/dist/{chunk-VWZOAB7K.js → chunk-GR5G2PVF.js} +9 -8
- package/dist/{chunk-4DGYLA73.js → chunk-GXNLGKAB.js} +80 -9
- package/dist/chunk-HHV2GANA.js +88 -0
- package/dist/{chunk-IIZWH4XA.js → chunk-HI63TFOG.js} +5 -4
- package/dist/{chunk-WLFILSD5.js → chunk-HJJTYUHX.js} +113 -83
- package/dist/{chunk-TT36MB5S.js → chunk-HLW2MRKE.js} +3 -1
- package/dist/{chunk-PMDBGQSJ.js → chunk-HWHKMHUA.js} +7 -7
- package/dist/chunk-HZHHCK24.js +1631 -0
- package/dist/{chunk-WSB3FPX7.js → chunk-I5VEOC6I.js} +39 -143
- package/dist/chunk-IBEBSCYA.js +564 -0
- package/dist/chunk-IQ7KR472.js +362 -0
- package/dist/{chunk-A3WNZD3P.js → chunk-J4W7KFM7.js} +949 -972
- package/dist/{chunk-TB5666IT.js → chunk-JDG2WCRO.js} +5 -5
- package/dist/{chunk-N22QMJKY.js → chunk-K5C3NCBD.js} +4 -4
- package/dist/{chunk-CGKSTWHD.js → chunk-K6BSR66V.js} +2 -1
- package/dist/{chunk-5C3AQNDW.js → chunk-KFZI4NIL.js} +216 -38
- package/dist/chunk-KMVISBZR.js +132 -0
- package/dist/{chunk-WXY7KU3G.js → chunk-LDQ2ZF2M.js} +2 -2
- package/dist/{chunk-XN3L4EYL.js → chunk-LQ3DZAMX.js} +3 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-LY4S7GJC.js} +173 -144
- package/dist/{chunk-4SPRNWDE.js → chunk-MLKNTWH2.js} +19 -19
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/chunk-NEKRRTYW.js +56 -0
- package/dist/{chunk-SPULKLCF.js → chunk-NHCZP4K7.js} +3 -3
- package/dist/chunk-NHLBIGRH.js +1506 -0
- package/dist/chunk-NQQH3YT7.js +302 -0
- package/dist/chunk-NYS75XW5.js +15 -0
- package/dist/{chunk-GN57SG4G.js → chunk-O4XIVISU.js} +10 -8
- package/dist/{chunk-EMYUUSFG.js → chunk-O6TL7WWY.js} +6 -6
- package/dist/chunk-OQBA45DZ.js +97 -0
- package/dist/{chunk-LU7P4LHA.js → chunk-P3FOHJT4.js} +2 -2
- package/dist/chunk-PMZCIOCJ.js +25 -0
- package/dist/{chunk-I4HZDVNP.js → chunk-PQEFIJ36.js} +2 -2
- package/dist/{chunk-J3YUBZWY.js → chunk-QBJA7R7N.js} +62 -6
- package/dist/chunk-QDC3K2U3.js +262 -0
- package/dist/{chunk-2HEJ2F35.js → chunk-QLFS5GO2.js} +22 -10
- package/dist/{chunk-7RGZWPB6.js → chunk-QLL7ILRG.js} +95 -32
- package/dist/{chunk-YS5VLNH5.js → chunk-QREDIESB.js} +6 -6
- package/dist/chunk-QSNYB6ZV.js +195 -0
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-P43ETTHK.js → chunk-SJ5ZKQ4S.js} +2 -2
- package/dist/{chunk-GPIEI3LY.js → chunk-SP2RXXYO.js} +6 -54
- package/dist/chunk-SUCTJL45.js +45 -0
- package/dist/{chunk-DYIM5TJT.js → chunk-SUW5DORT.js} +263 -7
- package/dist/chunk-T56WDKA5.js +183 -0
- package/dist/chunk-TVHHYFHE.js +255 -0
- package/dist/{chunk-FJ3H4MN5.js → chunk-TZ3SGWZZ.js} +3 -3
- package/dist/{chunk-MXKJU4JB.js → chunk-U77AMWDL.js} +91 -10
- package/dist/{chunk-5DHKRSMQ.js → chunk-ULC6OTWO.js} +11 -7
- package/dist/{chunk-RWSI4YD7.js → chunk-UM7N4G5A.js} +33 -12
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-IGWKHNIQ.js → chunk-UXMFQ54G.js} +44 -37
- package/dist/{chunk-HFSBBKSQ.js → chunk-V5DHCITQ.js} +171 -3
- package/dist/{chunk-5WIGXA4T.js → chunk-VAZSBTKF.js} +111 -4
- package/dist/{chunk-VHN4MY6O.js → chunk-VEO4AP2K.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-VFA6GDY5.js} +65 -4
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/chunk-VO67MWHC.js +75 -0
- package/dist/{chunk-XK56QHLX.js → chunk-VPKWYKEY.js} +19 -5
- package/dist/{chunk-TANS5ZJS.js → chunk-VYMXRQI6.js} +36 -22
- package/dist/{chunk-WWCZ5F23.js → chunk-W5VSYASO.js} +77 -16
- package/dist/{chunk-WNIJTQQK.js → chunk-WZR7K7ZX.js} +72 -116
- package/dist/{chunk-5Q2VVUKB.js → chunk-X3YGUTOB.js} +4 -4
- package/dist/chunk-X75E3D2N.js +686 -0
- package/dist/{chunk-VAWNZU7Z.js → chunk-YDFRH54B.js} +4 -4
- package/dist/chunk-YJX4SHTD.js +40 -0
- package/dist/{chunk-ZI647VB5.js → chunk-YPI3QQCF.js} +2 -2
- package/dist/chunk-Z2RR6MAK.js +127 -0
- package/dist/{chunk-FBVTI2TJ.js → chunk-Z4TXYIEG.js} +12 -131
- package/dist/cli/index.js +47 -36
- package/dist/{clio-QVTYJ57A.js → clio-2JXHBBY5.js} +7 -7
- package/dist/{code-nav-FGGFIE7L.js → code-nav-3YYRMYNF.js} +8 -8
- package/dist/{compile-cache-CVJMMODC.js → compile-cache-7FPE6PS3.js} +3 -3
- package/dist/{components-ZFA3SAER.js → components-RYZV4JGP.js} +5 -5
- package/dist/{config-LW5IJFQN.js → config-QZPCMYSO.js} +99 -63
- package/dist/{configure-7XIZCOU4.js → configure-TEGEBYCA.js} +23 -22
- package/dist/{context-Y6Y7QPR6.js → context-AV7OEZ4D.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-E6H5RNMC.js} +56 -47
- package/dist/{context-N52ZA626.js → context-GSXUE4CT.js} +29 -27
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-SHIBYK6T.js} +55 -46
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-5ZGKPGZQ.js} +13 -13
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-EFMJT4LD.js} +93 -64
- package/dist/{docs-7LQ23DLM.js → docs-23KQS3XK.js} +5 -5
- package/dist/doctor-QOA5FNY5.js +313 -0
- package/dist/{eval-BEC2WHDA.js → eval-TFBYQH4H.js} +2032 -156
- package/dist/eval-inventory-SXH7PDKX.js +316 -0
- package/dist/{evidence-REJUMSKM.js → evidence-ERGESKGN.js} +203 -48
- package/dist/{evolve-PY5ZBA5K.js → evolve-VDXTSYCJ.js} +52 -43
- package/dist/{extensions-HVKU65YU.js → extensions-7BGBHN57.js} +13 -7
- package/dist/{fleet-7WZEWRFA.js → fleet-2RRVDF2V.js} +228 -108
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-VJ726XIA.js} +11 -11
- package/dist/fleet-decisions-EPAPM3XJ.js +157 -0
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-JF5QOATM.js} +18 -15
- package/dist/fleet-inspect-VLY4S7QM.js +442 -0
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-AIZUEJOY.js} +5 -5
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-AJRPDMDV.js} +22 -19
- package/dist/fleet-verify-JFEL2L3H.js +175 -0
- package/dist/fleet-view-ZCON35AG.js +102 -0
- package/dist/{init-OG3TPGQG.js → init-DN2WWLFE.js} +72 -62
- package/dist/install-XGLBQY5E.js +13 -0
- package/dist/interop-OZBKXAYL.js +114 -0
- package/dist/{library-CNTMPLRF.js → library-YWZG7IMW.js} +21 -18
- package/dist/{memory-6IS7F275.js → memory-I4C4HMLW.js} +54 -45
- package/dist/{models-ENRJDA5W.js → models-CEYXJBO6.js} +33 -30
- package/dist/{monitor-XLDVO7TN.js → monitor-NZ6GCI3P.js} +59 -52
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-GCGQ4N5I.js} +7827 -7328
- package/dist/panes-HMABYVO4.js +58 -0
- package/dist/panes-KY6W3V2E.js +103 -0
- package/dist/{paths-DBXMZMDU.js → paths-II4K7DNR.js} +5 -5
- package/dist/{reset-RZ4ER727.js → reset-DQ6FGCSH.js} +13 -11
- package/dist/resources-BB3MVJMD.js +111 -0
- package/dist/{run-Y2CNK5RU.js → run-H2GQDUER.js} +119 -87
- package/dist/{share-A55GYP6Z.js → share-GTJN6A5O.js} +20 -17
- package/dist/{skills-ALC5J6AT.js → skills-L55TEW6R.js} +33 -24
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-XVXPH2JI.js} +67 -56
- package/dist/skills-inventory-S4MXPJFV.js +126 -0
- package/dist/slash-commands-ZSGASKJC.js +77 -0
- package/dist/{steer-GGWFUJUD.js → steer-RZGSCY4R.js} +4 -4
- package/dist/{support-MIETYA5E.js → support-PKEUNNQL.js} +6 -6
- package/dist/{targets-VGNXIR3S.js → targets-NCPZ644J.js} +68 -40
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-44SV3YCN.js} +6 -4
- package/dist/tools-DAF3DI3C.js +27 -0
- package/dist/{trace-PNCASAXC.js → trace-FYVW2MQA.js} +207 -10
- package/dist/tui-primitives-2AKXQNZK.js +13 -0
- package/dist/{uninstall-ZJF5H5ZN.js → uninstall-DW2PNOIC.js} +5 -5
- package/dist/{upgrade-FUSUAGHR.js → upgrade-3XPP6OQL.js} +27 -24
- package/dist/{usage-N4MKVHKD.js → usage-3NLHGTU2.js} +114 -62
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-SSQONKRT.js} +172 -13
- package/dist/{verify-LTDHYBGY.js → verify-3U6J7FZI.js} +10 -10
- package/dist/{web-fetch-2YHJ3KTG.js → web-fetch-S7RR6GZ7.js} +3 -3
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-CEHYGPGQ.js} +75 -65
- package/dist/with-panes-MKB46MPQ.js +782 -0
- package/dist/worker/entry.js +106 -75
- package/docs/README.md +3 -2
- package/docs/acp.md +24 -3
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +9 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +62 -3
- package/docs/commands-and-modes.md +31 -2
- package/docs/configuration-and-targets.md +69 -12
- package/docs/context-engine.md +63 -4
- package/docs/development-pipeline.md +19 -0
- package/docs/dispatch-typed-intent.md +385 -0
- package/docs/documentation-coverage.md +5 -5
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +3 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +12 -11
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +27 -1
- package/docs/fleet-dispatch.md +25 -4
- package/docs/installation-and-lifecycle.md +15 -2
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +10 -1
- package/docs/observability.md +55 -4
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +2 -2
- package/docs/scientific-validation.md +3 -3
- package/docs/session-lifecycle.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +18 -8
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +88 -1
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +5 -2
- package/src/cli/acp.ts +6 -2
- package/src/cli/agents.ts +1 -1
- package/src/cli/argv.ts +25 -0
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/configure.ts +23 -21
- package/src/cli/doctor-panes.ts +124 -0
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor-toolchain.ts +57 -0
- package/src/cli/doctor.ts +22 -1
- package/src/cli/eval-inventory.ts +436 -0
- package/src/cli/eval.ts +93 -16
- package/src/cli/evidence-detail.ts +88 -0
- package/src/cli/evidence-inventory.ts +183 -0
- package/src/cli/evidence.ts +30 -5
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet-decisions.ts +69 -0
- package/src/cli/fleet-inspect.ts +334 -0
- package/src/cli/fleet-verify.ts +133 -0
- package/src/cli/fleet-view.ts +810 -0
- package/src/cli/fleet.ts +179 -39
- package/src/cli/index.ts +14 -2
- package/src/cli/interop-inspect.ts +128 -0
- package/src/cli/interop.ts +34 -0
- package/src/cli/panes.ts +35 -0
- package/src/cli/reset.ts +5 -2
- package/src/cli/run.ts +58 -0
- package/src/cli/skills-inventory.ts +185 -0
- package/src/cli/skills.ts +16 -13
- package/src/cli/targets.ts +44 -13
- package/src/cli/tools.ts +321 -0
- package/src/cli/trace-inspect.ts +252 -0
- package/src/cli/trace.ts +85 -5
- package/src/cli/usage.ts +63 -14
- package/src/cli/verifiers-inspect.ts +347 -0
- package/src/cli/verifiers.ts +9 -0
- package/src/core/bus-events.ts +33 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +67 -0
- package/src/core/defaults.ts +124 -8
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +80 -6
- package/src/core/theme-token-hex.ts +43 -0
- package/src/core/tool-names.ts +2 -1
- package/src/core/xdg.ts +1 -1
- package/src/domains/agents/fleets/build-review.md +0 -3
- package/src/domains/agents/fleets/build-test.md +0 -3
- package/src/domains/agents/result-contract-filesystem.ts +32 -0
- package/src/domains/agents/result-contract.ts +164 -35
- package/src/domains/config/classify.ts +6 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission-error.ts +9 -0
- package/src/domains/dispatch/admission.ts +52 -14
- package/src/domains/dispatch/capacity-lease.ts +118 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/council-topology.ts +398 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +378 -82
- package/src/domains/dispatch/fleet-node-prompt.ts +62 -0
- package/src/domains/dispatch/fleet-plan.ts +7 -2
- package/src/domains/dispatch/fleet-run.ts +87 -4
- package/src/domains/dispatch/gate-decisions.ts +11 -1
- package/src/domains/dispatch/gate-role-prompts.ts +9 -0
- package/src/domains/dispatch/gate-topology.ts +289 -0
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +22 -0
- package/src/domains/dispatch/intent-compatibility.ts +330 -0
- package/src/domains/dispatch/intent.ts +85 -1
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +139 -11
- package/src/domains/dispatch/run-event-journal-bridge.ts +149 -0
- package/src/domains/dispatch/run-event-journal.ts +598 -0
- package/src/domains/dispatch/state.ts +49 -1
- package/src/domains/dispatch/types.ts +13 -0
- package/src/domains/dispatch/validation.ts +33 -8
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +342 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/inventory.ts +113 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +105 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/store.ts +6 -0
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -1
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +82 -17
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +3 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +97 -21
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/mux/contract.ts +434 -0
- package/src/domains/mux/detect.ts +158 -0
- package/src/domains/mux/extension.ts +47 -0
- package/src/domains/mux/index.ts +96 -0
- package/src/domains/mux/manifest.ts +6 -0
- package/src/domains/mux/operations.ts +164 -0
- package/src/domains/mux/pane-registry.ts +90 -0
- package/src/domains/mux/protocol.ts +49 -0
- package/src/domains/mux/socket-client.ts +816 -0
- package/src/domains/mux/types.ts +222 -0
- package/src/domains/mux/viewer-command.ts +59 -0
- package/src/domains/mux/yazi/assets/init.lua +2 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/LICENSE +21 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/README.md +78 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/main.lua +255 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/types.lua +12 -0
- package/src/domains/mux/yazi/assets/yazi.toml +17 -0
- package/src/domains/mux/yazi/event-stream.ts +180 -0
- package/src/domains/mux/yazi/profile.ts +299 -0
- package/src/domains/mux/yazi/session.ts +228 -0
- package/src/domains/mux/yazi/theme.ts +30 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +22 -1
- package/src/domains/observability/index.ts +9 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +234 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +228 -0
- package/src/domains/providers/endpoint-slots-store.ts +189 -0
- package/src/domains/providers/extension.ts +20 -3
- package/src/domains/providers/index.ts +32 -0
- package/src/domains/providers/model-runtime-capabilities.ts +32 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/boot-manifest.ts +1 -0
- package/src/domains/providers/runtimes/builtins.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +375 -0
- package/src/domains/providers/support.ts +1 -0
- package/src/domains/providers/target-model-cache.ts +124 -0
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/resources/skills/loader.ts +33 -0
- package/src/domains/safety/action-classifier.ts +6 -0
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/domains/toolchain/archive.ts +175 -0
- package/src/domains/toolchain/contract.ts +28 -0
- package/src/domains/toolchain/extension.ts +47 -0
- package/src/domains/toolchain/index.ts +39 -0
- package/src/domains/toolchain/install.ts +327 -0
- package/src/domains/toolchain/manifest.ts +8 -0
- package/src/domains/toolchain/paths.ts +34 -0
- package/src/domains/toolchain/registry.ts +265 -0
- package/src/domains/toolchain/remove.ts +218 -0
- package/src/domains/toolchain/resolve.ts +182 -0
- package/src/domains/toolchain/types.ts +113 -0
- package/src/domains/toolchain/version.ts +88 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/acp/server.ts +413 -70
- package/src/engine/acp/types.ts +19 -1
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/claude/sdk-module.ts +98 -0
- package/src/engine/claude/sdk-runtime.ts +19 -11
- package/src/engine/provider-payload.ts +29 -1
- package/src/engine/tui-primitives.ts +21 -0
- package/src/engine/tui.ts +1 -0
- package/src/engine/worker-runtime.ts +2 -13
- package/src/entry/boot-options.ts +2 -0
- package/src/entry/orchestrator.ts +279 -35
- package/src/entry/panes-activation.ts +31 -0
- package/src/entry/with-panes.ts +20 -0
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +41 -10
- package/src/interactive/cost-overlay.ts +66 -6
- package/src/interactive/council-grid.ts +1 -3
- package/src/interactive/council.ts +11 -0
- package/src/interactive/dispatch-board.ts +102 -20
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +177 -7
- package/src/interactive/interactive-input-runtime.ts +15 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +63 -10
- package/src/interactive/memory-overlay.ts +9 -0
- package/src/interactive/modal-marker.ts +170 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/mux-bridge.ts +214 -0
- package/src/interactive/overlay-frame.ts +58 -2
- package/src/interactive/overlay-general-openers.ts +17 -0
- package/src/interactive/overlay-key-routing.ts +52 -3
- package/src/interactive/overlay-lifecycle.ts +56 -7
- package/src/interactive/overlay-model-selectors.ts +40 -3
- package/src/interactive/overlay-permission-lifecycle.ts +112 -24
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlay-transitions.ts +18 -4
- package/src/interactive/overlays/agents.ts +1 -0
- package/src/interactive/overlays/ask-user.ts +227 -49
- package/src/interactive/overlays/auth-dialog.ts +1 -0
- package/src/interactive/overlays/context-reset.ts +1 -0
- package/src/interactive/overlays/cwd-fallback.ts +1 -0
- package/src/interactive/overlays/decisions.ts +11 -11
- package/src/interactive/overlays/extensions.ts +1 -0
- package/src/interactive/overlays/fleet-run-approval.ts +1 -0
- package/src/interactive/overlays/handoff-review.ts +1 -0
- package/src/interactive/overlays/help-reference.ts +6 -0
- package/src/interactive/overlays/interop.ts +1 -0
- package/src/interactive/overlays/library-install-confirm.ts +1 -0
- package/src/interactive/overlays/library-tabs.ts +28 -0
- package/src/interactive/overlays/list-overlay.ts +10 -1
- package/src/interactive/overlays/message-picker.ts +1 -0
- package/src/interactive/overlays/model-scope.ts +86 -0
- package/src/interactive/overlays/model-selector.ts +1 -0
- package/src/interactive/overlays/prompts.ts +12 -1
- package/src/interactive/overlays/session-selector.ts +1 -0
- package/src/interactive/overlays/settings-sections.ts +30 -0
- package/src/interactive/overlays/settings.ts +614 -45
- package/src/interactive/overlays/side-question.ts +1 -0
- package/src/interactive/overlays/skills-hub.ts +3 -11
- package/src/interactive/overlays/tree-selector.ts +1 -0
- package/src/interactive/pane-policy.ts +46 -0
- package/src/interactive/panes-runtime.ts +292 -0
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/compaction-summary.ts +29 -0
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/renderers/worker-entry.ts +122 -14
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/slash-commands.ts +251 -15
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/tasks-overlay.ts +1 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/theme/tokens.ts +3 -14
- package/src/interactive/turn-context.ts +346 -33
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/artifacts.ts +109 -1
- package/src/interactive/view/view-overlay.ts +29 -3
- package/src/interactive/watch-pane.ts +152 -0
- package/src/interactive/worker-progress.ts +7 -1
- package/src/interactive/worker-receipts.ts +19 -1
- package/src/interactive/worker-stream.ts +5 -0
- package/src/interactive/yazi-bridge.ts +444 -0
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/bootstrap.ts +26 -2
- package/src/tools/builtin-tool-catalog.ts +15 -0
- package/src/tools/compete-worktrees.ts +83 -2
- package/src/tools/core-bootstrap.ts +2 -1
- package/src/tools/dispatch-admission.ts +14 -3
- package/src/tools/dispatch-arguments.ts +20 -20
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-run-events.ts +134 -19
- package/src/tools/dispatch-runner.ts +29 -7
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/dispatch-types.ts +15 -3
- package/src/tools/dispatch.ts +1 -1
- package/src/tools/executables.ts +17 -14
- package/src/tools/observation.ts +54 -4
- package/src/tools/panes-surface.ts +38 -0
- package/src/tools/panes.ts +112 -0
- package/src/tools/policy.ts +10 -1
- package/src/tools/presentation.ts +1 -0
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HLE42MG7.js +0 -37
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-JOZYP4GM.js +0 -279
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -3,7 +3,19 @@
|
|
|
3
3
|
Clio Coder is designed to be self-contained and platform-compliant. This document outlines the default directory paths, file purposes, permission levels, and lifecycle commands (`install`, `reset`, `upgrade`, and `uninstall`). Clio Coder installs from npm as `@iowarp/clio-coder` (`npm install -g @iowarp/clio-coder`, published since v0.3.0) or from a source checkout with a deterministic local symlink; the CLI classifies both install kinds and `clio-coder upgrade` handles each.
|
|
4
4
|
|
|
5
5
|
> [!TIP]
|
|
6
|
-
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.
|
|
6
|
+
> **Interactive Spec Available:** An interactive dashboard with a path simulator and visual flowcharts is located at [docs/html/lifecycle_blueprint.html](html/lifecycle_blueprint.html) (Version: 0.4.0). You can open it directly in any web browser to view details dynamically.
|
|
7
|
+
|
|
8
|
+
### Optional dependency: the Claude Agent SDK
|
|
9
|
+
|
|
10
|
+
`@anthropic-ai/claude-agent-sdk` is an `optionalDependencies` entry, not a hard dependency. Its platform package carries a proprietary binary of roughly 224MB per platform, and only the `claude-sdk` runtime uses it. Skip it with:
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
npm install -g @iowarp/clio-coder --omit=optional
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Measured on Linux x64 with a production install (`--omit=dev`): 387MB across 118 packages with the SDK, 143MB across 109 packages without it. That is 244MB and nine packages saved, a 63% smaller tree, and what remains is fully open-licensed.
|
|
17
|
+
|
|
18
|
+
Everything except the `claude-sdk` runtime works on the smaller install: boot, `clio-coder doctor`, every other target and worker runtime. Dispatching a `claude-sdk` target on an install that omitted the package fails that run with a diagnostic naming the package and the command that fixes it (`npm install @anthropic-ai/claude-agent-sdk@0.3.186`); nothing else degrades, and nothing fails at startup.
|
|
7
19
|
|
|
8
20
|
---
|
|
9
21
|
|
|
@@ -84,6 +96,7 @@ The core files are created automatically during the first run. `credentials.yaml
|
|
|
84
96
|
| **State** | `install.json` | Install metadata: Clio version, node, platform, `installedAt` (written once at first install), `upgradedAt` and `upgradedFrom` (stamped on a version change), and `noticedVersion` (the version whose one-time upgrade notice the interactive launch has shown). | Writer/umask default | Removed by uninstall / `reset --state`. |
|
|
85
97
|
| **State** | `migrations.json` | Log of successfully applied schema/state migrations. | Writer/umask default | Removed by uninstall / `reset --state`. |
|
|
86
98
|
| **Data** | `memory/records.json` | Long-term learning memories (up to 500 records) proposed/approved from runs. | Writer/umask default | Removed by uninstall / `reset --data`. |
|
|
99
|
+
| **Data** | `tools/<id>/<version>/` | One pinned external program Clio downloaded on request (`clio-coder tools install <id>`), with its upstream license text and a `clio-install.json` recording url, sha256, platform and install time. Binaries `0o755`, documents `0o644`. Only the pinned version is kept: a successful install prunes the versions it supersedes. | `0o755` dir | `clio-coder tools remove <id>` deletes every version of one tool; removed by uninstall / `reset --data`. |
|
|
87
100
|
| **State** | `audit/YYYY-MM-DD.jsonl` | Daily safety audit logs showing allowed/blocked tool actions. | Writer/umask default | Removed by uninstall / `reset --state`. |
|
|
88
101
|
| **State** | `sessions/<cwdHash>/<id>/` | Session details: `meta.json`, `current.jsonl`, and fork hierarchies `tree.json`. | Writer/umask default | Removed by uninstall / `reset --state`. |
|
|
89
102
|
|
|
@@ -256,7 +269,7 @@ inventory of what a level covers, because a remembered list drifts as soon as a
|
|
|
256
269
|
new artifact is written into a root.
|
|
257
270
|
|
|
258
271
|
* `--state` *(Default)*: Deletes the state root only. It holds every session transcript and the audit trail beside it, so a reset is the end of `resume`, `/view`, and their history. This is the level a bare `clio-coder reset` selects, and it carries that note in its preview.
|
|
259
|
-
* `--data`: Deletes the data root only: memory, evidence, evals (durable products).
|
|
272
|
+
* `--data`: Deletes the data root only: memory, evidence, evals, and any vendored external tools (durable products). The vendored tools are the one entry a reset cannot regenerate locally; `clio-coder tools install <id>` downloads them again.
|
|
260
273
|
* `--cache`: Deletes the cache root only.
|
|
261
274
|
* `--auth`: Deletes `credentials.yaml`. Removes all saved keys.
|
|
262
275
|
* `--config`: Deletes `settings.yaml` to revert preferences to default.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Middleware and Component Registry
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard with an interactive component scanner and a dynamic hook-and-effect pipeline is located at [docs/html/middleware_blueprint.html](html/middleware_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Clio Coder has two related but separate surfaces:
|
|
7
7
|
|
package/docs/model-catalog.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Model Catalog, Runtime Refresh, and Field Notes
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard mapping capabilities, probe discovery, and target resolution is located at [docs/html/models_blueprint.html](html/models_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Clio Coder treats a selectable model as the intersection of three sources:
|
|
7
7
|
|
|
@@ -161,6 +161,15 @@ The Context Engine evaluates thinking mechanisms per model target and manages li
|
|
|
161
161
|
|
|
162
162
|
- **Ollama Native (`ollama-native`):** Ollama utilizes the native `thinking` field in the request and response payloads. The engine handles Ollama-specific effort levels and streams reasoning increments cleanly through the native thinking channel.
|
|
163
163
|
- **LM Studio (`lmstudio`):** Chat uses the OpenAI-compatible `/v1/chat/completions` surface, including its `reasoning` stream field. Clio controls thinking only with `reasoning_effort` and never sends `chat_template_kwargs` to LM Studio. See <https://lmstudio.ai/docs/developer/openai-compat/chat-completions>.
|
|
164
|
+
- **LiteLLM (`litellm`):** This is a gateway runtime, not an `openai-compat`
|
|
165
|
+
alias. Discovery checks `/health/liveliness`, reads aliases and capability
|
|
166
|
+
metadata from `/v1/model/info`, and records the physical deployment reported
|
|
167
|
+
by `x-litellm-*` response headers. Defaults stay conservative when metadata is
|
|
168
|
+
absent: tools, vision, and reasoning are not inferred. The runtime advertises
|
|
169
|
+
standard `json_schema` structured output and treats schema-plus-tools as a
|
|
170
|
+
conflict because the gateway alias does not identify one stable upstream.
|
|
171
|
+
Residency is observe-only because LiteLLM owns loading and eviction behind the
|
|
172
|
+
alias.
|
|
164
173
|
- **OpenAI Completions (`openai-completions`):** The OpenAI-compatible completions provider preserves reasoning blocks within assistant messages. It replays thinking blocks via the `reasoning_content` parameter in the message history, ensuring that the model maintains its chain-of-thought across conversational turns without stripping the data.
|
|
165
174
|
- **Anthropic OAuth / API (`anthropic-max`):** Uses the `anthropic-extended` thinking format. The engine supports Anthropic's native extended thinking block protocol, streaming thinking increments and outputting them wrapped appropriately or natively depending on target capabilities.
|
|
166
175
|
- **Reasoning-Never Models (`thinking.mechanism: none`):** When a model is configured or cataloged with `thinking.mechanism: none`, it is treated as a reasoning-never model. For these models, Clio must not send any thinking fields or parameters in requests, must not replay thinking blocks, must not surface thinking events to the TUI, and must not preserve or log reasoning token usage in metrics.
|
package/docs/observability.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Observability Viewer
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/observability_blueprint.html](html/observability_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
`/view` is the interactive artifact viewer for a Clio session. It keeps the live transcript compact while preserving a full inspection path for durable artifacts, task ledgers, and successful workspace outputs.
|
|
7
7
|
|
|
@@ -13,6 +13,35 @@
|
|
|
13
13
|
|
|
14
14
|
`/view` opens a full-screen split viewer. The left pane groups artifacts by category and supports type-to-filter. The right pane renders the selected artifact with pager controls. `Tab` or `Shift+Tab` switches between the artifact list and details. `Left` and `Right` jump to the previous or next non-empty category from either pane; `Up` and `Down` select artifacts in the list or scroll details in the content pane. Category jumps honor the active filter and wrap at the ends. `v` verifies a selected receipt. `o` shows the absolute backing path through the notice channel when the selected artifact has one; pathless artifacts produce a warning notice instead. In the list pane, `Esc` clears a non-empty filter before a second `Esc` closes the viewer.
|
|
15
15
|
|
|
16
|
+
## Trace retention and state usage
|
|
17
|
+
|
|
18
|
+
The SQLite trace mirror at `<state-dir>/trace.sqlite` is rebuildable and bounded. By default Clio retains terminal runs for 30 days and limits the allocated database to 128 MiB (134,217,728 bytes), whichever limit is reached first. The policy runs after each dispatch or interactive turn becomes terminal. It deletes a run as one unit across `runs`, `phases`, `events`, `envelopes`, `gate_results`, `agent_sessions`, and `processes`. A `queued` or `running` run is never a candidate, even when its start time is older than the age cutoff or its rows put the store over the byte limit.
|
|
19
|
+
|
|
20
|
+
Two environment variables configure the automatic policy:
|
|
21
|
+
|
|
22
|
+
| Variable | Default | Valid values |
|
|
23
|
+
| --- | ---: | --- |
|
|
24
|
+
| `CLIO_CODER_TRACE_RETENTION_DAYS` | `30` | An integer of at least 1. |
|
|
25
|
+
| `CLIO_CODER_TRACE_MAX_BYTES` | `134217728` | An integer of at least 1,048,576. |
|
|
26
|
+
|
|
27
|
+
An operator can apply the current policy immediately or supply one-command overrides:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
clio-coder trace prune
|
|
31
|
+
clio-coder trace prune --max-age-days 14 --max-bytes 67108864
|
|
32
|
+
clio-coder trace prune --json
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
The command reports terminal runs removed, total rows removed across the seven run-owned tables, physical bytes reclaimed from `trace.sqlite` and its WAL sidecars, whether `VACUUM` ran, and how many live runs were protected. Age pruning uses a terminal run's `ended_at`. Size pruning removes the oldest terminal runs until the live database pages fit or no terminal candidate remains.
|
|
36
|
+
|
|
37
|
+
Deleting SQLite rows creates reusable pages but does not normally reduce the file. Clio runs `VACUUM` when at least 20 percent of allocated pages are reclaimable, or whenever reclaiming deleted pages is necessary to enforce the 128 MiB bound. It then truncates the WAL. Smaller deletions remain available for SQLite to reuse and avoid rewriting the whole database on every completed run.
|
|
38
|
+
|
|
39
|
+
`clio-coder doctor` includes a `state storage` row with the recursive byte total for the state directory and the largest top-level contributor. For example:
|
|
40
|
+
|
|
41
|
+
```text
|
|
42
|
+
OK state storage 96.4 MiB (101,082,624 bytes); largest contributor trace.sqlite at 89.9 MiB (94,248,960 bytes)
|
|
43
|
+
```
|
|
44
|
+
|
|
16
45
|
---
|
|
17
46
|
|
|
18
47
|
## The Evidence Spine End-to-End
|
|
@@ -69,8 +98,28 @@ An `EvidenceIndexRow` has the following schema:
|
|
|
69
98
|
| `turns` | `turns in window` and the `tokens` fact | Folded calls that were turns, so labelled calls are subtracted exactly as `/cost` subtracts them. |
|
|
70
99
|
| `sideQuestions` | `side questions in window` and the `tokens` fact | `/btw` rounds in the window. |
|
|
71
100
|
| `handoffs` | `handoffs in window` and the `tokens` fact | `/handoff` extraction rounds in the window. |
|
|
101
|
+
| `prewarms` | `pre-warms in window` and the `tokens` fact | Prompt pre-warm rounds in the window. |
|
|
102
|
+
| `backgroundMemorySteps` | `background memory steps in window` and the `tokens` fact | Proactive-memory model steps in the window. |
|
|
103
|
+
|
|
104
|
+
The last five fields appear only when at least one labelled call falls in the window, and each individual line is printed only when its own count is above zero. An archive with no labelled call in it renders exactly as it did before those fields existed, so their presence is itself the signal that money was spent beside a session. All four labelled kinds are subtracted from `turns` the same way, so a session's turn count never includes a round the operator did not take.
|
|
105
|
+
|
|
106
|
+
The report also prints a prompt-cache block, one row per session that recorded any cache telemetry:
|
|
107
|
+
|
|
108
|
+
```text
|
|
109
|
+
prompt cache by session (from backend timings and persisted verdicts)
|
|
110
|
+
session uncached prefill hot/partial/cold/small
|
|
111
|
+
3vpu6z19ee7t 130353 4/3/2/0
|
|
112
|
+
```
|
|
72
113
|
|
|
73
|
-
|
|
114
|
+
`uncached prefill` is the sum of the backend's own newly evaluated prompt tokens across every persisted call in that session, and reads `n/a` rather than `0` when the server reported no cache figure to subtract. The four counts are the per-call verdicts. Both facts are also in `--json` under a `session-cache` fact per session.
|
|
115
|
+
|
|
116
|
+
`clio-coder doctor` reports the same evidence for the latest session only, as one row, so a cache problem is visible without opening the TUI or a report:
|
|
117
|
+
|
|
118
|
+
```text
|
|
119
|
+
OK cache telemetry last session 3vpu6z19ee7t: hot 4 · partial 3 · cold 2 · small 0; top expected reason dispatch (3)
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The row reads `top expected reason none` when the session recorded verdicts but no expected-cold reason, and it degrades to a warning saying `no prompt-cache telemetry recorded` when the latest session has none at all, which is the honest answer for a target whose backend reports nothing rather than a claim of a perfect cache. "Latest" selects the most recent `current.jsonl` by its newest entry timestamp, falling back to the file's mtime, so the other diagnostic JSONL files in a session directory cannot be mistaken for the conversation.
|
|
74
123
|
|
|
75
124
|
### The Out-of-Turn Usage Store
|
|
76
125
|
|
|
@@ -102,6 +151,8 @@ A row has the following schema:
|
|
|
102
151
|
|
|
103
152
|
`repoIdentity` is the same cwd hash the session ledger is filed under, which is what lets `usage report --repo <path>` select these rows with the hash it already computes for the ledgers.
|
|
104
153
|
|
|
154
|
+
`label` is one of `side-question`, `handoff`, `prewarm`, or `background-memory`. The last two joined for the same reason as the first two: a prompt pre-warm and a proactive-memory step are provider calls the operator did not ask for and would otherwise never see, and neither appends anything to the session JSONL. A row may also carry `timing { durationMs }` and a `promptCache` block built from the backend's own prefill facts when the server reported them; a backend that reports no timings simply omits the block, as LM Studio's OpenAI-compatible port does.
|
|
155
|
+
|
|
105
156
|
---
|
|
106
157
|
|
|
107
158
|
## Artifact Categories and Path Layouts
|
|
@@ -206,8 +257,8 @@ The base provenance sets, steering, routing, quality, worker identity, result-co
|
|
|
206
257
|
| `safety.toolTelemetry.ingestionErrors` | `number` | Current dispatch receipts | Malformed or lost frames, event-fold/source errors, and drain timeouts that make otherwise mediated telemetry incomplete | experimental |
|
|
207
258
|
| `safety.toolTelemetry.unfinished` | `{ tool, count }[]` | Current dispatch receipts | Tool starts that had no matching finish when the receipt sealed | experimental |
|
|
208
259
|
| `safety.toolTelemetry.workspaceMutationPossible` | `boolean` | Current dispatch receipts | Whether incomplete or unavailable telemetry could conceal a shared-workspace mutation; retry admission fails closed when true | experimental |
|
|
209
|
-
| `autonomyEnforcement.grade` | `string` | Always in v0.
|
|
210
|
-
| `autonomyEnforcement.autonomy` | `string` | Always in v0.
|
|
260
|
+
| `autonomyEnforcement.grade` | `string` | Always in v0.4.0 | The autonomy grade level enforced for the run | experimental |
|
|
261
|
+
| `autonomyEnforcement.autonomy` | `string` | Always in v0.4.0 | The effective autonomy level name (e.g. auto-edit, suggest, read-only, full-auto) | experimental |
|
|
211
262
|
| `autonomyEnforcement.externalMode` | `string` | When running external worker | The execution mode of the external worker runtime | experimental |
|
|
212
263
|
| `autonomyEnforcement.dangerousBypass` | `boolean` | When running external worker | Whether a safety bypass was explicitly activated | experimental |
|
|
213
264
|
| `validationGrounding.claimed` | `number` | Validation grounding evaluated | Count of validations claimed by worker | experimental |
|
package/docs/proactive-memory.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Proactive task memory
|
|
2
2
|
|
|
3
|
-
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.
|
|
3
|
+
> **Interactive Spec Available:** An interactive memory lifecycle dashboard and simulator is located at [docs/html/memory_blueprint.html](html/memory_blueprint.html) (Version: 0.4.0).
|
|
4
4
|
|
|
5
5
|
Clio's proactive task memory protects long-running work from behavioral state
|
|
6
6
|
decay: a requirement, environment fact, failed attempt, or diagnosis can still
|
|
@@ -120,7 +120,7 @@ malformed response, or telemetry failure is silent and never blocks a tool.
|
|
|
120
120
|
- `memory.intervention.everyNTools` (default `10`): Minimum completed-tool interval between background interventions.
|
|
121
121
|
- `memory.intervention.windowSteps` (default `8`): Completed tool-trajectory window analyzed during background evaluation.
|
|
122
122
|
- `memory.intervention.maxTokens` (default `400`): Bounds the rendered memory-bank and reminder context budget; the policy model output cap is a separate fixed `4,000`-token contract in `task-memory-policy.ts`, sized so that a model which reasons anyway still reaches its envelope.
|
|
123
|
-
- `memory.intervention.timeoutMs` (default `
|
|
123
|
+
- `memory.intervention.timeoutMs` (default `30000`): Wall-clock limit for one background memory-policy request. The step is detached, so this deadline never delays a turn, but it does hold a request slot on a real inference endpoint that your own turns and your dispatched workers queue against. The default is what a turn boundary can wait for rather than what a long-tailed route eventually answers in: on the reference route below, 23 of 60 steps ran past 30 seconds and 531 of the measured 1,666 seconds were spent beyond that mark. Raise it only if you have measured that your route's slow steps are the ones producing reminders, and read the trade in "Cost and the default decision" first.
|
|
124
124
|
|
|
125
125
|
## Trigger semantics
|
|
126
126
|
|
|
@@ -203,6 +203,88 @@ Thus `last` remains `injected` across such continuations until a later
|
|
|
203
203
|
tool-bearing or explicitly triggered memory step produces a new outcome (e.g.,
|
|
204
204
|
a healthy tool leading to `silent`).
|
|
205
205
|
|
|
206
|
+
## Cost and the default decision
|
|
207
|
+
|
|
208
|
+
The LLM tier costs real tokens, real seconds of model time, and a request slot on
|
|
209
|
+
a server that is usually the same machine the operator's own turns run on. Every
|
|
210
|
+
step is therefore accounted for the way a `/btw` side question is: one cost entry
|
|
211
|
+
under the `background-memory` label, which `/cost` shows as its own `memory steps`
|
|
212
|
+
row, and one durable row in `<stateDir>/usage/out-of-turn.jsonl` carrying the
|
|
213
|
+
usage, the call's duration, and the backend's prefill facts, which
|
|
214
|
+
`clio-coder usage report` folds after the process exits. `/memory` shows the
|
|
215
|
+
lifetime figures folded from `steps.jsonl`: steps, tokens, model time, and the
|
|
216
|
+
hit rate.
|
|
217
|
+
|
|
218
|
+
### The measurement
|
|
219
|
+
|
|
220
|
+
From one operator's `steps.jsonl`, 274 rows spanning 2026-08-14 to 2026-08-29 on
|
|
221
|
+
a small local background route:
|
|
222
|
+
|
|
223
|
+
| Figure | Value |
|
|
224
|
+
| --- | --- |
|
|
225
|
+
| Model-tier steps | 60 |
|
|
226
|
+
| Tokens | 137,205 |
|
|
227
|
+
| Model time | 1,666.6 s |
|
|
228
|
+
| Step latency | median 18.7 s, p90 70.2 s, max 102.5 s |
|
|
229
|
+
| Injections produced by the model tier | 6 |
|
|
230
|
+
| Hit rate | 10.0 percent |
|
|
231
|
+
| Cost per injection | 22,868 tokens and 278 s of model time |
|
|
232
|
+
| Model-tier injections in the last 5 days | 0 of 4 steps |
|
|
233
|
+
|
|
234
|
+
Four further injections in the same window came from the free rules tier, so the
|
|
235
|
+
lifetime total of 10 injections is not the model tier's score. Rules-tier
|
|
236
|
+
injections cost nothing.
|
|
237
|
+
|
|
238
|
+
### The decision
|
|
239
|
+
|
|
240
|
+
The default does not change, and it is a deliberate default rather than an
|
|
241
|
+
unexamined one:
|
|
242
|
+
|
|
243
|
+
- `memory.intervention.enabled` stays `true`. It runs the rules tier, which makes
|
|
244
|
+
no model calls, spends no tokens, and produced 4 of the 10 injections.
|
|
245
|
+
- The LLM tier stays opt-in through `background.target` and `background.model`,
|
|
246
|
+
which is already the case: an unset background role never resolves a client.
|
|
247
|
+
A 10 percent hit rate at 22,868 tokens per injection does not earn a default-on
|
|
248
|
+
position, and it is not so poor that it earns removal from an operator who has
|
|
249
|
+
measured their own route and wants it.
|
|
250
|
+
- The step deadline drops from 180 s to 30 s. This is the one behavioral change,
|
|
251
|
+
and it is a genuine trade: at 30 s, two of the six observed injections, at
|
|
252
|
+
53.6 s and 57.7 s, would have been cut, while 531 s of the 1,666 s spent would
|
|
253
|
+
not have been spent at all. The deadline is the bound on what one optional call
|
|
254
|
+
may hold a shared local server for, not a prediction of when a route answers.
|
|
255
|
+
- A step that would run on the endpoint the chat target is streaming against is
|
|
256
|
+
skipped with reason `endpoint_busy`, and the skip is recorded. On a single-slot
|
|
257
|
+
llama.cpp router the alternative is queueing behind the operator's own decoding
|
|
258
|
+
or evicting the resident model, and neither is a cost an optional call may
|
|
259
|
+
impose.
|
|
260
|
+
|
|
261
|
+
### What a background target costs on a shared local server
|
|
262
|
+
|
|
263
|
+
If `background.target` names the same server as `orchestrator.target`, that
|
|
264
|
+
server's slots are shared. On a llama.cpp router started with `--parallel 1`
|
|
265
|
+
there is exactly one, and the memory step and the operator's turn contend for it.
|
|
266
|
+
|
|
267
|
+
The consequence is worth stating plainly: a shared endpoint suppresses the model
|
|
268
|
+
tier rather than merely delaying it. A step is started from the `turn_end` hook,
|
|
269
|
+
which fires inside the streaming run at `agent_end`
|
|
270
|
+
(`src/interactive/turn-runtime.ts`), while the chat loop still holds its
|
|
271
|
+
foreground registration on that endpoint; the loop releases the hold afterwards,
|
|
272
|
+
in the `finally` around the run (`src/interactive/chat-loop.ts`). Every boundary
|
|
273
|
+
therefore finds the endpoint busy and records `dropped`/`endpoint_busy`. That is
|
|
274
|
+
the intended trade: an optional call may not take the one slot the operator's own
|
|
275
|
+
turn is using, and it may not make the server swap the resident model out. The
|
|
276
|
+
`/memory` step list and `steps.jsonl` say so on every boundary, so the tier is
|
|
277
|
+
visibly declining rather than quietly idle.
|
|
278
|
+
|
|
279
|
+
The second mechanism is an `expected cold` stamp, for the case where a step did
|
|
280
|
+
run on the chat endpoint. Its prompt is a trajectory rather than the chat prefix,
|
|
281
|
+
so the next turn's prefill is expected to be cold; `/context` names
|
|
282
|
+
`background_memory` as the reason instead of reporting an unexplained cold
|
|
283
|
+
prefix.
|
|
284
|
+
|
|
285
|
+
Pointing the background role at a second machine avoids both effects and is the
|
|
286
|
+
arrangement the tier is designed for.
|
|
287
|
+
|
|
206
288
|
## Choosing a background model
|
|
207
289
|
|
|
208
290
|
Memory reads a trajectory and writes a fixed envelope. It does not plan, and it
|
|
@@ -257,7 +339,7 @@ memory:
|
|
|
257
339
|
everyNTools: 10
|
|
258
340
|
windowSteps: 8
|
|
259
341
|
maxTokens: 400
|
|
260
|
-
timeoutMs:
|
|
342
|
+
timeoutMs: 30000
|
|
261
343
|
```
|
|
262
344
|
|
|
263
345
|
With `background.target` and `background.model` unset, Clio stays in the
|
|
@@ -289,12 +371,12 @@ percentile of 79.9, and a 95th of 131.6. Capability is not the constraint;
|
|
|
289
371
|
latency is, its spread is wide, and the detached step above is what makes the
|
|
290
372
|
tier usable anyway.
|
|
291
373
|
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
374
|
+
The deadline is a bound on what an optional call may hold that server for, not a
|
|
375
|
+
figure sized to capture the tail. The shipped 30000 sits above the median and
|
|
376
|
+
below the tail deliberately, and a step that exceeds it records `timeout` with
|
|
377
|
+
its work discarded. A route whose steps mostly record `timeout` is a
|
|
378
|
+
misconfigured deadline before it is a slow model, so read the ledger before
|
|
379
|
+
raising it: `/memory` shows the hit rate the raise would be buying.
|
|
298
380
|
|
|
299
381
|
The target ID is not hard-coded. Any configured orchestrator-eligible local
|
|
300
382
|
target and wire model can fill the background role. Before enabling it, use the
|
|
@@ -316,6 +398,25 @@ For an immediate kill switch, set `memory.intervention.enabled` to `false` in
|
|
|
316
398
|
`/settings`. Removing the background target instead returns to rules-only
|
|
317
399
|
operation while leaving deterministic protection active.
|
|
318
400
|
|
|
401
|
+
## Where what the tier writes ends up
|
|
402
|
+
|
|
403
|
+
A bank entry lives and dies with its session. When a reminder actually reaches
|
|
404
|
+
the operator, the entries it cited are also proposed into the durable store at
|
|
405
|
+
`<dataDir>/memory/records.json`, unapproved, scoped to the repository the session
|
|
406
|
+
is working in, with provenance naming the session and the source entry. That is
|
|
407
|
+
the one automatic writer of that file; everything else about it is unchanged.
|
|
408
|
+
`/memory` and `clio-coder memory list` show the proposal, and
|
|
409
|
+
`clio-coder memory approve <id>` is still a separate operator action, so nothing
|
|
410
|
+
the background plane produced reaches a system prompt without review. A step with
|
|
411
|
+
no session, or one running outside a canonical repository, proposes nothing:
|
|
412
|
+
global scope broadens applicability to every future session and is not a claim a
|
|
413
|
+
background step may make on the operator's behalf.
|
|
414
|
+
|
|
415
|
+
Rules-tier reminders are not proposed. Their entries are this middleware's own
|
|
416
|
+
one-line records of a repeated tool failure, and filing each one as a durable
|
|
417
|
+
lesson would fill the review queue with rows nobody asked for. They remain
|
|
418
|
+
promotable by hand from `/memory`.
|
|
419
|
+
|
|
319
420
|
## What the LLM tier actually writes
|
|
320
421
|
|
|
321
422
|
Measured on the shipped prompt against `google/gemma-4-26b-a4b-qat`, across ten
|
|
@@ -392,11 +493,23 @@ keeps one previous generation as `steps.jsonl.1`. Every exact-schema record has:
|
|
|
392
493
|
- `silent`, `injected`, `gated`, `timeout`, `malformed`, or `dropped` decision;
|
|
393
494
|
- count of cited entries, input/output/total memory-model tokens, and latency.
|
|
394
495
|
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
496
|
+
The same steps are also billed. See "Cost and the default decision" for the
|
|
497
|
+
`/cost` row, the durable out-of-turn usage row, and the lifetime figures `/memory`
|
|
498
|
+
folds out of this file.
|
|
499
|
+
|
|
500
|
+
`dropped` is the one outcome that ran no step. It has two causes, separated by
|
|
501
|
+
the row's reason: `step_in_flight` means the boundary triggered while an earlier
|
|
502
|
+
step still held the single in-flight slot, and `endpoint_busy` means the step
|
|
503
|
+
would have called the endpoint the chat target was streaming against. Both cost
|
|
504
|
+
no tokens and no latency, both leave their triggers pending for the next free
|
|
505
|
+
boundary, and neither replaces the operator-visible last decision. Counting
|
|
506
|
+
`dropped` rows against `llm` rows over a session is how a starved cadence becomes
|
|
507
|
+
visible.
|
|
508
|
+
|
|
509
|
+
A step that exceeds the deadline records `timeout`, never `silent`: reason
|
|
510
|
+
`deadline` when the policy's own race fired first, and `timed_out` when the
|
|
511
|
+
transport aborted at its deadline. Both are distinct from `client_error`, which
|
|
512
|
+
is a route that refused rather than a route that was slow.
|
|
400
513
|
|
|
401
514
|
The log contains no task, trajectory, bank, error, or reminder text. File creation,
|
|
402
515
|
rotation, serialization, and injected sinks are all best effort; a read-only
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Prompt Envelope and Tools
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/tools_blueprint.html](html/tools_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Clio Coder keeps the model-facing envelope stable and moves enforcement into the runtime registry and safety policy.
|
|
7
7
|
|
|
@@ -13,6 +13,24 @@ The chat loop compiles one provider-facing system prompt for a session. The comp
|
|
|
13
13
|
|
|
14
14
|
The compiled prompt is reused byte-for-byte on ordinary submits. It recompiles only when that key changes or when config hot-reload invalidates the prompt cache. Path-scoped project rules can therefore recompile the prompt when a matching file enters working context. When recompilation changes the text, the session ledger records a `promptRecompiled` entry with the previous hash, new hash, and token estimate.
|
|
15
15
|
|
|
16
|
+
## Section order: stable prefix first
|
|
17
|
+
|
|
18
|
+
The compiled prompt lays its sections down in `SESSION_PROMPT_SECTION_ORDER` (`src/domains/prompts/compiler.ts`): identity, operating contract, delegation, skills, safety, tool contract, fleet, retrieval hints, project context, memory, runtime, then the operator-editable tail fragments (workspace root, Clio repo awareness, project rules, operator profile) in their own order.
|
|
19
|
+
|
|
20
|
+
One rule fixes that list. A section goes as late as its volatility, and anything that reads a clock, a probe, or a mutable store goes after everything that does not. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so a section that can change between two turns must not sit ahead of sections that cannot. The runtime block is last of the compiled sections because its `Context window: N` moves when the backend reloads a model or a co-residency clamp lands; memory sits just ahead of it because an approved memory record rewrites that section mid-session; project rules are dead last because path-scoped rules join the prompt when a matching file enters working context.
|
|
21
|
+
|
|
22
|
+
`Context window: N` is the window the backend will actually serve. A recorded loaded window outranks a probe, which reports a figure the target advertises without saying it is what is open, so a resumed session states the window its ledger measured rather than a re-probed server-wide number. Each prompt-manifest record carries that window and the layer that answered it (`contextWindow`, `contextWindowSource`) alongside a `version` for the prompt layout itself, so a recompile whose only cause was the window moving is explained by the record rather than inferred.
|
|
23
|
+
|
|
24
|
+
`PROMPT_MANIFEST_VERSION` (`src/domains/session/prompt-manifest.ts`) is `2` as of this release, and the reordering above is what moved it. The field is additive: a record written by 0.3.8 carries no `version` and reads back as version 1, so a `prompt-manifest.jsonl` from an older session still parses. The rule for the field is that it tracks the layout rather than the inputs. Bump it when the compiled text moves for a reason other than a changed fragment, a changed tool surface, or a changed setting, so that a resumed session has the version in hand to explain the single `promptRecompiled` entry its first compile writes.
|
|
25
|
+
|
|
26
|
+
### What not to add to the prefix
|
|
27
|
+
|
|
28
|
+
Two additions look free and are not.
|
|
29
|
+
|
|
30
|
+
The first is a terseness rule. It is tempting to cap the prose a model emits between tool calls, because that text is generated tokens on every hop of a long turn. Anthropic measured that exact change on Claude Code and reported a 3 percent quality regression, so a word-count or verbosity limit on inter-tool text is a bad trade: the tokens it saves are the cheapest ones in the turn, and the model's own narration of what it is about to do is load-bearing for what it then does. Bound tool results instead, where a single `grep` can cost thousands of tokens and the envelope caps already do the work.
|
|
31
|
+
|
|
32
|
+
The second is anything that varies with the wall clock or the working tree. No timestamp, no `git status`, no branch name, no session id, and no run id belongs anywhere in the compiled prefix. Every backend Clio targets caches by exact prefix and re-prefills from the earliest changed byte, so one such field turns the whole prompt into a cache miss on every turn for no information the model could not have asked a tool for. On the sprint's measurement server that is a whole 2,778-token prompt re-prefilled at 2.6 s where the same change behind the stable sections cost 516 tokens and 0.72 s. Volatile facts belong in the user message, in a tool result, or in the runtime block, which is last for this reason.
|
|
33
|
+
|
|
16
34
|
The disk fragments under `src/domains/prompts/fragments/` are layered by who reads them. `identity.clio` and `operating.contract` are constitutional: they render for every reader, name no tool, and state what is always true about Clio and her harness. `operating.delegation` (fleet coordination, receipts, spot-checks, shared `[worker result]` notes) renders only when `dispatch` is on the session's tool surface, and `operating.skills` (skill-shaped tasks, `/skill <name>` suggestions) only when `context` is; a fragment that teaches a tool is absent when the tool is, the same rule the Fleet block follows. `identity.docs-routing`, the directive to call `context(scope="docs")` before answering a question about Clio herself, follows the `context` gate too, while `identity.self-awareness` (installed paths, code outranks docs, configuration locations) names no tool and is unconditional. `operating.worker` (the assigned-task contract) renders only for dispatched workers, which never see the coordinator fragments. `safety.<level>` states what runs, what is approval-required, and what is blocked at the effective autonomy, in the safety net's action-class vocabulary (read, write, command, `system_modify`, `git_destructive`) and never by tool name, so the same body is true on every surface; the session and every worker read that one body, and what "approval-required" resolves to is the only role text (one operator confirmation for the session, the worker's `onPermission` routing for a worker).
|
|
17
35
|
|
|
18
36
|
Prompt extensions can add dynamic fragments for project rules, the operator profile, and Clio source-tree awareness. Pending skill requests and middleware reminders are visible text in the user message, not hidden prompt machinery.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Provider Adapter Cookbook
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive runtime adapter descriptor builder and probe sequence capability checklist is located at [docs/html/provider_adapter_blueprint.html](html/provider_adapter_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
This cookbook guides developers through implementing custom model runtimes and inference server integrations within Clio Coder. It explains the runtime descriptor interfaces, probing protocols, model synthesis, and how to configure reasoning and thinking behaviors.
|
|
7
7
|
|
|
@@ -39,8 +39,14 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
39
39
|
7. `npm run ci` (runs 1 through 6)
|
|
40
40
|
8. `npm run ci:release` (7 plus `scripts/check-release.mjs`: dist shebang
|
|
41
41
|
integrity, version coherence between `package.json` and the top
|
|
42
|
-
`CHANGELOG.md` heading, the
|
|
43
|
-
|
|
42
|
+
`CHANGELOG.md` heading, the deterministic 26-scenario behavioral machinery
|
|
43
|
+
corpus against its checked baseline, the forbidden-file list, the required
|
|
44
|
+
runtime resources, and the tarball and unpacked size budgets). A baseline
|
|
45
|
+
mismatch prints reviewable evidence and names prompt- or recipe-affected
|
|
46
|
+
corpus results. For an intentional change, inspect that diff, run
|
|
47
|
+
`node benchmarks/eval/check-behavioral-release.mjs --update` (with `TMPDIR` on a disk-backed path if `/tmp` is a small tmpfs),
|
|
48
|
+
review `benchmarks/eval/behavioral-machinery-baseline.json`, and commit it
|
|
49
|
+
with the change.
|
|
44
50
|
9. Optional: step 8 again under Node 24. Hosted CI gates on Node 22 alone,
|
|
45
51
|
the `engines` floor; the weekly `flake-hunt` workflow carries Node 24.
|
|
46
52
|
Repeat locally only when the cut touches runtime-sensitive code.
|
|
@@ -58,7 +64,17 @@ Run against the exact final candidate with `NO_COLOR` unset and
|
|
|
58
64
|
12. Install that tarball into a clean temporary prefix with an empty
|
|
59
65
|
`CLIO_CODER_HOME` and verify `--version`, `--help`, an empty-state non-TTY
|
|
60
66
|
launch, `doctor`, and `uninstall --dry-run` without developer-local state.
|
|
61
|
-
13.
|
|
67
|
+
13. Before interactive release testing, run the model-required public
|
|
68
|
+
behavioral corpus manually against the release target and built CLI:
|
|
69
|
+
`node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model.yaml --target mini --clio-coder-entry dist/cli/index.js`
|
|
70
|
+
and
|
|
71
|
+
`node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model-negative-control.yaml --target mini --clio-coder-entry dist/cli/index.js`.
|
|
72
|
+
Retain both Artifact v4 files as release evidence. The positive corpus must
|
|
73
|
+
report its scenario and role rows without an undeclared envelope mismatch;
|
|
74
|
+
the negative control must still record violated exploration and safety
|
|
75
|
+
labels. These model-dependent runs are manual and are never required by
|
|
76
|
+
ordinary deterministic CI. Continue with interactive release testing,
|
|
77
|
+
which this cut added because the release is
|
|
62
78
|
almost entirely interactive surface: a tester agent drives the step-12
|
|
63
79
|
install through real TUI sessions in a throwaway repository, one session
|
|
64
80
|
per shipped feature, against local targets for the main session and a
|
package/docs/safety-model.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio Coder Safety Model
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/safety_blueprint.html](html/safety_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Clio Coder's safety posture is code-enforced, not prompt-only. As the orchestrator coding agent in the [IOWarp](https://iowarp.ai) ecosystem developed by the [Gnosis Research Center](https://grc.iit.edu) at Illinois Tech under NSF Award [#2411318](https://www.nsf.gov/awardsearch/showAward?AWD_ID=2411318), Clio gates execution by target capabilities, the tool registry, the safety policy engine, project policies, protected-artifact checks, and audit receipts.
|
|
7
7
|
|
|
@@ -13,7 +13,7 @@ Source of truth: `src/domains/safety/**`, `src/tools/registry.ts`, `src/tools/bo
|
|
|
13
13
|
|
|
14
14
|
The `autonomy` setting (`read-only` | `suggest` | `auto-edit` | `full-auto`) is an enforced dial. It controls exactly one thing: which action classes run immediately, which park for operator approval, and which are auto-denied. The safety net (damage-control rules, path policy, protected artifacts, loop guard, dispatch scope admission) is independent of the dial and identical at every level. When a `[safety-net]` notice appears at full-auto, that is the always-on net working as designed, not a contradiction of the level.
|
|
15
15
|
|
|
16
|
-
In Clio Coder v0.
|
|
16
|
+
In Clio Coder v0.4.0, effective autonomy resolution is strictly centralized in `src/entry/orchestrator.ts` through `resolveEffectiveAutonomy` and `resolveBaselineAutonomy`. Every admission surface (tool registry admission, dispatch plan provenance, and ACP session snapshots) delegates to this pair of functions so that fallback paths cannot diverge across execution contexts. `resolveBaselineAutonomy` evaluates dispatch settings overrides, headless CLI options, and configuration settings before applying the default `auto-edit` level. `resolveEffectiveAutonomy` combines any active ACP session autonomy level with the baseline resolution.
|
|
17
17
|
|
|
18
18
|
### Autonomy levels
|
|
19
19
|
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
# Clio Coder Scientific Validation Contracts
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive numerical tolerance calculator and HPC queue execution simulator is located at [docs/html/validation_blueprint.html](html/validation_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Scientific software development cannot treat simple file presence as proof of correctness. A simulation script that crashes on rank 48, or writes out NetCDF arrays filled with `NaN`s, may still successfully write a file to the disk.
|
|
7
7
|
|
|
8
|
-
Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.
|
|
8
|
+
Clio Coder recognizes **scientific validation contract files** as an opt-in signal for a higher evidence bar. In v0.4.0, the session rigor resolver does not parse or enforce a scientific contract schema. The presence of `.clio-coder/validation.yaml`, `.clio-coder/validation.yml`, `validation.yaml`, `validation.yml`, or `VALIDATION.md` at the workspace root raises the default rigor level to `high`; the file contents are advisory material for developers, project agents, and external validators.
|
|
9
9
|
|
|
10
10
|
This advisory convention is separate from the executable project verifier catalog at `.clio-coder/verifiers.yaml`. The verifier catalog has a strict version-1 schema and admits exact argv vectors to the `verify` tool. Scientific validation contracts and handbook expectations do not grant command authority: prose such as `validators: ["python tools/check_grid.py"]` remains guidance until the project owner confirms the equivalent argv, cwd, timeout, and tags in `verifiers.yaml`. The executable catalog does not interpret numerical tolerances or artifact expectations; it only runs the explicitly declared process vector through safe-exec.
|
|
11
11
|
|
|
@@ -95,7 +95,7 @@ Comparing floating-point values in scientific computations must accommodate roun
|
|
|
95
95
|
|
|
96
96
|
## Common Scientific Artifact Families
|
|
97
97
|
|
|
98
|
-
The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.
|
|
98
|
+
The following labels are useful project conventions for validation contracts and reports. They are not a closed, core-enforced enum in v0.4.0:
|
|
99
99
|
|
|
100
100
|
- **`HDF5` / `NetCDF` / `Zarr`:** Multi-dimensional scientific array files.
|
|
101
101
|
- **`FITS`:** Flexible Image Transport System (used in astrophysics).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Session Lifecycle
|
|
2
2
|
|
|
3
|
-
This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.
|
|
3
|
+
This document is the authoritative specification for Clio Coder interactive and headless session lifecycles, on-disk ledger structures, tree-based conversation branching, checkpoints, and recovery protocols in `v0.4.0`.
|
|
4
4
|
|
|
5
5
|
Source implementations: `src/engine/session.ts` and `src/domains/session/`.
|
|
6
6
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Skills Marketplace
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive dashboard is located at [docs/html/skills_blueprint.html](html/skills_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
The Skills Hub (`/skill`) shows project skills, user skills, and the marketplace. Every marketplace row comes from the same local lookup that `clio-coder skills install <name>` and `/skill <name>` resolve through, so the hub lists nothing it cannot install.
|
|
7
7
|
|
package/docs/tool-usage.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
# Tool Usage Reference
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive seven-plane tool atlas and observation envelope truncation/offload calculator is located at [docs/html/tool_usage_blueprint.html](html/tool_usage_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
This is the deep usage reference behind the deliberately terse tool descriptions in the prompt envelope. Toolkit v2 keeps rich guidance out of tool descriptions and puts it here, where `context(scope="docs", query=...)` retrieves it section by section. Each tool below has its own self-contained `##` section covering the argument surface, defaults, truncation and continuation behavior, and concrete calls. Source of truth is `src/tools/`.
|
|
7
7
|
|
|
8
|
-
In Clio Coder v0.
|
|
8
|
+
In Clio Coder v0.4.0, `src/tools/agent-tools.ts` serves as the single agent-tool adapter across both orchestrator and worker runtimes. Both surfaces resolve their executable tools through the exact same `effectiveToolNames` narrowing, ensuring that attested tool schemas never drift from the tools available at runtime. Tools are keyed strictly by the `ToolName` union with no alias table. Argument leniency for weak-model callers is provided exclusively by per-tool `prepareArguments` normalizers declared on `ToolSpec`.
|
|
9
9
|
|
|
10
10
|
## Observation envelope: truncation notices, offload, next hints, and the turn budget
|
|
11
11
|
|
|
@@ -253,7 +253,7 @@ Arguments:
|
|
|
253
253
|
- `cwd` (optional). Default agent working directory.
|
|
254
254
|
- `timeout_ms` (optional). Aborts the whole dispatch; in sequential mode remaining tasks are skipped and the skip is reported.
|
|
255
255
|
- `briefing` (optional string, top-level default or per-task override). Parent-composed context/data, not worker instructions: it cannot replace `task`. It is trimmed and omitted when blank, rejected above 12,000 UTF-8 bytes, sent as its own delimited untrusted dynamic message, and retained only as byte/hash provenance. The shared value applies to string tasks and object tasks without an override; an object-level briefing wins.
|
|
256
|
-
- `intent` (
|
|
256
|
+
- `intent` (object, top-level default or per-task override, and the default way to dispatch). Declares `read_roots`, `write_roots`, `relevant_paths`, `expected_outputs`, and `verification`. Path arrays contain normalized repository-relative POSIX paths. Verification entries contain a declared `check` id and optional `timeout_ms`; ids are resolved from package scripts and `.clio-coder/verifiers.yaml` before approval. Checks are ids, not shell commands. Declared paths select the project rules that apply to them and pin worker context; omitting `intent` falls back to reading path-like tokens out of the task and briefing, which can miss an applicable rule. In a batch, per-task `intent` shallow-merges over the top-level object and is then checked against it as a ceiling: a task may narrow the shared scope and is refused with `intent_scope_widening` if it reaches outside. Declaring `write_roots` that disagree with a legacy `writeRoots`, an `expected_outputs` entry outside every declared write root, or an `intent.version` other than 2 are each terminal refusals carrying a stable reason code. See [dispatch-typed-intent.md](dispatch-typed-intent.md).
|
|
257
257
|
- `gate` (optional string, top-level default or per-task override). Exact shorthand for `intent.verification=[{check: gate}]`. Supplying it together with `intent.verification` is refused.
|
|
258
258
|
- `max_output_bytes` (optional). Summary byte budget; default 20000, split across runs with at least 1024 bytes each.
|
|
259
259
|
|
|
@@ -269,13 +269,23 @@ Sealed receipts are the durable evidence; worker prose remains advisory until ve
|
|
|
269
269
|
|
|
270
270
|
```text
|
|
271
271
|
dispatch(list=true)
|
|
272
|
-
dispatch(agent="debugger", task="Adversarially verify the strict v19 receipt boundary", briefing="Prior receipt R1 cited receipt-integrity.ts and left these claims unresolved", detach=true)
|
|
273
|
-
dispatch(tasks=["Run the contract tests in tests/contracts/dispatch.test.ts and report each failure with its assertion"])
|
|
272
|
+
dispatch(agent="debugger", task="Adversarially verify the strict v19 receipt boundary", briefing="Prior receipt R1 cited receipt-integrity.ts and left these claims unresolved", intent={read_roots: ["src/domains/dispatch/"]}, detach=true)
|
|
274
273
|
dispatch(tasks=[
|
|
275
|
-
{
|
|
276
|
-
|
|
274
|
+
{task: "Run the contract tests in tests/contracts/dispatch.test.ts and report each failure with its assertion",
|
|
275
|
+
intent: {read_roots: ["tests/contracts/", "src/domains/dispatch/"], verification: [{check: "test"}]}}
|
|
276
|
+
])
|
|
277
|
+
dispatch(tasks=[
|
|
278
|
+
{agent: "researcher", task: "Map every caller of finalizeObservation and summarize the envelope shapes",
|
|
279
|
+
intent: {read_roots: ["src/domains/"]}},
|
|
280
|
+
{agent: "coder", task: "Fix the failing assertion in tests/contracts/safety.test.ts",
|
|
281
|
+
intent: {write_roots: ["tests/contracts/"], expected_outputs: ["tests/contracts/safety.test.ts"], verification: [{check: "test"}]}}
|
|
277
282
|
], mode="parallel")
|
|
278
|
-
dispatch(
|
|
283
|
+
dispatch(
|
|
284
|
+
intent={read_roots: ["src/"], write_roots: ["src/domains/"]},
|
|
285
|
+
tasks=[
|
|
286
|
+
{task: "Refactor step 1", intent: {write_roots: ["src/domains/dispatch/"]}},
|
|
287
|
+
{task: "Refactor step 2", intent: {write_roots: ["src/domains/context/"]}}
|
|
288
|
+
], mode="sequential", timeout_ms=600000)
|
|
279
289
|
```
|
|
280
290
|
|
|
281
291
|
## verify: run declared verification checks
|
package/docs/trace-store.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Trace store contract
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.
|
|
4
|
+
> **Interactive Spec Available:** An interactive trace database viewer, schema inspector, and SQL query validator simulator is located at [docs/html/trace_blueprint.html](html/trace_blueprint.html) (Version: 0.4.0).
|
|
5
5
|
|
|
6
6
|
Clio's trace database is a rebuildable, queryable mirror. Receipts, session
|
|
7
7
|
ledgers, gate artifacts, and evidence remain the source of truth. Removing
|