@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -16,6 +16,30 @@
|
|
|
16
16
|
# models curated for Clio's local coding workflows and leaves cloud GPT models
|
|
17
17
|
# to pi-ai's native OpenAI and openai-codex catalogs.
|
|
18
18
|
#
|
|
19
|
+
# Serving configuration is a quality variable, not a constant. Context length,
|
|
20
|
+
# KV quantization, batch and ubatch size, speculative decoding, and the server's
|
|
21
|
+
# own sampler defaults all move what a model does, so a quirk measured under one
|
|
22
|
+
# configuration is not a property of the weights. Every family entry whose
|
|
23
|
+
# quirks came from a measurement rather than from the model card carries a
|
|
24
|
+
# `measuredUnder` block naming the hardware class, the runtime and its build
|
|
25
|
+
# string, and the llama.cpp flags the measurement ran under. Fields that could
|
|
26
|
+
# not be established from git history or the release-test reports under
|
|
27
|
+
# docs/release-notes/ read `unknown` rather than a plausible guess. A
|
|
28
|
+
# `measuredUnder` block is provenance for a reader; nothing in the engine
|
|
29
|
+
# consumes it (`extractLocalModelQuirks` narrows only kvCache, sampling, and
|
|
30
|
+
# thinking).
|
|
31
|
+
#
|
|
32
|
+
# `llamaCpp.parallel` is the recommended `--parallel` value for starting the
|
|
33
|
+
# server, and families that carry one also carry a `parallelSlots` note. It is
|
|
34
|
+
# not what dispatch admission counts. Endpoint capacity resolves a target's
|
|
35
|
+
# request-slot limit in this order: an explicit `maxConcurrentRequests` on the
|
|
36
|
+
# target, then `parallelSlots` discovered from the server (llama.cpp reads
|
|
37
|
+
# `total_slots` from `/props`, falling back to the selected worker's
|
|
38
|
+
# `/props?model=<id>` and then to its `--parallel` argv), then one slot for any
|
|
39
|
+
# other local-native runtime. vLLM and SGLang stay unbounded without an explicit
|
|
40
|
+
# override. The catalog value therefore says what to start the server with; the
|
|
41
|
+
# server's own answer is what Clio admits against.
|
|
42
|
+
#
|
|
19
43
|
# Thinking semantics for these local families: the chain-of-thought is emitted
|
|
20
44
|
# by the model's chat template (Qwen-style <think> blocks, Gemma 4 thinking
|
|
21
45
|
# template, Nemotron reasoning template). LM Studio's OpenAI-compatible HTTP
|
|
@@ -89,6 +113,18 @@
|
|
|
89
113
|
contextWindow: 262144
|
|
90
114
|
maxTokens: 65536
|
|
91
115
|
quirks:
|
|
116
|
+
measuredUnder:
|
|
117
|
+
hardware: "32 GiB class GPU; the exact card was not recorded"
|
|
118
|
+
runtime: llamacpp
|
|
119
|
+
build: unknown
|
|
120
|
+
llamaCpp: "ctx 262144, --parallel 4, --batch-size 2048, --ubatch-size 512, KV q8_0/q8_0 (the block below); no argv was recorded"
|
|
121
|
+
date: unknown
|
|
122
|
+
source: unknown
|
|
123
|
+
note: |
|
|
124
|
+
Only the VRAM-headroom figure in the 32gb tier reads as measured. It
|
|
125
|
+
entered the catalog with the initial local-model set and no commit,
|
|
126
|
+
report, or server argv records where it was taken, so everything but
|
|
127
|
+
the recommended serving profile is unknown.
|
|
92
128
|
sampling:
|
|
93
129
|
thinking:
|
|
94
130
|
temperature: 0.5
|
|
@@ -111,6 +147,7 @@
|
|
|
111
147
|
cacheTypeK: q8_0
|
|
112
148
|
cacheTypeV: q8_0
|
|
113
149
|
parallel: 4
|
|
150
|
+
parallelSlots: "Start the server with --parallel 4. Clio discovers total_slots 4 from /props and admits four concurrent requests on this endpoint, one of which the orchestrator's own turn holds while it streams. Add --kv-unified, or the 262144-token context is divided four ways."
|
|
114
151
|
batchSize: 2048
|
|
115
152
|
ubatchSize: 512
|
|
116
153
|
thinking:
|
|
@@ -174,6 +211,7 @@
|
|
|
174
211
|
flashAttn: true
|
|
175
212
|
nGpuLayers: 99
|
|
176
213
|
parallel: 4
|
|
214
|
+
parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 204800 tokens rather than the whole 819200. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
|
|
177
215
|
thinking:
|
|
178
216
|
mechanism: budget-tokens
|
|
179
217
|
budgetByLevel:
|
|
@@ -267,6 +305,19 @@
|
|
|
267
305
|
contextWindow: 262144
|
|
268
306
|
maxTokens: 65536
|
|
269
307
|
quirks:
|
|
308
|
+
measuredUnder:
|
|
309
|
+
hardware: unknown
|
|
310
|
+
runtime: lmstudio
|
|
311
|
+
build: "unknown; the host was the operator's `dynamo` LM Studio machine, whose bundled build string was not recorded on this date"
|
|
312
|
+
llamaCpp: "unknown; the measurement went through LM Studio's OpenAI-compatible port, which does not expose the underlying server argv"
|
|
313
|
+
date: "2026-08-08"
|
|
314
|
+
source: "commit b3db3b82 (fix(providers): thinking off reaches the wire for models that reason by default)"
|
|
315
|
+
note: |
|
|
316
|
+
The always-on classification is the measured part. That session found
|
|
317
|
+
Ornith reasoning by default with no thinking field set, in the same
|
|
318
|
+
family of behavior as qwopus3.6-35b-a3b-coder. The recommended
|
|
319
|
+
llama.cpp serving profile below is the model card's, not that
|
|
320
|
+
measurement's, and no argv from the measuring host survives.
|
|
270
321
|
kvCache:
|
|
271
322
|
kQuant: q8_0
|
|
272
323
|
vQuant: q8_0
|
|
@@ -295,6 +346,7 @@
|
|
|
295
346
|
flashAttn: true
|
|
296
347
|
nGpuLayers: 99
|
|
297
348
|
parallel: 1
|
|
349
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator is streaming is refused with the endpoint denial rather than queued; point workers at a second server, or raise --parallel and re-probe."
|
|
298
350
|
batchSize: 2048
|
|
299
351
|
ubatchSize: 512
|
|
300
352
|
thinking:
|
|
@@ -326,6 +378,20 @@
|
|
|
326
378
|
contextWindow: 262144
|
|
327
379
|
maxTokens: 32768
|
|
328
380
|
quirks:
|
|
381
|
+
measuredUnder:
|
|
382
|
+
hardware: unknown
|
|
383
|
+
runtime: "lmstudio and its OpenAI-compatible port"
|
|
384
|
+
build: unknown
|
|
385
|
+
llamaCpp: "unknown; this family carries no llamaCpp block and the parity check ran over LM Studio's HTTP surface"
|
|
386
|
+
date: unknown
|
|
387
|
+
source: unknown
|
|
388
|
+
note: |
|
|
389
|
+
The measured claim is the openaiCompat parity line: tool calls, a
|
|
390
|
+
reasoning_content field, and completion_tokens_details.reasoning_tokens
|
|
391
|
+
were all observed on the wire. It entered the catalog with the initial
|
|
392
|
+
local-model set, so the host, the LM Studio version, and the loaded
|
|
393
|
+
quantization are all unrecorded. The sampler and the 32gb VRAM
|
|
394
|
+
arithmetic come from the official Qwen card rather than a run.
|
|
329
395
|
sampling:
|
|
330
396
|
thinking:
|
|
331
397
|
temperature: 0.6
|
|
@@ -380,6 +446,28 @@
|
|
|
380
446
|
contextWindow: 262144
|
|
381
447
|
maxTokens: 131072
|
|
382
448
|
quirks:
|
|
449
|
+
measuredUnder:
|
|
450
|
+
hardware: "AMD Radeon AI PRO R9700, gfx1201, 32 GB, ROCm 7.2.0"
|
|
451
|
+
runtime: llamacpp
|
|
452
|
+
build: b226-2115b73d8
|
|
453
|
+
model: "Qwen3.8-27B IQ4_NL, mmproj F16"
|
|
454
|
+
llamaCpp: >-
|
|
455
|
+
--ctx-size 262144 --parallel 1 --kv-unified --cache-type-k q8_0
|
|
456
|
+
--cache-type-v q8_0 --flash-attn true --batch-size 2048 --ubatch-size 512
|
|
457
|
+
--spec-type draft-mtp --spec-draft-n-max 4 --n-gpu-layers-draft 99
|
|
458
|
+
--swa-checkpoints 64 --checkpoint-min-step 1024 --jinja --reasoning on
|
|
459
|
+
--reasoning-effort high --temperature 1.0 --top-k 20 --top-p 0.95
|
|
460
|
+
--min-p 0.0 --presence-penalty 0.0 --repeat-penalty 1.0
|
|
461
|
+
date: "2026-08-29"
|
|
462
|
+
source: docs/release-notes/v0.3.9-local-economics.md
|
|
463
|
+
alsoMeasuredOn: "LM Studio 2.29.0 (OpenAI-compatible HTTP port), host build string llama.cpp-win-x86_64-nvidia-cuda12-avx2"
|
|
464
|
+
note: |
|
|
465
|
+
The server above is deliberately not the 24gb-rocm tier below: it runs
|
|
466
|
+
ctx=262144 and parallel=1 where that tier recommends ctx=131072 and
|
|
467
|
+
parallel=2. The reasoning_effort vocabulary and the enable_thinking
|
|
468
|
+
control were established earlier, on 2026-08-18, from a llama.cpp host
|
|
469
|
+
and an LM Studio host whose build strings were not recorded; treat those
|
|
470
|
+
two findings as unknown-build.
|
|
383
471
|
kvCache:
|
|
384
472
|
kQuant: q8_0
|
|
385
473
|
vQuant: q8_0
|
|
@@ -415,6 +503,7 @@
|
|
|
415
503
|
flashAttn: true
|
|
416
504
|
nGpuLayers: 99
|
|
417
505
|
parallel: 2
|
|
506
|
+
parallelSlots: "Start the server with --parallel 2 to serve the orchestrator and one worker at once. Clio then discovers total_slots from /props and admits two concurrent requests against this endpoint; a server started with --parallel 1 discovers one slot, and a dispatch during the orchestrator's own turn is refused rather than queued."
|
|
418
507
|
batchSize: 1024
|
|
419
508
|
ubatchSize: 256
|
|
420
509
|
mmproj: true
|
|
@@ -443,7 +532,26 @@
|
|
|
443
532
|
and is a separate mechanism from the template's enable_thinking flag;
|
|
444
533
|
both were verified independently on llama.cpp and on LM Studio's HTTP
|
|
445
534
|
port.
|
|
446
|
-
|
|
535
|
+
|
|
536
|
+
Thinking off is verified for this family, and no floor is imposed on
|
|
537
|
+
it. JetBrains report that Qwen3.8 without reasoning gets stuck in a
|
|
538
|
+
loop repeating the same tool call indefinitely under Junie. Clio's
|
|
539
|
+
shipped default is orchestrator.thinkingLevel: off, which on llama.cpp
|
|
540
|
+
sends chat_template_kwargs.enable_thinking:false and on LM Studio's
|
|
541
|
+
OpenAI-compatible port sends reasoning_effort:"none". Clio 0.3.8 ran
|
|
542
|
+
that configuration on both runtimes (llama.cpp build b226-2115b73d8 on
|
|
543
|
+
ROCm, LM Studio 2.29.0) through the v0.3.9 local-inference campaign on
|
|
544
|
+
2026-08-29 and 2026-08-30: an interactive build-and-resume session, a
|
|
545
|
+
three-worker fleet step, and a ten-task eval suite totalling 116 model
|
|
546
|
+
calls, none of which repeated a tool call into a loop. The catalog
|
|
547
|
+
therefore imposes no thinking floor and coerces no level for this
|
|
548
|
+
family. If a future build or quantization regresses into the loop
|
|
549
|
+
JetBrains describe, the guards that catch it are Clio's own, not a
|
|
550
|
+
catalog setting: the tool-prose-loop detector
|
|
551
|
+
(src/interactive/tool-prose-loop.ts), which is armed on the
|
|
552
|
+
local-native tier, and the stalled-turn middleware
|
|
553
|
+
(src/domains/middleware/stalled-turn.ts).
|
|
554
|
+
serving: "Qwen3.8-27B (unsloth/Qwen3.8-27B-GGUF), hybrid qwen3_5 architecture: 3 linear-attention (Gated DeltaNet) layers per 1 full-attention layer, so only 16 of 64 layers hold a conventional KV cache. Native ctx 262144, extensible to 1M via YaRN (out of scope for the hardware classes above; leave rope at default). Tool calls use an XML <tool_call><function=...> form in the stock template; llama.cpp's qwen3_coder parser converts this to standard tool_calls JSON, verified live on ROCm, CUDA, and Vulkan backends. Only messages[0] may be a system message and only system/user/assistant/tool roles are accepted; a second system message or a developer role raises a template error. The recurrent layers set the prefix-cache economics: llama.cpp cannot roll gated-deltanet state back to an arbitrary token, so a divergence anywhere inside a cached prefix re-prefills from the last context checkpoint rather than from the changed byte. Checkpoints are written at the end of each processed prompt and only when at least --checkpoint-min-step tokens separate them, so the server's checkpoint count and --checkpoint-min-step are a quality variable on this family and not a tuning detail: measured on the configuration recorded in measuredUnder, one changed line inside an earlier turn of a four-turn conversation cost cache_n 9 and 17.9 s at the build defaults (32 checkpoints, min step 8192) and cache_n 7570 and 10.9 s at 64 checkpoints and min step 1024. A single prefill writes one checkpoint at its end, so a 16.7K-token prompt re-sent with one changed line re-prefills from zero (18.4 s) under both settings, while changing only the final user message re-prefills the last ubatch (prompt_n 517, 1.4 s). Serving this family with --parallel greater than 1 divides --ctx-size across slots unless --kv-unified is set."
|
|
447
555
|
|
|
448
556
|
- family: gemma-4-31b-it-nvfp4-turbo
|
|
449
557
|
matchPatterns:
|
|
@@ -467,6 +575,21 @@
|
|
|
467
575
|
contextWindow: 122880
|
|
468
576
|
maxTokens: 32768
|
|
469
577
|
quirks:
|
|
578
|
+
measuredUnder:
|
|
579
|
+
hardware: unknown
|
|
580
|
+
runtime: "lmstudio and its OpenAI-compatible port"
|
|
581
|
+
build: unknown
|
|
582
|
+
llamaCpp: "unknown; this family carries no llamaCpp block, and the channel-marker leakage was observed through LM Studio's stream"
|
|
583
|
+
date: unknown
|
|
584
|
+
source: unknown
|
|
585
|
+
note: |
|
|
586
|
+
Two claims here are observations rather than card values: the
|
|
587
|
+
channel-marker leakage in leakageNote, and the instruct temperature
|
|
588
|
+
pinned at 0.7 because NVFP4 quantization corrupts tool-call JSON under
|
|
589
|
+
the card's 1.0. Neither records the host, the LM Studio version, or the
|
|
590
|
+
exact NVFP4 build it was seen on. The KV-budget arithmetic in the
|
|
591
|
+
gpuTiers and the serving note is derived from the official Google
|
|
592
|
+
Gemma 4 31B card.
|
|
470
593
|
kvCache:
|
|
471
594
|
kQuant: q8_0
|
|
472
595
|
vQuant: q8_0
|
|
@@ -567,6 +690,7 @@
|
|
|
567
690
|
flashAttn: true
|
|
568
691
|
nGpuLayers: 99
|
|
569
692
|
parallel: 1
|
|
693
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised during the orchestrator's turn is refused rather than queued."
|
|
570
694
|
batchSize: 1024
|
|
571
695
|
ubatchSize: 256
|
|
572
696
|
mmproj: true
|
|
@@ -621,6 +745,19 @@
|
|
|
621
745
|
minP: 0.0
|
|
622
746
|
maxTokens: 32768
|
|
623
747
|
chatTemplate: "gemma-channel"
|
|
748
|
+
measuredUnder:
|
|
749
|
+
hardware: unknown
|
|
750
|
+
runtime: "lmstudio and its OpenAI-compatible port"
|
|
751
|
+
build: unknown
|
|
752
|
+
llamaCpp: "unknown; this family carries no llamaCpp block"
|
|
753
|
+
date: unknown
|
|
754
|
+
source: unknown
|
|
755
|
+
note: |
|
|
756
|
+
The measured claims are the empty thought region the model emits when
|
|
757
|
+
the '<|think|>' token is absent from the system prompt, and the absence
|
|
758
|
+
of reasoning_content from the openai-compat reasoning probe. Both
|
|
759
|
+
entered the catalog with the initial local-model set, with no host, LM
|
|
760
|
+
Studio version, or quantization recorded.
|
|
624
761
|
thinkingControl: |
|
|
625
762
|
Jackrong's distilled gemma-4 31B uses the same channel-marker template.
|
|
626
763
|
Thinking mode is gated by the literal '<|think|>' token in the system
|
|
@@ -661,6 +798,18 @@
|
|
|
661
798
|
contextWindow: 262144
|
|
662
799
|
maxTokens: 32768
|
|
663
800
|
quirks:
|
|
801
|
+
measuredUnder:
|
|
802
|
+
hardware: unknown
|
|
803
|
+
runtime: "lmstudio and its OpenAI-compatible port"
|
|
804
|
+
build: unknown
|
|
805
|
+
llamaCpp: "unknown; this family carries no llamaCpp block"
|
|
806
|
+
date: unknown
|
|
807
|
+
source: unknown
|
|
808
|
+
note: |
|
|
809
|
+
The measured claim is the openaiCompat tool-call extraction parity and
|
|
810
|
+
the reasoning_content field surfacing. It entered the catalog with the
|
|
811
|
+
initial local-model set and records no host, LM Studio version, or
|
|
812
|
+
quantization. The sampler comes from the model card.
|
|
664
813
|
sampling:
|
|
665
814
|
thinking:
|
|
666
815
|
temperature: 0.6
|
|
@@ -732,6 +881,26 @@
|
|
|
732
881
|
contextWindow: 262144
|
|
733
882
|
maxTokens: 32768
|
|
734
883
|
quirks:
|
|
884
|
+
measuredUnder:
|
|
885
|
+
hardware: unknown
|
|
886
|
+
runtime: "lmstudio and its OpenAI-compatible port for the reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
|
|
887
|
+
build: "unknown for both dates; the operator's `dynamo` host is a Windows LM Studio machine whose bundled build string was first recorded on 2026-08-29 as llama.cpp-win-x86_64-nvidia-cuda12-avx2, which is three weeks after the reasoning measurement and cannot be backdated to it"
|
|
888
|
+
llamaCpp: "unknown; the reasoning measurement went through LM Studio's HTTP port, which does not expose the underlying server argv, and no argv survives for the dispatch that produced the presence-penalty value"
|
|
889
|
+
date: "2026-08-08 (reasoning_effort and enable_thinking), 2026-07-06 (presence penalty)"
|
|
890
|
+
source: "commits b3db3b82 and f7d76bf4"
|
|
891
|
+
note: |
|
|
892
|
+
Three numbers here are measurements. On 2026-08-08 on `dynamo` this
|
|
893
|
+
model spent 98 of 103 completion tokens reasoning on "what is 17+25"
|
|
894
|
+
with no thinking field set; chat_template_kwargs enable_thinking was
|
|
895
|
+
inert, re-checked with unique prompts to rule out prompt-cache hits;
|
|
896
|
+
and reasoning_effort "none" took the same prompt to 0. The same session
|
|
897
|
+
recorded a wiki planning dispatch over a 1007-file repository going
|
|
898
|
+
from 89,501 reasoning tokens, 47 tool calls and exit 1 at 459 s to 0
|
|
899
|
+
reasoning tokens, 8 tool calls and exit 0 at 218 s. On 2026-07-06 a
|
|
900
|
+
live dispatch measured presence_penalty 1.5: without it a coder worker
|
|
901
|
+
repeated one code_nav call into the loop-guard abort on 3 of 3 runs,
|
|
902
|
+
and with it the same task passed 3 of 3. Neither commit records the
|
|
903
|
+
hardware or the server configuration, so both stay unknown.
|
|
735
904
|
sampling:
|
|
736
905
|
instruct:
|
|
737
906
|
temperature: 0.2
|
|
@@ -757,6 +926,7 @@
|
|
|
757
926
|
flashAttn: true
|
|
758
927
|
nGpuLayers: 99
|
|
759
928
|
parallel: 1
|
|
929
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint. This family is often the workers.default model on a router that also serves the orchestrator, and a one-slot router refuses the worker batch rather than queueing it; a second server or an explicit maxConcurrentRequests on the target is the remedy."
|
|
760
930
|
batchSize: 2048
|
|
761
931
|
ubatchSize: 512
|
|
762
932
|
mmproj: true
|
|
@@ -801,6 +971,20 @@
|
|
|
801
971
|
contextWindow: 262144
|
|
802
972
|
maxTokens: 32768
|
|
803
973
|
quirks:
|
|
974
|
+
measuredUnder:
|
|
975
|
+
hardware: unknown
|
|
976
|
+
runtime: unknown
|
|
977
|
+
build: unknown
|
|
978
|
+
llamaCpp: unknown
|
|
979
|
+
date: unknown
|
|
980
|
+
source: "commit b3db3b82, which states the exclusion directly"
|
|
981
|
+
note: |
|
|
982
|
+
Nothing in this entry was measured on this model. The reasoning:false
|
|
983
|
+
pin and the presence penalty are both carried over from the 35B-A3B
|
|
984
|
+
Coder-MTP design; the commit that reclassified the 35B on live evidence
|
|
985
|
+
says in as many words that the 27B and 9B Coder variants keep their
|
|
986
|
+
"never" pin because nothing has measured them. Read the reasoning class
|
|
987
|
+
as an untested inheritance until someone runs this model.
|
|
804
988
|
sampling:
|
|
805
989
|
instruct:
|
|
806
990
|
temperature: 0.2
|
|
@@ -841,6 +1025,19 @@
|
|
|
841
1025
|
contextWindow: 262144
|
|
842
1026
|
maxTokens: 32768
|
|
843
1027
|
quirks:
|
|
1028
|
+
measuredUnder:
|
|
1029
|
+
hardware: unknown
|
|
1030
|
+
runtime: unknown
|
|
1031
|
+
build: unknown
|
|
1032
|
+
llamaCpp: unknown
|
|
1033
|
+
date: unknown
|
|
1034
|
+
source: "commit b3db3b82, which states the exclusion directly"
|
|
1035
|
+
note: |
|
|
1036
|
+
Nothing in this entry was measured on this model. The reasoning:false
|
|
1037
|
+
pin is carried over from the Coder-MTP design; the commit that
|
|
1038
|
+
reclassified the 35B-A3B on live evidence says in as many words that
|
|
1039
|
+
the 27B and 9B Coder variants keep their "never" pin because nothing
|
|
1040
|
+
has measured them.
|
|
844
1041
|
sampling:
|
|
845
1042
|
instruct:
|
|
846
1043
|
temperature: 0.2
|
|
@@ -936,6 +1133,21 @@
|
|
|
936
1133
|
contextWindow: 1048576
|
|
937
1134
|
maxTokens: 65536
|
|
938
1135
|
quirks:
|
|
1136
|
+
measuredUnder:
|
|
1137
|
+
hardware: unknown
|
|
1138
|
+
runtime: "llamacpp; the anthropicCompat probe ran against a llama.cpp target's /v1/messages and /v1/models"
|
|
1139
|
+
build: unknown
|
|
1140
|
+
llamaCpp: "unknown; the recommended profile below was not recorded as the profile the probe ran under"
|
|
1141
|
+
date: unknown
|
|
1142
|
+
source: unknown
|
|
1143
|
+
note: |
|
|
1144
|
+
The measured claim is the anthropicCompat line: /v1/messages returned
|
|
1145
|
+
404 on the llama.cpp target profile in use while /v1/models and the
|
|
1146
|
+
OpenAI-compatible chat surface both worked. That is a fact about one
|
|
1147
|
+
server's mounted routes, so it is exactly the kind of finding a
|
|
1148
|
+
different build or launch configuration can invalidate, and no build or
|
|
1149
|
+
argv was recorded with it. Everything else in this entry comes from the
|
|
1150
|
+
NVIDIA card.
|
|
939
1151
|
sampling:
|
|
940
1152
|
instruct:
|
|
941
1153
|
temperature: 0.3
|
|
@@ -956,6 +1168,7 @@
|
|
|
956
1168
|
flashAttn: true
|
|
957
1169
|
nGpuLayers: 99
|
|
958
1170
|
parallel: 4
|
|
1171
|
+
parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 262144 tokens rather than the whole 1048576. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
|
|
959
1172
|
chatTemplateKwargs:
|
|
960
1173
|
enable_thinking: false
|
|
961
1174
|
thinking:
|
|
@@ -990,6 +1203,19 @@
|
|
|
990
1203
|
contextWindow: 262144
|
|
991
1204
|
maxTokens: 32768
|
|
992
1205
|
quirks:
|
|
1206
|
+
measuredUnder:
|
|
1207
|
+
hardware: unknown
|
|
1208
|
+
runtime: lmstudio
|
|
1209
|
+
build: unknown
|
|
1210
|
+
llamaCpp: "unknown; this family carries no llamaCpp block"
|
|
1211
|
+
date: unknown
|
|
1212
|
+
source: unknown
|
|
1213
|
+
note: |
|
|
1214
|
+
The only measured claim is the lmstudio line, that the model loads and
|
|
1215
|
+
runs usably under the Qwen-style preset. It entered the catalog with
|
|
1216
|
+
the initial local-model set and records no host, LM Studio version, or
|
|
1217
|
+
quantization. The sampler and the thinking budgets follow the Qwen3.5
|
|
1218
|
+
family defaults rather than a run on this distillation.
|
|
993
1219
|
sampling:
|
|
994
1220
|
thinking:
|
|
995
1221
|
temperature: 0.6
|
|
@@ -1041,6 +1267,22 @@
|
|
|
1041
1267
|
contextWindow: 262144
|
|
1042
1268
|
maxTokens: 32768
|
|
1043
1269
|
quirks:
|
|
1270
|
+
measuredUnder:
|
|
1271
|
+
hardware: "unknown; the host was the operator's `dynamo` LM Studio machine"
|
|
1272
|
+
runtime: "lmstudio, SDK 1.5.0 and its OpenAI-compatible port"
|
|
1273
|
+
build: "unknown; LM Studio's bundled llama.cpp build string was not recorded on this date"
|
|
1274
|
+
llamaCpp: "unknown; the measurement went through LM Studio, which does not expose the underlying server argv"
|
|
1275
|
+
date: "2026-08-11"
|
|
1276
|
+
source: "commit fbaa0492 (fix: make thinking off reach the model on LM Studio routes)"
|
|
1277
|
+
note: |
|
|
1278
|
+
The always-on classification is measured, and it is measured against
|
|
1279
|
+
every control that could have disproved it. Baseline 56 reasoning
|
|
1280
|
+
tokens; reasoning_effort none 120; reasoning_effort minimal 243;
|
|
1281
|
+
chat_template_kwargs.enable_thinking false 45; both together 44. No
|
|
1282
|
+
spelling silences it, and the budget-tokens mechanism the entry
|
|
1283
|
+
previously claimed never reached the wire on either LM Studio or
|
|
1284
|
+
llama.cpp. The gpuTiers 16gb line is a role assignment rather than a
|
|
1285
|
+
VRAM measurement.
|
|
1044
1286
|
sampling:
|
|
1045
1287
|
instruct:
|
|
1046
1288
|
temperature: 0.2
|
|
@@ -147,6 +147,13 @@ export interface ResolveRuntimeTargetInput {
|
|
|
147
147
|
requireTools?: boolean;
|
|
148
148
|
requireStreaming?: boolean;
|
|
149
149
|
requireOutputBudget?: boolean;
|
|
150
|
+
/**
|
|
151
|
+
* A loaded window this target and model were already observed serving, from
|
|
152
|
+
* a caller that remembers across processes. Used only when live discovery
|
|
153
|
+
* reports no loaded window, so a resumed session stops budgeting against a
|
|
154
|
+
* probed server-wide figure for its first turn (issue #227).
|
|
155
|
+
*/
|
|
156
|
+
knownLoadedContextWindow?: number | null;
|
|
150
157
|
}
|
|
151
158
|
|
|
152
159
|
function diagnostic(severity: RuntimeResolutionSeverity, code: string, message: string): RuntimeResolutionDiagnostic {
|
|
@@ -405,7 +412,7 @@ export function resolveRuntimeTarget(
|
|
|
405
412
|
// Discovery's per-model loaded window, which the probe capabilities cannot
|
|
406
413
|
// carry: `probeCapabilitiesForModel` answers for the target's default model
|
|
407
414
|
// and reports a window without saying whether it is the one being served.
|
|
408
|
-
const loadedContextWindow = loadedContextWindowForModel(status, wireModelId);
|
|
415
|
+
const loadedContextWindow = loadedContextWindowForModel(status, wireModelId) ?? input.knownLoadedContextWindow ?? null;
|
|
409
416
|
const contextWindowDetails = resolveContextWindowDetails(
|
|
410
417
|
target,
|
|
411
418
|
runtime,
|
|
@@ -310,6 +310,7 @@ interface LlamaCppProps {
|
|
|
310
310
|
default_generation_settings?: { n_ctx?: unknown; n_predict?: unknown };
|
|
311
311
|
modalities?: { vision?: unknown };
|
|
312
312
|
build_info?: unknown;
|
|
313
|
+
total_slots?: unknown;
|
|
313
314
|
}
|
|
314
315
|
|
|
315
316
|
export interface LlamaCppPropsEnrichment {
|
|
@@ -499,6 +500,9 @@ export async function probeLlamaCppModelStatus(
|
|
|
499
500
|
if (flags.reasoning === true || flags.reasoningBudget !== undefined) caps.reasoning = true;
|
|
500
501
|
if (flags.mmproj) caps.vision = true;
|
|
501
502
|
if (flags.jinja === true) caps.tools = true;
|
|
503
|
+
if (flags.parallel !== undefined && Number.isInteger(flags.parallel) && flags.parallel > 0) {
|
|
504
|
+
caps.parallelSlots = flags.parallel;
|
|
505
|
+
}
|
|
502
506
|
const enrichment: LlamaCppStatusEnrichment = { modelId: selected.id, serverFlags: flags };
|
|
503
507
|
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
504
508
|
const notes = statusNotes(selected.id, selected.status);
|
|
@@ -532,13 +536,28 @@ export async function detectModelMismatch(
|
|
|
532
536
|
return `wire model id ${expected} does not match server's loaded model ${loaded}; llama.cpp serves a single fixed model`;
|
|
533
537
|
}
|
|
534
538
|
|
|
535
|
-
export async function probeLlamaCppProps(
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
539
|
+
export async function probeLlamaCppProps(
|
|
540
|
+
base: string,
|
|
541
|
+
ctx: ProbeContext,
|
|
542
|
+
modelId?: string,
|
|
543
|
+
): Promise<LlamaCppPropsEnrichment> {
|
|
544
|
+
const request = async (url: string): Promise<LlamaCppProps | null> => {
|
|
545
|
+
const opts = { url, timeoutMs: ctx.httpTimeoutMs } as const;
|
|
546
|
+
const result = await (ctx.signal
|
|
547
|
+
? probeJson<LlamaCppProps>({ ...opts, signal: ctx.signal })
|
|
548
|
+
: probeJson<LlamaCppProps>(opts));
|
|
549
|
+
return result.ok && result.data ? result.data : null;
|
|
550
|
+
};
|
|
551
|
+
const router = await request(`${base}/props`);
|
|
552
|
+
if (router === null) return {};
|
|
553
|
+
// A llama.cpp router reports its own role at /props and the selected
|
|
554
|
+
// worker's request slots at /props?model=<id>. A fixed-model server answers
|
|
555
|
+
// the first request directly, so the second GET is only made when needed.
|
|
556
|
+
const selected =
|
|
557
|
+
typeof router.total_slots === "number" || !modelId
|
|
558
|
+
? null
|
|
559
|
+
: await request(`${base}/props?model=${encodeURIComponent(modelId)}`);
|
|
560
|
+
const data = selected ?? router;
|
|
542
561
|
const enrichment: LlamaCppPropsEnrichment = {};
|
|
543
562
|
const caps: Partial<CapabilityFlags> = {};
|
|
544
563
|
const nCtx = data.default_generation_settings?.n_ctx;
|
|
@@ -547,9 +566,12 @@ export async function probeLlamaCppProps(base: string, ctx: ProbeContext): Promi
|
|
|
547
566
|
if (typeof nPredict === "number" && nPredict > 0) caps.maxTokens = nPredict;
|
|
548
567
|
const vision = data.modalities?.vision;
|
|
549
568
|
if (typeof vision === "boolean") caps.vision = vision;
|
|
569
|
+
const totalSlots = data.total_slots;
|
|
570
|
+
if (typeof totalSlots === "number" && Number.isInteger(totalSlots) && totalSlots > 0) caps.parallelSlots = totalSlots;
|
|
550
571
|
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
551
|
-
|
|
552
|
-
|
|
572
|
+
const buildInfo = data.build_info ?? router.build_info;
|
|
573
|
+
if (typeof buildInfo === "string" && buildInfo.length > 0) {
|
|
574
|
+
enrichment.serverVersion = buildInfo;
|
|
553
575
|
}
|
|
554
576
|
return enrichment;
|
|
555
577
|
}
|
|
@@ -52,7 +52,7 @@ const llamacppAnthropicRuntime: RuntimeDescriptor = {
|
|
|
52
52
|
};
|
|
53
53
|
const head = await (ctx.signal ? probeHttp({ ...headOpts, signal: ctx.signal }) : probeHttp(headOpts));
|
|
54
54
|
if (!head.ok) return head;
|
|
55
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
55
|
+
const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
|
|
56
56
|
const enriched: ProbeResult = { ...head };
|
|
57
57
|
if (props.discoveredCapabilities) enriched.discoveredCapabilities = props.discoveredCapabilities;
|
|
58
58
|
if (props.serverVersion) enriched.serverVersion = props.serverVersion;
|
|
@@ -157,7 +157,7 @@ const llamacppCompletionRuntime: RuntimeDescriptor = {
|
|
|
157
157
|
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs } as const;
|
|
158
158
|
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
159
159
|
if (!health.ok) return health;
|
|
160
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
160
|
+
const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
|
|
161
161
|
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
162
162
|
const enriched: ProbeResult = { ...health };
|
|
163
163
|
const discoveredCapabilities = {
|
|
@@ -101,7 +101,7 @@ const llamacppEmbedRuntime: RuntimeDescriptor = {
|
|
|
101
101
|
if (!probeResponse.ok) {
|
|
102
102
|
return { ok: false, error: `/embedding not available: HTTP ${probeResponse.status}` };
|
|
103
103
|
}
|
|
104
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
104
|
+
const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
|
|
105
105
|
const result: ProbeResult = { ok: true };
|
|
106
106
|
if (health.latencyMs !== undefined) result.latencyMs = health.latencyMs;
|
|
107
107
|
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
@@ -60,7 +60,7 @@ const llamacppRerankRuntime: RuntimeDescriptor = {
|
|
|
60
60
|
if (!(probeResponse.status === 200 || probeResponse.status === 202)) {
|
|
61
61
|
return { ok: false, error: `/reranking not available: HTTP ${probeResponse.status}` };
|
|
62
62
|
}
|
|
63
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
63
|
+
const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
|
|
64
64
|
const result: ProbeResult = { ok: true };
|
|
65
65
|
if (health.latencyMs !== undefined) result.latencyMs = health.latencyMs;
|
|
66
66
|
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
@@ -55,8 +55,8 @@ const llamacppRuntime: RuntimeDescriptor = {
|
|
|
55
55
|
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs } as const;
|
|
56
56
|
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
57
57
|
if (!health.ok) return health;
|
|
58
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
59
58
|
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
59
|
+
const props = await probeLlamaCppProps(base, ctx, status.modelId ?? target.defaultModel);
|
|
60
60
|
const catalog = await probeOpenAIModelCatalog(base, ctx);
|
|
61
61
|
const result: ProbeResult = { ok: true };
|
|
62
62
|
if (catalog.models.length > 0) result.models = catalog.models;
|
|
@@ -66,6 +66,9 @@ const llamacppRuntime: RuntimeDescriptor = {
|
|
|
66
66
|
const discoveredCapabilities = {
|
|
67
67
|
...(props.discoveredCapabilities ?? {}),
|
|
68
68
|
...(status.discoveredCapabilities ?? {}),
|
|
69
|
+
...(props.discoveredCapabilities?.parallelSlots !== undefined
|
|
70
|
+
? { parallelSlots: props.discoveredCapabilities.parallelSlots }
|
|
71
|
+
: {}),
|
|
69
72
|
};
|
|
70
73
|
if (Object.keys(discoveredCapabilities).length > 0) {
|
|
71
74
|
result.discoveredCapabilities = discoveredCapabilities;
|
|
@@ -84,6 +84,8 @@ function capabilities(model: LmStudioModelInfo, instance?: LmStudioLoadedInstanc
|
|
|
84
84
|
out.reasoning = model.reasoningOptions.some((option) => option !== "off");
|
|
85
85
|
const contextWindow = loadedContextLength(instance) ?? model.maxContextLength;
|
|
86
86
|
if (contextWindow !== undefined) out.contextWindow = contextWindow;
|
|
87
|
+
const parallel = instance?.config.parallel;
|
|
88
|
+
if (typeof parallel === "number" && Number.isInteger(parallel) && parallel > 0) out.parallelSlots = parallel;
|
|
87
89
|
return out;
|
|
88
90
|
}
|
|
89
91
|
|
|
@@ -144,6 +146,7 @@ function probeFromCatalog(catalog: LmStudioCatalog, target: TargetDescriptor): P
|
|
|
144
146
|
const selected = resolveModel(models, configuredModel(target));
|
|
145
147
|
const result: ProbeResult = {
|
|
146
148
|
ok: true,
|
|
149
|
+
discoveredCapabilities: { parallelSlots: 1 },
|
|
147
150
|
models: ids,
|
|
148
151
|
modelCapabilities,
|
|
149
152
|
modelStates,
|
|
@@ -161,7 +164,7 @@ function probeFromCatalog(catalog: LmStudioCatalog, target: TargetDescriptor): P
|
|
|
161
164
|
};
|
|
162
165
|
if (catalog.latencyMs !== undefined) result.latencyMs = catalog.latencyMs;
|
|
163
166
|
if (selected) {
|
|
164
|
-
result.discoveredCapabilities = capabilities(selected.model, selected.instance);
|
|
167
|
+
result.discoveredCapabilities = { parallelSlots: 1, ...capabilities(selected.model, selected.instance) };
|
|
165
168
|
result.capabilityModelId = configuredModel(target) ?? selected.model.key;
|
|
166
169
|
}
|
|
167
170
|
const loaded = models.flatMap((model) =>
|
|
@@ -33,6 +33,11 @@ function positiveNumber(value: unknown): number | undefined {
|
|
|
33
33
|
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : undefined;
|
|
34
34
|
}
|
|
35
35
|
|
|
36
|
+
function ollamaParallelSlots(): number {
|
|
37
|
+
const parsed = Number(process.env.OLLAMA_NUM_PARALLEL);
|
|
38
|
+
return Number.isInteger(parsed) && parsed > 0 ? parsed : 1;
|
|
39
|
+
}
|
|
40
|
+
|
|
36
41
|
/**
|
|
37
42
|
* Resident models reported by `/api/ps`, keyed by wire id. Best-effort: the
|
|
38
43
|
* probe still succeeds on `/api/tags` alone, so an `/api/ps` failure (older
|
|
@@ -83,7 +88,7 @@ const ollamaNativeRuntime: RuntimeDescriptor = {
|
|
|
83
88
|
if (result.latencyMs !== undefined) failed.latencyMs = result.latencyMs;
|
|
84
89
|
return failed;
|
|
85
90
|
}
|
|
86
|
-
const out: ProbeResult = { ok: true };
|
|
91
|
+
const out: ProbeResult = { ok: true, discoveredCapabilities: { parallelSlots: ollamaParallelSlots() } };
|
|
87
92
|
if (result.latencyMs !== undefined) out.latencyMs = result.latencyMs;
|
|
88
93
|
const modelStates = await probeResidentModelStates(base, ctx);
|
|
89
94
|
if (modelStates) out.modelStates = modelStates;
|
|
@@ -28,6 +28,8 @@ export interface CapabilityFlags {
|
|
|
28
28
|
fim: boolean;
|
|
29
29
|
contextWindow: number;
|
|
30
30
|
maxTokens: number;
|
|
31
|
+
/** Request slots reported by the endpoint probe. This is deployment capacity, not a model trait. */
|
|
32
|
+
parallelSlots?: number;
|
|
31
33
|
}
|
|
32
34
|
|
|
33
35
|
export const EMPTY_CAPABILITIES: CapabilityFlags = {
|