@iowarp/clio-coder 0.3.1 → 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +233 -373
- package/CONTRIBUTING.md +23 -23
- package/README.md +284 -613
- package/dist/{acp-FPR54DGL.js → acp-P2AQILE2.js} +43 -53
- package/dist/{agents-OGPIHPJH.js → agents-72W3BI7I.js} +43 -26
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-IC3K6NIZ.js → auth-5TWEIYDN.js} +20 -12
- package/dist/chunk-2DJ2KNFG.js +2095 -0
- package/dist/{chunk-IS3ONKU3.js → chunk-2IR2NMPA.js} +6 -4
- package/dist/chunk-2SFS6XQE.js +122 -0
- package/dist/{chunk-PV4JUBVJ.js → chunk-2TLUCQVG.js} +40 -21
- package/dist/chunk-2VTFPG5O.js +48 -0
- package/dist/chunk-4BJ5BYCE.js +61 -0
- package/dist/{chunk-474KN5II.js → chunk-4BPJXDWC.js} +111 -181
- package/dist/chunk-4VP4KH3K.js +962 -0
- package/dist/chunk-4XUGQOHA.js +797 -0
- package/dist/chunk-4ZG3XFUR.js +77 -0
- package/dist/chunk-5B2AEOW5.js +5407 -0
- package/dist/{chunk-K2ITRMHZ.js → chunk-5TSRNF4G.js} +6 -138
- package/dist/{chunk-4QKXUHSR.js → chunk-5UFT4SUX.js} +70 -20
- package/dist/{chunk-OLBBMFRD.js → chunk-5UUP6MWO.js} +24 -62
- package/dist/chunk-65DEGPJ6.js +52 -0
- package/dist/chunk-6EJMN2Y3.js +17 -0
- package/dist/chunk-6N5PTWMY.js +136 -0
- package/dist/chunk-6SGHMWE3.js +277 -0
- package/dist/chunk-6XLNIQDB.js +27 -0
- package/dist/chunk-7CR24IG7.js +242 -0
- package/dist/chunk-7MNJORFF.js +22 -0
- package/dist/{chunk-KY56HMHH.js → chunk-A3CYT5EX.js} +125 -31
- package/dist/chunk-AGYYIBLL.js +1069 -0
- package/dist/{chunk-GB6QRBXN.js → chunk-APJ265NV.js} +54 -1187
- package/dist/{chunk-673JJUWJ.js → chunk-BMEMKKIT.js} +2 -2
- package/dist/chunk-CBCAPZAA.js +229 -0
- package/dist/chunk-CMZWFGD2.js +352 -0
- package/dist/chunk-COU2UHX6.js +400 -0
- package/dist/chunk-DSELYM6W.js +1077 -0
- package/dist/chunk-DUYJ5IO6.js +644 -0
- package/dist/chunk-ECH6PKUQ.js +39 -0
- package/dist/chunk-ED4KHGC3.js +143 -0
- package/dist/chunk-EKMEHE4H.js +340 -0
- package/dist/chunk-FCSXB6T2.js +338 -0
- package/dist/chunk-FJ3H4MN5.js +48 -0
- package/dist/{chunk-RPTR2H26.js → chunk-FNTMWMX5.js} +21 -15
- package/dist/chunk-FQ4SKYE4.js +29 -0
- package/dist/chunk-G4BMMOKF.js +182 -0
- package/dist/{chunk-ZPY3JZ5E.js → chunk-GGXXDWE4.js} +183 -1233
- package/dist/chunk-HC4CLZ2Y.js +68 -0
- package/dist/{chunk-LU4TK2PR.js → chunk-HFSBBKSQ.js} +5 -56
- package/dist/{chunk-PIUMUEMV.js → chunk-HKIYEGME.js} +10 -6
- package/dist/chunk-I4HZDVNP.js +73 -0
- package/dist/chunk-IKCO5N3L.js +162 -0
- package/dist/chunk-IR4CFBFN.js +56 -0
- package/dist/{chunk-R5KLMSBV.js → chunk-J5Q24KAG.js} +2 -2
- package/dist/chunk-J7CWMCQD.js +255 -0
- package/dist/{chunk-K5XEMXTI.js → chunk-JVCV3ICN.js} +1 -1
- package/dist/chunk-KZWTDYJF.js +217 -0
- package/dist/chunk-LBMZMYH2.js +285 -0
- package/dist/{chunk-G34LV2PF.js → chunk-LM5TQCJZ.js} +84 -170
- package/dist/chunk-LW6DSM3M.js +5135 -0
- package/dist/chunk-LWLEKMDQ.js +3482 -0
- package/dist/{chunk-H6F6BYOH.js → chunk-LZSJBIVT.js} +7003 -7434
- package/dist/{chunk-HQQID6OA.js → chunk-M6SHUN7Q.js} +5 -5
- package/dist/{chunk-FST4FYJB.js → chunk-MFFY33HR.js} +99 -140
- package/dist/{chunk-BSU2YIWB.js → chunk-MVVUPGPW.js} +131 -136
- package/dist/chunk-OAO4GE4M.js +619 -0
- package/dist/{chunk-Q3RUPKEJ.js → chunk-OC7FIQPC.js} +58 -189
- package/dist/chunk-OKGUZO2U.js +34 -0
- package/dist/{chunk-GAEBEQVI.js → chunk-OOJYHWRB.js} +32 -346
- package/dist/{chunk-Q5WJOSJ7.js → chunk-OQ33BKR3.js} +2 -1
- package/dist/chunk-OQE5J4C6.js +73 -0
- package/dist/{chunk-KKNLWXI6.js → chunk-ORBHGJC5.js} +8 -8
- package/dist/{chunk-MAR7Y6HW.js → chunk-PAJK6MAQ.js} +23 -16
- package/dist/{chunk-M5T5VO65.js → chunk-PIWWS5BL.js} +837 -635
- package/dist/chunk-POHLU5DW.js +1186 -0
- package/dist/chunk-QKMUKYO7.js +4961 -0
- package/dist/chunk-SRDMMSEP.js +16405 -0
- package/dist/chunk-SST6Z5JA.js +80 -0
- package/dist/chunk-STBPMHSX.js +2456 -0
- package/dist/chunk-T6YILFSB.js +80 -0
- package/dist/chunk-TZK7PACC.js +174 -0
- package/dist/chunk-TZTZS7QK.js +227 -0
- package/dist/{chunk-ASND7OZK.js → chunk-UFIIWP2H.js} +13 -13
- package/dist/chunk-UOV2BYIW.js +107 -0
- package/dist/{chunk-PFEFKVGL.js → chunk-V6RTAOC2.js} +13 -11
- package/dist/chunk-VAKQQHWR.js +434 -0
- package/dist/chunk-VG7TBQIY.js +128 -0
- package/dist/chunk-VJWL6YS5.js +244 -0
- package/dist/{chunk-EYOKLTMF.js → chunk-VPAYEGVX.js} +17 -3
- package/dist/chunk-WEH5XRJQ.js +32 -0
- package/dist/chunk-X4RCMKVQ.js +641 -0
- package/dist/{chunk-TEKV33Q5.js → chunk-X6IAEBZR.js} +65 -33
- package/dist/chunk-XBXAASKX.js +18 -0
- package/dist/chunk-XN3L4EYL.js +46 -0
- package/dist/{chunk-RDLVBZEO.js → chunk-YCWGATWI.js} +6 -4
- package/dist/chunk-YHZX5GEU.js +193 -0
- package/dist/chunk-YXLYO42X.js +91 -0
- package/dist/{chunk-NMOX6HFD.js → chunk-ZDOOVTXZ.js} +29 -77
- package/dist/chunk-ZI647VB5.js +37 -0
- package/dist/{chunk-C4PTHK7P.js → chunk-ZWLZP4ZT.js} +5 -5
- package/dist/chunk-ZWMF7253.js +1882 -0
- package/dist/cli/index.js +62 -54
- package/dist/clio-JOU4FXVA.js +25 -0
- package/dist/code-nav-7AX6FYE6.js +600 -0
- package/dist/codewiki/build-worker.js +66 -0
- package/dist/compile-cache-CVJMMODC.js +18 -0
- package/dist/{components-DMAOEKFB.js → components-KELWS457.js} +11 -6
- package/dist/{config-IRUQ7SE4.js → config-XCDVKR23.js} +92 -55
- package/dist/configure-4GAP54ZW.js +42 -0
- package/dist/{context-5RADCKTR.js → context-4UOGGLQ5.js} +71 -35
- package/dist/context-5VKGUVJJ.js +866 -0
- package/dist/{context-3KWFLHJG.js → context-77FM5DV5.js} +15 -13
- package/dist/{context-clear-7TSNPAAI.js → context-clear-XXJRLCJJ.js} +54 -28
- package/dist/{context-index-W4RLWOQH.js → context-index-BZ4UYMTC.js} +30 -24
- package/dist/dispatch-runner-QPRDDBDX.js +1997 -0
- package/dist/{docs-5AWSPS37.js → docs-2C2LTVT2.js} +23 -10
- package/dist/{doctor-UC5NAJYQ.js → doctor-HR46URBJ.js} +27 -17
- package/dist/{eval-U6TJHRLX.js → eval-XSSNATB4.js} +29 -16
- package/dist/{evidence-YEGUW4L3.js → evidence-6HG2PY2B.js} +46 -26
- package/dist/{evolve-TXARCTPG.js → evolve-K7YU3NCY.js} +45 -25
- package/dist/{extensions-OZFJ3A3G.js → extensions-QVDOHDGJ.js} +16 -7
- package/dist/{fleet-6G3DHNYE.js → fleet-VY3HHKN6.js} +163 -54
- package/dist/{fleet-preflight-DSNT37JK.js → fleet-preflight-DDN536IT.js} +7 -4
- package/dist/{init-KZ5QTF6M.js → init-JYGXI3FK.js} +69 -32
- package/dist/{memory-73ESV5YC.js → memory-WFZMGYHX.js} +48 -27
- package/dist/{models-A4PVNWJK.js → models-I5QWSEOM.js} +39 -25
- package/dist/monitor-GE4ID3IA.js +661 -0
- package/dist/{chunk-FCIH3BIZ.js → orchestrator-EM5MC3HM.js} +15979 -12407
- package/dist/{paths-C4H6IV77.js → paths-UXLN5YYZ.js} +10 -5
- package/dist/{preload-6WVMHX3A.js → preload-P6DGH2PZ.js} +2 -2
- package/dist/{reset-BGW6OGMV.js → reset-L2FQEE3E.js} +16 -10
- package/dist/{run-YTPEYQOH.js → run-ZU3QMZPZ.js} +101 -61
- package/dist/{share-YIFFV4NQ.js → share-S5BZQC5I.js} +15 -8
- package/dist/{skills-2V6RA3OQ.js → skills-X5VXCRNQ.js} +34 -14
- package/dist/{skills-eval-S2TVJO4F.js → skills-eval-WKIHWTHR.js} +70 -34
- package/dist/steer-GGWFUJUD.js +77 -0
- package/dist/{targets-TYXLPB23.js → targets-SNCPI2NR.js} +43 -27
- package/dist/terminal-lease-BNAHVHBS.js +395 -0
- package/dist/{trace-GGOJ6Q6Z.js → trace-PNCASAXC.js} +41 -16
- package/dist/{chunk-N6F52NLF.js → tree-sitter-HGKH6LG4.js} +28 -2306
- package/dist/{uninstall-LLLT4F4W.js → uninstall-FZCQCDKC.js} +10 -5
- package/dist/{upgrade-33G2LMM5.js → upgrade-JQHHPQ4K.js} +45 -25
- package/dist/{usage-ZAFSXKKG.js → usage-OR4O5SMZ.js} +62 -31
- package/dist/verify-375KUB3Y.js +716 -0
- package/dist/web-fetch-2YHJ3KTG.js +638 -0
- package/dist/{wiki-generate-NUQCVOQ3.js → wiki-generate-UEXP2ARI.js} +74 -34
- package/dist/worker/entry.js +221 -36
- package/dist/workspace-G4ZWUIPR.js +22 -0
- package/docs/README.md +22 -17
- package/docs/acp.md +168 -16
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +136 -7
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +114 -71
- package/docs/config-knobs-audit.md +1 -3
- package/docs/configuration-and-targets.md +174 -46
- package/docs/context-engine.md +29 -6
- package/docs/development-pipeline.md +26 -1
- package/docs/dispatch-architecture-rationale.md +1 -1
- package/docs/documentation-coverage.md +2 -2
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +13 -5
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +1 -1
- package/docs/evidence-and-memory.md +6 -2
- package/docs/evolution.md +2 -2
- package/docs/exit-codes-and-output.md +15 -9
- package/docs/extensions-and-sharing.md +9 -9
- package/docs/fleet-dispatch.md +7 -5
- package/docs/git-commit-provenance.md +120 -0
- package/docs/glossary.md +1 -1
- package/docs/installation-and-lifecycle.md +34 -27
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +45 -14
- package/docs/observability.md +8 -5
- package/docs/performance-methodology.md +491 -0
- package/docs/pi-boundary.md +72 -0
- package/docs/proactive-memory.md +3 -3
- package/docs/prompt-envelope-and-tools.md +24 -3
- package/docs/provider-adapter-cookbook.md +57 -4
- package/docs/release-cut-checklist.md +129 -115
- package/docs/safety-model.md +9 -5
- package/docs/scientific-validation.md +3 -3
- package/docs/session-lifecycle.md +55 -12
- package/docs/skills-marketplace.md +12 -8
- package/docs/time-conventions.md +1 -1
- package/docs/tool-usage.md +3 -3
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +10 -7
- package/docs/tui-design.md +47 -10
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +19 -22
- package/skills/coding/ast-grep/SKILL.md +136 -0
- package/skills/coding/ast-grep/evals.md +56 -0
- package/skills/coding/ast-grep/references/rule_reference.md +297 -0
- package/skills/coding/coding-standards/SKILL.md +113 -0
- package/skills/coding/coding-standards/evals.md +34 -0
- package/skills/coding/prototype/SKILL.md +86 -0
- package/skills/coding/prototype/evals.md +42 -0
- package/skills/coding/prototype/references/LOGIC.md +67 -0
- package/skills/coding/prototype/references/UI.md +112 -0
- package/skills/coding/tdd/SKILL.md +101 -0
- package/skills/coding/tdd/evals.md +41 -0
- package/skills/coding/tdd/references/mocking.md +59 -0
- package/skills/coding/tdd/references/tests.md +77 -0
- package/skills/context/context-handoff/SKILL.md +126 -0
- package/skills/context/context-handoff/evals.md +57 -0
- package/skills/context/context-handoff/scripts/new-handoff.sh +26 -0
- package/skills/context/context-prime/SKILL.md +95 -0
- package/skills/context/context-prime/evals.md +54 -0
- package/skills/meta/clio-dev/SKILL.md +91 -0
- package/skills/meta/clio-dev/evals.md +45 -0
- package/skills/meta/clio-test/SKILL.md +130 -0
- package/skills/meta/clio-test/evals.md +43 -0
- package/skills/meta/clio-test/references/harness.md +97 -0
- package/skills/meta/clio-test/references/test-map.md +59 -0
- package/skills/meta/credentials/SKILL.md +125 -0
- package/skills/meta/credentials/evals.md +104 -0
- package/skills/meta/find-skills/SKILL.md +72 -0
- package/skills/meta/find-skills/evals.md +47 -0
- package/skills/meta/herdr/SKILL.md +127 -0
- package/skills/meta/herdr/evals.md +38 -0
- package/skills/meta/skill-craft/SKILL.md +102 -0
- package/skills/meta/skill-craft/evals.md +41 -0
- package/skills/planning/architecture/SKILL.md +129 -0
- package/skills/planning/architecture/evals.md +36 -0
- package/skills/planning/backlog/SKILL.md +90 -0
- package/skills/planning/backlog/evals.md +43 -0
- package/skills/planning/prd/SKILL.md +82 -0
- package/skills/planning/prd/evals.md +49 -0
- package/skills/planning/product-intent/SKILL.md +112 -0
- package/skills/planning/product-intent/evals.md +36 -0
- package/skills/planning/tech-spec/SKILL.md +115 -0
- package/skills/planning/tech-spec/evals.md +47 -0
- package/skills/registry.yaml +136 -0
- package/skills/research/arxiv-literature/SKILL.md +104 -0
- package/skills/research/arxiv-literature/evals.md +58 -0
- package/skills/research/experiment-protocol/SKILL.md +122 -0
- package/skills/research/experiment-protocol/evals.md +91 -0
- package/skills/research/scientific-debugging/SKILL.md +119 -0
- package/skills/research/scientific-debugging/evals.md +138 -0
- package/skills/research/scientific-modernization/SKILL.md +138 -0
- package/skills/research/scientific-modernization/evals.md +84 -0
- package/skills/workflow/design-council/SKILL.md +139 -0
- package/skills/workflow/design-council/evals.md +97 -0
- package/skills/workflow/grill-me/SKILL.md +186 -0
- package/skills/workflow/grill-me/evals.md +78 -0
- package/skills/workflow/workflow-distiller/SKILL.md +136 -0
- package/skills/workflow/workflow-distiller/evals.md +107 -0
- package/src/cli/acp.ts +31 -4
- package/src/cli/clio.ts +68 -6
- package/src/cli/config-inspect.ts +28 -22
- package/src/cli/configure.ts +47 -9
- package/src/cli/context-clear.ts +2 -2
- package/src/cli/context-index.ts +21 -23
- package/src/cli/context.ts +13 -8
- package/src/cli/default-target.ts +9 -17
- package/src/cli/docs.ts +11 -5
- package/src/cli/evidence.ts +4 -1
- package/src/cli/extensions.ts +10 -1
- package/src/cli/fleet.ts +47 -6
- package/src/cli/index.ts +55 -26
- package/src/cli/memory.ts +3 -1
- package/src/cli/models.ts +1 -1
- package/src/cli/modes/json-stream.ts +37 -1
- package/src/cli/modes/print.ts +24 -9
- package/src/cli/run.ts +2 -2
- package/src/cli/skills-eval.ts +23 -8
- package/src/cli/skills.ts +19 -4
- package/src/cli/targets.ts +4 -0
- package/src/cli/text-layout.ts +15 -5
- package/src/cli/trace.ts +62 -14
- package/src/cli/upgrade.ts +18 -2
- package/src/cli/usage.ts +10 -3
- package/src/cli/wiki-generate.ts +2 -1
- package/src/core/agent-environment.ts +7 -0
- package/src/core/bash-exec.ts +72 -1
- package/src/core/boot-trace.ts +9 -4
- package/src/core/bus-events.ts +20 -4
- package/src/core/commit-attribution.ts +157 -0
- package/src/core/compile-cache.ts +159 -0
- package/src/core/config.ts +131 -2
- package/src/core/defaults.ts +39 -5
- package/src/core/domain-loader.ts +12 -5
- package/src/core/git-commit-attribution.ts +387 -0
- package/src/core/incomplete-installation.ts +45 -0
- package/src/core/response-schema.ts +1 -1
- package/src/core/safe-exec.ts +13 -1
- package/src/core/settings-layers.ts +155 -21
- package/src/core/skill-activation.ts +1 -1
- package/src/core/startup-timer.ts +3 -3
- package/src/core/state-file-lock.ts +13 -1
- package/src/core/termination.ts +78 -5
- package/src/domains/config/classify.ts +15 -3
- package/src/domains/config/extension.ts +19 -13
- package/src/domains/config/index.ts +10 -0
- package/src/domains/config/keybindings.ts +45 -9
- package/src/domains/context/bootstrap-prompt.ts +1 -1
- package/src/domains/context/bootstrap.ts +111 -18
- package/src/domains/context/clear.ts +16 -11
- package/src/domains/context/clio-md.ts +111 -9
- package/src/domains/context/codewiki/artifact.ts +400 -0
- package/src/domains/context/codewiki/build-worker-protocol.ts +24 -0
- package/src/domains/context/codewiki/build-worker.ts +54 -0
- package/src/domains/context/codewiki/coordinator.ts +182 -0
- package/src/domains/context/codewiki/indexer.ts +59 -144
- package/src/domains/context/codewiki/paths.ts +67 -0
- package/src/domains/context/codewiki/schema.ts +80 -0
- package/src/domains/context/codewiki/tree-sitter.ts +1 -1
- package/src/domains/context/contract.ts +11 -5
- package/src/domains/context/extension.ts +94 -143
- package/src/domains/context/fingerprint.ts +3 -1
- package/src/domains/context/index.ts +12 -22
- package/src/domains/context/project-metadata.ts +19 -0
- package/src/domains/context/prompt-context.ts +9 -10
- package/src/domains/context/refresh.ts +29 -21
- package/src/domains/context/runtime.ts +17 -0
- package/src/domains/context/wiki/generate.ts +39 -34
- package/src/domains/context/wiki/plan.ts +1 -1
- package/src/domains/context/wiki/prompts.ts +21 -8
- package/src/domains/dispatch/code-step.ts +20 -1
- package/src/domains/dispatch/extension.ts +158 -21
- package/src/domains/dispatch/failure-classification.ts +6 -0
- package/src/domains/dispatch/fleet-commit-attribution.ts +56 -0
- package/src/domains/dispatch/orphan-recovery.ts +50 -8
- package/src/domains/dispatch/receipt-integrity.ts +5 -0
- package/src/domains/dispatch/state.ts +31 -6
- package/src/domains/dispatch/transport.ts +2 -1
- package/src/domains/dispatch/types.ts +14 -0
- package/src/domains/dispatch/worker-spawn.ts +21 -2
- package/src/domains/eval/metrics/context.ts +1 -1
- package/src/domains/eval/types.ts +0 -1
- package/src/domains/evidence/build.ts +41 -1
- package/src/domains/lifecycle/migrations/2026-08-18-lmstudio-runtime-id.ts +52 -0
- package/src/domains/lifecycle/migrations/index.ts +24 -4
- package/src/domains/middleware/hooks-io.ts +12 -0
- package/src/domains/middleware/skills-reminder.ts +30 -15
- package/src/domains/prompts/compiler.ts +142 -84
- package/src/domains/prompts/contract.ts +18 -2
- package/src/domains/prompts/extension.ts +39 -7
- package/src/domains/prompts/fragment-loader.ts +0 -1
- package/src/domains/prompts/fragments/identity/clio.md +2 -4
- package/src/domains/prompts/fragments/identity/docs-routing.md +10 -0
- package/src/domains/prompts/fragments/identity/self-awareness.md +1 -45
- package/src/domains/prompts/fragments/operating/contract.md +4 -50
- package/src/domains/prompts/fragments/operating/delegation.md +42 -0
- package/src/domains/prompts/fragments/operating/skills.md +26 -0
- package/src/domains/prompts/fragments/operating/worker.md +16 -0
- package/src/domains/prompts/fragments/safety/auto-edit.md +5 -5
- package/src/domains/prompts/fragments/safety/full-auto.md +3 -3
- package/src/domains/prompts/fragments/safety/read-only.md +4 -4
- package/src/domains/prompts/fragments/safety/suggest.md +2 -2
- package/src/domains/prompts/fragments/wiki/page.md +10 -0
- package/src/domains/prompts/fragments/wiki/plan.md +10 -0
- package/src/domains/prompts/preload.ts +3 -3
- package/src/domains/providers/auth/api-key.ts +1 -1
- package/src/domains/providers/auth/backend-file.ts +20 -10
- package/src/domains/providers/auth/backend-memory.ts +59 -4
- package/src/domains/providers/auth/boot-status.ts +65 -0
- package/src/domains/providers/auth/oauth.ts +2 -1
- package/src/domains/providers/auth/storage.ts +97 -38
- package/src/domains/providers/capabilities.ts +12 -4
- package/src/domains/providers/contract.ts +15 -4
- package/src/domains/providers/extension.ts +18 -6
- package/src/domains/providers/model-runtime-capabilities.ts +15 -4
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +118 -35
- package/src/domains/providers/plugins.ts +5 -3
- package/src/domains/providers/probe/fingerprint.ts +25 -5
- package/src/domains/providers/registry.ts +31 -10
- package/src/domains/providers/runtimes/boot-manifest.ts +55 -0
- package/src/domains/providers/runtimes/builtins.ts +2 -2
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +423 -0
- package/src/domains/providers/runtimes/common/local-synth.ts +6 -7
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +241 -0
- package/src/domains/providers/support.ts +6 -3
- package/src/domains/providers/types/local-model-quirks.ts +7 -9
- package/src/domains/providers/types/runtime-descriptor.ts +12 -1
- package/src/domains/providers/types/target-descriptor.ts +22 -0
- package/src/domains/resources/contract.ts +0 -1
- package/src/domains/resources/extension.ts +1 -3
- package/src/domains/resources/loader.ts +3 -4
- package/src/domains/resources/prompts/loader.ts +16 -2
- package/src/domains/resources/prompts/substitute.ts +1 -65
- package/src/domains/resources/skills/content-hash.ts +2 -0
- package/src/domains/resources/skills/install.ts +17 -0
- package/src/domains/resources/skills/loader.ts +17 -10
- package/src/domains/resources/skills/marketplace.ts +55 -9
- package/src/domains/safety/action-classifier.ts +4 -2
- package/src/domains/safety/audit.ts +8 -2
- package/src/domains/safety/extension.ts +1 -1
- package/src/domains/session/compaction/branch-summary.ts +3 -2
- package/src/domains/session/compaction/cut-point.ts +2 -1
- package/src/domains/session/compaction/tokens.ts +2 -1
- package/src/domains/session/context-ledger.ts +14 -0
- package/src/domains/session/contract.ts +15 -0
- package/src/domains/session/decision-board.ts +190 -0
- package/src/domains/session/entries.ts +66 -3
- package/src/domains/session/extension.ts +93 -12
- package/src/domains/session/retry.ts +10 -18
- package/src/domains/session/session-artifacts.ts +107 -0
- package/src/domains/session/task-board.ts +207 -13
- package/src/domains/session/tree/active-path.ts +44 -5
- package/src/domains/session/tree/fork.ts +26 -27
- package/src/domains/session/tree/preview.ts +2 -2
- package/src/domains/session/workspace/git-probe.ts +17 -11
- package/src/domains/user-tasks/store.ts +297 -0
- package/src/engine/acp/errors.ts +96 -0
- package/src/engine/acp/server.ts +1728 -146
- package/src/engine/acp/transport.ts +135 -14
- package/src/engine/acp/types.ts +26 -0
- package/src/engine/agent.ts +3 -3
- package/src/engine/ai.ts +32 -27
- package/src/engine/alcf-oauth.ts +26 -19
- package/src/engine/api-registry.ts +223 -0
- package/src/engine/apis/index.ts +3 -7
- package/src/engine/apis/llamacpp-residency.ts +49 -9
- package/src/engine/apis/lmstudio-residency.ts +5 -21
- package/src/engine/apis/lmstudio.ts +243 -0
- package/src/engine/apis/ollama-native.ts +24 -3
- package/src/engine/apis/openai-completions.ts +170 -91
- package/src/engine/apis/residency.ts +139 -3
- package/src/engine/apis/types.ts +16 -0
- package/src/engine/env-api-keys.ts +98 -0
- package/src/engine/gemma-channel-filter.ts +223 -0
- package/src/engine/instrumented-tui.ts +192 -0
- package/src/engine/messages.ts +14 -0
- package/src/engine/models.ts +42 -0
- package/src/engine/oauth.ts +16 -12
- package/src/engine/prompt-templates.ts +1 -0
- package/src/engine/provider-payload.ts +16 -59
- package/src/engine/strip-tokenizer-sentinels.ts +1 -1
- package/src/engine/truncate.ts +9 -0
- package/src/engine/tui.ts +17 -9
- package/src/engine/types.ts +3 -6
- package/src/engine/worker-runtime-capabilities.ts +5 -0
- package/src/engine/worker-runtime.ts +1 -1
- package/src/engine/worker-tools.ts +9 -4
- package/src/entry/boot-options.ts +50 -0
- package/src/entry/orchestrator.ts +288 -150
- package/src/interactive/application-controller.ts +89 -2
- package/src/interactive/chat-loop.ts +266 -41
- package/src/interactive/chat-panel.ts +715 -278
- package/src/interactive/chat-renderer.ts +306 -85
- package/src/interactive/clio-editor.ts +3 -8
- package/src/interactive/command-fallbacks.ts +2 -2
- package/src/interactive/context-overlay.ts +27 -1
- package/src/interactive/editor-submit.ts +253 -24
- package/src/interactive/export-html/ansi-to-html.ts +161 -0
- package/src/interactive/export-html/index.ts +51 -0
- package/src/interactive/export-html/template.ts +45 -0
- package/src/interactive/export-html/tool-renderer.ts +54 -0
- package/src/interactive/footer/dashboard.ts +4 -0
- package/src/interactive/footer/notifications.ts +1 -1
- package/src/interactive/footer/widgets.ts +42 -22
- package/src/interactive/footer-panel.ts +8 -3
- package/src/interactive/format-time.ts +14 -2
- package/src/interactive/interactive-application.ts +203 -17
- package/src/interactive/interactive-event-projection.ts +18 -1
- package/src/interactive/interactive-input-runtime.ts +50 -4
- package/src/interactive/interactive-presentation.ts +151 -19
- package/src/interactive/interactive-shell.ts +268 -14
- package/src/interactive/interactive-slash-runtime.ts +184 -117
- package/src/interactive/interactive-tickers.ts +38 -7
- package/src/interactive/keybinding-manager.ts +1 -1
- package/src/interactive/layout.ts +40 -3
- package/src/interactive/overlay-frame.ts +1 -1
- package/src/interactive/overlay-general-openers.ts +58 -1
- package/src/interactive/overlay-key-routing.ts +3 -0
- package/src/interactive/overlay-lifecycle.ts +13 -0
- package/src/interactive/overlay-permission-lifecycle.ts +2 -1
- package/src/interactive/overlay-session-lifecycle.ts +69 -12
- package/src/interactive/overlays/ask-user.ts +146 -24
- package/src/interactive/overlays/decisions.ts +300 -0
- package/src/interactive/overlays/help-reference.ts +15 -10
- package/src/interactive/overlays/model-selector.ts +34 -16
- package/src/interactive/overlays/session-selector.ts +18 -0
- package/src/interactive/overlays/settings.ts +105 -17
- package/src/interactive/overlays/skills-hub.ts +4 -4
- package/src/interactive/overlays/tree-selector.ts +41 -6
- package/src/interactive/render-trace.ts +499 -90
- package/src/interactive/renderers/compaction-summary.ts +2 -2
- package/src/interactive/renderers/diff.ts +115 -104
- package/src/interactive/renderers/mermaid.ts +53 -0
- package/src/interactive/renderers/tool-execution.ts +516 -168
- package/src/interactive/renderers/worker-entry.ts +20 -4
- package/src/interactive/session-switch-settlement.ts +10 -0
- package/src/interactive/slash-autocomplete.ts +6 -114
- package/src/interactive/slash-commands.ts +135 -47
- package/src/interactive/slash-spec.ts +9 -38
- package/src/interactive/status/controller.ts +5 -1
- package/src/interactive/status/index.ts +12 -1
- package/src/interactive/status/reasoning.ts +87 -0
- package/src/interactive/status/summary.ts +13 -2
- package/src/interactive/stdout-backpressure.ts +99 -0
- package/src/interactive/stream-pacer.ts +530 -0
- package/src/interactive/stream-pacing-policy.ts +66 -0
- package/src/interactive/tasks-overlay.ts +368 -14
- package/src/interactive/terminal-lease.ts +485 -0
- package/src/interactive/theme/tokens.ts +1 -1
- package/src/interactive/transcript-detail.ts +120 -0
- package/src/interactive/turn-context.ts +4 -3
- package/src/interactive/turn-persistence.ts +30 -13
- package/src/interactive/turn-queues.ts +12 -0
- package/src/interactive/turn-recovery.ts +25 -8
- package/src/interactive/turn-runtime.ts +79 -12
- package/src/interactive/turn-state.ts +10 -0
- package/src/interactive/view/artifacts.ts +114 -4
- package/src/interactive/view/view-overlay.ts +3 -0
- package/src/interactive/welcome-dashboard.ts +17 -16
- package/src/interactive/worker-receipts.ts +52 -3
- package/src/interactive/worker-stream.ts +5 -1
- package/src/tools/agent-tools.ts +23 -3
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +23 -13
- package/src/tools/bash.ts +30 -2
- package/src/tools/bootstrap.ts +34 -431
- package/src/tools/builtin-tool-catalog.ts +271 -0
- package/src/tools/codewiki/code-nav-surface.ts +29 -0
- package/src/tools/codewiki/code-nav.ts +8 -22
- package/src/tools/codewiki/shared.ts +41 -38
- package/src/tools/context/docs-engine.ts +14 -3
- package/src/tools/context/index.ts +107 -28
- package/src/tools/context/surface.ts +19 -0
- package/src/tools/core-bootstrap.ts +168 -0
- package/src/tools/credential-present.ts +5 -5
- package/src/tools/dispatch-admission.ts +533 -0
- package/src/tools/dispatch-background.ts +54 -0
- package/src/tools/dispatch-event-text.ts +6 -0
- package/src/tools/dispatch-plan.ts +9 -4
- package/src/tools/dispatch-run-events.ts +238 -0
- package/src/tools/dispatch-runner.ts +2370 -0
- package/src/tools/dispatch-scout-admission.ts +295 -0
- package/src/tools/dispatch-types.ts +77 -0
- package/src/tools/dispatch.ts +67 -3161
- package/src/tools/find.ts +4 -2
- package/src/tools/grep.ts +2 -2
- package/src/tools/lazy-tool.ts +60 -0
- package/src/tools/ledger.ts +3 -3
- package/src/tools/monitor-surface.ts +36 -0
- package/src/tools/monitor.ts +2 -32
- package/src/tools/observers.ts +2 -2
- package/src/tools/presentation.ts +107 -0
- package/src/tools/registry.ts +45 -27
- package/src/tools/safe-exec.ts +2 -2
- package/src/tools/steer-surface.ts +17 -0
- package/src/tools/steer.ts +2 -13
- package/src/tools/tasks.ts +108 -11
- package/src/tools/truncate.ts +25 -184
- package/src/tools/verify/frontend.ts +3 -1
- package/src/tools/verify/index.ts +3 -38
- package/src/tools/verify/surface.ts +46 -0
- package/src/tools/web-fetch-surface.ts +23 -0
- package/src/tools/web-fetch.ts +2 -20
- package/src/tools/write.ts +7 -2
- package/src/worker/entry.ts +39 -2
- package/src/worker/spec-contract.ts +26 -5
- package/dist/chunk-7SS2CTV2.js +0 -61361
- package/dist/chunk-DKGKUHFA.js +0 -924
- package/dist/chunk-GEP36Y4X.js +0 -12796
- package/dist/chunk-XYWBQRDM.js +0 -137
- package/dist/clio-BZVGEUFJ.js +0 -58
- package/dist/configure-S7S6F6CL.js +0 -32
- package/docs/html/agents_blueprint.html +0 -936
- package/docs/html/alcf_blueprint.html +0 -324
- package/docs/html/architecture_blueprint.html +0 -850
- package/docs/html/commands_blueprint.html +0 -939
- package/docs/html/config_knobs_audit_blueprint.html +0 -178
- package/docs/html/configuration_blueprint.html +0 -1080
- package/docs/html/context_blueprint.html +0 -603
- package/docs/html/documentation_blueprint.html +0 -832
- package/docs/html/environment_blueprint.html +0 -404
- package/docs/html/eval_blueprint.html +0 -743
- package/docs/html/evals_internal_blueprint.html +0 -190
- package/docs/html/evolution_blueprint.html +0 -674
- package/docs/html/extensions_blueprint.html +0 -2065
- package/docs/html/fleet_dispatch_blueprint.html +0 -286
- package/docs/html/index.html +0 -919
- package/docs/html/lifecycle_blueprint.html +0 -723
- package/docs/html/memory_blueprint.html +0 -699
- package/docs/html/middleware_blueprint.html +0 -664
- package/docs/html/models_blueprint.html +0 -2366
- package/docs/html/observability_blueprint.html +0 -683
- package/docs/html/provider_adapter_blueprint.html +0 -245
- package/docs/html/safety_blueprint.html +0 -1386
- package/docs/html/shared.css +0 -571
- package/docs/html/shared.js +0 -143
- package/docs/html/skills_blueprint.html +0 -671
- package/docs/html/soak_blueprint.html +0 -182
- package/docs/html/tool_usage_blueprint.html +0 -350
- package/docs/html/tools_blueprint.html +0 -2249
- package/docs/html/trace_blueprint.html +0 -235
- package/docs/html/tui_design_blueprint.html +0 -374
- package/docs/html/validation_blueprint.html +0 -961
- package/docs/html/worker_dispatch_blueprint.html +0 -231
- package/src/core/release.ts +0 -2
- package/src/domains/providers/runtimes/common/lmstudio-logger.ts +0 -32
- package/src/domains/providers/runtimes/local-native/lmstudio-native.ts +0 -491
- package/src/engine/apis/lmstudio-native.ts +0 -1438
- package/src/engine/apis/thinking-replay.ts +0 -11
- package/src/tools/string-enum.ts +0 -15
|
@@ -1,1438 +0,0 @@
|
|
|
1
|
-
import { randomUUID } from "node:crypto";
|
|
2
|
-
import type {
|
|
3
|
-
Api,
|
|
4
|
-
AssistantMessage,
|
|
5
|
-
AssistantMessageEventStream,
|
|
6
|
-
Context,
|
|
7
|
-
ImageContent,
|
|
8
|
-
Message,
|
|
9
|
-
Model,
|
|
10
|
-
SimpleStreamOptions,
|
|
11
|
-
StreamOptions,
|
|
12
|
-
TextContent,
|
|
13
|
-
ThinkingContent,
|
|
14
|
-
Tool,
|
|
15
|
-
ToolCall,
|
|
16
|
-
Usage,
|
|
17
|
-
} from "@earendil-works/pi-ai";
|
|
18
|
-
import { createAssistantMessageEventStream } from "@earendil-works/pi-ai";
|
|
19
|
-
import type { ApiProvider } from "@earendil-works/pi-ai/compat";
|
|
20
|
-
import {
|
|
21
|
-
type ChatHistoryData,
|
|
22
|
-
type ChatMessageData,
|
|
23
|
-
type ChatMessagePartFileData,
|
|
24
|
-
type ChatMessagePartTextData,
|
|
25
|
-
type ChatMessagePartToolCallRequestData,
|
|
26
|
-
type ChatMessagePartToolCallResultData,
|
|
27
|
-
type FileHandle,
|
|
28
|
-
type FunctionToolCallRequest,
|
|
29
|
-
type LLMLoadModelConfig,
|
|
30
|
-
type LLMPredictionStopReason,
|
|
31
|
-
type LLMRespondOpts,
|
|
32
|
-
type LLMTool,
|
|
33
|
-
LMStudioClient,
|
|
34
|
-
} from "@lmstudio/sdk";
|
|
35
|
-
import { runOverrides } from "../../core/run-overrides.js";
|
|
36
|
-
import {
|
|
37
|
-
reasoningClassForMechanism,
|
|
38
|
-
resolveModelRuntimeCapabilitiesForModel,
|
|
39
|
-
} from "../../domains/providers/model-runtime-capabilities.js";
|
|
40
|
-
import { lmStudioQuietLogger } from "../../domains/providers/runtimes/common/lmstudio-logger.js";
|
|
41
|
-
import type { ThinkingLevel } from "../../domains/providers/types/capability-flags.js";
|
|
42
|
-
import {
|
|
43
|
-
asKvCacheQuant,
|
|
44
|
-
KV_CACHE_QUANTS,
|
|
45
|
-
type LocalModelQuirks,
|
|
46
|
-
type SamplingProfile,
|
|
47
|
-
} from "../../domains/providers/types/local-model-quirks.js";
|
|
48
|
-
import { ceilChars } from "../../domains/session/context-accounting.js";
|
|
49
|
-
import { calculateEngineCost, parseEngineJsonWithRepair, parseEngineStreamingJson } from "../ai.js";
|
|
50
|
-
import { HarmonyResponseParser } from "../harmony-response.js";
|
|
51
|
-
import { createSentinelStripper } from "../strip-tokenizer-sentinels.js";
|
|
52
|
-
import { type DegradedInferenceWatchdog, startDegradedInferenceWatchdog } from "./degraded-inference.js";
|
|
53
|
-
import { coResidentContextCeiling, duplicateInstances, fitLoadContextLength } from "./lmstudio-residency.js";
|
|
54
|
-
import { openAICompletionsApiProvider } from "./openai-completions.js";
|
|
55
|
-
import { remainingContextMaxTokens } from "./output-budget.js";
|
|
56
|
-
import {
|
|
57
|
-
emitResidencyNotice,
|
|
58
|
-
type ResidencyAdapter,
|
|
59
|
-
type ResidencyPlan,
|
|
60
|
-
reconcileResidency,
|
|
61
|
-
residencyManagedFor,
|
|
62
|
-
} from "./residency.js";
|
|
63
|
-
import { withResidencyLock } from "./residency-lock.js";
|
|
64
|
-
import type { ResidentModelInfo } from "./resident-models.js";
|
|
65
|
-
import { mergeSamplingOverride } from "./sampling-overrides.js";
|
|
66
|
-
import { formatThinkingForReplay } from "./thinking-replay.js";
|
|
67
|
-
|
|
68
|
-
const EMPTY_TOOL_ARGUMENTS_ERROR =
|
|
69
|
-
"LM Studio SDK returned empty tool-call arguments; this model's chat template may not be compatible. Try the openai-compat runtime against the same gateway.";
|
|
70
|
-
|
|
71
|
-
type RuntimeLifecycle = "user-managed" | "clio-managed";
|
|
72
|
-
|
|
73
|
-
interface ClioRuntimeMetadata {
|
|
74
|
-
clio?: {
|
|
75
|
-
targetId: string;
|
|
76
|
-
runtimeId: string;
|
|
77
|
-
/** Present only when settings set the target lifecycle explicitly. */
|
|
78
|
-
lifecycle?: RuntimeLifecycle;
|
|
79
|
-
gateway?: boolean;
|
|
80
|
-
quirks?: LocalModelQuirks;
|
|
81
|
-
};
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
function normalizeBaseUrl(url: string): string {
|
|
85
|
-
const trimmed = url.endsWith("/") ? url.slice(0, -1) : url;
|
|
86
|
-
if (trimmed.startsWith("ws://") || trimmed.startsWith("wss://")) return trimmed;
|
|
87
|
-
if (trimmed.startsWith("https://")) return `wss://${trimmed.slice("https://".length)}`;
|
|
88
|
-
if (trimmed.startsWith("http://")) return `ws://${trimmed.slice("http://".length)}`;
|
|
89
|
-
return trimmed;
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
function normalizeHttpBaseUrl(url: string): string {
|
|
93
|
-
const trimmed = url.endsWith("/") ? url.slice(0, -1) : url;
|
|
94
|
-
if (trimmed.startsWith("http://") || trimmed.startsWith("https://")) return trimmed;
|
|
95
|
-
if (trimmed.startsWith("ws://")) return `http://${trimmed.slice("ws://".length)}`;
|
|
96
|
-
if (trimmed.startsWith("wss://")) return `https://${trimmed.slice("wss://".length)}`;
|
|
97
|
-
return `http://${trimmed}`;
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
// One loaded model instance in LM Studio's resident set, as the SDK socket
|
|
101
|
-
// reports it. The reconciler (residency.ts) lists residents through these
|
|
102
|
-
// entries, and a post-failure fallback swap unloads through them; loads go
|
|
103
|
-
// through the SDK's JIT model open. `identifier` is per-instance rather than
|
|
104
|
-
// per-model: LM Studio can hold two instances of one model key, and telling
|
|
105
|
-
// them apart is what lets the duplicate be released.
|
|
106
|
-
export interface ResidentModelEntry {
|
|
107
|
-
readonly modelKey: string;
|
|
108
|
-
readonly identifier?: string;
|
|
109
|
-
readonly sizeBytes?: number;
|
|
110
|
-
unload(): Promise<void>;
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
/**
|
|
114
|
-
* Collapse LM Studio's per-instance listing into the per-model view the
|
|
115
|
-
* reconciler decides on. Two instances of one model key are one resident model
|
|
116
|
-
* as far as eviction is concerned; the duplicate sweep handles the extra
|
|
117
|
-
* instance separately, before any eviction question arises.
|
|
118
|
-
*/
|
|
119
|
-
function residentModelInfos(entries: ReadonlyArray<ResidentModelEntry>): ResidentModelInfo[] {
|
|
120
|
-
const byKey = new Map<string, ResidentModelInfo>();
|
|
121
|
-
for (const entry of entries) {
|
|
122
|
-
if (byKey.has(entry.modelKey)) continue;
|
|
123
|
-
byKey.set(entry.modelKey, {
|
|
124
|
-
modelId: entry.modelKey,
|
|
125
|
-
...(entry.sizeBytes !== undefined ? { sizeBytes: entry.sizeBytes } : {}),
|
|
126
|
-
});
|
|
127
|
-
}
|
|
128
|
-
return [...byKey.values()];
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
function toolToLmStudio(tool: Tool): LLMTool {
|
|
132
|
-
const fn: LLMTool["function"] = {
|
|
133
|
-
name: tool.name,
|
|
134
|
-
parameters: tool.parameters as unknown as NonNullable<LLMTool["function"]["parameters"]>,
|
|
135
|
-
};
|
|
136
|
-
if (tool.description) fn.description = tool.description;
|
|
137
|
-
return { type: "function", function: fn };
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
type UserPart = ChatMessagePartTextData | ChatMessagePartFileData;
|
|
141
|
-
type AssistantPart = ChatMessagePartTextData | ChatMessagePartFileData | ChatMessagePartToolCallRequestData;
|
|
142
|
-
|
|
143
|
-
interface PredictionStatsLike {
|
|
144
|
-
promptTokensCount?: number;
|
|
145
|
-
predictedTokensCount?: number;
|
|
146
|
-
totalTokensCount?: number;
|
|
147
|
-
stopReason?: LLMPredictionStopReason;
|
|
148
|
-
}
|
|
149
|
-
|
|
150
|
-
interface PredictionResultLike {
|
|
151
|
-
stats: PredictionStatsLike;
|
|
152
|
-
}
|
|
153
|
-
|
|
154
|
-
interface OngoingPredictionLike {
|
|
155
|
-
result(): Promise<PredictionResultLike>;
|
|
156
|
-
}
|
|
157
|
-
|
|
158
|
-
interface LmStudioPredictionHandle {
|
|
159
|
-
respond(history: ChatHistoryData, opts: LLMRespondOpts<unknown>): OngoingPredictionLike;
|
|
160
|
-
}
|
|
161
|
-
|
|
162
|
-
interface LmStudioRunClient {
|
|
163
|
-
files: {
|
|
164
|
-
prepareImageBase64(fileName: string, contentBase64: string): Promise<FileHandle>;
|
|
165
|
-
};
|
|
166
|
-
llm: {
|
|
167
|
-
listLoaded(): Promise<ReadonlyArray<ResidentModelEntry>>;
|
|
168
|
-
model(
|
|
169
|
-
modelId: string,
|
|
170
|
-
opts: { signal: AbortSignal; verbose: boolean; config?: LLMLoadModelConfig },
|
|
171
|
-
): Promise<LmStudioPredictionHandle>;
|
|
172
|
-
};
|
|
173
|
-
}
|
|
174
|
-
|
|
175
|
-
export interface LmStudioRunDeps {
|
|
176
|
-
createClient(opts: ConstructorParameters<typeof LMStudioClient>[0]): LmStudioRunClient;
|
|
177
|
-
reconcile(adapter: ResidencyAdapter): Promise<ResidencyPlan>;
|
|
178
|
-
discoverLoadedContext(baseUrl: string, modelId: string, signal: AbortSignal): Promise<number | undefined>;
|
|
179
|
-
/** Cross-process residency-mutation serializer; defaults to the state-dir lock file. */
|
|
180
|
-
lock?<T>(targetKey: string, fn: () => Promise<T>): Promise<T>;
|
|
181
|
-
}
|
|
182
|
-
|
|
183
|
-
/**
|
|
184
|
-
* Out-of-band hints from the api-provider wrapper. `thinkingLevel` is the
|
|
185
|
-
* Clio ThinkingLevel for the in-flight turn; `runStream` resolves it through
|
|
186
|
-
* the provider-domain runtime capability layer before choosing catalog
|
|
187
|
-
* sampling. The bare `stream` path (no SimpleStreamOptions) leaves it
|
|
188
|
-
* undefined, in which case the helper falls back to the model's `reasoning`
|
|
189
|
-
* capability flag.
|
|
190
|
-
*/
|
|
191
|
-
export interface RunStreamHints {
|
|
192
|
-
thinkingLevel?: ThinkingLevel;
|
|
193
|
-
}
|
|
194
|
-
|
|
195
|
-
// Worker-process-scoped cache of LMStudioClient instances keyed on
|
|
196
|
-
// `${baseUrl}|${clientPasskey ?? ""}`. Creating a fresh client per turn
|
|
197
|
-
// allocates a new WebSocket session against the LM Studio server, and abandoning
|
|
198
|
-
// it without [Symbol.asyncDispose] leaks server-side channel state. The server
|
|
199
|
-
// then logs `Received channelSend for unknown channel` warnings on the next
|
|
200
|
-
// abort because the previous turn's controller fires after the SDK has already
|
|
201
|
-
// torn the channel down. Reusing one client across turns avoids that race and
|
|
202
|
-
// keeps the WebSocket warm for high-latency remote LM Studio hosts. The
|
|
203
|
-
// cache lives in the worker subprocess and dies with it; it is not shared
|
|
204
|
-
// across worker processes.
|
|
205
|
-
const lmStudioClientCache = new Map<string, LMStudioClient>();
|
|
206
|
-
|
|
207
|
-
function lmStudioCacheKey(baseUrl: string, clientPasskey: string | undefined): string {
|
|
208
|
-
return `${baseUrl}|${clientPasskey ?? ""}`;
|
|
209
|
-
}
|
|
210
|
-
|
|
211
|
-
export async function disposeLmStudioClients(): Promise<void> {
|
|
212
|
-
const clients = Array.from(lmStudioClientCache.values());
|
|
213
|
-
lmStudioClientCache.clear();
|
|
214
|
-
await Promise.all(
|
|
215
|
-
clients.map(async (client) => {
|
|
216
|
-
try {
|
|
217
|
-
await client[Symbol.asyncDispose]();
|
|
218
|
-
} catch {
|
|
219
|
-
// Disposal is best-effort: the worker is already shutting down,
|
|
220
|
-
// and a noisy close should not block process exit.
|
|
221
|
-
}
|
|
222
|
-
}),
|
|
223
|
-
);
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
function getOrCreateLmStudioClient(
|
|
227
|
-
opts: ConstructorParameters<typeof LMStudioClient>[0],
|
|
228
|
-
create: (o: ConstructorParameters<typeof LMStudioClient>[0]) => LmStudioRunClient,
|
|
229
|
-
): LmStudioRunClient {
|
|
230
|
-
const baseUrl = opts?.baseUrl ?? "";
|
|
231
|
-
const passkey = opts?.clientPasskey;
|
|
232
|
-
const key = lmStudioCacheKey(baseUrl, passkey);
|
|
233
|
-
const cached = lmStudioClientCache.get(key);
|
|
234
|
-
if (cached) return cached as unknown as LmStudioRunClient;
|
|
235
|
-
const created = create(opts);
|
|
236
|
-
// `created` is the structural LmStudioRunClient view of an LMStudioClient.
|
|
237
|
-
// In production `defaultRunDeps.createClient` returns `new LMStudioClient(...)`,
|
|
238
|
-
// which satisfies the cache value type. Tests inject fakes through the
|
|
239
|
-
// `deps.createClient` parameter and bypass this cache entirely (see
|
|
240
|
-
// `runStream`'s caller-provided-deps branch below).
|
|
241
|
-
lmStudioClientCache.set(key, created as unknown as LMStudioClient);
|
|
242
|
-
return created;
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
const defaultRunDeps: LmStudioRunDeps = {
|
|
246
|
-
createClient: (opts) =>
|
|
247
|
-
getOrCreateLmStudioClient(opts, (o) => new LMStudioClient({ ...(o ?? {}), logger: lmStudioQuietLogger })),
|
|
248
|
-
reconcile: reconcileResidency,
|
|
249
|
-
discoverLoadedContext: discoverLoadedContextLength,
|
|
250
|
-
lock: withResidencyLock,
|
|
251
|
-
};
|
|
252
|
-
|
|
253
|
-
function fileHandleToPart(handle: FileHandle): ChatMessagePartFileData {
|
|
254
|
-
return {
|
|
255
|
-
type: "file",
|
|
256
|
-
name: handle.name,
|
|
257
|
-
identifier: handle.identifier,
|
|
258
|
-
sizeBytes: handle.sizeBytes,
|
|
259
|
-
fileType: handle.type,
|
|
260
|
-
};
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
function imageFileName(mimeType: string, index: number): string {
|
|
264
|
-
const slash = mimeType.indexOf("/");
|
|
265
|
-
const ext = slash >= 0 ? mimeType.slice(slash + 1) : "png";
|
|
266
|
-
const safeExt = ext.replace(/[^a-zA-Z0-9]/g, "") || "png";
|
|
267
|
-
return `clio-image-${index}.${safeExt}`;
|
|
268
|
-
}
|
|
269
|
-
|
|
270
|
-
async function userMessage(
|
|
271
|
-
client: Pick<LmStudioRunClient, "files">,
|
|
272
|
-
content: string | (TextContent | ImageContent)[],
|
|
273
|
-
imageCounter: { next: number },
|
|
274
|
-
): Promise<ChatMessageData> {
|
|
275
|
-
if (typeof content === "string") {
|
|
276
|
-
return { role: "user", content: [{ type: "text", text: content }] };
|
|
277
|
-
}
|
|
278
|
-
const parts: UserPart[] = [];
|
|
279
|
-
for (const block of content) {
|
|
280
|
-
if (block.type === "text") {
|
|
281
|
-
parts.push({ type: "text", text: block.text });
|
|
282
|
-
continue;
|
|
283
|
-
}
|
|
284
|
-
if (block.type === "image") {
|
|
285
|
-
const fileName = imageFileName(block.mimeType, imageCounter.next++);
|
|
286
|
-
let handle: FileHandle;
|
|
287
|
-
try {
|
|
288
|
-
handle = await client.files.prepareImageBase64(fileName, block.data);
|
|
289
|
-
} catch (err) {
|
|
290
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
291
|
-
throw new Error(`LM Studio prepareImage failed for ${fileName}: ${msg}`);
|
|
292
|
-
}
|
|
293
|
-
parts.push(fileHandleToPart(handle));
|
|
294
|
-
}
|
|
295
|
-
}
|
|
296
|
-
return { role: "user", content: parts };
|
|
297
|
-
}
|
|
298
|
-
|
|
299
|
-
export function assistantMessage(
|
|
300
|
-
content: AssistantMessage["content"],
|
|
301
|
-
opts?: { harmony?: boolean; preserveThinking?: boolean },
|
|
302
|
-
): ChatMessageData {
|
|
303
|
-
const parts: AssistantPart[] = [];
|
|
304
|
-
const thinkingParts: string[] = [];
|
|
305
|
-
const preserveThinking = opts?.preserveThinking ?? true;
|
|
306
|
-
for (const block of content) {
|
|
307
|
-
if (block.type === "text") {
|
|
308
|
-
parts.push({ type: "text", text: block.text });
|
|
309
|
-
} else if (block.type === "toolCall") {
|
|
310
|
-
const req: FunctionToolCallRequest = {
|
|
311
|
-
type: "function",
|
|
312
|
-
name: block.name,
|
|
313
|
-
arguments: block.arguments,
|
|
314
|
-
};
|
|
315
|
-
if (block.id) req.id = block.id;
|
|
316
|
-
parts.push({ type: "toolCallRequest", toolCallRequest: req });
|
|
317
|
-
} else if (preserveThinking && block.type === "thinking") {
|
|
318
|
-
const thinkingVal = (block as ThinkingContent).thinking;
|
|
319
|
-
if (thinkingVal) {
|
|
320
|
-
thinkingParts.push(thinkingVal);
|
|
321
|
-
}
|
|
322
|
-
}
|
|
323
|
-
}
|
|
324
|
-
if (thinkingParts.length > 0) {
|
|
325
|
-
const joined = thinkingParts.join("\n");
|
|
326
|
-
const harmony = opts?.harmony ?? false;
|
|
327
|
-
parts.unshift({ type: "text", text: formatThinkingForReplay(joined, { harmony }) });
|
|
328
|
-
}
|
|
329
|
-
return { role: "assistant", content: parts };
|
|
330
|
-
}
|
|
331
|
-
|
|
332
|
-
function toolResultMessage(msg: Extract<Message, { role: "toolResult" }>): ChatMessageData {
|
|
333
|
-
const text = msg.content
|
|
334
|
-
.filter((b): b is TextContent => b.type === "text")
|
|
335
|
-
.map((b) => b.text)
|
|
336
|
-
.join("\n");
|
|
337
|
-
const result: ChatMessagePartToolCallResultData = {
|
|
338
|
-
type: "toolCallResult",
|
|
339
|
-
content: text,
|
|
340
|
-
toolCallId: msg.toolCallId,
|
|
341
|
-
};
|
|
342
|
-
return { role: "tool", content: [result] };
|
|
343
|
-
}
|
|
344
|
-
|
|
345
|
-
async function buildChatHistory(
|
|
346
|
-
client: Pick<LmStudioRunClient, "files">,
|
|
347
|
-
context: Context,
|
|
348
|
-
opts?: { harmony?: boolean; preserveThinking?: boolean },
|
|
349
|
-
): Promise<ChatHistoryData> {
|
|
350
|
-
const messages: ChatMessageData[] = [];
|
|
351
|
-
const imageCounter = { next: 0 };
|
|
352
|
-
if (context.systemPrompt && context.systemPrompt.length > 0) {
|
|
353
|
-
messages.push({ role: "system", content: [{ type: "text", text: context.systemPrompt }] });
|
|
354
|
-
}
|
|
355
|
-
for (const msg of context.messages) {
|
|
356
|
-
if (msg.role === "user") messages.push(await userMessage(client, msg.content, imageCounter));
|
|
357
|
-
else if (msg.role === "assistant") messages.push(assistantMessage(msg.content, opts));
|
|
358
|
-
else if (msg.role === "toolResult") messages.push(toolResultMessage(msg));
|
|
359
|
-
}
|
|
360
|
-
return { messages };
|
|
361
|
-
}
|
|
362
|
-
|
|
363
|
-
function mapStopReason(
|
|
364
|
-
reason: LLMPredictionStopReason | undefined,
|
|
365
|
-
aborted: boolean,
|
|
366
|
-
hadToolCall: boolean,
|
|
367
|
-
): AssistantMessage["stopReason"] {
|
|
368
|
-
if (aborted || reason === "userStopped") return "aborted";
|
|
369
|
-
if (reason === "failed" || reason === "modelUnloaded") return "error";
|
|
370
|
-
if (reason === "toolCalls" || hadToolCall) return "toolUse";
|
|
371
|
-
if (reason === "maxPredictedTokensReached" || reason === "contextLengthReached") return "length";
|
|
372
|
-
return "stop";
|
|
373
|
-
}
|
|
374
|
-
|
|
375
|
-
function asDoneReason(
|
|
376
|
-
reason: AssistantMessage["stopReason"],
|
|
377
|
-
): Extract<AssistantMessage["stopReason"], "stop" | "length" | "toolUse"> {
|
|
378
|
-
if (reason === "length" || reason === "toolUse") return reason;
|
|
379
|
-
return "stop";
|
|
380
|
-
}
|
|
381
|
-
|
|
382
|
-
interface PendingToolCall {
|
|
383
|
-
contentIndex: number;
|
|
384
|
-
name: string;
|
|
385
|
-
argBuffer: string;
|
|
386
|
-
assistantIndex: number;
|
|
387
|
-
toolCallSlot: ToolCall;
|
|
388
|
-
}
|
|
389
|
-
|
|
390
|
-
class LmStudioToolCallExtractionError extends Error {
|
|
391
|
-
constructor() {
|
|
392
|
-
super(EMPTY_TOOL_ARGUMENTS_ERROR);
|
|
393
|
-
this.name = "LmStudioToolCallExtractionError";
|
|
394
|
-
}
|
|
395
|
-
}
|
|
396
|
-
|
|
397
|
-
function hasNonEmptyGeneratedContent(output: AssistantMessage): boolean {
|
|
398
|
-
return output.content.some((block) => {
|
|
399
|
-
if (block.type === "text") return block.text.trim().length > 0;
|
|
400
|
-
if (block.type === "thinking") return block.thinking.trim().length > 0;
|
|
401
|
-
return false;
|
|
402
|
-
});
|
|
403
|
-
}
|
|
404
|
-
|
|
405
|
-
function hasEmptyToolArguments(value: unknown): boolean {
|
|
406
|
-
if (value === undefined || value === null) return true;
|
|
407
|
-
if (typeof value === "string") return value.trim().length === 0;
|
|
408
|
-
if (value && typeof value === "object" && !Array.isArray(value)) return Object.keys(value).length === 0;
|
|
409
|
-
return false;
|
|
410
|
-
}
|
|
411
|
-
|
|
412
|
-
const MAX_AUTOMATIC_LOAD_CONTEXT = 262_144;
|
|
413
|
-
const MIN_AUTOMATIC_LOAD_CONTEXT = 32_768;
|
|
414
|
-
|
|
415
|
-
function automaticLoadContextLength(model: Model<"lmstudio-native">): number {
|
|
416
|
-
// When `model.contextWindow` is set (from a knowledge-base entry or an
|
|
417
|
-
// explicit `--context-window` override on the target) it is the
|
|
418
|
-
// authoritative budget for the load. Earlier versions also clamped against
|
|
419
|
-
// `maxTokens * 2`, but agent workloads are dominated by *input* tokens, so
|
|
420
|
-
// that clamp silently undersized the KV cache (e.g. 262K → 65K).
|
|
421
|
-
if (model.contextWindow > 0) {
|
|
422
|
-
return Math.min(model.contextWindow, MAX_AUTOMATIC_LOAD_CONTEXT);
|
|
423
|
-
}
|
|
424
|
-
const requestedOutput = model.maxTokens > 0 ? model.maxTokens : MIN_AUTOMATIC_LOAD_CONTEXT;
|
|
425
|
-
const target = Math.max(MIN_AUTOMATIC_LOAD_CONTEXT, requestedOutput * 2);
|
|
426
|
-
return Math.min(target, MAX_AUTOMATIC_LOAD_CONTEXT);
|
|
427
|
-
}
|
|
428
|
-
|
|
429
|
-
function clioQuirks(model: Model<Api>): LocalModelQuirks | undefined {
|
|
430
|
-
return (model as Model<Api> & ClioRuntimeMetadata).clio?.quirks;
|
|
431
|
-
}
|
|
432
|
-
|
|
433
|
-
function pickSamplingProfile(
|
|
434
|
-
quirks: LocalModelQuirks | undefined,
|
|
435
|
-
thinkingActive: boolean,
|
|
436
|
-
): SamplingProfile | undefined {
|
|
437
|
-
const sampling = quirks?.sampling;
|
|
438
|
-
const profile = sampling ? (thinkingActive ? (sampling.thinking ?? sampling.instruct) : sampling.instruct) : undefined;
|
|
439
|
-
return mergeSamplingOverride(profile);
|
|
440
|
-
}
|
|
441
|
-
|
|
442
|
-
function thinkingLevelFromHintOrModel(hints: RunStreamHints, model: Model<"lmstudio-native">): ThinkingLevel {
|
|
443
|
-
if (hints.thinkingLevel) return hints.thinkingLevel;
|
|
444
|
-
return model.reasoning === true ? "medium" : "off";
|
|
445
|
-
}
|
|
446
|
-
|
|
447
|
-
const VALID_ENV_KV_CACHE_QUANTS: ReadonlySet<string> = new Set(KV_CACHE_QUANTS);
|
|
448
|
-
|
|
449
|
-
export function loadModelConfig(
|
|
450
|
-
model: Model<"lmstudio-native">,
|
|
451
|
-
overrides?: { contextLength?: number },
|
|
452
|
-
): LLMLoadModelConfig {
|
|
453
|
-
// LM Studio's REST `/api/v1/models/load` does not expose KV cache quant or
|
|
454
|
-
// fp16 KV options; those only round-trip through the SDK's WebSocket
|
|
455
|
-
// protocol (LLMLoadModelConfig.llama{K,V}CacheQuantizationType,
|
|
456
|
-
// useFp16ForKVCache). Honor catalog quirks here so dense gemma-4 NVFP4 loads
|
|
457
|
-
// fit at f16 KV with parallel=1 and drop to q8_0 KV at parallel=4 without
|
|
458
|
-
// the user editing settings.yaml by hand.
|
|
459
|
-
const config: LLMLoadModelConfig = {
|
|
460
|
-
contextLength: overrides?.contextLength ?? automaticLoadContextLength(model),
|
|
461
|
-
flashAttention: true,
|
|
462
|
-
gpu: { ratio: "max" },
|
|
463
|
-
gpuStrictVramCap: true,
|
|
464
|
-
offloadKVCacheToGpu: true,
|
|
465
|
-
};
|
|
466
|
-
const kvCache = clioQuirks(model)?.kvCache;
|
|
467
|
-
if (kvCache) {
|
|
468
|
-
if (kvCache.kQuant !== undefined && kvCache.kQuant !== false) config.llamaKCacheQuantizationType = kvCache.kQuant;
|
|
469
|
-
if (kvCache.vQuant !== undefined && kvCache.vQuant !== false) config.llamaVCacheQuantizationType = kvCache.vQuant;
|
|
470
|
-
if (kvCache.useFp16 !== undefined) config.useFp16ForKVCache = kvCache.useFp16;
|
|
471
|
-
}
|
|
472
|
-
// One-run CLI override (clio-coder run --kv-cache-mode), delivered over the
|
|
473
|
-
// run-overrides transport; see core/run-overrides.ts.
|
|
474
|
-
const kvCacheModeOverride = runOverrides().kvCacheMode;
|
|
475
|
-
if (kvCacheModeOverride) {
|
|
476
|
-
if (kvCacheModeOverride === "f16") {
|
|
477
|
-
config.llamaKCacheQuantizationType = "f16";
|
|
478
|
-
config.llamaVCacheQuantizationType = "f16";
|
|
479
|
-
config.useFp16ForKVCache = true;
|
|
480
|
-
} else if (kvCacheModeOverride === "f32") {
|
|
481
|
-
config.llamaKCacheQuantizationType = "f32";
|
|
482
|
-
config.llamaVCacheQuantizationType = "f32";
|
|
483
|
-
config.useFp16ForKVCache = false;
|
|
484
|
-
} else if (kvCacheModeOverride === "none" || kvCacheModeOverride === "false") {
|
|
485
|
-
delete config.llamaKCacheQuantizationType;
|
|
486
|
-
delete config.llamaVCacheQuantizationType;
|
|
487
|
-
delete config.useFp16ForKVCache;
|
|
488
|
-
} else {
|
|
489
|
-
const quant = asKvCacheQuant(kvCacheModeOverride);
|
|
490
|
-
if (quant !== undefined && quant !== false && VALID_ENV_KV_CACHE_QUANTS.has(quant)) {
|
|
491
|
-
config.llamaKCacheQuantizationType = quant;
|
|
492
|
-
config.llamaVCacheQuantizationType = quant;
|
|
493
|
-
config.useFp16ForKVCache = false;
|
|
494
|
-
} else {
|
|
495
|
-
process.stderr.write(`clio: ignoring invalid kv-cache-mode override '${kvCacheModeOverride}'\n`);
|
|
496
|
-
}
|
|
497
|
-
}
|
|
498
|
-
}
|
|
499
|
-
return config;
|
|
500
|
-
}
|
|
501
|
-
|
|
502
|
-
interface LmStudioApiV0ModelEntry {
|
|
503
|
-
id?: unknown;
|
|
504
|
-
loaded_context_length?: unknown;
|
|
505
|
-
}
|
|
506
|
-
|
|
507
|
-
interface LmStudioApiV0ModelsResponse {
|
|
508
|
-
data?: unknown;
|
|
509
|
-
}
|
|
510
|
-
|
|
511
|
-
interface LmStudioApiV1ModelEntry {
|
|
512
|
-
key?: unknown;
|
|
513
|
-
loaded_instances?: unknown;
|
|
514
|
-
}
|
|
515
|
-
|
|
516
|
-
interface LmStudioApiV1ModelsResponse {
|
|
517
|
-
models?: unknown;
|
|
518
|
-
}
|
|
519
|
-
|
|
520
|
-
function positiveNumber(value: unknown): number | undefined {
|
|
521
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : undefined;
|
|
522
|
-
}
|
|
523
|
-
|
|
524
|
-
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
525
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
526
|
-
}
|
|
527
|
-
|
|
528
|
-
function apiV0ModelEntries(payload: LmStudioApiV0ModelsResponse | undefined): LmStudioApiV0ModelEntry[] {
|
|
529
|
-
if (!Array.isArray(payload?.data)) return [];
|
|
530
|
-
return payload.data.filter((entry): entry is LmStudioApiV0ModelEntry => isRecord(entry));
|
|
531
|
-
}
|
|
532
|
-
|
|
533
|
-
function apiV1ModelEntries(payload: LmStudioApiV1ModelsResponse | undefined): LmStudioApiV1ModelEntry[] {
|
|
534
|
-
if (!Array.isArray(payload?.models)) return [];
|
|
535
|
-
return payload.models.filter((entry): entry is LmStudioApiV1ModelEntry => isRecord(entry));
|
|
536
|
-
}
|
|
537
|
-
|
|
538
|
-
function loadedContextFromV1Instance(value: unknown): number | undefined {
|
|
539
|
-
if (!isRecord(value) || !isRecord(value.config)) return undefined;
|
|
540
|
-
return positiveNumber(value.config.context_length);
|
|
541
|
-
}
|
|
542
|
-
|
|
543
|
-
function loadedContextFromV1Entry(entry: LmStudioApiV1ModelEntry): number | undefined {
|
|
544
|
-
if (!Array.isArray(entry.loaded_instances)) return undefined;
|
|
545
|
-
for (const instance of entry.loaded_instances) {
|
|
546
|
-
const contextLength = loadedContextFromV1Instance(instance);
|
|
547
|
-
if (contextLength !== undefined) return contextLength;
|
|
548
|
-
}
|
|
549
|
-
return undefined;
|
|
550
|
-
}
|
|
551
|
-
|
|
552
|
-
async function discoverLoadedContextLength(
|
|
553
|
-
baseUrl: string,
|
|
554
|
-
modelId: string,
|
|
555
|
-
signal: AbortSignal,
|
|
556
|
-
): Promise<number | undefined> {
|
|
557
|
-
const base = normalizeHttpBaseUrl(baseUrl);
|
|
558
|
-
const v1 = await discoverLoadedContextFromV1(base, modelId, signal);
|
|
559
|
-
if (v1 !== undefined) return v1;
|
|
560
|
-
return discoverLoadedContextFromV0(base, modelId, signal);
|
|
561
|
-
}
|
|
562
|
-
|
|
563
|
-
async function fetchLmStudioJson<T>(url: string, signal: AbortSignal): Promise<T | undefined> {
|
|
564
|
-
const controller = new AbortController();
|
|
565
|
-
const timer = setTimeout(() => controller.abort(), 1500);
|
|
566
|
-
const onAbort = () => controller.abort();
|
|
567
|
-
if (signal.aborted) controller.abort();
|
|
568
|
-
else signal.addEventListener("abort", onAbort, { once: true });
|
|
569
|
-
try {
|
|
570
|
-
const response = await fetch(url, { signal: controller.signal });
|
|
571
|
-
if (!response.ok) return undefined;
|
|
572
|
-
return (await response.json()) as T;
|
|
573
|
-
} catch {
|
|
574
|
-
return undefined;
|
|
575
|
-
} finally {
|
|
576
|
-
clearTimeout(timer);
|
|
577
|
-
signal.removeEventListener("abort", onAbort);
|
|
578
|
-
}
|
|
579
|
-
}
|
|
580
|
-
|
|
581
|
-
async function discoverLoadedContextFromV1(
|
|
582
|
-
baseUrl: string,
|
|
583
|
-
modelId: string,
|
|
584
|
-
signal: AbortSignal,
|
|
585
|
-
): Promise<number | undefined> {
|
|
586
|
-
const payload = await fetchLmStudioJson<LmStudioApiV1ModelsResponse>(`${baseUrl}/api/v1/models`, signal);
|
|
587
|
-
const entry = apiV1ModelEntries(payload).find((row) => row.key === modelId);
|
|
588
|
-
return entry ? loadedContextFromV1Entry(entry) : undefined;
|
|
589
|
-
}
|
|
590
|
-
|
|
591
|
-
async function discoverLoadedContextFromV0(
|
|
592
|
-
baseUrl: string,
|
|
593
|
-
modelId: string,
|
|
594
|
-
signal: AbortSignal,
|
|
595
|
-
): Promise<number | undefined> {
|
|
596
|
-
const payload = await fetchLmStudioJson<LmStudioApiV0ModelsResponse>(`${baseUrl}/api/v0/models`, signal);
|
|
597
|
-
const entry = apiV0ModelEntries(payload).find((row) => row.id === modelId);
|
|
598
|
-
return positiveNumber(entry?.loaded_context_length);
|
|
599
|
-
}
|
|
600
|
-
|
|
601
|
-
interface ResolvedRuntimeMetadata {
|
|
602
|
-
targetId: string;
|
|
603
|
-
runtimeId: string;
|
|
604
|
-
/** Explicit target lifecycle from settings; absent means Clio manages by default. */
|
|
605
|
-
lifecycle?: RuntimeLifecycle;
|
|
606
|
-
gateway: boolean;
|
|
607
|
-
}
|
|
608
|
-
|
|
609
|
-
function runtimeMetadata(model: Model<Api>): ResolvedRuntimeMetadata {
|
|
610
|
-
const metadata = (model as Model<Api> & ClioRuntimeMetadata).clio;
|
|
611
|
-
return {
|
|
612
|
-
targetId: metadata?.targetId ?? model.provider,
|
|
613
|
-
runtimeId: metadata?.runtimeId ?? model.provider,
|
|
614
|
-
...(metadata?.lifecycle ? { lifecycle: metadata.lifecycle } : {}),
|
|
615
|
-
gateway: metadata?.gateway ?? false,
|
|
616
|
-
};
|
|
617
|
-
}
|
|
618
|
-
|
|
619
|
-
/**
|
|
620
|
-
* Errnos raised before the server answered anything. A load that never reached
|
|
621
|
-
* the model cannot have been refused for its size, so the sizing advice below
|
|
622
|
-
* is wrong in both halves for these: the cause is the route, and no
|
|
623
|
-
* contextWindow override fixes a host that did not respond.
|
|
624
|
-
*/
|
|
625
|
-
const CONNECT_FAILURE_RE = /\b(ENETUNREACH|EHOSTUNREACH|ECONNREFUSED|ETIMEDOUT|ENOTFOUND|ECONNRESET|EAI_AGAIN)\b/;
|
|
626
|
-
|
|
627
|
-
function isConnectFailure(err: unknown, cause: string): boolean {
|
|
628
|
-
const code = typeof err === "object" && err !== null ? (err as { code?: unknown }).code : undefined;
|
|
629
|
-
if (typeof code === "string" && CONNECT_FAILURE_RE.test(code)) return true;
|
|
630
|
-
return CONNECT_FAILURE_RE.test(cause);
|
|
631
|
-
}
|
|
632
|
-
|
|
633
|
-
export function describeLoadFailure(
|
|
634
|
-
baseUrl: string,
|
|
635
|
-
model: Model<"lmstudio-native">,
|
|
636
|
-
loadConfig: LLMLoadModelConfig | undefined,
|
|
637
|
-
requestedMaxTokens: number | false | undefined,
|
|
638
|
-
err: unknown,
|
|
639
|
-
): string {
|
|
640
|
-
const metadata = runtimeMetadata(model);
|
|
641
|
-
const cause = err instanceof Error ? err.message : String(err);
|
|
642
|
-
if (isConnectFailure(err, cause)) {
|
|
643
|
-
return [
|
|
644
|
-
`LM Studio at ${baseUrl} did not answer for target '${metadata.targetId}' model '${model.id}'.`,
|
|
645
|
-
"The connection failed before a load was attempted, so this is reachability rather than model sizing.",
|
|
646
|
-
"Check that the server is running and that this host can reach it, then retry.",
|
|
647
|
-
`SDK error: ${cause}`,
|
|
648
|
-
].join(" ");
|
|
649
|
-
}
|
|
650
|
-
const context = loadConfig?.contextLength ?? model.contextWindow;
|
|
651
|
-
const output = requestedMaxTokens === false || requestedMaxTokens === undefined ? model.maxTokens : requestedMaxTokens;
|
|
652
|
-
return [
|
|
653
|
-
`LM Studio could not load target '${metadata.targetId}' model '${model.id}' at ${baseUrl}.`,
|
|
654
|
-
`Requested context ${context} and output ${output}.`,
|
|
655
|
-
`Likely cause: VRAM pressure or a context length above the quantized model/server limit.`,
|
|
656
|
-
"Try a lower contextWindow/maxTokens override, a smaller quant/tier, or openai-compat against the same LM Studio gateway when the model is already user-managed.",
|
|
657
|
-
`SDK error: ${cause}`,
|
|
658
|
-
].join(" ");
|
|
659
|
-
}
|
|
660
|
-
|
|
661
|
-
/**
|
|
662
|
-
* LM Studio's native SDK carries no thinking control. Measured against SDK
|
|
663
|
-
* 1.5.0 and a live server on 2026-08-11: `reasoning_effort`,
|
|
664
|
-
* `chat_template_kwargs`, and every `raw` KVConfig spelling are accepted and
|
|
665
|
-
* ignored (a deliberately bogus key behaved identically), and prompt-level
|
|
666
|
-
* markers such as `/no_think` change nothing. The same server's
|
|
667
|
-
* OpenAI-compatible port honours `reasoning_effort: "none"` and suppresses
|
|
668
|
-
* reasoning outright. Predictions therefore run over HTTP, and the SDK keeps
|
|
669
|
-
* the work it is the only surface for: listing, loading, and unloading models.
|
|
670
|
-
* Set CLIO_CODER_LMSTUDIO_SDK_PREDICT=1 to send predictions over the SDK again.
|
|
671
|
-
*/
|
|
672
|
-
function sdkPredictionEnabled(): boolean {
|
|
673
|
-
return process.env.CLIO_CODER_LMSTUDIO_SDK_PREDICT === "1";
|
|
674
|
-
}
|
|
675
|
-
|
|
676
|
-
/**
|
|
677
|
-
* Project the model onto LM Studio's OpenAI-compatible surface. Identity,
|
|
678
|
-
* catalog metadata, and budgets carry over verbatim, so capability resolution
|
|
679
|
-
* still keys on runtime id `lmstudio-native` and picks LM Studio's wire
|
|
680
|
-
* spelling for the thinking control. Only the transport changes.
|
|
681
|
-
*/
|
|
682
|
-
function toOpenAICompletionsModel(model: Model<"lmstudio-native">): Model<"openai-completions"> {
|
|
683
|
-
const projected = {
|
|
684
|
-
...model,
|
|
685
|
-
api: "openai-completions",
|
|
686
|
-
baseUrl: `${normalizeHttpBaseUrl(model.baseUrl)}/v1`,
|
|
687
|
-
// Matches what every other local OpenAI-compatible target synthesizes.
|
|
688
|
-
// Clio injects the thinking fields from its own `onPayload`, so pi-ai's
|
|
689
|
-
// reasoning-effort path stays off here rather than writing a second,
|
|
690
|
-
// unresolved spelling into the same body.
|
|
691
|
-
compat: {
|
|
692
|
-
supportsStore: false,
|
|
693
|
-
supportsDeveloperRole: false,
|
|
694
|
-
supportsReasoningEffort: false,
|
|
695
|
-
supportsUsageInStreaming: true,
|
|
696
|
-
maxTokensField: "max_tokens",
|
|
697
|
-
supportsStrictMode: false,
|
|
698
|
-
},
|
|
699
|
-
} as unknown as Model<"openai-completions">;
|
|
700
|
-
return projected;
|
|
701
|
-
}
|
|
702
|
-
|
|
703
|
-
/**
|
|
704
|
-
* Forward one prediction over LM Studio's OpenAI-compatible port and republish
|
|
705
|
-
* its events on this stream. Residency has already run, so the model is loaded
|
|
706
|
-
* with the context length Clio asked for and the output budget was computed
|
|
707
|
-
* against the window the server actually has open.
|
|
708
|
-
*/
|
|
709
|
-
async function pumpOpenAICompletions(
|
|
710
|
-
stream: AssistantMessageEventStream,
|
|
711
|
-
model: Model<"lmstudio-native">,
|
|
712
|
-
context: Context,
|
|
713
|
-
options: StreamOptions | undefined,
|
|
714
|
-
hints: RunStreamHints,
|
|
715
|
-
maxTokens: number,
|
|
716
|
-
onGeneratedChars?: (chars: number) => void,
|
|
717
|
-
): Promise<void> {
|
|
718
|
-
const simple: SimpleStreamOptions = {
|
|
719
|
-
...(options ?? {}),
|
|
720
|
-
maxTokens,
|
|
721
|
-
// The SDK treats a passkey as optional and an unsecured LM Studio server
|
|
722
|
-
// accepts any bearer, but the HTTP client refuses to build a request with
|
|
723
|
-
// no key at all. `lm-studio` is LM Studio's own documented placeholder.
|
|
724
|
-
apiKey: options?.apiKey ?? "lm-studio",
|
|
725
|
-
...(hints.thinkingLevel && hints.thinkingLevel !== "off" ? { reasoning: hints.thinkingLevel } : {}),
|
|
726
|
-
};
|
|
727
|
-
for await (const event of openAICompletionsApiProvider.streamSimple(
|
|
728
|
-
toOpenAICompletionsModel(model),
|
|
729
|
-
context,
|
|
730
|
-
simple,
|
|
731
|
-
)) {
|
|
732
|
-
if (onGeneratedChars && (event.type === "text_delta" || event.type === "thinking_delta")) {
|
|
733
|
-
onGeneratedChars(event.delta.length);
|
|
734
|
-
}
|
|
735
|
-
stream.push(event);
|
|
736
|
-
}
|
|
737
|
-
stream.end();
|
|
738
|
-
}
|
|
739
|
-
|
|
740
|
-
export function runStream(
|
|
741
|
-
model: Model<"lmstudio-native">,
|
|
742
|
-
context: Context,
|
|
743
|
-
options: StreamOptions | undefined,
|
|
744
|
-
deps: LmStudioRunDeps = defaultRunDeps,
|
|
745
|
-
hints: RunStreamHints = {},
|
|
746
|
-
): AssistantMessageEventStream {
|
|
747
|
-
const stream: AssistantMessageEventStream = createAssistantMessageEventStream();
|
|
748
|
-
const output: AssistantMessage = {
|
|
749
|
-
role: "assistant",
|
|
750
|
-
content: [],
|
|
751
|
-
api: model.api,
|
|
752
|
-
provider: model.provider,
|
|
753
|
-
model: model.id,
|
|
754
|
-
// `cacheRead` is deliberately absent. LM Studio reports no cached-token
|
|
755
|
-
// stat on either surface: the SDK's PredictionStats has no such field and
|
|
756
|
-
// its OpenAI-compat `usage` carries no `prompt_tokens_details`. Reporting 0
|
|
757
|
-
// reads as "measured no reuse" when the truth is "not measured", which is
|
|
758
|
-
// how #55 mistook a working prompt cache for a broken one. Undefined until
|
|
759
|
-
// a response actually carries the stat; consumers already treat it as
|
|
760
|
-
// optional.
|
|
761
|
-
usage: {
|
|
762
|
-
input: 0,
|
|
763
|
-
output: 0,
|
|
764
|
-
cacheWrite: 0,
|
|
765
|
-
totalTokens: 0,
|
|
766
|
-
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
767
|
-
} as Usage,
|
|
768
|
-
stopReason: "stop",
|
|
769
|
-
timestamp: Date.now(),
|
|
770
|
-
};
|
|
771
|
-
const controller = new AbortController();
|
|
772
|
-
const signal = options?.signal;
|
|
773
|
-
let aborted = signal?.aborted === true;
|
|
774
|
-
// `predictionDone` flips to true the moment `prediction.result()` resolves
|
|
775
|
-
// successfully, before any post-result work. Once it is set we must not call
|
|
776
|
-
// `controller.abort()` again: the SDK has already closed its channel cleanly,
|
|
777
|
-
// and a late abort raises "Received channelSend for unknown channel" on the
|
|
778
|
-
// LM Studio server. `controllerAborted` collapses repeated abort signals so
|
|
779
|
-
// `controller.abort()` fires at most once.
|
|
780
|
-
let predictionDone = false;
|
|
781
|
-
let controllerAborted = false;
|
|
782
|
-
const abortControllerOnce = () => {
|
|
783
|
-
if (controllerAborted) return;
|
|
784
|
-
controllerAborted = true;
|
|
785
|
-
controller.abort();
|
|
786
|
-
};
|
|
787
|
-
const onAbort = () => {
|
|
788
|
-
aborted = true;
|
|
789
|
-
if (predictionDone) return;
|
|
790
|
-
abortControllerOnce();
|
|
791
|
-
};
|
|
792
|
-
if (signal && !signal.aborted) signal.addEventListener("abort", onAbort, { once: true });
|
|
793
|
-
else if (aborted) abortControllerOnce();
|
|
794
|
-
// Started once the model handle is open, so a slow first load is never
|
|
795
|
-
// mistaken for slow generation, and stopped in the `finally` below so a
|
|
796
|
-
// failed turn leaves no timer behind.
|
|
797
|
-
let degradedWatchdog: DegradedInferenceWatchdog | null = null;
|
|
798
|
-
(async () => {
|
|
799
|
-
try {
|
|
800
|
-
if (aborted) throw new Error("Request was aborted");
|
|
801
|
-
const baseUrl = normalizeBaseUrl(model.baseUrl);
|
|
802
|
-
const clientOpts: ConstructorParameters<typeof LMStudioClient>[0] = { baseUrl };
|
|
803
|
-
const passkey = options?.apiKey;
|
|
804
|
-
if (passkey) clientOpts.clientPasskey = passkey;
|
|
805
|
-
const client = deps.createClient(clientOpts);
|
|
806
|
-
const metadata = runtimeMetadata(model);
|
|
807
|
-
const verbose = process.env.CLIO_CODER_RUNTIME_VERBOSE === "1";
|
|
808
|
-
const loadedContextWindow = await deps.discoverLoadedContext(baseUrl, model.id, controller.signal);
|
|
809
|
-
const budgetLimits = loadedContextWindow !== undefined ? { contextWindow: loadedContextWindow } : undefined;
|
|
810
|
-
const requestedMaxTokens = remainingContextMaxTokens(model, context, options, budgetLimits);
|
|
811
|
-
const requestedLoadContext = loadModelConfig(model).contextLength ?? model.contextWindow;
|
|
812
|
-
const targetKey = `lmstudio-native|${baseUrl}`;
|
|
813
|
-
// One reconciler decides LM Studio residency for both the interactive
|
|
814
|
-
// and headless paths. LM Studio loads just-in-time, so the plan never
|
|
815
|
-
// evicts up front: Clio attempts the co-resident load first and swaps the
|
|
816
|
-
// plan's ranked fallback candidates only after a load failure.
|
|
817
|
-
// `loadedEntries` captures the SDK handles so the duplicate sweep, the
|
|
818
|
-
// context-fit clamp, and a fallback swap all reuse one listLoaded
|
|
819
|
-
// round-trip.
|
|
820
|
-
let loadedEntries: ReadonlyArray<ResidentModelEntry> = [];
|
|
821
|
-
const plan = await deps.reconcile({
|
|
822
|
-
targetKey,
|
|
823
|
-
targetId: metadata.targetId,
|
|
824
|
-
runtimeId: "lmstudio-native",
|
|
825
|
-
keepModelId: model.id,
|
|
826
|
-
managed: residencyManagedFor(metadata.lifecycle),
|
|
827
|
-
strategy: "jit",
|
|
828
|
-
contextLength: requestedLoadContext,
|
|
829
|
-
...(model.contextWindow > 0 ? { modelMaxContext: model.contextWindow } : {}),
|
|
830
|
-
listResident: async () => {
|
|
831
|
-
loadedEntries = await client.llm.listLoaded();
|
|
832
|
-
return residentModelInfos(loadedEntries);
|
|
833
|
-
},
|
|
834
|
-
unload: async (id) => {
|
|
835
|
-
// Every instance of the id, so an eviction that frees a slot really
|
|
836
|
-
// frees it when LM Studio holds the model twice.
|
|
837
|
-
for (const entry of loadedEntries.filter((handle) => handle.modelKey === id)) await entry.unload();
|
|
838
|
-
},
|
|
839
|
-
});
|
|
840
|
-
const observeOnly = plan.decision === "observe";
|
|
841
|
-
const residencyLock = deps.lock ?? withResidencyLock;
|
|
842
|
-
// A second instance of the model Clio is about to use is pure loss: it
|
|
843
|
-
// holds another full weight copy and KV cache while serving the same
|
|
844
|
-
// requests. Releasing it takes no role's model away, so it happens before
|
|
845
|
-
// the fit question is even asked.
|
|
846
|
-
const duplicates = observeOnly ? [] : duplicateInstances(loadedEntries, model.id);
|
|
847
|
-
if (duplicates.length > 0) {
|
|
848
|
-
emitResidencyNotice({
|
|
849
|
-
kind: "co-resident",
|
|
850
|
-
level: "warning",
|
|
851
|
-
targetId: metadata.targetId,
|
|
852
|
-
runtimeId: "lmstudio-native",
|
|
853
|
-
model: model.id,
|
|
854
|
-
message: `'${metadata.targetId}' holds ${duplicates.length + 1} instances of '${model.id}'; releasing ${duplicates.length} duplicate instance(s) so one copy serves every request.`,
|
|
855
|
-
detail: { duplicateInstances: duplicates.length },
|
|
856
|
-
});
|
|
857
|
-
await residencyLock(targetKey, async () => {
|
|
858
|
-
for (const duplicate of duplicates) {
|
|
859
|
-
try {
|
|
860
|
-
await duplicate.unload();
|
|
861
|
-
} catch {
|
|
862
|
-
// Best-effort: a duplicate that survives is reported again next turn.
|
|
863
|
-
}
|
|
864
|
-
}
|
|
865
|
-
});
|
|
866
|
-
loadedEntries = loadedEntries.filter((entry) => !duplicates.includes(entry));
|
|
867
|
-
}
|
|
868
|
-
// Context-length fit. LM Studio's `gpuStrictVramCap` caps GPU offload
|
|
869
|
-
// instead of refusing an oversized load, so a KV cache that does not fit
|
|
870
|
-
// is served from CPU at a crawl rather than failing. While another model
|
|
871
|
-
// is resident, cap the load at the co-resident ceiling and say so before
|
|
872
|
-
// the first turn runs.
|
|
873
|
-
const fit = fitLoadContextLength({
|
|
874
|
-
requested: requestedLoadContext,
|
|
875
|
-
resident: loadedEntries,
|
|
876
|
-
keepModelId: model.id,
|
|
877
|
-
ceiling: coResidentContextCeiling(),
|
|
878
|
-
});
|
|
879
|
-
if (!observeOnly && !plan.keepResident && fit.clampedFrom !== undefined) {
|
|
880
|
-
emitResidencyNotice({
|
|
881
|
-
kind: "stress",
|
|
882
|
-
level: "warning",
|
|
883
|
-
targetId: metadata.targetId,
|
|
884
|
-
runtimeId: "lmstudio-native",
|
|
885
|
-
model: model.id,
|
|
886
|
-
message: `loading '${model.id}' on '${metadata.targetId}' alongside ${fit.neighbours.join(", ")}: context clamped ${fit.clampedFrom} -> ${fit.contextLength} tokens to keep the KV cache in GPU memory. Unload the co-resident model or raise CLIO_CODER_LMSTUDIO_CORESIDENT_CONTEXT for the full window.`,
|
|
887
|
-
detail: {
|
|
888
|
-
requestedContext: fit.clampedFrom,
|
|
889
|
-
loadContext: fit.contextLength,
|
|
890
|
-
residentCount: fit.neighbours.length + 1,
|
|
891
|
-
},
|
|
892
|
-
});
|
|
893
|
-
}
|
|
894
|
-
const loadConfig = loadModelConfig(model, { contextLength: fit.contextLength });
|
|
895
|
-
// Never hand `config` to client.llm.model for a model that is already
|
|
896
|
-
// resident: LM Studio answers a configured open of a loaded key with a
|
|
897
|
-
// second instance (observed as `google/gemma-4-26b-a4b-qat:2`) or with a
|
|
898
|
-
// no-progress reload wait. A resident model is reused as loaded, and the
|
|
899
|
-
// output budget already follows the window the server actually has open.
|
|
900
|
-
const modelOpenConfig = observeOnly || plan.keepResident ? undefined : loadConfig;
|
|
901
|
-
// `loadedEntries` is non-empty only when this stream actually listed the
|
|
902
|
-
// resident set, so the reconciler's TTL fast path stays silent instead of
|
|
903
|
-
// repeating the same line every turn.
|
|
904
|
-
if (
|
|
905
|
-
plan.keepResident &&
|
|
906
|
-
loadedEntries.length > 0 &&
|
|
907
|
-
loadedContextWindow !== undefined &&
|
|
908
|
-
loadedContextWindow < requestedLoadContext
|
|
909
|
-
) {
|
|
910
|
-
emitResidencyNotice({
|
|
911
|
-
kind: "co-resident",
|
|
912
|
-
level: "info",
|
|
913
|
-
targetId: metadata.targetId,
|
|
914
|
-
runtimeId: "lmstudio-native",
|
|
915
|
-
model: model.id,
|
|
916
|
-
message: `'${model.id}' is resident on '${metadata.targetId}' with a ${loadedContextWindow}-token window, below the ${requestedLoadContext} Clio would load; reusing the loaded instance and budgeting against ${loadedContextWindow}. Unload it in LM Studio to have Clio reload it larger.`,
|
|
917
|
-
detail: { loadedContext: loadedContextWindow, requestedContext: requestedLoadContext },
|
|
918
|
-
});
|
|
919
|
-
}
|
|
920
|
-
const modelOpenOpts: { signal: AbortSignal; verbose: boolean; config?: LLMLoadModelConfig } = {
|
|
921
|
-
signal: controller.signal,
|
|
922
|
-
verbose,
|
|
923
|
-
};
|
|
924
|
-
if (modelOpenConfig !== undefined) modelOpenOpts.config = modelOpenConfig;
|
|
925
|
-
const failWillNotFit: (err: unknown) => never = (err) => {
|
|
926
|
-
const message = describeLoadFailure(baseUrl, model, modelOpenConfig, requestedMaxTokens, err);
|
|
927
|
-
// gpuStrictVramCap turns an oversized load into a failure; surface it
|
|
928
|
-
// as a will-not-fit notice so it reads like every other VRAM miss.
|
|
929
|
-
emitResidencyNotice({
|
|
930
|
-
kind: "will-not-fit",
|
|
931
|
-
level: "error",
|
|
932
|
-
targetId: metadata.targetId,
|
|
933
|
-
runtimeId: "lmstudio-native",
|
|
934
|
-
model: model.id,
|
|
935
|
-
message,
|
|
936
|
-
});
|
|
937
|
-
throw new Error(message);
|
|
938
|
-
};
|
|
939
|
-
let llm: LmStudioPredictionHandle;
|
|
940
|
-
try {
|
|
941
|
-
llm = await client.llm.model(model.id, modelOpenOpts);
|
|
942
|
-
} catch (err) {
|
|
943
|
-
if (observeOnly || plan.fallbackEvict.length === 0) failWillNotFit(err);
|
|
944
|
-
// The co-resident load did not fit. Swap the ranked candidates and
|
|
945
|
-
// retry once, serialized against other Clio processes mutating this
|
|
946
|
-
// server; a second failure is a genuine VRAM miss.
|
|
947
|
-
try {
|
|
948
|
-
llm = await residencyLock(targetKey, async () => {
|
|
949
|
-
for (const entry of plan.fallbackEvict) {
|
|
950
|
-
emitResidencyNotice({
|
|
951
|
-
kind: "swap",
|
|
952
|
-
level: "warning",
|
|
953
|
-
targetId: metadata.targetId,
|
|
954
|
-
runtimeId: "lmstudio-native",
|
|
955
|
-
model: model.id,
|
|
956
|
-
message: `swapping resident '${entry.modelId}' for requested '${model.id}' on '${metadata.targetId}' after the co-resident load failed to fit.`,
|
|
957
|
-
detail: { swappedOut: entry.modelId },
|
|
958
|
-
});
|
|
959
|
-
try {
|
|
960
|
-
for (const handle of loadedEntries.filter((held) => held.modelKey === entry.modelId)) {
|
|
961
|
-
await handle.unload();
|
|
962
|
-
}
|
|
963
|
-
} catch {
|
|
964
|
-
// Best-effort: the retry load reports the real fit verdict.
|
|
965
|
-
}
|
|
966
|
-
}
|
|
967
|
-
return client.llm.model(model.id, modelOpenOpts);
|
|
968
|
-
});
|
|
969
|
-
} catch (retryErr) {
|
|
970
|
-
failWillNotFit(retryErr);
|
|
971
|
-
}
|
|
972
|
-
}
|
|
973
|
-
// A model whose weights or KV cache spilled to CPU still answers, at a
|
|
974
|
-
// crawl and with no error. Watch the token rate so that shows up as a
|
|
975
|
-
// notice naming the resident set instead of an indefinite spinner.
|
|
976
|
-
const residentList = [...new Set([model.id, ...loadedEntries.map((entry) => entry.modelKey)])].join(", ");
|
|
977
|
-
degradedWatchdog = startDegradedInferenceWatchdog({
|
|
978
|
-
onDegraded: (report) => {
|
|
979
|
-
emitResidencyNotice({
|
|
980
|
-
kind: "degraded",
|
|
981
|
-
level: "warning",
|
|
982
|
-
targetId: metadata.targetId,
|
|
983
|
-
runtimeId: "lmstudio-native",
|
|
984
|
-
model: model.id,
|
|
985
|
-
message: `'${model.id}' on '${metadata.targetId}' has generated ${report.tokens} tokens in ${Math.round(report.elapsedMs / 1000)}s (${report.tokensPerSecond.toFixed(2)} tok/s); inference is running far below GPU speed, which is what a spill to CPU looks like. Resident there: ${residentList}. Unload a co-resident model or lower the context window.`,
|
|
986
|
-
detail: {
|
|
987
|
-
tokens: report.tokens,
|
|
988
|
-
elapsedMs: report.elapsedMs,
|
|
989
|
-
tokensPerSecond: Number(report.tokensPerSecond.toFixed(2)),
|
|
990
|
-
residents: residentList,
|
|
991
|
-
},
|
|
992
|
-
});
|
|
993
|
-
},
|
|
994
|
-
});
|
|
995
|
-
if (!sdkPredictionEnabled()) {
|
|
996
|
-
// The SDK opened no prediction channel on this path, so nothing is
|
|
997
|
-
// left for a late abort to race; the HTTP transport owns the signal
|
|
998
|
-
// from here.
|
|
999
|
-
predictionDone = true;
|
|
1000
|
-
await pumpOpenAICompletions(stream, model, context, options, hints, requestedMaxTokens, (chars) =>
|
|
1001
|
-
degradedWatchdog?.addTokens(ceilChars(chars)),
|
|
1002
|
-
);
|
|
1003
|
-
return;
|
|
1004
|
-
}
|
|
1005
|
-
stream.push({ type: "start", partial: output });
|
|
1006
|
-
// Resolve once per request and make the resolved reasoning class
|
|
1007
|
-
// authoritative for what the operator is shown. LM Studio may classify
|
|
1008
|
-
// <think> tags even when the catalog says the selected family is
|
|
1009
|
-
// reasoning-never; in that case Clio suppresses those fragments rather
|
|
1010
|
-
// than surfacing thinking blocks. It never suppresses the token count:
|
|
1011
|
-
// the SDK carries no chat-template-kwargs channel, so `enable_thinking`
|
|
1012
|
-
// cannot reach this transport and a catalog `reasoning: false` is a
|
|
1013
|
-
// belief about the model, not a control over it. The reasoning a server
|
|
1014
|
-
// spends anyway has to stay visible in usage, or the belief cannot be
|
|
1015
|
-
// discovered to be wrong.
|
|
1016
|
-
const requestedThinkingLevel = thinkingLevelFromHintOrModel(hints, model);
|
|
1017
|
-
const resolved = resolveModelRuntimeCapabilitiesForModel(model, requestedThinkingLevel);
|
|
1018
|
-
const applied = resolved.thinking;
|
|
1019
|
-
const suppressThinking = reasoningClassForMechanism(applied.mechanism) === "never";
|
|
1020
|
-
// LM Studio's `result.stats.predictedTokensCount` is the total of all generated
|
|
1021
|
-
// tokens with no separate reasoning column. Sum the per-fragment `tokensCount`
|
|
1022
|
-
// for any fragment whose `reasoningType` belongs to a reasoning block (the
|
|
1023
|
-
// chain-of-thought content plus the literal start/end tag tokens) so the
|
|
1024
|
-
// receipt and TUI footer can report `reasoningTokens` truthfully even when the
|
|
1025
|
-
// model emits a chain-of-thought via its chat template that the SDK has no API
|
|
1026
|
-
// to disable. Per-fragment counts are approximate per the SDK docs, but the
|
|
1027
|
-
// per-run sum tracks the actual reasoning total closely.
|
|
1028
|
-
let reasoningTokensAccum = 0;
|
|
1029
|
-
const activeTextRef: { block: TextContent | null; idx: number } = { block: null, idx: -1 };
|
|
1030
|
-
const activeThinkingRef: { block: ThinkingContent | null; idx: number } = { block: null, idx: -1 };
|
|
1031
|
-
const sentinelStripper = createSentinelStripper();
|
|
1032
|
-
const closeActiveThinking = () => {
|
|
1033
|
-
const current = activeThinkingRef.block;
|
|
1034
|
-
if (!current) return;
|
|
1035
|
-
stream.push({
|
|
1036
|
-
type: "thinking_end",
|
|
1037
|
-
contentIndex: activeThinkingRef.idx,
|
|
1038
|
-
content: current.thinking,
|
|
1039
|
-
partial: output,
|
|
1040
|
-
});
|
|
1041
|
-
activeThinkingRef.block = null;
|
|
1042
|
-
activeThinkingRef.idx = -1;
|
|
1043
|
-
};
|
|
1044
|
-
const pushSafeText = (safe: string) => {
|
|
1045
|
-
if (!safe) return;
|
|
1046
|
-
closeActiveThinking();
|
|
1047
|
-
let current = activeTextRef.block;
|
|
1048
|
-
if (!current) {
|
|
1049
|
-
current = { type: "text", text: "" };
|
|
1050
|
-
output.content.push(current);
|
|
1051
|
-
activeTextRef.block = current;
|
|
1052
|
-
activeTextRef.idx = output.content.length - 1;
|
|
1053
|
-
stream.push({ type: "text_start", contentIndex: activeTextRef.idx, partial: output });
|
|
1054
|
-
}
|
|
1055
|
-
current.text += safe;
|
|
1056
|
-
stream.push({
|
|
1057
|
-
type: "text_delta",
|
|
1058
|
-
contentIndex: activeTextRef.idx,
|
|
1059
|
-
delta: safe,
|
|
1060
|
-
partial: output,
|
|
1061
|
-
});
|
|
1062
|
-
};
|
|
1063
|
-
const flushTextSentinelBuffer = () => {
|
|
1064
|
-
const tail = sentinelStripper.flush();
|
|
1065
|
-
if (tail) pushSafeText(tail);
|
|
1066
|
-
};
|
|
1067
|
-
const emitText = (chunk: string) => {
|
|
1068
|
-
if (!chunk) return;
|
|
1069
|
-
const safe = sentinelStripper.push(chunk);
|
|
1070
|
-
pushSafeText(safe);
|
|
1071
|
-
};
|
|
1072
|
-
const closeActiveText = () => {
|
|
1073
|
-
// Drain any sentinel-prefix bytes the streaming stripper held
|
|
1074
|
-
// back across the last delta. The buffered tail can never grow
|
|
1075
|
-
// past `MAX_SENTINEL_LEN - 1` characters and only contains
|
|
1076
|
-
// matter that turned out not to be a sentinel; emitting it now
|
|
1077
|
-
// keeps the visible block whole without leaking sentinels.
|
|
1078
|
-
flushTextSentinelBuffer();
|
|
1079
|
-
const current = activeTextRef.block;
|
|
1080
|
-
if (!current) return;
|
|
1081
|
-
stream.push({
|
|
1082
|
-
type: "text_end",
|
|
1083
|
-
contentIndex: activeTextRef.idx,
|
|
1084
|
-
content: current.text,
|
|
1085
|
-
partial: output,
|
|
1086
|
-
});
|
|
1087
|
-
activeTextRef.block = null;
|
|
1088
|
-
activeTextRef.idx = -1;
|
|
1089
|
-
};
|
|
1090
|
-
const emitThinking = (chunk: string, tokensHint?: number) => {
|
|
1091
|
-
if (!chunk) return;
|
|
1092
|
-
if (suppressThinking) return;
|
|
1093
|
-
closeActiveText();
|
|
1094
|
-
let current = activeThinkingRef.block;
|
|
1095
|
-
if (!current) {
|
|
1096
|
-
current = { type: "thinking", thinking: "" };
|
|
1097
|
-
output.content.push(current);
|
|
1098
|
-
activeThinkingRef.block = current;
|
|
1099
|
-
activeThinkingRef.idx = output.content.length - 1;
|
|
1100
|
-
stream.push({ type: "thinking_start", contentIndex: activeThinkingRef.idx, partial: output });
|
|
1101
|
-
}
|
|
1102
|
-
current.thinking += chunk;
|
|
1103
|
-
// When we have an upstream token count from the SDK use it; otherwise
|
|
1104
|
-
// approximate at 1 token per 4 chars (the same chars/4 estimator the
|
|
1105
|
-
// openai-compat wrapper uses for streamed thinking content).
|
|
1106
|
-
reasoningTokensAccum += tokensHint ?? ceilChars(chunk.length);
|
|
1107
|
-
stream.push({
|
|
1108
|
-
type: "thinking_delta",
|
|
1109
|
-
contentIndex: activeThinkingRef.idx,
|
|
1110
|
-
delta: chunk,
|
|
1111
|
-
partial: output,
|
|
1112
|
-
});
|
|
1113
|
-
};
|
|
1114
|
-
// Buffered state machine for gemma-4 family chat-template channel markers.
|
|
1115
|
-
// The model emits these as plain text fragments (reasoningType = "none")
|
|
1116
|
-
// rather than tagging them as reasoning, so without re-classification the
|
|
1117
|
-
// chain-of-thought leaks into the visible TUI text. The thought-start
|
|
1118
|
-
// pattern is regex-based because gemma emits channel-name variants like
|
|
1119
|
-
// `<|channel>thought\n`, `<|channel>own-thought\n`, `<|channel>own-think\n`.
|
|
1120
|
-
// LM Studio can also strip the `<|channel>` prefix and leave only the
|
|
1121
|
-
// bare channel label, e.g. `ownthought\n`, before a structured tool call.
|
|
1122
|
-
// Orphan `<channel|>` close markers (where the open was already consumed
|
|
1123
|
-
// via the SDK's reasoning-fragment path) are dropped in idle state.
|
|
1124
|
-
const GEMMA_THOUGHT_START_RE = /<\|channel>[^\n]*\n/;
|
|
1125
|
-
const GEMMA_BARE_THOUGHT_START_RE = /^\s*(?:thought|own[- ]?(?:thought|think))\s*\n/i;
|
|
1126
|
-
const GEMMA_BARE_THOUGHT_ONLY_RE = /^\s*(?:thought|own[- ]?(?:thought|think))\s*$/i;
|
|
1127
|
-
const GEMMA_THOUGHT_END = "<channel|>";
|
|
1128
|
-
const GEMMA_TOOLCALL_START = "<tool_call|>";
|
|
1129
|
-
const GEMMA_TOOLCALL_END = "<|tool_call|>";
|
|
1130
|
-
const GEMMA_BUFFER_MAX = 64;
|
|
1131
|
-
type GemmaState = "idle" | "thought" | "toolcall";
|
|
1132
|
-
let gemmaPending = "";
|
|
1133
|
-
let gemmaState: GemmaState = "idle";
|
|
1134
|
-
const responseParser = resolved.response.parser;
|
|
1135
|
-
const harmonyParser = responseParser === "harmony" ? new HarmonyResponseParser() : null;
|
|
1136
|
-
const flushGemmaPending = () => {
|
|
1137
|
-
if (gemmaPending.length === 0) return;
|
|
1138
|
-
if (gemmaState === "thought") emitThinking(gemmaPending);
|
|
1139
|
-
else if (gemmaState === "idle" && !GEMMA_BARE_THOUGHT_ONLY_RE.test(gemmaPending)) emitText(gemmaPending);
|
|
1140
|
-
gemmaPending = "";
|
|
1141
|
-
};
|
|
1142
|
-
const flushNonReasoningPending = () => {
|
|
1143
|
-
if (harmonyParser) {
|
|
1144
|
-
const parsed = harmonyParser.flush();
|
|
1145
|
-
emitThinking(parsed.thinking);
|
|
1146
|
-
emitText(parsed.text);
|
|
1147
|
-
return;
|
|
1148
|
-
}
|
|
1149
|
-
flushGemmaPending();
|
|
1150
|
-
};
|
|
1151
|
-
const routeNonReasoningChunk = (chunk: string) => {
|
|
1152
|
-
if (harmonyParser) {
|
|
1153
|
-
const parsed = harmonyParser.push(chunk);
|
|
1154
|
-
emitThinking(parsed.thinking);
|
|
1155
|
-
emitText(parsed.text);
|
|
1156
|
-
return;
|
|
1157
|
-
}
|
|
1158
|
-
gemmaPending += chunk;
|
|
1159
|
-
while (true) {
|
|
1160
|
-
if (gemmaState === "thought") {
|
|
1161
|
-
const endIdx = gemmaPending.indexOf(GEMMA_THOUGHT_END);
|
|
1162
|
-
if (endIdx === -1) {
|
|
1163
|
-
if (gemmaPending.length > GEMMA_BUFFER_MAX) {
|
|
1164
|
-
const safe = gemmaPending.slice(0, gemmaPending.length - GEMMA_BUFFER_MAX);
|
|
1165
|
-
emitThinking(safe);
|
|
1166
|
-
gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
|
|
1167
|
-
}
|
|
1168
|
-
return;
|
|
1169
|
-
}
|
|
1170
|
-
emitThinking(gemmaPending.slice(0, endIdx));
|
|
1171
|
-
gemmaPending = gemmaPending.slice(endIdx + GEMMA_THOUGHT_END.length);
|
|
1172
|
-
gemmaState = "idle";
|
|
1173
|
-
} else if (gemmaState === "toolcall") {
|
|
1174
|
-
// Discard SDK fallback text inside <tool_call|> regions; structured
|
|
1175
|
-
// tool calls arrive via the toolcall callbacks instead.
|
|
1176
|
-
const endIdx = gemmaPending.indexOf(GEMMA_TOOLCALL_END);
|
|
1177
|
-
if (endIdx === -1) {
|
|
1178
|
-
if (gemmaPending.length > GEMMA_BUFFER_MAX) {
|
|
1179
|
-
gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
|
|
1180
|
-
}
|
|
1181
|
-
return;
|
|
1182
|
-
}
|
|
1183
|
-
gemmaPending = gemmaPending.slice(endIdx + GEMMA_TOOLCALL_END.length);
|
|
1184
|
-
gemmaState = "idle";
|
|
1185
|
-
} else {
|
|
1186
|
-
const thoughtMatch = GEMMA_THOUGHT_START_RE.exec(gemmaPending);
|
|
1187
|
-
const thoughtIdx = thoughtMatch?.index ?? -1;
|
|
1188
|
-
const bareThoughtMatch = GEMMA_BARE_THOUGHT_START_RE.exec(gemmaPending);
|
|
1189
|
-
const bareThoughtIdx = bareThoughtMatch ? 0 : -1;
|
|
1190
|
-
const toolcallIdx = gemmaPending.indexOf(GEMMA_TOOLCALL_START);
|
|
1191
|
-
// Orphan close marker: SDK consumed `<|channel>...` via a
|
|
1192
|
-
// reasoning fragment, leaving only the standalone close in
|
|
1193
|
-
// the plain-text stream. Drop it silently.
|
|
1194
|
-
const orphanCloseIdx = gemmaPending.indexOf(GEMMA_THOUGHT_END);
|
|
1195
|
-
const candidates = [
|
|
1196
|
-
{ idx: thoughtIdx, kind: "thought" as const, advance: thoughtMatch?.[0].length ?? 0 },
|
|
1197
|
-
{ idx: bareThoughtIdx, kind: "thought" as const, advance: bareThoughtMatch?.[0].length ?? 0 },
|
|
1198
|
-
{ idx: toolcallIdx, kind: "toolcall" as const, advance: GEMMA_TOOLCALL_START.length },
|
|
1199
|
-
{ idx: orphanCloseIdx, kind: "orphan" as const, advance: GEMMA_THOUGHT_END.length },
|
|
1200
|
-
].filter((c) => c.idx !== -1);
|
|
1201
|
-
if (candidates.length === 0) {
|
|
1202
|
-
if (gemmaPending.length > GEMMA_BUFFER_MAX) {
|
|
1203
|
-
const safe = gemmaPending.slice(0, gemmaPending.length - GEMMA_BUFFER_MAX);
|
|
1204
|
-
emitText(safe);
|
|
1205
|
-
gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
|
|
1206
|
-
}
|
|
1207
|
-
return;
|
|
1208
|
-
}
|
|
1209
|
-
candidates.sort((a, b) => a.idx - b.idx);
|
|
1210
|
-
const next = candidates[0];
|
|
1211
|
-
if (!next) return;
|
|
1212
|
-
emitText(gemmaPending.slice(0, next.idx));
|
|
1213
|
-
gemmaPending = gemmaPending.slice(next.idx + next.advance);
|
|
1214
|
-
if (next.kind === "thought") gemmaState = "thought";
|
|
1215
|
-
else if (next.kind === "toolcall") gemmaState = "toolcall";
|
|
1216
|
-
// orphan stays in idle: just drop the marker bytes
|
|
1217
|
-
}
|
|
1218
|
-
}
|
|
1219
|
-
};
|
|
1220
|
-
const pending = new Map<number, PendingToolCall>();
|
|
1221
|
-
let toolExtractionError: LmStudioToolCallExtractionError | null = null;
|
|
1222
|
-
const predictionOpts: LLMRespondOpts<unknown> = {
|
|
1223
|
-
signal: controller.signal,
|
|
1224
|
-
onPredictionFragment: (fragment) => {
|
|
1225
|
-
if (!fragment.content) return;
|
|
1226
|
-
degradedWatchdog?.addTokens(fragment.tokensCount ?? ceilChars(fragment.content.length));
|
|
1227
|
-
// LM Studio SDK reasoningType values:
|
|
1228
|
-
// "none" normal content (text)
|
|
1229
|
-
// "reasoning" chain-of-thought inside the block
|
|
1230
|
-
// "reasoningStartTag" literal <think> token
|
|
1231
|
-
// "reasoningEndTag" literal </think> token
|
|
1232
|
-
// Drop the start/end tags so they never leak into text or thinking;
|
|
1233
|
-
// route reasoning fragments into a ThinkingContent block so the
|
|
1234
|
-
// agent message is non-empty and pi-agent-core's loop can chain
|
|
1235
|
-
// correctly when the model only emits reasoning + tool calls.
|
|
1236
|
-
if (fragment.reasoningType === "reasoningStartTag" || fragment.reasoningType === "reasoningEndTag") {
|
|
1237
|
-
reasoningTokensAccum += fragment.tokensCount ?? 0;
|
|
1238
|
-
return;
|
|
1239
|
-
}
|
|
1240
|
-
if (fragment.reasoningType === "reasoning") {
|
|
1241
|
-
flushNonReasoningPending();
|
|
1242
|
-
// Count first, show second. Suppression is a statement about what
|
|
1243
|
-
// the operator is shown, never about what the server spent. Gating
|
|
1244
|
-
// the accrual on it made a reasoning-never model that reasons
|
|
1245
|
-
// anyway report zero reasoning tokens, which is the one reading
|
|
1246
|
-
// that would have revealed the catalog was wrong about it.
|
|
1247
|
-
reasoningTokensAccum += fragment.tokensCount ?? 0;
|
|
1248
|
-
emitThinking(fragment.content, 0);
|
|
1249
|
-
return;
|
|
1250
|
-
}
|
|
1251
|
-
routeNonReasoningChunk(fragment.content);
|
|
1252
|
-
},
|
|
1253
|
-
onToolCallRequestStart: (callId) => {
|
|
1254
|
-
flushNonReasoningPending();
|
|
1255
|
-
gemmaState = "idle";
|
|
1256
|
-
closeActiveText();
|
|
1257
|
-
closeActiveThinking();
|
|
1258
|
-
const slot: ToolCall = { type: "toolCall", id: randomUUID(), name: "", arguments: {} };
|
|
1259
|
-
output.content.push(slot);
|
|
1260
|
-
const idx = output.content.length - 1;
|
|
1261
|
-
stream.push({ type: "toolcall_start", contentIndex: idx, partial: output });
|
|
1262
|
-
pending.set(callId, {
|
|
1263
|
-
contentIndex: idx,
|
|
1264
|
-
name: "",
|
|
1265
|
-
argBuffer: "",
|
|
1266
|
-
assistantIndex: idx,
|
|
1267
|
-
toolCallSlot: slot,
|
|
1268
|
-
});
|
|
1269
|
-
},
|
|
1270
|
-
onToolCallRequestNameReceived: (callId, name) => {
|
|
1271
|
-
const entry = pending.get(callId);
|
|
1272
|
-
if (!entry) return;
|
|
1273
|
-
entry.name = name;
|
|
1274
|
-
entry.toolCallSlot.name = name;
|
|
1275
|
-
},
|
|
1276
|
-
onToolCallRequestArgumentFragmentGenerated: (callId, fragment) => {
|
|
1277
|
-
const entry = pending.get(callId);
|
|
1278
|
-
if (!entry) return;
|
|
1279
|
-
entry.argBuffer += fragment;
|
|
1280
|
-
entry.toolCallSlot.arguments = parseStreamingArgs(entry.argBuffer);
|
|
1281
|
-
stream.push({
|
|
1282
|
-
type: "toolcall_delta",
|
|
1283
|
-
contentIndex: entry.contentIndex,
|
|
1284
|
-
delta: fragment,
|
|
1285
|
-
partial: output,
|
|
1286
|
-
});
|
|
1287
|
-
},
|
|
1288
|
-
onToolCallRequestEnd: (callId, info) => {
|
|
1289
|
-
const entry = pending.get(callId);
|
|
1290
|
-
if (!entry) return;
|
|
1291
|
-
const req = info.toolCallRequest;
|
|
1292
|
-
entry.toolCallSlot.name = req.name || entry.name || "";
|
|
1293
|
-
if (hasEmptyToolArguments(req.arguments) && entry.argBuffer.length === 0 && hasNonEmptyGeneratedContent(output)) {
|
|
1294
|
-
toolExtractionError = new LmStudioToolCallExtractionError();
|
|
1295
|
-
pending.delete(callId);
|
|
1296
|
-
return;
|
|
1297
|
-
}
|
|
1298
|
-
entry.toolCallSlot.arguments =
|
|
1299
|
-
req.arguments && typeof req.arguments === "object"
|
|
1300
|
-
? (req.arguments as Record<string, unknown>)
|
|
1301
|
-
: parseFinalArgs(entry.argBuffer);
|
|
1302
|
-
if (req.id) entry.toolCallSlot.id = req.id;
|
|
1303
|
-
stream.push({
|
|
1304
|
-
type: "toolcall_end",
|
|
1305
|
-
contentIndex: entry.contentIndex,
|
|
1306
|
-
toolCall: entry.toolCallSlot,
|
|
1307
|
-
partial: output,
|
|
1308
|
-
});
|
|
1309
|
-
pending.delete(callId);
|
|
1310
|
-
},
|
|
1311
|
-
};
|
|
1312
|
-
if (context.tools && context.tools.length > 0) {
|
|
1313
|
-
predictionOpts.rawTools = {
|
|
1314
|
-
type: "toolArray",
|
|
1315
|
-
tools: context.tools.map(toolToLmStudio),
|
|
1316
|
-
};
|
|
1317
|
-
}
|
|
1318
|
-
predictionOpts.maxTokens = requestedMaxTokens;
|
|
1319
|
-
// Apply catalog sampling quirks first; explicit StreamOptions overrides
|
|
1320
|
-
// (set on `options`) win where they are present. The catalog profile is
|
|
1321
|
-
// chosen by thinking activity, derived through the central resolver so
|
|
1322
|
-
// the sampler choice matches the actual surface the model exposes
|
|
1323
|
-
// (effort-levels, budget-tokens, on-off, always-on, none). The bare
|
|
1324
|
-
// `stream` path leaves `hints.thinkingLevel` unset and falls back to
|
|
1325
|
-
// medium when the model advertises reasoning.
|
|
1326
|
-
// The LM Studio SDK has no separate thinking-budget channel; the budget
|
|
1327
|
-
// from `applied.budgetTokens` is informational only here and surfaces
|
|
1328
|
-
// through the prompt Runtime block. `maxPredictedTokens` stays driven
|
|
1329
|
-
// by the remaining-context budget so a budget-tokens family does not
|
|
1330
|
-
// unexpectedly truncate output.
|
|
1331
|
-
const samplingProfile = pickSamplingProfile(resolved.quirks ?? clioQuirks(model), applied.thinkingActive);
|
|
1332
|
-
if (samplingProfile) {
|
|
1333
|
-
if (samplingProfile.temperature !== undefined) predictionOpts.temperature = samplingProfile.temperature;
|
|
1334
|
-
if (samplingProfile.topP !== undefined) predictionOpts.topPSampling = samplingProfile.topP;
|
|
1335
|
-
if (samplingProfile.topK !== undefined) predictionOpts.topKSampling = samplingProfile.topK;
|
|
1336
|
-
if (samplingProfile.minP !== undefined) predictionOpts.minPSampling = samplingProfile.minP;
|
|
1337
|
-
if (samplingProfile.repeatPenalty !== undefined) predictionOpts.repeatPenalty = samplingProfile.repeatPenalty;
|
|
1338
|
-
}
|
|
1339
|
-
if (options?.temperature !== undefined) predictionOpts.temperature = options.temperature;
|
|
1340
|
-
const harmony = resolved.response.parser === "harmony";
|
|
1341
|
-
const history = await buildChatHistory(client, context, { harmony, preserveThinking: !suppressThinking });
|
|
1342
|
-
if (aborted) throw new Error("Request was aborted");
|
|
1343
|
-
const prediction = llm.respond(history, predictionOpts);
|
|
1344
|
-
const result = await prediction.result();
|
|
1345
|
-
// The SDK has now closed its prediction channel cleanly. Block any
|
|
1346
|
-
// future `onAbort` from racing a second `controller.abort()` against
|
|
1347
|
-
// that closed channel; the post-result `if (aborted) throw` below
|
|
1348
|
-
// still surfaces a late user-driven abort to the caller.
|
|
1349
|
-
predictionDone = true;
|
|
1350
|
-
flushNonReasoningPending();
|
|
1351
|
-
closeActiveText();
|
|
1352
|
-
closeActiveThinking();
|
|
1353
|
-
// Write usage before any throw so the error path (tool-extraction failure,
|
|
1354
|
-
// post-result aborts) still surfaces real token counts to dispatch and the TUI.
|
|
1355
|
-
// Probed off the raw stats rather than PredictionStatsLike: no LM Studio
|
|
1356
|
-
// build ships a cached-token count today, so the field only exists for
|
|
1357
|
-
// the runtime that starts sending one. Anything else stays undefined.
|
|
1358
|
-
const reportedCache = asRecord(result.stats).cachedTokensCount;
|
|
1359
|
-
const cacheRead = typeof reportedCache === "number" && reportedCache >= 0 ? reportedCache : undefined;
|
|
1360
|
-
output.usage.input = Math.max(0, (result.stats.promptTokensCount ?? 0) - (cacheRead ?? 0));
|
|
1361
|
-
output.usage.output = result.stats.predictedTokensCount ?? 0;
|
|
1362
|
-
output.usage.cacheRead = cacheRead as number;
|
|
1363
|
-
output.usage.totalTokens = result.stats.totalTokensCount ?? output.usage.input + output.usage.output;
|
|
1364
|
-
if (reasoningTokensAccum > 0) {
|
|
1365
|
-
(output.usage as Usage & { reasoningTokens?: number }).reasoningTokens = reasoningTokensAccum;
|
|
1366
|
-
}
|
|
1367
|
-
// Costed against a numeric view: pi's cost math multiplies `cacheRead`
|
|
1368
|
-
// directly, so handing it undefined would make every total NaN.
|
|
1369
|
-
output.usage.cost = calculateEngineCost(model, { ...output.usage, cacheRead: cacheRead ?? 0 });
|
|
1370
|
-
if (aborted) throw new Error("Request was aborted");
|
|
1371
|
-
if (toolExtractionError) throw toolExtractionError;
|
|
1372
|
-
const hadToolCall = output.content.some((block) => block.type === "toolCall");
|
|
1373
|
-
output.stopReason = mapStopReason(result.stats.stopReason, aborted, hadToolCall);
|
|
1374
|
-
if (output.stopReason === "error" || output.stopReason === "aborted") {
|
|
1375
|
-
output.errorMessage = `prediction stopped: ${result.stats.stopReason ?? "unknown"}`;
|
|
1376
|
-
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
1377
|
-
} else {
|
|
1378
|
-
stream.push({ type: "done", reason: asDoneReason(output.stopReason), message: output });
|
|
1379
|
-
}
|
|
1380
|
-
stream.end();
|
|
1381
|
-
} catch (err) {
|
|
1382
|
-
output.stopReason = aborted ? "aborted" : "error";
|
|
1383
|
-
output.errorMessage = err instanceof Error ? err.message : String(err);
|
|
1384
|
-
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
1385
|
-
stream.end();
|
|
1386
|
-
} finally {
|
|
1387
|
-
degradedWatchdog?.stop();
|
|
1388
|
-
if (signal) signal.removeEventListener("abort", onAbort);
|
|
1389
|
-
}
|
|
1390
|
-
})();
|
|
1391
|
-
return stream;
|
|
1392
|
-
}
|
|
1393
|
-
|
|
1394
|
-
function asRecord(value: unknown): Record<string, unknown> {
|
|
1395
|
-
return value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
|
|
1396
|
-
}
|
|
1397
|
-
|
|
1398
|
-
function parseStreamingArgs(raw: string): Record<string, unknown> {
|
|
1399
|
-
if (!raw) return {};
|
|
1400
|
-
try {
|
|
1401
|
-
return asRecord(parseEngineStreamingJson<unknown>(raw));
|
|
1402
|
-
} catch {
|
|
1403
|
-
return {};
|
|
1404
|
-
}
|
|
1405
|
-
}
|
|
1406
|
-
|
|
1407
|
-
function parseFinalArgs(raw: string): Record<string, unknown> {
|
|
1408
|
-
if (!raw) return {};
|
|
1409
|
-
try {
|
|
1410
|
-
return asRecord(parseEngineJsonWithRepair<unknown>(raw));
|
|
1411
|
-
} catch {
|
|
1412
|
-
return parseStreamingArgs(raw);
|
|
1413
|
-
}
|
|
1414
|
-
}
|
|
1415
|
-
|
|
1416
|
-
function stripReasoning(options: SimpleStreamOptions | undefined): StreamOptions | undefined {
|
|
1417
|
-
if (!options) return undefined;
|
|
1418
|
-
const { reasoning: _r, thinkingBudgets: _b, ...rest } = options;
|
|
1419
|
-
return rest;
|
|
1420
|
-
}
|
|
1421
|
-
|
|
1422
|
-
// pi-ai's SimpleStreamOptions.reasoning is the ThinkingLevel for this turn,
|
|
1423
|
-
// or undefined when thinking is off. The bare `stream` path cannot reach the
|
|
1424
|
-
// level so runStream falls back to the model's `reasoning` capability flag.
|
|
1425
|
-
function thinkingLevelFromSimple(options: SimpleStreamOptions | undefined): ThinkingLevel {
|
|
1426
|
-
const reasoning = options?.reasoning;
|
|
1427
|
-
if (reasoning === undefined) return "off";
|
|
1428
|
-
return reasoning as ThinkingLevel;
|
|
1429
|
-
}
|
|
1430
|
-
|
|
1431
|
-
export const lmstudioNativeApiProvider: ApiProvider<"lmstudio-native"> = {
|
|
1432
|
-
api: "lmstudio-native",
|
|
1433
|
-
stream: (model, context, options) => runStream(model, context, options),
|
|
1434
|
-
streamSimple: (model, context, options?: SimpleStreamOptions) =>
|
|
1435
|
-
runStream(model, context, stripReasoning(options), defaultRunDeps, {
|
|
1436
|
-
thinkingLevel: thinkingLevelFromSimple(options),
|
|
1437
|
-
}),
|
|
1438
|
-
};
|