@iowarp/clio-coder 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (595) hide show
  1. package/CHANGELOG.md +90 -2
  2. package/CONTRIBUTING.md +23 -23
  3. package/README.md +284 -613
  4. package/dist/{acp-FPR54DGL.js → acp-BIYHVZIM.js} +43 -53
  5. package/dist/{agents-OGPIHPJH.js → agents-YT6SSRIT.js} +43 -26
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-IC3K6NIZ.js → auth-5TWEIYDN.js} +20 -12
  8. package/dist/{chunk-PV4JUBVJ.js → chunk-2EHAIA3X.js} +40 -21
  9. package/dist/{chunk-IS3ONKU3.js → chunk-2IR2NMPA.js} +6 -4
  10. package/dist/chunk-2SFS6XQE.js +122 -0
  11. package/dist/chunk-2VTFPG5O.js +48 -0
  12. package/dist/{chunk-MAR7Y6HW.js → chunk-3ZXDFGR5.js} +23 -16
  13. package/dist/chunk-4BJ5BYCE.js +61 -0
  14. package/dist/{chunk-474KN5II.js → chunk-4BPJXDWC.js} +111 -181
  15. package/dist/chunk-4KLWL3UC.js +18 -0
  16. package/dist/chunk-4VP4KH3K.js +962 -0
  17. package/dist/chunk-4ZG3XFUR.js +77 -0
  18. package/dist/chunk-5B2AEOW5.js +5407 -0
  19. package/dist/{chunk-K2ITRMHZ.js → chunk-5TSRNF4G.js} +6 -138
  20. package/dist/{chunk-OLBBMFRD.js → chunk-5UUP6MWO.js} +24 -62
  21. package/dist/chunk-65DEGPJ6.js +52 -0
  22. package/dist/chunk-6EJMN2Y3.js +17 -0
  23. package/dist/chunk-6EJV5X2W.js +16405 -0
  24. package/dist/chunk-6N5PTWMY.js +136 -0
  25. package/dist/chunk-6XLNIQDB.js +27 -0
  26. package/dist/{chunk-TEKV33Q5.js → chunk-77VKQEHF.js} +65 -33
  27. package/dist/chunk-7CR24IG7.js +242 -0
  28. package/dist/{chunk-M5T5VO65.js → chunk-7EYHLWU7.js} +837 -635
  29. package/dist/chunk-7MNJORFF.js +22 -0
  30. package/dist/{chunk-KY56HMHH.js → chunk-A3CYT5EX.js} +125 -31
  31. package/dist/chunk-AGYYIBLL.js +1069 -0
  32. package/dist/chunk-AO4RKG4M.js +277 -0
  33. package/dist/{chunk-GB6QRBXN.js → chunk-APJ265NV.js} +54 -1187
  34. package/dist/chunk-ARBGF5F7.js +174 -0
  35. package/dist/{chunk-673JJUWJ.js → chunk-BMEMKKIT.js} +2 -2
  36. package/dist/chunk-CBCAPZAA.js +229 -0
  37. package/dist/chunk-CMZWFGD2.js +352 -0
  38. package/dist/chunk-ECH6PKUQ.js +39 -0
  39. package/dist/chunk-ED4KHGC3.js +143 -0
  40. package/dist/chunk-EKMEHE4H.js +340 -0
  41. package/dist/{chunk-RPTR2H26.js → chunk-EPVUXGXG.js} +21 -15
  42. package/dist/chunk-FCSXB6T2.js +338 -0
  43. package/dist/chunk-FJ3H4MN5.js +48 -0
  44. package/dist/chunk-FQ4SKYE4.js +29 -0
  45. package/dist/chunk-G2DE3C7R.js +644 -0
  46. package/dist/chunk-G4BMMOKF.js +182 -0
  47. package/dist/{chunk-ZPY3JZ5E.js → chunk-GGXXDWE4.js} +183 -1233
  48. package/dist/chunk-HC4CLZ2Y.js +68 -0
  49. package/dist/{chunk-LU4TK2PR.js → chunk-HFSBBKSQ.js} +5 -56
  50. package/dist/{chunk-PIUMUEMV.js → chunk-HKIYEGME.js} +10 -6
  51. package/dist/chunk-I4HZDVNP.js +73 -0
  52. package/dist/{chunk-4QKXUHSR.js → chunk-IGLFWIYI.js} +70 -20
  53. package/dist/chunk-IKCO5N3L.js +162 -0
  54. package/dist/chunk-IR4CFBFN.js +56 -0
  55. package/dist/{chunk-PFEFKVGL.js → chunk-J5HN4RYU.js} +13 -11
  56. package/dist/{chunk-R5KLMSBV.js → chunk-J5Q24KAG.js} +2 -2
  57. package/dist/{chunk-K5XEMXTI.js → chunk-JVCV3ICN.js} +1 -1
  58. package/dist/chunk-KJ5LWLOE.js +1077 -0
  59. package/dist/chunk-LBMZMYH2.js +285 -0
  60. package/dist/{chunk-G34LV2PF.js → chunk-LBNRH5WM.js} +84 -170
  61. package/dist/{chunk-H6F6BYOH.js → chunk-LZSJBIVT.js} +7003 -7434
  62. package/dist/{chunk-HQQID6OA.js → chunk-M6SHUN7Q.js} +5 -5
  63. package/dist/chunk-MAW544W2.js +1882 -0
  64. package/dist/chunk-MBS4V7ZP.js +217 -0
  65. package/dist/{chunk-FST4FYJB.js → chunk-MFFY33HR.js} +99 -140
  66. package/dist/chunk-MNA4JGU4.js +255 -0
  67. package/dist/chunk-MQSRRFWA.js +3428 -0
  68. package/dist/{chunk-BSU2YIWB.js → chunk-MVVUPGPW.js} +131 -136
  69. package/dist/chunk-OAO4GE4M.js +619 -0
  70. package/dist/chunk-OHHN2SO4.js +5135 -0
  71. package/dist/chunk-OKGUZO2U.js +34 -0
  72. package/dist/{chunk-GAEBEQVI.js → chunk-OOJYHWRB.js} +32 -346
  73. package/dist/{chunk-Q5WJOSJ7.js → chunk-OQ33BKR3.js} +2 -1
  74. package/dist/chunk-OQE5J4C6.js +73 -0
  75. package/dist/{chunk-KKNLWXI6.js → chunk-ORBHGJC5.js} +8 -8
  76. package/dist/chunk-POHLU5DW.js +1186 -0
  77. package/dist/chunk-QKMUKYO7.js +4961 -0
  78. package/dist/{chunk-ASND7OZK.js → chunk-QTYWRVRA.js} +13 -13
  79. package/dist/{chunk-EYOKLTMF.js → chunk-SRF2PJNW.js} +17 -3
  80. package/dist/chunk-SST6Z5JA.js +80 -0
  81. package/dist/chunk-STBPMHSX.js +2456 -0
  82. package/dist/chunk-T6YILFSB.js +80 -0
  83. package/dist/chunk-TZTZS7QK.js +227 -0
  84. package/dist/chunk-UOV2BYIW.js +107 -0
  85. package/dist/{chunk-Q3RUPKEJ.js → chunk-V4RXGQ5Q.js} +58 -189
  86. package/dist/chunk-VAKQQHWR.js +434 -0
  87. package/dist/chunk-VG7TBQIY.js +128 -0
  88. package/dist/chunk-VJWL6YS5.js +244 -0
  89. package/dist/chunk-WEH5XRJQ.js +32 -0
  90. package/dist/chunk-WMSVI4G2.js +2095 -0
  91. package/dist/chunk-WVO7V2QY.js +797 -0
  92. package/dist/chunk-X4RCMKVQ.js +641 -0
  93. package/dist/chunk-X75S7HFS.js +374 -0
  94. package/dist/chunk-XN3L4EYL.js +46 -0
  95. package/dist/{chunk-RDLVBZEO.js → chunk-YCWGATWI.js} +6 -4
  96. package/dist/chunk-YHZX5GEU.js +193 -0
  97. package/dist/chunk-YXLYO42X.js +91 -0
  98. package/dist/{chunk-NMOX6HFD.js → chunk-ZDOOVTXZ.js} +29 -77
  99. package/dist/chunk-ZI647VB5.js +37 -0
  100. package/dist/{chunk-C4PTHK7P.js → chunk-ZWLZP4ZT.js} +5 -5
  101. package/dist/cli/index.js +62 -54
  102. package/dist/clio-4LY5K2AC.js +25 -0
  103. package/dist/code-nav-7AX6FYE6.js +600 -0
  104. package/dist/codewiki/build-worker.js +66 -0
  105. package/dist/compile-cache-CVJMMODC.js +18 -0
  106. package/dist/{components-DMAOEKFB.js → components-KELWS457.js} +11 -6
  107. package/dist/{config-IRUQ7SE4.js → config-GTLUW2PR.js} +92 -55
  108. package/dist/configure-R6A64DHX.js +42 -0
  109. package/dist/context-5VKGUVJJ.js +866 -0
  110. package/dist/{context-3KWFLHJG.js → context-JFZEJ7W5.js} +15 -13
  111. package/dist/{context-5RADCKTR.js → context-RW5HC47S.js} +71 -35
  112. package/dist/{context-clear-7TSNPAAI.js → context-clear-6ZHBAZZT.js} +54 -28
  113. package/dist/{context-index-W4RLWOQH.js → context-index-BZ4UYMTC.js} +30 -24
  114. package/dist/dispatch-runner-VKBRCWQC.js +1997 -0
  115. package/dist/{docs-5AWSPS37.js → docs-2C2LTVT2.js} +23 -10
  116. package/dist/{doctor-UC5NAJYQ.js → doctor-KI767GSN.js} +27 -17
  117. package/dist/{eval-U6TJHRLX.js → eval-XSSNATB4.js} +29 -16
  118. package/dist/{evidence-YEGUW4L3.js → evidence-UA6AWDQQ.js} +46 -26
  119. package/dist/{evolve-TXARCTPG.js → evolve-QNTFGV6Z.js} +45 -25
  120. package/dist/{extensions-OZFJ3A3G.js → extensions-QVDOHDGJ.js} +16 -7
  121. package/dist/{fleet-6G3DHNYE.js → fleet-Q7UOMUSG.js} +163 -54
  122. package/dist/{fleet-preflight-DSNT37JK.js → fleet-preflight-DDN536IT.js} +7 -4
  123. package/dist/{init-KZ5QTF6M.js → init-WBB65ZHQ.js} +69 -32
  124. package/dist/{memory-73ESV5YC.js → memory-MD3O64RI.js} +48 -27
  125. package/dist/{models-A4PVNWJK.js → models-BZU34YWD.js} +39 -25
  126. package/dist/monitor-MEQA5C3I.js +661 -0
  127. package/dist/{chunk-FCIH3BIZ.js → orchestrator-CGFKEP27.js} +11832 -8687
  128. package/dist/{paths-C4H6IV77.js → paths-UXLN5YYZ.js} +10 -5
  129. package/dist/{preload-6WVMHX3A.js → preload-P6DGH2PZ.js} +2 -2
  130. package/dist/{reset-BGW6OGMV.js → reset-L2FQEE3E.js} +16 -10
  131. package/dist/{run-YTPEYQOH.js → run-IV4Q6RLN.js} +101 -61
  132. package/dist/{share-YIFFV4NQ.js → share-S5BZQC5I.js} +15 -8
  133. package/dist/{skills-2V6RA3OQ.js → skills-LQEKRDTN.js} +34 -14
  134. package/dist/{skills-eval-S2TVJO4F.js → skills-eval-3DC4HEWS.js} +70 -34
  135. package/dist/steer-GGWFUJUD.js +77 -0
  136. package/dist/{targets-TYXLPB23.js → targets-C4SSGQOB.js} +43 -27
  137. package/dist/terminal-lease-IT5JW2NR.js +395 -0
  138. package/dist/{trace-GGOJ6Q6Z.js → trace-PNCASAXC.js} +41 -16
  139. package/dist/{chunk-N6F52NLF.js → tree-sitter-HGKH6LG4.js} +28 -2306
  140. package/dist/{uninstall-LLLT4F4W.js → uninstall-FZCQCDKC.js} +10 -5
  141. package/dist/{upgrade-33G2LMM5.js → upgrade-7TT7SQ3G.js} +45 -25
  142. package/dist/{usage-ZAFSXKKG.js → usage-GV4PKT3M.js} +62 -31
  143. package/dist/verify-G6V4D2G7.js +716 -0
  144. package/dist/web-fetch-2YHJ3KTG.js +638 -0
  145. package/dist/{wiki-generate-NUQCVOQ3.js → wiki-generate-DQF6Z66B.js} +74 -34
  146. package/dist/worker/entry.js +221 -36
  147. package/dist/workspace-G4ZWUIPR.js +22 -0
  148. package/docs/README.md +22 -17
  149. package/docs/acp.md +168 -16
  150. package/docs/alcf-provider.md +1 -1
  151. package/docs/architecture.md +136 -7
  152. package/docs/artifact-versions.md +1 -1
  153. package/docs/built-in-agents.md +1 -1
  154. package/docs/capacity-and-scheduling.md +1 -1
  155. package/docs/commands-and-modes.md +114 -71
  156. package/docs/config-knobs-audit.md +1 -3
  157. package/docs/configuration-and-targets.md +174 -46
  158. package/docs/context-engine.md +29 -6
  159. package/docs/development-pipeline.md +26 -1
  160. package/docs/dispatch-architecture-rationale.md +1 -1
  161. package/docs/documentation-coverage.md +2 -2
  162. package/docs/documentation-guide.md +1 -1
  163. package/docs/environment-variables.md +13 -5
  164. package/docs/eval-runner.md +1 -1
  165. package/docs/evals-internal.md +1 -1
  166. package/docs/evidence-and-memory.md +6 -2
  167. package/docs/evolution.md +2 -2
  168. package/docs/exit-codes-and-output.md +15 -9
  169. package/docs/extensions-and-sharing.md +9 -9
  170. package/docs/fleet-dispatch.md +7 -5
  171. package/docs/git-commit-provenance.md +120 -0
  172. package/docs/glossary.md +1 -1
  173. package/docs/installation-and-lifecycle.md +34 -27
  174. package/docs/middleware-and-components.md +1 -1
  175. package/docs/model-catalog.md +45 -14
  176. package/docs/observability.md +8 -5
  177. package/docs/performance-methodology.md +491 -0
  178. package/docs/pi-boundary.md +72 -0
  179. package/docs/proactive-memory.md +3 -3
  180. package/docs/prompt-envelope-and-tools.md +24 -3
  181. package/docs/provider-adapter-cookbook.md +57 -4
  182. package/docs/release-cut-checklist.md +129 -115
  183. package/docs/safety-model.md +9 -5
  184. package/docs/scientific-validation.md +3 -3
  185. package/docs/session-lifecycle.md +55 -12
  186. package/docs/skills-marketplace.md +12 -8
  187. package/docs/time-conventions.md +1 -1
  188. package/docs/tool-usage.md +3 -3
  189. package/docs/trace-store.md +1 -1
  190. package/docs/troubleshooting.md +10 -7
  191. package/docs/tui-design.md +47 -10
  192. package/docs/worker-dispatch-mechanics.md +1 -1
  193. package/package.json +19 -22
  194. package/skills/coding/ast-grep/SKILL.md +136 -0
  195. package/skills/coding/ast-grep/evals.md +56 -0
  196. package/skills/coding/ast-grep/references/rule_reference.md +297 -0
  197. package/skills/coding/coding-standards/SKILL.md +113 -0
  198. package/skills/coding/coding-standards/evals.md +34 -0
  199. package/skills/coding/prototype/SKILL.md +86 -0
  200. package/skills/coding/prototype/evals.md +42 -0
  201. package/skills/coding/prototype/references/LOGIC.md +67 -0
  202. package/skills/coding/prototype/references/UI.md +112 -0
  203. package/skills/coding/tdd/SKILL.md +101 -0
  204. package/skills/coding/tdd/evals.md +41 -0
  205. package/skills/coding/tdd/references/mocking.md +59 -0
  206. package/skills/coding/tdd/references/tests.md +77 -0
  207. package/skills/context/context-handoff/SKILL.md +126 -0
  208. package/skills/context/context-handoff/evals.md +57 -0
  209. package/skills/context/context-handoff/scripts/new-handoff.sh +26 -0
  210. package/skills/context/context-prime/SKILL.md +95 -0
  211. package/skills/context/context-prime/evals.md +54 -0
  212. package/skills/meta/clio-dev/SKILL.md +91 -0
  213. package/skills/meta/clio-dev/evals.md +45 -0
  214. package/skills/meta/clio-test/SKILL.md +130 -0
  215. package/skills/meta/clio-test/evals.md +43 -0
  216. package/skills/meta/clio-test/references/harness.md +97 -0
  217. package/skills/meta/clio-test/references/test-map.md +59 -0
  218. package/skills/meta/credentials/SKILL.md +125 -0
  219. package/skills/meta/credentials/evals.md +104 -0
  220. package/skills/meta/find-skills/SKILL.md +72 -0
  221. package/skills/meta/find-skills/evals.md +47 -0
  222. package/skills/meta/herdr/SKILL.md +127 -0
  223. package/skills/meta/herdr/evals.md +38 -0
  224. package/skills/meta/skill-craft/SKILL.md +102 -0
  225. package/skills/meta/skill-craft/evals.md +41 -0
  226. package/skills/planning/architecture/SKILL.md +129 -0
  227. package/skills/planning/architecture/evals.md +36 -0
  228. package/skills/planning/backlog/SKILL.md +90 -0
  229. package/skills/planning/backlog/evals.md +43 -0
  230. package/skills/planning/prd/SKILL.md +82 -0
  231. package/skills/planning/prd/evals.md +49 -0
  232. package/skills/planning/product-intent/SKILL.md +112 -0
  233. package/skills/planning/product-intent/evals.md +36 -0
  234. package/skills/planning/tech-spec/SKILL.md +115 -0
  235. package/skills/planning/tech-spec/evals.md +47 -0
  236. package/skills/registry.yaml +136 -0
  237. package/skills/research/arxiv-literature/SKILL.md +104 -0
  238. package/skills/research/arxiv-literature/evals.md +58 -0
  239. package/skills/research/experiment-protocol/SKILL.md +122 -0
  240. package/skills/research/experiment-protocol/evals.md +91 -0
  241. package/skills/research/scientific-debugging/SKILL.md +119 -0
  242. package/skills/research/scientific-debugging/evals.md +138 -0
  243. package/skills/research/scientific-modernization/SKILL.md +138 -0
  244. package/skills/research/scientific-modernization/evals.md +84 -0
  245. package/skills/workflow/design-council/SKILL.md +139 -0
  246. package/skills/workflow/design-council/evals.md +97 -0
  247. package/skills/workflow/grill-me/SKILL.md +186 -0
  248. package/skills/workflow/grill-me/evals.md +78 -0
  249. package/skills/workflow/workflow-distiller/SKILL.md +136 -0
  250. package/skills/workflow/workflow-distiller/evals.md +107 -0
  251. package/src/cli/acp.ts +31 -4
  252. package/src/cli/clio.ts +68 -6
  253. package/src/cli/config-inspect.ts +28 -22
  254. package/src/cli/configure.ts +47 -9
  255. package/src/cli/context-clear.ts +2 -2
  256. package/src/cli/context-index.ts +21 -23
  257. package/src/cli/context.ts +13 -8
  258. package/src/cli/default-target.ts +9 -17
  259. package/src/cli/docs.ts +11 -5
  260. package/src/cli/evidence.ts +4 -1
  261. package/src/cli/extensions.ts +10 -1
  262. package/src/cli/fleet.ts +47 -6
  263. package/src/cli/index.ts +55 -26
  264. package/src/cli/memory.ts +3 -1
  265. package/src/cli/models.ts +1 -1
  266. package/src/cli/modes/json-stream.ts +37 -1
  267. package/src/cli/modes/print.ts +24 -9
  268. package/src/cli/run.ts +2 -2
  269. package/src/cli/skills-eval.ts +23 -8
  270. package/src/cli/skills.ts +19 -4
  271. package/src/cli/targets.ts +4 -0
  272. package/src/cli/text-layout.ts +15 -5
  273. package/src/cli/trace.ts +62 -14
  274. package/src/cli/upgrade.ts +18 -2
  275. package/src/cli/usage.ts +10 -3
  276. package/src/cli/wiki-generate.ts +2 -1
  277. package/src/core/agent-environment.ts +7 -0
  278. package/src/core/bash-exec.ts +72 -1
  279. package/src/core/boot-trace.ts +9 -4
  280. package/src/core/bus-events.ts +20 -4
  281. package/src/core/commit-attribution.ts +157 -0
  282. package/src/core/compile-cache.ts +159 -0
  283. package/src/core/config.ts +131 -2
  284. package/src/core/defaults.ts +39 -5
  285. package/src/core/domain-loader.ts +12 -5
  286. package/src/core/git-commit-attribution.ts +362 -0
  287. package/src/core/incomplete-installation.ts +45 -0
  288. package/src/core/response-schema.ts +1 -1
  289. package/src/core/safe-exec.ts +13 -1
  290. package/src/core/settings-layers.ts +155 -21
  291. package/src/core/skill-activation.ts +1 -1
  292. package/src/core/startup-timer.ts +3 -3
  293. package/src/core/state-file-lock.ts +13 -1
  294. package/src/core/termination.ts +78 -5
  295. package/src/domains/config/classify.ts +15 -3
  296. package/src/domains/config/extension.ts +19 -13
  297. package/src/domains/config/index.ts +10 -0
  298. package/src/domains/config/keybindings.ts +42 -6
  299. package/src/domains/context/bootstrap-prompt.ts +1 -1
  300. package/src/domains/context/bootstrap.ts +111 -18
  301. package/src/domains/context/clear.ts +16 -11
  302. package/src/domains/context/clio-md.ts +111 -9
  303. package/src/domains/context/codewiki/artifact.ts +400 -0
  304. package/src/domains/context/codewiki/build-worker-protocol.ts +24 -0
  305. package/src/domains/context/codewiki/build-worker.ts +54 -0
  306. package/src/domains/context/codewiki/coordinator.ts +182 -0
  307. package/src/domains/context/codewiki/indexer.ts +59 -144
  308. package/src/domains/context/codewiki/paths.ts +67 -0
  309. package/src/domains/context/codewiki/schema.ts +80 -0
  310. package/src/domains/context/codewiki/tree-sitter.ts +1 -1
  311. package/src/domains/context/contract.ts +11 -5
  312. package/src/domains/context/extension.ts +94 -143
  313. package/src/domains/context/fingerprint.ts +3 -1
  314. package/src/domains/context/index.ts +12 -22
  315. package/src/domains/context/project-metadata.ts +19 -0
  316. package/src/domains/context/prompt-context.ts +9 -10
  317. package/src/domains/context/refresh.ts +29 -21
  318. package/src/domains/context/runtime.ts +17 -0
  319. package/src/domains/context/wiki/generate.ts +39 -34
  320. package/src/domains/context/wiki/plan.ts +1 -1
  321. package/src/domains/context/wiki/prompts.ts +21 -8
  322. package/src/domains/dispatch/code-step.ts +20 -1
  323. package/src/domains/dispatch/extension.ts +158 -21
  324. package/src/domains/dispatch/failure-classification.ts +6 -0
  325. package/src/domains/dispatch/fleet-commit-attribution.ts +56 -0
  326. package/src/domains/dispatch/orphan-recovery.ts +50 -8
  327. package/src/domains/dispatch/receipt-integrity.ts +5 -0
  328. package/src/domains/dispatch/state.ts +31 -6
  329. package/src/domains/dispatch/transport.ts +2 -1
  330. package/src/domains/dispatch/types.ts +14 -0
  331. package/src/domains/dispatch/worker-spawn.ts +21 -2
  332. package/src/domains/eval/metrics/context.ts +1 -1
  333. package/src/domains/eval/types.ts +0 -1
  334. package/src/domains/evidence/build.ts +41 -1
  335. package/src/domains/lifecycle/migrations/2026-08-18-lmstudio-runtime-id.ts +52 -0
  336. package/src/domains/lifecycle/migrations/index.ts +24 -4
  337. package/src/domains/middleware/hooks-io.ts +12 -0
  338. package/src/domains/middleware/skills-reminder.ts +30 -15
  339. package/src/domains/prompts/compiler.ts +142 -84
  340. package/src/domains/prompts/contract.ts +18 -2
  341. package/src/domains/prompts/extension.ts +39 -7
  342. package/src/domains/prompts/fragment-loader.ts +0 -1
  343. package/src/domains/prompts/fragments/identity/clio.md +2 -4
  344. package/src/domains/prompts/fragments/identity/docs-routing.md +10 -0
  345. package/src/domains/prompts/fragments/identity/self-awareness.md +1 -45
  346. package/src/domains/prompts/fragments/operating/contract.md +4 -50
  347. package/src/domains/prompts/fragments/operating/delegation.md +42 -0
  348. package/src/domains/prompts/fragments/operating/skills.md +26 -0
  349. package/src/domains/prompts/fragments/operating/worker.md +16 -0
  350. package/src/domains/prompts/fragments/safety/auto-edit.md +5 -5
  351. package/src/domains/prompts/fragments/safety/full-auto.md +3 -3
  352. package/src/domains/prompts/fragments/safety/read-only.md +4 -4
  353. package/src/domains/prompts/fragments/safety/suggest.md +2 -2
  354. package/src/domains/prompts/fragments/wiki/page.md +10 -0
  355. package/src/domains/prompts/fragments/wiki/plan.md +10 -0
  356. package/src/domains/prompts/preload.ts +3 -3
  357. package/src/domains/providers/auth/api-key.ts +1 -1
  358. package/src/domains/providers/auth/backend-file.ts +20 -10
  359. package/src/domains/providers/auth/backend-memory.ts +59 -4
  360. package/src/domains/providers/auth/boot-status.ts +65 -0
  361. package/src/domains/providers/auth/oauth.ts +2 -1
  362. package/src/domains/providers/auth/storage.ts +97 -38
  363. package/src/domains/providers/capabilities.ts +12 -4
  364. package/src/domains/providers/contract.ts +15 -4
  365. package/src/domains/providers/extension.ts +18 -6
  366. package/src/domains/providers/model-runtime-capabilities.ts +15 -4
  367. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +118 -35
  368. package/src/domains/providers/plugins.ts +5 -3
  369. package/src/domains/providers/probe/fingerprint.ts +25 -5
  370. package/src/domains/providers/registry.ts +31 -10
  371. package/src/domains/providers/runtimes/boot-manifest.ts +55 -0
  372. package/src/domains/providers/runtimes/builtins.ts +2 -2
  373. package/src/domains/providers/runtimes/common/lmstudio-http.ts +423 -0
  374. package/src/domains/providers/runtimes/common/local-synth.ts +6 -7
  375. package/src/domains/providers/runtimes/local-native/lmstudio.ts +241 -0
  376. package/src/domains/providers/support.ts +6 -3
  377. package/src/domains/providers/types/local-model-quirks.ts +7 -9
  378. package/src/domains/providers/types/runtime-descriptor.ts +12 -1
  379. package/src/domains/providers/types/target-descriptor.ts +22 -0
  380. package/src/domains/resources/contract.ts +0 -1
  381. package/src/domains/resources/extension.ts +1 -3
  382. package/src/domains/resources/loader.ts +3 -4
  383. package/src/domains/resources/prompts/loader.ts +16 -2
  384. package/src/domains/resources/prompts/substitute.ts +1 -65
  385. package/src/domains/resources/skills/content-hash.ts +2 -0
  386. package/src/domains/resources/skills/install.ts +17 -0
  387. package/src/domains/resources/skills/loader.ts +17 -10
  388. package/src/domains/resources/skills/marketplace.ts +55 -9
  389. package/src/domains/safety/action-classifier.ts +4 -2
  390. package/src/domains/safety/audit.ts +8 -2
  391. package/src/domains/safety/extension.ts +1 -1
  392. package/src/domains/session/compaction/branch-summary.ts +3 -2
  393. package/src/domains/session/compaction/cut-point.ts +2 -1
  394. package/src/domains/session/compaction/tokens.ts +2 -1
  395. package/src/domains/session/context-ledger.ts +14 -0
  396. package/src/domains/session/contract.ts +15 -0
  397. package/src/domains/session/decision-board.ts +190 -0
  398. package/src/domains/session/entries.ts +66 -3
  399. package/src/domains/session/extension.ts +93 -12
  400. package/src/domains/session/retry.ts +10 -18
  401. package/src/domains/session/session-artifacts.ts +107 -0
  402. package/src/domains/session/task-board.ts +207 -13
  403. package/src/domains/session/tree/active-path.ts +44 -5
  404. package/src/domains/session/tree/fork.ts +26 -27
  405. package/src/domains/session/tree/preview.ts +2 -2
  406. package/src/domains/session/workspace/git-probe.ts +17 -11
  407. package/src/domains/user-tasks/store.ts +297 -0
  408. package/src/engine/acp/errors.ts +96 -0
  409. package/src/engine/acp/server.ts +1728 -146
  410. package/src/engine/acp/transport.ts +135 -14
  411. package/src/engine/acp/types.ts +26 -0
  412. package/src/engine/agent.ts +3 -3
  413. package/src/engine/ai.ts +32 -27
  414. package/src/engine/alcf-oauth.ts +26 -19
  415. package/src/engine/api-registry.ts +223 -0
  416. package/src/engine/apis/index.ts +3 -7
  417. package/src/engine/apis/llamacpp-residency.ts +49 -9
  418. package/src/engine/apis/lmstudio-residency.ts +5 -21
  419. package/src/engine/apis/lmstudio.ts +243 -0
  420. package/src/engine/apis/ollama-native.ts +24 -3
  421. package/src/engine/apis/openai-completions.ts +170 -91
  422. package/src/engine/apis/residency.ts +139 -3
  423. package/src/engine/apis/types.ts +16 -0
  424. package/src/engine/env-api-keys.ts +98 -0
  425. package/src/engine/gemma-channel-filter.ts +223 -0
  426. package/src/engine/instrumented-tui.ts +192 -0
  427. package/src/engine/messages.ts +14 -0
  428. package/src/engine/models.ts +42 -0
  429. package/src/engine/oauth.ts +16 -12
  430. package/src/engine/prompt-templates.ts +1 -0
  431. package/src/engine/provider-payload.ts +16 -59
  432. package/src/engine/strip-tokenizer-sentinels.ts +1 -1
  433. package/src/engine/truncate.ts +9 -0
  434. package/src/engine/tui.ts +17 -9
  435. package/src/engine/types.ts +3 -6
  436. package/src/engine/worker-runtime-capabilities.ts +5 -0
  437. package/src/engine/worker-runtime.ts +1 -1
  438. package/src/engine/worker-tools.ts +9 -4
  439. package/src/entry/boot-options.ts +50 -0
  440. package/src/entry/orchestrator.ts +288 -150
  441. package/src/interactive/application-controller.ts +89 -2
  442. package/src/interactive/chat-loop.ts +266 -41
  443. package/src/interactive/chat-panel.ts +173 -47
  444. package/src/interactive/chat-renderer.ts +262 -72
  445. package/src/interactive/clio-editor.ts +3 -8
  446. package/src/interactive/command-fallbacks.ts +2 -2
  447. package/src/interactive/context-overlay.ts +27 -1
  448. package/src/interactive/editor-submit.ts +228 -24
  449. package/src/interactive/export-html/ansi-to-html.ts +161 -0
  450. package/src/interactive/export-html/index.ts +51 -0
  451. package/src/interactive/export-html/template.ts +45 -0
  452. package/src/interactive/export-html/tool-renderer.ts +54 -0
  453. package/src/interactive/footer/dashboard.ts +4 -0
  454. package/src/interactive/footer/notifications.ts +1 -1
  455. package/src/interactive/footer/widgets.ts +20 -2
  456. package/src/interactive/footer-panel.ts +2 -2
  457. package/src/interactive/format-time.ts +14 -2
  458. package/src/interactive/interactive-application.ts +201 -17
  459. package/src/interactive/interactive-event-projection.ts +6 -1
  460. package/src/interactive/interactive-input-runtime.ts +50 -4
  461. package/src/interactive/interactive-presentation.ts +151 -19
  462. package/src/interactive/interactive-shell.ts +268 -14
  463. package/src/interactive/interactive-slash-runtime.ts +176 -114
  464. package/src/interactive/interactive-tickers.ts +38 -7
  465. package/src/interactive/keybinding-manager.ts +1 -1
  466. package/src/interactive/layout.ts +40 -3
  467. package/src/interactive/overlay-frame.ts +1 -1
  468. package/src/interactive/overlay-general-openers.ts +58 -1
  469. package/src/interactive/overlay-key-routing.ts +3 -0
  470. package/src/interactive/overlay-lifecycle.ts +13 -0
  471. package/src/interactive/overlay-permission-lifecycle.ts +2 -1
  472. package/src/interactive/overlay-session-lifecycle.ts +69 -12
  473. package/src/interactive/overlays/decisions.ts +300 -0
  474. package/src/interactive/overlays/help-reference.ts +15 -10
  475. package/src/interactive/overlays/model-selector.ts +34 -16
  476. package/src/interactive/overlays/session-selector.ts +18 -0
  477. package/src/interactive/overlays/settings.ts +105 -17
  478. package/src/interactive/overlays/skills-hub.ts +4 -4
  479. package/src/interactive/overlays/tree-selector.ts +41 -6
  480. package/src/interactive/render-trace.ts +499 -90
  481. package/src/interactive/renderers/compaction-summary.ts +2 -2
  482. package/src/interactive/renderers/diff.ts +115 -104
  483. package/src/interactive/renderers/mermaid.ts +53 -0
  484. package/src/interactive/renderers/tool-execution.ts +386 -133
  485. package/src/interactive/renderers/worker-entry.ts +20 -4
  486. package/src/interactive/session-switch-settlement.ts +10 -0
  487. package/src/interactive/slash-autocomplete.ts +6 -114
  488. package/src/interactive/slash-commands.ts +135 -47
  489. package/src/interactive/slash-spec.ts +9 -38
  490. package/src/interactive/status/controller.ts +5 -1
  491. package/src/interactive/stdout-backpressure.ts +99 -0
  492. package/src/interactive/stream-pacer.ts +530 -0
  493. package/src/interactive/stream-pacing-policy.ts +66 -0
  494. package/src/interactive/tasks-overlay.ts +368 -14
  495. package/src/interactive/terminal-lease.ts +485 -0
  496. package/src/interactive/theme/tokens.ts +1 -1
  497. package/src/interactive/turn-context.ts +4 -3
  498. package/src/interactive/turn-persistence.ts +30 -13
  499. package/src/interactive/turn-queues.ts +12 -0
  500. package/src/interactive/turn-recovery.ts +25 -8
  501. package/src/interactive/turn-runtime.ts +79 -12
  502. package/src/interactive/turn-state.ts +10 -0
  503. package/src/interactive/view/artifacts.ts +114 -4
  504. package/src/interactive/view/view-overlay.ts +3 -0
  505. package/src/interactive/welcome-dashboard.ts +17 -16
  506. package/src/interactive/worker-receipts.ts +52 -3
  507. package/src/interactive/worker-stream.ts +5 -1
  508. package/src/tools/agent-tools.ts +23 -3
  509. package/src/tools/artifact.ts +2 -2
  510. package/src/tools/ask-user.ts +23 -13
  511. package/src/tools/bash.ts +30 -2
  512. package/src/tools/bootstrap.ts +34 -431
  513. package/src/tools/builtin-tool-catalog.ts +265 -0
  514. package/src/tools/codewiki/code-nav-surface.ts +29 -0
  515. package/src/tools/codewiki/code-nav.ts +8 -22
  516. package/src/tools/codewiki/shared.ts +41 -38
  517. package/src/tools/context/docs-engine.ts +14 -3
  518. package/src/tools/context/index.ts +107 -28
  519. package/src/tools/context/surface.ts +19 -0
  520. package/src/tools/core-bootstrap.ts +168 -0
  521. package/src/tools/credential-present.ts +5 -5
  522. package/src/tools/dispatch-admission.ts +533 -0
  523. package/src/tools/dispatch-background.ts +54 -0
  524. package/src/tools/dispatch-event-text.ts +6 -0
  525. package/src/tools/dispatch-plan.ts +9 -4
  526. package/src/tools/dispatch-run-events.ts +238 -0
  527. package/src/tools/dispatch-runner.ts +2370 -0
  528. package/src/tools/dispatch-scout-admission.ts +295 -0
  529. package/src/tools/dispatch-types.ts +77 -0
  530. package/src/tools/dispatch.ts +67 -3161
  531. package/src/tools/find.ts +4 -2
  532. package/src/tools/grep.ts +2 -2
  533. package/src/tools/lazy-tool.ts +60 -0
  534. package/src/tools/ledger.ts +3 -3
  535. package/src/tools/monitor-surface.ts +36 -0
  536. package/src/tools/monitor.ts +2 -32
  537. package/src/tools/observers.ts +2 -2
  538. package/src/tools/registry.ts +39 -27
  539. package/src/tools/safe-exec.ts +2 -2
  540. package/src/tools/steer-surface.ts +17 -0
  541. package/src/tools/steer.ts +2 -13
  542. package/src/tools/tasks.ts +108 -11
  543. package/src/tools/truncate.ts +25 -184
  544. package/src/tools/verify/frontend.ts +3 -1
  545. package/src/tools/verify/index.ts +3 -38
  546. package/src/tools/verify/surface.ts +46 -0
  547. package/src/tools/web-fetch-surface.ts +23 -0
  548. package/src/tools/web-fetch.ts +2 -20
  549. package/src/tools/write.ts +7 -2
  550. package/src/worker/entry.ts +39 -2
  551. package/src/worker/spec-contract.ts +26 -5
  552. package/dist/chunk-7SS2CTV2.js +0 -61361
  553. package/dist/chunk-DKGKUHFA.js +0 -924
  554. package/dist/chunk-GEP36Y4X.js +0 -12796
  555. package/dist/chunk-XYWBQRDM.js +0 -137
  556. package/dist/clio-BZVGEUFJ.js +0 -58
  557. package/dist/configure-S7S6F6CL.js +0 -32
  558. package/docs/html/agents_blueprint.html +0 -936
  559. package/docs/html/alcf_blueprint.html +0 -324
  560. package/docs/html/architecture_blueprint.html +0 -850
  561. package/docs/html/commands_blueprint.html +0 -939
  562. package/docs/html/config_knobs_audit_blueprint.html +0 -178
  563. package/docs/html/configuration_blueprint.html +0 -1080
  564. package/docs/html/context_blueprint.html +0 -603
  565. package/docs/html/documentation_blueprint.html +0 -832
  566. package/docs/html/environment_blueprint.html +0 -404
  567. package/docs/html/eval_blueprint.html +0 -743
  568. package/docs/html/evals_internal_blueprint.html +0 -190
  569. package/docs/html/evolution_blueprint.html +0 -674
  570. package/docs/html/extensions_blueprint.html +0 -2065
  571. package/docs/html/fleet_dispatch_blueprint.html +0 -286
  572. package/docs/html/index.html +0 -919
  573. package/docs/html/lifecycle_blueprint.html +0 -723
  574. package/docs/html/memory_blueprint.html +0 -699
  575. package/docs/html/middleware_blueprint.html +0 -664
  576. package/docs/html/models_blueprint.html +0 -2366
  577. package/docs/html/observability_blueprint.html +0 -683
  578. package/docs/html/provider_adapter_blueprint.html +0 -245
  579. package/docs/html/safety_blueprint.html +0 -1386
  580. package/docs/html/shared.css +0 -571
  581. package/docs/html/shared.js +0 -143
  582. package/docs/html/skills_blueprint.html +0 -671
  583. package/docs/html/soak_blueprint.html +0 -182
  584. package/docs/html/tool_usage_blueprint.html +0 -350
  585. package/docs/html/tools_blueprint.html +0 -2249
  586. package/docs/html/trace_blueprint.html +0 -235
  587. package/docs/html/tui_design_blueprint.html +0 -374
  588. package/docs/html/validation_blueprint.html +0 -961
  589. package/docs/html/worker_dispatch_blueprint.html +0 -231
  590. package/src/core/release.ts +0 -2
  591. package/src/domains/providers/runtimes/common/lmstudio-logger.ts +0 -32
  592. package/src/domains/providers/runtimes/local-native/lmstudio-native.ts +0 -491
  593. package/src/engine/apis/lmstudio-native.ts +0 -1438
  594. package/src/engine/apis/thinking-replay.ts +0 -11
  595. package/src/tools/string-enum.ts +0 -15
@@ -1,1438 +0,0 @@
1
- import { randomUUID } from "node:crypto";
2
- import type {
3
- Api,
4
- AssistantMessage,
5
- AssistantMessageEventStream,
6
- Context,
7
- ImageContent,
8
- Message,
9
- Model,
10
- SimpleStreamOptions,
11
- StreamOptions,
12
- TextContent,
13
- ThinkingContent,
14
- Tool,
15
- ToolCall,
16
- Usage,
17
- } from "@earendil-works/pi-ai";
18
- import { createAssistantMessageEventStream } from "@earendil-works/pi-ai";
19
- import type { ApiProvider } from "@earendil-works/pi-ai/compat";
20
- import {
21
- type ChatHistoryData,
22
- type ChatMessageData,
23
- type ChatMessagePartFileData,
24
- type ChatMessagePartTextData,
25
- type ChatMessagePartToolCallRequestData,
26
- type ChatMessagePartToolCallResultData,
27
- type FileHandle,
28
- type FunctionToolCallRequest,
29
- type LLMLoadModelConfig,
30
- type LLMPredictionStopReason,
31
- type LLMRespondOpts,
32
- type LLMTool,
33
- LMStudioClient,
34
- } from "@lmstudio/sdk";
35
- import { runOverrides } from "../../core/run-overrides.js";
36
- import {
37
- reasoningClassForMechanism,
38
- resolveModelRuntimeCapabilitiesForModel,
39
- } from "../../domains/providers/model-runtime-capabilities.js";
40
- import { lmStudioQuietLogger } from "../../domains/providers/runtimes/common/lmstudio-logger.js";
41
- import type { ThinkingLevel } from "../../domains/providers/types/capability-flags.js";
42
- import {
43
- asKvCacheQuant,
44
- KV_CACHE_QUANTS,
45
- type LocalModelQuirks,
46
- type SamplingProfile,
47
- } from "../../domains/providers/types/local-model-quirks.js";
48
- import { ceilChars } from "../../domains/session/context-accounting.js";
49
- import { calculateEngineCost, parseEngineJsonWithRepair, parseEngineStreamingJson } from "../ai.js";
50
- import { HarmonyResponseParser } from "../harmony-response.js";
51
- import { createSentinelStripper } from "../strip-tokenizer-sentinels.js";
52
- import { type DegradedInferenceWatchdog, startDegradedInferenceWatchdog } from "./degraded-inference.js";
53
- import { coResidentContextCeiling, duplicateInstances, fitLoadContextLength } from "./lmstudio-residency.js";
54
- import { openAICompletionsApiProvider } from "./openai-completions.js";
55
- import { remainingContextMaxTokens } from "./output-budget.js";
56
- import {
57
- emitResidencyNotice,
58
- type ResidencyAdapter,
59
- type ResidencyPlan,
60
- reconcileResidency,
61
- residencyManagedFor,
62
- } from "./residency.js";
63
- import { withResidencyLock } from "./residency-lock.js";
64
- import type { ResidentModelInfo } from "./resident-models.js";
65
- import { mergeSamplingOverride } from "./sampling-overrides.js";
66
- import { formatThinkingForReplay } from "./thinking-replay.js";
67
-
68
- const EMPTY_TOOL_ARGUMENTS_ERROR =
69
- "LM Studio SDK returned empty tool-call arguments; this model's chat template may not be compatible. Try the openai-compat runtime against the same gateway.";
70
-
71
- type RuntimeLifecycle = "user-managed" | "clio-managed";
72
-
73
- interface ClioRuntimeMetadata {
74
- clio?: {
75
- targetId: string;
76
- runtimeId: string;
77
- /** Present only when settings set the target lifecycle explicitly. */
78
- lifecycle?: RuntimeLifecycle;
79
- gateway?: boolean;
80
- quirks?: LocalModelQuirks;
81
- };
82
- }
83
-
84
- function normalizeBaseUrl(url: string): string {
85
- const trimmed = url.endsWith("/") ? url.slice(0, -1) : url;
86
- if (trimmed.startsWith("ws://") || trimmed.startsWith("wss://")) return trimmed;
87
- if (trimmed.startsWith("https://")) return `wss://${trimmed.slice("https://".length)}`;
88
- if (trimmed.startsWith("http://")) return `ws://${trimmed.slice("http://".length)}`;
89
- return trimmed;
90
- }
91
-
92
- function normalizeHttpBaseUrl(url: string): string {
93
- const trimmed = url.endsWith("/") ? url.slice(0, -1) : url;
94
- if (trimmed.startsWith("http://") || trimmed.startsWith("https://")) return trimmed;
95
- if (trimmed.startsWith("ws://")) return `http://${trimmed.slice("ws://".length)}`;
96
- if (trimmed.startsWith("wss://")) return `https://${trimmed.slice("wss://".length)}`;
97
- return `http://${trimmed}`;
98
- }
99
-
100
- // One loaded model instance in LM Studio's resident set, as the SDK socket
101
- // reports it. The reconciler (residency.ts) lists residents through these
102
- // entries, and a post-failure fallback swap unloads through them; loads go
103
- // through the SDK's JIT model open. `identifier` is per-instance rather than
104
- // per-model: LM Studio can hold two instances of one model key, and telling
105
- // them apart is what lets the duplicate be released.
106
- export interface ResidentModelEntry {
107
- readonly modelKey: string;
108
- readonly identifier?: string;
109
- readonly sizeBytes?: number;
110
- unload(): Promise<void>;
111
- }
112
-
113
- /**
114
- * Collapse LM Studio's per-instance listing into the per-model view the
115
- * reconciler decides on. Two instances of one model key are one resident model
116
- * as far as eviction is concerned; the duplicate sweep handles the extra
117
- * instance separately, before any eviction question arises.
118
- */
119
- function residentModelInfos(entries: ReadonlyArray<ResidentModelEntry>): ResidentModelInfo[] {
120
- const byKey = new Map<string, ResidentModelInfo>();
121
- for (const entry of entries) {
122
- if (byKey.has(entry.modelKey)) continue;
123
- byKey.set(entry.modelKey, {
124
- modelId: entry.modelKey,
125
- ...(entry.sizeBytes !== undefined ? { sizeBytes: entry.sizeBytes } : {}),
126
- });
127
- }
128
- return [...byKey.values()];
129
- }
130
-
131
- function toolToLmStudio(tool: Tool): LLMTool {
132
- const fn: LLMTool["function"] = {
133
- name: tool.name,
134
- parameters: tool.parameters as unknown as NonNullable<LLMTool["function"]["parameters"]>,
135
- };
136
- if (tool.description) fn.description = tool.description;
137
- return { type: "function", function: fn };
138
- }
139
-
140
- type UserPart = ChatMessagePartTextData | ChatMessagePartFileData;
141
- type AssistantPart = ChatMessagePartTextData | ChatMessagePartFileData | ChatMessagePartToolCallRequestData;
142
-
143
- interface PredictionStatsLike {
144
- promptTokensCount?: number;
145
- predictedTokensCount?: number;
146
- totalTokensCount?: number;
147
- stopReason?: LLMPredictionStopReason;
148
- }
149
-
150
- interface PredictionResultLike {
151
- stats: PredictionStatsLike;
152
- }
153
-
154
- interface OngoingPredictionLike {
155
- result(): Promise<PredictionResultLike>;
156
- }
157
-
158
- interface LmStudioPredictionHandle {
159
- respond(history: ChatHistoryData, opts: LLMRespondOpts<unknown>): OngoingPredictionLike;
160
- }
161
-
162
- interface LmStudioRunClient {
163
- files: {
164
- prepareImageBase64(fileName: string, contentBase64: string): Promise<FileHandle>;
165
- };
166
- llm: {
167
- listLoaded(): Promise<ReadonlyArray<ResidentModelEntry>>;
168
- model(
169
- modelId: string,
170
- opts: { signal: AbortSignal; verbose: boolean; config?: LLMLoadModelConfig },
171
- ): Promise<LmStudioPredictionHandle>;
172
- };
173
- }
174
-
175
- export interface LmStudioRunDeps {
176
- createClient(opts: ConstructorParameters<typeof LMStudioClient>[0]): LmStudioRunClient;
177
- reconcile(adapter: ResidencyAdapter): Promise<ResidencyPlan>;
178
- discoverLoadedContext(baseUrl: string, modelId: string, signal: AbortSignal): Promise<number | undefined>;
179
- /** Cross-process residency-mutation serializer; defaults to the state-dir lock file. */
180
- lock?<T>(targetKey: string, fn: () => Promise<T>): Promise<T>;
181
- }
182
-
183
- /**
184
- * Out-of-band hints from the api-provider wrapper. `thinkingLevel` is the
185
- * Clio ThinkingLevel for the in-flight turn; `runStream` resolves it through
186
- * the provider-domain runtime capability layer before choosing catalog
187
- * sampling. The bare `stream` path (no SimpleStreamOptions) leaves it
188
- * undefined, in which case the helper falls back to the model's `reasoning`
189
- * capability flag.
190
- */
191
- export interface RunStreamHints {
192
- thinkingLevel?: ThinkingLevel;
193
- }
194
-
195
- // Worker-process-scoped cache of LMStudioClient instances keyed on
196
- // `${baseUrl}|${clientPasskey ?? ""}`. Creating a fresh client per turn
197
- // allocates a new WebSocket session against the LM Studio server, and abandoning
198
- // it without [Symbol.asyncDispose] leaks server-side channel state. The server
199
- // then logs `Received channelSend for unknown channel` warnings on the next
200
- // abort because the previous turn's controller fires after the SDK has already
201
- // torn the channel down. Reusing one client across turns avoids that race and
202
- // keeps the WebSocket warm for high-latency remote LM Studio hosts. The
203
- // cache lives in the worker subprocess and dies with it; it is not shared
204
- // across worker processes.
205
- const lmStudioClientCache = new Map<string, LMStudioClient>();
206
-
207
- function lmStudioCacheKey(baseUrl: string, clientPasskey: string | undefined): string {
208
- return `${baseUrl}|${clientPasskey ?? ""}`;
209
- }
210
-
211
- export async function disposeLmStudioClients(): Promise<void> {
212
- const clients = Array.from(lmStudioClientCache.values());
213
- lmStudioClientCache.clear();
214
- await Promise.all(
215
- clients.map(async (client) => {
216
- try {
217
- await client[Symbol.asyncDispose]();
218
- } catch {
219
- // Disposal is best-effort: the worker is already shutting down,
220
- // and a noisy close should not block process exit.
221
- }
222
- }),
223
- );
224
- }
225
-
226
- function getOrCreateLmStudioClient(
227
- opts: ConstructorParameters<typeof LMStudioClient>[0],
228
- create: (o: ConstructorParameters<typeof LMStudioClient>[0]) => LmStudioRunClient,
229
- ): LmStudioRunClient {
230
- const baseUrl = opts?.baseUrl ?? "";
231
- const passkey = opts?.clientPasskey;
232
- const key = lmStudioCacheKey(baseUrl, passkey);
233
- const cached = lmStudioClientCache.get(key);
234
- if (cached) return cached as unknown as LmStudioRunClient;
235
- const created = create(opts);
236
- // `created` is the structural LmStudioRunClient view of an LMStudioClient.
237
- // In production `defaultRunDeps.createClient` returns `new LMStudioClient(...)`,
238
- // which satisfies the cache value type. Tests inject fakes through the
239
- // `deps.createClient` parameter and bypass this cache entirely (see
240
- // `runStream`'s caller-provided-deps branch below).
241
- lmStudioClientCache.set(key, created as unknown as LMStudioClient);
242
- return created;
243
- }
244
-
245
- const defaultRunDeps: LmStudioRunDeps = {
246
- createClient: (opts) =>
247
- getOrCreateLmStudioClient(opts, (o) => new LMStudioClient({ ...(o ?? {}), logger: lmStudioQuietLogger })),
248
- reconcile: reconcileResidency,
249
- discoverLoadedContext: discoverLoadedContextLength,
250
- lock: withResidencyLock,
251
- };
252
-
253
- function fileHandleToPart(handle: FileHandle): ChatMessagePartFileData {
254
- return {
255
- type: "file",
256
- name: handle.name,
257
- identifier: handle.identifier,
258
- sizeBytes: handle.sizeBytes,
259
- fileType: handle.type,
260
- };
261
- }
262
-
263
- function imageFileName(mimeType: string, index: number): string {
264
- const slash = mimeType.indexOf("/");
265
- const ext = slash >= 0 ? mimeType.slice(slash + 1) : "png";
266
- const safeExt = ext.replace(/[^a-zA-Z0-9]/g, "") || "png";
267
- return `clio-image-${index}.${safeExt}`;
268
- }
269
-
270
- async function userMessage(
271
- client: Pick<LmStudioRunClient, "files">,
272
- content: string | (TextContent | ImageContent)[],
273
- imageCounter: { next: number },
274
- ): Promise<ChatMessageData> {
275
- if (typeof content === "string") {
276
- return { role: "user", content: [{ type: "text", text: content }] };
277
- }
278
- const parts: UserPart[] = [];
279
- for (const block of content) {
280
- if (block.type === "text") {
281
- parts.push({ type: "text", text: block.text });
282
- continue;
283
- }
284
- if (block.type === "image") {
285
- const fileName = imageFileName(block.mimeType, imageCounter.next++);
286
- let handle: FileHandle;
287
- try {
288
- handle = await client.files.prepareImageBase64(fileName, block.data);
289
- } catch (err) {
290
- const msg = err instanceof Error ? err.message : String(err);
291
- throw new Error(`LM Studio prepareImage failed for ${fileName}: ${msg}`);
292
- }
293
- parts.push(fileHandleToPart(handle));
294
- }
295
- }
296
- return { role: "user", content: parts };
297
- }
298
-
299
- export function assistantMessage(
300
- content: AssistantMessage["content"],
301
- opts?: { harmony?: boolean; preserveThinking?: boolean },
302
- ): ChatMessageData {
303
- const parts: AssistantPart[] = [];
304
- const thinkingParts: string[] = [];
305
- const preserveThinking = opts?.preserveThinking ?? true;
306
- for (const block of content) {
307
- if (block.type === "text") {
308
- parts.push({ type: "text", text: block.text });
309
- } else if (block.type === "toolCall") {
310
- const req: FunctionToolCallRequest = {
311
- type: "function",
312
- name: block.name,
313
- arguments: block.arguments,
314
- };
315
- if (block.id) req.id = block.id;
316
- parts.push({ type: "toolCallRequest", toolCallRequest: req });
317
- } else if (preserveThinking && block.type === "thinking") {
318
- const thinkingVal = (block as ThinkingContent).thinking;
319
- if (thinkingVal) {
320
- thinkingParts.push(thinkingVal);
321
- }
322
- }
323
- }
324
- if (thinkingParts.length > 0) {
325
- const joined = thinkingParts.join("\n");
326
- const harmony = opts?.harmony ?? false;
327
- parts.unshift({ type: "text", text: formatThinkingForReplay(joined, { harmony }) });
328
- }
329
- return { role: "assistant", content: parts };
330
- }
331
-
332
- function toolResultMessage(msg: Extract<Message, { role: "toolResult" }>): ChatMessageData {
333
- const text = msg.content
334
- .filter((b): b is TextContent => b.type === "text")
335
- .map((b) => b.text)
336
- .join("\n");
337
- const result: ChatMessagePartToolCallResultData = {
338
- type: "toolCallResult",
339
- content: text,
340
- toolCallId: msg.toolCallId,
341
- };
342
- return { role: "tool", content: [result] };
343
- }
344
-
345
- async function buildChatHistory(
346
- client: Pick<LmStudioRunClient, "files">,
347
- context: Context,
348
- opts?: { harmony?: boolean; preserveThinking?: boolean },
349
- ): Promise<ChatHistoryData> {
350
- const messages: ChatMessageData[] = [];
351
- const imageCounter = { next: 0 };
352
- if (context.systemPrompt && context.systemPrompt.length > 0) {
353
- messages.push({ role: "system", content: [{ type: "text", text: context.systemPrompt }] });
354
- }
355
- for (const msg of context.messages) {
356
- if (msg.role === "user") messages.push(await userMessage(client, msg.content, imageCounter));
357
- else if (msg.role === "assistant") messages.push(assistantMessage(msg.content, opts));
358
- else if (msg.role === "toolResult") messages.push(toolResultMessage(msg));
359
- }
360
- return { messages };
361
- }
362
-
363
- function mapStopReason(
364
- reason: LLMPredictionStopReason | undefined,
365
- aborted: boolean,
366
- hadToolCall: boolean,
367
- ): AssistantMessage["stopReason"] {
368
- if (aborted || reason === "userStopped") return "aborted";
369
- if (reason === "failed" || reason === "modelUnloaded") return "error";
370
- if (reason === "toolCalls" || hadToolCall) return "toolUse";
371
- if (reason === "maxPredictedTokensReached" || reason === "contextLengthReached") return "length";
372
- return "stop";
373
- }
374
-
375
- function asDoneReason(
376
- reason: AssistantMessage["stopReason"],
377
- ): Extract<AssistantMessage["stopReason"], "stop" | "length" | "toolUse"> {
378
- if (reason === "length" || reason === "toolUse") return reason;
379
- return "stop";
380
- }
381
-
382
- interface PendingToolCall {
383
- contentIndex: number;
384
- name: string;
385
- argBuffer: string;
386
- assistantIndex: number;
387
- toolCallSlot: ToolCall;
388
- }
389
-
390
- class LmStudioToolCallExtractionError extends Error {
391
- constructor() {
392
- super(EMPTY_TOOL_ARGUMENTS_ERROR);
393
- this.name = "LmStudioToolCallExtractionError";
394
- }
395
- }
396
-
397
- function hasNonEmptyGeneratedContent(output: AssistantMessage): boolean {
398
- return output.content.some((block) => {
399
- if (block.type === "text") return block.text.trim().length > 0;
400
- if (block.type === "thinking") return block.thinking.trim().length > 0;
401
- return false;
402
- });
403
- }
404
-
405
- function hasEmptyToolArguments(value: unknown): boolean {
406
- if (value === undefined || value === null) return true;
407
- if (typeof value === "string") return value.trim().length === 0;
408
- if (value && typeof value === "object" && !Array.isArray(value)) return Object.keys(value).length === 0;
409
- return false;
410
- }
411
-
412
- const MAX_AUTOMATIC_LOAD_CONTEXT = 262_144;
413
- const MIN_AUTOMATIC_LOAD_CONTEXT = 32_768;
414
-
415
- function automaticLoadContextLength(model: Model<"lmstudio-native">): number {
416
- // When `model.contextWindow` is set (from a knowledge-base entry or an
417
- // explicit `--context-window` override on the target) it is the
418
- // authoritative budget for the load. Earlier versions also clamped against
419
- // `maxTokens * 2`, but agent workloads are dominated by *input* tokens, so
420
- // that clamp silently undersized the KV cache (e.g. 262K → 65K).
421
- if (model.contextWindow > 0) {
422
- return Math.min(model.contextWindow, MAX_AUTOMATIC_LOAD_CONTEXT);
423
- }
424
- const requestedOutput = model.maxTokens > 0 ? model.maxTokens : MIN_AUTOMATIC_LOAD_CONTEXT;
425
- const target = Math.max(MIN_AUTOMATIC_LOAD_CONTEXT, requestedOutput * 2);
426
- return Math.min(target, MAX_AUTOMATIC_LOAD_CONTEXT);
427
- }
428
-
429
- function clioQuirks(model: Model<Api>): LocalModelQuirks | undefined {
430
- return (model as Model<Api> & ClioRuntimeMetadata).clio?.quirks;
431
- }
432
-
433
- function pickSamplingProfile(
434
- quirks: LocalModelQuirks | undefined,
435
- thinkingActive: boolean,
436
- ): SamplingProfile | undefined {
437
- const sampling = quirks?.sampling;
438
- const profile = sampling ? (thinkingActive ? (sampling.thinking ?? sampling.instruct) : sampling.instruct) : undefined;
439
- return mergeSamplingOverride(profile);
440
- }
441
-
442
- function thinkingLevelFromHintOrModel(hints: RunStreamHints, model: Model<"lmstudio-native">): ThinkingLevel {
443
- if (hints.thinkingLevel) return hints.thinkingLevel;
444
- return model.reasoning === true ? "medium" : "off";
445
- }
446
-
447
- const VALID_ENV_KV_CACHE_QUANTS: ReadonlySet<string> = new Set(KV_CACHE_QUANTS);
448
-
449
- export function loadModelConfig(
450
- model: Model<"lmstudio-native">,
451
- overrides?: { contextLength?: number },
452
- ): LLMLoadModelConfig {
453
- // LM Studio's REST `/api/v1/models/load` does not expose KV cache quant or
454
- // fp16 KV options; those only round-trip through the SDK's WebSocket
455
- // protocol (LLMLoadModelConfig.llama{K,V}CacheQuantizationType,
456
- // useFp16ForKVCache). Honor catalog quirks here so dense gemma-4 NVFP4 loads
457
- // fit at f16 KV with parallel=1 and drop to q8_0 KV at parallel=4 without
458
- // the user editing settings.yaml by hand.
459
- const config: LLMLoadModelConfig = {
460
- contextLength: overrides?.contextLength ?? automaticLoadContextLength(model),
461
- flashAttention: true,
462
- gpu: { ratio: "max" },
463
- gpuStrictVramCap: true,
464
- offloadKVCacheToGpu: true,
465
- };
466
- const kvCache = clioQuirks(model)?.kvCache;
467
- if (kvCache) {
468
- if (kvCache.kQuant !== undefined && kvCache.kQuant !== false) config.llamaKCacheQuantizationType = kvCache.kQuant;
469
- if (kvCache.vQuant !== undefined && kvCache.vQuant !== false) config.llamaVCacheQuantizationType = kvCache.vQuant;
470
- if (kvCache.useFp16 !== undefined) config.useFp16ForKVCache = kvCache.useFp16;
471
- }
472
- // One-run CLI override (clio-coder run --kv-cache-mode), delivered over the
473
- // run-overrides transport; see core/run-overrides.ts.
474
- const kvCacheModeOverride = runOverrides().kvCacheMode;
475
- if (kvCacheModeOverride) {
476
- if (kvCacheModeOverride === "f16") {
477
- config.llamaKCacheQuantizationType = "f16";
478
- config.llamaVCacheQuantizationType = "f16";
479
- config.useFp16ForKVCache = true;
480
- } else if (kvCacheModeOverride === "f32") {
481
- config.llamaKCacheQuantizationType = "f32";
482
- config.llamaVCacheQuantizationType = "f32";
483
- config.useFp16ForKVCache = false;
484
- } else if (kvCacheModeOverride === "none" || kvCacheModeOverride === "false") {
485
- delete config.llamaKCacheQuantizationType;
486
- delete config.llamaVCacheQuantizationType;
487
- delete config.useFp16ForKVCache;
488
- } else {
489
- const quant = asKvCacheQuant(kvCacheModeOverride);
490
- if (quant !== undefined && quant !== false && VALID_ENV_KV_CACHE_QUANTS.has(quant)) {
491
- config.llamaKCacheQuantizationType = quant;
492
- config.llamaVCacheQuantizationType = quant;
493
- config.useFp16ForKVCache = false;
494
- } else {
495
- process.stderr.write(`clio: ignoring invalid kv-cache-mode override '${kvCacheModeOverride}'\n`);
496
- }
497
- }
498
- }
499
- return config;
500
- }
501
-
502
- interface LmStudioApiV0ModelEntry {
503
- id?: unknown;
504
- loaded_context_length?: unknown;
505
- }
506
-
507
- interface LmStudioApiV0ModelsResponse {
508
- data?: unknown;
509
- }
510
-
511
- interface LmStudioApiV1ModelEntry {
512
- key?: unknown;
513
- loaded_instances?: unknown;
514
- }
515
-
516
- interface LmStudioApiV1ModelsResponse {
517
- models?: unknown;
518
- }
519
-
520
- function positiveNumber(value: unknown): number | undefined {
521
- return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : undefined;
522
- }
523
-
524
- function isRecord(value: unknown): value is Record<string, unknown> {
525
- return value !== null && typeof value === "object" && !Array.isArray(value);
526
- }
527
-
528
- function apiV0ModelEntries(payload: LmStudioApiV0ModelsResponse | undefined): LmStudioApiV0ModelEntry[] {
529
- if (!Array.isArray(payload?.data)) return [];
530
- return payload.data.filter((entry): entry is LmStudioApiV0ModelEntry => isRecord(entry));
531
- }
532
-
533
- function apiV1ModelEntries(payload: LmStudioApiV1ModelsResponse | undefined): LmStudioApiV1ModelEntry[] {
534
- if (!Array.isArray(payload?.models)) return [];
535
- return payload.models.filter((entry): entry is LmStudioApiV1ModelEntry => isRecord(entry));
536
- }
537
-
538
- function loadedContextFromV1Instance(value: unknown): number | undefined {
539
- if (!isRecord(value) || !isRecord(value.config)) return undefined;
540
- return positiveNumber(value.config.context_length);
541
- }
542
-
543
- function loadedContextFromV1Entry(entry: LmStudioApiV1ModelEntry): number | undefined {
544
- if (!Array.isArray(entry.loaded_instances)) return undefined;
545
- for (const instance of entry.loaded_instances) {
546
- const contextLength = loadedContextFromV1Instance(instance);
547
- if (contextLength !== undefined) return contextLength;
548
- }
549
- return undefined;
550
- }
551
-
552
- async function discoverLoadedContextLength(
553
- baseUrl: string,
554
- modelId: string,
555
- signal: AbortSignal,
556
- ): Promise<number | undefined> {
557
- const base = normalizeHttpBaseUrl(baseUrl);
558
- const v1 = await discoverLoadedContextFromV1(base, modelId, signal);
559
- if (v1 !== undefined) return v1;
560
- return discoverLoadedContextFromV0(base, modelId, signal);
561
- }
562
-
563
- async function fetchLmStudioJson<T>(url: string, signal: AbortSignal): Promise<T | undefined> {
564
- const controller = new AbortController();
565
- const timer = setTimeout(() => controller.abort(), 1500);
566
- const onAbort = () => controller.abort();
567
- if (signal.aborted) controller.abort();
568
- else signal.addEventListener("abort", onAbort, { once: true });
569
- try {
570
- const response = await fetch(url, { signal: controller.signal });
571
- if (!response.ok) return undefined;
572
- return (await response.json()) as T;
573
- } catch {
574
- return undefined;
575
- } finally {
576
- clearTimeout(timer);
577
- signal.removeEventListener("abort", onAbort);
578
- }
579
- }
580
-
581
- async function discoverLoadedContextFromV1(
582
- baseUrl: string,
583
- modelId: string,
584
- signal: AbortSignal,
585
- ): Promise<number | undefined> {
586
- const payload = await fetchLmStudioJson<LmStudioApiV1ModelsResponse>(`${baseUrl}/api/v1/models`, signal);
587
- const entry = apiV1ModelEntries(payload).find((row) => row.key === modelId);
588
- return entry ? loadedContextFromV1Entry(entry) : undefined;
589
- }
590
-
591
- async function discoverLoadedContextFromV0(
592
- baseUrl: string,
593
- modelId: string,
594
- signal: AbortSignal,
595
- ): Promise<number | undefined> {
596
- const payload = await fetchLmStudioJson<LmStudioApiV0ModelsResponse>(`${baseUrl}/api/v0/models`, signal);
597
- const entry = apiV0ModelEntries(payload).find((row) => row.id === modelId);
598
- return positiveNumber(entry?.loaded_context_length);
599
- }
600
-
601
- interface ResolvedRuntimeMetadata {
602
- targetId: string;
603
- runtimeId: string;
604
- /** Explicit target lifecycle from settings; absent means Clio manages by default. */
605
- lifecycle?: RuntimeLifecycle;
606
- gateway: boolean;
607
- }
608
-
609
- function runtimeMetadata(model: Model<Api>): ResolvedRuntimeMetadata {
610
- const metadata = (model as Model<Api> & ClioRuntimeMetadata).clio;
611
- return {
612
- targetId: metadata?.targetId ?? model.provider,
613
- runtimeId: metadata?.runtimeId ?? model.provider,
614
- ...(metadata?.lifecycle ? { lifecycle: metadata.lifecycle } : {}),
615
- gateway: metadata?.gateway ?? false,
616
- };
617
- }
618
-
619
- /**
620
- * Errnos raised before the server answered anything. A load that never reached
621
- * the model cannot have been refused for its size, so the sizing advice below
622
- * is wrong in both halves for these: the cause is the route, and no
623
- * contextWindow override fixes a host that did not respond.
624
- */
625
- const CONNECT_FAILURE_RE = /\b(ENETUNREACH|EHOSTUNREACH|ECONNREFUSED|ETIMEDOUT|ENOTFOUND|ECONNRESET|EAI_AGAIN)\b/;
626
-
627
- function isConnectFailure(err: unknown, cause: string): boolean {
628
- const code = typeof err === "object" && err !== null ? (err as { code?: unknown }).code : undefined;
629
- if (typeof code === "string" && CONNECT_FAILURE_RE.test(code)) return true;
630
- return CONNECT_FAILURE_RE.test(cause);
631
- }
632
-
633
- export function describeLoadFailure(
634
- baseUrl: string,
635
- model: Model<"lmstudio-native">,
636
- loadConfig: LLMLoadModelConfig | undefined,
637
- requestedMaxTokens: number | false | undefined,
638
- err: unknown,
639
- ): string {
640
- const metadata = runtimeMetadata(model);
641
- const cause = err instanceof Error ? err.message : String(err);
642
- if (isConnectFailure(err, cause)) {
643
- return [
644
- `LM Studio at ${baseUrl} did not answer for target '${metadata.targetId}' model '${model.id}'.`,
645
- "The connection failed before a load was attempted, so this is reachability rather than model sizing.",
646
- "Check that the server is running and that this host can reach it, then retry.",
647
- `SDK error: ${cause}`,
648
- ].join(" ");
649
- }
650
- const context = loadConfig?.contextLength ?? model.contextWindow;
651
- const output = requestedMaxTokens === false || requestedMaxTokens === undefined ? model.maxTokens : requestedMaxTokens;
652
- return [
653
- `LM Studio could not load target '${metadata.targetId}' model '${model.id}' at ${baseUrl}.`,
654
- `Requested context ${context} and output ${output}.`,
655
- `Likely cause: VRAM pressure or a context length above the quantized model/server limit.`,
656
- "Try a lower contextWindow/maxTokens override, a smaller quant/tier, or openai-compat against the same LM Studio gateway when the model is already user-managed.",
657
- `SDK error: ${cause}`,
658
- ].join(" ");
659
- }
660
-
661
- /**
662
- * LM Studio's native SDK carries no thinking control. Measured against SDK
663
- * 1.5.0 and a live server on 2026-08-11: `reasoning_effort`,
664
- * `chat_template_kwargs`, and every `raw` KVConfig spelling are accepted and
665
- * ignored (a deliberately bogus key behaved identically), and prompt-level
666
- * markers such as `/no_think` change nothing. The same server's
667
- * OpenAI-compatible port honours `reasoning_effort: "none"` and suppresses
668
- * reasoning outright. Predictions therefore run over HTTP, and the SDK keeps
669
- * the work it is the only surface for: listing, loading, and unloading models.
670
- * Set CLIO_CODER_LMSTUDIO_SDK_PREDICT=1 to send predictions over the SDK again.
671
- */
672
- function sdkPredictionEnabled(): boolean {
673
- return process.env.CLIO_CODER_LMSTUDIO_SDK_PREDICT === "1";
674
- }
675
-
676
- /**
677
- * Project the model onto LM Studio's OpenAI-compatible surface. Identity,
678
- * catalog metadata, and budgets carry over verbatim, so capability resolution
679
- * still keys on runtime id `lmstudio-native` and picks LM Studio's wire
680
- * spelling for the thinking control. Only the transport changes.
681
- */
682
- function toOpenAICompletionsModel(model: Model<"lmstudio-native">): Model<"openai-completions"> {
683
- const projected = {
684
- ...model,
685
- api: "openai-completions",
686
- baseUrl: `${normalizeHttpBaseUrl(model.baseUrl)}/v1`,
687
- // Matches what every other local OpenAI-compatible target synthesizes.
688
- // Clio injects the thinking fields from its own `onPayload`, so pi-ai's
689
- // reasoning-effort path stays off here rather than writing a second,
690
- // unresolved spelling into the same body.
691
- compat: {
692
- supportsStore: false,
693
- supportsDeveloperRole: false,
694
- supportsReasoningEffort: false,
695
- supportsUsageInStreaming: true,
696
- maxTokensField: "max_tokens",
697
- supportsStrictMode: false,
698
- },
699
- } as unknown as Model<"openai-completions">;
700
- return projected;
701
- }
702
-
703
- /**
704
- * Forward one prediction over LM Studio's OpenAI-compatible port and republish
705
- * its events on this stream. Residency has already run, so the model is loaded
706
- * with the context length Clio asked for and the output budget was computed
707
- * against the window the server actually has open.
708
- */
709
- async function pumpOpenAICompletions(
710
- stream: AssistantMessageEventStream,
711
- model: Model<"lmstudio-native">,
712
- context: Context,
713
- options: StreamOptions | undefined,
714
- hints: RunStreamHints,
715
- maxTokens: number,
716
- onGeneratedChars?: (chars: number) => void,
717
- ): Promise<void> {
718
- const simple: SimpleStreamOptions = {
719
- ...(options ?? {}),
720
- maxTokens,
721
- // The SDK treats a passkey as optional and an unsecured LM Studio server
722
- // accepts any bearer, but the HTTP client refuses to build a request with
723
- // no key at all. `lm-studio` is LM Studio's own documented placeholder.
724
- apiKey: options?.apiKey ?? "lm-studio",
725
- ...(hints.thinkingLevel && hints.thinkingLevel !== "off" ? { reasoning: hints.thinkingLevel } : {}),
726
- };
727
- for await (const event of openAICompletionsApiProvider.streamSimple(
728
- toOpenAICompletionsModel(model),
729
- context,
730
- simple,
731
- )) {
732
- if (onGeneratedChars && (event.type === "text_delta" || event.type === "thinking_delta")) {
733
- onGeneratedChars(event.delta.length);
734
- }
735
- stream.push(event);
736
- }
737
- stream.end();
738
- }
739
-
740
- export function runStream(
741
- model: Model<"lmstudio-native">,
742
- context: Context,
743
- options: StreamOptions | undefined,
744
- deps: LmStudioRunDeps = defaultRunDeps,
745
- hints: RunStreamHints = {},
746
- ): AssistantMessageEventStream {
747
- const stream: AssistantMessageEventStream = createAssistantMessageEventStream();
748
- const output: AssistantMessage = {
749
- role: "assistant",
750
- content: [],
751
- api: model.api,
752
- provider: model.provider,
753
- model: model.id,
754
- // `cacheRead` is deliberately absent. LM Studio reports no cached-token
755
- // stat on either surface: the SDK's PredictionStats has no such field and
756
- // its OpenAI-compat `usage` carries no `prompt_tokens_details`. Reporting 0
757
- // reads as "measured no reuse" when the truth is "not measured", which is
758
- // how #55 mistook a working prompt cache for a broken one. Undefined until
759
- // a response actually carries the stat; consumers already treat it as
760
- // optional.
761
- usage: {
762
- input: 0,
763
- output: 0,
764
- cacheWrite: 0,
765
- totalTokens: 0,
766
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
767
- } as Usage,
768
- stopReason: "stop",
769
- timestamp: Date.now(),
770
- };
771
- const controller = new AbortController();
772
- const signal = options?.signal;
773
- let aborted = signal?.aborted === true;
774
- // `predictionDone` flips to true the moment `prediction.result()` resolves
775
- // successfully, before any post-result work. Once it is set we must not call
776
- // `controller.abort()` again: the SDK has already closed its channel cleanly,
777
- // and a late abort raises "Received channelSend for unknown channel" on the
778
- // LM Studio server. `controllerAborted` collapses repeated abort signals so
779
- // `controller.abort()` fires at most once.
780
- let predictionDone = false;
781
- let controllerAborted = false;
782
- const abortControllerOnce = () => {
783
- if (controllerAborted) return;
784
- controllerAborted = true;
785
- controller.abort();
786
- };
787
- const onAbort = () => {
788
- aborted = true;
789
- if (predictionDone) return;
790
- abortControllerOnce();
791
- };
792
- if (signal && !signal.aborted) signal.addEventListener("abort", onAbort, { once: true });
793
- else if (aborted) abortControllerOnce();
794
- // Started once the model handle is open, so a slow first load is never
795
- // mistaken for slow generation, and stopped in the `finally` below so a
796
- // failed turn leaves no timer behind.
797
- let degradedWatchdog: DegradedInferenceWatchdog | null = null;
798
- (async () => {
799
- try {
800
- if (aborted) throw new Error("Request was aborted");
801
- const baseUrl = normalizeBaseUrl(model.baseUrl);
802
- const clientOpts: ConstructorParameters<typeof LMStudioClient>[0] = { baseUrl };
803
- const passkey = options?.apiKey;
804
- if (passkey) clientOpts.clientPasskey = passkey;
805
- const client = deps.createClient(clientOpts);
806
- const metadata = runtimeMetadata(model);
807
- const verbose = process.env.CLIO_CODER_RUNTIME_VERBOSE === "1";
808
- const loadedContextWindow = await deps.discoverLoadedContext(baseUrl, model.id, controller.signal);
809
- const budgetLimits = loadedContextWindow !== undefined ? { contextWindow: loadedContextWindow } : undefined;
810
- const requestedMaxTokens = remainingContextMaxTokens(model, context, options, budgetLimits);
811
- const requestedLoadContext = loadModelConfig(model).contextLength ?? model.contextWindow;
812
- const targetKey = `lmstudio-native|${baseUrl}`;
813
- // One reconciler decides LM Studio residency for both the interactive
814
- // and headless paths. LM Studio loads just-in-time, so the plan never
815
- // evicts up front: Clio attempts the co-resident load first and swaps the
816
- // plan's ranked fallback candidates only after a load failure.
817
- // `loadedEntries` captures the SDK handles so the duplicate sweep, the
818
- // context-fit clamp, and a fallback swap all reuse one listLoaded
819
- // round-trip.
820
- let loadedEntries: ReadonlyArray<ResidentModelEntry> = [];
821
- const plan = await deps.reconcile({
822
- targetKey,
823
- targetId: metadata.targetId,
824
- runtimeId: "lmstudio-native",
825
- keepModelId: model.id,
826
- managed: residencyManagedFor(metadata.lifecycle),
827
- strategy: "jit",
828
- contextLength: requestedLoadContext,
829
- ...(model.contextWindow > 0 ? { modelMaxContext: model.contextWindow } : {}),
830
- listResident: async () => {
831
- loadedEntries = await client.llm.listLoaded();
832
- return residentModelInfos(loadedEntries);
833
- },
834
- unload: async (id) => {
835
- // Every instance of the id, so an eviction that frees a slot really
836
- // frees it when LM Studio holds the model twice.
837
- for (const entry of loadedEntries.filter((handle) => handle.modelKey === id)) await entry.unload();
838
- },
839
- });
840
- const observeOnly = plan.decision === "observe";
841
- const residencyLock = deps.lock ?? withResidencyLock;
842
- // A second instance of the model Clio is about to use is pure loss: it
843
- // holds another full weight copy and KV cache while serving the same
844
- // requests. Releasing it takes no role's model away, so it happens before
845
- // the fit question is even asked.
846
- const duplicates = observeOnly ? [] : duplicateInstances(loadedEntries, model.id);
847
- if (duplicates.length > 0) {
848
- emitResidencyNotice({
849
- kind: "co-resident",
850
- level: "warning",
851
- targetId: metadata.targetId,
852
- runtimeId: "lmstudio-native",
853
- model: model.id,
854
- message: `'${metadata.targetId}' holds ${duplicates.length + 1} instances of '${model.id}'; releasing ${duplicates.length} duplicate instance(s) so one copy serves every request.`,
855
- detail: { duplicateInstances: duplicates.length },
856
- });
857
- await residencyLock(targetKey, async () => {
858
- for (const duplicate of duplicates) {
859
- try {
860
- await duplicate.unload();
861
- } catch {
862
- // Best-effort: a duplicate that survives is reported again next turn.
863
- }
864
- }
865
- });
866
- loadedEntries = loadedEntries.filter((entry) => !duplicates.includes(entry));
867
- }
868
- // Context-length fit. LM Studio's `gpuStrictVramCap` caps GPU offload
869
- // instead of refusing an oversized load, so a KV cache that does not fit
870
- // is served from CPU at a crawl rather than failing. While another model
871
- // is resident, cap the load at the co-resident ceiling and say so before
872
- // the first turn runs.
873
- const fit = fitLoadContextLength({
874
- requested: requestedLoadContext,
875
- resident: loadedEntries,
876
- keepModelId: model.id,
877
- ceiling: coResidentContextCeiling(),
878
- });
879
- if (!observeOnly && !plan.keepResident && fit.clampedFrom !== undefined) {
880
- emitResidencyNotice({
881
- kind: "stress",
882
- level: "warning",
883
- targetId: metadata.targetId,
884
- runtimeId: "lmstudio-native",
885
- model: model.id,
886
- message: `loading '${model.id}' on '${metadata.targetId}' alongside ${fit.neighbours.join(", ")}: context clamped ${fit.clampedFrom} -> ${fit.contextLength} tokens to keep the KV cache in GPU memory. Unload the co-resident model or raise CLIO_CODER_LMSTUDIO_CORESIDENT_CONTEXT for the full window.`,
887
- detail: {
888
- requestedContext: fit.clampedFrom,
889
- loadContext: fit.contextLength,
890
- residentCount: fit.neighbours.length + 1,
891
- },
892
- });
893
- }
894
- const loadConfig = loadModelConfig(model, { contextLength: fit.contextLength });
895
- // Never hand `config` to client.llm.model for a model that is already
896
- // resident: LM Studio answers a configured open of a loaded key with a
897
- // second instance (observed as `google/gemma-4-26b-a4b-qat:2`) or with a
898
- // no-progress reload wait. A resident model is reused as loaded, and the
899
- // output budget already follows the window the server actually has open.
900
- const modelOpenConfig = observeOnly || plan.keepResident ? undefined : loadConfig;
901
- // `loadedEntries` is non-empty only when this stream actually listed the
902
- // resident set, so the reconciler's TTL fast path stays silent instead of
903
- // repeating the same line every turn.
904
- if (
905
- plan.keepResident &&
906
- loadedEntries.length > 0 &&
907
- loadedContextWindow !== undefined &&
908
- loadedContextWindow < requestedLoadContext
909
- ) {
910
- emitResidencyNotice({
911
- kind: "co-resident",
912
- level: "info",
913
- targetId: metadata.targetId,
914
- runtimeId: "lmstudio-native",
915
- model: model.id,
916
- message: `'${model.id}' is resident on '${metadata.targetId}' with a ${loadedContextWindow}-token window, below the ${requestedLoadContext} Clio would load; reusing the loaded instance and budgeting against ${loadedContextWindow}. Unload it in LM Studio to have Clio reload it larger.`,
917
- detail: { loadedContext: loadedContextWindow, requestedContext: requestedLoadContext },
918
- });
919
- }
920
- const modelOpenOpts: { signal: AbortSignal; verbose: boolean; config?: LLMLoadModelConfig } = {
921
- signal: controller.signal,
922
- verbose,
923
- };
924
- if (modelOpenConfig !== undefined) modelOpenOpts.config = modelOpenConfig;
925
- const failWillNotFit: (err: unknown) => never = (err) => {
926
- const message = describeLoadFailure(baseUrl, model, modelOpenConfig, requestedMaxTokens, err);
927
- // gpuStrictVramCap turns an oversized load into a failure; surface it
928
- // as a will-not-fit notice so it reads like every other VRAM miss.
929
- emitResidencyNotice({
930
- kind: "will-not-fit",
931
- level: "error",
932
- targetId: metadata.targetId,
933
- runtimeId: "lmstudio-native",
934
- model: model.id,
935
- message,
936
- });
937
- throw new Error(message);
938
- };
939
- let llm: LmStudioPredictionHandle;
940
- try {
941
- llm = await client.llm.model(model.id, modelOpenOpts);
942
- } catch (err) {
943
- if (observeOnly || plan.fallbackEvict.length === 0) failWillNotFit(err);
944
- // The co-resident load did not fit. Swap the ranked candidates and
945
- // retry once, serialized against other Clio processes mutating this
946
- // server; a second failure is a genuine VRAM miss.
947
- try {
948
- llm = await residencyLock(targetKey, async () => {
949
- for (const entry of plan.fallbackEvict) {
950
- emitResidencyNotice({
951
- kind: "swap",
952
- level: "warning",
953
- targetId: metadata.targetId,
954
- runtimeId: "lmstudio-native",
955
- model: model.id,
956
- message: `swapping resident '${entry.modelId}' for requested '${model.id}' on '${metadata.targetId}' after the co-resident load failed to fit.`,
957
- detail: { swappedOut: entry.modelId },
958
- });
959
- try {
960
- for (const handle of loadedEntries.filter((held) => held.modelKey === entry.modelId)) {
961
- await handle.unload();
962
- }
963
- } catch {
964
- // Best-effort: the retry load reports the real fit verdict.
965
- }
966
- }
967
- return client.llm.model(model.id, modelOpenOpts);
968
- });
969
- } catch (retryErr) {
970
- failWillNotFit(retryErr);
971
- }
972
- }
973
- // A model whose weights or KV cache spilled to CPU still answers, at a
974
- // crawl and with no error. Watch the token rate so that shows up as a
975
- // notice naming the resident set instead of an indefinite spinner.
976
- const residentList = [...new Set([model.id, ...loadedEntries.map((entry) => entry.modelKey)])].join(", ");
977
- degradedWatchdog = startDegradedInferenceWatchdog({
978
- onDegraded: (report) => {
979
- emitResidencyNotice({
980
- kind: "degraded",
981
- level: "warning",
982
- targetId: metadata.targetId,
983
- runtimeId: "lmstudio-native",
984
- model: model.id,
985
- message: `'${model.id}' on '${metadata.targetId}' has generated ${report.tokens} tokens in ${Math.round(report.elapsedMs / 1000)}s (${report.tokensPerSecond.toFixed(2)} tok/s); inference is running far below GPU speed, which is what a spill to CPU looks like. Resident there: ${residentList}. Unload a co-resident model or lower the context window.`,
986
- detail: {
987
- tokens: report.tokens,
988
- elapsedMs: report.elapsedMs,
989
- tokensPerSecond: Number(report.tokensPerSecond.toFixed(2)),
990
- residents: residentList,
991
- },
992
- });
993
- },
994
- });
995
- if (!sdkPredictionEnabled()) {
996
- // The SDK opened no prediction channel on this path, so nothing is
997
- // left for a late abort to race; the HTTP transport owns the signal
998
- // from here.
999
- predictionDone = true;
1000
- await pumpOpenAICompletions(stream, model, context, options, hints, requestedMaxTokens, (chars) =>
1001
- degradedWatchdog?.addTokens(ceilChars(chars)),
1002
- );
1003
- return;
1004
- }
1005
- stream.push({ type: "start", partial: output });
1006
- // Resolve once per request and make the resolved reasoning class
1007
- // authoritative for what the operator is shown. LM Studio may classify
1008
- // <think> tags even when the catalog says the selected family is
1009
- // reasoning-never; in that case Clio suppresses those fragments rather
1010
- // than surfacing thinking blocks. It never suppresses the token count:
1011
- // the SDK carries no chat-template-kwargs channel, so `enable_thinking`
1012
- // cannot reach this transport and a catalog `reasoning: false` is a
1013
- // belief about the model, not a control over it. The reasoning a server
1014
- // spends anyway has to stay visible in usage, or the belief cannot be
1015
- // discovered to be wrong.
1016
- const requestedThinkingLevel = thinkingLevelFromHintOrModel(hints, model);
1017
- const resolved = resolveModelRuntimeCapabilitiesForModel(model, requestedThinkingLevel);
1018
- const applied = resolved.thinking;
1019
- const suppressThinking = reasoningClassForMechanism(applied.mechanism) === "never";
1020
- // LM Studio's `result.stats.predictedTokensCount` is the total of all generated
1021
- // tokens with no separate reasoning column. Sum the per-fragment `tokensCount`
1022
- // for any fragment whose `reasoningType` belongs to a reasoning block (the
1023
- // chain-of-thought content plus the literal start/end tag tokens) so the
1024
- // receipt and TUI footer can report `reasoningTokens` truthfully even when the
1025
- // model emits a chain-of-thought via its chat template that the SDK has no API
1026
- // to disable. Per-fragment counts are approximate per the SDK docs, but the
1027
- // per-run sum tracks the actual reasoning total closely.
1028
- let reasoningTokensAccum = 0;
1029
- const activeTextRef: { block: TextContent | null; idx: number } = { block: null, idx: -1 };
1030
- const activeThinkingRef: { block: ThinkingContent | null; idx: number } = { block: null, idx: -1 };
1031
- const sentinelStripper = createSentinelStripper();
1032
- const closeActiveThinking = () => {
1033
- const current = activeThinkingRef.block;
1034
- if (!current) return;
1035
- stream.push({
1036
- type: "thinking_end",
1037
- contentIndex: activeThinkingRef.idx,
1038
- content: current.thinking,
1039
- partial: output,
1040
- });
1041
- activeThinkingRef.block = null;
1042
- activeThinkingRef.idx = -1;
1043
- };
1044
- const pushSafeText = (safe: string) => {
1045
- if (!safe) return;
1046
- closeActiveThinking();
1047
- let current = activeTextRef.block;
1048
- if (!current) {
1049
- current = { type: "text", text: "" };
1050
- output.content.push(current);
1051
- activeTextRef.block = current;
1052
- activeTextRef.idx = output.content.length - 1;
1053
- stream.push({ type: "text_start", contentIndex: activeTextRef.idx, partial: output });
1054
- }
1055
- current.text += safe;
1056
- stream.push({
1057
- type: "text_delta",
1058
- contentIndex: activeTextRef.idx,
1059
- delta: safe,
1060
- partial: output,
1061
- });
1062
- };
1063
- const flushTextSentinelBuffer = () => {
1064
- const tail = sentinelStripper.flush();
1065
- if (tail) pushSafeText(tail);
1066
- };
1067
- const emitText = (chunk: string) => {
1068
- if (!chunk) return;
1069
- const safe = sentinelStripper.push(chunk);
1070
- pushSafeText(safe);
1071
- };
1072
- const closeActiveText = () => {
1073
- // Drain any sentinel-prefix bytes the streaming stripper held
1074
- // back across the last delta. The buffered tail can never grow
1075
- // past `MAX_SENTINEL_LEN - 1` characters and only contains
1076
- // matter that turned out not to be a sentinel; emitting it now
1077
- // keeps the visible block whole without leaking sentinels.
1078
- flushTextSentinelBuffer();
1079
- const current = activeTextRef.block;
1080
- if (!current) return;
1081
- stream.push({
1082
- type: "text_end",
1083
- contentIndex: activeTextRef.idx,
1084
- content: current.text,
1085
- partial: output,
1086
- });
1087
- activeTextRef.block = null;
1088
- activeTextRef.idx = -1;
1089
- };
1090
- const emitThinking = (chunk: string, tokensHint?: number) => {
1091
- if (!chunk) return;
1092
- if (suppressThinking) return;
1093
- closeActiveText();
1094
- let current = activeThinkingRef.block;
1095
- if (!current) {
1096
- current = { type: "thinking", thinking: "" };
1097
- output.content.push(current);
1098
- activeThinkingRef.block = current;
1099
- activeThinkingRef.idx = output.content.length - 1;
1100
- stream.push({ type: "thinking_start", contentIndex: activeThinkingRef.idx, partial: output });
1101
- }
1102
- current.thinking += chunk;
1103
- // When we have an upstream token count from the SDK use it; otherwise
1104
- // approximate at 1 token per 4 chars (the same chars/4 estimator the
1105
- // openai-compat wrapper uses for streamed thinking content).
1106
- reasoningTokensAccum += tokensHint ?? ceilChars(chunk.length);
1107
- stream.push({
1108
- type: "thinking_delta",
1109
- contentIndex: activeThinkingRef.idx,
1110
- delta: chunk,
1111
- partial: output,
1112
- });
1113
- };
1114
- // Buffered state machine for gemma-4 family chat-template channel markers.
1115
- // The model emits these as plain text fragments (reasoningType = "none")
1116
- // rather than tagging them as reasoning, so without re-classification the
1117
- // chain-of-thought leaks into the visible TUI text. The thought-start
1118
- // pattern is regex-based because gemma emits channel-name variants like
1119
- // `<|channel>thought\n`, `<|channel>own-thought\n`, `<|channel>own-think\n`.
1120
- // LM Studio can also strip the `<|channel>` prefix and leave only the
1121
- // bare channel label, e.g. `ownthought\n`, before a structured tool call.
1122
- // Orphan `<channel|>` close markers (where the open was already consumed
1123
- // via the SDK's reasoning-fragment path) are dropped in idle state.
1124
- const GEMMA_THOUGHT_START_RE = /<\|channel>[^\n]*\n/;
1125
- const GEMMA_BARE_THOUGHT_START_RE = /^\s*(?:thought|own[- ]?(?:thought|think))\s*\n/i;
1126
- const GEMMA_BARE_THOUGHT_ONLY_RE = /^\s*(?:thought|own[- ]?(?:thought|think))\s*$/i;
1127
- const GEMMA_THOUGHT_END = "<channel|>";
1128
- const GEMMA_TOOLCALL_START = "<tool_call|>";
1129
- const GEMMA_TOOLCALL_END = "<|tool_call|>";
1130
- const GEMMA_BUFFER_MAX = 64;
1131
- type GemmaState = "idle" | "thought" | "toolcall";
1132
- let gemmaPending = "";
1133
- let gemmaState: GemmaState = "idle";
1134
- const responseParser = resolved.response.parser;
1135
- const harmonyParser = responseParser === "harmony" ? new HarmonyResponseParser() : null;
1136
- const flushGemmaPending = () => {
1137
- if (gemmaPending.length === 0) return;
1138
- if (gemmaState === "thought") emitThinking(gemmaPending);
1139
- else if (gemmaState === "idle" && !GEMMA_BARE_THOUGHT_ONLY_RE.test(gemmaPending)) emitText(gemmaPending);
1140
- gemmaPending = "";
1141
- };
1142
- const flushNonReasoningPending = () => {
1143
- if (harmonyParser) {
1144
- const parsed = harmonyParser.flush();
1145
- emitThinking(parsed.thinking);
1146
- emitText(parsed.text);
1147
- return;
1148
- }
1149
- flushGemmaPending();
1150
- };
1151
- const routeNonReasoningChunk = (chunk: string) => {
1152
- if (harmonyParser) {
1153
- const parsed = harmonyParser.push(chunk);
1154
- emitThinking(parsed.thinking);
1155
- emitText(parsed.text);
1156
- return;
1157
- }
1158
- gemmaPending += chunk;
1159
- while (true) {
1160
- if (gemmaState === "thought") {
1161
- const endIdx = gemmaPending.indexOf(GEMMA_THOUGHT_END);
1162
- if (endIdx === -1) {
1163
- if (gemmaPending.length > GEMMA_BUFFER_MAX) {
1164
- const safe = gemmaPending.slice(0, gemmaPending.length - GEMMA_BUFFER_MAX);
1165
- emitThinking(safe);
1166
- gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
1167
- }
1168
- return;
1169
- }
1170
- emitThinking(gemmaPending.slice(0, endIdx));
1171
- gemmaPending = gemmaPending.slice(endIdx + GEMMA_THOUGHT_END.length);
1172
- gemmaState = "idle";
1173
- } else if (gemmaState === "toolcall") {
1174
- // Discard SDK fallback text inside <tool_call|> regions; structured
1175
- // tool calls arrive via the toolcall callbacks instead.
1176
- const endIdx = gemmaPending.indexOf(GEMMA_TOOLCALL_END);
1177
- if (endIdx === -1) {
1178
- if (gemmaPending.length > GEMMA_BUFFER_MAX) {
1179
- gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
1180
- }
1181
- return;
1182
- }
1183
- gemmaPending = gemmaPending.slice(endIdx + GEMMA_TOOLCALL_END.length);
1184
- gemmaState = "idle";
1185
- } else {
1186
- const thoughtMatch = GEMMA_THOUGHT_START_RE.exec(gemmaPending);
1187
- const thoughtIdx = thoughtMatch?.index ?? -1;
1188
- const bareThoughtMatch = GEMMA_BARE_THOUGHT_START_RE.exec(gemmaPending);
1189
- const bareThoughtIdx = bareThoughtMatch ? 0 : -1;
1190
- const toolcallIdx = gemmaPending.indexOf(GEMMA_TOOLCALL_START);
1191
- // Orphan close marker: SDK consumed `<|channel>...` via a
1192
- // reasoning fragment, leaving only the standalone close in
1193
- // the plain-text stream. Drop it silently.
1194
- const orphanCloseIdx = gemmaPending.indexOf(GEMMA_THOUGHT_END);
1195
- const candidates = [
1196
- { idx: thoughtIdx, kind: "thought" as const, advance: thoughtMatch?.[0].length ?? 0 },
1197
- { idx: bareThoughtIdx, kind: "thought" as const, advance: bareThoughtMatch?.[0].length ?? 0 },
1198
- { idx: toolcallIdx, kind: "toolcall" as const, advance: GEMMA_TOOLCALL_START.length },
1199
- { idx: orphanCloseIdx, kind: "orphan" as const, advance: GEMMA_THOUGHT_END.length },
1200
- ].filter((c) => c.idx !== -1);
1201
- if (candidates.length === 0) {
1202
- if (gemmaPending.length > GEMMA_BUFFER_MAX) {
1203
- const safe = gemmaPending.slice(0, gemmaPending.length - GEMMA_BUFFER_MAX);
1204
- emitText(safe);
1205
- gemmaPending = gemmaPending.slice(gemmaPending.length - GEMMA_BUFFER_MAX);
1206
- }
1207
- return;
1208
- }
1209
- candidates.sort((a, b) => a.idx - b.idx);
1210
- const next = candidates[0];
1211
- if (!next) return;
1212
- emitText(gemmaPending.slice(0, next.idx));
1213
- gemmaPending = gemmaPending.slice(next.idx + next.advance);
1214
- if (next.kind === "thought") gemmaState = "thought";
1215
- else if (next.kind === "toolcall") gemmaState = "toolcall";
1216
- // orphan stays in idle: just drop the marker bytes
1217
- }
1218
- }
1219
- };
1220
- const pending = new Map<number, PendingToolCall>();
1221
- let toolExtractionError: LmStudioToolCallExtractionError | null = null;
1222
- const predictionOpts: LLMRespondOpts<unknown> = {
1223
- signal: controller.signal,
1224
- onPredictionFragment: (fragment) => {
1225
- if (!fragment.content) return;
1226
- degradedWatchdog?.addTokens(fragment.tokensCount ?? ceilChars(fragment.content.length));
1227
- // LM Studio SDK reasoningType values:
1228
- // "none" normal content (text)
1229
- // "reasoning" chain-of-thought inside the block
1230
- // "reasoningStartTag" literal <think> token
1231
- // "reasoningEndTag" literal </think> token
1232
- // Drop the start/end tags so they never leak into text or thinking;
1233
- // route reasoning fragments into a ThinkingContent block so the
1234
- // agent message is non-empty and pi-agent-core's loop can chain
1235
- // correctly when the model only emits reasoning + tool calls.
1236
- if (fragment.reasoningType === "reasoningStartTag" || fragment.reasoningType === "reasoningEndTag") {
1237
- reasoningTokensAccum += fragment.tokensCount ?? 0;
1238
- return;
1239
- }
1240
- if (fragment.reasoningType === "reasoning") {
1241
- flushNonReasoningPending();
1242
- // Count first, show second. Suppression is a statement about what
1243
- // the operator is shown, never about what the server spent. Gating
1244
- // the accrual on it made a reasoning-never model that reasons
1245
- // anyway report zero reasoning tokens, which is the one reading
1246
- // that would have revealed the catalog was wrong about it.
1247
- reasoningTokensAccum += fragment.tokensCount ?? 0;
1248
- emitThinking(fragment.content, 0);
1249
- return;
1250
- }
1251
- routeNonReasoningChunk(fragment.content);
1252
- },
1253
- onToolCallRequestStart: (callId) => {
1254
- flushNonReasoningPending();
1255
- gemmaState = "idle";
1256
- closeActiveText();
1257
- closeActiveThinking();
1258
- const slot: ToolCall = { type: "toolCall", id: randomUUID(), name: "", arguments: {} };
1259
- output.content.push(slot);
1260
- const idx = output.content.length - 1;
1261
- stream.push({ type: "toolcall_start", contentIndex: idx, partial: output });
1262
- pending.set(callId, {
1263
- contentIndex: idx,
1264
- name: "",
1265
- argBuffer: "",
1266
- assistantIndex: idx,
1267
- toolCallSlot: slot,
1268
- });
1269
- },
1270
- onToolCallRequestNameReceived: (callId, name) => {
1271
- const entry = pending.get(callId);
1272
- if (!entry) return;
1273
- entry.name = name;
1274
- entry.toolCallSlot.name = name;
1275
- },
1276
- onToolCallRequestArgumentFragmentGenerated: (callId, fragment) => {
1277
- const entry = pending.get(callId);
1278
- if (!entry) return;
1279
- entry.argBuffer += fragment;
1280
- entry.toolCallSlot.arguments = parseStreamingArgs(entry.argBuffer);
1281
- stream.push({
1282
- type: "toolcall_delta",
1283
- contentIndex: entry.contentIndex,
1284
- delta: fragment,
1285
- partial: output,
1286
- });
1287
- },
1288
- onToolCallRequestEnd: (callId, info) => {
1289
- const entry = pending.get(callId);
1290
- if (!entry) return;
1291
- const req = info.toolCallRequest;
1292
- entry.toolCallSlot.name = req.name || entry.name || "";
1293
- if (hasEmptyToolArguments(req.arguments) && entry.argBuffer.length === 0 && hasNonEmptyGeneratedContent(output)) {
1294
- toolExtractionError = new LmStudioToolCallExtractionError();
1295
- pending.delete(callId);
1296
- return;
1297
- }
1298
- entry.toolCallSlot.arguments =
1299
- req.arguments && typeof req.arguments === "object"
1300
- ? (req.arguments as Record<string, unknown>)
1301
- : parseFinalArgs(entry.argBuffer);
1302
- if (req.id) entry.toolCallSlot.id = req.id;
1303
- stream.push({
1304
- type: "toolcall_end",
1305
- contentIndex: entry.contentIndex,
1306
- toolCall: entry.toolCallSlot,
1307
- partial: output,
1308
- });
1309
- pending.delete(callId);
1310
- },
1311
- };
1312
- if (context.tools && context.tools.length > 0) {
1313
- predictionOpts.rawTools = {
1314
- type: "toolArray",
1315
- tools: context.tools.map(toolToLmStudio),
1316
- };
1317
- }
1318
- predictionOpts.maxTokens = requestedMaxTokens;
1319
- // Apply catalog sampling quirks first; explicit StreamOptions overrides
1320
- // (set on `options`) win where they are present. The catalog profile is
1321
- // chosen by thinking activity, derived through the central resolver so
1322
- // the sampler choice matches the actual surface the model exposes
1323
- // (effort-levels, budget-tokens, on-off, always-on, none). The bare
1324
- // `stream` path leaves `hints.thinkingLevel` unset and falls back to
1325
- // medium when the model advertises reasoning.
1326
- // The LM Studio SDK has no separate thinking-budget channel; the budget
1327
- // from `applied.budgetTokens` is informational only here and surfaces
1328
- // through the prompt Runtime block. `maxPredictedTokens` stays driven
1329
- // by the remaining-context budget so a budget-tokens family does not
1330
- // unexpectedly truncate output.
1331
- const samplingProfile = pickSamplingProfile(resolved.quirks ?? clioQuirks(model), applied.thinkingActive);
1332
- if (samplingProfile) {
1333
- if (samplingProfile.temperature !== undefined) predictionOpts.temperature = samplingProfile.temperature;
1334
- if (samplingProfile.topP !== undefined) predictionOpts.topPSampling = samplingProfile.topP;
1335
- if (samplingProfile.topK !== undefined) predictionOpts.topKSampling = samplingProfile.topK;
1336
- if (samplingProfile.minP !== undefined) predictionOpts.minPSampling = samplingProfile.minP;
1337
- if (samplingProfile.repeatPenalty !== undefined) predictionOpts.repeatPenalty = samplingProfile.repeatPenalty;
1338
- }
1339
- if (options?.temperature !== undefined) predictionOpts.temperature = options.temperature;
1340
- const harmony = resolved.response.parser === "harmony";
1341
- const history = await buildChatHistory(client, context, { harmony, preserveThinking: !suppressThinking });
1342
- if (aborted) throw new Error("Request was aborted");
1343
- const prediction = llm.respond(history, predictionOpts);
1344
- const result = await prediction.result();
1345
- // The SDK has now closed its prediction channel cleanly. Block any
1346
- // future `onAbort` from racing a second `controller.abort()` against
1347
- // that closed channel; the post-result `if (aborted) throw` below
1348
- // still surfaces a late user-driven abort to the caller.
1349
- predictionDone = true;
1350
- flushNonReasoningPending();
1351
- closeActiveText();
1352
- closeActiveThinking();
1353
- // Write usage before any throw so the error path (tool-extraction failure,
1354
- // post-result aborts) still surfaces real token counts to dispatch and the TUI.
1355
- // Probed off the raw stats rather than PredictionStatsLike: no LM Studio
1356
- // build ships a cached-token count today, so the field only exists for
1357
- // the runtime that starts sending one. Anything else stays undefined.
1358
- const reportedCache = asRecord(result.stats).cachedTokensCount;
1359
- const cacheRead = typeof reportedCache === "number" && reportedCache >= 0 ? reportedCache : undefined;
1360
- output.usage.input = Math.max(0, (result.stats.promptTokensCount ?? 0) - (cacheRead ?? 0));
1361
- output.usage.output = result.stats.predictedTokensCount ?? 0;
1362
- output.usage.cacheRead = cacheRead as number;
1363
- output.usage.totalTokens = result.stats.totalTokensCount ?? output.usage.input + output.usage.output;
1364
- if (reasoningTokensAccum > 0) {
1365
- (output.usage as Usage & { reasoningTokens?: number }).reasoningTokens = reasoningTokensAccum;
1366
- }
1367
- // Costed against a numeric view: pi's cost math multiplies `cacheRead`
1368
- // directly, so handing it undefined would make every total NaN.
1369
- output.usage.cost = calculateEngineCost(model, { ...output.usage, cacheRead: cacheRead ?? 0 });
1370
- if (aborted) throw new Error("Request was aborted");
1371
- if (toolExtractionError) throw toolExtractionError;
1372
- const hadToolCall = output.content.some((block) => block.type === "toolCall");
1373
- output.stopReason = mapStopReason(result.stats.stopReason, aborted, hadToolCall);
1374
- if (output.stopReason === "error" || output.stopReason === "aborted") {
1375
- output.errorMessage = `prediction stopped: ${result.stats.stopReason ?? "unknown"}`;
1376
- stream.push({ type: "error", reason: output.stopReason, error: output });
1377
- } else {
1378
- stream.push({ type: "done", reason: asDoneReason(output.stopReason), message: output });
1379
- }
1380
- stream.end();
1381
- } catch (err) {
1382
- output.stopReason = aborted ? "aborted" : "error";
1383
- output.errorMessage = err instanceof Error ? err.message : String(err);
1384
- stream.push({ type: "error", reason: output.stopReason, error: output });
1385
- stream.end();
1386
- } finally {
1387
- degradedWatchdog?.stop();
1388
- if (signal) signal.removeEventListener("abort", onAbort);
1389
- }
1390
- })();
1391
- return stream;
1392
- }
1393
-
1394
- function asRecord(value: unknown): Record<string, unknown> {
1395
- return value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
1396
- }
1397
-
1398
- function parseStreamingArgs(raw: string): Record<string, unknown> {
1399
- if (!raw) return {};
1400
- try {
1401
- return asRecord(parseEngineStreamingJson<unknown>(raw));
1402
- } catch {
1403
- return {};
1404
- }
1405
- }
1406
-
1407
- function parseFinalArgs(raw: string): Record<string, unknown> {
1408
- if (!raw) return {};
1409
- try {
1410
- return asRecord(parseEngineJsonWithRepair<unknown>(raw));
1411
- } catch {
1412
- return parseStreamingArgs(raw);
1413
- }
1414
- }
1415
-
1416
- function stripReasoning(options: SimpleStreamOptions | undefined): StreamOptions | undefined {
1417
- if (!options) return undefined;
1418
- const { reasoning: _r, thinkingBudgets: _b, ...rest } = options;
1419
- return rest;
1420
- }
1421
-
1422
- // pi-ai's SimpleStreamOptions.reasoning is the ThinkingLevel for this turn,
1423
- // or undefined when thinking is off. The bare `stream` path cannot reach the
1424
- // level so runStream falls back to the model's `reasoning` capability flag.
1425
- function thinkingLevelFromSimple(options: SimpleStreamOptions | undefined): ThinkingLevel {
1426
- const reasoning = options?.reasoning;
1427
- if (reasoning === undefined) return "off";
1428
- return reasoning as ThinkingLevel;
1429
- }
1430
-
1431
- export const lmstudioNativeApiProvider: ApiProvider<"lmstudio-native"> = {
1432
- api: "lmstudio-native",
1433
- stream: (model, context, options) => runStream(model, context, options),
1434
- streamSimple: (model, context, options?: SimpleStreamOptions) =>
1435
- runStream(model, context, stripReasoning(options), defaultRunDeps, {
1436
- thinkingLevel: thinkingLevelFromSimple(options),
1437
- }),
1438
- };