@iowarp/clio-coder 0.3.8 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (552) hide show
  1. package/CHANGELOG.md +102 -0
  2. package/NOTICE +33 -0
  3. package/README.md +12 -3
  4. package/dist/{acp-U67UHUK2.js → acp-G5WJBNCT.js} +13 -12
  5. package/dist/{agents-YU6SGALZ.js → agents-FMV2Q5G4.js} +41 -36
  6. package/dist/assets/codewiki.json +1 -1
  7. package/dist/{auth-5ZPJOIVG.js → auth-3IDSJEIK.js} +18 -18
  8. package/dist/{builtins-C6JMZVV6.js → builtins-XCZWXSC7.js} +5 -5
  9. package/dist/{chunk-IFBNV6H6.js → chunk-2ANTL7MR.js} +3 -3
  10. package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
  11. package/dist/{chunk-XWSF374K.js → chunk-2OQE55CK.js} +3 -3
  12. package/dist/{chunk-5H3GB5BO.js → chunk-32KWKNSF.js} +8 -384
  13. package/dist/{chunk-KTYTFRMB.js → chunk-36EJLSQQ.js} +34 -36
  14. package/dist/{chunk-FHJEP5SW.js → chunk-3BT2XMV4.js} +19 -13
  15. package/dist/chunk-3DPEIQKN.js +113 -0
  16. package/dist/{chunk-GU2UIAFZ.js → chunk-3URVFKWK.js} +7 -7
  17. package/dist/{chunk-WEH5XRJQ.js → chunk-3XML7CDN.js} +3 -3
  18. package/dist/chunk-42FMPA75.js +101 -0
  19. package/dist/{chunk-A2NJGIB3.js → chunk-5LXZXPKX.js} +2 -2
  20. package/dist/{chunk-VCBR6CU7.js → chunk-5PSMVOLM.js} +2 -2
  21. package/dist/{chunk-NMPKI6XL.js → chunk-6DB53AJS.js} +203 -25
  22. package/dist/{chunk-3BINW3FP.js → chunk-76ONBSIA.js} +2 -2
  23. package/dist/{chunk-26LEYJZH.js → chunk-7MCTRUCE.js} +2 -2
  24. package/dist/{chunk-TYPGUK6W.js → chunk-AB44T6BB.js} +111 -5
  25. package/dist/{chunk-K4XHGFR5.js → chunk-B7OBL7PK.js} +317 -721
  26. package/dist/{chunk-B5CSFE7B.js → chunk-BBVJUZHB.js} +2 -2
  27. package/dist/{chunk-EQ63NRB7.js → chunk-BBVYXMFO.js} +2 -2
  28. package/dist/{chunk-ZVJ5BLO2.js → chunk-BKFJHQCA.js} +154 -16
  29. package/dist/chunk-BKFM6EJV.js +462 -0
  30. package/dist/chunk-BMS5RKQY.js +27 -0
  31. package/dist/{chunk-ZNLWCMVZ.js → chunk-BPKCPIL7.js} +2 -2
  32. package/dist/{chunk-TLQJPP24.js → chunk-BUMFYQFY.js} +1394 -1349
  33. package/dist/{chunk-BNAZZHFG.js → chunk-BYP5D4HI.js} +1 -1
  34. package/dist/chunk-C2LTL2W6.js +2447 -0
  35. package/dist/{chunk-ME6CCNFO.js → chunk-CODPRO7Q.js} +8 -8
  36. package/dist/{chunk-E77JEWSD.js → chunk-CTJ4RNAA.js} +7 -37
  37. package/dist/{chunk-ODFEOB4F.js → chunk-CY6FY24N.js} +26 -8
  38. package/dist/{chunk-7RFXX52T.js → chunk-DQITNCXG.js} +642 -172
  39. package/dist/{chunk-TTHACPOM.js → chunk-DYJP44XW.js} +578 -117
  40. package/dist/{chunk-2HFZQUHL.js → chunk-F4CKPOEQ.js} +18 -8
  41. package/dist/{chunk-DGSYXYMX.js → chunk-FEFIFZTL.js} +3 -3
  42. package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
  43. package/dist/{chunk-MV3K5QF2.js → chunk-GCSMB2KY.js} +2 -2
  44. package/dist/chunk-GKF55TAZ.js +391 -0
  45. package/dist/{chunk-VWZOAB7K.js → chunk-GR5G2PVF.js} +9 -8
  46. package/dist/{chunk-4DGYLA73.js → chunk-GXNLGKAB.js} +80 -9
  47. package/dist/chunk-HHV2GANA.js +88 -0
  48. package/dist/{chunk-IIZWH4XA.js → chunk-HI63TFOG.js} +5 -4
  49. package/dist/{chunk-WLFILSD5.js → chunk-HJJTYUHX.js} +113 -83
  50. package/dist/{chunk-TT36MB5S.js → chunk-HLW2MRKE.js} +3 -1
  51. package/dist/{chunk-PMDBGQSJ.js → chunk-HWHKMHUA.js} +7 -7
  52. package/dist/chunk-HZHHCK24.js +1631 -0
  53. package/dist/{chunk-WSB3FPX7.js → chunk-I5VEOC6I.js} +39 -143
  54. package/dist/chunk-IBEBSCYA.js +564 -0
  55. package/dist/chunk-IQ7KR472.js +362 -0
  56. package/dist/{chunk-A3WNZD3P.js → chunk-J4W7KFM7.js} +949 -972
  57. package/dist/{chunk-TB5666IT.js → chunk-JDG2WCRO.js} +5 -5
  58. package/dist/{chunk-N22QMJKY.js → chunk-K5C3NCBD.js} +4 -4
  59. package/dist/{chunk-CGKSTWHD.js → chunk-K6BSR66V.js} +2 -1
  60. package/dist/{chunk-5C3AQNDW.js → chunk-KFZI4NIL.js} +216 -38
  61. package/dist/chunk-KMVISBZR.js +132 -0
  62. package/dist/{chunk-WXY7KU3G.js → chunk-LDQ2ZF2M.js} +2 -2
  63. package/dist/{chunk-XN3L4EYL.js → chunk-LQ3DZAMX.js} +3 -3
  64. package/dist/{chunk-U6MBIEMB.js → chunk-LY4S7GJC.js} +173 -144
  65. package/dist/{chunk-4SPRNWDE.js → chunk-MLKNTWH2.js} +19 -19
  66. package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
  67. package/dist/chunk-NEKRRTYW.js +56 -0
  68. package/dist/{chunk-SPULKLCF.js → chunk-NHCZP4K7.js} +3 -3
  69. package/dist/chunk-NHLBIGRH.js +1506 -0
  70. package/dist/chunk-NQQH3YT7.js +302 -0
  71. package/dist/chunk-NYS75XW5.js +15 -0
  72. package/dist/{chunk-GN57SG4G.js → chunk-O4XIVISU.js} +10 -8
  73. package/dist/{chunk-EMYUUSFG.js → chunk-O6TL7WWY.js} +6 -6
  74. package/dist/chunk-OQBA45DZ.js +97 -0
  75. package/dist/{chunk-LU7P4LHA.js → chunk-P3FOHJT4.js} +2 -2
  76. package/dist/chunk-PMZCIOCJ.js +25 -0
  77. package/dist/{chunk-I4HZDVNP.js → chunk-PQEFIJ36.js} +2 -2
  78. package/dist/{chunk-J3YUBZWY.js → chunk-QBJA7R7N.js} +62 -6
  79. package/dist/chunk-QDC3K2U3.js +262 -0
  80. package/dist/{chunk-2HEJ2F35.js → chunk-QLFS5GO2.js} +22 -10
  81. package/dist/{chunk-7RGZWPB6.js → chunk-QLL7ILRG.js} +95 -32
  82. package/dist/{chunk-YS5VLNH5.js → chunk-QREDIESB.js} +6 -6
  83. package/dist/chunk-QSNYB6ZV.js +195 -0
  84. package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
  85. package/dist/chunk-RKKLTLYB.js +45 -0
  86. package/dist/{chunk-P43ETTHK.js → chunk-SJ5ZKQ4S.js} +2 -2
  87. package/dist/{chunk-GPIEI3LY.js → chunk-SP2RXXYO.js} +6 -54
  88. package/dist/chunk-SUCTJL45.js +45 -0
  89. package/dist/{chunk-DYIM5TJT.js → chunk-SUW5DORT.js} +263 -7
  90. package/dist/chunk-T56WDKA5.js +183 -0
  91. package/dist/chunk-TVHHYFHE.js +255 -0
  92. package/dist/{chunk-FJ3H4MN5.js → chunk-TZ3SGWZZ.js} +3 -3
  93. package/dist/{chunk-MXKJU4JB.js → chunk-U77AMWDL.js} +91 -10
  94. package/dist/{chunk-5DHKRSMQ.js → chunk-ULC6OTWO.js} +11 -7
  95. package/dist/{chunk-RWSI4YD7.js → chunk-UM7N4G5A.js} +33 -12
  96. package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
  97. package/dist/{chunk-IGWKHNIQ.js → chunk-UXMFQ54G.js} +44 -37
  98. package/dist/{chunk-HFSBBKSQ.js → chunk-V5DHCITQ.js} +171 -3
  99. package/dist/{chunk-5WIGXA4T.js → chunk-VAZSBTKF.js} +111 -4
  100. package/dist/{chunk-VHN4MY6O.js → chunk-VEO4AP2K.js} +2 -2
  101. package/dist/{chunk-IJ7RPIYJ.js → chunk-VFA6GDY5.js} +65 -4
  102. package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
  103. package/dist/chunk-VO67MWHC.js +75 -0
  104. package/dist/{chunk-XK56QHLX.js → chunk-VPKWYKEY.js} +19 -5
  105. package/dist/{chunk-TANS5ZJS.js → chunk-VYMXRQI6.js} +36 -22
  106. package/dist/{chunk-WWCZ5F23.js → chunk-W5VSYASO.js} +77 -16
  107. package/dist/{chunk-WNIJTQQK.js → chunk-WZR7K7ZX.js} +72 -116
  108. package/dist/{chunk-5Q2VVUKB.js → chunk-X3YGUTOB.js} +4 -4
  109. package/dist/chunk-X75E3D2N.js +686 -0
  110. package/dist/{chunk-VAWNZU7Z.js → chunk-YDFRH54B.js} +4 -4
  111. package/dist/chunk-YJX4SHTD.js +40 -0
  112. package/dist/{chunk-ZI647VB5.js → chunk-YPI3QQCF.js} +2 -2
  113. package/dist/chunk-Z2RR6MAK.js +127 -0
  114. package/dist/{chunk-FBVTI2TJ.js → chunk-Z4TXYIEG.js} +12 -131
  115. package/dist/cli/index.js +47 -36
  116. package/dist/{clio-QVTYJ57A.js → clio-2JXHBBY5.js} +7 -7
  117. package/dist/{code-nav-FGGFIE7L.js → code-nav-3YYRMYNF.js} +8 -8
  118. package/dist/{compile-cache-CVJMMODC.js → compile-cache-7FPE6PS3.js} +3 -3
  119. package/dist/{components-ZFA3SAER.js → components-RYZV4JGP.js} +5 -5
  120. package/dist/{config-LW5IJFQN.js → config-QZPCMYSO.js} +99 -63
  121. package/dist/{configure-7XIZCOU4.js → configure-TEGEBYCA.js} +23 -22
  122. package/dist/{context-Y6Y7QPR6.js → context-AV7OEZ4D.js} +12 -12
  123. package/dist/{context-L3WL3X7K.js → context-E6H5RNMC.js} +56 -47
  124. package/dist/{context-N52ZA626.js → context-GSXUE4CT.js} +29 -27
  125. package/dist/{context-clear-MBQRLSDQ.js → context-clear-SHIBYK6T.js} +55 -46
  126. package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
  127. package/dist/{context-working-set-GS6DSO7F.js → context-working-set-5ZGKPGZQ.js} +13 -13
  128. package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-EFMJT4LD.js} +93 -64
  129. package/dist/{docs-7LQ23DLM.js → docs-23KQS3XK.js} +5 -5
  130. package/dist/doctor-QOA5FNY5.js +313 -0
  131. package/dist/{eval-BEC2WHDA.js → eval-TFBYQH4H.js} +2032 -156
  132. package/dist/eval-inventory-SXH7PDKX.js +316 -0
  133. package/dist/{evidence-REJUMSKM.js → evidence-ERGESKGN.js} +203 -48
  134. package/dist/{evolve-PY5ZBA5K.js → evolve-VDXTSYCJ.js} +52 -43
  135. package/dist/{extensions-HVKU65YU.js → extensions-7BGBHN57.js} +13 -7
  136. package/dist/{fleet-7WZEWRFA.js → fleet-2RRVDF2V.js} +228 -108
  137. package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-VJ726XIA.js} +11 -11
  138. package/dist/fleet-decisions-EPAPM3XJ.js +157 -0
  139. package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-JF5QOATM.js} +18 -15
  140. package/dist/fleet-inspect-VLY4S7QM.js +442 -0
  141. package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-AIZUEJOY.js} +5 -5
  142. package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-AJRPDMDV.js} +22 -19
  143. package/dist/fleet-verify-JFEL2L3H.js +175 -0
  144. package/dist/fleet-view-ZCON35AG.js +102 -0
  145. package/dist/{init-OG3TPGQG.js → init-DN2WWLFE.js} +72 -62
  146. package/dist/install-XGLBQY5E.js +13 -0
  147. package/dist/interop-OZBKXAYL.js +114 -0
  148. package/dist/{library-CNTMPLRF.js → library-YWZG7IMW.js} +21 -18
  149. package/dist/{memory-6IS7F275.js → memory-I4C4HMLW.js} +54 -45
  150. package/dist/{models-ENRJDA5W.js → models-CEYXJBO6.js} +33 -30
  151. package/dist/{monitor-XLDVO7TN.js → monitor-NZ6GCI3P.js} +59 -52
  152. package/dist/{orchestrator-6KSPYRHA.js → orchestrator-GCGQ4N5I.js} +7827 -7328
  153. package/dist/panes-HMABYVO4.js +58 -0
  154. package/dist/panes-KY6W3V2E.js +103 -0
  155. package/dist/{paths-DBXMZMDU.js → paths-II4K7DNR.js} +5 -5
  156. package/dist/{reset-RZ4ER727.js → reset-DQ6FGCSH.js} +13 -11
  157. package/dist/resources-BB3MVJMD.js +111 -0
  158. package/dist/{run-Y2CNK5RU.js → run-H2GQDUER.js} +119 -87
  159. package/dist/{share-A55GYP6Z.js → share-GTJN6A5O.js} +20 -17
  160. package/dist/{skills-ALC5J6AT.js → skills-L55TEW6R.js} +33 -24
  161. package/dist/{skills-eval-JPBEBYQU.js → skills-eval-XVXPH2JI.js} +67 -56
  162. package/dist/skills-inventory-S4MXPJFV.js +126 -0
  163. package/dist/slash-commands-ZSGASKJC.js +77 -0
  164. package/dist/{steer-GGWFUJUD.js → steer-RZGSCY4R.js} +4 -4
  165. package/dist/{support-MIETYA5E.js → support-PKEUNNQL.js} +6 -6
  166. package/dist/{targets-VGNXIR3S.js → targets-NCPZ644J.js} +68 -40
  167. package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-44SV3YCN.js} +6 -4
  168. package/dist/tools-DAF3DI3C.js +27 -0
  169. package/dist/{trace-PNCASAXC.js → trace-FYVW2MQA.js} +207 -10
  170. package/dist/tui-primitives-2AKXQNZK.js +13 -0
  171. package/dist/{uninstall-ZJF5H5ZN.js → uninstall-DW2PNOIC.js} +5 -5
  172. package/dist/{upgrade-FUSUAGHR.js → upgrade-3XPP6OQL.js} +27 -24
  173. package/dist/{usage-N4MKVHKD.js → usage-3NLHGTU2.js} +114 -62
  174. package/dist/{verifiers-YAWOJ3H2.js → verifiers-SSQONKRT.js} +172 -13
  175. package/dist/{verify-LTDHYBGY.js → verify-3U6J7FZI.js} +10 -10
  176. package/dist/{web-fetch-2YHJ3KTG.js → web-fetch-S7RR6GZ7.js} +3 -3
  177. package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-CEHYGPGQ.js} +75 -65
  178. package/dist/with-panes-MKB46MPQ.js +782 -0
  179. package/dist/worker/entry.js +106 -75
  180. package/docs/README.md +3 -2
  181. package/docs/acp.md +24 -3
  182. package/docs/alcf-provider.md +1 -1
  183. package/docs/architecture.md +2 -2
  184. package/docs/artifact-versions.md +9 -1
  185. package/docs/built-in-agents.md +1 -1
  186. package/docs/capacity-and-scheduling.md +62 -3
  187. package/docs/commands-and-modes.md +31 -2
  188. package/docs/configuration-and-targets.md +69 -12
  189. package/docs/context-engine.md +63 -4
  190. package/docs/development-pipeline.md +19 -0
  191. package/docs/dispatch-typed-intent.md +385 -0
  192. package/docs/documentation-coverage.md +5 -5
  193. package/docs/documentation-guide.md +1 -1
  194. package/docs/environment-variables.md +3 -0
  195. package/docs/eval-runner.md +262 -11
  196. package/docs/evals-internal.md +72 -2
  197. package/docs/evidence-and-memory.md +12 -11
  198. package/docs/evolution.md +1 -1
  199. package/docs/exit-codes-and-output.md +1 -1
  200. package/docs/extensions-and-sharing.md +27 -1
  201. package/docs/fleet-dispatch.md +25 -4
  202. package/docs/installation-and-lifecycle.md +15 -2
  203. package/docs/middleware-and-components.md +1 -1
  204. package/docs/model-catalog.md +10 -1
  205. package/docs/observability.md +55 -4
  206. package/docs/proactive-memory.md +127 -14
  207. package/docs/prompt-envelope-and-tools.md +19 -1
  208. package/docs/provider-adapter-cookbook.md +1 -1
  209. package/docs/release-cut-checklist.md +19 -3
  210. package/docs/safety-model.md +2 -2
  211. package/docs/scientific-validation.md +3 -3
  212. package/docs/session-lifecycle.md +1 -1
  213. package/docs/skills-marketplace.md +1 -1
  214. package/docs/tool-usage.md +18 -8
  215. package/docs/trace-store.md +1 -1
  216. package/docs/troubleshooting.md +88 -1
  217. package/docs/tui-design.md +1 -1
  218. package/docs/worker-dispatch-mechanics.md +1 -1
  219. package/package.json +5 -2
  220. package/src/cli/acp.ts +6 -2
  221. package/src/cli/agents.ts +1 -1
  222. package/src/cli/argv.ts +25 -0
  223. package/src/cli/config-inspect.ts +33 -6
  224. package/src/cli/config.ts +1 -1
  225. package/src/cli/configure.ts +23 -21
  226. package/src/cli/doctor-panes.ts +124 -0
  227. package/src/cli/doctor-state-size.ts +82 -0
  228. package/src/cli/doctor-toolchain.ts +57 -0
  229. package/src/cli/doctor.ts +22 -1
  230. package/src/cli/eval-inventory.ts +436 -0
  231. package/src/cli/eval.ts +93 -16
  232. package/src/cli/evidence-detail.ts +88 -0
  233. package/src/cli/evidence-inventory.ts +183 -0
  234. package/src/cli/evidence.ts +30 -5
  235. package/src/cli/extensions.ts +5 -1
  236. package/src/cli/fleet-decisions.ts +69 -0
  237. package/src/cli/fleet-inspect.ts +334 -0
  238. package/src/cli/fleet-verify.ts +133 -0
  239. package/src/cli/fleet-view.ts +810 -0
  240. package/src/cli/fleet.ts +179 -39
  241. package/src/cli/index.ts +14 -2
  242. package/src/cli/interop-inspect.ts +128 -0
  243. package/src/cli/interop.ts +34 -0
  244. package/src/cli/panes.ts +35 -0
  245. package/src/cli/reset.ts +5 -2
  246. package/src/cli/run.ts +58 -0
  247. package/src/cli/skills-inventory.ts +185 -0
  248. package/src/cli/skills.ts +16 -13
  249. package/src/cli/targets.ts +44 -13
  250. package/src/cli/tools.ts +321 -0
  251. package/src/cli/trace-inspect.ts +252 -0
  252. package/src/cli/trace.ts +85 -5
  253. package/src/cli/usage.ts +63 -14
  254. package/src/cli/verifiers-inspect.ts +347 -0
  255. package/src/cli/verifiers.ts +9 -0
  256. package/src/core/bus-events.ts +33 -1
  257. package/src/core/cache-telemetry.ts +42 -0
  258. package/src/core/config.ts +67 -0
  259. package/src/core/defaults.ts +124 -8
  260. package/src/core/endpoint-key.ts +27 -0
  261. package/src/core/residency-target-key.ts +25 -0
  262. package/src/core/response-schema.ts +80 -6
  263. package/src/core/theme-token-hex.ts +43 -0
  264. package/src/core/tool-names.ts +2 -1
  265. package/src/core/xdg.ts +1 -1
  266. package/src/domains/agents/fleets/build-review.md +0 -3
  267. package/src/domains/agents/fleets/build-test.md +0 -3
  268. package/src/domains/agents/result-contract-filesystem.ts +32 -0
  269. package/src/domains/agents/result-contract.ts +164 -35
  270. package/src/domains/config/classify.ts +6 -0
  271. package/src/domains/context/codewiki/coordinator.ts +12 -4
  272. package/src/domains/dispatch/admission-error.ts +9 -0
  273. package/src/domains/dispatch/admission.ts +52 -14
  274. package/src/domains/dispatch/capacity-lease.ts +118 -9
  275. package/src/domains/dispatch/contract.ts +11 -0
  276. package/src/domains/dispatch/council-topology.ts +398 -0
  277. package/src/domains/dispatch/execution-plan.ts +44 -4
  278. package/src/domains/dispatch/extension.ts +378 -82
  279. package/src/domains/dispatch/fleet-node-prompt.ts +62 -0
  280. package/src/domains/dispatch/fleet-plan.ts +7 -2
  281. package/src/domains/dispatch/fleet-run.ts +87 -4
  282. package/src/domains/dispatch/gate-decisions.ts +11 -1
  283. package/src/domains/dispatch/gate-role-prompts.ts +9 -0
  284. package/src/domains/dispatch/gate-topology.ts +289 -0
  285. package/src/domains/dispatch/heartbeat.ts +32 -8
  286. package/src/domains/dispatch/index.ts +22 -0
  287. package/src/domains/dispatch/intent-compatibility.ts +330 -0
  288. package/src/domains/dispatch/intent.ts +85 -1
  289. package/src/domains/dispatch/orphan-recovery.ts +5 -0
  290. package/src/domains/dispatch/reservation-store.ts +139 -11
  291. package/src/domains/dispatch/run-event-journal-bridge.ts +149 -0
  292. package/src/domains/dispatch/run-event-journal.ts +598 -0
  293. package/src/domains/dispatch/state.ts +49 -1
  294. package/src/domains/dispatch/types.ts +13 -0
  295. package/src/domains/dispatch/validation.ts +33 -8
  296. package/src/domains/dispatch/worker-spawn.ts +25 -11
  297. package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
  298. package/src/domains/dispatch/write-boundary.ts +62 -1
  299. package/src/domains/eval/artifacts/store.ts +62 -0
  300. package/src/domains/eval/compare/behavioral.ts +224 -0
  301. package/src/domains/eval/compare/compare.ts +342 -2
  302. package/src/domains/eval/compare/envelope.ts +128 -0
  303. package/src/domains/eval/compare/gates.ts +24 -6
  304. package/src/domains/eval/compare/thresholds.ts +30 -3
  305. package/src/domains/eval/execution-provenance.ts +240 -0
  306. package/src/domains/eval/inventory.ts +113 -0
  307. package/src/domains/eval/metrics/aggregate.ts +136 -0
  308. package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
  309. package/src/domains/eval/metrics/tracked.ts +413 -0
  310. package/src/domains/eval/provenance.ts +117 -0
  311. package/src/domains/eval/reports/comparison.ts +128 -0
  312. package/src/domains/eval/reports/junit.ts +17 -3
  313. package/src/domains/eval/reports/markdown.ts +3 -3
  314. package/src/domains/eval/reports/text.ts +14 -0
  315. package/src/domains/eval/run-compare.ts +20 -0
  316. package/src/domains/eval/runners/clio-run.ts +127 -0
  317. package/src/domains/eval/runners/external-command.ts +28 -3
  318. package/src/domains/eval/schema/adapter.ts +111 -0
  319. package/src/domains/eval/schema/artifact.ts +20 -0
  320. package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
  321. package/src/domains/eval/schema/behavioral.ts +520 -0
  322. package/src/domains/eval/schema/execution-envelope.ts +194 -0
  323. package/src/domains/eval/schema/serving.ts +105 -0
  324. package/src/domains/eval/schema/suite.ts +38 -8
  325. package/src/domains/eval/schema/validate.ts +58 -3
  326. package/src/domains/eval/schema/verdict.ts +237 -0
  327. package/src/domains/eval/suites/resolve.ts +2 -0
  328. package/src/domains/eval/suites/run.ts +264 -33
  329. package/src/domains/eval/verifiers/command.ts +2 -1
  330. package/src/domains/eval/workspaces/temp-copy.ts +145 -13
  331. package/src/domains/evidence/build.ts +2 -13
  332. package/src/domains/evidence/eval.ts +2 -12
  333. package/src/domains/evidence/findings-markdown.ts +33 -0
  334. package/src/domains/evidence/run-trust.ts +7 -113
  335. package/src/domains/evidence/store.ts +6 -0
  336. package/src/domains/evidence/trust-projection.ts +2 -2
  337. package/src/domains/extensions/compatibility.ts +285 -0
  338. package/src/domains/extensions/discovery.ts +38 -3
  339. package/src/domains/extensions/resources.ts +1 -1
  340. package/src/domains/extensions/state.ts +12 -3
  341. package/src/domains/extensions/types.ts +2 -0
  342. package/src/domains/lifecycle/doctor.ts +69 -1
  343. package/src/domains/memory/index.ts +14 -1
  344. package/src/domains/memory/task-bank-promotion.ts +64 -0
  345. package/src/domains/memory/task-memory-policy.ts +82 -17
  346. package/src/domains/memory/task-memory-spend.ts +131 -0
  347. package/src/domains/memory/task-memory-status.ts +7 -0
  348. package/src/domains/memory/task-memory-telemetry.ts +3 -0
  349. package/src/domains/middleware/index.ts +1 -0
  350. package/src/domains/middleware/memory-intervention.ts +97 -21
  351. package/src/domains/middleware/memory-step-endpoint.ts +71 -0
  352. package/src/domains/mux/contract.ts +434 -0
  353. package/src/domains/mux/detect.ts +158 -0
  354. package/src/domains/mux/extension.ts +47 -0
  355. package/src/domains/mux/index.ts +96 -0
  356. package/src/domains/mux/manifest.ts +6 -0
  357. package/src/domains/mux/operations.ts +164 -0
  358. package/src/domains/mux/pane-registry.ts +90 -0
  359. package/src/domains/mux/protocol.ts +49 -0
  360. package/src/domains/mux/socket-client.ts +816 -0
  361. package/src/domains/mux/types.ts +222 -0
  362. package/src/domains/mux/viewer-command.ts +59 -0
  363. package/src/domains/mux/yazi/assets/init.lua +2 -0
  364. package/src/domains/mux/yazi/assets/plugins/git.yazi/LICENSE +21 -0
  365. package/src/domains/mux/yazi/assets/plugins/git.yazi/README.md +78 -0
  366. package/src/domains/mux/yazi/assets/plugins/git.yazi/main.lua +255 -0
  367. package/src/domains/mux/yazi/assets/plugins/git.yazi/types.lua +12 -0
  368. package/src/domains/mux/yazi/assets/yazi.toml +17 -0
  369. package/src/domains/mux/yazi/event-stream.ts +180 -0
  370. package/src/domains/mux/yazi/profile.ts +299 -0
  371. package/src/domains/mux/yazi/session.ts +228 -0
  372. package/src/domains/mux/yazi/theme.ts +30 -0
  373. package/src/domains/observability/background-memory-usage.ts +140 -0
  374. package/src/domains/observability/cost.ts +22 -1
  375. package/src/domains/observability/index.ts +9 -0
  376. package/src/domains/observability/out-of-turn-usage.ts +51 -2
  377. package/src/domains/observability/trace-store.ts +234 -2
  378. package/src/domains/prompts/compiler.ts +100 -13
  379. package/src/domains/providers/endpoint-capacity.ts +228 -0
  380. package/src/domains/providers/endpoint-slots-store.ts +189 -0
  381. package/src/domains/providers/extension.ts +20 -3
  382. package/src/domains/providers/index.ts +32 -0
  383. package/src/domains/providers/model-runtime-capabilities.ts +32 -0
  384. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
  385. package/src/domains/providers/runtime-resolution.ts +8 -1
  386. package/src/domains/providers/runtimes/boot-manifest.ts +1 -0
  387. package/src/domains/providers/runtimes/builtins.ts +2 -0
  388. package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
  389. package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
  390. package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
  391. package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
  392. package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
  393. package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
  394. package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
  395. package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
  396. package/src/domains/providers/runtimes/protocol/litellm.ts +375 -0
  397. package/src/domains/providers/support.ts +1 -0
  398. package/src/domains/providers/target-model-cache.ts +124 -0
  399. package/src/domains/providers/types/capability-flags.ts +2 -0
  400. package/src/domains/providers/types/target-descriptor.ts +2 -0
  401. package/src/domains/resources/index.ts +3 -0
  402. package/src/domains/resources/prompts/loader.ts +95 -33
  403. package/src/domains/resources/skills/loader.ts +33 -0
  404. package/src/domains/safety/action-classifier.ts +6 -0
  405. package/src/domains/safety/call-target.ts +52 -0
  406. package/src/domains/safety/run-effects.ts +35 -4
  407. package/src/domains/session/context-accounting.ts +52 -1
  408. package/src/domains/session/context-ledger.ts +37 -13
  409. package/src/domains/session/index.ts +6 -0
  410. package/src/domains/session/prompt-cache.ts +140 -0
  411. package/src/domains/session/prompt-manifest.ts +42 -0
  412. package/src/domains/toolchain/archive.ts +175 -0
  413. package/src/domains/toolchain/contract.ts +28 -0
  414. package/src/domains/toolchain/extension.ts +47 -0
  415. package/src/domains/toolchain/index.ts +39 -0
  416. package/src/domains/toolchain/install.ts +327 -0
  417. package/src/domains/toolchain/manifest.ts +8 -0
  418. package/src/domains/toolchain/paths.ts +34 -0
  419. package/src/domains/toolchain/registry.ts +265 -0
  420. package/src/domains/toolchain/remove.ts +218 -0
  421. package/src/domains/toolchain/resolve.ts +182 -0
  422. package/src/domains/toolchain/types.ts +113 -0
  423. package/src/domains/toolchain/version.ts +88 -0
  424. package/src/engine/acp/adapter.ts +18 -3
  425. package/src/engine/acp/server.ts +413 -70
  426. package/src/engine/acp/types.ts +19 -1
  427. package/src/engine/ai.ts +35 -0
  428. package/src/engine/apis/llamacpp-residency.ts +55 -3
  429. package/src/engine/apis/lmstudio.ts +25 -5
  430. package/src/engine/apis/ollama-native.ts +2 -1
  431. package/src/engine/apis/openai-completions.ts +80 -17
  432. package/src/engine/apis/residency-lock.ts +3 -1
  433. package/src/engine/apis/residency.ts +34 -1
  434. package/src/engine/claude/sdk-module.ts +98 -0
  435. package/src/engine/claude/sdk-runtime.ts +19 -11
  436. package/src/engine/provider-payload.ts +29 -1
  437. package/src/engine/tui-primitives.ts +21 -0
  438. package/src/engine/tui.ts +1 -0
  439. package/src/engine/worker-runtime.ts +2 -13
  440. package/src/entry/boot-options.ts +2 -0
  441. package/src/entry/orchestrator.ts +279 -35
  442. package/src/entry/panes-activation.ts +31 -0
  443. package/src/entry/with-panes.ts +20 -0
  444. package/src/interactive/chat-loop-messages.ts +26 -7
  445. package/src/interactive/chat-loop.ts +318 -41
  446. package/src/interactive/chat-panel.ts +62 -8
  447. package/src/interactive/clio-editor.ts +45 -8
  448. package/src/interactive/context-activity.ts +5 -1
  449. package/src/interactive/context-meter.ts +1 -1
  450. package/src/interactive/context-overlay.ts +41 -10
  451. package/src/interactive/cost-overlay.ts +66 -6
  452. package/src/interactive/council-grid.ts +1 -3
  453. package/src/interactive/council.ts +11 -0
  454. package/src/interactive/dispatch-board.ts +102 -20
  455. package/src/interactive/fleet-run-preview.ts +41 -15
  456. package/src/interactive/handoff-round.ts +41 -2
  457. package/src/interactive/interactive-application.ts +177 -7
  458. package/src/interactive/interactive-input-runtime.ts +15 -0
  459. package/src/interactive/interactive-presentation.ts +4 -0
  460. package/src/interactive/interactive-shell.ts +20 -17
  461. package/src/interactive/interactive-slash-runtime.ts +63 -10
  462. package/src/interactive/memory-overlay.ts +9 -0
  463. package/src/interactive/modal-marker.ts +170 -0
  464. package/src/interactive/mutation-preview.ts +295 -0
  465. package/src/interactive/mux-bridge.ts +214 -0
  466. package/src/interactive/overlay-frame.ts +58 -2
  467. package/src/interactive/overlay-general-openers.ts +17 -0
  468. package/src/interactive/overlay-key-routing.ts +52 -3
  469. package/src/interactive/overlay-lifecycle.ts +56 -7
  470. package/src/interactive/overlay-model-selectors.ts +40 -3
  471. package/src/interactive/overlay-permission-lifecycle.ts +112 -24
  472. package/src/interactive/overlay-session-lifecycle.ts +73 -9
  473. package/src/interactive/overlay-transitions.ts +18 -4
  474. package/src/interactive/overlays/agents.ts +1 -0
  475. package/src/interactive/overlays/ask-user.ts +227 -49
  476. package/src/interactive/overlays/auth-dialog.ts +1 -0
  477. package/src/interactive/overlays/context-reset.ts +1 -0
  478. package/src/interactive/overlays/cwd-fallback.ts +1 -0
  479. package/src/interactive/overlays/decisions.ts +11 -11
  480. package/src/interactive/overlays/extensions.ts +1 -0
  481. package/src/interactive/overlays/fleet-run-approval.ts +1 -0
  482. package/src/interactive/overlays/handoff-review.ts +1 -0
  483. package/src/interactive/overlays/help-reference.ts +6 -0
  484. package/src/interactive/overlays/interop.ts +1 -0
  485. package/src/interactive/overlays/library-install-confirm.ts +1 -0
  486. package/src/interactive/overlays/library-tabs.ts +28 -0
  487. package/src/interactive/overlays/list-overlay.ts +10 -1
  488. package/src/interactive/overlays/message-picker.ts +1 -0
  489. package/src/interactive/overlays/model-scope.ts +86 -0
  490. package/src/interactive/overlays/model-selector.ts +1 -0
  491. package/src/interactive/overlays/prompts.ts +12 -1
  492. package/src/interactive/overlays/session-selector.ts +1 -0
  493. package/src/interactive/overlays/settings-sections.ts +30 -0
  494. package/src/interactive/overlays/settings.ts +614 -45
  495. package/src/interactive/overlays/side-question.ts +1 -0
  496. package/src/interactive/overlays/skills-hub.ts +3 -11
  497. package/src/interactive/overlays/tree-selector.ts +1 -0
  498. package/src/interactive/pane-policy.ts +46 -0
  499. package/src/interactive/panes-runtime.ts +292 -0
  500. package/src/interactive/permission-hint.ts +34 -2
  501. package/src/interactive/permission-overlay.ts +159 -9
  502. package/src/interactive/prewarm.ts +197 -0
  503. package/src/interactive/render-trace.ts +162 -15
  504. package/src/interactive/renderers/compaction-summary.ts +29 -0
  505. package/src/interactive/renderers/tool-execution.ts +4 -0
  506. package/src/interactive/renderers/worker-entry.ts +122 -14
  507. package/src/interactive/side-question.ts +58 -1
  508. package/src/interactive/slash-commands.ts +251 -15
  509. package/src/interactive/status/controller.ts +11 -0
  510. package/src/interactive/status/state-machine.ts +54 -2
  511. package/src/interactive/status/types.ts +7 -0
  512. package/src/interactive/tasks-overlay.ts +1 -0
  513. package/src/interactive/terminal-lease.ts +2 -0
  514. package/src/interactive/theme/tokens.ts +3 -14
  515. package/src/interactive/turn-context.ts +346 -33
  516. package/src/interactive/turn-persistence.ts +14 -4
  517. package/src/interactive/turn-prewarm.ts +364 -0
  518. package/src/interactive/turn-queues.ts +7 -4
  519. package/src/interactive/turn-runtime.ts +8 -1
  520. package/src/interactive/turn-state.ts +23 -0
  521. package/src/interactive/view/artifacts.ts +109 -1
  522. package/src/interactive/view/view-overlay.ts +29 -3
  523. package/src/interactive/watch-pane.ts +152 -0
  524. package/src/interactive/worker-progress.ts +7 -1
  525. package/src/interactive/worker-receipts.ts +19 -1
  526. package/src/interactive/worker-stream.ts +5 -0
  527. package/src/interactive/yazi-bridge.ts +444 -0
  528. package/src/tools/ask-user.ts +43 -2
  529. package/src/tools/bootstrap.ts +26 -2
  530. package/src/tools/builtin-tool-catalog.ts +15 -0
  531. package/src/tools/compete-worktrees.ts +83 -2
  532. package/src/tools/core-bootstrap.ts +2 -1
  533. package/src/tools/dispatch-admission.ts +14 -3
  534. package/src/tools/dispatch-arguments.ts +20 -20
  535. package/src/tools/dispatch-plan.ts +17 -9
  536. package/src/tools/dispatch-run-events.ts +134 -19
  537. package/src/tools/dispatch-runner.ts +29 -7
  538. package/src/tools/dispatch-scout.ts +1 -1
  539. package/src/tools/dispatch-types.ts +15 -3
  540. package/src/tools/dispatch.ts +1 -1
  541. package/src/tools/executables.ts +17 -14
  542. package/src/tools/observation.ts +54 -4
  543. package/src/tools/panes-surface.ts +38 -0
  544. package/src/tools/panes.ts +112 -0
  545. package/src/tools/policy.ts +10 -1
  546. package/src/tools/presentation.ts +1 -0
  547. package/src/tools/registry.ts +16 -0
  548. package/dist/chunk-AOCYTWAV.js +0 -449
  549. package/dist/chunk-HLE42MG7.js +0 -37
  550. package/dist/chunk-HWUFFB6L.js +0 -83
  551. package/dist/chunk-JOZYP4GM.js +0 -279
  552. package/dist/doctor-M7YEDGAE.js +0 -91
@@ -27,6 +27,31 @@ export {
27
27
  export { AGENT_ROLE_TOOLS_REQUIRED_REASON, mergeCapabilities, supportsAgentRoleTools } from "./capabilities.js";
28
28
  export type { ProvidersContract, TargetHealth, TargetStatus } from "./contract.js";
29
29
  export { isDispatchEligibleRuntime, isOrchestratorEligibleRuntime, isTargetEligibleRuntime } from "./eligibility.js";
30
+ export {
31
+ canonicalEndpointKey,
32
+ type EndpointCapacity,
33
+ type EndpointCapacityInput,
34
+ type EndpointCapacitySource,
35
+ type EndpointRuntimeIdentity,
36
+ type EndpointSlotPriors,
37
+ endpointCapacitiesForStatuses,
38
+ endpointCapacityFor,
39
+ endpointCapacityForStatus,
40
+ endpointLabel,
41
+ foregroundStreamUsage,
42
+ recordEndpointSlotsFromStatus,
43
+ registerForegroundStream,
44
+ resolveEndpointCapacities,
45
+ } from "./endpoint-capacity.js";
46
+ export {
47
+ DEFAULT_ENDPOINT_SLOTS_TTL_MS,
48
+ type DiscoveredEndpointSlots,
49
+ ENDPOINT_SLOTS_TTL_ENV_VAR,
50
+ endpointSlotsPath,
51
+ endpointSlotsTtlMs,
52
+ readDiscoveredEndpointSlots,
53
+ recordDiscoveredEndpointSlots,
54
+ } from "./endpoint-slots-store.js";
30
55
  export { ProvidersManifest } from "./manifest.js";
31
56
  export type { ModelCapabilityPatchTarget } from "./model-capabilities.js";
32
57
  export { applyModelCapabilityPatch, resolveModelCapabilities } from "./model-capabilities.js";
@@ -116,6 +141,13 @@ export {
116
141
  runtimeModelListSource,
117
142
  supportGroupLabel,
118
143
  } from "./support.js";
144
+ export {
145
+ readTargetModelSnapshot,
146
+ recordTargetModelSnapshot,
147
+ TARGET_MODEL_CACHE_TTL_MS,
148
+ type TargetModelSnapshot,
149
+ targetModelSnapshotPath,
150
+ } from "./target-model-cache.js";
119
151
  export type {
120
152
  CapabilityFlags,
121
153
  StructuredOutputMode,
@@ -178,6 +178,38 @@ export function harmonyReasoningEffort(level: string | undefined): HarmonyReason
178
178
  */
179
179
  export type ReasoningClass = "never" | "switchable" | "always";
180
180
 
181
+ /**
182
+ * Minimum completion room for a memory envelope on an always-on-thinking
183
+ * model. A 9B local route has been measured reasoning for more than 1,800
184
+ * tokens before its first `<operations>` byte, while an always-on 35B spent a
185
+ * 400-token allowance entirely on reasoning and ended at length. Four thousand
186
+ * leaves the configured 2,000-token content budget after that preamble.
187
+ */
188
+ export const MEMORY_ALWAYS_ON_MODEL_MIN_OUTPUT_TOKENS = 4_000;
189
+
190
+ /**
191
+ * Derive the memory step's request budget from the resolved model capability.
192
+ * Switchable models are requested with thinking off and need only the normal
193
+ * content budget. An always-on mechanism cannot honor that request, so it gets
194
+ * explicit reasoning headroom. The runtime's known cap remains authoritative.
195
+ */
196
+ export function memoryInterventionModelMaxTokens(input: {
197
+ configuredMaxTokens: number;
198
+ thinkingMechanism: ThinkingMechanism;
199
+ modelMaxTokens?: number;
200
+ }): number {
201
+ const configured =
202
+ Number.isSafeInteger(input.configuredMaxTokens) && input.configuredMaxTokens > 0 ? input.configuredMaxTokens : 1;
203
+ const requested =
204
+ input.thinkingMechanism === "always-on"
205
+ ? Math.max(configured * 2, MEMORY_ALWAYS_ON_MODEL_MIN_OUTPUT_TOKENS)
206
+ : configured;
207
+ const modelCap = input.modelMaxTokens;
208
+ return typeof modelCap === "number" && Number.isSafeInteger(modelCap) && modelCap > 0
209
+ ? Math.min(requested, modelCap)
210
+ : requested;
211
+ }
212
+
181
213
  export function reasoningClassForMechanism(mechanism: ThinkingMechanism | null | undefined): ReasoningClass {
182
214
  if (mechanism === "none") return "never";
183
215
  if (mechanism === "always-on") return "always";
@@ -16,6 +16,30 @@
16
16
  # models curated for Clio's local coding workflows and leaves cloud GPT models
17
17
  # to pi-ai's native OpenAI and openai-codex catalogs.
18
18
  #
19
+ # Serving configuration is a quality variable, not a constant. Context length,
20
+ # KV quantization, batch and ubatch size, speculative decoding, and the server's
21
+ # own sampler defaults all move what a model does, so a quirk measured under one
22
+ # configuration is not a property of the weights. Every family entry whose
23
+ # quirks came from a measurement rather than from the model card carries a
24
+ # `measuredUnder` block naming the hardware class, the runtime and its build
25
+ # string, and the llama.cpp flags the measurement ran under. Fields that could
26
+ # not be established from git history or the release-test reports under
27
+ # docs/release-notes/ read `unknown` rather than a plausible guess. A
28
+ # `measuredUnder` block is provenance for a reader; nothing in the engine
29
+ # consumes it (`extractLocalModelQuirks` narrows only kvCache, sampling, and
30
+ # thinking).
31
+ #
32
+ # `llamaCpp.parallel` is the recommended `--parallel` value for starting the
33
+ # server, and families that carry one also carry a `parallelSlots` note. It is
34
+ # not what dispatch admission counts. Endpoint capacity resolves a target's
35
+ # request-slot limit in this order: an explicit `maxConcurrentRequests` on the
36
+ # target, then `parallelSlots` discovered from the server (llama.cpp reads
37
+ # `total_slots` from `/props`, falling back to the selected worker's
38
+ # `/props?model=<id>` and then to its `--parallel` argv), then one slot for any
39
+ # other local-native runtime. vLLM and SGLang stay unbounded without an explicit
40
+ # override. The catalog value therefore says what to start the server with; the
41
+ # server's own answer is what Clio admits against.
42
+ #
19
43
  # Thinking semantics for these local families: the chain-of-thought is emitted
20
44
  # by the model's chat template (Qwen-style <think> blocks, Gemma 4 thinking
21
45
  # template, Nemotron reasoning template). LM Studio's OpenAI-compatible HTTP
@@ -89,6 +113,18 @@
89
113
  contextWindow: 262144
90
114
  maxTokens: 65536
91
115
  quirks:
116
+ measuredUnder:
117
+ hardware: "32 GiB class GPU; the exact card was not recorded"
118
+ runtime: llamacpp
119
+ build: unknown
120
+ llamaCpp: "ctx 262144, --parallel 4, --batch-size 2048, --ubatch-size 512, KV q8_0/q8_0 (the block below); no argv was recorded"
121
+ date: unknown
122
+ source: unknown
123
+ note: |
124
+ Only the VRAM-headroom figure in the 32gb tier reads as measured. It
125
+ entered the catalog with the initial local-model set and no commit,
126
+ report, or server argv records where it was taken, so everything but
127
+ the recommended serving profile is unknown.
92
128
  sampling:
93
129
  thinking:
94
130
  temperature: 0.5
@@ -111,6 +147,7 @@
111
147
  cacheTypeK: q8_0
112
148
  cacheTypeV: q8_0
113
149
  parallel: 4
150
+ parallelSlots: "Start the server with --parallel 4. Clio discovers total_slots 4 from /props and admits four concurrent requests on this endpoint, one of which the orchestrator's own turn holds while it streams. Add --kv-unified, or the 262144-token context is divided four ways."
114
151
  batchSize: 2048
115
152
  ubatchSize: 512
116
153
  thinking:
@@ -174,6 +211,7 @@
174
211
  flashAttn: true
175
212
  nGpuLayers: 99
176
213
  parallel: 4
214
+ parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 204800 tokens rather than the whole 819200. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
177
215
  thinking:
178
216
  mechanism: budget-tokens
179
217
  budgetByLevel:
@@ -267,6 +305,19 @@
267
305
  contextWindow: 262144
268
306
  maxTokens: 65536
269
307
  quirks:
308
+ measuredUnder:
309
+ hardware: unknown
310
+ runtime: lmstudio
311
+ build: "unknown; the host was the operator's `dynamo` LM Studio machine, whose bundled build string was not recorded on this date"
312
+ llamaCpp: "unknown; the measurement went through LM Studio's OpenAI-compatible port, which does not expose the underlying server argv"
313
+ date: "2026-08-08"
314
+ source: "commit b3db3b82 (fix(providers): thinking off reaches the wire for models that reason by default)"
315
+ note: |
316
+ The always-on classification is the measured part. That session found
317
+ Ornith reasoning by default with no thinking field set, in the same
318
+ family of behavior as qwopus3.6-35b-a3b-coder. The recommended
319
+ llama.cpp serving profile below is the model card's, not that
320
+ measurement's, and no argv from the measuring host survives.
270
321
  kvCache:
271
322
  kQuant: q8_0
272
323
  vQuant: q8_0
@@ -295,6 +346,7 @@
295
346
  flashAttn: true
296
347
  nGpuLayers: 99
297
348
  parallel: 1
349
+ parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator is streaming is refused with the endpoint denial rather than queued; point workers at a second server, or raise --parallel and re-probe."
298
350
  batchSize: 2048
299
351
  ubatchSize: 512
300
352
  thinking:
@@ -326,6 +378,20 @@
326
378
  contextWindow: 262144
327
379
  maxTokens: 32768
328
380
  quirks:
381
+ measuredUnder:
382
+ hardware: unknown
383
+ runtime: "lmstudio and its OpenAI-compatible port"
384
+ build: unknown
385
+ llamaCpp: "unknown; this family carries no llamaCpp block and the parity check ran over LM Studio's HTTP surface"
386
+ date: unknown
387
+ source: unknown
388
+ note: |
389
+ The measured claim is the openaiCompat parity line: tool calls, a
390
+ reasoning_content field, and completion_tokens_details.reasoning_tokens
391
+ were all observed on the wire. It entered the catalog with the initial
392
+ local-model set, so the host, the LM Studio version, and the loaded
393
+ quantization are all unrecorded. The sampler and the 32gb VRAM
394
+ arithmetic come from the official Qwen card rather than a run.
329
395
  sampling:
330
396
  thinking:
331
397
  temperature: 0.6
@@ -380,6 +446,28 @@
380
446
  contextWindow: 262144
381
447
  maxTokens: 131072
382
448
  quirks:
449
+ measuredUnder:
450
+ hardware: "AMD Radeon AI PRO R9700, gfx1201, 32 GB, ROCm 7.2.0"
451
+ runtime: llamacpp
452
+ build: b226-2115b73d8
453
+ model: "Qwen3.8-27B IQ4_NL, mmproj F16"
454
+ llamaCpp: >-
455
+ --ctx-size 262144 --parallel 1 --kv-unified --cache-type-k q8_0
456
+ --cache-type-v q8_0 --flash-attn true --batch-size 2048 --ubatch-size 512
457
+ --spec-type draft-mtp --spec-draft-n-max 4 --n-gpu-layers-draft 99
458
+ --swa-checkpoints 64 --checkpoint-min-step 1024 --jinja --reasoning on
459
+ --reasoning-effort high --temperature 1.0 --top-k 20 --top-p 0.95
460
+ --min-p 0.0 --presence-penalty 0.0 --repeat-penalty 1.0
461
+ date: "2026-08-29"
462
+ source: docs/release-notes/v0.3.9-local-economics.md
463
+ alsoMeasuredOn: "LM Studio 2.29.0 (OpenAI-compatible HTTP port), host build string llama.cpp-win-x86_64-nvidia-cuda12-avx2"
464
+ note: |
465
+ The server above is deliberately not the 24gb-rocm tier below: it runs
466
+ ctx=262144 and parallel=1 where that tier recommends ctx=131072 and
467
+ parallel=2. The reasoning_effort vocabulary and the enable_thinking
468
+ control were established earlier, on 2026-08-18, from a llama.cpp host
469
+ and an LM Studio host whose build strings were not recorded; treat those
470
+ two findings as unknown-build.
383
471
  kvCache:
384
472
  kQuant: q8_0
385
473
  vQuant: q8_0
@@ -415,6 +503,7 @@
415
503
  flashAttn: true
416
504
  nGpuLayers: 99
417
505
  parallel: 2
506
+ parallelSlots: "Start the server with --parallel 2 to serve the orchestrator and one worker at once. Clio then discovers total_slots from /props and admits two concurrent requests against this endpoint; a server started with --parallel 1 discovers one slot, and a dispatch during the orchestrator's own turn is refused rather than queued."
418
507
  batchSize: 1024
419
508
  ubatchSize: 256
420
509
  mmproj: true
@@ -443,7 +532,26 @@
443
532
  and is a separate mechanism from the template's enable_thinking flag;
444
533
  both were verified independently on llama.cpp and on LM Studio's HTTP
445
534
  port.
446
- serving: "Qwen3.8-27B (unsloth/Qwen3.8-27B-GGUF), hybrid qwen3_5 architecture: 3 linear-attention (Gated DeltaNet) layers per 1 full-attention layer, so only 16 of 64 layers hold a conventional KV cache. Native ctx 262144, extensible to 1M via YaRN (out of scope for the hardware classes above; leave rope at default). Tool calls use an XML <tool_call><function=...> form in the stock template; llama.cpp's qwen3_coder parser converts this to standard tool_calls JSON, verified live on ROCm, CUDA, and Vulkan backends. Only messages[0] may be a system message and only system/user/assistant/tool roles are accepted; a second system message or a developer role raises a template error."
535
+
536
+ Thinking off is verified for this family, and no floor is imposed on
537
+ it. JetBrains report that Qwen3.8 without reasoning gets stuck in a
538
+ loop repeating the same tool call indefinitely under Junie. Clio's
539
+ shipped default is orchestrator.thinkingLevel: off, which on llama.cpp
540
+ sends chat_template_kwargs.enable_thinking:false and on LM Studio's
541
+ OpenAI-compatible port sends reasoning_effort:"none". Clio 0.3.8 ran
542
+ that configuration on both runtimes (llama.cpp build b226-2115b73d8 on
543
+ ROCm, LM Studio 2.29.0) through the v0.3.9 local-inference campaign on
544
+ 2026-08-29 and 2026-08-30: an interactive build-and-resume session, a
545
+ three-worker fleet step, and a ten-task eval suite totalling 116 model
546
+ calls, none of which repeated a tool call into a loop. The catalog
547
+ therefore imposes no thinking floor and coerces no level for this
548
+ family. If a future build or quantization regresses into the loop
549
+ JetBrains describe, the guards that catch it are Clio's own, not a
550
+ catalog setting: the tool-prose-loop detector
551
+ (src/interactive/tool-prose-loop.ts), which is armed on the
552
+ local-native tier, and the stalled-turn middleware
553
+ (src/domains/middleware/stalled-turn.ts).
554
+ serving: "Qwen3.8-27B (unsloth/Qwen3.8-27B-GGUF), hybrid qwen3_5 architecture: 3 linear-attention (Gated DeltaNet) layers per 1 full-attention layer, so only 16 of 64 layers hold a conventional KV cache. Native ctx 262144, extensible to 1M via YaRN (out of scope for the hardware classes above; leave rope at default). Tool calls use an XML <tool_call><function=...> form in the stock template; llama.cpp's qwen3_coder parser converts this to standard tool_calls JSON, verified live on ROCm, CUDA, and Vulkan backends. Only messages[0] may be a system message and only system/user/assistant/tool roles are accepted; a second system message or a developer role raises a template error. The recurrent layers set the prefix-cache economics: llama.cpp cannot roll gated-deltanet state back to an arbitrary token, so a divergence anywhere inside a cached prefix re-prefills from the last context checkpoint rather than from the changed byte. Checkpoints are written at the end of each processed prompt and only when at least --checkpoint-min-step tokens separate them, so the server's checkpoint count and --checkpoint-min-step are a quality variable on this family and not a tuning detail: measured on the configuration recorded in measuredUnder, one changed line inside an earlier turn of a four-turn conversation cost cache_n 9 and 17.9 s at the build defaults (32 checkpoints, min step 8192) and cache_n 7570 and 10.9 s at 64 checkpoints and min step 1024. A single prefill writes one checkpoint at its end, so a 16.7K-token prompt re-sent with one changed line re-prefills from zero (18.4 s) under both settings, while changing only the final user message re-prefills the last ubatch (prompt_n 517, 1.4 s). Serving this family with --parallel greater than 1 divides --ctx-size across slots unless --kv-unified is set."
447
555
 
448
556
  - family: gemma-4-31b-it-nvfp4-turbo
449
557
  matchPatterns:
@@ -467,6 +575,21 @@
467
575
  contextWindow: 122880
468
576
  maxTokens: 32768
469
577
  quirks:
578
+ measuredUnder:
579
+ hardware: unknown
580
+ runtime: "lmstudio and its OpenAI-compatible port"
581
+ build: unknown
582
+ llamaCpp: "unknown; this family carries no llamaCpp block, and the channel-marker leakage was observed through LM Studio's stream"
583
+ date: unknown
584
+ source: unknown
585
+ note: |
586
+ Two claims here are observations rather than card values: the
587
+ channel-marker leakage in leakageNote, and the instruct temperature
588
+ pinned at 0.7 because NVFP4 quantization corrupts tool-call JSON under
589
+ the card's 1.0. Neither records the host, the LM Studio version, or the
590
+ exact NVFP4 build it was seen on. The KV-budget arithmetic in the
591
+ gpuTiers and the serving note is derived from the official Google
592
+ Gemma 4 31B card.
470
593
  kvCache:
471
594
  kQuant: q8_0
472
595
  vQuant: q8_0
@@ -567,6 +690,7 @@
567
690
  flashAttn: true
568
691
  nGpuLayers: 99
569
692
  parallel: 1
693
+ parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised during the orchestrator's turn is refused rather than queued."
570
694
  batchSize: 1024
571
695
  ubatchSize: 256
572
696
  mmproj: true
@@ -621,6 +745,19 @@
621
745
  minP: 0.0
622
746
  maxTokens: 32768
623
747
  chatTemplate: "gemma-channel"
748
+ measuredUnder:
749
+ hardware: unknown
750
+ runtime: "lmstudio and its OpenAI-compatible port"
751
+ build: unknown
752
+ llamaCpp: "unknown; this family carries no llamaCpp block"
753
+ date: unknown
754
+ source: unknown
755
+ note: |
756
+ The measured claims are the empty thought region the model emits when
757
+ the '<|think|>' token is absent from the system prompt, and the absence
758
+ of reasoning_content from the openai-compat reasoning probe. Both
759
+ entered the catalog with the initial local-model set, with no host, LM
760
+ Studio version, or quantization recorded.
624
761
  thinkingControl: |
625
762
  Jackrong's distilled gemma-4 31B uses the same channel-marker template.
626
763
  Thinking mode is gated by the literal '<|think|>' token in the system
@@ -661,6 +798,18 @@
661
798
  contextWindow: 262144
662
799
  maxTokens: 32768
663
800
  quirks:
801
+ measuredUnder:
802
+ hardware: unknown
803
+ runtime: "lmstudio and its OpenAI-compatible port"
804
+ build: unknown
805
+ llamaCpp: "unknown; this family carries no llamaCpp block"
806
+ date: unknown
807
+ source: unknown
808
+ note: |
809
+ The measured claim is the openaiCompat tool-call extraction parity and
810
+ the reasoning_content field surfacing. It entered the catalog with the
811
+ initial local-model set and records no host, LM Studio version, or
812
+ quantization. The sampler comes from the model card.
664
813
  sampling:
665
814
  thinking:
666
815
  temperature: 0.6
@@ -732,6 +881,26 @@
732
881
  contextWindow: 262144
733
882
  maxTokens: 32768
734
883
  quirks:
884
+ measuredUnder:
885
+ hardware: unknown
886
+ runtime: "lmstudio and its OpenAI-compatible port for the reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
887
+ build: "unknown for both dates; the operator's `dynamo` host is a Windows LM Studio machine whose bundled build string was first recorded on 2026-08-29 as llama.cpp-win-x86_64-nvidia-cuda12-avx2, which is three weeks after the reasoning measurement and cannot be backdated to it"
888
+ llamaCpp: "unknown; the reasoning measurement went through LM Studio's HTTP port, which does not expose the underlying server argv, and no argv survives for the dispatch that produced the presence-penalty value"
889
+ date: "2026-08-08 (reasoning_effort and enable_thinking), 2026-07-06 (presence penalty)"
890
+ source: "commits b3db3b82 and f7d76bf4"
891
+ note: |
892
+ Three numbers here are measurements. On 2026-08-08 on `dynamo` this
893
+ model spent 98 of 103 completion tokens reasoning on "what is 17+25"
894
+ with no thinking field set; chat_template_kwargs enable_thinking was
895
+ inert, re-checked with unique prompts to rule out prompt-cache hits;
896
+ and reasoning_effort "none" took the same prompt to 0. The same session
897
+ recorded a wiki planning dispatch over a 1007-file repository going
898
+ from 89,501 reasoning tokens, 47 tool calls and exit 1 at 459 s to 0
899
+ reasoning tokens, 8 tool calls and exit 0 at 218 s. On 2026-07-06 a
900
+ live dispatch measured presence_penalty 1.5: without it a coder worker
901
+ repeated one code_nav call into the loop-guard abort on 3 of 3 runs,
902
+ and with it the same task passed 3 of 3. Neither commit records the
903
+ hardware or the server configuration, so both stay unknown.
735
904
  sampling:
736
905
  instruct:
737
906
  temperature: 0.2
@@ -757,6 +926,7 @@
757
926
  flashAttn: true
758
927
  nGpuLayers: 99
759
928
  parallel: 1
929
+ parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 262144-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint. This family is often the workers.default model on a router that also serves the orchestrator, and a one-slot router refuses the worker batch rather than queueing it; a second server or an explicit maxConcurrentRequests on the target is the remedy."
760
930
  batchSize: 2048
761
931
  ubatchSize: 512
762
932
  mmproj: true
@@ -801,6 +971,20 @@
801
971
  contextWindow: 262144
802
972
  maxTokens: 32768
803
973
  quirks:
974
+ measuredUnder:
975
+ hardware: unknown
976
+ runtime: unknown
977
+ build: unknown
978
+ llamaCpp: unknown
979
+ date: unknown
980
+ source: "commit b3db3b82, which states the exclusion directly"
981
+ note: |
982
+ Nothing in this entry was measured on this model. The reasoning:false
983
+ pin and the presence penalty are both carried over from the 35B-A3B
984
+ Coder-MTP design; the commit that reclassified the 35B on live evidence
985
+ says in as many words that the 27B and 9B Coder variants keep their
986
+ "never" pin because nothing has measured them. Read the reasoning class
987
+ as an untested inheritance until someone runs this model.
804
988
  sampling:
805
989
  instruct:
806
990
  temperature: 0.2
@@ -841,6 +1025,19 @@
841
1025
  contextWindow: 262144
842
1026
  maxTokens: 32768
843
1027
  quirks:
1028
+ measuredUnder:
1029
+ hardware: unknown
1030
+ runtime: unknown
1031
+ build: unknown
1032
+ llamaCpp: unknown
1033
+ date: unknown
1034
+ source: "commit b3db3b82, which states the exclusion directly"
1035
+ note: |
1036
+ Nothing in this entry was measured on this model. The reasoning:false
1037
+ pin is carried over from the Coder-MTP design; the commit that
1038
+ reclassified the 35B-A3B on live evidence says in as many words that
1039
+ the 27B and 9B Coder variants keep their "never" pin because nothing
1040
+ has measured them.
844
1041
  sampling:
845
1042
  instruct:
846
1043
  temperature: 0.2
@@ -936,6 +1133,21 @@
936
1133
  contextWindow: 1048576
937
1134
  maxTokens: 65536
938
1135
  quirks:
1136
+ measuredUnder:
1137
+ hardware: unknown
1138
+ runtime: "llamacpp; the anthropicCompat probe ran against a llama.cpp target's /v1/messages and /v1/models"
1139
+ build: unknown
1140
+ llamaCpp: "unknown; the recommended profile below was not recorded as the profile the probe ran under"
1141
+ date: unknown
1142
+ source: unknown
1143
+ note: |
1144
+ The measured claim is the anthropicCompat line: /v1/messages returned
1145
+ 404 on the llama.cpp target profile in use while /v1/models and the
1146
+ OpenAI-compatible chat surface both worked. That is a fact about one
1147
+ server's mounted routes, so it is exactly the kind of finding a
1148
+ different build or launch configuration can invalidate, and no build or
1149
+ argv was recorded with it. Everything else in this entry comes from the
1150
+ NVIDIA card.
939
1151
  sampling:
940
1152
  instruct:
941
1153
  temperature: 0.3
@@ -956,6 +1168,7 @@
956
1168
  flashAttn: true
957
1169
  nGpuLayers: 99
958
1170
  parallel: 4
1171
+ parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 262144 tokens rather than the whole 1048576. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
959
1172
  chatTemplateKwargs:
960
1173
  enable_thinking: false
961
1174
  thinking:
@@ -990,6 +1203,19 @@
990
1203
  contextWindow: 262144
991
1204
  maxTokens: 32768
992
1205
  quirks:
1206
+ measuredUnder:
1207
+ hardware: unknown
1208
+ runtime: lmstudio
1209
+ build: unknown
1210
+ llamaCpp: "unknown; this family carries no llamaCpp block"
1211
+ date: unknown
1212
+ source: unknown
1213
+ note: |
1214
+ The only measured claim is the lmstudio line, that the model loads and
1215
+ runs usably under the Qwen-style preset. It entered the catalog with
1216
+ the initial local-model set and records no host, LM Studio version, or
1217
+ quantization. The sampler and the thinking budgets follow the Qwen3.5
1218
+ family defaults rather than a run on this distillation.
993
1219
  sampling:
994
1220
  thinking:
995
1221
  temperature: 0.6
@@ -1041,6 +1267,22 @@
1041
1267
  contextWindow: 262144
1042
1268
  maxTokens: 32768
1043
1269
  quirks:
1270
+ measuredUnder:
1271
+ hardware: "unknown; the host was the operator's `dynamo` LM Studio machine"
1272
+ runtime: "lmstudio, SDK 1.5.0 and its OpenAI-compatible port"
1273
+ build: "unknown; LM Studio's bundled llama.cpp build string was not recorded on this date"
1274
+ llamaCpp: "unknown; the measurement went through LM Studio, which does not expose the underlying server argv"
1275
+ date: "2026-08-11"
1276
+ source: "commit fbaa0492 (fix: make thinking off reach the model on LM Studio routes)"
1277
+ note: |
1278
+ The always-on classification is measured, and it is measured against
1279
+ every control that could have disproved it. Baseline 56 reasoning
1280
+ tokens; reasoning_effort none 120; reasoning_effort minimal 243;
1281
+ chat_template_kwargs.enable_thinking false 45; both together 44. No
1282
+ spelling silences it, and the budget-tokens mechanism the entry
1283
+ previously claimed never reached the wire on either LM Studio or
1284
+ llama.cpp. The gpuTiers 16gb line is a role assignment rather than a
1285
+ VRAM measurement.
1044
1286
  sampling:
1045
1287
  instruct:
1046
1288
  temperature: 0.2
@@ -147,6 +147,13 @@ export interface ResolveRuntimeTargetInput {
147
147
  requireTools?: boolean;
148
148
  requireStreaming?: boolean;
149
149
  requireOutputBudget?: boolean;
150
+ /**
151
+ * A loaded window this target and model were already observed serving, from
152
+ * a caller that remembers across processes. Used only when live discovery
153
+ * reports no loaded window, so a resumed session stops budgeting against a
154
+ * probed server-wide figure for its first turn (issue #227).
155
+ */
156
+ knownLoadedContextWindow?: number | null;
150
157
  }
151
158
 
152
159
  function diagnostic(severity: RuntimeResolutionSeverity, code: string, message: string): RuntimeResolutionDiagnostic {
@@ -405,7 +412,7 @@ export function resolveRuntimeTarget(
405
412
  // Discovery's per-model loaded window, which the probe capabilities cannot
406
413
  // carry: `probeCapabilitiesForModel` answers for the target's default model
407
414
  // and reports a window without saying whether it is the one being served.
408
- const loadedContextWindow = loadedContextWindowForModel(status, wireModelId);
415
+ const loadedContextWindow = loadedContextWindowForModel(status, wireModelId) ?? input.knownLoadedContextWindow ?? null;
409
416
  const contextWindowDetails = resolveContextWindowDetails(
410
417
  target,
411
418
  runtime,
@@ -42,6 +42,7 @@ export const BUILTIN_RUNTIME_BOOT_MANIFEST: ReadonlyArray<RuntimeBootMetadata> =
42
42
  { id: "lmstudio", aliases: ["lmstudio-native"], kind: "http", tier: "local-native", auth: "api-key" },
43
43
  { id: "ollama-native", kind: "http", tier: "local-native", auth: "none" },
44
44
  { id: "anthropic-compat", kind: "http", tier: "protocol", auth: "api-key" },
45
+ { id: "litellm", kind: "http", tier: "protocol", auth: "api-key" },
45
46
  { id: "openai-compat", kind: "http", tier: "protocol", auth: "api-key" },
46
47
  { id: "sglang", kind: "http", tier: "local-native", auth: "api-key" },
47
48
  { id: "vllm", kind: "http", tier: "local-native", auth: "api-key" },
@@ -36,6 +36,7 @@ import ollamaNative from "./local-native/ollama-native.js";
36
36
  import sglang from "./local-native/sglang.js";
37
37
  import vllm from "./local-native/vllm.js";
38
38
  import anthropicCompat from "./protocol/anthropic-compat.js";
39
+ import litellm from "./protocol/litellm.js";
39
40
  import openaiCompat from "./protocol/openai-compat.js";
40
41
 
41
42
  const BUILTIN_RUNTIMES: ReadonlyArray<RuntimeDescriptor> = [
@@ -60,6 +61,7 @@ const BUILTIN_RUNTIMES: ReadonlyArray<RuntimeDescriptor> = [
60
61
  lmstudio,
61
62
  ollamaNative,
62
63
  anthropicCompat,
64
+ litellm,
63
65
  openaiCompat,
64
66
  sglang,
65
67
  vllm,
@@ -310,6 +310,7 @@ interface LlamaCppProps {
310
310
  default_generation_settings?: { n_ctx?: unknown; n_predict?: unknown };
311
311
  modalities?: { vision?: unknown };
312
312
  build_info?: unknown;
313
+ total_slots?: unknown;
313
314
  }
314
315
 
315
316
  export interface LlamaCppPropsEnrichment {
@@ -499,6 +500,9 @@ export async function probeLlamaCppModelStatus(
499
500
  if (flags.reasoning === true || flags.reasoningBudget !== undefined) caps.reasoning = true;
500
501
  if (flags.mmproj) caps.vision = true;
501
502
  if (flags.jinja === true) caps.tools = true;
503
+ if (flags.parallel !== undefined && Number.isInteger(flags.parallel) && flags.parallel > 0) {
504
+ caps.parallelSlots = flags.parallel;
505
+ }
502
506
  const enrichment: LlamaCppStatusEnrichment = { modelId: selected.id, serverFlags: flags };
503
507
  if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
504
508
  const notes = statusNotes(selected.id, selected.status);
@@ -532,13 +536,28 @@ export async function detectModelMismatch(
532
536
  return `wire model id ${expected} does not match server's loaded model ${loaded}; llama.cpp serves a single fixed model`;
533
537
  }
534
538
 
535
- export async function probeLlamaCppProps(base: string, ctx: ProbeContext): Promise<LlamaCppPropsEnrichment> {
536
- const opts = { url: `${base}/props`, timeoutMs: ctx.httpTimeoutMs } as const;
537
- const result = await (ctx.signal
538
- ? probeJson<LlamaCppProps>({ ...opts, signal: ctx.signal })
539
- : probeJson<LlamaCppProps>(opts));
540
- if (!result.ok || !result.data) return {};
541
- const data = result.data;
539
+ export async function probeLlamaCppProps(
540
+ base: string,
541
+ ctx: ProbeContext,
542
+ modelId?: string,
543
+ ): Promise<LlamaCppPropsEnrichment> {
544
+ const request = async (url: string): Promise<LlamaCppProps | null> => {
545
+ const opts = { url, timeoutMs: ctx.httpTimeoutMs } as const;
546
+ const result = await (ctx.signal
547
+ ? probeJson<LlamaCppProps>({ ...opts, signal: ctx.signal })
548
+ : probeJson<LlamaCppProps>(opts));
549
+ return result.ok && result.data ? result.data : null;
550
+ };
551
+ const router = await request(`${base}/props`);
552
+ if (router === null) return {};
553
+ // A llama.cpp router reports its own role at /props and the selected
554
+ // worker's request slots at /props?model=<id>. A fixed-model server answers
555
+ // the first request directly, so the second GET is only made when needed.
556
+ const selected =
557
+ typeof router.total_slots === "number" || !modelId
558
+ ? null
559
+ : await request(`${base}/props?model=${encodeURIComponent(modelId)}`);
560
+ const data = selected ?? router;
542
561
  const enrichment: LlamaCppPropsEnrichment = {};
543
562
  const caps: Partial<CapabilityFlags> = {};
544
563
  const nCtx = data.default_generation_settings?.n_ctx;
@@ -547,9 +566,12 @@ export async function probeLlamaCppProps(base: string, ctx: ProbeContext): Promi
547
566
  if (typeof nPredict === "number" && nPredict > 0) caps.maxTokens = nPredict;
548
567
  const vision = data.modalities?.vision;
549
568
  if (typeof vision === "boolean") caps.vision = vision;
569
+ const totalSlots = data.total_slots;
570
+ if (typeof totalSlots === "number" && Number.isInteger(totalSlots) && totalSlots > 0) caps.parallelSlots = totalSlots;
550
571
  if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
551
- if (typeof data.build_info === "string" && data.build_info.length > 0) {
552
- enrichment.serverVersion = data.build_info;
572
+ const buildInfo = data.build_info ?? router.build_info;
573
+ if (typeof buildInfo === "string" && buildInfo.length > 0) {
574
+ enrichment.serverVersion = buildInfo;
553
575
  }
554
576
  return enrichment;
555
577
  }
@@ -52,7 +52,7 @@ const llamacppAnthropicRuntime: RuntimeDescriptor = {
52
52
  };
53
53
  const head = await (ctx.signal ? probeHttp({ ...headOpts, signal: ctx.signal }) : probeHttp(headOpts));
54
54
  if (!head.ok) return head;
55
- const props = await probeLlamaCppProps(base, ctx);
55
+ const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
56
56
  const enriched: ProbeResult = { ...head };
57
57
  if (props.discoveredCapabilities) enriched.discoveredCapabilities = props.discoveredCapabilities;
58
58
  if (props.serverVersion) enriched.serverVersion = props.serverVersion;
@@ -157,7 +157,7 @@ const llamacppCompletionRuntime: RuntimeDescriptor = {
157
157
  const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs } as const;
158
158
  const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
159
159
  if (!health.ok) return health;
160
- const props = await probeLlamaCppProps(base, ctx);
160
+ const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
161
161
  const status = await probeLlamaCppModelStatus(base, target, ctx);
162
162
  const enriched: ProbeResult = { ...health };
163
163
  const discoveredCapabilities = {
@@ -101,7 +101,7 @@ const llamacppEmbedRuntime: RuntimeDescriptor = {
101
101
  if (!probeResponse.ok) {
102
102
  return { ok: false, error: `/embedding not available: HTTP ${probeResponse.status}` };
103
103
  }
104
- const props = await probeLlamaCppProps(base, ctx);
104
+ const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
105
105
  const result: ProbeResult = { ok: true };
106
106
  if (health.latencyMs !== undefined) result.latencyMs = health.latencyMs;
107
107
  if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
@@ -60,7 +60,7 @@ const llamacppRerankRuntime: RuntimeDescriptor = {
60
60
  if (!(probeResponse.status === 200 || probeResponse.status === 202)) {
61
61
  return { ok: false, error: `/reranking not available: HTTP ${probeResponse.status}` };
62
62
  }
63
- const props = await probeLlamaCppProps(base, ctx);
63
+ const props = await probeLlamaCppProps(base, ctx, target.defaultModel);
64
64
  const result: ProbeResult = { ok: true };
65
65
  if (health.latencyMs !== undefined) result.latencyMs = health.latencyMs;
66
66
  if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
@@ -55,8 +55,8 @@ const llamacppRuntime: RuntimeDescriptor = {
55
55
  const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs } as const;
56
56
  const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
57
57
  if (!health.ok) return health;
58
- const props = await probeLlamaCppProps(base, ctx);
59
58
  const status = await probeLlamaCppModelStatus(base, target, ctx);
59
+ const props = await probeLlamaCppProps(base, ctx, status.modelId ?? target.defaultModel);
60
60
  const catalog = await probeOpenAIModelCatalog(base, ctx);
61
61
  const result: ProbeResult = { ok: true };
62
62
  if (catalog.models.length > 0) result.models = catalog.models;
@@ -66,6 +66,9 @@ const llamacppRuntime: RuntimeDescriptor = {
66
66
  const discoveredCapabilities = {
67
67
  ...(props.discoveredCapabilities ?? {}),
68
68
  ...(status.discoveredCapabilities ?? {}),
69
+ ...(props.discoveredCapabilities?.parallelSlots !== undefined
70
+ ? { parallelSlots: props.discoveredCapabilities.parallelSlots }
71
+ : {}),
69
72
  };
70
73
  if (Object.keys(discoveredCapabilities).length > 0) {
71
74
  result.discoveredCapabilities = discoveredCapabilities;