@iowarp/clio-coder 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. package/CHANGELOG.md +407 -0
  2. package/CODE_OF_CONDUCT.md +21 -0
  3. package/CONTRIBUTING.md +224 -0
  4. package/LICENSE +202 -0
  5. package/NOTICE +9 -0
  6. package/README.md +798 -0
  7. package/SECURITY.md +72 -0
  8. package/assets/clio-coder-logo-128.webp +0 -0
  9. package/damage-control-rules.yaml +419 -0
  10. package/dist/acp-UMLFVA3F.js +92 -0
  11. package/dist/agents-Q4MYPMUW.js +91 -0
  12. package/dist/auth-O6HYIJ6J.js +521 -0
  13. package/dist/chunk-262G75JS.js +35 -0
  14. package/dist/chunk-26BZQOAD.js +1281 -0
  15. package/dist/chunk-2J63S4SF.js +508 -0
  16. package/dist/chunk-3DANZDGR.js +717 -0
  17. package/dist/chunk-4UQA7NCT.js +29 -0
  18. package/dist/chunk-527KG6XR.js +497 -0
  19. package/dist/chunk-5LDRNKX2.js +1063 -0
  20. package/dist/chunk-5N2FG33Q.js +25 -0
  21. package/dist/chunk-67MTHP2E.js +135 -0
  22. package/dist/chunk-6CWDTGUC.js +20 -0
  23. package/dist/chunk-7BHLZB3A.js +2115 -0
  24. package/dist/chunk-7RBKDI66.js +348 -0
  25. package/dist/chunk-AMFR5YA3.js +541 -0
  26. package/dist/chunk-BBUH4VAA.js +1224 -0
  27. package/dist/chunk-BYEU76JP.js +899 -0
  28. package/dist/chunk-CLJ5HLUD.js +458 -0
  29. package/dist/chunk-D5YD55AR.js +116 -0
  30. package/dist/chunk-DXQNI4PC.js +61 -0
  31. package/dist/chunk-E3NYWENM.js +1004 -0
  32. package/dist/chunk-GNGDQYDU.js +34688 -0
  33. package/dist/chunk-GOTUR54M.js +9 -0
  34. package/dist/chunk-HBU5MTAM.js +41 -0
  35. package/dist/chunk-HMYNFFY4.js +28 -0
  36. package/dist/chunk-JPOWPFCU.js +1010 -0
  37. package/dist/chunk-JWHCJDCI.js +1215 -0
  38. package/dist/chunk-KBR4MZZR.js +41 -0
  39. package/dist/chunk-KKKPTZLM.js +93 -0
  40. package/dist/chunk-ME6DNWIU.js +66 -0
  41. package/dist/chunk-NI4DEJMC.js +88 -0
  42. package/dist/chunk-O4EJEDHO.js +659 -0
  43. package/dist/chunk-PIDUD6M2.js +31 -0
  44. package/dist/chunk-PS4PFJQP.js +29459 -0
  45. package/dist/chunk-QV47YRF4.js +48 -0
  46. package/dist/chunk-RQDWMVRB.js +279 -0
  47. package/dist/chunk-TFSSEXL6.js +136 -0
  48. package/dist/chunk-TKHQ4DGZ.js +8290 -0
  49. package/dist/chunk-TPOCL34A.js +2876 -0
  50. package/dist/chunk-UGYAX5YI.js +565 -0
  51. package/dist/chunk-UHTSULZS.js +461 -0
  52. package/dist/chunk-UU3R62TT.js +128 -0
  53. package/dist/chunk-UWIJNAOB.js +3906 -0
  54. package/dist/chunk-VOO7NYPP.js +914 -0
  55. package/dist/chunk-VPAWTYLY.js +117 -0
  56. package/dist/chunk-WD6AJM35.js +1216 -0
  57. package/dist/chunk-X3BR7HWV.js +115 -0
  58. package/dist/chunk-X3NE4WVW.js +120 -0
  59. package/dist/chunk-XNISANGE.js +1395 -0
  60. package/dist/chunk-XV4ZJ6ZM.js +3177 -0
  61. package/dist/cli/index.js +236 -0
  62. package/dist/clio-KIQ5SNDS.js +53 -0
  63. package/dist/components-JVHMUBEB.js +653 -0
  64. package/dist/config-ZFCDBMDC.js +372 -0
  65. package/dist/configure-G4E3A2PG.js +27 -0
  66. package/dist/context-CDXTP2MP.js +293 -0
  67. package/dist/context-E3KIFVXI.js +185 -0
  68. package/dist/context-clear-3F4PLXOS.js +102 -0
  69. package/dist/context-index-Q7YSYTR3.js +106 -0
  70. package/dist/docs-YIETIWZI.js +280 -0
  71. package/dist/doctor-M5HJJZOL.js +61 -0
  72. package/dist/domains/agents/builtins/architect.md +33 -0
  73. package/dist/domains/agents/builtins/coder.md +31 -0
  74. package/dist/domains/agents/builtins/context-bootstrap.md +38 -0
  75. package/dist/domains/agents/builtins/debugger.md +30 -0
  76. package/dist/domains/agents/builtins/documenter.md +31 -0
  77. package/dist/domains/agents/builtins/git-master.md +30 -0
  78. package/dist/domains/agents/builtins/provenance.md +30 -0
  79. package/dist/domains/agents/builtins/researcher.md +71 -0
  80. package/dist/domains/agents/builtins/scout.md +42 -0
  81. package/dist/domains/agents/builtins/tester.md +31 -0
  82. package/dist/domains/agents/builtins/verifier.md +30 -0
  83. package/dist/domains/agents/builtins/wiki-writer.md +41 -0
  84. package/dist/eval-B3KZZESM.js +2674 -0
  85. package/dist/evidence-V67CHM35.js +233 -0
  86. package/dist/evolve-YDZSUQYA.js +518 -0
  87. package/dist/extensions-SRG7XCAH.js +207 -0
  88. package/dist/fleet-CA2CRTVG.js +760 -0
  89. package/dist/fleet-preflight-CLIAX7YR.js +21 -0
  90. package/dist/init-2OZDJE2D.js +227 -0
  91. package/dist/memory-3PIQQAKX.js +207 -0
  92. package/dist/models-DY35XI7Y.js +237 -0
  93. package/dist/paths-5OMXW7Z4.js +57 -0
  94. package/dist/preload-KZVHET2B.js +11 -0
  95. package/dist/reset-PIFYNOS3.js +216 -0
  96. package/dist/run-3VSPP24F.js +735 -0
  97. package/dist/share-D36RQCXM.js +241 -0
  98. package/dist/skills-F2MRLELY.js +445 -0
  99. package/dist/skills-eval-E2ZTW4PL.js +932 -0
  100. package/dist/targets-DZMEZAH4.js +977 -0
  101. package/dist/trace-7NYCUI2J.js +250 -0
  102. package/dist/uninstall-AD3JWHBB.js +322 -0
  103. package/dist/upgrade-WYYBKGDY.js +301 -0
  104. package/dist/usage-ULIDAGFF.js +755 -0
  105. package/dist/version-ROZ6CZKH.js +16 -0
  106. package/dist/wiki-generate-PKFIX6OB.js +377 -0
  107. package/dist/worker/entry.js +1739 -0
  108. package/docs/README.md +93 -0
  109. package/docs/acp.md +120 -0
  110. package/docs/alcf-provider.md +72 -0
  111. package/docs/architecture.md +172 -0
  112. package/docs/artifact-versions.md +54 -0
  113. package/docs/built-in-agents.md +265 -0
  114. package/docs/capacity-and-scheduling.md +97 -0
  115. package/docs/commands-and-modes.md +554 -0
  116. package/docs/config-knobs-audit.md +115 -0
  117. package/docs/configuration-and-targets.md +812 -0
  118. package/docs/context-engine.md +236 -0
  119. package/docs/dispatch-architecture-rationale.md +126 -0
  120. package/docs/documentation-coverage.md +46 -0
  121. package/docs/documentation-guide.md +166 -0
  122. package/docs/environment-variables.md +105 -0
  123. package/docs/eval-runner.md +205 -0
  124. package/docs/evals-internal.md +298 -0
  125. package/docs/evidence-and-memory.md +243 -0
  126. package/docs/evolution.md +143 -0
  127. package/docs/exit-codes-and-output.md +74 -0
  128. package/docs/extensions-and-sharing.md +306 -0
  129. package/docs/fleet-demo-runbook.md +179 -0
  130. package/docs/fleet-dispatch.md +591 -0
  131. package/docs/glossary.md +75 -0
  132. package/docs/html/agents_blueprint.html +936 -0
  133. package/docs/html/alcf_blueprint.html +324 -0
  134. package/docs/html/architecture_blueprint.html +850 -0
  135. package/docs/html/commands_blueprint.html +794 -0
  136. package/docs/html/config_knobs_audit_blueprint.html +178 -0
  137. package/docs/html/configuration_blueprint.html +1080 -0
  138. package/docs/html/context_blueprint.html +603 -0
  139. package/docs/html/documentation_blueprint.html +832 -0
  140. package/docs/html/environment_blueprint.html +404 -0
  141. package/docs/html/eval_blueprint.html +743 -0
  142. package/docs/html/evals_internal_blueprint.html +190 -0
  143. package/docs/html/evolution_blueprint.html +674 -0
  144. package/docs/html/extensions_blueprint.html +2065 -0
  145. package/docs/html/fleet_dispatch_blueprint.html +286 -0
  146. package/docs/html/index.html +919 -0
  147. package/docs/html/lifecycle_blueprint.html +723 -0
  148. package/docs/html/memory_blueprint.html +699 -0
  149. package/docs/html/middleware_blueprint.html +664 -0
  150. package/docs/html/models_blueprint.html +2366 -0
  151. package/docs/html/observability_blueprint.html +683 -0
  152. package/docs/html/provider_adapter_blueprint.html +245 -0
  153. package/docs/html/safety_blueprint.html +1386 -0
  154. package/docs/html/shared.css +571 -0
  155. package/docs/html/shared.js +143 -0
  156. package/docs/html/skills_blueprint.html +671 -0
  157. package/docs/html/soak_blueprint.html +182 -0
  158. package/docs/html/tool_usage_blueprint.html +350 -0
  159. package/docs/html/tools_blueprint.html +2249 -0
  160. package/docs/html/trace_blueprint.html +235 -0
  161. package/docs/html/tui_design_blueprint.html +314 -0
  162. package/docs/html/validation_blueprint.html +961 -0
  163. package/docs/html/worker_dispatch_blueprint.html +231 -0
  164. package/docs/installation-and-lifecycle.md +308 -0
  165. package/docs/middleware-and-components.md +148 -0
  166. package/docs/model-catalog.md +189 -0
  167. package/docs/observability.md +233 -0
  168. package/docs/proactive-memory.md +452 -0
  169. package/docs/prompt-envelope-and-tools.md +142 -0
  170. package/docs/provider-adapter-cookbook.md +148 -0
  171. package/docs/release-cut-checklist.md +138 -0
  172. package/docs/safety-model.md +357 -0
  173. package/docs/scientific-validation.md +105 -0
  174. package/docs/session-lifecycle.md +156 -0
  175. package/docs/skills-marketplace.md +46 -0
  176. package/docs/tool-usage.md +527 -0
  177. package/docs/trace-store.md +132 -0
  178. package/docs/troubleshooting.md +33 -0
  179. package/docs/tui-design.md +239 -0
  180. package/docs/worker-dispatch-mechanics.md +242 -0
  181. package/package.json +132 -0
  182. package/skills/README.md +408 -0
  183. package/skills/git/commit-crafting/SKILL.md +79 -0
  184. package/skills/git/commit-crafting/evals.md +92 -0
  185. package/skills/git/create-pr/SKILL.md +116 -0
  186. package/skills/git/create-pr/evals.md +114 -0
  187. package/skills/git/investigate-issue/SKILL.md +139 -0
  188. package/skills/git/investigate-issue/evals.md +94 -0
  189. package/skills/git/resolve-merge-conflicts/SKILL.md +96 -0
  190. package/skills/git/resolve-merge-conflicts/evals.md +58 -0
  191. package/skills/git/review-changes/SKILL.md +103 -0
  192. package/skills/git/review-changes/evals.md +85 -0
  193. package/skills/git/worktree-create/SKILL.md +92 -0
  194. package/skills/git/worktree-create/evals.md +97 -0
  195. package/skills/git/worktree-create/references/worktree-setup.md +66 -0
  196. package/skills/git/worktree-merge/SKILL.md +95 -0
  197. package/skills/git/worktree-merge/evals.md +114 -0
  198. package/skills/skill-marketplace.json +261 -0
  199. package/skills/workflow/cut-it/SKILL.md +86 -0
  200. package/skills/workflow/cut-it/evals.md +42 -0
  201. package/src/domains/agents/builtins/architect.md +33 -0
  202. package/src/domains/agents/builtins/coder.md +31 -0
  203. package/src/domains/agents/builtins/context-bootstrap.md +38 -0
  204. package/src/domains/agents/builtins/debugger.md +30 -0
  205. package/src/domains/agents/builtins/documenter.md +31 -0
  206. package/src/domains/agents/builtins/git-master.md +30 -0
  207. package/src/domains/agents/builtins/provenance.md +30 -0
  208. package/src/domains/agents/builtins/researcher.md +71 -0
  209. package/src/domains/agents/builtins/scout.md +42 -0
  210. package/src/domains/agents/builtins/tester.md +31 -0
  211. package/src/domains/agents/builtins/verifier.md +30 -0
  212. package/src/domains/agents/builtins/wiki-writer.md +41 -0
  213. package/src/domains/agents/fleets/build-review.md +34 -0
  214. package/src/domains/agents/fleets/build-test.md +35 -0
  215. package/src/domains/agents/fleets/sdlc.md +86 -0
  216. package/src/domains/prompts/fragments/identity/clio-worker.md +11 -0
  217. package/src/domains/prompts/fragments/identity/clio.md +26 -0
  218. package/src/domains/prompts/fragments/operating/contract.md +64 -0
  219. package/src/domains/prompts/fragments/safety/auto-edit.md +14 -0
  220. package/src/domains/prompts/fragments/safety/full-auto.md +14 -0
  221. package/src/domains/prompts/fragments/safety/read-only.md +13 -0
  222. package/src/domains/prompts/fragments/safety/suggest.md +13 -0
  223. package/src/domains/prompts/fragments/wiki/page.md +75 -0
  224. package/src/domains/prompts/fragments/wiki/plan.md +48 -0
  225. package/src/domains/providers/models/cloud-models/alcf.yaml +40 -0
  226. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +993 -0
@@ -0,0 +1,993 @@
1
+ # Sources:
2
+ # https://huggingface.co/Qwen/Qwen3.6-35B-A3B
3
+ # https://huggingface.co/Qwen/Qwen3.6-27B
4
+ # https://huggingface.co/google/gemma-4-26B-A4B
5
+ # https://huggingface.co/LilaRest/gemma-4-31B-it-NVFP4-turbo
6
+ # https://huggingface.co/Jackrong/Gemopus-4-31B-it-GGUF
7
+ # https://huggingface.co/Jackrong/Qwopus3.6-27B-v1-preview-GGUF
8
+ # https://huggingface.co/Jackrong/Qwopus3.6-35B-A3B-Coder-MTP-GGUF
9
+ # https://huggingface.co/nvidia/Nemotron-Cascade-2-30B-A3B
10
+ # https://huggingface.co/unsloth/NVIDIA-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-GGUF
11
+ # https://huggingface.co/mradermacher/Qwen3.5-35B-A3B-Claude-4.6-Opus-Reasoning-Distilled-i1-GGUF
12
+ # https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-GGUF
13
+ #
14
+ # This knowledge base is intentionally narrow. It describes only the local
15
+ # models curated for Clio's local coding workflows and leaves cloud GPT models
16
+ # to pi-ai's native OpenAI and openai-codex catalogs.
17
+ #
18
+ # Thinking semantics for these local families: the chain-of-thought is emitted
19
+ # by the model's chat template (Qwen-style <think> blocks, Gemma 4 thinking
20
+ # template, Nemotron reasoning template). LM Studio's SDK and llama.cpp's
21
+ # OpenAI-compatible surface do not expose a reliable "no thinking" flag, so
22
+ # Clio cannot disable template-driven reasoning at the API layer. With
23
+ # `--thinking off` Clio simply does not request reasoning; if the model emits
24
+ # it anyway, the chain-of-thought is captured into a ThinkingContent block,
25
+ # counted as reasoning tokens (separate from output tokens) on the receipt
26
+ # and TUI footer, and surfaced to the user truthfully rather than hidden.
27
+ - family: openai-gpt-oss
28
+ matchPatterns:
29
+ - openai/gpt-oss
30
+ - gpt-oss
31
+ - gpt-oss-20b
32
+ - gpt-oss-120b
33
+ capabilities:
34
+ chat: true
35
+ tools: true
36
+ toolCallFormat: openai
37
+ reasoning: true
38
+ thinkingFormat: harmony
39
+ structuredOutputs: json-schema
40
+ vision: false
41
+ audio: false
42
+ embeddings: false
43
+ rerank: false
44
+ fim: false
45
+ contextWindow: 131072
46
+ maxTokens: 32768
47
+ quirks:
48
+ sampling:
49
+ thinking:
50
+ temperature: 1
51
+ topP: 1
52
+ maxTokens: 32768
53
+ instruct:
54
+ temperature: 1
55
+ topP: 1
56
+ maxTokens: 8192
57
+ runtimePreference:
58
+ llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high."
59
+ openaiCompat: "Preferred for Harmony reasoning-effort control when the local gateway exposes chat_template_kwargs."
60
+ thinking:
61
+ mechanism: effort-levels
62
+ effortByLevel:
63
+ low: low
64
+ medium: medium
65
+ high: high
66
+ guidance: |
67
+ GPT-OSS uses OpenAI Harmony formatting and supports low, medium, and
68
+ high reasoning effort. Clio coerces off/minimal to low, captures
69
+ non-final Harmony channels as ThinkingContent, and strips Harmony
70
+ control tokens from visible output.
71
+
72
+ - family: agenticqwen-30b-a3b-i1
73
+ matchPatterns:
74
+ - agenticqwen-30b-a3b-i1
75
+ - agenticqwen-30b-a3b-i1-q4_k_m
76
+ - agenticqwen
77
+ capabilities:
78
+ chat: true
79
+ tools: true
80
+ toolCallFormat: qwen
81
+ reasoning: true
82
+ thinkingFormat: qwen-chat-template
83
+ structuredOutputs: json-schema
84
+ vision: false
85
+ audio: false
86
+ embeddings: false
87
+ rerank: false
88
+ fim: false
89
+ contextWindow: 262144
90
+ maxTokens: 65536
91
+ quirks:
92
+ sampling:
93
+ thinking:
94
+ temperature: 0.5
95
+ topP: 0.9
96
+ topK: 20
97
+ maxTokens: 16384
98
+ reasoningBudget: 4096
99
+ instruct:
100
+ temperature: 0.5
101
+ topP: 0.9
102
+ topK: 20
103
+ maxTokens: 4096
104
+ gpuTiers:
105
+ "32gb": "Default 32 GB local llama.cpp orchestrator profile; benchmarked with roughly 1 GiB VRAM headroom on a 32 GiB class card."
106
+ runtimePreference:
107
+ llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile uses ctx=262144, parallel=4, batch=2048, ubatch=512, and q8 KV cache."
108
+ openaiCompat: "Use only when a gateway fronts the same llama.cpp chat-completions surface."
109
+ llamaCpp:
110
+ ctxSize: 262144
111
+ cacheTypeK: q8_0
112
+ cacheTypeV: q8_0
113
+ parallel: 4
114
+ batchSize: 2048
115
+ ubatchSize: 512
116
+ thinking:
117
+ mechanism: budget-tokens
118
+ budgetByLevel:
119
+ low: 1024
120
+ medium: 4096
121
+ high: 16384
122
+ guidance: |
123
+ AgenticQwen is the tuned local default for Clio source work on
124
+ a local llama.cpp server. Thinking emits through the qwen chat template;
125
+ keep high reasoning for main-agent work and preserve visible thinking
126
+ blocks in receipts.
127
+ serving: "Use the llama.cpp settings listed above for a 262144-token local profile."
128
+
129
+ - family: nemotron-3-nano-omni-30b-a3b-reasoning
130
+ matchPatterns:
131
+ - nvidia-nemotron-3-nano-omni-30b-a3b-reasoning
132
+ - nemotron-3-nano-omni-30b-a3b-reasoning-ud-q4_k_m
133
+ - nemotron-3-nano-omni-30b-a3b-reasoning
134
+ - nemotron-3-nano-omni
135
+ - nemotron-nano-omni
136
+ capabilities:
137
+ chat: true
138
+ tools: true
139
+ toolCallFormat: qwen
140
+ reasoning: true
141
+ thinkingFormat: qwen-chat-template
142
+ structuredOutputs: json-schema
143
+ vision: true
144
+ audio: true
145
+ embeddings: false
146
+ rerank: false
147
+ fim: false
148
+ contextWindow: 1048576
149
+ maxTokens: 131072
150
+ quirks:
151
+ sampling:
152
+ thinking:
153
+ temperature: 0.6
154
+ topP: 0.95
155
+ topK: 20
156
+ maxTokens: 20480
157
+ reasoningBudget: 16384
158
+ gracePeriod: 1024
159
+ instruct:
160
+ temperature: 0.2
161
+ topK: 1
162
+ maxTokens: 1024
163
+ gpuTiers:
164
+ "24gb": "Use a 4-bit quant with reduced load context for evaluation. Prefer the 32GB tier for sustained Clio writes."
165
+ "32gb": "Primary strong local model for multimodal main-agent work. Keep openai-compat available for tool-call fallback."
166
+ runtimePreference:
167
+ lmstudioNative: "Use only when the target's server lifecycle is Clio-managed or the model is already loaded at the requested context."
168
+ openaiCompat: "Preferred fallback for LM Studio gateway tool-call extraction."
169
+ llamaCpp: "OpenAI-compatible chat completions with --jinja."
170
+ llamaCpp:
171
+ ctxSize: 819200
172
+ cacheTypeK: q8_0
173
+ cacheTypeV: q8_0
174
+ flashAttn: true
175
+ nGpuLayers: 99
176
+ parallel: 4
177
+ thinking:
178
+ mechanism: budget-tokens
179
+ budgetByLevel:
180
+ low: 1024
181
+ medium: 4096
182
+ high: 16384
183
+ guidance: |
184
+ Reasoning is template-driven; thinking blocks emit through the qwen
185
+ chat template. Preserve thinking across coding-task turns when the
186
+ chain matters; otherwise let the budget cap output growth.
187
+ serving: "Use qwen3_coder tool parsing when served through OpenAI-compatible HTTP. LM Studio native uses SDK rawTools, but openai-compat remains the fallback for template-specific tool extraction issues."
188
+
189
+ - family: qwen3.6-35b-a3b
190
+ matchPatterns:
191
+ - qwen3.6-35b-a3b
192
+ - qwen3_6-35b-a3b
193
+ - qwen3-6-35b-a3b
194
+ - qwen3.6-35b
195
+ - qwen3.6-35b-a3b-ud
196
+ - qwen3.6-35b-a3b-ud-q4_k_xl
197
+ capabilities:
198
+ chat: true
199
+ tools: true
200
+ toolCallFormat: qwen
201
+ reasoning: true
202
+ thinkingFormat: qwen-chat-template
203
+ structuredOutputs: json-schema
204
+ vision: true
205
+ audio: false
206
+ embeddings: false
207
+ rerank: false
208
+ fim: false
209
+ contextWindow: 262144
210
+ maxTokens: 65536
211
+ quirks:
212
+ sampling:
213
+ thinking:
214
+ temperature: 0.6
215
+ topP: 0.95
216
+ topK: 20
217
+ maxTokens: 16384
218
+ reasoningBudget: 4096
219
+ instruct:
220
+ temperature: 0.7
221
+ topP: 0.80
222
+ topK: 20
223
+ minP: 0.0
224
+ gpuTiers:
225
+ "32gb": "Main-agent candidate on 32GB targets when Nemotron Omni is unavailable or tool behavior is better for text-only work."
226
+ runtimePreference:
227
+ lmstudioNative: "Usable for text and vision when loaded in LM Studio. Keep openai-compat available for tool fallback."
228
+ openaiCompat: "Preferred local gateway path for tool-call stress tests."
229
+ llamaCpp: "OpenAI-compatible chat completions with --jinja. Local llama.cpp profile emits reasoning_content without a separate --reasoning flag."
230
+ llamaCpp:
231
+ ctxSize: 262144
232
+ cacheTypeK: q8_0
233
+ cacheTypeV: q8_0
234
+ flashAttn: true
235
+ nGpuLayers: 99
236
+ thinking:
237
+ mechanism: budget-tokens
238
+ budgetByLevel:
239
+ low: 1024
240
+ medium: 4096
241
+ high: 16384
242
+ guidance: |
243
+ Qwen chat template emits thinking blocks. The card recommends at least
244
+ 128K context for thinking quality. Preserve thinking blocks across
245
+ coding-task turns so iterative reasoning chains do not restart cold.
246
+ serving: "Official card recommends at least 128K context for thinking quality. Use 262K when memory allows."
247
+
248
+ - family: ornith-1.0-35b
249
+ matchPatterns:
250
+ - ornith-1.0-35b
251
+ - ornith-1.0-35b-q4_k_m
252
+ - ornith-1.0-35b-gguf
253
+ - deepreinforce-ai/ornith-1.0-35b-gguf
254
+ - ornith
255
+ capabilities:
256
+ chat: true
257
+ tools: true
258
+ toolCallFormat: qwen
259
+ reasoning: true
260
+ thinkingFormat: qwen-chat-template
261
+ structuredOutputs: json-schema
262
+ vision: false
263
+ audio: false
264
+ embeddings: false
265
+ rerank: false
266
+ fim: false
267
+ contextWindow: 262144
268
+ maxTokens: 65536
269
+ quirks:
270
+ kvCache:
271
+ kQuant: q8_0
272
+ vQuant: q8_0
273
+ sampling:
274
+ thinking:
275
+ temperature: 0.6
276
+ topP: 0.95
277
+ topK: 20
278
+ maxTokens: 16384
279
+ reasoningBudget: 4096
280
+ instruct:
281
+ temperature: 0.6
282
+ topP: 0.95
283
+ topK: 20
284
+ repeatPenalty: 1.05
285
+ maxTokens: 4096
286
+ gpuTiers:
287
+ "32gb": "32 GB local profile with Q4_K_M, ctx=262144, q8 KV, flash attention, and parallel=1."
288
+ runtimePreference:
289
+ llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. Recommended local profile uses ctx=262144, parallel=1, batch=2048, ubatch=512, flash attention, and q8 KV cache."
290
+ openaiCompat: "Preferred when a gateway fronts the same llama.cpp chat-completions surface with qwen reasoning/tool parsing."
291
+ llamaCpp:
292
+ ctxSize: 262144
293
+ cacheTypeK: q8_0
294
+ cacheTypeV: q8_0
295
+ flashAttn: true
296
+ nGpuLayers: 99
297
+ parallel: 1
298
+ batchSize: 2048
299
+ ubatchSize: 512
300
+ thinking:
301
+ mechanism: always-on
302
+ guidance: |
303
+ Ornith emits a Qwen-style <think> block by default. The local llama.cpp
304
+ router is configured with --reasoning on, so the OpenAI-compatible
305
+ response separates this trace into reasoning_content while content
306
+ contains the final answer.
307
+ serving: "Ornith-1.0-35B is an agentic coding model; the card recommends 262K context, qwen3 reasoning parsing, qwen3_coder/qwen3_xml tool parsing, and sampler temperature=0.6, top_p=0.95, top_k=20."
308
+
309
+ - family: qwen3.6-27b
310
+ matchPatterns:
311
+ - qwen3.6-27b
312
+ - qwen3_6-27b
313
+ - qwen3-6-27b
314
+ capabilities:
315
+ chat: true
316
+ tools: true
317
+ toolCallFormat: qwen
318
+ reasoning: true
319
+ thinkingFormat: qwen-chat-template
320
+ structuredOutputs: json-schema
321
+ vision: true
322
+ audio: false
323
+ embeddings: false
324
+ rerank: false
325
+ fim: false
326
+ contextWindow: 262144
327
+ maxTokens: 32768
328
+ quirks:
329
+ sampling:
330
+ thinking:
331
+ temperature: 0.6
332
+ topP: 0.95
333
+ topK: 20
334
+ minP: 0.0
335
+ presencePenalty: 0.0
336
+ repetitionPenalty: 1.0
337
+ maxTokens: 32768
338
+ instruct:
339
+ temperature: 0.7
340
+ topP: 0.80
341
+ topK: 20
342
+ minP: 0.0
343
+ presencePenalty: 1.5
344
+ repetitionPenalty: 1.0
345
+ maxTokens: 32768
346
+ gpuTiers:
347
+ "32gb": "Fits at f16 KV with parallel=1 and ctx=262144 (Q4_K_XL weights ~19.5 GB plus ~32 GB KV)."
348
+ runtimePreference:
349
+ lmstudioNative: "Primary path; standard <think> tagging works correctly with the SDK classifier."
350
+ openaiCompat: "Verified parity for tool calls; reasoning_content surfaces and completion_tokens_details.reasoning_tokens is reported."
351
+ thinking:
352
+ mechanism: budget-tokens
353
+ budgetByLevel:
354
+ low: 1024
355
+ medium: 4096
356
+ high: 16384
357
+ guidance: |
358
+ Qwen chat template drives thinking via standard <think> blocks.
359
+ Coding sampler is the precise variant. Preserve thinking blocks across
360
+ turns when the workflow is iterative; the budget caps runaway chains.
361
+ serving: "Official Qwen card recommends max output 32768 tokens for standard queries (81920 only for math/code competitions). Coding sampler is the precise variant; preserve thinking blocks across turns for iterative workflows."
362
+
363
+ - family: gemma-4-31b-it-nvfp4-turbo
364
+ matchPatterns:
365
+ - gemma-4-31b-it-nvfp4-turbo
366
+ - gemma-4-31b-it-nvfp4
367
+ - gemma-4-31b-it-turbo
368
+ - gemma-4-31b-it
369
+ - gemma-4-31b
370
+ capabilities:
371
+ chat: true
372
+ tools: true
373
+ toolCallFormat: openai
374
+ reasoning: true
375
+ thinkingFormat: qwen-chat-template
376
+ structuredOutputs: json-schema
377
+ vision: true
378
+ audio: false
379
+ embeddings: false
380
+ rerank: false
381
+ fim: false
382
+ contextWindow: 122880
383
+ maxTokens: 32768
384
+ quirks:
385
+ kvCache:
386
+ kQuant: q8_0
387
+ vQuant: q8_0
388
+ sampling:
389
+ thinking:
390
+ temperature: 1.0
391
+ topP: 0.95
392
+ topK: 64
393
+ minP: 0.0
394
+ presencePenalty: 0.0
395
+ repetitionPenalty: 1.0
396
+ maxTokens: 32768
397
+ instruct:
398
+ temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
399
+ topP: 0.95
400
+ topK: 64
401
+ minP: 0.0
402
+ maxTokens: 32768
403
+ chatTemplate: "gemma-channel"
404
+ leakageNote: |
405
+ LM Studio's gemma4 chat template emits chain-of-thought as plain text
406
+ bracketed by literal '<|channel>thought\n' and '<channel|>' markers
407
+ rather than reasoningType=reasoning fragments. Clio's lmstudio-native
408
+ adapter buffers and re-classifies these regions into ThinkingContent
409
+ so the markers do not leak into the visible TUI text and reasoning
410
+ tokens are accounted via the chars/4 estimator.
411
+ gpuTiers:
412
+ "48gb": "Fits at f16 KV when parallel=1 and ctx=122880; drop to q8_0 KV at parallel=4."
413
+ "80gb": "Comfortable at f16 KV with parallel=1 and ctx=122880."
414
+ runtimePreference:
415
+ lmstudioNative: "Verified path. Pin parallel=1 in the LM Studio per-model preset to give a single agent the full 122880 context."
416
+ openaiCompat: "Fallback path; the openai-compat surface does not expose the channel-marker classifier so visible text shows the raw markers."
417
+ thinking:
418
+ mechanism: on-off
419
+ guidance: |
420
+ Gemma 4 chat template exposes thinking only as on or off; intermediate
421
+ levels coerce to on. NVFP4 quantization can corrupt JSON tool-call
422
+ output under aggressive sampling, so the instruct profile pins
423
+ temperature at 0.7. The lmstudio-native adapter re-classifies the
424
+ channel-marker thought region into ThinkingContent so markers do not
425
+ leak into visible text.
426
+ serving: "LilaRest publishes no per-model sampler; values inherit the gemma family default per the official Google Gemma 4 31B card. KV cache budget at f16, ctx=122880 is roughly 57 GB on top of ~21 GB of NVFP4 weights."
427
+
428
+ # The qat/UD-Q4_K_XL/MTP build is the long-context llama.cpp gemma profile and is a
429
+ # different serving profile from the NVFP4 turbo family above: 262K context
430
+ # (not 122880), q8 KV, F16 gemma4v mmproj vision, and MTP speculative decoding.
431
+ # Its longest matchPattern (gemma-4-31b-it-qat-...) is longer than the NVFP4
432
+ # family's `gemma-4-31b-it` (14 chars), so the longest-substring matcher in
433
+ # knowledge-base.ts resolves this id here rather than to the NVFP4 profile.
434
+ - family: gemma-4-31b-it-qat-mtp
435
+ matchPatterns:
436
+ - gemma-4-31b-it-qat-ud-q4_k_xl-mtp-262k
437
+ - gemma-4-31b-it-qat-ud-q4_k_xl-mtp
438
+ - gemma-4-31b-it-qat-ud
439
+ - gemma-4-31b-it-qat
440
+ capabilities:
441
+ chat: true
442
+ tools: true
443
+ toolCallFormat: openai
444
+ reasoning: true
445
+ thinkingFormat: qwen-chat-template
446
+ structuredOutputs: json-schema
447
+ vision: true
448
+ audio: false
449
+ embeddings: false
450
+ rerank: false
451
+ fim: false
452
+ contextWindow: 262144
453
+ maxTokens: 32768
454
+ quirks:
455
+ kvCache:
456
+ kQuant: q8_0
457
+ vQuant: q8_0
458
+ sampling:
459
+ thinking:
460
+ temperature: 1.0
461
+ topP: 0.95
462
+ topK: 64
463
+ minP: 0.0
464
+ presencePenalty: 0.0
465
+ repetitionPenalty: 1.0
466
+ maxTokens: 32768
467
+ instruct:
468
+ temperature: 0.7 # 0.7 keeps tool-call JSON parseable; router profiles may serve this qat build at 0.8
469
+ topP: 0.95
470
+ topK: 64
471
+ minP: 0.0
472
+ repeatPenalty: 1.05
473
+ maxTokens: 32768
474
+ chatTemplate: "gemma-channel"
475
+ gpuTiers:
476
+ "32gb": "32 GB Vulkan llama.cpp profile with UD-Q4_K_XL weights, ctx=262144, q8 KV, flash attention, parallel=1, F16 gemma4v mmproj (vision), and MTP draft (spec-type draft-mtp, n_max 4) both active."
477
+ runtimePreference:
478
+ llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning off. Recommended local profile serves at ctx=262144, q8 KV, flash attention, parallel=1, batch=1024, ubatch=256, with F16 mmproj vision and draft-mtp speculative decoding; sampler temperature 0.8, top_p 0.95, top_k 64, repeat_penalty 1.05."
479
+ openaiCompat: "Fallback when a gateway fronts the same llama.cpp chat-completions surface; the openai-compat surface does not expose the channel-marker classifier, so visible text may show raw thought markers."
480
+ llamaCpp:
481
+ ctxSize: 262144
482
+ cacheTypeK: q8_0
483
+ cacheTypeV: q8_0
484
+ flashAttn: true
485
+ nGpuLayers: 99
486
+ parallel: 1
487
+ batchSize: 1024
488
+ ubatchSize: 256
489
+ mmproj: true
490
+ thinking:
491
+ mechanism: on-off
492
+ guidance: |
493
+ Gemma 4 31B IT QAT exposes thinking only as on or off via the gemma-4
494
+ enable_thinking template; intermediate levels coerce to on. A local
495
+ llama.cpp profile serves it with --reasoning off (default_reasoning off),
496
+ so the chain-of-thought is not requested; if the template emits one it
497
+ surfaces via reasoning_content and is accounted as reasoning tokens. The
498
+ QAT weights hold tool-call JSON together better than the NVFP4 turbo
499
+ build, but the instruct profile still pins a conservative temperature.
500
+ serving: "Gemma 4 31B IT QAT, UD-Q4_K_XL, served as a local llama.cpp router model (vulkan) with MTP speculative decoding (spec-type draft-mtp, n_max 4, draft mtp-gemma-4-31B-it) and F16 gemma4v mmproj vision both active at 262k. Reasoning off by default; gemma-4 thinking is on/off. Sampler temperature 0.8, top_p 0.95, top_k 64, repeat_penalty 1.05."
501
+
502
+ - family: gemopus-4-31b-it
503
+ matchPatterns:
504
+ - gemopus-4-31b-it
505
+ - gemopus-4-31b
506
+ - gemopus-4
507
+ capabilities:
508
+ chat: true
509
+ tools: true
510
+ toolCallFormat: openai
511
+ reasoning: true
512
+ thinkingFormat: qwen-chat-template
513
+ structuredOutputs: json-schema
514
+ vision: false
515
+ audio: false
516
+ embeddings: false
517
+ rerank: false
518
+ fim: false
519
+ contextWindow: 122880
520
+ maxTokens: 32768
521
+ quirks:
522
+ kvCache:
523
+ kQuant: q8_0
524
+ vQuant: q8_0
525
+ sampling:
526
+ thinking:
527
+ temperature: 1.0
528
+ topP: 0.95
529
+ topK: 64
530
+ minP: 0.0
531
+ presencePenalty: 0.0
532
+ repetitionPenalty: 1.0
533
+ maxTokens: 32768
534
+ instruct:
535
+ temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
536
+ topP: 0.95
537
+ topK: 64
538
+ minP: 0.0
539
+ maxTokens: 32768
540
+ chatTemplate: "gemma-channel"
541
+ thinkingControl: |
542
+ Jackrong's distilled gemma-4 31B uses the same channel-marker template.
543
+ Thinking mode is gated by the literal '<|think|>' token in the system
544
+ prompt; Clio does not currently inject it, so the model emits an empty
545
+ thought region followed by the answer.
546
+ gpuTiers:
547
+ "48gb": "Fits at f16 KV with parallel=1; q8_0 KV recommended at parallel=4."
548
+ "80gb": "Comfortable at f16 KV with parallel=1 and ctx=122880."
549
+ runtimePreference:
550
+ lmstudioNative: "Verified path. Channel-marker handler in clio's adapter recovers reasoning tokens that the SDK classifier drops."
551
+ openaiCompat: "Fallback; reasoning_content is absent from the openai-compat reasoning probe for this family."
552
+ thinking:
553
+ mechanism: always-on
554
+ guidance: |
555
+ Distilled traces emit unconditionally. The thinking level setting is
556
+ ignored at the API surface; chain-of-thought arrives whether or not
557
+ Clio asks for it. Reasoning tokens are accounted via the channel-marker
558
+ re-classifier in lmstudio-native.
559
+ serving: "Distilled from Claude 4.6 Opus reasoning traces onto Google Gemma 4 31B; sampler matches the gemopus card values."
560
+
561
+ - family: qwopus3.6-27b-v1-preview
562
+ matchPatterns:
563
+ - qwopus3.6-27b-v1-preview
564
+ - qwopus3.6-27b
565
+ - qwopus3.6
566
+ capabilities:
567
+ chat: true
568
+ tools: true
569
+ toolCallFormat: qwen
570
+ reasoning: true
571
+ thinkingFormat: qwen-chat-template
572
+ structuredOutputs: json-schema
573
+ vision: false
574
+ audio: false
575
+ embeddings: false
576
+ rerank: false
577
+ fim: false
578
+ contextWindow: 262144
579
+ maxTokens: 32768
580
+ quirks:
581
+ sampling:
582
+ thinking:
583
+ temperature: 0.6
584
+ topP: 0.95
585
+ topK: 20
586
+ minP: 0.0
587
+ presencePenalty: 0.0
588
+ repetitionPenalty: 1.0
589
+ maxTokens: 32768
590
+ instruct:
591
+ temperature: 0.7
592
+ topP: 0.80
593
+ topK: 20
594
+ minP: 0.0
595
+ presencePenalty: 1.5
596
+ repetitionPenalty: 1.0
597
+ maxTokens: 32768
598
+ gpuTiers:
599
+ "32gb": "Fits at f16 KV with parallel=1 and ctx=262144 thanks to the Q4_K_M weights (~16.5 GB)."
600
+ runtimePreference:
601
+ lmstudioNative: "Primary path; thinking is emitted via standard <think> tags so the SDK classifier accounts reasoning correctly."
602
+ openaiCompat: "Verified for tool-call extraction parity; reasoning_content surfaces."
603
+ thinking:
604
+ mechanism: budget-tokens
605
+ budgetByLevel:
606
+ low: 1024
607
+ medium: 4096
608
+ high: 16384
609
+ guidance: |
610
+ Distilled Opus reasoning traces over Qwen3.6 27B; thinking emits via
611
+ standard <think> tags. Coding-task sampler is the precise variant.
612
+ Preserve thinking blocks across turns for iterative workflows.
613
+ serving: "Distilled from Claude 4.6 Opus reasoning traces onto Qwen3.6 27B. Coding-task sampler is the precise variant: temperature 0.6, top_p 0.95, top_k 20."
614
+
615
+ - family: qwopus3.6-35b-a3b-coder
616
+ matchPatterns:
617
+ - qwopus3.6-35b-a3b-coder-mtp-q4_k_m-262k
618
+ - qwopus3.6-35b-a3b-coder-mtp
619
+ - qwopus3.6-35b-a3b-coder
620
+ - qwopus3.6-35b-a3b
621
+ capabilities:
622
+ chat: true
623
+ tools: true
624
+ toolCallFormat: qwen
625
+ # Reasoning class: effort-levels, measured against an LM Studio host on
626
+ # 2026-08-08 over its OpenAI-compatible surface. The card
627
+ # frames the Coder-MTP line as thinking-off execution, but the wire
628
+ # disagrees: on "what is 17+25" this model spends 98 of 103 completion
629
+ # tokens on reasoning, and a wiki planning dispatch spent 89,501. It is an
630
+ # always-reasoning model unless told otherwise.
631
+ #
632
+ # chat_template_kwargs enable_thinking is inert here (verified with unique
633
+ # prompts to rule out cache hits). reasoning_effort "none" is the control
634
+ # that works, taking the same prompt from 98 reasoning tokens to 0, which
635
+ # is why `off` is mapped explicitly below rather than left unsent.
636
+ #
637
+ # reasoning=false previously resolved to mechanism "none", and mechanism
638
+ # "none" makes openai-completions strip reasoning_effort from the payload.
639
+ # The flag set to suppress thinking was the reason it could not be
640
+ # suppressed.
641
+ reasoning: true
642
+ thinkingFormat: qwen-chat-template
643
+ structuredOutputs: json-schema
644
+ vision: true
645
+ audio: false
646
+ embeddings: false
647
+ rerank: false
648
+ fim: false
649
+ contextWindow: 262144
650
+ maxTokens: 32768
651
+ quirks:
652
+ sampling:
653
+ instruct:
654
+ temperature: 0.2
655
+ topP: 0.9
656
+ topK: 20
657
+ repeatPenalty: 1.05
658
+ # Upstream Qwen3.6 card recommends presence_penalty 1.5 for
659
+ # non-thinking modes to suppress repetition loops. Measured on live
660
+ # dispatch (2026-07-06): without it a coder worker repeated one
661
+ # code_nav call into the loop-guard abort on 3 of 3 runs; with it the
662
+ # same task passed 3 of 3 with clean edit-then-validate trajectories.
663
+ presencePenalty: 1.5
664
+ maxTokens: 32768
665
+ gpuTiers:
666
+ "32gb": "Default 32 GB local profile: Q4_K_M, ctx=262144, q8 KV, flash attention, parallel=1, with F32 mmproj (vision) and draft-mtp speculative decoding both active. Leaves several GiB free at 262k on a 32 GiB class card."
667
+ runtimePreference:
668
+ llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile serves text-only with reasoning off and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
669
+ openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen tool parsing."
670
+ llamaCpp:
671
+ ctxSize: 262144
672
+ cacheTypeK: q8_0
673
+ cacheTypeV: q8_0
674
+ flashAttn: true
675
+ nGpuLayers: 99
676
+ parallel: 1
677
+ batchSize: 2048
678
+ ubatchSize: 512
679
+ mmproj: true
680
+ chatTemplateKwargs:
681
+ enable_thinking: false
682
+ thinking:
683
+ mechanism: effort-levels
684
+ effortByLevel:
685
+ # "none" is the only value that silences this family. It is carried on
686
+ # the wire because the model reasons by default, so sending nothing is
687
+ # in effect a request to keep reasoning.
688
+ off: none
689
+ low: low
690
+ medium: medium
691
+ high: high
692
+ guidance: |
693
+ Qwopus3.6 35B-A3B Coder reasons unless reasoning_effort is set to
694
+ "none", despite the creator's card describing the Coder-MTP variants as
695
+ thinking-off. That dial only reaches the wire on openai-compat; the LM
696
+ Studio native transport has no field for it, so a native-routed run
697
+ reasons at full rate whatever the configured thinking level says.
698
+ serving: "Jackrong Qwopus3.6 35B-A3B Coder MTP, Q4_K_M. Thinking-off coder/agent tuned for tool-use loops. Served as a local router default with MTP speculative decoding (spec-type draft-mtp, n_max 2) and F32 mmproj vision both active at 262k."
699
+
700
+ - family: qwopus3.6-27b-coder
701
+ matchPatterns:
702
+ - qwopus3.6-27b-coder-mtp-q5_k_m-262k
703
+ - qwopus3.6-27b-coder-mtp
704
+ - qwopus3.6-27b-coder
705
+ capabilities:
706
+ chat: true
707
+ tools: true
708
+ toolCallFormat: qwen
709
+ # Reasoning class: never, same Coder-MTP thinking-off design as the 35B.
710
+ reasoning: false
711
+ thinkingFormat: qwen-chat-template
712
+ structuredOutputs: json-schema
713
+ vision: false
714
+ audio: false
715
+ embeddings: false
716
+ rerank: false
717
+ fim: false
718
+ contextWindow: 262144
719
+ maxTokens: 32768
720
+ quirks:
721
+ sampling:
722
+ instruct:
723
+ temperature: 0.2
724
+ topP: 0.9
725
+ topK: 20
726
+ repeatPenalty: 1.05
727
+ # Same Coder-MTP non-thinking design as the 35B; see the measured
728
+ # presence-penalty note there.
729
+ presencePenalty: 1.5
730
+ maxTokens: 32768
731
+ runtimePreference:
732
+ llamaCpp: "Mini router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
733
+ lmstudioNative: "Dense 27B Coder-MTP; the creator's template ships thinking-off, and no per-request field re-enables it."
734
+ thinking:
735
+ mechanism: none
736
+ guidance: |
737
+ Qwopus3.6 27B Coder-MTP is the dense thinking-off coder variant; the
738
+ dial clamps to off at every level.
739
+ serving: "Jackrong Qwopus3.6 27B Coder MTP, Q5_K_M. Thinking-off dense coder for tool-use loops."
740
+
741
+ - family: qwopus3.5-9b-coder
742
+ matchPatterns:
743
+ - qwopus3.5-9b-coder-q8_0-262k
744
+ - qwopus3.5-9b-coder
745
+ capabilities:
746
+ chat: true
747
+ tools: true
748
+ toolCallFormat: qwen
749
+ # Reasoning class: never (Coder thinking-off design, 3.5 generation).
750
+ reasoning: false
751
+ thinkingFormat: qwen-chat-template
752
+ structuredOutputs: json-schema
753
+ vision: false
754
+ audio: false
755
+ embeddings: false
756
+ rerank: false
757
+ fim: false
758
+ contextWindow: 262144
759
+ maxTokens: 32768
760
+ quirks:
761
+ sampling:
762
+ instruct:
763
+ temperature: 0.2
764
+ topP: 0.9
765
+ topK: 20
766
+ repeatPenalty: 1.05
767
+ maxTokens: 32768
768
+ runtimePreference:
769
+ llamaCpp: "Mini router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
770
+ thinking:
771
+ mechanism: none
772
+ guidance: |
773
+ Qwopus3.5 9B Coder is the small thinking-off coder; the dial clamps
774
+ to off at every level.
775
+ serving: "Jackrong Qwopus3.5 9B Coder, Q8_0. Thinking-off small coder for fast tool loops."
776
+
777
+ - family: gemma4-26b-a4b
778
+ matchPatterns:
779
+ - gemma-4-26b-a4b
780
+ - gemma4-26b-a4b
781
+ - gemma-4-26b-a4b-it
782
+ - gemma4-26b-a4b-it
783
+ - gemma-4-26b-a4b-it-q4
784
+ - gemma-4-26b-a4b-it-q4_k_m
785
+ capabilities:
786
+ chat: true
787
+ tools: true
788
+ toolCallFormat: openai
789
+ reasoning: true
790
+ thinkingFormat: qwen-chat-template
791
+ structuredOutputs: json-schema
792
+ vision: true
793
+ audio: false
794
+ embeddings: false
795
+ rerank: false
796
+ fim: false
797
+ contextWindow: 262144
798
+ maxTokens: 65536
799
+ quirks:
800
+ sampling:
801
+ thinking:
802
+ temperature: 0.3
803
+ topP: 0.9
804
+ topK: 20
805
+ maxTokens: 8192
806
+ reasoningBudget: 4096
807
+ instruct:
808
+ temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
809
+ topP: 0.95
810
+ topK: 64
811
+ minP: 0.0
812
+ gpuTiers:
813
+ "32gb": "Main-agent candidate for 32 GB and larger local systems when Gemma behavior is preferred."
814
+ runtimePreference:
815
+ lmstudioNative: "Use for already-loaded LM Studio sessions and multimodal checks."
816
+ openaiCompat: "Preferred fallback when SDK rawTools do not match the template."
817
+ llamaCpp: "OpenAI-compatible chat completions with --jinja, --reasoning on, and the matching mmproj."
818
+ llamaCpp:
819
+ ctxSize: 262144
820
+ cacheTypeK: q8_0
821
+ cacheTypeV: q8_0
822
+ flashAttn: true
823
+ nGpuLayers: 99
824
+ mmproj: true
825
+ thinkingControl: "Gemma 4 exposes thinking through chat template enable_thinking support."
826
+ thinking:
827
+ mechanism: on-off
828
+ guidance: |
829
+ Gemma 4 chat template exposes thinking only as on or off via
830
+ enable_thinking. Intermediate levels coerce to on. The MoE A4B variant
831
+ keeps the same template behavior as the dense gemma-4 family.
832
+
833
+ - family: nemotron-cascade-2-30b-a3b
834
+ matchPatterns:
835
+ - nemotron-cascade-2
836
+ - nemotron-cascade-2-30b-a3b
837
+ - nemotron-cascade-2-30b-a3b-i1
838
+ - nemotron-cascade-2-30b-a3b-i1-q4_k_m
839
+ - nemotron-cascade2
840
+ - nemotron-cascade-2-30b
841
+ capabilities:
842
+ chat: true
843
+ tools: true
844
+ toolCallFormat: qwen
845
+ reasoning: true
846
+ thinkingFormat: qwen-chat-template
847
+ structuredOutputs: json-schema
848
+ vision: false
849
+ audio: false
850
+ embeddings: false
851
+ rerank: false
852
+ fim: false
853
+ contextWindow: 1048576
854
+ maxTokens: 65536
855
+ quirks:
856
+ sampling:
857
+ instruct:
858
+ temperature: 0.3
859
+ topP: 0.9
860
+ topK: 20
861
+ maxTokens: 8192
862
+ gpuTiers:
863
+ "32gb": "Preferred local worker model when a text-only worker is enough."
864
+ runtimePreference:
865
+ lmstudioNative: "Use when LM Studio SDK tool extraction is verified for the loaded template."
866
+ openaiCompat: "Preferred fallback for LM Studio gateway tool calls."
867
+ anthropicCompat: "Do not select for the current llama.cpp target profile; /v1/messages returned 404 while /v1/models and OpenAI-compatible chat worked."
868
+ llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile disables thinking in chat_template_kwargs."
869
+ llamaCpp:
870
+ ctxSize: 1048576
871
+ cacheTypeK: q8_0
872
+ cacheTypeV: q8_0
873
+ flashAttn: true
874
+ nGpuLayers: 99
875
+ parallel: 4
876
+ chatTemplateKwargs:
877
+ enable_thinking: false
878
+ thinking:
879
+ mechanism: on-off
880
+ guidance: |
881
+ Nemotron Cascade 2 chat template exposes thinking through
882
+ chat_template_kwargs.enable_thinking. Intermediate levels coerce to on.
883
+ The recommended llama.cpp profile disables thinking in the template kwargs by
884
+ default; flip via the runtime override when reasoning is needed.
885
+ serving: "NVIDIA card recommends vLLM with nemotron_v3 reasoning and qwen3_coder tool parsing."
886
+
887
+ - family: qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled
888
+ matchPatterns:
889
+ - qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled
890
+ - qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled-i1
891
+ - qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled-i1-q4_k_m
892
+ - qwen35-distilled
893
+ - qwen35-distilled-i1
894
+ - qwen35-distilled-i1-q4_k_m
895
+ capabilities:
896
+ chat: true
897
+ tools: true
898
+ toolCallFormat: qwen
899
+ reasoning: true
900
+ thinkingFormat: qwen-chat-template
901
+ structuredOutputs: json-schema
902
+ vision: false
903
+ audio: false
904
+ embeddings: false
905
+ rerank: false
906
+ fim: false
907
+ contextWindow: 262144
908
+ maxTokens: 32768
909
+ quirks:
910
+ sampling:
911
+ thinking:
912
+ temperature: 0.6
913
+ topP: 0.95
914
+ topK: 20
915
+ maxTokens: 16384
916
+ reasoningBudget: 4096
917
+ instruct:
918
+ temperature: 0.2
919
+ topP: 0.9
920
+ topK: 20
921
+ maxTokens: 4096
922
+ gpuTiers:
923
+ "32gb": "Distilled Opus-style reasoning at Qwen3.5 35B A3B scale. Use as a strong local main agent when GPU room is available."
924
+ runtimePreference:
925
+ lmstudioNative: "Verified usable on LM Studio when loaded with the Qwen-style preset."
926
+ openaiCompat: "Preferred fallback when SDK rawTools yield empty arguments for the distilled template."
927
+ llamaCpp: "OpenAI-compatible chat completions with --jinja. Match qwen3 reasoning parser flags upstream."
928
+ thinking:
929
+ mechanism: budget-tokens
930
+ budgetByLevel:
931
+ low: 1024
932
+ medium: 4096
933
+ high: 16384
934
+ guidance: |
935
+ Distilled Opus reasoning over Qwen3.5 35B A3B; thinking emits via the
936
+ qwen chat template. Preserve thinking blocks across coding-task turns;
937
+ the budget caps runaway chains while keeping iterative reasoning
938
+ coherent.
939
+ serving: "Distilled from Claude 4.6 Opus reasoning traces; expect stronger thinking-then-answer behavior than vanilla Qwen3.5 35B A3B."
940
+
941
+ - family: qwopus3.5-9b-v3
942
+ matchPatterns:
943
+ - qwopus3.5-9b-v3
944
+ - qwopus-9b
945
+ - qwopus 9b
946
+ capabilities:
947
+ chat: true
948
+ tools: true
949
+ toolCallFormat: qwen
950
+ reasoning: true
951
+ thinkingFormat: qwen-chat-template
952
+ structuredOutputs: json-schema
953
+ vision: false
954
+ audio: false
955
+ embeddings: false
956
+ rerank: false
957
+ fim: false
958
+ contextWindow: 262144
959
+ maxTokens: 32768
960
+ quirks:
961
+ sampling:
962
+ instruct:
963
+ temperature: 0.2
964
+ topP: 0.9
965
+ topK: 20
966
+ maxTokens: 4096
967
+ thinking:
968
+ temperature: 0.6
969
+ topP: 0.95
970
+ topK: 20
971
+ maxTokens: 8192
972
+ reasoningBudget: 2048
973
+ gpuTiers:
974
+ "16gb": "Use this as the local 16GB emulation target. It keeps the output cap below the 32GB-class MoE defaults."
975
+ runtimePreference:
976
+ lmstudioNative: "Primary local laptop path. Verified through LM Studio /api/v1/models with 262144 loaded context, flash attention, GPU KV offload, and trained_for_tool_use."
977
+ openaiCompat: "Verified through the LM Studio gateway for read and artifact tool calls; use when the native SDK produces thinking-only length stops or malformed tools."
978
+ anthropicCompat: "LM Studio's Anthropic Messages surface streams simple text with the server root URL as baseUrl; keep as a protocol fallback until tool and reasoning behavior are proven."
979
+ thinking:
980
+ # Measured 2026-08-11 against LM Studio on this exact model: baseline 56
981
+ # reasoning tokens, reasoning_effort none 120, reasoning_effort minimal
982
+ # 243, chat_template_kwargs.enable_thinking false 45, both together 44.
983
+ # No spelling silences it, and the budget-tokens mechanism it used to
984
+ # claim is informational on both LM Studio and llama.cpp, so nothing ever
985
+ # reached the wire. Always-on is what the model actually does.
986
+ mechanism: always-on
987
+ guidance: |
988
+ Qwopus 9B distilled from Opus reasoning; the chain-of-thought comes out
989
+ of the chat template unconditionally and the dial cannot stop it, so
990
+ Clio shows the level as forced. Output cap is tighter than the
991
+ 32GB-class MoE families because this is the 16GB emulation target;
992
+ budget for a reasoning preamble ahead of every answer.
993
+ serving: "LM Studio reports trained_for_tool_use for the local qwopus3.5-9b-v3 target. Native model override successfully switched residency to granite-4.0-h-350m through the SDK."