@iowarp/clio-coder 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +407 -0
- package/CODE_OF_CONDUCT.md +21 -0
- package/CONTRIBUTING.md +224 -0
- package/LICENSE +202 -0
- package/NOTICE +9 -0
- package/README.md +798 -0
- package/SECURITY.md +72 -0
- package/assets/clio-coder-logo-128.webp +0 -0
- package/damage-control-rules.yaml +419 -0
- package/dist/acp-UMLFVA3F.js +92 -0
- package/dist/agents-Q4MYPMUW.js +91 -0
- package/dist/auth-O6HYIJ6J.js +521 -0
- package/dist/chunk-262G75JS.js +35 -0
- package/dist/chunk-26BZQOAD.js +1281 -0
- package/dist/chunk-2J63S4SF.js +508 -0
- package/dist/chunk-3DANZDGR.js +717 -0
- package/dist/chunk-4UQA7NCT.js +29 -0
- package/dist/chunk-527KG6XR.js +497 -0
- package/dist/chunk-5LDRNKX2.js +1063 -0
- package/dist/chunk-5N2FG33Q.js +25 -0
- package/dist/chunk-67MTHP2E.js +135 -0
- package/dist/chunk-6CWDTGUC.js +20 -0
- package/dist/chunk-7BHLZB3A.js +2115 -0
- package/dist/chunk-7RBKDI66.js +348 -0
- package/dist/chunk-AMFR5YA3.js +541 -0
- package/dist/chunk-BBUH4VAA.js +1224 -0
- package/dist/chunk-BYEU76JP.js +899 -0
- package/dist/chunk-CLJ5HLUD.js +458 -0
- package/dist/chunk-D5YD55AR.js +116 -0
- package/dist/chunk-DXQNI4PC.js +61 -0
- package/dist/chunk-E3NYWENM.js +1004 -0
- package/dist/chunk-GNGDQYDU.js +34688 -0
- package/dist/chunk-GOTUR54M.js +9 -0
- package/dist/chunk-HBU5MTAM.js +41 -0
- package/dist/chunk-HMYNFFY4.js +28 -0
- package/dist/chunk-JPOWPFCU.js +1010 -0
- package/dist/chunk-JWHCJDCI.js +1215 -0
- package/dist/chunk-KBR4MZZR.js +41 -0
- package/dist/chunk-KKKPTZLM.js +93 -0
- package/dist/chunk-ME6DNWIU.js +66 -0
- package/dist/chunk-NI4DEJMC.js +88 -0
- package/dist/chunk-O4EJEDHO.js +659 -0
- package/dist/chunk-PIDUD6M2.js +31 -0
- package/dist/chunk-PS4PFJQP.js +29459 -0
- package/dist/chunk-QV47YRF4.js +48 -0
- package/dist/chunk-RQDWMVRB.js +279 -0
- package/dist/chunk-TFSSEXL6.js +136 -0
- package/dist/chunk-TKHQ4DGZ.js +8290 -0
- package/dist/chunk-TPOCL34A.js +2876 -0
- package/dist/chunk-UGYAX5YI.js +565 -0
- package/dist/chunk-UHTSULZS.js +461 -0
- package/dist/chunk-UU3R62TT.js +128 -0
- package/dist/chunk-UWIJNAOB.js +3906 -0
- package/dist/chunk-VOO7NYPP.js +914 -0
- package/dist/chunk-VPAWTYLY.js +117 -0
- package/dist/chunk-WD6AJM35.js +1216 -0
- package/dist/chunk-X3BR7HWV.js +115 -0
- package/dist/chunk-X3NE4WVW.js +120 -0
- package/dist/chunk-XNISANGE.js +1395 -0
- package/dist/chunk-XV4ZJ6ZM.js +3177 -0
- package/dist/cli/index.js +236 -0
- package/dist/clio-KIQ5SNDS.js +53 -0
- package/dist/components-JVHMUBEB.js +653 -0
- package/dist/config-ZFCDBMDC.js +372 -0
- package/dist/configure-G4E3A2PG.js +27 -0
- package/dist/context-CDXTP2MP.js +293 -0
- package/dist/context-E3KIFVXI.js +185 -0
- package/dist/context-clear-3F4PLXOS.js +102 -0
- package/dist/context-index-Q7YSYTR3.js +106 -0
- package/dist/docs-YIETIWZI.js +280 -0
- package/dist/doctor-M5HJJZOL.js +61 -0
- package/dist/domains/agents/builtins/architect.md +33 -0
- package/dist/domains/agents/builtins/coder.md +31 -0
- package/dist/domains/agents/builtins/context-bootstrap.md +38 -0
- package/dist/domains/agents/builtins/debugger.md +30 -0
- package/dist/domains/agents/builtins/documenter.md +31 -0
- package/dist/domains/agents/builtins/git-master.md +30 -0
- package/dist/domains/agents/builtins/provenance.md +30 -0
- package/dist/domains/agents/builtins/researcher.md +71 -0
- package/dist/domains/agents/builtins/scout.md +42 -0
- package/dist/domains/agents/builtins/tester.md +31 -0
- package/dist/domains/agents/builtins/verifier.md +30 -0
- package/dist/domains/agents/builtins/wiki-writer.md +41 -0
- package/dist/eval-B3KZZESM.js +2674 -0
- package/dist/evidence-V67CHM35.js +233 -0
- package/dist/evolve-YDZSUQYA.js +518 -0
- package/dist/extensions-SRG7XCAH.js +207 -0
- package/dist/fleet-CA2CRTVG.js +760 -0
- package/dist/fleet-preflight-CLIAX7YR.js +21 -0
- package/dist/init-2OZDJE2D.js +227 -0
- package/dist/memory-3PIQQAKX.js +207 -0
- package/dist/models-DY35XI7Y.js +237 -0
- package/dist/paths-5OMXW7Z4.js +57 -0
- package/dist/preload-KZVHET2B.js +11 -0
- package/dist/reset-PIFYNOS3.js +216 -0
- package/dist/run-3VSPP24F.js +735 -0
- package/dist/share-D36RQCXM.js +241 -0
- package/dist/skills-F2MRLELY.js +445 -0
- package/dist/skills-eval-E2ZTW4PL.js +932 -0
- package/dist/targets-DZMEZAH4.js +977 -0
- package/dist/trace-7NYCUI2J.js +250 -0
- package/dist/uninstall-AD3JWHBB.js +322 -0
- package/dist/upgrade-WYYBKGDY.js +301 -0
- package/dist/usage-ULIDAGFF.js +755 -0
- package/dist/version-ROZ6CZKH.js +16 -0
- package/dist/wiki-generate-PKFIX6OB.js +377 -0
- package/dist/worker/entry.js +1739 -0
- package/docs/README.md +93 -0
- package/docs/acp.md +120 -0
- package/docs/alcf-provider.md +72 -0
- package/docs/architecture.md +172 -0
- package/docs/artifact-versions.md +54 -0
- package/docs/built-in-agents.md +265 -0
- package/docs/capacity-and-scheduling.md +97 -0
- package/docs/commands-and-modes.md +554 -0
- package/docs/config-knobs-audit.md +115 -0
- package/docs/configuration-and-targets.md +812 -0
- package/docs/context-engine.md +236 -0
- package/docs/dispatch-architecture-rationale.md +126 -0
- package/docs/documentation-coverage.md +46 -0
- package/docs/documentation-guide.md +166 -0
- package/docs/environment-variables.md +105 -0
- package/docs/eval-runner.md +205 -0
- package/docs/evals-internal.md +298 -0
- package/docs/evidence-and-memory.md +243 -0
- package/docs/evolution.md +143 -0
- package/docs/exit-codes-and-output.md +74 -0
- package/docs/extensions-and-sharing.md +306 -0
- package/docs/fleet-demo-runbook.md +179 -0
- package/docs/fleet-dispatch.md +591 -0
- package/docs/glossary.md +75 -0
- package/docs/html/agents_blueprint.html +936 -0
- package/docs/html/alcf_blueprint.html +324 -0
- package/docs/html/architecture_blueprint.html +850 -0
- package/docs/html/commands_blueprint.html +794 -0
- package/docs/html/config_knobs_audit_blueprint.html +178 -0
- package/docs/html/configuration_blueprint.html +1080 -0
- package/docs/html/context_blueprint.html +603 -0
- package/docs/html/documentation_blueprint.html +832 -0
- package/docs/html/environment_blueprint.html +404 -0
- package/docs/html/eval_blueprint.html +743 -0
- package/docs/html/evals_internal_blueprint.html +190 -0
- package/docs/html/evolution_blueprint.html +674 -0
- package/docs/html/extensions_blueprint.html +2065 -0
- package/docs/html/fleet_dispatch_blueprint.html +286 -0
- package/docs/html/index.html +919 -0
- package/docs/html/lifecycle_blueprint.html +723 -0
- package/docs/html/memory_blueprint.html +699 -0
- package/docs/html/middleware_blueprint.html +664 -0
- package/docs/html/models_blueprint.html +2366 -0
- package/docs/html/observability_blueprint.html +683 -0
- package/docs/html/provider_adapter_blueprint.html +245 -0
- package/docs/html/safety_blueprint.html +1386 -0
- package/docs/html/shared.css +571 -0
- package/docs/html/shared.js +143 -0
- package/docs/html/skills_blueprint.html +671 -0
- package/docs/html/soak_blueprint.html +182 -0
- package/docs/html/tool_usage_blueprint.html +350 -0
- package/docs/html/tools_blueprint.html +2249 -0
- package/docs/html/trace_blueprint.html +235 -0
- package/docs/html/tui_design_blueprint.html +314 -0
- package/docs/html/validation_blueprint.html +961 -0
- package/docs/html/worker_dispatch_blueprint.html +231 -0
- package/docs/installation-and-lifecycle.md +308 -0
- package/docs/middleware-and-components.md +148 -0
- package/docs/model-catalog.md +189 -0
- package/docs/observability.md +233 -0
- package/docs/proactive-memory.md +452 -0
- package/docs/prompt-envelope-and-tools.md +142 -0
- package/docs/provider-adapter-cookbook.md +148 -0
- package/docs/release-cut-checklist.md +138 -0
- package/docs/safety-model.md +357 -0
- package/docs/scientific-validation.md +105 -0
- package/docs/session-lifecycle.md +156 -0
- package/docs/skills-marketplace.md +46 -0
- package/docs/tool-usage.md +527 -0
- package/docs/trace-store.md +132 -0
- package/docs/troubleshooting.md +33 -0
- package/docs/tui-design.md +239 -0
- package/docs/worker-dispatch-mechanics.md +242 -0
- package/package.json +132 -0
- package/skills/README.md +408 -0
- package/skills/git/commit-crafting/SKILL.md +79 -0
- package/skills/git/commit-crafting/evals.md +92 -0
- package/skills/git/create-pr/SKILL.md +116 -0
- package/skills/git/create-pr/evals.md +114 -0
- package/skills/git/investigate-issue/SKILL.md +139 -0
- package/skills/git/investigate-issue/evals.md +94 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +96 -0
- package/skills/git/resolve-merge-conflicts/evals.md +58 -0
- package/skills/git/review-changes/SKILL.md +103 -0
- package/skills/git/review-changes/evals.md +85 -0
- package/skills/git/worktree-create/SKILL.md +92 -0
- package/skills/git/worktree-create/evals.md +97 -0
- package/skills/git/worktree-create/references/worktree-setup.md +66 -0
- package/skills/git/worktree-merge/SKILL.md +95 -0
- package/skills/git/worktree-merge/evals.md +114 -0
- package/skills/skill-marketplace.json +261 -0
- package/skills/workflow/cut-it/SKILL.md +86 -0
- package/skills/workflow/cut-it/evals.md +42 -0
- package/src/domains/agents/builtins/architect.md +33 -0
- package/src/domains/agents/builtins/coder.md +31 -0
- package/src/domains/agents/builtins/context-bootstrap.md +38 -0
- package/src/domains/agents/builtins/debugger.md +30 -0
- package/src/domains/agents/builtins/documenter.md +31 -0
- package/src/domains/agents/builtins/git-master.md +30 -0
- package/src/domains/agents/builtins/provenance.md +30 -0
- package/src/domains/agents/builtins/researcher.md +71 -0
- package/src/domains/agents/builtins/scout.md +42 -0
- package/src/domains/agents/builtins/tester.md +31 -0
- package/src/domains/agents/builtins/verifier.md +30 -0
- package/src/domains/agents/builtins/wiki-writer.md +41 -0
- package/src/domains/agents/fleets/build-review.md +34 -0
- package/src/domains/agents/fleets/build-test.md +35 -0
- package/src/domains/agents/fleets/sdlc.md +86 -0
- package/src/domains/prompts/fragments/identity/clio-worker.md +11 -0
- package/src/domains/prompts/fragments/identity/clio.md +26 -0
- package/src/domains/prompts/fragments/operating/contract.md +64 -0
- package/src/domains/prompts/fragments/safety/auto-edit.md +14 -0
- package/src/domains/prompts/fragments/safety/full-auto.md +14 -0
- package/src/domains/prompts/fragments/safety/read-only.md +13 -0
- package/src/domains/prompts/fragments/safety/suggest.md +13 -0
- package/src/domains/prompts/fragments/wiki/page.md +75 -0
- package/src/domains/prompts/fragments/wiki/plan.md +48 -0
- package/src/domains/providers/models/cloud-models/alcf.yaml +40 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +993 -0
|
@@ -0,0 +1,993 @@
|
|
|
1
|
+
# Sources:
|
|
2
|
+
# https://huggingface.co/Qwen/Qwen3.6-35B-A3B
|
|
3
|
+
# https://huggingface.co/Qwen/Qwen3.6-27B
|
|
4
|
+
# https://huggingface.co/google/gemma-4-26B-A4B
|
|
5
|
+
# https://huggingface.co/LilaRest/gemma-4-31B-it-NVFP4-turbo
|
|
6
|
+
# https://huggingface.co/Jackrong/Gemopus-4-31B-it-GGUF
|
|
7
|
+
# https://huggingface.co/Jackrong/Qwopus3.6-27B-v1-preview-GGUF
|
|
8
|
+
# https://huggingface.co/Jackrong/Qwopus3.6-35B-A3B-Coder-MTP-GGUF
|
|
9
|
+
# https://huggingface.co/nvidia/Nemotron-Cascade-2-30B-A3B
|
|
10
|
+
# https://huggingface.co/unsloth/NVIDIA-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-GGUF
|
|
11
|
+
# https://huggingface.co/mradermacher/Qwen3.5-35B-A3B-Claude-4.6-Opus-Reasoning-Distilled-i1-GGUF
|
|
12
|
+
# https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-GGUF
|
|
13
|
+
#
|
|
14
|
+
# This knowledge base is intentionally narrow. It describes only the local
|
|
15
|
+
# models curated for Clio's local coding workflows and leaves cloud GPT models
|
|
16
|
+
# to pi-ai's native OpenAI and openai-codex catalogs.
|
|
17
|
+
#
|
|
18
|
+
# Thinking semantics for these local families: the chain-of-thought is emitted
|
|
19
|
+
# by the model's chat template (Qwen-style <think> blocks, Gemma 4 thinking
|
|
20
|
+
# template, Nemotron reasoning template). LM Studio's SDK and llama.cpp's
|
|
21
|
+
# OpenAI-compatible surface do not expose a reliable "no thinking" flag, so
|
|
22
|
+
# Clio cannot disable template-driven reasoning at the API layer. With
|
|
23
|
+
# `--thinking off` Clio simply does not request reasoning; if the model emits
|
|
24
|
+
# it anyway, the chain-of-thought is captured into a ThinkingContent block,
|
|
25
|
+
# counted as reasoning tokens (separate from output tokens) on the receipt
|
|
26
|
+
# and TUI footer, and surfaced to the user truthfully rather than hidden.
|
|
27
|
+
- family: openai-gpt-oss
|
|
28
|
+
matchPatterns:
|
|
29
|
+
- openai/gpt-oss
|
|
30
|
+
- gpt-oss
|
|
31
|
+
- gpt-oss-20b
|
|
32
|
+
- gpt-oss-120b
|
|
33
|
+
capabilities:
|
|
34
|
+
chat: true
|
|
35
|
+
tools: true
|
|
36
|
+
toolCallFormat: openai
|
|
37
|
+
reasoning: true
|
|
38
|
+
thinkingFormat: harmony
|
|
39
|
+
structuredOutputs: json-schema
|
|
40
|
+
vision: false
|
|
41
|
+
audio: false
|
|
42
|
+
embeddings: false
|
|
43
|
+
rerank: false
|
|
44
|
+
fim: false
|
|
45
|
+
contextWindow: 131072
|
|
46
|
+
maxTokens: 32768
|
|
47
|
+
quirks:
|
|
48
|
+
sampling:
|
|
49
|
+
thinking:
|
|
50
|
+
temperature: 1
|
|
51
|
+
topP: 1
|
|
52
|
+
maxTokens: 32768
|
|
53
|
+
instruct:
|
|
54
|
+
temperature: 1
|
|
55
|
+
topP: 1
|
|
56
|
+
maxTokens: 8192
|
|
57
|
+
runtimePreference:
|
|
58
|
+
llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high."
|
|
59
|
+
openaiCompat: "Preferred for Harmony reasoning-effort control when the local gateway exposes chat_template_kwargs."
|
|
60
|
+
thinking:
|
|
61
|
+
mechanism: effort-levels
|
|
62
|
+
effortByLevel:
|
|
63
|
+
low: low
|
|
64
|
+
medium: medium
|
|
65
|
+
high: high
|
|
66
|
+
guidance: |
|
|
67
|
+
GPT-OSS uses OpenAI Harmony formatting and supports low, medium, and
|
|
68
|
+
high reasoning effort. Clio coerces off/minimal to low, captures
|
|
69
|
+
non-final Harmony channels as ThinkingContent, and strips Harmony
|
|
70
|
+
control tokens from visible output.
|
|
71
|
+
|
|
72
|
+
- family: agenticqwen-30b-a3b-i1
|
|
73
|
+
matchPatterns:
|
|
74
|
+
- agenticqwen-30b-a3b-i1
|
|
75
|
+
- agenticqwen-30b-a3b-i1-q4_k_m
|
|
76
|
+
- agenticqwen
|
|
77
|
+
capabilities:
|
|
78
|
+
chat: true
|
|
79
|
+
tools: true
|
|
80
|
+
toolCallFormat: qwen
|
|
81
|
+
reasoning: true
|
|
82
|
+
thinkingFormat: qwen-chat-template
|
|
83
|
+
structuredOutputs: json-schema
|
|
84
|
+
vision: false
|
|
85
|
+
audio: false
|
|
86
|
+
embeddings: false
|
|
87
|
+
rerank: false
|
|
88
|
+
fim: false
|
|
89
|
+
contextWindow: 262144
|
|
90
|
+
maxTokens: 65536
|
|
91
|
+
quirks:
|
|
92
|
+
sampling:
|
|
93
|
+
thinking:
|
|
94
|
+
temperature: 0.5
|
|
95
|
+
topP: 0.9
|
|
96
|
+
topK: 20
|
|
97
|
+
maxTokens: 16384
|
|
98
|
+
reasoningBudget: 4096
|
|
99
|
+
instruct:
|
|
100
|
+
temperature: 0.5
|
|
101
|
+
topP: 0.9
|
|
102
|
+
topK: 20
|
|
103
|
+
maxTokens: 4096
|
|
104
|
+
gpuTiers:
|
|
105
|
+
"32gb": "Default 32 GB local llama.cpp orchestrator profile; benchmarked with roughly 1 GiB VRAM headroom on a 32 GiB class card."
|
|
106
|
+
runtimePreference:
|
|
107
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile uses ctx=262144, parallel=4, batch=2048, ubatch=512, and q8 KV cache."
|
|
108
|
+
openaiCompat: "Use only when a gateway fronts the same llama.cpp chat-completions surface."
|
|
109
|
+
llamaCpp:
|
|
110
|
+
ctxSize: 262144
|
|
111
|
+
cacheTypeK: q8_0
|
|
112
|
+
cacheTypeV: q8_0
|
|
113
|
+
parallel: 4
|
|
114
|
+
batchSize: 2048
|
|
115
|
+
ubatchSize: 512
|
|
116
|
+
thinking:
|
|
117
|
+
mechanism: budget-tokens
|
|
118
|
+
budgetByLevel:
|
|
119
|
+
low: 1024
|
|
120
|
+
medium: 4096
|
|
121
|
+
high: 16384
|
|
122
|
+
guidance: |
|
|
123
|
+
AgenticQwen is the tuned local default for Clio source work on
|
|
124
|
+
a local llama.cpp server. Thinking emits through the qwen chat template;
|
|
125
|
+
keep high reasoning for main-agent work and preserve visible thinking
|
|
126
|
+
blocks in receipts.
|
|
127
|
+
serving: "Use the llama.cpp settings listed above for a 262144-token local profile."
|
|
128
|
+
|
|
129
|
+
- family: nemotron-3-nano-omni-30b-a3b-reasoning
|
|
130
|
+
matchPatterns:
|
|
131
|
+
- nvidia-nemotron-3-nano-omni-30b-a3b-reasoning
|
|
132
|
+
- nemotron-3-nano-omni-30b-a3b-reasoning-ud-q4_k_m
|
|
133
|
+
- nemotron-3-nano-omni-30b-a3b-reasoning
|
|
134
|
+
- nemotron-3-nano-omni
|
|
135
|
+
- nemotron-nano-omni
|
|
136
|
+
capabilities:
|
|
137
|
+
chat: true
|
|
138
|
+
tools: true
|
|
139
|
+
toolCallFormat: qwen
|
|
140
|
+
reasoning: true
|
|
141
|
+
thinkingFormat: qwen-chat-template
|
|
142
|
+
structuredOutputs: json-schema
|
|
143
|
+
vision: true
|
|
144
|
+
audio: true
|
|
145
|
+
embeddings: false
|
|
146
|
+
rerank: false
|
|
147
|
+
fim: false
|
|
148
|
+
contextWindow: 1048576
|
|
149
|
+
maxTokens: 131072
|
|
150
|
+
quirks:
|
|
151
|
+
sampling:
|
|
152
|
+
thinking:
|
|
153
|
+
temperature: 0.6
|
|
154
|
+
topP: 0.95
|
|
155
|
+
topK: 20
|
|
156
|
+
maxTokens: 20480
|
|
157
|
+
reasoningBudget: 16384
|
|
158
|
+
gracePeriod: 1024
|
|
159
|
+
instruct:
|
|
160
|
+
temperature: 0.2
|
|
161
|
+
topK: 1
|
|
162
|
+
maxTokens: 1024
|
|
163
|
+
gpuTiers:
|
|
164
|
+
"24gb": "Use a 4-bit quant with reduced load context for evaluation. Prefer the 32GB tier for sustained Clio writes."
|
|
165
|
+
"32gb": "Primary strong local model for multimodal main-agent work. Keep openai-compat available for tool-call fallback."
|
|
166
|
+
runtimePreference:
|
|
167
|
+
lmstudioNative: "Use only when the target's server lifecycle is Clio-managed or the model is already loaded at the requested context."
|
|
168
|
+
openaiCompat: "Preferred fallback for LM Studio gateway tool-call extraction."
|
|
169
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja."
|
|
170
|
+
llamaCpp:
|
|
171
|
+
ctxSize: 819200
|
|
172
|
+
cacheTypeK: q8_0
|
|
173
|
+
cacheTypeV: q8_0
|
|
174
|
+
flashAttn: true
|
|
175
|
+
nGpuLayers: 99
|
|
176
|
+
parallel: 4
|
|
177
|
+
thinking:
|
|
178
|
+
mechanism: budget-tokens
|
|
179
|
+
budgetByLevel:
|
|
180
|
+
low: 1024
|
|
181
|
+
medium: 4096
|
|
182
|
+
high: 16384
|
|
183
|
+
guidance: |
|
|
184
|
+
Reasoning is template-driven; thinking blocks emit through the qwen
|
|
185
|
+
chat template. Preserve thinking across coding-task turns when the
|
|
186
|
+
chain matters; otherwise let the budget cap output growth.
|
|
187
|
+
serving: "Use qwen3_coder tool parsing when served through OpenAI-compatible HTTP. LM Studio native uses SDK rawTools, but openai-compat remains the fallback for template-specific tool extraction issues."
|
|
188
|
+
|
|
189
|
+
- family: qwen3.6-35b-a3b
|
|
190
|
+
matchPatterns:
|
|
191
|
+
- qwen3.6-35b-a3b
|
|
192
|
+
- qwen3_6-35b-a3b
|
|
193
|
+
- qwen3-6-35b-a3b
|
|
194
|
+
- qwen3.6-35b
|
|
195
|
+
- qwen3.6-35b-a3b-ud
|
|
196
|
+
- qwen3.6-35b-a3b-ud-q4_k_xl
|
|
197
|
+
capabilities:
|
|
198
|
+
chat: true
|
|
199
|
+
tools: true
|
|
200
|
+
toolCallFormat: qwen
|
|
201
|
+
reasoning: true
|
|
202
|
+
thinkingFormat: qwen-chat-template
|
|
203
|
+
structuredOutputs: json-schema
|
|
204
|
+
vision: true
|
|
205
|
+
audio: false
|
|
206
|
+
embeddings: false
|
|
207
|
+
rerank: false
|
|
208
|
+
fim: false
|
|
209
|
+
contextWindow: 262144
|
|
210
|
+
maxTokens: 65536
|
|
211
|
+
quirks:
|
|
212
|
+
sampling:
|
|
213
|
+
thinking:
|
|
214
|
+
temperature: 0.6
|
|
215
|
+
topP: 0.95
|
|
216
|
+
topK: 20
|
|
217
|
+
maxTokens: 16384
|
|
218
|
+
reasoningBudget: 4096
|
|
219
|
+
instruct:
|
|
220
|
+
temperature: 0.7
|
|
221
|
+
topP: 0.80
|
|
222
|
+
topK: 20
|
|
223
|
+
minP: 0.0
|
|
224
|
+
gpuTiers:
|
|
225
|
+
"32gb": "Main-agent candidate on 32GB targets when Nemotron Omni is unavailable or tool behavior is better for text-only work."
|
|
226
|
+
runtimePreference:
|
|
227
|
+
lmstudioNative: "Usable for text and vision when loaded in LM Studio. Keep openai-compat available for tool fallback."
|
|
228
|
+
openaiCompat: "Preferred local gateway path for tool-call stress tests."
|
|
229
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Local llama.cpp profile emits reasoning_content without a separate --reasoning flag."
|
|
230
|
+
llamaCpp:
|
|
231
|
+
ctxSize: 262144
|
|
232
|
+
cacheTypeK: q8_0
|
|
233
|
+
cacheTypeV: q8_0
|
|
234
|
+
flashAttn: true
|
|
235
|
+
nGpuLayers: 99
|
|
236
|
+
thinking:
|
|
237
|
+
mechanism: budget-tokens
|
|
238
|
+
budgetByLevel:
|
|
239
|
+
low: 1024
|
|
240
|
+
medium: 4096
|
|
241
|
+
high: 16384
|
|
242
|
+
guidance: |
|
|
243
|
+
Qwen chat template emits thinking blocks. The card recommends at least
|
|
244
|
+
128K context for thinking quality. Preserve thinking blocks across
|
|
245
|
+
coding-task turns so iterative reasoning chains do not restart cold.
|
|
246
|
+
serving: "Official card recommends at least 128K context for thinking quality. Use 262K when memory allows."
|
|
247
|
+
|
|
248
|
+
- family: ornith-1.0-35b
|
|
249
|
+
matchPatterns:
|
|
250
|
+
- ornith-1.0-35b
|
|
251
|
+
- ornith-1.0-35b-q4_k_m
|
|
252
|
+
- ornith-1.0-35b-gguf
|
|
253
|
+
- deepreinforce-ai/ornith-1.0-35b-gguf
|
|
254
|
+
- ornith
|
|
255
|
+
capabilities:
|
|
256
|
+
chat: true
|
|
257
|
+
tools: true
|
|
258
|
+
toolCallFormat: qwen
|
|
259
|
+
reasoning: true
|
|
260
|
+
thinkingFormat: qwen-chat-template
|
|
261
|
+
structuredOutputs: json-schema
|
|
262
|
+
vision: false
|
|
263
|
+
audio: false
|
|
264
|
+
embeddings: false
|
|
265
|
+
rerank: false
|
|
266
|
+
fim: false
|
|
267
|
+
contextWindow: 262144
|
|
268
|
+
maxTokens: 65536
|
|
269
|
+
quirks:
|
|
270
|
+
kvCache:
|
|
271
|
+
kQuant: q8_0
|
|
272
|
+
vQuant: q8_0
|
|
273
|
+
sampling:
|
|
274
|
+
thinking:
|
|
275
|
+
temperature: 0.6
|
|
276
|
+
topP: 0.95
|
|
277
|
+
topK: 20
|
|
278
|
+
maxTokens: 16384
|
|
279
|
+
reasoningBudget: 4096
|
|
280
|
+
instruct:
|
|
281
|
+
temperature: 0.6
|
|
282
|
+
topP: 0.95
|
|
283
|
+
topK: 20
|
|
284
|
+
repeatPenalty: 1.05
|
|
285
|
+
maxTokens: 4096
|
|
286
|
+
gpuTiers:
|
|
287
|
+
"32gb": "32 GB local profile with Q4_K_M, ctx=262144, q8 KV, flash attention, and parallel=1."
|
|
288
|
+
runtimePreference:
|
|
289
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. Recommended local profile uses ctx=262144, parallel=1, batch=2048, ubatch=512, flash attention, and q8 KV cache."
|
|
290
|
+
openaiCompat: "Preferred when a gateway fronts the same llama.cpp chat-completions surface with qwen reasoning/tool parsing."
|
|
291
|
+
llamaCpp:
|
|
292
|
+
ctxSize: 262144
|
|
293
|
+
cacheTypeK: q8_0
|
|
294
|
+
cacheTypeV: q8_0
|
|
295
|
+
flashAttn: true
|
|
296
|
+
nGpuLayers: 99
|
|
297
|
+
parallel: 1
|
|
298
|
+
batchSize: 2048
|
|
299
|
+
ubatchSize: 512
|
|
300
|
+
thinking:
|
|
301
|
+
mechanism: always-on
|
|
302
|
+
guidance: |
|
|
303
|
+
Ornith emits a Qwen-style <think> block by default. The local llama.cpp
|
|
304
|
+
router is configured with --reasoning on, so the OpenAI-compatible
|
|
305
|
+
response separates this trace into reasoning_content while content
|
|
306
|
+
contains the final answer.
|
|
307
|
+
serving: "Ornith-1.0-35B is an agentic coding model; the card recommends 262K context, qwen3 reasoning parsing, qwen3_coder/qwen3_xml tool parsing, and sampler temperature=0.6, top_p=0.95, top_k=20."
|
|
308
|
+
|
|
309
|
+
- family: qwen3.6-27b
|
|
310
|
+
matchPatterns:
|
|
311
|
+
- qwen3.6-27b
|
|
312
|
+
- qwen3_6-27b
|
|
313
|
+
- qwen3-6-27b
|
|
314
|
+
capabilities:
|
|
315
|
+
chat: true
|
|
316
|
+
tools: true
|
|
317
|
+
toolCallFormat: qwen
|
|
318
|
+
reasoning: true
|
|
319
|
+
thinkingFormat: qwen-chat-template
|
|
320
|
+
structuredOutputs: json-schema
|
|
321
|
+
vision: true
|
|
322
|
+
audio: false
|
|
323
|
+
embeddings: false
|
|
324
|
+
rerank: false
|
|
325
|
+
fim: false
|
|
326
|
+
contextWindow: 262144
|
|
327
|
+
maxTokens: 32768
|
|
328
|
+
quirks:
|
|
329
|
+
sampling:
|
|
330
|
+
thinking:
|
|
331
|
+
temperature: 0.6
|
|
332
|
+
topP: 0.95
|
|
333
|
+
topK: 20
|
|
334
|
+
minP: 0.0
|
|
335
|
+
presencePenalty: 0.0
|
|
336
|
+
repetitionPenalty: 1.0
|
|
337
|
+
maxTokens: 32768
|
|
338
|
+
instruct:
|
|
339
|
+
temperature: 0.7
|
|
340
|
+
topP: 0.80
|
|
341
|
+
topK: 20
|
|
342
|
+
minP: 0.0
|
|
343
|
+
presencePenalty: 1.5
|
|
344
|
+
repetitionPenalty: 1.0
|
|
345
|
+
maxTokens: 32768
|
|
346
|
+
gpuTiers:
|
|
347
|
+
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 (Q4_K_XL weights ~19.5 GB plus ~32 GB KV)."
|
|
348
|
+
runtimePreference:
|
|
349
|
+
lmstudioNative: "Primary path; standard <think> tagging works correctly with the SDK classifier."
|
|
350
|
+
openaiCompat: "Verified parity for tool calls; reasoning_content surfaces and completion_tokens_details.reasoning_tokens is reported."
|
|
351
|
+
thinking:
|
|
352
|
+
mechanism: budget-tokens
|
|
353
|
+
budgetByLevel:
|
|
354
|
+
low: 1024
|
|
355
|
+
medium: 4096
|
|
356
|
+
high: 16384
|
|
357
|
+
guidance: |
|
|
358
|
+
Qwen chat template drives thinking via standard <think> blocks.
|
|
359
|
+
Coding sampler is the precise variant. Preserve thinking blocks across
|
|
360
|
+
turns when the workflow is iterative; the budget caps runaway chains.
|
|
361
|
+
serving: "Official Qwen card recommends max output 32768 tokens for standard queries (81920 only for math/code competitions). Coding sampler is the precise variant; preserve thinking blocks across turns for iterative workflows."
|
|
362
|
+
|
|
363
|
+
- family: gemma-4-31b-it-nvfp4-turbo
|
|
364
|
+
matchPatterns:
|
|
365
|
+
- gemma-4-31b-it-nvfp4-turbo
|
|
366
|
+
- gemma-4-31b-it-nvfp4
|
|
367
|
+
- gemma-4-31b-it-turbo
|
|
368
|
+
- gemma-4-31b-it
|
|
369
|
+
- gemma-4-31b
|
|
370
|
+
capabilities:
|
|
371
|
+
chat: true
|
|
372
|
+
tools: true
|
|
373
|
+
toolCallFormat: openai
|
|
374
|
+
reasoning: true
|
|
375
|
+
thinkingFormat: qwen-chat-template
|
|
376
|
+
structuredOutputs: json-schema
|
|
377
|
+
vision: true
|
|
378
|
+
audio: false
|
|
379
|
+
embeddings: false
|
|
380
|
+
rerank: false
|
|
381
|
+
fim: false
|
|
382
|
+
contextWindow: 122880
|
|
383
|
+
maxTokens: 32768
|
|
384
|
+
quirks:
|
|
385
|
+
kvCache:
|
|
386
|
+
kQuant: q8_0
|
|
387
|
+
vQuant: q8_0
|
|
388
|
+
sampling:
|
|
389
|
+
thinking:
|
|
390
|
+
temperature: 1.0
|
|
391
|
+
topP: 0.95
|
|
392
|
+
topK: 64
|
|
393
|
+
minP: 0.0
|
|
394
|
+
presencePenalty: 0.0
|
|
395
|
+
repetitionPenalty: 1.0
|
|
396
|
+
maxTokens: 32768
|
|
397
|
+
instruct:
|
|
398
|
+
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
399
|
+
topP: 0.95
|
|
400
|
+
topK: 64
|
|
401
|
+
minP: 0.0
|
|
402
|
+
maxTokens: 32768
|
|
403
|
+
chatTemplate: "gemma-channel"
|
|
404
|
+
leakageNote: |
|
|
405
|
+
LM Studio's gemma4 chat template emits chain-of-thought as plain text
|
|
406
|
+
bracketed by literal '<|channel>thought\n' and '<channel|>' markers
|
|
407
|
+
rather than reasoningType=reasoning fragments. Clio's lmstudio-native
|
|
408
|
+
adapter buffers and re-classifies these regions into ThinkingContent
|
|
409
|
+
so the markers do not leak into the visible TUI text and reasoning
|
|
410
|
+
tokens are accounted via the chars/4 estimator.
|
|
411
|
+
gpuTiers:
|
|
412
|
+
"48gb": "Fits at f16 KV when parallel=1 and ctx=122880; drop to q8_0 KV at parallel=4."
|
|
413
|
+
"80gb": "Comfortable at f16 KV with parallel=1 and ctx=122880."
|
|
414
|
+
runtimePreference:
|
|
415
|
+
lmstudioNative: "Verified path. Pin parallel=1 in the LM Studio per-model preset to give a single agent the full 122880 context."
|
|
416
|
+
openaiCompat: "Fallback path; the openai-compat surface does not expose the channel-marker classifier so visible text shows the raw markers."
|
|
417
|
+
thinking:
|
|
418
|
+
mechanism: on-off
|
|
419
|
+
guidance: |
|
|
420
|
+
Gemma 4 chat template exposes thinking only as on or off; intermediate
|
|
421
|
+
levels coerce to on. NVFP4 quantization can corrupt JSON tool-call
|
|
422
|
+
output under aggressive sampling, so the instruct profile pins
|
|
423
|
+
temperature at 0.7. The lmstudio-native adapter re-classifies the
|
|
424
|
+
channel-marker thought region into ThinkingContent so markers do not
|
|
425
|
+
leak into visible text.
|
|
426
|
+
serving: "LilaRest publishes no per-model sampler; values inherit the gemma family default per the official Google Gemma 4 31B card. KV cache budget at f16, ctx=122880 is roughly 57 GB on top of ~21 GB of NVFP4 weights."
|
|
427
|
+
|
|
428
|
+
# The qat/UD-Q4_K_XL/MTP build is the long-context llama.cpp gemma profile and is a
|
|
429
|
+
# different serving profile from the NVFP4 turbo family above: 262K context
|
|
430
|
+
# (not 122880), q8 KV, F16 gemma4v mmproj vision, and MTP speculative decoding.
|
|
431
|
+
# Its longest matchPattern (gemma-4-31b-it-qat-...) is longer than the NVFP4
|
|
432
|
+
# family's `gemma-4-31b-it` (14 chars), so the longest-substring matcher in
|
|
433
|
+
# knowledge-base.ts resolves this id here rather than to the NVFP4 profile.
|
|
434
|
+
- family: gemma-4-31b-it-qat-mtp
|
|
435
|
+
matchPatterns:
|
|
436
|
+
- gemma-4-31b-it-qat-ud-q4_k_xl-mtp-262k
|
|
437
|
+
- gemma-4-31b-it-qat-ud-q4_k_xl-mtp
|
|
438
|
+
- gemma-4-31b-it-qat-ud
|
|
439
|
+
- gemma-4-31b-it-qat
|
|
440
|
+
capabilities:
|
|
441
|
+
chat: true
|
|
442
|
+
tools: true
|
|
443
|
+
toolCallFormat: openai
|
|
444
|
+
reasoning: true
|
|
445
|
+
thinkingFormat: qwen-chat-template
|
|
446
|
+
structuredOutputs: json-schema
|
|
447
|
+
vision: true
|
|
448
|
+
audio: false
|
|
449
|
+
embeddings: false
|
|
450
|
+
rerank: false
|
|
451
|
+
fim: false
|
|
452
|
+
contextWindow: 262144
|
|
453
|
+
maxTokens: 32768
|
|
454
|
+
quirks:
|
|
455
|
+
kvCache:
|
|
456
|
+
kQuant: q8_0
|
|
457
|
+
vQuant: q8_0
|
|
458
|
+
sampling:
|
|
459
|
+
thinking:
|
|
460
|
+
temperature: 1.0
|
|
461
|
+
topP: 0.95
|
|
462
|
+
topK: 64
|
|
463
|
+
minP: 0.0
|
|
464
|
+
presencePenalty: 0.0
|
|
465
|
+
repetitionPenalty: 1.0
|
|
466
|
+
maxTokens: 32768
|
|
467
|
+
instruct:
|
|
468
|
+
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; router profiles may serve this qat build at 0.8
|
|
469
|
+
topP: 0.95
|
|
470
|
+
topK: 64
|
|
471
|
+
minP: 0.0
|
|
472
|
+
repeatPenalty: 1.05
|
|
473
|
+
maxTokens: 32768
|
|
474
|
+
chatTemplate: "gemma-channel"
|
|
475
|
+
gpuTiers:
|
|
476
|
+
"32gb": "32 GB Vulkan llama.cpp profile with UD-Q4_K_XL weights, ctx=262144, q8 KV, flash attention, parallel=1, F16 gemma4v mmproj (vision), and MTP draft (spec-type draft-mtp, n_max 4) both active."
|
|
477
|
+
runtimePreference:
|
|
478
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning off. Recommended local profile serves at ctx=262144, q8 KV, flash attention, parallel=1, batch=1024, ubatch=256, with F16 mmproj vision and draft-mtp speculative decoding; sampler temperature 0.8, top_p 0.95, top_k 64, repeat_penalty 1.05."
|
|
479
|
+
openaiCompat: "Fallback when a gateway fronts the same llama.cpp chat-completions surface; the openai-compat surface does not expose the channel-marker classifier, so visible text may show raw thought markers."
|
|
480
|
+
llamaCpp:
|
|
481
|
+
ctxSize: 262144
|
|
482
|
+
cacheTypeK: q8_0
|
|
483
|
+
cacheTypeV: q8_0
|
|
484
|
+
flashAttn: true
|
|
485
|
+
nGpuLayers: 99
|
|
486
|
+
parallel: 1
|
|
487
|
+
batchSize: 1024
|
|
488
|
+
ubatchSize: 256
|
|
489
|
+
mmproj: true
|
|
490
|
+
thinking:
|
|
491
|
+
mechanism: on-off
|
|
492
|
+
guidance: |
|
|
493
|
+
Gemma 4 31B IT QAT exposes thinking only as on or off via the gemma-4
|
|
494
|
+
enable_thinking template; intermediate levels coerce to on. A local
|
|
495
|
+
llama.cpp profile serves it with --reasoning off (default_reasoning off),
|
|
496
|
+
so the chain-of-thought is not requested; if the template emits one it
|
|
497
|
+
surfaces via reasoning_content and is accounted as reasoning tokens. The
|
|
498
|
+
QAT weights hold tool-call JSON together better than the NVFP4 turbo
|
|
499
|
+
build, but the instruct profile still pins a conservative temperature.
|
|
500
|
+
serving: "Gemma 4 31B IT QAT, UD-Q4_K_XL, served as a local llama.cpp router model (vulkan) with MTP speculative decoding (spec-type draft-mtp, n_max 4, draft mtp-gemma-4-31B-it) and F16 gemma4v mmproj vision both active at 262k. Reasoning off by default; gemma-4 thinking is on/off. Sampler temperature 0.8, top_p 0.95, top_k 64, repeat_penalty 1.05."
|
|
501
|
+
|
|
502
|
+
- family: gemopus-4-31b-it
|
|
503
|
+
matchPatterns:
|
|
504
|
+
- gemopus-4-31b-it
|
|
505
|
+
- gemopus-4-31b
|
|
506
|
+
- gemopus-4
|
|
507
|
+
capabilities:
|
|
508
|
+
chat: true
|
|
509
|
+
tools: true
|
|
510
|
+
toolCallFormat: openai
|
|
511
|
+
reasoning: true
|
|
512
|
+
thinkingFormat: qwen-chat-template
|
|
513
|
+
structuredOutputs: json-schema
|
|
514
|
+
vision: false
|
|
515
|
+
audio: false
|
|
516
|
+
embeddings: false
|
|
517
|
+
rerank: false
|
|
518
|
+
fim: false
|
|
519
|
+
contextWindow: 122880
|
|
520
|
+
maxTokens: 32768
|
|
521
|
+
quirks:
|
|
522
|
+
kvCache:
|
|
523
|
+
kQuant: q8_0
|
|
524
|
+
vQuant: q8_0
|
|
525
|
+
sampling:
|
|
526
|
+
thinking:
|
|
527
|
+
temperature: 1.0
|
|
528
|
+
topP: 0.95
|
|
529
|
+
topK: 64
|
|
530
|
+
minP: 0.0
|
|
531
|
+
presencePenalty: 0.0
|
|
532
|
+
repetitionPenalty: 1.0
|
|
533
|
+
maxTokens: 32768
|
|
534
|
+
instruct:
|
|
535
|
+
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
536
|
+
topP: 0.95
|
|
537
|
+
topK: 64
|
|
538
|
+
minP: 0.0
|
|
539
|
+
maxTokens: 32768
|
|
540
|
+
chatTemplate: "gemma-channel"
|
|
541
|
+
thinkingControl: |
|
|
542
|
+
Jackrong's distilled gemma-4 31B uses the same channel-marker template.
|
|
543
|
+
Thinking mode is gated by the literal '<|think|>' token in the system
|
|
544
|
+
prompt; Clio does not currently inject it, so the model emits an empty
|
|
545
|
+
thought region followed by the answer.
|
|
546
|
+
gpuTiers:
|
|
547
|
+
"48gb": "Fits at f16 KV with parallel=1; q8_0 KV recommended at parallel=4."
|
|
548
|
+
"80gb": "Comfortable at f16 KV with parallel=1 and ctx=122880."
|
|
549
|
+
runtimePreference:
|
|
550
|
+
lmstudioNative: "Verified path. Channel-marker handler in clio's adapter recovers reasoning tokens that the SDK classifier drops."
|
|
551
|
+
openaiCompat: "Fallback; reasoning_content is absent from the openai-compat reasoning probe for this family."
|
|
552
|
+
thinking:
|
|
553
|
+
mechanism: always-on
|
|
554
|
+
guidance: |
|
|
555
|
+
Distilled traces emit unconditionally. The thinking level setting is
|
|
556
|
+
ignored at the API surface; chain-of-thought arrives whether or not
|
|
557
|
+
Clio asks for it. Reasoning tokens are accounted via the channel-marker
|
|
558
|
+
re-classifier in lmstudio-native.
|
|
559
|
+
serving: "Distilled from Claude 4.6 Opus reasoning traces onto Google Gemma 4 31B; sampler matches the gemopus card values."
|
|
560
|
+
|
|
561
|
+
- family: qwopus3.6-27b-v1-preview
|
|
562
|
+
matchPatterns:
|
|
563
|
+
- qwopus3.6-27b-v1-preview
|
|
564
|
+
- qwopus3.6-27b
|
|
565
|
+
- qwopus3.6
|
|
566
|
+
capabilities:
|
|
567
|
+
chat: true
|
|
568
|
+
tools: true
|
|
569
|
+
toolCallFormat: qwen
|
|
570
|
+
reasoning: true
|
|
571
|
+
thinkingFormat: qwen-chat-template
|
|
572
|
+
structuredOutputs: json-schema
|
|
573
|
+
vision: false
|
|
574
|
+
audio: false
|
|
575
|
+
embeddings: false
|
|
576
|
+
rerank: false
|
|
577
|
+
fim: false
|
|
578
|
+
contextWindow: 262144
|
|
579
|
+
maxTokens: 32768
|
|
580
|
+
quirks:
|
|
581
|
+
sampling:
|
|
582
|
+
thinking:
|
|
583
|
+
temperature: 0.6
|
|
584
|
+
topP: 0.95
|
|
585
|
+
topK: 20
|
|
586
|
+
minP: 0.0
|
|
587
|
+
presencePenalty: 0.0
|
|
588
|
+
repetitionPenalty: 1.0
|
|
589
|
+
maxTokens: 32768
|
|
590
|
+
instruct:
|
|
591
|
+
temperature: 0.7
|
|
592
|
+
topP: 0.80
|
|
593
|
+
topK: 20
|
|
594
|
+
minP: 0.0
|
|
595
|
+
presencePenalty: 1.5
|
|
596
|
+
repetitionPenalty: 1.0
|
|
597
|
+
maxTokens: 32768
|
|
598
|
+
gpuTiers:
|
|
599
|
+
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 thanks to the Q4_K_M weights (~16.5 GB)."
|
|
600
|
+
runtimePreference:
|
|
601
|
+
lmstudioNative: "Primary path; thinking is emitted via standard <think> tags so the SDK classifier accounts reasoning correctly."
|
|
602
|
+
openaiCompat: "Verified for tool-call extraction parity; reasoning_content surfaces."
|
|
603
|
+
thinking:
|
|
604
|
+
mechanism: budget-tokens
|
|
605
|
+
budgetByLevel:
|
|
606
|
+
low: 1024
|
|
607
|
+
medium: 4096
|
|
608
|
+
high: 16384
|
|
609
|
+
guidance: |
|
|
610
|
+
Distilled Opus reasoning traces over Qwen3.6 27B; thinking emits via
|
|
611
|
+
standard <think> tags. Coding-task sampler is the precise variant.
|
|
612
|
+
Preserve thinking blocks across turns for iterative workflows.
|
|
613
|
+
serving: "Distilled from Claude 4.6 Opus reasoning traces onto Qwen3.6 27B. Coding-task sampler is the precise variant: temperature 0.6, top_p 0.95, top_k 20."
|
|
614
|
+
|
|
615
|
+
- family: qwopus3.6-35b-a3b-coder
|
|
616
|
+
matchPatterns:
|
|
617
|
+
- qwopus3.6-35b-a3b-coder-mtp-q4_k_m-262k
|
|
618
|
+
- qwopus3.6-35b-a3b-coder-mtp
|
|
619
|
+
- qwopus3.6-35b-a3b-coder
|
|
620
|
+
- qwopus3.6-35b-a3b
|
|
621
|
+
capabilities:
|
|
622
|
+
chat: true
|
|
623
|
+
tools: true
|
|
624
|
+
toolCallFormat: qwen
|
|
625
|
+
# Reasoning class: effort-levels, measured against an LM Studio host on
|
|
626
|
+
# 2026-08-08 over its OpenAI-compatible surface. The card
|
|
627
|
+
# frames the Coder-MTP line as thinking-off execution, but the wire
|
|
628
|
+
# disagrees: on "what is 17+25" this model spends 98 of 103 completion
|
|
629
|
+
# tokens on reasoning, and a wiki planning dispatch spent 89,501. It is an
|
|
630
|
+
# always-reasoning model unless told otherwise.
|
|
631
|
+
#
|
|
632
|
+
# chat_template_kwargs enable_thinking is inert here (verified with unique
|
|
633
|
+
# prompts to rule out cache hits). reasoning_effort "none" is the control
|
|
634
|
+
# that works, taking the same prompt from 98 reasoning tokens to 0, which
|
|
635
|
+
# is why `off` is mapped explicitly below rather than left unsent.
|
|
636
|
+
#
|
|
637
|
+
# reasoning=false previously resolved to mechanism "none", and mechanism
|
|
638
|
+
# "none" makes openai-completions strip reasoning_effort from the payload.
|
|
639
|
+
# The flag set to suppress thinking was the reason it could not be
|
|
640
|
+
# suppressed.
|
|
641
|
+
reasoning: true
|
|
642
|
+
thinkingFormat: qwen-chat-template
|
|
643
|
+
structuredOutputs: json-schema
|
|
644
|
+
vision: true
|
|
645
|
+
audio: false
|
|
646
|
+
embeddings: false
|
|
647
|
+
rerank: false
|
|
648
|
+
fim: false
|
|
649
|
+
contextWindow: 262144
|
|
650
|
+
maxTokens: 32768
|
|
651
|
+
quirks:
|
|
652
|
+
sampling:
|
|
653
|
+
instruct:
|
|
654
|
+
temperature: 0.2
|
|
655
|
+
topP: 0.9
|
|
656
|
+
topK: 20
|
|
657
|
+
repeatPenalty: 1.05
|
|
658
|
+
# Upstream Qwen3.6 card recommends presence_penalty 1.5 for
|
|
659
|
+
# non-thinking modes to suppress repetition loops. Measured on live
|
|
660
|
+
# dispatch (2026-07-06): without it a coder worker repeated one
|
|
661
|
+
# code_nav call into the loop-guard abort on 3 of 3 runs; with it the
|
|
662
|
+
# same task passed 3 of 3 with clean edit-then-validate trajectories.
|
|
663
|
+
presencePenalty: 1.5
|
|
664
|
+
maxTokens: 32768
|
|
665
|
+
gpuTiers:
|
|
666
|
+
"32gb": "Default 32 GB local profile: Q4_K_M, ctx=262144, q8 KV, flash attention, parallel=1, with F32 mmproj (vision) and draft-mtp speculative decoding both active. Leaves several GiB free at 262k on a 32 GiB class card."
|
|
667
|
+
runtimePreference:
|
|
668
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile serves text-only with reasoning off and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
669
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen tool parsing."
|
|
670
|
+
llamaCpp:
|
|
671
|
+
ctxSize: 262144
|
|
672
|
+
cacheTypeK: q8_0
|
|
673
|
+
cacheTypeV: q8_0
|
|
674
|
+
flashAttn: true
|
|
675
|
+
nGpuLayers: 99
|
|
676
|
+
parallel: 1
|
|
677
|
+
batchSize: 2048
|
|
678
|
+
ubatchSize: 512
|
|
679
|
+
mmproj: true
|
|
680
|
+
chatTemplateKwargs:
|
|
681
|
+
enable_thinking: false
|
|
682
|
+
thinking:
|
|
683
|
+
mechanism: effort-levels
|
|
684
|
+
effortByLevel:
|
|
685
|
+
# "none" is the only value that silences this family. It is carried on
|
|
686
|
+
# the wire because the model reasons by default, so sending nothing is
|
|
687
|
+
# in effect a request to keep reasoning.
|
|
688
|
+
off: none
|
|
689
|
+
low: low
|
|
690
|
+
medium: medium
|
|
691
|
+
high: high
|
|
692
|
+
guidance: |
|
|
693
|
+
Qwopus3.6 35B-A3B Coder reasons unless reasoning_effort is set to
|
|
694
|
+
"none", despite the creator's card describing the Coder-MTP variants as
|
|
695
|
+
thinking-off. That dial only reaches the wire on openai-compat; the LM
|
|
696
|
+
Studio native transport has no field for it, so a native-routed run
|
|
697
|
+
reasons at full rate whatever the configured thinking level says.
|
|
698
|
+
serving: "Jackrong Qwopus3.6 35B-A3B Coder MTP, Q4_K_M. Thinking-off coder/agent tuned for tool-use loops. Served as a local router default with MTP speculative decoding (spec-type draft-mtp, n_max 2) and F32 mmproj vision both active at 262k."
|
|
699
|
+
|
|
700
|
+
- family: qwopus3.6-27b-coder
|
|
701
|
+
matchPatterns:
|
|
702
|
+
- qwopus3.6-27b-coder-mtp-q5_k_m-262k
|
|
703
|
+
- qwopus3.6-27b-coder-mtp
|
|
704
|
+
- qwopus3.6-27b-coder
|
|
705
|
+
capabilities:
|
|
706
|
+
chat: true
|
|
707
|
+
tools: true
|
|
708
|
+
toolCallFormat: qwen
|
|
709
|
+
# Reasoning class: never, same Coder-MTP thinking-off design as the 35B.
|
|
710
|
+
reasoning: false
|
|
711
|
+
thinkingFormat: qwen-chat-template
|
|
712
|
+
structuredOutputs: json-schema
|
|
713
|
+
vision: false
|
|
714
|
+
audio: false
|
|
715
|
+
embeddings: false
|
|
716
|
+
rerank: false
|
|
717
|
+
fim: false
|
|
718
|
+
contextWindow: 262144
|
|
719
|
+
maxTokens: 32768
|
|
720
|
+
quirks:
|
|
721
|
+
sampling:
|
|
722
|
+
instruct:
|
|
723
|
+
temperature: 0.2
|
|
724
|
+
topP: 0.9
|
|
725
|
+
topK: 20
|
|
726
|
+
repeatPenalty: 1.05
|
|
727
|
+
# Same Coder-MTP non-thinking design as the 35B; see the measured
|
|
728
|
+
# presence-penalty note there.
|
|
729
|
+
presencePenalty: 1.5
|
|
730
|
+
maxTokens: 32768
|
|
731
|
+
runtimePreference:
|
|
732
|
+
llamaCpp: "Mini router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
733
|
+
lmstudioNative: "Dense 27B Coder-MTP; the creator's template ships thinking-off, and no per-request field re-enables it."
|
|
734
|
+
thinking:
|
|
735
|
+
mechanism: none
|
|
736
|
+
guidance: |
|
|
737
|
+
Qwopus3.6 27B Coder-MTP is the dense thinking-off coder variant; the
|
|
738
|
+
dial clamps to off at every level.
|
|
739
|
+
serving: "Jackrong Qwopus3.6 27B Coder MTP, Q5_K_M. Thinking-off dense coder for tool-use loops."
|
|
740
|
+
|
|
741
|
+
- family: qwopus3.5-9b-coder
|
|
742
|
+
matchPatterns:
|
|
743
|
+
- qwopus3.5-9b-coder-q8_0-262k
|
|
744
|
+
- qwopus3.5-9b-coder
|
|
745
|
+
capabilities:
|
|
746
|
+
chat: true
|
|
747
|
+
tools: true
|
|
748
|
+
toolCallFormat: qwen
|
|
749
|
+
# Reasoning class: never (Coder thinking-off design, 3.5 generation).
|
|
750
|
+
reasoning: false
|
|
751
|
+
thinkingFormat: qwen-chat-template
|
|
752
|
+
structuredOutputs: json-schema
|
|
753
|
+
vision: false
|
|
754
|
+
audio: false
|
|
755
|
+
embeddings: false
|
|
756
|
+
rerank: false
|
|
757
|
+
fim: false
|
|
758
|
+
contextWindow: 262144
|
|
759
|
+
maxTokens: 32768
|
|
760
|
+
quirks:
|
|
761
|
+
sampling:
|
|
762
|
+
instruct:
|
|
763
|
+
temperature: 0.2
|
|
764
|
+
topP: 0.9
|
|
765
|
+
topK: 20
|
|
766
|
+
repeatPenalty: 1.05
|
|
767
|
+
maxTokens: 32768
|
|
768
|
+
runtimePreference:
|
|
769
|
+
llamaCpp: "Mini router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
770
|
+
thinking:
|
|
771
|
+
mechanism: none
|
|
772
|
+
guidance: |
|
|
773
|
+
Qwopus3.5 9B Coder is the small thinking-off coder; the dial clamps
|
|
774
|
+
to off at every level.
|
|
775
|
+
serving: "Jackrong Qwopus3.5 9B Coder, Q8_0. Thinking-off small coder for fast tool loops."
|
|
776
|
+
|
|
777
|
+
- family: gemma4-26b-a4b
|
|
778
|
+
matchPatterns:
|
|
779
|
+
- gemma-4-26b-a4b
|
|
780
|
+
- gemma4-26b-a4b
|
|
781
|
+
- gemma-4-26b-a4b-it
|
|
782
|
+
- gemma4-26b-a4b-it
|
|
783
|
+
- gemma-4-26b-a4b-it-q4
|
|
784
|
+
- gemma-4-26b-a4b-it-q4_k_m
|
|
785
|
+
capabilities:
|
|
786
|
+
chat: true
|
|
787
|
+
tools: true
|
|
788
|
+
toolCallFormat: openai
|
|
789
|
+
reasoning: true
|
|
790
|
+
thinkingFormat: qwen-chat-template
|
|
791
|
+
structuredOutputs: json-schema
|
|
792
|
+
vision: true
|
|
793
|
+
audio: false
|
|
794
|
+
embeddings: false
|
|
795
|
+
rerank: false
|
|
796
|
+
fim: false
|
|
797
|
+
contextWindow: 262144
|
|
798
|
+
maxTokens: 65536
|
|
799
|
+
quirks:
|
|
800
|
+
sampling:
|
|
801
|
+
thinking:
|
|
802
|
+
temperature: 0.3
|
|
803
|
+
topP: 0.9
|
|
804
|
+
topK: 20
|
|
805
|
+
maxTokens: 8192
|
|
806
|
+
reasoningBudget: 4096
|
|
807
|
+
instruct:
|
|
808
|
+
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
809
|
+
topP: 0.95
|
|
810
|
+
topK: 64
|
|
811
|
+
minP: 0.0
|
|
812
|
+
gpuTiers:
|
|
813
|
+
"32gb": "Main-agent candidate for 32 GB and larger local systems when Gemma behavior is preferred."
|
|
814
|
+
runtimePreference:
|
|
815
|
+
lmstudioNative: "Use for already-loaded LM Studio sessions and multimodal checks."
|
|
816
|
+
openaiCompat: "Preferred fallback when SDK rawTools do not match the template."
|
|
817
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja, --reasoning on, and the matching mmproj."
|
|
818
|
+
llamaCpp:
|
|
819
|
+
ctxSize: 262144
|
|
820
|
+
cacheTypeK: q8_0
|
|
821
|
+
cacheTypeV: q8_0
|
|
822
|
+
flashAttn: true
|
|
823
|
+
nGpuLayers: 99
|
|
824
|
+
mmproj: true
|
|
825
|
+
thinkingControl: "Gemma 4 exposes thinking through chat template enable_thinking support."
|
|
826
|
+
thinking:
|
|
827
|
+
mechanism: on-off
|
|
828
|
+
guidance: |
|
|
829
|
+
Gemma 4 chat template exposes thinking only as on or off via
|
|
830
|
+
enable_thinking. Intermediate levels coerce to on. The MoE A4B variant
|
|
831
|
+
keeps the same template behavior as the dense gemma-4 family.
|
|
832
|
+
|
|
833
|
+
- family: nemotron-cascade-2-30b-a3b
|
|
834
|
+
matchPatterns:
|
|
835
|
+
- nemotron-cascade-2
|
|
836
|
+
- nemotron-cascade-2-30b-a3b
|
|
837
|
+
- nemotron-cascade-2-30b-a3b-i1
|
|
838
|
+
- nemotron-cascade-2-30b-a3b-i1-q4_k_m
|
|
839
|
+
- nemotron-cascade2
|
|
840
|
+
- nemotron-cascade-2-30b
|
|
841
|
+
capabilities:
|
|
842
|
+
chat: true
|
|
843
|
+
tools: true
|
|
844
|
+
toolCallFormat: qwen
|
|
845
|
+
reasoning: true
|
|
846
|
+
thinkingFormat: qwen-chat-template
|
|
847
|
+
structuredOutputs: json-schema
|
|
848
|
+
vision: false
|
|
849
|
+
audio: false
|
|
850
|
+
embeddings: false
|
|
851
|
+
rerank: false
|
|
852
|
+
fim: false
|
|
853
|
+
contextWindow: 1048576
|
|
854
|
+
maxTokens: 65536
|
|
855
|
+
quirks:
|
|
856
|
+
sampling:
|
|
857
|
+
instruct:
|
|
858
|
+
temperature: 0.3
|
|
859
|
+
topP: 0.9
|
|
860
|
+
topK: 20
|
|
861
|
+
maxTokens: 8192
|
|
862
|
+
gpuTiers:
|
|
863
|
+
"32gb": "Preferred local worker model when a text-only worker is enough."
|
|
864
|
+
runtimePreference:
|
|
865
|
+
lmstudioNative: "Use when LM Studio SDK tool extraction is verified for the loaded template."
|
|
866
|
+
openaiCompat: "Preferred fallback for LM Studio gateway tool calls."
|
|
867
|
+
anthropicCompat: "Do not select for the current llama.cpp target profile; /v1/messages returned 404 while /v1/models and OpenAI-compatible chat worked."
|
|
868
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Recommended local profile disables thinking in chat_template_kwargs."
|
|
869
|
+
llamaCpp:
|
|
870
|
+
ctxSize: 1048576
|
|
871
|
+
cacheTypeK: q8_0
|
|
872
|
+
cacheTypeV: q8_0
|
|
873
|
+
flashAttn: true
|
|
874
|
+
nGpuLayers: 99
|
|
875
|
+
parallel: 4
|
|
876
|
+
chatTemplateKwargs:
|
|
877
|
+
enable_thinking: false
|
|
878
|
+
thinking:
|
|
879
|
+
mechanism: on-off
|
|
880
|
+
guidance: |
|
|
881
|
+
Nemotron Cascade 2 chat template exposes thinking through
|
|
882
|
+
chat_template_kwargs.enable_thinking. Intermediate levels coerce to on.
|
|
883
|
+
The recommended llama.cpp profile disables thinking in the template kwargs by
|
|
884
|
+
default; flip via the runtime override when reasoning is needed.
|
|
885
|
+
serving: "NVIDIA card recommends vLLM with nemotron_v3 reasoning and qwen3_coder tool parsing."
|
|
886
|
+
|
|
887
|
+
- family: qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled
|
|
888
|
+
matchPatterns:
|
|
889
|
+
- qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled
|
|
890
|
+
- qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled-i1
|
|
891
|
+
- qwen3.5-35b-a3b-claude-4.6-opus-reasoning-distilled-i1-q4_k_m
|
|
892
|
+
- qwen35-distilled
|
|
893
|
+
- qwen35-distilled-i1
|
|
894
|
+
- qwen35-distilled-i1-q4_k_m
|
|
895
|
+
capabilities:
|
|
896
|
+
chat: true
|
|
897
|
+
tools: true
|
|
898
|
+
toolCallFormat: qwen
|
|
899
|
+
reasoning: true
|
|
900
|
+
thinkingFormat: qwen-chat-template
|
|
901
|
+
structuredOutputs: json-schema
|
|
902
|
+
vision: false
|
|
903
|
+
audio: false
|
|
904
|
+
embeddings: false
|
|
905
|
+
rerank: false
|
|
906
|
+
fim: false
|
|
907
|
+
contextWindow: 262144
|
|
908
|
+
maxTokens: 32768
|
|
909
|
+
quirks:
|
|
910
|
+
sampling:
|
|
911
|
+
thinking:
|
|
912
|
+
temperature: 0.6
|
|
913
|
+
topP: 0.95
|
|
914
|
+
topK: 20
|
|
915
|
+
maxTokens: 16384
|
|
916
|
+
reasoningBudget: 4096
|
|
917
|
+
instruct:
|
|
918
|
+
temperature: 0.2
|
|
919
|
+
topP: 0.9
|
|
920
|
+
topK: 20
|
|
921
|
+
maxTokens: 4096
|
|
922
|
+
gpuTiers:
|
|
923
|
+
"32gb": "Distilled Opus-style reasoning at Qwen3.5 35B A3B scale. Use as a strong local main agent when GPU room is available."
|
|
924
|
+
runtimePreference:
|
|
925
|
+
lmstudioNative: "Verified usable on LM Studio when loaded with the Qwen-style preset."
|
|
926
|
+
openaiCompat: "Preferred fallback when SDK rawTools yield empty arguments for the distilled template."
|
|
927
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Match qwen3 reasoning parser flags upstream."
|
|
928
|
+
thinking:
|
|
929
|
+
mechanism: budget-tokens
|
|
930
|
+
budgetByLevel:
|
|
931
|
+
low: 1024
|
|
932
|
+
medium: 4096
|
|
933
|
+
high: 16384
|
|
934
|
+
guidance: |
|
|
935
|
+
Distilled Opus reasoning over Qwen3.5 35B A3B; thinking emits via the
|
|
936
|
+
qwen chat template. Preserve thinking blocks across coding-task turns;
|
|
937
|
+
the budget caps runaway chains while keeping iterative reasoning
|
|
938
|
+
coherent.
|
|
939
|
+
serving: "Distilled from Claude 4.6 Opus reasoning traces; expect stronger thinking-then-answer behavior than vanilla Qwen3.5 35B A3B."
|
|
940
|
+
|
|
941
|
+
- family: qwopus3.5-9b-v3
|
|
942
|
+
matchPatterns:
|
|
943
|
+
- qwopus3.5-9b-v3
|
|
944
|
+
- qwopus-9b
|
|
945
|
+
- qwopus 9b
|
|
946
|
+
capabilities:
|
|
947
|
+
chat: true
|
|
948
|
+
tools: true
|
|
949
|
+
toolCallFormat: qwen
|
|
950
|
+
reasoning: true
|
|
951
|
+
thinkingFormat: qwen-chat-template
|
|
952
|
+
structuredOutputs: json-schema
|
|
953
|
+
vision: false
|
|
954
|
+
audio: false
|
|
955
|
+
embeddings: false
|
|
956
|
+
rerank: false
|
|
957
|
+
fim: false
|
|
958
|
+
contextWindow: 262144
|
|
959
|
+
maxTokens: 32768
|
|
960
|
+
quirks:
|
|
961
|
+
sampling:
|
|
962
|
+
instruct:
|
|
963
|
+
temperature: 0.2
|
|
964
|
+
topP: 0.9
|
|
965
|
+
topK: 20
|
|
966
|
+
maxTokens: 4096
|
|
967
|
+
thinking:
|
|
968
|
+
temperature: 0.6
|
|
969
|
+
topP: 0.95
|
|
970
|
+
topK: 20
|
|
971
|
+
maxTokens: 8192
|
|
972
|
+
reasoningBudget: 2048
|
|
973
|
+
gpuTiers:
|
|
974
|
+
"16gb": "Use this as the local 16GB emulation target. It keeps the output cap below the 32GB-class MoE defaults."
|
|
975
|
+
runtimePreference:
|
|
976
|
+
lmstudioNative: "Primary local laptop path. Verified through LM Studio /api/v1/models with 262144 loaded context, flash attention, GPU KV offload, and trained_for_tool_use."
|
|
977
|
+
openaiCompat: "Verified through the LM Studio gateway for read and artifact tool calls; use when the native SDK produces thinking-only length stops or malformed tools."
|
|
978
|
+
anthropicCompat: "LM Studio's Anthropic Messages surface streams simple text with the server root URL as baseUrl; keep as a protocol fallback until tool and reasoning behavior are proven."
|
|
979
|
+
thinking:
|
|
980
|
+
# Measured 2026-08-11 against LM Studio on this exact model: baseline 56
|
|
981
|
+
# reasoning tokens, reasoning_effort none 120, reasoning_effort minimal
|
|
982
|
+
# 243, chat_template_kwargs.enable_thinking false 45, both together 44.
|
|
983
|
+
# No spelling silences it, and the budget-tokens mechanism it used to
|
|
984
|
+
# claim is informational on both LM Studio and llama.cpp, so nothing ever
|
|
985
|
+
# reached the wire. Always-on is what the model actually does.
|
|
986
|
+
mechanism: always-on
|
|
987
|
+
guidance: |
|
|
988
|
+
Qwopus 9B distilled from Opus reasoning; the chain-of-thought comes out
|
|
989
|
+
of the chat template unconditionally and the dial cannot stop it, so
|
|
990
|
+
Clio shows the level as forced. Output cap is tighter than the
|
|
991
|
+
32GB-class MoE families because this is the 16GB emulation target;
|
|
992
|
+
budget for a reasoning preamble ahead of every answer.
|
|
993
|
+
serving: "LM Studio reports trained_for_tool_use for the local qwopus3.5-9b-v3 target. Native model override successfully switched residency to granite-4.0-h-350m through the SDK."
|