@iowarp/clio-coder 0.4.1 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +92 -0
- package/CONTRIBUTING.md +59 -36
- package/README.md +404 -472
- package/SECURITY.md +2 -1
- package/dist/{acp-ZILU3AUO.js → acp-TMDQZDIG.js} +7 -7
- package/dist/{agents-HYWGBGQR.js → agents-5N5NG3XG.js} +28 -28
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-N3QT7CBO.js → auth-Z5CCBXKQ.js} +8 -9
- package/dist/{builtins-UJLMOVOV.js → builtins-K6TNDT24.js} +4 -4
- package/dist/{chunk-GVQJ5CCZ.js → chunk-2HFQNRV3.js} +7 -7
- package/dist/{chunk-QMXC4JB7.js → chunk-2NHR3NAY.js} +163 -1401
- package/dist/chunk-2X4RYJTJ.js +39 -0
- package/dist/{chunk-Y45G3AXC.js → chunk-2Z2IKEXI.js} +6 -10
- package/dist/{chunk-EIMVLWB3.js → chunk-34BHNEE3.js} +7 -3
- package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
- package/dist/chunk-3EBYEESD.js +314 -0
- package/dist/{chunk-CTJ4RNAA.js → chunk-3F7VUY77.js} +2 -2
- package/dist/{chunk-AP73CFDC.js → chunk-3KIPBMUA.js} +2 -2
- package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
- package/dist/{chunk-VEGN6WIQ.js → chunk-462T4EGZ.js} +2 -2
- package/dist/{chunk-AFKWHWXF.js → chunk-4JDLP6ZS.js} +33 -16
- package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
- package/dist/{chunk-AKB4GYDL.js → chunk-54ODD65L.js} +5 -5
- package/dist/{chunk-BBTJOK6Y.js → chunk-5KW52TEP.js} +3 -3
- package/dist/{chunk-6CCS4G3W.js → chunk-5PFYMY2V.js} +2 -2
- package/dist/chunk-77QIVUZB.js +1334 -0
- package/dist/{chunk-7OBGU7UB.js → chunk-7BHIY2MW.js} +7 -13
- package/dist/{chunk-3QSOM6PA.js → chunk-AZ4WMN4W.js} +2 -2
- package/dist/{chunk-6NJQITNH.js → chunk-B74PXLU7.js} +6 -3
- package/dist/{chunk-R23Z6K6I.js → chunk-B7HM5Z7T.js} +15 -15
- package/dist/{chunk-R32CLGZ6.js → chunk-BO7Y52RY.js} +81 -20
- package/dist/{chunk-UEDMSP56.js → chunk-BYMNWQ7O.js} +123 -148
- package/dist/{chunk-ZJLUDYFY.js → chunk-CRFOIAX3.js} +4 -4
- package/dist/{chunk-2NM363SV.js → chunk-CYZW7JHJ.js} +7 -7
- package/dist/{chunk-6HMJX2VU.js → chunk-DYHAXKHD.js} +38 -10
- package/dist/{chunk-THYWACCR.js → chunk-DZAW46HP.js} +3 -3
- package/dist/{chunk-FYUN5KZ3.js → chunk-DZEK6CJN.js} +17 -17
- package/dist/{chunk-3I5NY75V.js → chunk-E7GT7O5N.js} +5 -5
- package/dist/{chunk-VKFQTNDV.js → chunk-F2I26BDK.js} +4 -4
- package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
- package/dist/{chunk-IXJT6DCX.js → chunk-FVDGR2ZL.js} +3 -3
- package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
- package/dist/{chunk-7EPLI7VL.js → chunk-HIICAHCJ.js} +2 -2
- package/dist/{chunk-E67WX76H.js → chunk-HKMD33FO.js} +29 -80
- package/dist/chunk-HLAFFSEK.js +360 -0
- package/dist/{chunk-UAPGZHYC.js → chunk-I64IFBLB.js} +9 -2
- package/dist/{chunk-XKA2ICR3.js → chunk-I66ZTYNP.js} +440 -175
- package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
- package/dist/{chunk-PVAMAVBB.js → chunk-IDNA72AH.js} +102 -2
- package/dist/{chunk-GCSMB2KY.js → chunk-IKOZFYBN.js} +1 -1
- package/dist/{chunk-2VG7KLYV.js → chunk-IKSLQ4XV.js} +5460 -3241
- package/dist/{chunk-QKIFBZKT.js → chunk-IMXMHHMQ.js} +166 -25
- package/dist/{chunk-74YWRRU5.js → chunk-JBCS7CRR.js} +2 -2
- package/dist/{chunk-BDPT6GTK.js → chunk-JWJGP5DQ.js} +2 -2
- package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
- package/dist/chunk-KPXDY6QF.js +47 -0
- package/dist/{chunk-ABLSQ6JX.js → chunk-LJID3DYZ.js} +7 -1
- package/dist/{chunk-VKRH2TCS.js → chunk-M2DAX4F6.js} +2 -2
- package/dist/{chunk-6I5ILFOF.js → chunk-M2WXEHER.js} +2 -2
- package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
- package/dist/{chunk-N5UK64DP.js → chunk-MCMZMDAC.js} +2 -2
- package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
- package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
- package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
- package/dist/{chunk-HUAS7ITX.js → chunk-O3YUNJZ2.js} +13 -21
- package/dist/{chunk-MA3H6DM5.js → chunk-P75RZCJW.js} +25 -3
- package/dist/{chunk-IG7BCQBA.js → chunk-PGF63K6I.js} +2 -2
- package/dist/chunk-PJX3WQUQ.js +42 -0
- package/dist/{chunk-6DWBAZ5U.js → chunk-Q4XWMHX6.js} +4 -6
- package/dist/{chunk-OJTRZGR3.js → chunk-QQLGQY2A.js} +8 -8
- package/dist/{chunk-J4HBWF6Y.js → chunk-RLYRBIYQ.js} +115 -20
- package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
- package/dist/{chunk-C537JADH.js → chunk-SSEYRH53.js} +6 -7
- package/dist/chunk-SZAA6XDG.js +30 -0
- package/dist/{chunk-MOPSG2X7.js → chunk-TPEQIQIE.js} +6 -6
- package/dist/{chunk-JA5QWE4Z.js → chunk-UBRFI4HS.js} +1879 -1650
- package/dist/{chunk-BTGG6BG2.js → chunk-UH347SHR.js} +154 -15
- package/dist/{chunk-5YHDIDBP.js → chunk-UH632ZYL.js} +2 -2
- package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
- package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
- package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
- package/dist/{chunk-UXN6JT4W.js → chunk-W4YEMFBX.js} +2 -2
- package/dist/{chunk-TD3PGPQA.js → chunk-W6NIE6OW.js} +2 -2
- package/dist/{chunk-TVH4ONAM.js → chunk-X7IARSHT.js} +3 -3
- package/dist/{chunk-PJJ6MY27.js → chunk-XE3PCIXH.js} +3 -3
- package/dist/{chunk-FEFIFZTL.js → chunk-XGDPUNND.js} +2 -2
- package/dist/{chunk-SCYB3HA4.js → chunk-XOXV5GKE.js} +51 -16
- package/dist/{chunk-QTFGO774.js → chunk-XQRY4DTA.js} +24 -11
- package/dist/{chunk-BJGUKIG4.js → chunk-YJISEZKC.js} +2 -2
- package/dist/{chunk-GPPB3JBE.js → chunk-ZGNYYXQ6.js} +2 -2
- package/dist/{chunk-SINK3QR6.js → chunk-ZNT2M6TG.js} +7 -7
- package/dist/{chunk-7RY5VZPH.js → chunk-ZW4HH5JJ.js} +6 -6
- package/dist/cli/index.js +33 -32
- package/dist/{clio-IT3G3VQH.js → clio-7VB377CC.js} +7 -7
- package/dist/{code-nav-RK6S7F6E.js → code-nav-YVLCYA7V.js} +85 -17
- package/dist/{config-3QZRWZJF.js → config-4HVOS65E.js} +88 -43
- package/dist/{configure-FL7Y3KJF.js → configure-PIWO7B24.js} +10 -10
- package/dist/{context-5HE7ODYK.js → context-IYEHL3WQ.js} +33 -31
- package/dist/{context-XNHL75JV.js → context-KQYIWPWT.js} +47 -34
- package/dist/{context-KYQFRVDC.js → context-N6ZE3LGJ.js} +11 -11
- package/dist/{context-clear-N545L53A.js → context-clear-G4OGZJDS.js} +33 -31
- package/dist/{context-working-set-QHKXSV2F.js → context-working-set-BWLF6LJP.js} +7 -7
- package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-2QQAITS3.js} +38 -38
- package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
- package/dist/{doctor-ZGPEGHIP.js → doctor-LHBD36VU.js} +23 -22
- package/dist/{eval-GXLL44RD.js → eval-C45FYRJ6.js} +21 -20
- package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-6DEJPLBF.js} +2 -2
- package/dist/{evidence-HWLBRH3Q.js → evidence-6SHONYAF.js} +30 -28
- package/dist/{evolve-FTZBMNVW.js → evolve-KRKMV72X.js} +30 -28
- package/dist/{extensions-VHRBEID7.js → extensions-KPZ2UHBB.js} +5 -3
- package/dist/{fleet-CKZHJWZJ.js → fleet-IVTCKDHT.js} +62 -61
- package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-EDWL3IT7.js} +5 -5
- package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-YP3YEFGK.js} +4 -4
- package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-ZFWKHY2M.js} +16 -14
- package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-FVUNCBML.js} +31 -29
- package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-UN5XED4R.js} +2 -2
- package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-XOWC4HSX.js} +17 -15
- package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-UN3SODEL.js} +30 -28
- package/dist/{fleet-view-WAMJYNDT.js → fleet-view-TWHJKCN6.js} +31 -29
- package/dist/{init-5XQRBOFV.js → init-T2QORQ3Y.js} +50 -49
- package/dist/{interop-34TVO25M.js → interop-IN5I2A66.js} +5 -5
- package/dist/{library-3QY6KF57.js → library-LSCATDLZ.js} +15 -13
- package/dist/{memory-L4UTIIIW.js → memory-HYOKAGGJ.js} +31 -29
- package/dist/{models-ZVX3QOWE.js → models-2GPMFYCM.js} +22 -21
- package/dist/{monitor-CEKVSYTS.js → monitor-E4ASVUJH.js} +34 -32
- package/dist/{orchestrator-77BAP6BC.js → orchestrator-DDMPR3PY.js} +984 -583
- package/dist/{panes-7STHOAUJ.js → panes-E3RUXOW5.js} +4 -4
- package/dist/{panes-SHAUIRXY.js → panes-IXKLOKA2.js} +23 -8
- package/dist/{reset-EOLM7GVE.js → reset-OAQP3W4O.js} +4 -4
- package/dist/{resources-74GKTLSF.js → resources-OTRSN34L.js} +15 -13
- package/dist/{run-HBAUJNNZ.js → run-5DEYH5QK.js} +60 -59
- package/dist/{share-G3APVLVP.js → share-IHWTLO3M.js} +19 -15
- package/dist/{skills-35HHUKCR.js → skills-IYMXMKW4.js} +17 -15
- package/dist/{skills-eval-QN4HSHDC.js → skills-eval-DROHSJAR.js} +36 -36
- package/dist/{skills-inventory-J357J34F.js → skills-inventory-D7X4L4ZX.js} +15 -13
- package/dist/{slash-commands-JZZCQA32.js → slash-commands-QBM7UZ3B.js} +21 -18
- package/dist/{steer-XAVHJM22.js → steer-Z5DO23FJ.js} +2 -2
- package/dist/{targets-DSM6CY3M.js → targets-P2FUC4IL.js} +25 -28
- package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-YREJ3JX2.js} +5 -5
- package/dist/{tools-MKNWVPBH.js → tools-5B7RO6MV.js} +4 -4
- package/dist/{trace-ECQ7TIYZ.js → trace-YMGMUM6A.js} +55 -7
- package/dist/{upgrade-H7TOM7YL.js → upgrade-PXK3S2YM.js} +11 -9
- package/dist/{usage-X52N3IDJ.js → usage-ME5MPXGX.js} +36 -34
- package/dist/{verifiers-EJTVVSMA.js → verifiers-BVZ7IWOO.js} +5 -5
- package/dist/{verify-YJL6XET2.js → verify-5K7ZKQFC.js} +4 -4
- package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
- package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-F5W5QTYY.js} +48 -47
- package/dist/{with-panes-OBOBFIIR.js → with-panes-BYOJCLAM.js} +51 -255
- package/dist/worker/entry.js +45 -30
- package/docs/README.md +176 -81
- package/docs/{acp.md → architecture/acp.md} +36 -20
- package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
- package/docs/{architecture.md → architecture/architecture.md} +43 -22
- package/docs/{artifact-placement.md → architecture/artifact-placement.md} +26 -23
- package/docs/architecture/artifact-versions.md +90 -0
- package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
- package/docs/{context-engine.md → architecture/context-engine.md} +25 -25
- package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
- package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
- package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
- package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
- package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
- package/docs/{model-catalog.md → architecture/model-catalog.md} +40 -17
- package/docs/{observability.md → architecture/observability.md} +26 -13
- package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
- package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +55 -20
- package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +35 -24
- package/docs/{safety-model.md → architecture/safety-model.md} +20 -15
- package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
- package/docs/architecture/time-conventions.md +125 -0
- package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
- package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
- package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
- package/docs/{built-in-agents.md → guide/built-in-agents.md} +50 -34
- package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +65 -60
- package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +227 -289
- package/docs/guide/configuration-reference.md +1158 -0
- package/docs/{environment-variables.md → guide/environment-variables.md} +31 -28
- package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
- package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
- package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
- package/docs/{glossary.md → guide/glossary.md} +14 -11
- package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +44 -13
- package/docs/guide/panes-and-files.md +290 -0
- package/docs/{proactive-memory.md → guide/proactive-memory.md} +79 -66
- package/docs/{resource-library.md → guide/resource-library.md} +13 -4
- package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +7 -3
- package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
- package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
- package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
- package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
- package/docs/{development-pipeline.md → process/development-pipeline.md} +24 -26
- package/docs/process/documentation-coverage.md +100 -0
- package/docs/process/documentation-guide.md +187 -0
- package/docs/{eval-runner.md → process/eval-runner.md} +41 -50
- package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
- package/docs/{evolution.md → process/evolution.md} +2 -2
- package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
- package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
- package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
- package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
- package/evals/README.md +2 -2
- package/package.json +9 -7
- package/skills/README.md +46 -37
- package/skills/coding/ast-grep/SKILL.md +2 -2
- package/skills/coding/coding-standards/SKILL.md +2 -2
- package/skills/coding/prototype/SKILL.md +2 -2
- package/skills/coding/tdd/SKILL.md +2 -2
- package/skills/context/context-handoff/SKILL.md +2 -2
- package/skills/context/context-prime/SKILL.md +2 -2
- package/skills/git/file-ticket/SKILL.md +2 -2
- package/skills/git/fix-issue/SKILL.md +3 -3
- package/skills/git/resolve-merge-conflicts/SKILL.md +2 -2
- package/skills/git/ship/SKILL.md +2 -2
- package/skills/git/worktree-create/SKILL.md +2 -2
- package/skills/git/worktree-merge/SKILL.md +2 -2
- package/skills/meta/clio-coder-dev/SKILL.md +9 -5
- package/skills/meta/clio-coder-dev/evals.md +3 -2
- package/skills/meta/clio-coder-test/SKILL.md +102 -95
- package/skills/meta/clio-coder-test/evals.md +9 -4
- package/skills/meta/clio-coder-test/references/harness.md +100 -124
- package/skills/meta/clio-coder-test/references/test-map.md +77 -50
- package/skills/meta/credentials/SKILL.md +2 -2
- package/skills/meta/find-skills/SKILL.md +2 -2
- package/skills/meta/herdr/SKILL.md +2 -2
- package/skills/meta/skill-craft/SKILL.md +22 -16
- package/skills/planning/architecture/SKILL.md +2 -2
- package/skills/planning/backlog/SKILL.md +2 -2
- package/skills/planning/prd/SKILL.md +2 -2
- package/skills/planning/product-intent/SKILL.md +2 -2
- package/skills/planning/tech-spec/SKILL.md +2 -2
- package/skills/registry.yaml +62 -62
- package/skills/research/arxiv-literature/SKILL.md +2 -2
- package/skills/research/experiment-protocol/SKILL.md +2 -2
- package/skills/research/scientific-debugging/SKILL.md +2 -2
- package/skills/research/scientific-modernization/SKILL.md +2 -2
- package/skills/skill-marketplace.json +62 -62
- package/skills/workflow/cut-it/SKILL.md +2 -2
- package/skills/workflow/design-council/SKILL.md +2 -2
- package/skills/workflow/grill-me/SKILL.md +2 -2
- package/skills/workflow/workflow-distiller/SKILL.md +2 -2
- package/src/cli/args.ts +2 -2
- package/src/cli/bootstrap-generate.ts +1 -1
- package/src/cli/config-inspect.ts +65 -12
- package/src/cli/configure.ts +0 -4
- package/src/cli/docs.ts +22 -14
- package/src/cli/doctor-naming.ts +5 -5
- package/src/cli/doctor-toolchain.ts +3 -3
- package/src/cli/eval.ts +1 -2
- package/src/cli/extensions.ts +2 -1
- package/src/cli/fleet.ts +1 -1
- package/src/cli/index.ts +2 -1
- package/src/cli/internal-dispatch.ts +3 -4
- package/src/cli/panes.ts +19 -5
- package/src/cli/run.ts +2 -2
- package/src/cli/share.ts +5 -1
- package/src/cli/skills-eval.ts +3 -3
- package/src/cli/targets.ts +2 -6
- package/src/cli/trace.ts +55 -4
- package/src/cli/wiki-generate.ts +1 -1
- package/src/core/artifact-paths.ts +1 -1
- package/src/core/bash-exec.ts +131 -86
- package/src/core/bus-events.ts +51 -6
- package/src/core/config.ts +5 -1
- package/src/core/defaults.ts +7 -4
- package/src/core/dispatch-outcome.ts +16 -0
- package/src/core/guardrails.ts +10 -49
- package/src/core/prompt-hint.ts +9 -0
- package/src/domains/agents/builtins/architect.md +2 -3
- package/src/domains/agents/builtins/coder.md +3 -2
- package/src/domains/agents/builtins/debugger.md +2 -2
- package/src/domains/agents/builtins/documenter.md +2 -2
- package/src/domains/agents/builtins/git-master.md +1 -1
- package/src/domains/agents/builtins/oracle.md +1 -1
- package/src/domains/agents/builtins/provenance.md +1 -1
- package/src/domains/agents/builtins/researcher.md +1 -1
- package/src/domains/agents/builtins/scout.md +1 -1
- package/src/domains/agents/builtins/tester.md +2 -2
- package/src/domains/agents/builtins/verifier.md +2 -2
- package/src/domains/agents/builtins/wiki-writer.md +1 -1
- package/src/domains/agents/catalog.ts +12 -14
- package/src/domains/agents/contract.ts +2 -0
- package/src/domains/agents/extension.ts +23 -1
- package/src/domains/config/keybindings.ts +8 -0
- package/src/domains/context/extension.ts +0 -3
- package/src/domains/context/working-set/path-index.ts +1 -0
- package/src/domains/dispatch/capability-match.ts +10 -0
- package/src/domains/dispatch/extension.ts +105 -22
- package/src/domains/dispatch/host-verification.ts +435 -39
- package/src/domains/dispatch/intent-requirements.ts +10 -0
- package/src/domains/dispatch/intent.ts +18 -1
- package/src/domains/dispatch/path-scope.ts +235 -24
- package/src/domains/dispatch/run-event-journal.ts +4 -15
- package/src/domains/dispatch/state.ts +2 -3
- package/src/domains/dispatch/transport.ts +45 -21
- package/src/domains/dispatch/types.ts +55 -3
- package/src/domains/eval/artifacts/store.ts +5 -0
- package/src/domains/eval/store.ts +8 -1
- package/src/domains/evidence/trust-status.ts +10 -1
- package/src/domains/extensions/contract.ts +15 -1
- package/src/domains/extensions/discovery.ts +238 -41
- package/src/domains/extensions/extension.ts +105 -6
- package/src/domains/extensions/index.ts +24 -0
- package/src/domains/extensions/integrity.ts +189 -0
- package/src/domains/extensions/manager.ts +17 -1
- package/src/domains/extensions/resource-path.ts +27 -0
- package/src/domains/extensions/resources.ts +18 -38
- package/src/domains/extensions/snapshot-store.ts +39 -0
- package/src/domains/extensions/snapshot.ts +180 -0
- package/src/domains/extensions/state.ts +385 -57
- package/src/domains/extensions/types.ts +118 -1
- package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
- package/src/domains/lifecycle/migrations/index.ts +2 -0
- package/src/domains/lifecycle/naming-resources.ts +19 -4
- package/src/domains/lifecycle/naming-yazi.ts +10 -5
- package/src/domains/middleware/contract.ts +26 -0
- package/src/domains/middleware/extension.ts +24 -24
- package/src/domains/middleware/hook-receipts.ts +27 -4
- package/src/domains/middleware/hooks-io.ts +65 -32
- package/src/domains/middleware/hooks.ts +64 -0
- package/src/domains/middleware/index.ts +28 -4
- package/src/domains/middleware/registrations.ts +326 -0
- package/src/domains/middleware/runtime.ts +28 -0
- package/src/domains/middleware/snapshot.ts +20 -7
- package/src/domains/mux/contract.ts +38 -0
- package/src/domains/mux/detect.ts +6 -13
- package/src/domains/mux/index.ts +1 -1
- package/src/domains/mux/operations.ts +44 -5
- package/src/domains/mux/yazi/assets/yazi.toml +2 -2
- package/src/domains/mux/yazi/session.ts +53 -4
- package/src/domains/mux/yazi/theme.ts +117 -17
- package/src/domains/observability/contract.ts +10 -11
- package/src/domains/observability/extension.ts +11 -3
- package/src/domains/observability/projection.ts +14 -90
- package/src/domains/observability/trace-store.ts +43 -7
- package/src/domains/prompts/compiler.ts +73 -53
- package/src/domains/prompts/contract.ts +15 -3
- package/src/domains/prompts/extension.ts +97 -9
- package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
- package/src/domains/prompts/fragments/identity/clio.md +6 -12
- package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
- package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
- package/src/domains/prompts/fragments/operating/contract.md +7 -15
- package/src/domains/prompts/fragments/operating/delegation.md +32 -34
- package/src/domains/prompts/fragments/operating/skills.md +10 -24
- package/src/domains/prompts/fragments/operating/worker.md +1 -8
- package/src/domains/providers/index.ts +1 -1
- package/src/domains/providers/model-runtime-capabilities.ts +85 -21
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +669 -104
- package/src/domains/providers/runtime-resolution.ts +31 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
- package/src/domains/providers/types/cost-provenance.ts +19 -0
- package/src/domains/providers/types/local-model-quirks.ts +85 -37
- package/src/domains/resources/skills/loader.ts +16 -19
- package/src/domains/safety/call-target.ts +1 -1
- package/src/domains/safety/loop-detector.ts +7 -4
- package/src/domains/session/task-board.ts +10 -9
- package/src/domains/share/archive.ts +164 -7
- package/src/engine/acp/server.ts +62 -9
- package/src/engine/apis/llamacpp-residency.ts +3 -4
- package/src/engine/apis/lmstudio.ts +3 -3
- package/src/engine/apis/ollama-native.ts +6 -6
- package/src/engine/apis/openai-completions.ts +28 -25
- package/src/engine/apis/output-budget.ts +8 -18
- package/src/engine/apis/residency.ts +8 -27
- package/src/engine/gemma-channel-filter.ts +19 -0
- package/src/engine/loop-guard.ts +92 -12
- package/src/engine/worker-runtime.ts +40 -11
- package/src/engine/worker-tools.ts +3 -1
- package/src/entry/extension-hook-sources.ts +28 -0
- package/src/entry/extension-reload.ts +309 -0
- package/src/entry/orchestrator.ts +59 -35
- package/src/interactive/application-controller.ts +2 -1
- package/src/interactive/bus-notices.ts +8 -1
- package/src/interactive/chat-loop-messages.ts +3 -13
- package/src/interactive/chat-loop.ts +10 -1
- package/src/interactive/chat-panel.ts +36 -13
- package/src/interactive/chat-renderer.ts +71 -7
- package/src/interactive/dispatch-board.ts +6 -11
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +39 -4
- package/src/interactive/interactive-input-runtime.ts +4 -0
- package/src/interactive/interactive-presentation.ts +2 -2
- package/src/interactive/interactive-slash-runtime.ts +2 -0
- package/src/interactive/overlays/extensions.ts +9 -1
- package/src/interactive/overlays/help-reference.ts +13 -0
- package/src/interactive/overlays/settings.ts +27 -16
- package/src/interactive/panes-runtime.ts +111 -35
- package/src/interactive/prompt-cache-identity.ts +88 -0
- package/src/interactive/slash-commands.ts +129 -14
- package/src/interactive/stream-pacing-policy.ts +0 -23
- package/src/interactive/turn-context.ts +30 -15
- package/src/interactive/yazi-bridge.ts +60 -6
- package/src/tools/agent-tools.ts +30 -1
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +3 -3
- package/src/tools/bash.ts +1 -1
- package/src/tools/bootstrap.ts +4 -0
- package/src/tools/builtin-tool-catalog.ts +52 -22
- package/src/tools/codewiki/code-nav-surface.ts +6 -0
- package/src/tools/codewiki/code-nav.ts +99 -13
- package/src/tools/context/docs-engine.ts +20 -7
- package/src/tools/context/index.ts +29 -12
- package/src/tools/core-bootstrap.ts +28 -6
- package/src/tools/credential-present.ts +1 -2
- package/src/tools/dispatch-arguments.ts +5 -1
- package/src/tools/dispatch-plan.ts +48 -4
- package/src/tools/dispatch-run-events.ts +1 -1
- package/src/tools/dispatch-schema.ts +338 -0
- package/src/tools/dispatch-types.ts +3 -0
- package/src/tools/dispatch.ts +9 -254
- package/src/tools/ledger.ts +3 -5
- package/src/tools/monitor-surface.ts +5 -13
- package/src/tools/observation.ts +4 -5
- package/src/tools/panes-surface.ts +4 -11
- package/src/tools/panes.ts +4 -2
- package/src/tools/policy.ts +15 -2
- package/src/tools/read.ts +5 -6
- package/src/tools/registry.ts +30 -7
- package/src/tools/result-shaping.ts +18 -14
- package/src/tools/steer-surface.ts +1 -1
- package/src/tools/tasks.ts +1 -1
- package/src/tools/truncate.ts +6 -5
- package/src/tools/verify/surface.ts +6 -12
- package/src/tools/web-fetch-surface.ts +1 -3
- package/dist/chunk-5QIAJV2D.js +0 -48
- package/dist/chunk-JZWT5J3Y.js +0 -814
- package/dist/chunk-K7VKOLQQ.js +0 -15
- package/dist/chunk-PMZCIOCJ.js +0 -25
- package/dist/chunk-SUW5DORT.js +0 -819
- package/dist/chunk-UOV2BYIW.js +0 -107
- package/dist/chunk-WR6U3OVP.js +0 -45
- package/docs/artifact-versions.md +0 -67
- package/docs/documentation-coverage.md +0 -46
- package/docs/documentation-guide.md +0 -167
- package/docs/time-conventions.md +0 -101
|
@@ -9,8 +9,11 @@
|
|
|
9
9
|
# https://huggingface.co/Jackrong/Qwopus3.6-35B-A3B-Coder-MTP-GGUF
|
|
10
10
|
# https://huggingface.co/nvidia/Nemotron-Cascade-2-30B-A3B
|
|
11
11
|
# https://huggingface.co/unsloth/NVIDIA-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-GGUF
|
|
12
|
+
# https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16
|
|
12
13
|
# https://huggingface.co/mradermacher/Qwen3.5-35B-A3B-Claude-4.6-Opus-Reasoning-Distilled-i1-GGUF
|
|
13
14
|
# https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-GGUF
|
|
15
|
+
# https://huggingface.co/meta-models/Muse-Glimmer-30B
|
|
16
|
+
# https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.6-27B
|
|
14
17
|
#
|
|
15
18
|
# This knowledge base is intentionally narrow. It describes only the local
|
|
16
19
|
# models curated for Clio's local coding workflows and leaves cloud GPT models
|
|
@@ -26,8 +29,7 @@
|
|
|
26
29
|
# not be established from git history or the release-test reports under
|
|
27
30
|
# docs/release-notes/ read `unknown` rather than a plausible guess. A
|
|
28
31
|
# `measuredUnder` block is provenance for a reader; nothing in the engine
|
|
29
|
-
# consumes it (`extractLocalModelQuirks` narrows only
|
|
30
|
-
# thinking).
|
|
32
|
+
# consumes it (`extractLocalModelQuirks` narrows only sampling and thinking).
|
|
31
33
|
#
|
|
32
34
|
# `llamaCpp.parallel` is the recommended `--parallel` value for starting the
|
|
33
35
|
# server, and families that carry one also carry a `parallelSlots` note. It is
|
|
@@ -73,13 +75,12 @@
|
|
|
73
75
|
thinking:
|
|
74
76
|
temperature: 1
|
|
75
77
|
topP: 1
|
|
76
|
-
maxTokens: 32768
|
|
77
78
|
instruct:
|
|
78
79
|
temperature: 1
|
|
79
80
|
topP: 1
|
|
80
|
-
maxTokens: 8192
|
|
81
81
|
runtimePreference:
|
|
82
|
-
llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high."
|
|
82
|
+
llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high. The analysis channel arrives as reasoning_content."
|
|
83
|
+
lmstudio: "Reasoning effort travels as the top-level reasoning_effort field. The analysis channel arrives as the OpenAI-compatible `reasoning` field, on the streamed delta and on the final message, with reasoning_content absent (measured 2026-09-02 on dynamo, gpt-oss-20b and gpt-oss-120b); Clio reads both spellings into thinking content."
|
|
83
84
|
openaiCompat: "Preferred for Harmony reasoning-effort control when the local gateway exposes chat_template_kwargs."
|
|
84
85
|
thinking:
|
|
85
86
|
mechanism: effort-levels
|
|
@@ -91,7 +92,9 @@
|
|
|
91
92
|
GPT-OSS uses OpenAI Harmony formatting and supports low, medium, and
|
|
92
93
|
high reasoning effort. Clio coerces off/minimal to low, captures
|
|
93
94
|
non-final Harmony channels as ThinkingContent, and strips Harmony
|
|
94
|
-
control tokens from visible output.
|
|
95
|
+
control tokens from visible output. The analysis channel reaches Clio
|
|
96
|
+
as reasoning_content on llama.cpp and as reasoning on LM Studio; both
|
|
97
|
+
are captured the same way.
|
|
95
98
|
|
|
96
99
|
- family: agenticqwen-30b-a3b-i1
|
|
97
100
|
matchPatterns:
|
|
@@ -130,13 +133,11 @@
|
|
|
130
133
|
temperature: 0.5
|
|
131
134
|
topP: 0.9
|
|
132
135
|
topK: 20
|
|
133
|
-
maxTokens: 16384
|
|
134
136
|
reasoningBudget: 4096
|
|
135
137
|
instruct:
|
|
136
138
|
temperature: 0.5
|
|
137
139
|
topP: 0.9
|
|
138
140
|
topK: 20
|
|
139
|
-
maxTokens: 4096
|
|
140
141
|
gpuTiers:
|
|
141
142
|
"32gb": "Default 32 GB local llama.cpp orchestrator profile; benchmarked with roughly 1 GiB VRAM headroom on a 32 GiB class card."
|
|
142
143
|
runtimePreference:
|
|
@@ -170,6 +171,10 @@
|
|
|
170
171
|
- nemotron-3-nano-omni-30b-a3b-reasoning
|
|
171
172
|
- nemotron-3-nano-omni
|
|
172
173
|
- nemotron-nano-omni
|
|
174
|
+
# Both router spellings. The live mini router advertises the id with `moe`
|
|
175
|
+
# in it; issue #263 and the round-5 handoffs use the shorter one.
|
|
176
|
+
- nemotron3-30b-moe-omni
|
|
177
|
+
- nemotron3-30b-omni
|
|
173
178
|
capabilities:
|
|
174
179
|
chat: true
|
|
175
180
|
tools: true
|
|
@@ -185,18 +190,45 @@
|
|
|
185
190
|
contextWindow: 1048576
|
|
186
191
|
maxTokens: 131072
|
|
187
192
|
quirks:
|
|
193
|
+
measuredUnder:
|
|
194
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 4 request slots"
|
|
195
|
+
runtime: llamacpp
|
|
196
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee (the string the ornith-1.5 entry below records for this same host on this date)"
|
|
197
|
+
model: "Nemotron-3-Nano-Omni-30B-A3B-Reasoning UD-Q4_K_M with Nemotron-Omni-mmproj-F16 (arch nemotron_h_moe)"
|
|
198
|
+
llamaCpp: >-
|
|
199
|
+
--ctx-size 1032192 --parallel 4 --kv-unified --cache-type-k f16
|
|
200
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
201
|
+
true --n-gpu-layers 99 --jinja --reasoning on --temperature 0.6 --top-k
|
|
202
|
+
40 --top-p 0.95 --repeat-penalty 1.05
|
|
203
|
+
date: "2026-09-02"
|
|
204
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
205
|
+
note: |
|
|
206
|
+
The on-off classification is measured, and it replaces a budget-tokens
|
|
207
|
+
dial that never reached this runtime. With no thinking field the model
|
|
208
|
+
spent 54 completion tokens with 140 characters of reasoning_content;
|
|
209
|
+
with chat_template_kwargs.enable_thinking false it answered in 4 tokens
|
|
210
|
+
with none; with it true it spent 42 tokens and 119 characters.
|
|
211
|
+
reasoning_effort moved nothing: none, low, and high produced 43, 48, and
|
|
212
|
+
39 tokens and a reasoning trace every time. The budget-tokens mechanism
|
|
213
|
+
this entry used to declare is downgraded to informational for
|
|
214
|
+
openai-completions on llama.cpp with a qwen chat template
|
|
215
|
+
(acceptsBudgetTokensField in model-runtime-capabilities.ts), so no
|
|
216
|
+
budget ever reached the server and the model reasoned at every level.
|
|
217
|
+
The router agrees with the measurement, advertising reasoning:on_off
|
|
218
|
+
with default_reasoning:on.
|
|
219
|
+
|
|
220
|
+
The measuring server is not the recommended profile below: it ran
|
|
221
|
+
ctx-size 1032192 with f16 KV against the 819200 and q8_0 recommended
|
|
222
|
+
here, and loaded only the vision mmproj, so the audio capability stays
|
|
223
|
+
the card's claim rather than this run's.
|
|
188
224
|
sampling:
|
|
189
225
|
thinking:
|
|
190
226
|
temperature: 0.6
|
|
191
227
|
topP: 0.95
|
|
192
228
|
topK: 20
|
|
193
|
-
maxTokens: 20480
|
|
194
|
-
reasoningBudget: 16384
|
|
195
|
-
gracePeriod: 1024
|
|
196
229
|
instruct:
|
|
197
230
|
temperature: 0.2
|
|
198
231
|
topK: 1
|
|
199
|
-
maxTokens: 1024
|
|
200
232
|
gpuTiers:
|
|
201
233
|
"24gb": "Use a 4-bit quant with reduced load context for evaluation. Prefer the 32GB tier for sustained Clio writes."
|
|
202
234
|
"32gb": "Primary strong local model for multimodal main-agent work. Keep openai-compat available for tool-call fallback."
|
|
@@ -213,17 +245,131 @@
|
|
|
213
245
|
parallel: 4
|
|
214
246
|
parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 204800 tokens rather than the whole 819200. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
|
|
215
247
|
thinking:
|
|
216
|
-
mechanism:
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
248
|
+
mechanism: on-off
|
|
249
|
+
chatTemplateKwargs:
|
|
250
|
+
static:
|
|
251
|
+
# Card recipe for the reasoning mode (scratchpad/catalog/models/nvidia-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-FP8.md:7): reasoning_budget=16384.
|
|
252
|
+
reasoning_budget: 16384
|
|
253
|
+
lmstudio: unsupported
|
|
221
254
|
guidance: |
|
|
222
|
-
Reasoning is template-driven
|
|
223
|
-
|
|
224
|
-
|
|
255
|
+
Reasoning is template-driven and the switch is
|
|
256
|
+
chat_template_kwargs.enable_thinking; intermediate levels coerce to on.
|
|
257
|
+
The checkpoint reasons when nothing is sent, so off has to be carried on
|
|
258
|
+
the wire rather than omitted. reasoning_effort is inert on llama.cpp;
|
|
259
|
+
LM Studio reads the same on-off switch from that field instead.
|
|
225
260
|
serving: "Use qwen3_coder tool parsing when served through OpenAI-compatible HTTP. LM Studio uses the same OpenAI-compatible tool surface."
|
|
226
261
|
|
|
262
|
+
# NVIDIA's 3.5 Lightning generation, not a requant of the Nano Omni family
|
|
263
|
+
# above: a different HF repo (nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B),
|
|
264
|
+
# text-only where Nano Omni carries vision and audio, and a Mamba-2 plus MoE
|
|
265
|
+
# plus attention hybrid rather than the omni checkpoint. They share the vendor,
|
|
266
|
+
# the nemotron_h_moe arch string, and the 30B-A3B parameter count, which is not
|
|
267
|
+
# the same as sharing weights.
|
|
268
|
+
- family: nemotron-3.5-lightning-30b-a3b
|
|
269
|
+
matchPatterns:
|
|
270
|
+
# No bare `nemotron-3.5-lightning`: the 30b spelling is the shortest one
|
|
271
|
+
# that cannot swallow a future 8B or 70B Lightning id, which would inherit
|
|
272
|
+
# this checkpoint's sampler, its 1M window and a dial measured elsewhere.
|
|
273
|
+
- nvidia-nemotron-3.5-lightning-30b-a3b
|
|
274
|
+
- nemotron-3.5-lightning-30b-a3b
|
|
275
|
+
- nemotron-3.5-lightning-30b
|
|
276
|
+
- nemo3.5-30b-moe
|
|
277
|
+
capabilities:
|
|
278
|
+
chat: true
|
|
279
|
+
tools: true
|
|
280
|
+
toolCallFormat: qwen
|
|
281
|
+
reasoning: true
|
|
282
|
+
thinkingFormat: qwen-chat-template
|
|
283
|
+
structuredOutputs: json-schema
|
|
284
|
+
vision: false
|
|
285
|
+
audio: false
|
|
286
|
+
embeddings: false
|
|
287
|
+
rerank: false
|
|
288
|
+
fim: false
|
|
289
|
+
contextWindow: 1048576
|
|
290
|
+
maxTokens: 65536
|
|
291
|
+
quirks:
|
|
292
|
+
measuredUnder:
|
|
293
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 1 request slot"
|
|
294
|
+
runtime: llamacpp
|
|
295
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee"
|
|
296
|
+
model: "NVIDIA-Nemotron-3.5-Lightning-30B-A3B Q4_K_M (arch nemotron_h_moe)"
|
|
297
|
+
llamaCpp: >-
|
|
298
|
+
--ctx-size 1048576 --parallel 1 --kv-unified --cache-type-k f16
|
|
299
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
300
|
+
true --fit off --n-gpu-layers 99 --jinja --reasoning on
|
|
301
|
+
--chat-template-kwargs {"enable_thinking": true} --temperature 1.0
|
|
302
|
+
--top-k 40 --top-p 0.95 --repeat-penalty 1.05
|
|
303
|
+
date: "2026-09-02"
|
|
304
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
305
|
+
note: |
|
|
306
|
+
The mechanism is measured on this runtime and the card agrees with it.
|
|
307
|
+
With no thinking field the model spent 97 completion tokens with 300
|
|
308
|
+
characters of reasoning_content; with
|
|
309
|
+
chat_template_kwargs.enable_thinking false it answered in 3 tokens with
|
|
310
|
+
none; with it true it spent 120 tokens and 353 characters.
|
|
311
|
+
reasoning_effort is inert: none, low, and high produced 127, 114, and
|
|
312
|
+
117 tokens with a reasoning trace every time. The NVIDIA card states the
|
|
313
|
+
same control, "Reasoning Mode: Configurable on/off via chat template
|
|
314
|
+
(enable_thinking=True/False)", and the router advertises reasoning:on_off
|
|
315
|
+
with default_reasoning:on.
|
|
316
|
+
|
|
317
|
+
The same divergence is already recorded in code for this checkpoint:
|
|
318
|
+
model-runtime-capabilities.ts notes a 2026-08-11 fleet measurement on
|
|
319
|
+
nvidia-nemotron-3.5-lightning-30b-a3b where LM Studio suppressed
|
|
320
|
+
reasoning for reasoning_effort "none" and ignored enable_thinking false
|
|
321
|
+
while llama.cpp did the reverse. That id is a matchPattern here.
|
|
322
|
+
|
|
323
|
+
Sampler provenance: temperature 1.0 and top_p 0.95 are the card's
|
|
324
|
+
"Recommended Sampling"; top_k 40 and repeat_penalty 1.05 are the router
|
|
325
|
+
preset's launch arguments and have no card value. The card publishes no
|
|
326
|
+
separate thinking and instruct profiles, so both carry the same numbers.
|
|
327
|
+
maxTokens follows nemotron-cascade-2, the other 1M-context Nemotron in
|
|
328
|
+
this catalog; the card caps output only by the context window.
|
|
329
|
+
sampling:
|
|
330
|
+
thinking:
|
|
331
|
+
temperature: 1.0
|
|
332
|
+
topP: 0.95
|
|
333
|
+
topK: 40
|
|
334
|
+
repeatPenalty: 1.05
|
|
335
|
+
instruct:
|
|
336
|
+
temperature: 1.0
|
|
337
|
+
topP: 0.95
|
|
338
|
+
topK: 40
|
|
339
|
+
repeatPenalty: 1.05
|
|
340
|
+
gpuTiers:
|
|
341
|
+
"32gb": "Text-only agent target on a 32 GB card. Q4_K_M weights plus f16 KV reach the full 1048576-token window at parallel 1 on the measured server."
|
|
342
|
+
runtimePreference:
|
|
343
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. The router profile pins chat_template_kwargs.enable_thinking true, so a request that wants silence must send false rather than omitting the field."
|
|
344
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen tool parsing."
|
|
345
|
+
lmstudio: "LM Studio reads the on-off switch from reasoning_effort, not from chat_template_kwargs; see the 2026-08-11 note above."
|
|
346
|
+
llamaCpp:
|
|
347
|
+
ctxSize: 1048576
|
|
348
|
+
cacheTypeK: f16
|
|
349
|
+
cacheTypeV: f16
|
|
350
|
+
flashAttn: true
|
|
351
|
+
nGpuLayers: 99
|
|
352
|
+
parallel: 1
|
|
353
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 1048576-token context, and add --kv-unified if --parallel is raised or the context is divided across slots. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
354
|
+
batchSize: 16384
|
|
355
|
+
ubatchSize: 2048
|
|
356
|
+
chatTemplateKwargs:
|
|
357
|
+
enable_thinking: true
|
|
358
|
+
thinking:
|
|
359
|
+
mechanism: on-off
|
|
360
|
+
chatTemplateKwargs:
|
|
361
|
+
static:
|
|
362
|
+
# Card: "For coding agents, add extra_body={"chat_template_kwargs": {"force_nonempty_content": True}}" (scratchpad/catalog/models/nvidia-NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.md:21).
|
|
363
|
+
force_nonempty_content: true
|
|
364
|
+
lmstudio: unsupported
|
|
365
|
+
guidance: |
|
|
366
|
+
Nemotron 3.5 Lightning's chat template exposes thinking as on or off
|
|
367
|
+
through enable_thinking; intermediate levels coerce to on. It reasons by
|
|
368
|
+
default, so off has to be carried on the wire. On LM Studio the same
|
|
369
|
+
switch is read from reasoning_effort instead, which Clio sends for
|
|
370
|
+
on-off models on that runtime.
|
|
371
|
+
serving: "NVIDIA-Nemotron-3.5-Lightning-30B-A3B, MoE hybrid of Mamba-2, MoE and attention layers with 3B active parameters. The card recommends nemotron_v3 reasoning parsing and qwen3_coder tool parsing; the mini router serves the Q4_K_M gguf with XML tool calls through llama.cpp's qwen3_coder parser. Native context is 1M."
|
|
372
|
+
|
|
227
373
|
- family: qwen3.6-35b-a3b
|
|
228
374
|
matchPatterns:
|
|
229
375
|
- qwen3.6-35b-a3b
|
|
@@ -245,14 +391,14 @@
|
|
|
245
391
|
rerank: false
|
|
246
392
|
fim: false
|
|
247
393
|
contextWindow: 262144
|
|
248
|
-
|
|
394
|
+
# Vendor maximum for complex problems. Evidence: scratchpad/catalog/models/Qwen-Qwen3.6-35B-A3B.md.
|
|
395
|
+
maxTokens: 81920
|
|
249
396
|
quirks:
|
|
250
397
|
sampling:
|
|
251
398
|
thinking:
|
|
252
399
|
temperature: 0.6
|
|
253
400
|
topP: 0.95
|
|
254
401
|
topK: 20
|
|
255
|
-
maxTokens: 16384
|
|
256
402
|
reasoningBudget: 4096
|
|
257
403
|
instruct:
|
|
258
404
|
temperature: 0.7
|
|
@@ -318,22 +464,17 @@
|
|
|
318
464
|
family of behavior as qwopus3.6-35b-a3b-coder. The recommended
|
|
319
465
|
llama.cpp serving profile below is the model card's, not that
|
|
320
466
|
measurement's, and no argv from the measuring host survives.
|
|
321
|
-
kvCache:
|
|
322
|
-
kQuant: q8_0
|
|
323
|
-
vQuant: q8_0
|
|
324
467
|
sampling:
|
|
325
468
|
thinking:
|
|
326
469
|
temperature: 0.6
|
|
327
470
|
topP: 0.95
|
|
328
471
|
topK: 20
|
|
329
|
-
maxTokens: 16384
|
|
330
472
|
reasoningBudget: 4096
|
|
331
473
|
instruct:
|
|
332
474
|
temperature: 0.6
|
|
333
475
|
topP: 0.95
|
|
334
476
|
topK: 20
|
|
335
477
|
repeatPenalty: 1.05
|
|
336
|
-
maxTokens: 4096
|
|
337
478
|
gpuTiers:
|
|
338
479
|
"32gb": "32 GB local profile with Q4_K_M, ctx=262144, q8 KV, flash attention, and parallel=1."
|
|
339
480
|
runtimePreference:
|
|
@@ -358,6 +499,72 @@
|
|
|
358
499
|
contains the final answer.
|
|
359
500
|
serving: "Ornith-1.0-35B is an agentic coding model; the card recommends 262K context, qwen3 reasoning parsing, qwen3_coder/qwen3_xml tool parsing, and sampler temperature=0.6, top_p=0.95, top_k=20."
|
|
360
501
|
|
|
502
|
+
- family: ornith-1.5
|
|
503
|
+
matchPatterns:
|
|
504
|
+
- ornith-1.5-35b-a3b
|
|
505
|
+
- ornith1.5-35b-moe
|
|
506
|
+
- ornith-1.5-35b
|
|
507
|
+
- ornith1.5-35b
|
|
508
|
+
- ornith-1.5
|
|
509
|
+
- ornith1.5
|
|
510
|
+
capabilities:
|
|
511
|
+
chat: true
|
|
512
|
+
tools: true
|
|
513
|
+
toolCallFormat: qwen
|
|
514
|
+
reasoning: true
|
|
515
|
+
thinkingFormat: qwen-chat-template
|
|
516
|
+
structuredOutputs: json-schema
|
|
517
|
+
vision: false
|
|
518
|
+
audio: false
|
|
519
|
+
embeddings: false
|
|
520
|
+
rerank: false
|
|
521
|
+
fim: false
|
|
522
|
+
contextWindow: 262144
|
|
523
|
+
maxTokens: 65536
|
|
524
|
+
quirks:
|
|
525
|
+
measuredUnder:
|
|
526
|
+
hardware: "mini fleet node, AMD GPU on the Vulkan backend, 4 request slots"
|
|
527
|
+
runtime: llamacpp
|
|
528
|
+
build: b1-c841aee
|
|
529
|
+
model: "Ornith-1.5-35B-A3B Q4_K_M (arch qwen35moe)"
|
|
530
|
+
llamaCpp: >-
|
|
531
|
+
--ctx-size 786432 --parallel 4 --no-kv-unified --cache-type-k q8_0
|
|
532
|
+
--cache-type-v q8_0 --batch-size 16384 --ubatch-size 2048 --jinja
|
|
533
|
+
--reasoning on
|
|
534
|
+
date: "2026-09-02"
|
|
535
|
+
source: "prompt-infrastructure round 2 (v0.4.2 development): live probe of the on-off switch, and a debugger dispatch receipt"
|
|
536
|
+
note: |
|
|
537
|
+
The 1.5 release is a Qwen3.5-MoE derivative and its chat template
|
|
538
|
+
honors enable_thinking, which the 1.0 measurement above did not find.
|
|
539
|
+
Measured on this date: with chat_template_kwargs.enable_thinking false
|
|
540
|
+
the model answered a one-line arithmetic prompt in 4 completion tokens
|
|
541
|
+
with no reasoning_content; with it true the same prompt spent 79
|
|
542
|
+
tokens, 158 characters of them reasoning. Under the inherited
|
|
543
|
+
always-on classification a debugger worker dispatched with thinking
|
|
544
|
+
off spent 1,061 of its 1,518 output tokens on a reasoning trace, and
|
|
545
|
+
the router itself advertises the model as reasoning:on_off. The
|
|
546
|
+
generic `ornith` pattern still routes 1.0 checkpoints to the entry
|
|
547
|
+
above; these longer patterns win for 1.5 ids.
|
|
548
|
+
sampling:
|
|
549
|
+
thinking:
|
|
550
|
+
temperature: 0.6
|
|
551
|
+
topP: 0.95
|
|
552
|
+
topK: 20
|
|
553
|
+
reasoningBudget: 4096
|
|
554
|
+
instruct:
|
|
555
|
+
temperature: 0.6
|
|
556
|
+
topP: 0.95
|
|
557
|
+
topK: 20
|
|
558
|
+
repeatPenalty: 1.05
|
|
559
|
+
thinking:
|
|
560
|
+
mechanism: on-off
|
|
561
|
+
guidance: |
|
|
562
|
+
Ornith 1.5's Qwen3.5 chat template exposes thinking as on or off
|
|
563
|
+
through enable_thinking; intermediate levels coerce to on. LM Studio
|
|
564
|
+
reads the same switch from reasoning_effort (none or low), which Clio
|
|
565
|
+
sends for on-off models on that runtime.
|
|
566
|
+
serving: "Ornith-1.5-35B-A3B follows the 1.0 card: 262K context, qwen3 reasoning parsing, qwen3_coder tool parsing, sampler temperature=0.6, top_p=0.95, top_k=20."
|
|
567
|
+
|
|
361
568
|
- family: qwen3.6-27b
|
|
362
569
|
matchPatterns:
|
|
363
570
|
- qwen3.6-27b
|
|
@@ -376,7 +583,8 @@
|
|
|
376
583
|
rerank: false
|
|
377
584
|
fim: false
|
|
378
585
|
contextWindow: 262144
|
|
379
|
-
|
|
586
|
+
# Vendor maximum for complex problems. Evidence: scratchpad/catalog/models/Qwen-Qwen3.6-27B.md.
|
|
587
|
+
maxTokens: 81920
|
|
380
588
|
quirks:
|
|
381
589
|
measuredUnder:
|
|
382
590
|
hardware: unknown
|
|
@@ -400,7 +608,6 @@
|
|
|
400
608
|
minP: 0.0
|
|
401
609
|
presencePenalty: 0.0
|
|
402
610
|
repetitionPenalty: 1.0
|
|
403
|
-
maxTokens: 32768
|
|
404
611
|
instruct:
|
|
405
612
|
temperature: 0.7
|
|
406
613
|
topP: 0.80
|
|
@@ -408,7 +615,6 @@
|
|
|
408
615
|
minP: 0.0
|
|
409
616
|
presencePenalty: 1.5
|
|
410
617
|
repetitionPenalty: 1.0
|
|
411
|
-
maxTokens: 32768
|
|
412
618
|
gpuTiers:
|
|
413
619
|
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 (Q4_K_XL weights ~19.5 GB plus ~32 GB KV)."
|
|
414
620
|
runtimePreference:
|
|
@@ -444,6 +650,7 @@
|
|
|
444
650
|
rerank: false
|
|
445
651
|
fim: false
|
|
446
652
|
contextWindow: 262144
|
|
653
|
+
# Vendor maximum for final responses. Evidence: scratchpad/catalog/models/Qwen-Qwen3.8-27B.md.
|
|
447
654
|
maxTokens: 131072
|
|
448
655
|
quirks:
|
|
449
656
|
measuredUnder:
|
|
@@ -468,9 +675,6 @@
|
|
|
468
675
|
control were established earlier, on 2026-08-18, from a llama.cpp host
|
|
469
676
|
and an LM Studio host whose build strings were not recorded; treat those
|
|
470
677
|
two findings as unknown-build.
|
|
471
|
-
kvCache:
|
|
472
|
-
kQuant: q8_0
|
|
473
|
-
vQuant: q8_0
|
|
474
678
|
sampling:
|
|
475
679
|
thinking:
|
|
476
680
|
temperature: 1.0
|
|
@@ -479,7 +683,6 @@
|
|
|
479
683
|
minP: 0.0
|
|
480
684
|
presencePenalty: 0.0
|
|
481
685
|
repetitionPenalty: 1.0
|
|
482
|
-
maxTokens: 32768
|
|
483
686
|
instruct:
|
|
484
687
|
temperature: 0.7
|
|
485
688
|
topP: 0.80
|
|
@@ -487,7 +690,6 @@
|
|
|
487
690
|
minP: 0.0
|
|
488
691
|
presencePenalty: 1.5
|
|
489
692
|
repetitionPenalty: 1.0
|
|
490
|
-
maxTokens: 32768
|
|
491
693
|
gpuTiers:
|
|
492
694
|
"24gb-rocm": "24 GB ROCm class: IQ4_NL uses about 17.3 GB and UD-Q4_K_XL uses about 19.5 GB. Serve ctx=131072 with q8_0/q8_0 KV, parallel=2, and draft-mtp speculative decoding."
|
|
493
695
|
"32gb-cuda": "32 GB CUDA class: Q6_K uses about 23.8 GB. Serve ctx=131072 with parallel=4 because the full 262144-token context is VRAM-bound."
|
|
@@ -590,9 +792,6 @@
|
|
|
590
792
|
exact NVFP4 build it was seen on. The KV-budget arithmetic in the
|
|
591
793
|
gpuTiers and the serving note is derived from the official Google
|
|
592
794
|
Gemma 4 31B card.
|
|
593
|
-
kvCache:
|
|
594
|
-
kQuant: q8_0
|
|
595
|
-
vQuant: q8_0
|
|
596
795
|
sampling:
|
|
597
796
|
thinking:
|
|
598
797
|
temperature: 1.0
|
|
@@ -601,13 +800,11 @@
|
|
|
601
800
|
minP: 0.0
|
|
602
801
|
presencePenalty: 0.0
|
|
603
802
|
repetitionPenalty: 1.0
|
|
604
|
-
maxTokens: 32768
|
|
605
803
|
instruct:
|
|
606
804
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
607
805
|
topP: 0.95
|
|
608
806
|
topK: 64
|
|
609
807
|
minP: 0.0
|
|
610
|
-
maxTokens: 32768
|
|
611
808
|
chatTemplate: "gemma-channel"
|
|
612
809
|
leakageNote: |
|
|
613
810
|
LM Studio's gemma4 chat template emits chain-of-thought as plain text
|
|
@@ -643,6 +840,11 @@
|
|
|
643
840
|
- gemma-4-31b-it-qat-ud-q4_k_xl-mtp
|
|
644
841
|
- gemma-4-31b-it-qat-ud
|
|
645
842
|
- gemma-4-31b-it-qat
|
|
843
|
+
# The mini router's short id for this exact build; its alias is
|
|
844
|
+
# Gemma-4-31B-it-qat-UD-Q4_K_XL-MTP-262K, the first pattern above. A bare
|
|
845
|
+
# `gemma4-31b` would also swallow a future `gemma4-31b-nvfp4` id that
|
|
846
|
+
# belongs to the turbo family above, so the full router id is the pattern.
|
|
847
|
+
- gemma4-31b-dense
|
|
646
848
|
capabilities:
|
|
647
849
|
chat: true
|
|
648
850
|
tools: true
|
|
@@ -658,9 +860,35 @@
|
|
|
658
860
|
contextWindow: 262144
|
|
659
861
|
maxTokens: 32768
|
|
660
862
|
quirks:
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
863
|
+
measuredUnder:
|
|
864
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
865
|
+
runtime: llamacpp
|
|
866
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
867
|
+
model: "Gemma-4-31B-it-qat UD-Q4_K_XL with mtp-gemma-4-31B-it draft and mmproj-F16 (arch gemma4)"
|
|
868
|
+
llamaCpp: >-
|
|
869
|
+
--ctx-size 139264 --parallel 1 --kv-unified --cache-type-k f16
|
|
870
|
+
--cache-type-v f16 --batch-size 1024 --ubatch-size 256 --flash-attn true
|
|
871
|
+
--fit off --n-gpu-layers 99 --spec-type draft-mtp --spec-draft-n-max 4
|
|
872
|
+
--n-gpu-layers-draft 99 --jinja --reasoning off --temperature 0.8
|
|
873
|
+
--top-k 64 --top-p 0.95 --repeat-penalty 1.05
|
|
874
|
+
date: "2026-09-02"
|
|
875
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
876
|
+
note: |
|
|
877
|
+
The on-off mechanism this entry already claimed is now measured on
|
|
878
|
+
llama.cpp rather than inherited from the gemma-4 template. With no
|
|
879
|
+
thinking field, and with chat_template_kwargs.enable_thinking false, the
|
|
880
|
+
model answered in 3 completion tokens with no reasoning_content; with it
|
|
881
|
+
true the same prompt spent 68 tokens, 128 characters of them reasoning.
|
|
882
|
+
reasoning_effort is inert: none, low, and high all stayed at 3 tokens
|
|
883
|
+
with no trace, so this family is on-off and not effort-levels on this
|
|
884
|
+
runtime. The router serves it with --reasoning off, which is why the
|
|
885
|
+
no-field baseline is silent.
|
|
886
|
+
|
|
887
|
+
The measuring server is not the 32gb profile recommended below. It ran
|
|
888
|
+
ROCm at ctx-size 139264 with f16 KV, batch 1024 and ubatch 256, where the
|
|
889
|
+
block below recommends Vulkan at 262144 with q8_0 KV. The serving note
|
|
890
|
+
and gpuTiers keep the 262k recommendation; treat the argv above as the
|
|
891
|
+
configuration the token counts came from.
|
|
664
892
|
sampling:
|
|
665
893
|
thinking:
|
|
666
894
|
temperature: 1.0
|
|
@@ -669,14 +897,12 @@
|
|
|
669
897
|
minP: 0.0
|
|
670
898
|
presencePenalty: 0.0
|
|
671
899
|
repetitionPenalty: 1.0
|
|
672
|
-
maxTokens: 32768
|
|
673
900
|
instruct:
|
|
674
901
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; router profiles may serve this qat build at 0.8
|
|
675
902
|
topP: 0.95
|
|
676
903
|
topK: 64
|
|
677
904
|
minP: 0.0
|
|
678
905
|
repeatPenalty: 1.05
|
|
679
|
-
maxTokens: 32768
|
|
680
906
|
chatTemplate: "gemma-channel"
|
|
681
907
|
gpuTiers:
|
|
682
908
|
"32gb": "32 GB Vulkan llama.cpp profile with UD-Q4_K_XL weights, ctx=262144, q8 KV, flash attention, parallel=1, F16 gemma4v mmproj (vision), and MTP draft (spec-type draft-mtp, n_max 4) both active."
|
|
@@ -726,9 +952,6 @@
|
|
|
726
952
|
contextWindow: 122880
|
|
727
953
|
maxTokens: 32768
|
|
728
954
|
quirks:
|
|
729
|
-
kvCache:
|
|
730
|
-
kQuant: q8_0
|
|
731
|
-
vQuant: q8_0
|
|
732
955
|
sampling:
|
|
733
956
|
thinking:
|
|
734
957
|
temperature: 1.0
|
|
@@ -737,13 +960,11 @@
|
|
|
737
960
|
minP: 0.0
|
|
738
961
|
presencePenalty: 0.0
|
|
739
962
|
repetitionPenalty: 1.0
|
|
740
|
-
maxTokens: 32768
|
|
741
963
|
instruct:
|
|
742
964
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
743
965
|
topP: 0.95
|
|
744
966
|
topK: 64
|
|
745
967
|
minP: 0.0
|
|
746
|
-
maxTokens: 32768
|
|
747
968
|
chatTemplate: "gemma-channel"
|
|
748
969
|
measuredUnder:
|
|
749
970
|
hardware: unknown
|
|
@@ -779,10 +1000,26 @@
|
|
|
779
1000
|
serving: "Distilled from Claude 4.6 Opus reasoning traces onto Google Gemma 4 31B; sampler matches the gemopus card values."
|
|
780
1001
|
|
|
781
1002
|
- family: qwopus3.6-27b-v1-preview
|
|
1003
|
+
# A bare `qwopus3.6` pattern used to sit here and swallowed every Qwopus3.6
|
|
1004
|
+
# router id that was not spelled 27b. The mini router's `qwopus3.6-35b-moe`
|
|
1005
|
+
# landed on this 27B distill and inherited its budget-tokens dial, which emits
|
|
1006
|
+
# nothing on llama.cpp, so that model ran with a four-level dial that never
|
|
1007
|
+
# reached the wire (issue #263). The matcher in knowledge-base.ts ranks by
|
|
1008
|
+
# pattern length and ignores file order, so the only durable guard against the
|
|
1009
|
+
# next `qwopus3.6-*` id landing here is not to declare the short pattern.
|
|
1010
|
+
#
|
|
1011
|
+
# The cost is real and is chosen deliberately, here and at the other four
|
|
1012
|
+
# families that keep only their full spellings. An unlisted `qwopus3.6-*` id
|
|
1013
|
+
# now matches nothing, and a model with no family gets whatever
|
|
1014
|
+
# `inferThinkingMechanism` reads off the live probe: `none` with the dial
|
|
1015
|
+
# pinned to off and every thinking field stripped when the server reports no
|
|
1016
|
+
# reasoning, `on-off` when it reports some. Both are honest about knowing
|
|
1017
|
+
# nothing. Inheriting a sibling's measured dial is not, and it is the failure
|
|
1018
|
+
# mode that costs a session its reasoning silently. Adding the id is a
|
|
1019
|
+
# one-line catalog edit; noticing that a dial reaches nothing takes a sweep.
|
|
782
1020
|
matchPatterns:
|
|
783
1021
|
- qwopus3.6-27b-v1-preview
|
|
784
1022
|
- qwopus3.6-27b
|
|
785
|
-
- qwopus3.6
|
|
786
1023
|
capabilities:
|
|
787
1024
|
chat: true
|
|
788
1025
|
tools: true
|
|
@@ -818,7 +1055,6 @@
|
|
|
818
1055
|
minP: 0.0
|
|
819
1056
|
presencePenalty: 0.0
|
|
820
1057
|
repetitionPenalty: 1.0
|
|
821
|
-
maxTokens: 32768
|
|
822
1058
|
instruct:
|
|
823
1059
|
temperature: 0.7
|
|
824
1060
|
topP: 0.80
|
|
@@ -826,7 +1062,6 @@
|
|
|
826
1062
|
minP: 0.0
|
|
827
1063
|
presencePenalty: 1.5
|
|
828
1064
|
repetitionPenalty: 1.0
|
|
829
|
-
maxTokens: 32768
|
|
830
1065
|
gpuTiers:
|
|
831
1066
|
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 thanks to the Q4_K_M weights (~16.5 GB)."
|
|
832
1067
|
runtimePreference:
|
|
@@ -850,21 +1085,31 @@
|
|
|
850
1085
|
- qwopus3.6-35b-a3b-coder-mtp
|
|
851
1086
|
- qwopus3.6-35b-a3b-coder
|
|
852
1087
|
- qwopus3.6-35b-a3b
|
|
1088
|
+
# The mini router's short id for this same gguf. At 17 characters it outbids
|
|
1089
|
+
# the 13-character `qwopus3.6-27b` of the preview family, whichever order
|
|
1090
|
+
# the two entries load in.
|
|
1091
|
+
- qwopus3.6-35b-moe
|
|
853
1092
|
capabilities:
|
|
854
1093
|
chat: true
|
|
855
1094
|
tools: true
|
|
856
1095
|
toolCallFormat: qwen
|
|
857
|
-
# Reasoning class:
|
|
858
|
-
#
|
|
859
|
-
#
|
|
860
|
-
#
|
|
861
|
-
#
|
|
862
|
-
#
|
|
1096
|
+
# Reasoning class: on-off, and which field carries the switch depends on the
|
|
1097
|
+
# runtime. The card frames the Coder-MTP line as thinking-off execution, but
|
|
1098
|
+
# the wire disagrees on both hosts measured: the model reasons unless it is
|
|
1099
|
+
# told not to, or stays silent unless it is told to, and never ignores the
|
|
1100
|
+
# control that runtime honors.
|
|
1101
|
+
#
|
|
1102
|
+
# llama.cpp, 2026-09-02: chat_template_kwargs.enable_thinking is the switch.
|
|
1103
|
+
# true took a one-line arithmetic prompt from 3 completion tokens to 130
|
|
1104
|
+
# with 338 characters of reasoning_content, while reasoning_effort none, low
|
|
1105
|
+
# and high all left it at 3 tokens with no trace.
|
|
863
1106
|
#
|
|
864
|
-
#
|
|
865
|
-
# prompts to rule out cache hits)
|
|
866
|
-
# that
|
|
867
|
-
#
|
|
1107
|
+
# LM Studio, 2026-08-08: exactly the reverse. enable_thinking was inert
|
|
1108
|
+
# (verified with unique prompts to rule out cache hits) and reasoning_effort
|
|
1109
|
+
# "none" was the control that worked, taking a prompt from 98 reasoning
|
|
1110
|
+
# tokens to 0. Clio's on-off mechanism sends the template flag on every
|
|
1111
|
+
# runtime and adds LM Studio's reasoning_effort spelling on that one, so
|
|
1112
|
+
# both surfaces get a dial that reaches the model.
|
|
868
1113
|
#
|
|
869
1114
|
# reasoning=false previously resolved to mechanism "none", and mechanism
|
|
870
1115
|
# "none" makes openai-completions strip reasoning_effort from the payload.
|
|
@@ -882,25 +1127,75 @@
|
|
|
882
1127
|
maxTokens: 32768
|
|
883
1128
|
quirks:
|
|
884
1129
|
measuredUnder:
|
|
885
|
-
hardware: unknown
|
|
886
|
-
runtime: "lmstudio and its OpenAI-compatible port for the reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
|
|
887
|
-
build: "
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
1130
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 4 request slots, for the 2026-09-02 run; unknown for both earlier dates"
|
|
1131
|
+
runtime: "llamacpp for the 2026-09-02 on-off measurement; lmstudio and its OpenAI-compatible port for the 2026-08-08 reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
|
|
1132
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin behind a router reporting build_info b1-c841aee, for the 2026-09-02 run. Unknown for both earlier dates; the Windows LM Studio measurement host's bundled build string was first recorded on 2026-08-29 as llama.cpp-win-x86_64-nvidia-cuda12-avx2, which is three weeks after the reasoning measurement and cannot be backdated to it"
|
|
1133
|
+
model: "Qwopus3.6-35B-A3B-Coder-MTP Q4_K_M (arch qwen35moe)"
|
|
1134
|
+
llamaCpp: >-
|
|
1135
|
+
2026-09-02 only: --ctx-size 786432 --parallel 4 --no-kv-unified
|
|
1136
|
+
--cache-type-k q8_0 --cache-type-v q8_0 --batch-size 16384 --ubatch-size
|
|
1137
|
+
2048 --flash-attn true --fit off --n-gpu-layers 99 --jinja --reasoning
|
|
1138
|
+
off --chat-template-kwargs {"enable_thinking": false} --temperature 0.6
|
|
1139
|
+
--top-k 20 --top-p 0.95 --min-p 0.0 --presence-penalty 0.0
|
|
1140
|
+
--repeat-last-n 256 --repeat-penalty 1.0. Unknown for both earlier
|
|
1141
|
+
dates; the reasoning measurement went through LM Studio's HTTP port,
|
|
1142
|
+
which does not expose the underlying server argv, and no argv survives
|
|
1143
|
+
for the dispatch that produced the presence-penalty value
|
|
1144
|
+
date: "2026-09-02 (llama.cpp on-off switch), 2026-08-08 (reasoning_effort and enable_thinking on LM Studio), 2026-07-06 (presence penalty)"
|
|
1145
|
+
source: "issue #263 thinking sweep for the llama.cpp date; commits b3db3b82 and f7d76bf4 for the two earlier ones"
|
|
891
1146
|
note: |
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
1147
|
+
Measured 2026-09-02 on llama.cpp: with no thinking field, and with
|
|
1148
|
+
chat_template_kwargs.enable_thinking false, the model answered a
|
|
1149
|
+
one-line arithmetic prompt in 3 completion tokens with no
|
|
1150
|
+
reasoning_content; with enable_thinking true the same prompt spent 130
|
|
1151
|
+
tokens, 338 characters of them reasoning. reasoning_effort none, low and
|
|
1152
|
+
high all left it at 3 tokens with no trace. The silent baseline is the
|
|
1153
|
+
router's doing rather than the checkpoint's: the preset launches this
|
|
1154
|
+
model with --reasoning off and chat_template_kwargs enable_thinking
|
|
1155
|
+
false, which is the opposite default from the LM Studio host below.
|
|
1156
|
+
|
|
1157
|
+
Three numbers from the earlier dates are also measurements. On
|
|
1158
|
+
2026-08-08 on that LM Studio host this model spent 98 of 103 completion
|
|
1159
|
+
tokens reasoning on "what is 17+25" with no thinking field set;
|
|
1160
|
+
chat_template_kwargs enable_thinking was inert, re-checked with unique
|
|
1161
|
+
prompts to rule out prompt-cache hits; and reasoning_effort "none" took
|
|
1162
|
+
the same prompt to 0. The same session recorded a wiki planning dispatch
|
|
1163
|
+
over a 1007-file repository going from 89,501 reasoning tokens, 47 tool
|
|
1164
|
+
calls and exit 1 at 459 s to 0 reasoning tokens, 8 tool calls and exit 0
|
|
1165
|
+
at 218 s. On 2026-07-06 a live dispatch measured presence_penalty 1.5:
|
|
1166
|
+
without it a coder worker repeated one code_nav call into the loop-guard
|
|
1167
|
+
abort on 3 of 3 runs, and with it the same task passed 3 of 3. Neither
|
|
1168
|
+
commit records the hardware or the server configuration, so both stay
|
|
1169
|
+
unknown.
|
|
1170
|
+
|
|
1171
|
+
The 2026-09-02 server also differs from the profile recommended below.
|
|
1172
|
+
It ran ctx-size 786432 with --parallel 4 and --no-kv-unified, giving each
|
|
1173
|
+
slot 196608 tokens, and served text-only with no mmproj and no MTP draft
|
|
1174
|
+
where the block below recommends 262144 at parallel 1 with both active.
|
|
1175
|
+
The vision capability stays the checkpoint's; that run simply did not
|
|
1176
|
+
load the projector. That has a consequence for any projector-less
|
|
1177
|
+
deployment, because the model Clio builds takes vision from this entry
|
|
1178
|
+
and from a target-level override only: synthLocalModel merges the
|
|
1179
|
+
runtime defaults, this entry and target.capabilities and passes no probe
|
|
1180
|
+
layer, then sets input to ["text","image"] whenever vision is true. Once
|
|
1181
|
+
this id resolves here rather than to the 27B preview it did before issue
|
|
1182
|
+
#263, mini's text-only preset would be offered image input it rejects,
|
|
1183
|
+
so a target serving this checkpoint without --mmproj needs
|
|
1184
|
+
capabilities.vision false on the target entry.
|
|
1185
|
+
|
|
1186
|
+
Why on-off replaced the four-level effort map, and what it costs. The
|
|
1187
|
+
map gave LM Studio a low/medium/high granularity nothing had measured,
|
|
1188
|
+
and gave llama.cpp an active level that put reasoning_effort on the wire
|
|
1189
|
+
and nothing else, which this template does not read. on-off is what both
|
|
1190
|
+
measurements support. The cost is that resolveRequestCapability adds the
|
|
1191
|
+
reasoning_effort spelling for an on-off family only when runtimeId is
|
|
1192
|
+
literally `lmstudio` (REASONING_EFFORT_ONLY_RUNTIMES), while an
|
|
1193
|
+
effort-levels family emitted it on every runtime. An LM Studio server
|
|
1194
|
+
reached through a generic openai-compat, litellm or vllm target
|
|
1195
|
+
therefore now receives only the template flag, which that host ignores,
|
|
1196
|
+
and the model reasons at every level. Point such a target at the
|
|
1197
|
+
`lmstudio` runtime, which is what the 2026-08-08 finding above was taken
|
|
1198
|
+
on and what carries the spelling that reaches it.
|
|
904
1199
|
sampling:
|
|
905
1200
|
instruct:
|
|
906
1201
|
temperature: 0.2
|
|
@@ -913,7 +1208,6 @@
|
|
|
913
1208
|
# code_nav call into the loop-guard abort on 3 of 3 runs; with it the
|
|
914
1209
|
# same task passed 3 of 3 with clean edit-then-validate trajectories.
|
|
915
1210
|
presencePenalty: 1.5
|
|
916
|
-
maxTokens: 32768
|
|
917
1211
|
gpuTiers:
|
|
918
1212
|
"32gb": "Default 32 GB local profile: Q4_K_M, ctx=262144, q8 KV, flash attention, parallel=1, with F32 mmproj (vision) and draft-mtp speculative decoding both active. Leaves several GiB free at 262k on a 32 GiB class card."
|
|
919
1213
|
runtimePreference:
|
|
@@ -933,21 +1227,13 @@
|
|
|
933
1227
|
chatTemplateKwargs:
|
|
934
1228
|
enable_thinking: false
|
|
935
1229
|
thinking:
|
|
936
|
-
mechanism:
|
|
937
|
-
effortByLevel:
|
|
938
|
-
# "none" is the only value that silences this family. It is carried on
|
|
939
|
-
# the wire because the model reasons by default, so sending nothing is
|
|
940
|
-
# in effect a request to keep reasoning.
|
|
941
|
-
off: none
|
|
942
|
-
low: low
|
|
943
|
-
medium: medium
|
|
944
|
-
high: high
|
|
1230
|
+
mechanism: on-off
|
|
945
1231
|
guidance: |
|
|
946
|
-
|
|
947
|
-
"none",
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
1232
|
+
One switch, two spellings: chat_template_kwargs.enable_thinking on
|
|
1233
|
+
llama.cpp, reasoning_effort ("none" off, "low" on) on LM Studio, each
|
|
1234
|
+
inert on the other host. Intermediate levels coerce to on, and off has
|
|
1235
|
+
to be carried on the wire because the checkpoint reasons unless told
|
|
1236
|
+
not to, whatever its card says about thinking-off execution.
|
|
951
1237
|
serving: "Jackrong Qwopus3.6 35B-A3B Coder MTP, Q4_K_M. Thinking-off coder/agent tuned for tool-use loops. Served as a local router default with MTP speculative decoding (spec-type draft-mtp, n_max 2) and F32 mmproj vision both active at 262k."
|
|
952
1238
|
|
|
953
1239
|
- family: qwopus3.6-27b-coder
|
|
@@ -994,7 +1280,6 @@
|
|
|
994
1280
|
# Same Coder-MTP non-thinking design as the 35B; see the measured
|
|
995
1281
|
# presence-penalty note there.
|
|
996
1282
|
presencePenalty: 1.5
|
|
997
|
-
maxTokens: 32768
|
|
998
1283
|
runtimePreference:
|
|
999
1284
|
llamaCpp: "A local llama.cpp router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
1000
1285
|
lmstudio: "Dense 27B Coder-MTP; configure MTP in LM Studio's per-model load settings."
|
|
@@ -1044,7 +1329,6 @@
|
|
|
1044
1329
|
topP: 0.9
|
|
1045
1330
|
topK: 20
|
|
1046
1331
|
repeatPenalty: 1.05
|
|
1047
|
-
maxTokens: 32768
|
|
1048
1332
|
runtimePreference:
|
|
1049
1333
|
llamaCpp: "A local llama.cpp router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
1050
1334
|
thinking:
|
|
@@ -1062,6 +1346,13 @@
|
|
|
1062
1346
|
- gemma4-26b-a4b-it
|
|
1063
1347
|
- gemma-4-26b-a4b-it-q4
|
|
1064
1348
|
- gemma-4-26b-a4b-it-q4_k_m
|
|
1349
|
+
# The mini router's short id for the same checkpoint; its alias is
|
|
1350
|
+
# Gemma-4-26B-A4B-it-Q4_K_M-262K and it loads
|
|
1351
|
+
# gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf. QAT changes how the weights are
|
|
1352
|
+
# represented, not the chat template, so the on-off mechanism below carries
|
|
1353
|
+
# over and there is no reason to split the family the way the 31B nvfp4 and
|
|
1354
|
+
# qat builds are split (those declare different context windows).
|
|
1355
|
+
- gemma4-26b-moe
|
|
1065
1356
|
capabilities:
|
|
1066
1357
|
chat: true
|
|
1067
1358
|
tools: true
|
|
@@ -1077,12 +1368,48 @@
|
|
|
1077
1368
|
contextWindow: 262144
|
|
1078
1369
|
maxTokens: 65536
|
|
1079
1370
|
quirks:
|
|
1371
|
+
measuredUnder:
|
|
1372
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 1 request slot"
|
|
1373
|
+
runtime: llamacpp
|
|
1374
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee"
|
|
1375
|
+
model: "Gemma-4-26B-A4B-it-qat UD-Q4_K_XL with mmproj-F16, no draft model (arch gemma4)"
|
|
1376
|
+
llamaCpp: >-
|
|
1377
|
+
--ctx-size 262144 --parallel 1 --kv-unified --cache-type-k f16
|
|
1378
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
1379
|
+
true --n-gpu-layers 99 --jinja --reasoning off --temperature 0.8
|
|
1380
|
+
--top-k 64 --top-p 0.95 --repeat-penalty 1.05
|
|
1381
|
+
date: "2026-09-02"
|
|
1382
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1383
|
+
note: |
|
|
1384
|
+
The on-off mechanism this entry already claimed is now measured on
|
|
1385
|
+
llama.cpp rather than inherited from the gemma-4 template. With no
|
|
1386
|
+
thinking field, and with chat_template_kwargs.enable_thinking false, the
|
|
1387
|
+
model answered in 3 completion tokens with no reasoning_content; with it
|
|
1388
|
+
true the same prompt spent 99 tokens, 227 characters of them reasoning.
|
|
1389
|
+
reasoning_effort is inert: none, low, and high all stayed at 3 tokens
|
|
1390
|
+
with no trace, so this family is on-off and not effort-levels on this
|
|
1391
|
+
runtime. The router serves it with --reasoning off, which is why the
|
|
1392
|
+
no-field baseline is silent.
|
|
1393
|
+
|
|
1394
|
+
Three serving facts differ from the recommended profile below. The
|
|
1395
|
+
measuring server ran f16 KV where the block below recommends q8_0, and
|
|
1396
|
+
it ran at batch 16384 and ubatch 2048. It also ran without speculative
|
|
1397
|
+
decoding: the preset's --tags string advertises
|
|
1398
|
+
spec:mtp,spec_draft_n:4,draft:mtp-gemma-4-26B-A4B-it, but the argv
|
|
1399
|
+
carries no --model-draft, --spec-type or --spec-draft-n-max, so the
|
|
1400
|
+
token counts above come from a plain decode. The 31B qat sibling above
|
|
1401
|
+
does pass those flags, which is where the tags were copied from.
|
|
1402
|
+
|
|
1403
|
+
Separately, the round-5 cache sweep on this same date measured the
|
|
1404
|
+
Gemma 4 tokenizer rendering Clio's full-capability first prompt at 7,353
|
|
1405
|
+
tokens against roughly 8.4k for the Qwen tokenizer on the identical
|
|
1406
|
+
prompt, so this family buys about a thousand tokens of headroom per turn
|
|
1407
|
+
over the Qwen-tokenized families.
|
|
1080
1408
|
sampling:
|
|
1081
1409
|
thinking:
|
|
1082
1410
|
temperature: 0.3
|
|
1083
1411
|
topP: 0.9
|
|
1084
1412
|
topK: 20
|
|
1085
|
-
maxTokens: 8192
|
|
1086
1413
|
reasoningBudget: 4096
|
|
1087
1414
|
instruct:
|
|
1088
1415
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
@@ -1153,7 +1480,6 @@
|
|
|
1153
1480
|
temperature: 0.3
|
|
1154
1481
|
topP: 0.9
|
|
1155
1482
|
topK: 20
|
|
1156
|
-
maxTokens: 8192
|
|
1157
1483
|
gpuTiers:
|
|
1158
1484
|
"32gb": "Preferred local worker model when a text-only worker is enough."
|
|
1159
1485
|
runtimePreference:
|
|
@@ -1221,13 +1547,11 @@
|
|
|
1221
1547
|
temperature: 0.6
|
|
1222
1548
|
topP: 0.95
|
|
1223
1549
|
topK: 20
|
|
1224
|
-
maxTokens: 16384
|
|
1225
1550
|
reasoningBudget: 4096
|
|
1226
1551
|
instruct:
|
|
1227
1552
|
temperature: 0.2
|
|
1228
1553
|
topP: 0.9
|
|
1229
1554
|
topK: 20
|
|
1230
|
-
maxTokens: 4096
|
|
1231
1555
|
gpuTiers:
|
|
1232
1556
|
"32gb": "Distilled Opus-style reasoning at Qwen3.5 35B A3B scale. Use as a strong local main agent when GPU room is available."
|
|
1233
1557
|
runtimePreference:
|
|
@@ -1288,12 +1612,10 @@
|
|
|
1288
1612
|
temperature: 0.2
|
|
1289
1613
|
topP: 0.9
|
|
1290
1614
|
topK: 20
|
|
1291
|
-
maxTokens: 4096
|
|
1292
1615
|
thinking:
|
|
1293
1616
|
temperature: 0.6
|
|
1294
1617
|
topP: 0.95
|
|
1295
1618
|
topK: 20
|
|
1296
|
-
maxTokens: 8192
|
|
1297
1619
|
reasoningBudget: 2048
|
|
1298
1620
|
gpuTiers:
|
|
1299
1621
|
"16gb": "Use this as the local 16GB emulation target. It keeps the output cap below the 32GB-class MoE defaults."
|
|
@@ -1316,3 +1638,246 @@
|
|
|
1316
1638
|
32GB-class MoE families because this is the 16GB emulation target;
|
|
1317
1639
|
budget for a reasoning preamble ahead of every answer.
|
|
1318
1640
|
serving: "LM Studio reports trained_for_tool_use for the local qwopus3.5-9b-v3 target. Configure residency through LM Studio or explicit lmstudio.load settings."
|
|
1641
|
+
|
|
1642
|
+
# Meta's Muse Glimmer is a new vendor, architecture and template for this
|
|
1643
|
+
# catalog: no existing family shares its perception encoder, its DFlash drafter,
|
|
1644
|
+
# or its reasoning control. It is listed under always-on because its dial is not
|
|
1645
|
+
# representable, not because the model reasons unconditionally by design; the
|
|
1646
|
+
# measuredUnder note below records what was tried.
|
|
1647
|
+
- family: muse-glimmer-30b
|
|
1648
|
+
matchPatterns:
|
|
1649
|
+
# No bare `muse-glimmer`: it would swallow a future Muse Glimmer of another
|
|
1650
|
+
# size, which shares neither this 131072 window nor a measured dial.
|
|
1651
|
+
- unsloth/muse-glimmer-30b-gguf
|
|
1652
|
+
- muse-glimmer-30b
|
|
1653
|
+
# The mini router serves the same weights under both this short id and the
|
|
1654
|
+
# unsloth repo name above.
|
|
1655
|
+
- muse-30b-dense
|
|
1656
|
+
capabilities:
|
|
1657
|
+
chat: true
|
|
1658
|
+
tools: true
|
|
1659
|
+
toolCallFormat: openai
|
|
1660
|
+
reasoning: true
|
|
1661
|
+
thinkingFormat: qwen-chat-template
|
|
1662
|
+
structuredOutputs: json-schema
|
|
1663
|
+
vision: true
|
|
1664
|
+
audio: false
|
|
1665
|
+
embeddings: false
|
|
1666
|
+
rerank: false
|
|
1667
|
+
fim: false
|
|
1668
|
+
contextWindow: 131072
|
|
1669
|
+
maxTokens: 32768
|
|
1670
|
+
quirks:
|
|
1671
|
+
measuredUnder:
|
|
1672
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
1673
|
+
runtime: llamacpp
|
|
1674
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
1675
|
+
model: "Muse-Glimmer-30B UD-Q4_K_XL with dflash-kquant draft and mmproj-Muse-Glimmer-30B-Q8_0 (arch muse-glimmer)"
|
|
1676
|
+
llamaCpp: >-
|
|
1677
|
+
--ctx-size 131072 --parallel 1 --kv-unified --cache-type-k q8_0
|
|
1678
|
+
--cache-type-v q8_0 --batch-size 2048 --ubatch-size 512 --flash-attn
|
|
1679
|
+
true --fit off --n-gpu-layers 99 --spec-type draft-dflash
|
|
1680
|
+
--spec-draft-n-max 16 --n-gpu-layers-draft 99 --jinja
|
|
1681
|
+
--chat-template-kwargs {"reasoning_strength": "xhigh"} --temperature 1.0
|
|
1682
|
+
--top-k 64 --top-p 0.95
|
|
1683
|
+
date: "2026-09-02"
|
|
1684
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1685
|
+
note: |
|
|
1686
|
+
always-on is the measured classification, and it is measured against
|
|
1687
|
+
both controls Clio can send. With no thinking field the model spent 99
|
|
1688
|
+
completion tokens with 358 characters of reasoning_content;
|
|
1689
|
+
chat_template_kwargs.enable_thinking false still produced 91 tokens with
|
|
1690
|
+
332 characters, and true produced 90 with 312; reasoning_effort none,
|
|
1691
|
+
low, and high produced 81, 83, and 61 tokens, every one of them with a
|
|
1692
|
+
trace. Nothing Clio can put on the wire silences it.
|
|
1693
|
+
|
|
1694
|
+
The reason is the model's own design rather than a template bug. The
|
|
1695
|
+
card's dial is reasoning strength with levels low, medium, high, and
|
|
1696
|
+
xhigh, set through the system prompt as "Reasoning strength: <value>",
|
|
1697
|
+
and the router passes it as chat_template_kwargs.reasoning_strength
|
|
1698
|
+
pinned to xhigh. There is no off level on that scale, and no
|
|
1699
|
+
ThinkingMechanism in this engine emits that key:
|
|
1700
|
+
resolveRequestCapability populates chat_template_kwargs only with
|
|
1701
|
+
enable_thinking or, on the harmony path, reasoning_effort. Encoding this
|
|
1702
|
+
family as on-off would put a flag on the wire that this template does
|
|
1703
|
+
not read while the TUI showed a working off/on dial, which is the
|
|
1704
|
+
failure qwopus3.5-9b-v3 above was reclassified to stop. A real dial for
|
|
1705
|
+
this family needs a per-family kwarg key in the engine, not a catalog
|
|
1706
|
+
edit.
|
|
1707
|
+
|
|
1708
|
+
Sampler provenance: temperature 1.0, top_p 0.95 and top_k 64 are the
|
|
1709
|
+
card's Best Practices values and the router preset uses the same three.
|
|
1710
|
+
The preset sets no repeat penalty and the card recommends none, so this
|
|
1711
|
+
entry declares none. contextWindow is the card's 131072, which the
|
|
1712
|
+
router matches at ctx-size 131072. maxTokens has no card value and
|
|
1713
|
+
follows the other 131072-context families here.
|
|
1714
|
+
sampling:
|
|
1715
|
+
thinking:
|
|
1716
|
+
temperature: 1.0
|
|
1717
|
+
topP: 0.95
|
|
1718
|
+
topK: 64
|
|
1719
|
+
instruct:
|
|
1720
|
+
temperature: 1.0
|
|
1721
|
+
topP: 0.95
|
|
1722
|
+
topK: 64
|
|
1723
|
+
gpuTiers:
|
|
1724
|
+
"32gb": "Agent-role target on a 32 GB card: UD-Q4_K_XL weights, q8_0 KV at ctx=131072, parallel=1, with the Q8_0 perception encoder and the DFlash drafter both resident."
|
|
1725
|
+
runtimePreference:
|
|
1726
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Set the reasoning budget at launch with --chat-template-kwargs {\"reasoning_strength\": \"...\"}; there is no per-request field Clio can send for it."
|
|
1727
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface. Expect a reasoning trace on every response whatever the configured thinking level says."
|
|
1728
|
+
llamaCpp:
|
|
1729
|
+
ctxSize: 131072
|
|
1730
|
+
cacheTypeK: q8_0
|
|
1731
|
+
cacheTypeV: q8_0
|
|
1732
|
+
flashAttn: true
|
|
1733
|
+
nGpuLayers: 99
|
|
1734
|
+
parallel: 1
|
|
1735
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 131072-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
1736
|
+
batchSize: 2048
|
|
1737
|
+
ubatchSize: 512
|
|
1738
|
+
mmproj: true
|
|
1739
|
+
specType: draft-dflash
|
|
1740
|
+
specDraftNMax: 16
|
|
1741
|
+
thinking:
|
|
1742
|
+
mechanism: always-on
|
|
1743
|
+
chatTemplateKwargs:
|
|
1744
|
+
byLevel:
|
|
1745
|
+
# Card: reasoning strength low|medium|high|xhigh (scratchpad/catalog/models/unsloth-Muse-Glimmer-30B.md:35). The per-family kwarg key the note above asked for.
|
|
1746
|
+
key: reasoning_strength
|
|
1747
|
+
values:
|
|
1748
|
+
low: low
|
|
1749
|
+
medium: medium
|
|
1750
|
+
high: high
|
|
1751
|
+
xhigh: xhigh
|
|
1752
|
+
lmstudio: unsupported
|
|
1753
|
+
lmstudio: unsupported
|
|
1754
|
+
guidance: |
|
|
1755
|
+
Muse Glimmer reasons on every turn and its strength scale has no off
|
|
1756
|
+
level, so Clio shows the dial as forced and budgets for a reasoning
|
|
1757
|
+
preamble ahead of every answer. Change the amount of thinking at the
|
|
1758
|
+
server, with the reasoning_strength chat-template kwarg; a per-request
|
|
1759
|
+
thinking level cannot reach this model.
|
|
1760
|
+
serving: "Meta Superintelligence Lab Muse-Glimmer-30B, a dense 29.6B causal transformer with a ~1.8B ViT-G/14 perception encoder, interleaved local and global attention with a 2048 sliding window, and a DFlash block-diffusion drafter that proposes 16 tokens per forward pass. Card context length is 131072. Input is text plus image; the mini router additionally tags video, which the card does not claim, so the catalog follows the card."
|
|
1761
|
+
|
|
1762
|
+
# ThinkingCap is a brevity finetune of Qwen3.6-27B and not the base model, so it
|
|
1763
|
+
# keeps its own family rather than borrowing qwen3.6-27b's. No pattern here may
|
|
1764
|
+
# be a bare `qwen3.6` spelling: `lookup` is a substring test, so `qwen3.6` alone
|
|
1765
|
+
# would match every base-model id too and win or lose on length rather than on
|
|
1766
|
+
# which checkpoint the id names. The `thinkingcap-` prefixed spellings below all
|
|
1767
|
+
# contain `qwen3.6` harmlessly, because a longer pattern only matches ids that
|
|
1768
|
+
# contain the whole longer string.
|
|
1769
|
+
- family: thinkingcap-qwen3.6-27b
|
|
1770
|
+
matchPatterns:
|
|
1771
|
+
# No bare `thinkingcap`: it would swallow a ThinkingCap built on another
|
|
1772
|
+
# base, such as a Qwen3.8 32B, and hand it this q4's sampler, its 131072
|
|
1773
|
+
# window and a dial measured on a different checkpoint.
|
|
1774
|
+
- thinkingcap-qwen3.6-27b
|
|
1775
|
+
- thinkingcap-27b-dense-q4
|
|
1776
|
+
- thinkingcap-27b-dense
|
|
1777
|
+
- thinkingcap-27b
|
|
1778
|
+
capabilities:
|
|
1779
|
+
chat: true
|
|
1780
|
+
tools: true
|
|
1781
|
+
toolCallFormat: qwen
|
|
1782
|
+
reasoning: true
|
|
1783
|
+
thinkingFormat: qwen-chat-template
|
|
1784
|
+
structuredOutputs: json-schema
|
|
1785
|
+
vision: false
|
|
1786
|
+
audio: false
|
|
1787
|
+
embeddings: false
|
|
1788
|
+
rerank: false
|
|
1789
|
+
fim: false
|
|
1790
|
+
contextWindow: 131072
|
|
1791
|
+
maxTokens: 32768
|
|
1792
|
+
quirks:
|
|
1793
|
+
measuredUnder:
|
|
1794
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
1795
|
+
runtime: llamacpp
|
|
1796
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
1797
|
+
model: "ThinkingCap-27B Q4_K_M with the MTP draft head baked into the gguf (arch qwen36)"
|
|
1798
|
+
llamaCpp: >-
|
|
1799
|
+
--ctx-size 131072 --parallel 1 --kv-unified --cache-type-k f16
|
|
1800
|
+
--cache-type-v f16 --batch-size 2048 --ubatch-size 512 --flash-attn true
|
|
1801
|
+
--fit off --n-gpu-layers 99 --spec-type draft-mtp --spec-draft-n-max 4
|
|
1802
|
+
--n-gpu-layers-draft 99 --jinja --reasoning on --chat-template-kwargs
|
|
1803
|
+
{"enable_thinking": true} --temperature 1.0 --top-k 20 --top-p 0.95
|
|
1804
|
+
--min-p 0.0 --repeat-penalty 1.0
|
|
1805
|
+
date: "2026-09-02"
|
|
1806
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1807
|
+
note: |
|
|
1808
|
+
The router tag is wrong for this model and the measurement wins. It is
|
|
1809
|
+
the only reasoning model on this router tagged reasoning:on rather than
|
|
1810
|
+
reasoning:on_off, and it carries no default_reasoning tag at all, which
|
|
1811
|
+
reads as always-on. It is not. On the q4 that this family is keyed to,
|
|
1812
|
+
the no-field baseline spent 125 completion tokens with 354 characters of
|
|
1813
|
+
reasoning_content, chat_template_kwargs.enable_thinking false answered
|
|
1814
|
+
in 3 tokens with none, and true spent 130 tokens with 388 characters.
|
|
1815
|
+
reasoning_effort is inert: none, low, and high produced 115, 140, and
|
|
1816
|
+
131 tokens with a trace every time. The default-on behavior comes from
|
|
1817
|
+
the preset rather than the checkpoint, because the
|
|
1818
|
+
[thinkingcap-27b-dense-q4] block launches the server with reasoning = on
|
|
1819
|
+
and chat-template-kwargs {"enable_thinking": true}.
|
|
1820
|
+
|
|
1821
|
+
Issue #263 names thinkingcap-27b-dense-q6, which no longer exists. That
|
|
1822
|
+
quant was measured on this same date (baseline 99 tokens with 246
|
|
1823
|
+
characters of reasoning, enable_thinking false 3 tokens with none, true
|
|
1824
|
+
98 tokens with 291 characters, the same on-off shape as the q4) and then
|
|
1825
|
+
deleted from mini: the weights are gone and the preset block is removed
|
|
1826
|
+
from the router, which now serves the q4 as the only ThinkingCap id.
|
|
1827
|
+
This family therefore names no q6 quant. The short `thinkingcap-27b`
|
|
1828
|
+
pattern still covers that spelling if the operator ever restores it.
|
|
1829
|
+
|
|
1830
|
+
One serving observation was taken on the q6 before it was removed and
|
|
1831
|
+
has not been repeated on the q4: in the round-5 cache sweep this model
|
|
1832
|
+
re-prefilled 661 tokens on a `--continue` turn where the other eight
|
|
1833
|
+
router models kept their prefix. Read it as a lead rather than a fact
|
|
1834
|
+
about the q4.
|
|
1835
|
+
|
|
1836
|
+
Sampler provenance: temperature 0.6, top_p 0.95 and top_k 20 are the
|
|
1837
|
+
quantizer's stated intended sampling for this finetune, which also warns
|
|
1838
|
+
that greedy decoding makes it loop without closing the think block. The
|
|
1839
|
+
router preset instead runs temperature 1.0 with the same top_p and
|
|
1840
|
+
top_k, plus min_p 0.0 and repeat_penalty 1.0. The catalog follows the
|
|
1841
|
+
card, as it does for every other family here, and records the preset
|
|
1842
|
+
divergence above.
|
|
1843
|
+
sampling:
|
|
1844
|
+
thinking:
|
|
1845
|
+
temperature: 0.6
|
|
1846
|
+
topP: 0.95
|
|
1847
|
+
topK: 20
|
|
1848
|
+
minP: 0.0
|
|
1849
|
+
repeatPenalty: 1.0
|
|
1850
|
+
instruct:
|
|
1851
|
+
temperature: 0.6
|
|
1852
|
+
topP: 0.95
|
|
1853
|
+
topK: 20
|
|
1854
|
+
minP: 0.0
|
|
1855
|
+
repeatPenalty: 1.0
|
|
1856
|
+
gpuTiers:
|
|
1857
|
+
"32gb": "Reasoning-role target on a 32 GB card: Q4_K_M weights (about 17 GB) with f16 KV at ctx=131072, parallel=1, and the baked-in MTP draft head active."
|
|
1858
|
+
runtimePreference:
|
|
1859
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. The MTP head ships inside the gguf, so --spec-type draft-mtp needs no second model. Never decode greedy: temperature 0 makes this finetune loop without closing its think block."
|
|
1860
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen reasoning and tool parsing."
|
|
1861
|
+
llamaCpp:
|
|
1862
|
+
ctxSize: 131072
|
|
1863
|
+
cacheTypeK: f16
|
|
1864
|
+
cacheTypeV: f16
|
|
1865
|
+
flashAttn: true
|
|
1866
|
+
nGpuLayers: 99
|
|
1867
|
+
parallel: 1
|
|
1868
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 131072-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
1869
|
+
batchSize: 2048
|
|
1870
|
+
ubatchSize: 512
|
|
1871
|
+
specType: draft-mtp
|
|
1872
|
+
specDraftNMax: 4
|
|
1873
|
+
chatTemplateKwargs:
|
|
1874
|
+
enable_thinking: true
|
|
1875
|
+
thinking:
|
|
1876
|
+
mechanism: on-off
|
|
1877
|
+
guidance: |
|
|
1878
|
+
ThinkingCap keeps the Qwen3.6 chat template's enable_thinking switch;
|
|
1879
|
+
intermediate levels coerce to on. The router serves it with thinking on
|
|
1880
|
+
by default, so off has to be carried on the wire rather than omitted,
|
|
1881
|
+
and reasoning_effort is inert on llama.cpp. The finetune's point is a
|
|
1882
|
+
shorter chain rather than no chain, so an active level is cheap.
|
|
1883
|
+
serving: "BottleCapAI ThinkingCap-Qwen3.6-27B, a brevity finetune of Qwen3.6-27B that holds accuracy at roughly 40 to 46 percent fewer thinking tokens (the quantizer measured a 15-prompt reasoning set at mean 675 tokens against 1401 for base Qwen3.6-27B). Served on mini as the Q4_K_M gguf with the MTP draft head embedded in the file, so speculative decoding needs no second model."
|