@iowarp/clio-coder 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/CONTRIBUTING.md +142 -52
- package/README.md +434 -473
- package/SECURITY.md +2 -1
- package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
- package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
- package/dist/builtins-IA7V7FUC.js +22 -0
- package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
- package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
- package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
- package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
- package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
- package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
- package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
- package/dist/chunk-3EBYEESD.js +314 -0
- package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
- package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
- package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
- package/dist/chunk-4UVU7BJ5.js +39 -0
- package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
- package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
- package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
- package/dist/chunk-5MEZN6CB.js +1334 -0
- package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
- package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
- package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
- package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
- package/dist/chunk-7DRAWPTZ.js +360 -0
- package/dist/chunk-7E7I3WLS.js +3762 -0
- package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
- package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
- package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
- package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
- package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
- package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
- package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
- package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
- package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
- package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
- package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
- package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
- package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
- package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
- package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
- package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
- package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
- package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
- package/dist/chunk-GYV6VZOC.js +26 -0
- package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
- package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
- package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
- package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
- package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
- package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
- package/dist/chunk-IJNZMHLA.js +101 -0
- package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
- package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
- package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
- package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
- package/dist/chunk-JEQ3XTHC.js +42 -0
- package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
- package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
- package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
- package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
- package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
- package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
- package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
- package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
- package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
- package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
- package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
- package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
- package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
- package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
- package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
- package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
- package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
- package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
- package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
- package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
- package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
- package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
- package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
- package/dist/chunk-R6Q67RJH.js +134 -0
- package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
- package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
- package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
- package/dist/chunk-SKHCAU7K.js +385 -0
- package/dist/chunk-SZAA6XDG.js +30 -0
- package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
- package/dist/chunk-UOIZ7DA4.js +41 -0
- package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
- package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
- package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
- package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
- package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
- package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
- package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
- package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
- package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
- package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
- package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
- package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
- package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
- package/dist/chunk-XRZT5WY5.js +47 -0
- package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
- package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
- package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
- package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
- package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
- package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
- package/dist/cli/index.js +41 -39
- package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
- package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
- package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
- package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
- package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
- package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
- package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
- package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
- package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
- package/dist/context-map-COB37XXN.js +505 -0
- package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
- package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
- package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
- package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
- package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
- package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
- package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
- package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
- package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
- package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
- package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
- package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
- package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
- package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
- package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
- package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
- package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
- package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
- package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
- package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
- package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
- package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
- package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
- package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
- package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
- package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
- package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
- package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
- package/dist/reset-EAJFFJVB.js +344 -0
- package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
- package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
- package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
- package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
- package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
- package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
- package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
- package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
- package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
- package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
- package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
- package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
- package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
- package/dist/uninstall-5PEVOE5B.js +408 -0
- package/dist/upgrade-M4WXY6KN.js +303 -0
- package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
- package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
- package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
- package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
- package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
- package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
- package/dist/worker/entry.js +90 -74
- package/docs/README.md +176 -81
- package/docs/{acp.md → architecture/acp.md} +36 -20
- package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
- package/docs/{architecture.md → architecture/architecture.md} +43 -22
- package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
- package/docs/architecture/artifact-versions.md +90 -0
- package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
- package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
- package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
- package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
- package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
- package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
- package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
- package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
- package/docs/{observability.md → architecture/observability.md} +38 -14
- package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
- package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
- package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
- package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
- package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
- package/docs/architecture/time-conventions.md +125 -0
- package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
- package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
- package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
- package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
- package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
- package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
- package/docs/guide/configuration-reference.md +1163 -0
- package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
- package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
- package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
- package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
- package/docs/{glossary.md → guide/glossary.md} +14 -11
- package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
- package/docs/guide/panes-and-files.md +290 -0
- package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
- package/docs/{resource-library.md → guide/resource-library.md} +13 -4
- package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
- package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
- package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
- package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
- package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
- package/docs/process/development-pipeline.md +152 -0
- package/docs/process/documentation-coverage.md +100 -0
- package/docs/process/documentation-guide.md +187 -0
- package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
- package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
- package/docs/{evolution.md → process/evolution.md} +2 -2
- package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
- package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
- package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
- package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
- package/evals/README.md +2 -2
- package/evals/behavioral-model.yaml +3 -2
- package/package.json +10 -8
- package/skills/README.md +52 -41
- package/skills/coding/ast-grep/SKILL.md +102 -31
- package/skills/coding/ast-grep/evals.md +26 -0
- package/skills/coding/coding-standards/SKILL.md +41 -6
- package/skills/coding/coding-standards/evals.md +23 -0
- package/skills/coding/prototype/SKILL.md +88 -29
- package/skills/coding/prototype/evals.md +19 -0
- package/skills/coding/tdd/SKILL.md +81 -54
- package/skills/coding/tdd/evals.md +20 -0
- package/skills/context/context-handoff/SKILL.md +44 -3
- package/skills/context/context-handoff/evals.md +44 -0
- package/skills/context/context-prime/SKILL.md +46 -16
- package/skills/context/context-prime/evals.md +45 -0
- package/skills/git/branch-closeout/SKILL.md +132 -0
- package/skills/git/branch-closeout/evals.md +133 -0
- package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
- package/skills/git/file-ticket/SKILL.md +78 -64
- package/skills/git/file-ticket/assets/issue-template.md +22 -0
- package/skills/git/file-ticket/evals.md +31 -26
- package/skills/git/file-ticket/references/issue-discovery.md +49 -0
- package/skills/git/fix-issue/SKILL.md +88 -65
- package/skills/git/fix-issue/evals.md +35 -31
- package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
- package/skills/git/resolve-merge-conflicts/evals.md +52 -25
- package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
- package/skills/git/ship/SKILL.md +103 -67
- package/skills/git/ship/assets/pr-template.md +21 -0
- package/skills/git/ship/evals.md +44 -28
- package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
- package/skills/git/worktree-create/SKILL.md +80 -50
- package/skills/git/worktree-create/evals.md +40 -33
- package/skills/git/worktree-create/references/worktree-setup.md +62 -66
- package/skills/git/worktree-merge/SKILL.md +112 -65
- package/skills/git/worktree-merge/evals.md +42 -34
- package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
- package/skills/meta/clio-coder-dev/SKILL.md +9 -5
- package/skills/meta/clio-coder-dev/evals.md +3 -2
- package/skills/meta/clio-coder-test/SKILL.md +102 -95
- package/skills/meta/clio-coder-test/evals.md +9 -4
- package/skills/meta/clio-coder-test/references/harness.md +100 -124
- package/skills/meta/clio-coder-test/references/test-map.md +77 -50
- package/skills/meta/credentials/SKILL.md +2 -2
- package/skills/meta/find-skills/SKILL.md +2 -2
- package/skills/meta/herdr/SKILL.md +2 -2
- package/skills/meta/skill-craft/SKILL.md +22 -16
- package/skills/planning/archify/SKILL.md +196 -0
- package/skills/planning/archify/evals.md +65 -0
- package/skills/planning/architecture/SKILL.md +62 -13
- package/skills/planning/architecture/evals.md +65 -0
- package/skills/planning/backlog/SKILL.md +131 -15
- package/skills/planning/backlog/evals.md +142 -0
- package/skills/planning/prd/SKILL.md +47 -7
- package/skills/planning/prd/evals.md +54 -0
- package/skills/planning/product-intent/SKILL.md +58 -3
- package/skills/planning/product-intent/evals.md +70 -0
- package/skills/planning/tech-spec/SKILL.md +54 -3
- package/skills/planning/tech-spec/evals.md +73 -0
- package/skills/registry.yaml +70 -62
- package/skills/remote.yaml +13 -0
- package/skills/research/arxiv-literature/SKILL.md +77 -19
- package/skills/research/arxiv-literature/evals.md +50 -0
- package/skills/research/experiment-protocol/SKILL.md +21 -2
- package/skills/research/experiment-protocol/evals.md +23 -0
- package/skills/research/scientific-debugging/SKILL.md +24 -2
- package/skills/research/scientific-debugging/evals.md +18 -0
- package/skills/research/scientific-modernization/SKILL.md +27 -2
- package/skills/research/scientific-modernization/evals.md +27 -0
- package/skills/skill-marketplace.json +97 -62
- package/skills/workflow/cut-it/SKILL.md +66 -6
- package/skills/workflow/cut-it/evals.md +101 -0
- package/skills/workflow/design-council/SKILL.md +118 -28
- package/skills/workflow/design-council/evals.md +161 -0
- package/skills/workflow/grill-me/SKILL.md +87 -11
- package/skills/workflow/grill-me/evals.md +153 -0
- package/skills/workflow/workflow-distiller/SKILL.md +77 -18
- package/skills/workflow/workflow-distiller/evals.md +118 -0
- package/src/cli/args.ts +2 -2
- package/src/cli/bootstrap-generate.ts +1 -1
- package/src/cli/config-inspect.ts +65 -12
- package/src/cli/configure-interop.ts +105 -13
- package/src/cli/configure-oauth.ts +57 -0
- package/src/cli/configure-onboarding.ts +980 -0
- package/src/cli/configure-target.ts +594 -0
- package/src/cli/configure.ts +1082 -532
- package/src/cli/context-map.ts +114 -0
- package/src/cli/context.ts +4 -0
- package/src/cli/docs.ts +22 -14
- package/src/cli/doctor-naming.ts +5 -5
- package/src/cli/doctor-toolchain.ts +3 -3
- package/src/cli/eval.ts +1 -2
- package/src/cli/extensions.ts +2 -1
- package/src/cli/fleet.ts +1 -1
- package/src/cli/index.ts +3 -1
- package/src/cli/internal-dispatch.ts +3 -4
- package/src/cli/lifecycle-presenter.ts +436 -0
- package/src/cli/models.ts +10 -2
- package/src/cli/modes/print.ts +5 -1
- package/src/cli/panes.ts +19 -5
- package/src/cli/reset.ts +228 -106
- package/src/cli/run.ts +9 -4
- package/src/cli/select.ts +664 -0
- package/src/cli/share.ts +5 -1
- package/src/cli/skills-eval.ts +3 -3
- package/src/cli/skills.ts +9 -2
- package/src/cli/targets.ts +5 -6
- package/src/cli/trace.ts +55 -4
- package/src/cli/uninstall.ts +233 -165
- package/src/cli/upgrade.ts +204 -149
- package/src/cli/usage.ts +86 -27
- package/src/cli/validate-model.ts +3 -3
- package/src/cli/wiki-generate.ts +1 -1
- package/src/core/artifact-paths.ts +1 -1
- package/src/core/bash-exec.ts +131 -86
- package/src/core/bus-events.ts +51 -6
- package/src/core/config.ts +61 -1
- package/src/core/defaults.ts +7 -4
- package/src/core/dispatch-outcome.ts +16 -0
- package/src/core/external-diagnostic.ts +44 -0
- package/src/core/gateway-routing.ts +157 -0
- package/src/core/guardrails.ts +10 -49
- package/src/core/prompt-hint.ts +9 -0
- package/src/core/safe-exec.ts +17 -2
- package/src/core/skill-activation.ts +89 -2
- package/src/domains/agents/builtins/architect.md +2 -3
- package/src/domains/agents/builtins/coder.md +3 -2
- package/src/domains/agents/builtins/debugger.md +2 -2
- package/src/domains/agents/builtins/documenter.md +2 -2
- package/src/domains/agents/builtins/git-master.md +1 -1
- package/src/domains/agents/builtins/oracle.md +1 -1
- package/src/domains/agents/builtins/provenance.md +1 -1
- package/src/domains/agents/builtins/researcher.md +1 -1
- package/src/domains/agents/builtins/scout.md +1 -1
- package/src/domains/agents/builtins/tester.md +2 -2
- package/src/domains/agents/builtins/verifier.md +2 -2
- package/src/domains/agents/builtins/wiki-writer.md +1 -1
- package/src/domains/agents/builtins/world-knowledge.md +31 -0
- package/src/domains/agents/catalog.ts +13 -15
- package/src/domains/agents/contract.ts +2 -0
- package/src/domains/agents/extension.ts +23 -1
- package/src/domains/agents/result-contract.ts +70 -0
- package/src/domains/config/keybindings.ts +8 -0
- package/src/domains/context/extension.ts +0 -3
- package/src/domains/context/wiki/map-seed.ts +589 -0
- package/src/domains/context/wiki/plan.ts +2 -2
- package/src/domains/context/working-set/path-index.ts +1 -0
- package/src/domains/dispatch/admission.ts +29 -0
- package/src/domains/dispatch/agent-candidates.ts +10 -0
- package/src/domains/dispatch/budget-envelope.ts +86 -1
- package/src/domains/dispatch/capability-match.ts +11 -0
- package/src/domains/dispatch/capacity-lease.ts +17 -0
- package/src/domains/dispatch/contract.ts +11 -1
- package/src/domains/dispatch/extension.ts +237 -49
- package/src/domains/dispatch/host-verification.ts +435 -39
- package/src/domains/dispatch/intent-requirements.ts +10 -0
- package/src/domains/dispatch/intent.ts +18 -1
- package/src/domains/dispatch/path-scope.ts +235 -24
- package/src/domains/dispatch/run-event-journal.ts +4 -15
- package/src/domains/dispatch/state.ts +2 -3
- package/src/domains/dispatch/transport.ts +45 -21
- package/src/domains/dispatch/types.ts +58 -3
- package/src/domains/dispatch/worker-model-metadata.ts +38 -0
- package/src/domains/eval/artifacts/store.ts +5 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
- package/src/domains/eval/metrics/token-stream.ts +201 -31
- package/src/domains/eval/metrics/tracked.ts +40 -4
- package/src/domains/eval/runners/clio-run.ts +5 -2
- package/src/domains/eval/schema/suite.ts +28 -0
- package/src/domains/eval/schema/verdict.ts +2 -2
- package/src/domains/eval/store.ts +8 -1
- package/src/domains/eval/suites/resolve.ts +13 -1
- package/src/domains/eval/suites/run.ts +24 -3
- package/src/domains/evidence/trust-status.ts +10 -1
- package/src/domains/extensions/contract.ts +15 -1
- package/src/domains/extensions/discovery.ts +238 -41
- package/src/domains/extensions/extension.ts +105 -6
- package/src/domains/extensions/index.ts +24 -0
- package/src/domains/extensions/integrity.ts +189 -0
- package/src/domains/extensions/manager.ts +17 -1
- package/src/domains/extensions/resource-path.ts +27 -0
- package/src/domains/extensions/resources.ts +18 -38
- package/src/domains/extensions/snapshot-store.ts +39 -0
- package/src/domains/extensions/snapshot.ts +180 -0
- package/src/domains/extensions/state.ts +385 -57
- package/src/domains/extensions/types.ts +118 -1
- package/src/domains/interop/registry.ts +6 -2
- package/src/domains/interop/types.ts +4 -0
- package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
- package/src/domains/lifecycle/migrations/index.ts +6 -0
- package/src/domains/lifecycle/naming-resources.ts +19 -4
- package/src/domains/lifecycle/naming-yazi.ts +10 -5
- package/src/domains/memory/task-memory-policy.ts +70 -26
- package/src/domains/memory/task-memory-telemetry.ts +1 -0
- package/src/domains/middleware/contract.ts +26 -0
- package/src/domains/middleware/extension.ts +24 -24
- package/src/domains/middleware/hook-receipts.ts +27 -4
- package/src/domains/middleware/hooks-io.ts +65 -32
- package/src/domains/middleware/hooks.ts +64 -0
- package/src/domains/middleware/index.ts +28 -5
- package/src/domains/middleware/marketplace-offer.ts +3 -35
- package/src/domains/middleware/memory-intervention.ts +127 -32
- package/src/domains/middleware/memory-step-endpoint.ts +3 -2
- package/src/domains/middleware/registrations.ts +326 -0
- package/src/domains/middleware/runtime.ts +28 -0
- package/src/domains/middleware/skills-reminder.ts +31 -2
- package/src/domains/middleware/snapshot.ts +20 -7
- package/src/domains/mux/contract.ts +38 -0
- package/src/domains/mux/detect.ts +6 -13
- package/src/domains/mux/index.ts +1 -1
- package/src/domains/mux/operations.ts +44 -5
- package/src/domains/mux/yazi/assets/yazi.toml +2 -2
- package/src/domains/mux/yazi/session.ts +53 -4
- package/src/domains/mux/yazi/theme.ts +117 -17
- package/src/domains/observability/compaction-usage.ts +118 -0
- package/src/domains/observability/contract.ts +10 -11
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/extension.ts +17 -4
- package/src/domains/observability/out-of-turn-usage.ts +52 -21
- package/src/domains/observability/projection.ts +14 -90
- package/src/domains/observability/trace-store.ts +43 -7
- package/src/domains/prompts/compiler.ts +73 -53
- package/src/domains/prompts/contract.ts +15 -3
- package/src/domains/prompts/extension.ts +97 -9
- package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
- package/src/domains/prompts/fragments/identity/clio.md +6 -12
- package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
- package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
- package/src/domains/prompts/fragments/operating/contract.md +7 -15
- package/src/domains/prompts/fragments/operating/delegation.md +32 -34
- package/src/domains/prompts/fragments/operating/skills.md +10 -24
- package/src/domains/prompts/fragments/operating/worker.md +1 -8
- package/src/domains/providers/contract.ts +4 -1
- package/src/domains/providers/extension.ts +40 -9
- package/src/domains/providers/index.ts +1 -1
- package/src/domains/providers/model-capabilities.ts +9 -0
- package/src/domains/providers/model-discovery.ts +2 -0
- package/src/domains/providers/model-runtime-capabilities.ts +99 -25
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
- package/src/domains/providers/runtime-resolution.ts +31 -0
- package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
- package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
- package/src/domains/providers/support.ts +11 -5
- package/src/domains/providers/target-model-cache.ts +25 -2
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/cost-provenance.ts +19 -0
- package/src/domains/providers/types/local-model-quirks.ts +85 -37
- package/src/domains/providers/types/runtime-descriptor.ts +20 -1
- package/src/domains/providers/types/target-descriptor.ts +19 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/skills/install.ts +72 -7
- package/src/domains/resources/skills/loader.ts +23 -19
- package/src/domains/resources/skills/marketplace.ts +63 -11
- package/src/domains/safety/autonomy.ts +15 -0
- package/src/domains/safety/call-target.ts +1 -1
- package/src/domains/safety/index.ts +1 -0
- package/src/domains/safety/loop-detector.ts +7 -4
- package/src/domains/safety/path-policy.ts +1 -1
- package/src/domains/safety/policy-engine.ts +34 -11
- package/src/domains/safety/protected-artifacts.ts +191 -88
- package/src/domains/safety/run-effects.ts +2 -22
- package/src/domains/safety/skill-authority.ts +55 -0
- package/src/domains/session/compaction/compact.ts +72 -22
- package/src/domains/session/entries.ts +6 -0
- package/src/domains/session/task-board.ts +10 -9
- package/src/domains/session/usage.ts +3 -3
- package/src/domains/share/archive.ts +164 -7
- package/src/engine/acp/server.ts +62 -9
- package/src/engine/agent.ts +13 -3
- package/src/engine/ai.ts +26 -8
- package/src/engine/antigravity/subprocess-runtime.ts +386 -120
- package/src/engine/api-registry.ts +3 -0
- package/src/engine/apis/llamacpp-residency.ts +3 -4
- package/src/engine/apis/lmstudio.ts +3 -3
- package/src/engine/apis/ollama-native.ts +6 -6
- package/src/engine/apis/openai-completions.ts +145 -39
- package/src/engine/apis/output-budget.ts +8 -18
- package/src/engine/apis/residency.ts +8 -27
- package/src/engine/external-subprocess.ts +114 -6
- package/src/engine/gemma-channel-filter.ts +19 -0
- package/src/engine/loop-guard.ts +92 -12
- package/src/engine/worker-runtime.ts +40 -11
- package/src/engine/worker-tools.ts +3 -1
- package/src/entry/background-model-metadata.ts +18 -0
- package/src/entry/compaction-prompt.ts +57 -0
- package/src/entry/extension-hook-sources.ts +28 -0
- package/src/entry/extension-reload.ts +309 -0
- package/src/entry/orchestrator.ts +464 -251
- package/src/entry/task-memory-lifecycle.ts +35 -0
- package/src/interactive/application-controller.ts +2 -1
- package/src/interactive/bus-notices.ts +8 -1
- package/src/interactive/chat-loop-messages.ts +16 -17
- package/src/interactive/chat-loop.ts +75 -3
- package/src/interactive/chat-panel.ts +36 -13
- package/src/interactive/chat-renderer.ts +72 -7
- package/src/interactive/cost-overlay.ts +26 -2
- package/src/interactive/dispatch-board.ts +6 -11
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +39 -4
- package/src/interactive/interactive-input-runtime.ts +4 -0
- package/src/interactive/interactive-presentation.ts +2 -2
- package/src/interactive/interactive-slash-runtime.ts +4 -1
- package/src/interactive/overlays/extensions.ts +9 -1
- package/src/interactive/overlays/help-reference.ts +13 -0
- package/src/interactive/overlays/settings.ts +27 -16
- package/src/interactive/panes-runtime.ts +111 -35
- package/src/interactive/prompt-cache-identity.ts +88 -0
- package/src/interactive/renderers/worker-entry.ts +32 -0
- package/src/interactive/slash-commands.ts +153 -20
- package/src/interactive/stream-pacing-policy.ts +0 -23
- package/src/interactive/theme/labels.ts +19 -13
- package/src/interactive/turn-context.ts +39 -20
- package/src/interactive/turn-recovery.ts +8 -0
- package/src/interactive/turn-runtime.ts +27 -11
- package/src/interactive/turn-state.ts +7 -0
- package/src/interactive/worker-receipts.ts +1 -0
- package/src/interactive/worker-stream.ts +6 -1
- package/src/interactive/yazi-bridge.ts +60 -6
- package/src/tools/agent-tools.ts +30 -1
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +3 -3
- package/src/tools/bash.ts +1 -1
- package/src/tools/bootstrap.ts +4 -0
- package/src/tools/builtin-tool-catalog.ts +52 -22
- package/src/tools/codewiki/code-nav-surface.ts +6 -0
- package/src/tools/codewiki/code-nav.ts +99 -13
- package/src/tools/context/docs-engine.ts +20 -7
- package/src/tools/context/index.ts +59 -21
- package/src/tools/core-bootstrap.ts +28 -6
- package/src/tools/credential-present.ts +1 -2
- package/src/tools/dispatch-arguments.ts +6 -1
- package/src/tools/dispatch-event-text.ts +10 -0
- package/src/tools/dispatch-plan.ts +49 -4
- package/src/tools/dispatch-run-events.ts +1 -1
- package/src/tools/dispatch-runner.ts +12 -0
- package/src/tools/dispatch-schema.ts +338 -0
- package/src/tools/dispatch-types.ts +3 -0
- package/src/tools/dispatch.ts +9 -254
- package/src/tools/ledger.ts +3 -5
- package/src/tools/monitor-surface.ts +5 -13
- package/src/tools/observation.ts +4 -5
- package/src/tools/panes-surface.ts +4 -11
- package/src/tools/panes.ts +4 -2
- package/src/tools/policy.ts +15 -2
- package/src/tools/read.ts +5 -6
- package/src/tools/registry.ts +41 -12
- package/src/tools/result-shaping.ts +18 -14
- package/src/tools/steer-surface.ts +1 -1
- package/src/tools/tasks.ts +1 -1
- package/src/tools/truncate.ts +6 -5
- package/src/tools/verify/surface.ts +6 -12
- package/src/tools/web-fetch-surface.ts +1 -3
- package/src/tools/worker-evidence.ts +3 -1
- package/src/worker/spec-contract.ts +4 -0
- package/dist/builtins-UJLMOVOV.js +0 -17
- package/dist/chunk-5QIAJV2D.js +0 -48
- package/dist/chunk-JZWT5J3Y.js +0 -814
- package/dist/chunk-K7VKOLQQ.js +0 -15
- package/dist/chunk-PMZCIOCJ.js +0 -25
- package/dist/chunk-SUW5DORT.js +0 -819
- package/dist/chunk-UOV2BYIW.js +0 -107
- package/dist/chunk-WR6U3OVP.js +0 -45
- package/dist/chunk-Y45G3AXC.js +0 -1558
- package/dist/reset-EOLM7GVE.js +0 -230
- package/dist/uninstall-N34PCTGJ.js +0 -331
- package/dist/upgrade-H7TOM7YL.js +0 -323
- package/docs/artifact-versions.md +0 -67
- package/docs/development-pipeline.md +0 -121
- package/docs/documentation-coverage.md +0 -46
- package/docs/documentation-guide.md +0 -167
- package/docs/time-conventions.md +0 -101
|
@@ -9,8 +9,11 @@
|
|
|
9
9
|
# https://huggingface.co/Jackrong/Qwopus3.6-35B-A3B-Coder-MTP-GGUF
|
|
10
10
|
# https://huggingface.co/nvidia/Nemotron-Cascade-2-30B-A3B
|
|
11
11
|
# https://huggingface.co/unsloth/NVIDIA-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-GGUF
|
|
12
|
+
# https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16
|
|
12
13
|
# https://huggingface.co/mradermacher/Qwen3.5-35B-A3B-Claude-4.6-Opus-Reasoning-Distilled-i1-GGUF
|
|
13
14
|
# https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B-GGUF
|
|
15
|
+
# https://huggingface.co/meta-models/Muse-Glimmer-30B
|
|
16
|
+
# https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.6-27B
|
|
14
17
|
#
|
|
15
18
|
# This knowledge base is intentionally narrow. It describes only the local
|
|
16
19
|
# models curated for Clio's local coding workflows and leaves cloud GPT models
|
|
@@ -26,8 +29,7 @@
|
|
|
26
29
|
# not be established from git history or the release-test reports under
|
|
27
30
|
# docs/release-notes/ read `unknown` rather than a plausible guess. A
|
|
28
31
|
# `measuredUnder` block is provenance for a reader; nothing in the engine
|
|
29
|
-
# consumes it (`extractLocalModelQuirks` narrows only
|
|
30
|
-
# thinking).
|
|
32
|
+
# consumes it (`extractLocalModelQuirks` narrows only sampling and thinking).
|
|
31
33
|
#
|
|
32
34
|
# `llamaCpp.parallel` is the recommended `--parallel` value for starting the
|
|
33
35
|
# server, and families that carry one also carry a `parallelSlots` note. It is
|
|
@@ -73,13 +75,12 @@
|
|
|
73
75
|
thinking:
|
|
74
76
|
temperature: 1
|
|
75
77
|
topP: 1
|
|
76
|
-
maxTokens: 32768
|
|
77
78
|
instruct:
|
|
78
79
|
temperature: 1
|
|
79
80
|
topP: 1
|
|
80
|
-
maxTokens: 8192
|
|
81
81
|
runtimePreference:
|
|
82
|
-
llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high."
|
|
82
|
+
llamaCpp: "Use the OpenAI-compatible chat-completions surface with chat_template_kwargs.reasoning_effort set to low, medium, or high. The analysis channel arrives as reasoning_content."
|
|
83
|
+
lmstudio: "Reasoning effort travels as the top-level reasoning_effort field. The analysis channel arrives as the OpenAI-compatible `reasoning` field, on the streamed delta and on the final message, with reasoning_content absent (measured 2026-09-02 on dynamo, gpt-oss-20b and gpt-oss-120b); Clio reads both spellings into thinking content."
|
|
83
84
|
openaiCompat: "Preferred for Harmony reasoning-effort control when the local gateway exposes chat_template_kwargs."
|
|
84
85
|
thinking:
|
|
85
86
|
mechanism: effort-levels
|
|
@@ -91,7 +92,9 @@
|
|
|
91
92
|
GPT-OSS uses OpenAI Harmony formatting and supports low, medium, and
|
|
92
93
|
high reasoning effort. Clio coerces off/minimal to low, captures
|
|
93
94
|
non-final Harmony channels as ThinkingContent, and strips Harmony
|
|
94
|
-
control tokens from visible output.
|
|
95
|
+
control tokens from visible output. The analysis channel reaches Clio
|
|
96
|
+
as reasoning_content on llama.cpp and as reasoning on LM Studio; both
|
|
97
|
+
are captured the same way.
|
|
95
98
|
|
|
96
99
|
- family: agenticqwen-30b-a3b-i1
|
|
97
100
|
matchPatterns:
|
|
@@ -130,13 +133,11 @@
|
|
|
130
133
|
temperature: 0.5
|
|
131
134
|
topP: 0.9
|
|
132
135
|
topK: 20
|
|
133
|
-
maxTokens: 16384
|
|
134
136
|
reasoningBudget: 4096
|
|
135
137
|
instruct:
|
|
136
138
|
temperature: 0.5
|
|
137
139
|
topP: 0.9
|
|
138
140
|
topK: 20
|
|
139
|
-
maxTokens: 4096
|
|
140
141
|
gpuTiers:
|
|
141
142
|
"32gb": "Default 32 GB local llama.cpp orchestrator profile; benchmarked with roughly 1 GiB VRAM headroom on a 32 GiB class card."
|
|
142
143
|
runtimePreference:
|
|
@@ -170,6 +171,10 @@
|
|
|
170
171
|
- nemotron-3-nano-omni-30b-a3b-reasoning
|
|
171
172
|
- nemotron-3-nano-omni
|
|
172
173
|
- nemotron-nano-omni
|
|
174
|
+
# Both router spellings. The live mini router advertises the id with `moe`
|
|
175
|
+
# in it; issue #263 and the round-5 handoffs use the shorter one.
|
|
176
|
+
- nemotron3-30b-moe-omni
|
|
177
|
+
- nemotron3-30b-omni
|
|
173
178
|
capabilities:
|
|
174
179
|
chat: true
|
|
175
180
|
tools: true
|
|
@@ -185,18 +190,45 @@
|
|
|
185
190
|
contextWindow: 1048576
|
|
186
191
|
maxTokens: 131072
|
|
187
192
|
quirks:
|
|
193
|
+
measuredUnder:
|
|
194
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 4 request slots"
|
|
195
|
+
runtime: llamacpp
|
|
196
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee (the string the ornith-1.5 entry below records for this same host on this date)"
|
|
197
|
+
model: "Nemotron-3-Nano-Omni-30B-A3B-Reasoning UD-Q4_K_M with Nemotron-Omni-mmproj-F16 (arch nemotron_h_moe)"
|
|
198
|
+
llamaCpp: >-
|
|
199
|
+
--ctx-size 1032192 --parallel 4 --kv-unified --cache-type-k f16
|
|
200
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
201
|
+
true --n-gpu-layers 99 --jinja --reasoning on --temperature 0.6 --top-k
|
|
202
|
+
40 --top-p 0.95 --repeat-penalty 1.05
|
|
203
|
+
date: "2026-09-02"
|
|
204
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
205
|
+
note: |
|
|
206
|
+
The on-off classification is measured, and it replaces a budget-tokens
|
|
207
|
+
dial that never reached this runtime. With no thinking field the model
|
|
208
|
+
spent 54 completion tokens with 140 characters of reasoning_content;
|
|
209
|
+
with chat_template_kwargs.enable_thinking false it answered in 4 tokens
|
|
210
|
+
with none; with it true it spent 42 tokens and 119 characters.
|
|
211
|
+
reasoning_effort moved nothing: none, low, and high produced 43, 48, and
|
|
212
|
+
39 tokens and a reasoning trace every time. The budget-tokens mechanism
|
|
213
|
+
this entry used to declare is downgraded to informational for
|
|
214
|
+
openai-completions on llama.cpp with a qwen chat template
|
|
215
|
+
(acceptsBudgetTokensField in model-runtime-capabilities.ts), so no
|
|
216
|
+
budget ever reached the server and the model reasoned at every level.
|
|
217
|
+
The router agrees with the measurement, advertising reasoning:on_off
|
|
218
|
+
with default_reasoning:on.
|
|
219
|
+
|
|
220
|
+
The measuring server is not the recommended profile below: it ran
|
|
221
|
+
ctx-size 1032192 with f16 KV against the 819200 and q8_0 recommended
|
|
222
|
+
here, and loaded only the vision mmproj, so the audio capability stays
|
|
223
|
+
the card's claim rather than this run's.
|
|
188
224
|
sampling:
|
|
189
225
|
thinking:
|
|
190
226
|
temperature: 0.6
|
|
191
227
|
topP: 0.95
|
|
192
228
|
topK: 20
|
|
193
|
-
maxTokens: 20480
|
|
194
|
-
reasoningBudget: 16384
|
|
195
|
-
gracePeriod: 1024
|
|
196
229
|
instruct:
|
|
197
230
|
temperature: 0.2
|
|
198
231
|
topK: 1
|
|
199
|
-
maxTokens: 1024
|
|
200
232
|
gpuTiers:
|
|
201
233
|
"24gb": "Use a 4-bit quant with reduced load context for evaluation. Prefer the 32GB tier for sustained Clio writes."
|
|
202
234
|
"32gb": "Primary strong local model for multimodal main-agent work. Keep openai-compat available for tool-call fallback."
|
|
@@ -213,17 +245,131 @@
|
|
|
213
245
|
parallel: 4
|
|
214
246
|
parallelSlots: "Start the server with --parallel 4, and add --kv-unified unless each of the four slots is meant to get 204800 tokens rather than the whole 819200. Clio admits four concurrent requests on this endpoint once /props reports total_slots 4."
|
|
215
247
|
thinking:
|
|
216
|
-
mechanism:
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
248
|
+
mechanism: on-off
|
|
249
|
+
chatTemplateKwargs:
|
|
250
|
+
static:
|
|
251
|
+
# Card recipe for the reasoning mode (scratchpad/catalog/models/nvidia-Nemotron-3-Nano-Omni-30B-A3B-Reasoning-FP8.md:7): reasoning_budget=16384.
|
|
252
|
+
reasoning_budget: 16384
|
|
253
|
+
lmstudio: unsupported
|
|
221
254
|
guidance: |
|
|
222
|
-
Reasoning is template-driven
|
|
223
|
-
|
|
224
|
-
|
|
255
|
+
Reasoning is template-driven and the switch is
|
|
256
|
+
chat_template_kwargs.enable_thinking; intermediate levels coerce to on.
|
|
257
|
+
The checkpoint reasons when nothing is sent, so off has to be carried on
|
|
258
|
+
the wire rather than omitted. reasoning_effort is inert on llama.cpp;
|
|
259
|
+
LM Studio reads the same on-off switch from that field instead.
|
|
225
260
|
serving: "Use qwen3_coder tool parsing when served through OpenAI-compatible HTTP. LM Studio uses the same OpenAI-compatible tool surface."
|
|
226
261
|
|
|
262
|
+
# NVIDIA's 3.5 Lightning generation, not a requant of the Nano Omni family
|
|
263
|
+
# above: a different HF repo (nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B),
|
|
264
|
+
# text-only where Nano Omni carries vision and audio, and a Mamba-2 plus MoE
|
|
265
|
+
# plus attention hybrid rather than the omni checkpoint. They share the vendor,
|
|
266
|
+
# the nemotron_h_moe arch string, and the 30B-A3B parameter count, which is not
|
|
267
|
+
# the same as sharing weights.
|
|
268
|
+
- family: nemotron-3.5-lightning-30b-a3b
|
|
269
|
+
matchPatterns:
|
|
270
|
+
# No bare `nemotron-3.5-lightning`: the 30b spelling is the shortest one
|
|
271
|
+
# that cannot swallow a future 8B or 70B Lightning id, which would inherit
|
|
272
|
+
# this checkpoint's sampler, its 1M window and a dial measured elsewhere.
|
|
273
|
+
- nvidia-nemotron-3.5-lightning-30b-a3b
|
|
274
|
+
- nemotron-3.5-lightning-30b-a3b
|
|
275
|
+
- nemotron-3.5-lightning-30b
|
|
276
|
+
- nemo3.5-30b-moe
|
|
277
|
+
capabilities:
|
|
278
|
+
chat: true
|
|
279
|
+
tools: true
|
|
280
|
+
toolCallFormat: qwen
|
|
281
|
+
reasoning: true
|
|
282
|
+
thinkingFormat: qwen-chat-template
|
|
283
|
+
structuredOutputs: json-schema
|
|
284
|
+
vision: false
|
|
285
|
+
audio: false
|
|
286
|
+
embeddings: false
|
|
287
|
+
rerank: false
|
|
288
|
+
fim: false
|
|
289
|
+
contextWindow: 1048576
|
|
290
|
+
maxTokens: 65536
|
|
291
|
+
quirks:
|
|
292
|
+
measuredUnder:
|
|
293
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 1 request slot"
|
|
294
|
+
runtime: llamacpp
|
|
295
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee"
|
|
296
|
+
model: "NVIDIA-Nemotron-3.5-Lightning-30B-A3B Q4_K_M (arch nemotron_h_moe)"
|
|
297
|
+
llamaCpp: >-
|
|
298
|
+
--ctx-size 1048576 --parallel 1 --kv-unified --cache-type-k f16
|
|
299
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
300
|
+
true --fit off --n-gpu-layers 99 --jinja --reasoning on
|
|
301
|
+
--chat-template-kwargs {"enable_thinking": true} --temperature 1.0
|
|
302
|
+
--top-k 40 --top-p 0.95 --repeat-penalty 1.05
|
|
303
|
+
date: "2026-09-02"
|
|
304
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
305
|
+
note: |
|
|
306
|
+
The mechanism is measured on this runtime and the card agrees with it.
|
|
307
|
+
With no thinking field the model spent 97 completion tokens with 300
|
|
308
|
+
characters of reasoning_content; with
|
|
309
|
+
chat_template_kwargs.enable_thinking false it answered in 3 tokens with
|
|
310
|
+
none; with it true it spent 120 tokens and 353 characters.
|
|
311
|
+
reasoning_effort is inert: none, low, and high produced 127, 114, and
|
|
312
|
+
117 tokens with a reasoning trace every time. The NVIDIA card states the
|
|
313
|
+
same control, "Reasoning Mode: Configurable on/off via chat template
|
|
314
|
+
(enable_thinking=True/False)", and the router advertises reasoning:on_off
|
|
315
|
+
with default_reasoning:on.
|
|
316
|
+
|
|
317
|
+
The same divergence is already recorded in code for this checkpoint:
|
|
318
|
+
model-runtime-capabilities.ts notes a 2026-08-11 fleet measurement on
|
|
319
|
+
nvidia-nemotron-3.5-lightning-30b-a3b where LM Studio suppressed
|
|
320
|
+
reasoning for reasoning_effort "none" and ignored enable_thinking false
|
|
321
|
+
while llama.cpp did the reverse. That id is a matchPattern here.
|
|
322
|
+
|
|
323
|
+
Sampler provenance: temperature 1.0 and top_p 0.95 are the card's
|
|
324
|
+
"Recommended Sampling"; top_k 40 and repeat_penalty 1.05 are the router
|
|
325
|
+
preset's launch arguments and have no card value. The card publishes no
|
|
326
|
+
separate thinking and instruct profiles, so both carry the same numbers.
|
|
327
|
+
maxTokens follows nemotron-cascade-2, the other 1M-context Nemotron in
|
|
328
|
+
this catalog; the card caps output only by the context window.
|
|
329
|
+
sampling:
|
|
330
|
+
thinking:
|
|
331
|
+
temperature: 1.0
|
|
332
|
+
topP: 0.95
|
|
333
|
+
topK: 40
|
|
334
|
+
repeatPenalty: 1.05
|
|
335
|
+
instruct:
|
|
336
|
+
temperature: 1.0
|
|
337
|
+
topP: 0.95
|
|
338
|
+
topK: 40
|
|
339
|
+
repeatPenalty: 1.05
|
|
340
|
+
gpuTiers:
|
|
341
|
+
"32gb": "Text-only agent target on a 32 GB card. Q4_K_M weights plus f16 KV reach the full 1048576-token window at parallel 1 on the measured server."
|
|
342
|
+
runtimePreference:
|
|
343
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. The router profile pins chat_template_kwargs.enable_thinking true, so a request that wants silence must send false rather than omitting the field."
|
|
344
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen tool parsing."
|
|
345
|
+
lmstudio: "LM Studio reads the on-off switch from reasoning_effort, not from chat_template_kwargs; see the 2026-08-11 note above."
|
|
346
|
+
llamaCpp:
|
|
347
|
+
ctxSize: 1048576
|
|
348
|
+
cacheTypeK: f16
|
|
349
|
+
cacheTypeV: f16
|
|
350
|
+
flashAttn: true
|
|
351
|
+
nGpuLayers: 99
|
|
352
|
+
parallel: 1
|
|
353
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 1048576-token context, and add --kv-unified if --parallel is raised or the context is divided across slots. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
354
|
+
batchSize: 16384
|
|
355
|
+
ubatchSize: 2048
|
|
356
|
+
chatTemplateKwargs:
|
|
357
|
+
enable_thinking: true
|
|
358
|
+
thinking:
|
|
359
|
+
mechanism: on-off
|
|
360
|
+
chatTemplateKwargs:
|
|
361
|
+
static:
|
|
362
|
+
# Card: "For coding agents, add extra_body={"chat_template_kwargs": {"force_nonempty_content": True}}" (scratchpad/catalog/models/nvidia-NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16.md:21).
|
|
363
|
+
force_nonempty_content: true
|
|
364
|
+
lmstudio: unsupported
|
|
365
|
+
guidance: |
|
|
366
|
+
Nemotron 3.5 Lightning's chat template exposes thinking as on or off
|
|
367
|
+
through enable_thinking; intermediate levels coerce to on. It reasons by
|
|
368
|
+
default, so off has to be carried on the wire. On LM Studio the same
|
|
369
|
+
switch is read from reasoning_effort instead, which Clio sends for
|
|
370
|
+
on-off models on that runtime.
|
|
371
|
+
serving: "NVIDIA-Nemotron-3.5-Lightning-30B-A3B, MoE hybrid of Mamba-2, MoE and attention layers with 3B active parameters. The card recommends nemotron_v3 reasoning parsing and qwen3_coder tool parsing; the mini router serves the Q4_K_M gguf with XML tool calls through llama.cpp's qwen3_coder parser. Native context is 1M."
|
|
372
|
+
|
|
227
373
|
- family: qwen3.6-35b-a3b
|
|
228
374
|
matchPatterns:
|
|
229
375
|
- qwen3.6-35b-a3b
|
|
@@ -245,14 +391,14 @@
|
|
|
245
391
|
rerank: false
|
|
246
392
|
fim: false
|
|
247
393
|
contextWindow: 262144
|
|
248
|
-
|
|
394
|
+
# Vendor maximum for complex problems. Evidence: scratchpad/catalog/models/Qwen-Qwen3.6-35B-A3B.md.
|
|
395
|
+
maxTokens: 81920
|
|
249
396
|
quirks:
|
|
250
397
|
sampling:
|
|
251
398
|
thinking:
|
|
252
399
|
temperature: 0.6
|
|
253
400
|
topP: 0.95
|
|
254
401
|
topK: 20
|
|
255
|
-
maxTokens: 16384
|
|
256
402
|
reasoningBudget: 4096
|
|
257
403
|
instruct:
|
|
258
404
|
temperature: 0.7
|
|
@@ -318,22 +464,17 @@
|
|
|
318
464
|
family of behavior as qwopus3.6-35b-a3b-coder. The recommended
|
|
319
465
|
llama.cpp serving profile below is the model card's, not that
|
|
320
466
|
measurement's, and no argv from the measuring host survives.
|
|
321
|
-
kvCache:
|
|
322
|
-
kQuant: q8_0
|
|
323
|
-
vQuant: q8_0
|
|
324
467
|
sampling:
|
|
325
468
|
thinking:
|
|
326
469
|
temperature: 0.6
|
|
327
470
|
topP: 0.95
|
|
328
471
|
topK: 20
|
|
329
|
-
maxTokens: 16384
|
|
330
472
|
reasoningBudget: 4096
|
|
331
473
|
instruct:
|
|
332
474
|
temperature: 0.6
|
|
333
475
|
topP: 0.95
|
|
334
476
|
topK: 20
|
|
335
477
|
repeatPenalty: 1.05
|
|
336
|
-
maxTokens: 4096
|
|
337
478
|
gpuTiers:
|
|
338
479
|
"32gb": "32 GB local profile with Q4_K_M, ctx=262144, q8 KV, flash attention, and parallel=1."
|
|
339
480
|
runtimePreference:
|
|
@@ -358,6 +499,76 @@
|
|
|
358
499
|
contains the final answer.
|
|
359
500
|
serving: "Ornith-1.0-35B is an agentic coding model; the card recommends 262K context, qwen3 reasoning parsing, qwen3_coder/qwen3_xml tool parsing, and sampler temperature=0.6, top_p=0.95, top_k=20."
|
|
360
501
|
|
|
502
|
+
- family: ornith-1.5
|
|
503
|
+
matchPatterns:
|
|
504
|
+
- ornith-1.5-35b-a3b
|
|
505
|
+
- ornith1.5-35b-moe
|
|
506
|
+
- ornith-1.5-35b
|
|
507
|
+
- ornith1.5-35b
|
|
508
|
+
- ornith-1.5
|
|
509
|
+
- ornith1.5
|
|
510
|
+
capabilities:
|
|
511
|
+
chat: true
|
|
512
|
+
tools: true
|
|
513
|
+
toolCallFormat: qwen
|
|
514
|
+
reasoning: true
|
|
515
|
+
thinkingFormat: qwen-chat-template
|
|
516
|
+
structuredOutputs: json-schema
|
|
517
|
+
vision: false
|
|
518
|
+
audio: false
|
|
519
|
+
embeddings: false
|
|
520
|
+
rerank: false
|
|
521
|
+
fim: false
|
|
522
|
+
contextWindow: 262144
|
|
523
|
+
maxTokens: 65536
|
|
524
|
+
quirks:
|
|
525
|
+
measuredUnder:
|
|
526
|
+
hardware: "mini fleet node, AMD GPU on the Vulkan backend, 4 request slots"
|
|
527
|
+
runtime: llamacpp
|
|
528
|
+
build: b1-c841aee
|
|
529
|
+
model: "Ornith-1.5-35B-A3B Q4_K_M (arch qwen35moe)"
|
|
530
|
+
llamaCpp: >-
|
|
531
|
+
--ctx-size 786432 --parallel 4 --no-kv-unified --cache-type-k q8_0
|
|
532
|
+
--cache-type-v q8_0 --batch-size 16384 --ubatch-size 2048 --jinja
|
|
533
|
+
--reasoning on
|
|
534
|
+
date: "2026-09-02"
|
|
535
|
+
source: "prompt-infrastructure round 2 (v0.4.2 development): live probe of the on-off switch, and a debugger dispatch receipt"
|
|
536
|
+
note: |
|
|
537
|
+
The 1.5 release is a Qwen3.5-MoE derivative and its chat template
|
|
538
|
+
honors enable_thinking, which the 1.0 measurement above did not find.
|
|
539
|
+
Measured on this date: with chat_template_kwargs.enable_thinking false
|
|
540
|
+
the model answered a one-line arithmetic prompt in 4 completion tokens
|
|
541
|
+
with no reasoning_content; with it true the same prompt spent 79
|
|
542
|
+
tokens, 158 characters of them reasoning. Under the inherited
|
|
543
|
+
always-on classification a debugger worker dispatched with thinking
|
|
544
|
+
off spent 1,061 of its 1,518 output tokens on a reasoning trace, and
|
|
545
|
+
the router itself advertises the model as reasoning:on_off. The
|
|
546
|
+
generic `ornith` pattern still routes 1.0 checkpoints to the entry
|
|
547
|
+
above; these longer patterns win for 1.5 ids.
|
|
548
|
+
sampling:
|
|
549
|
+
thinking:
|
|
550
|
+
temperature: 0.6
|
|
551
|
+
topP: 0.95
|
|
552
|
+
topK: 20
|
|
553
|
+
reasoningBudget: 4096
|
|
554
|
+
instruct:
|
|
555
|
+
temperature: 0.6
|
|
556
|
+
topP: 0.95
|
|
557
|
+
topK: 20
|
|
558
|
+
repeatPenalty: 1.05
|
|
559
|
+
thinking:
|
|
560
|
+
mechanism: on-off
|
|
561
|
+
guidance: |
|
|
562
|
+
Ornith 1.5's Qwen3.5 chat template exposes thinking as on or off
|
|
563
|
+
through enable_thinking; intermediate levels coerce to on. LM Studio
|
|
564
|
+
2026-09-03 exposed a split vocabulary on this exact Q4_K_M: native
|
|
565
|
+
discovery advertised {off,on}, while its OpenAI-compatible request
|
|
566
|
+
validator rejected literal `on` and accepted only effort-level values.
|
|
567
|
+
`low` ran only after a warning and fallback to on, so Clio now omits an
|
|
568
|
+
active override for this default-on model and sends `reasoning_effort:
|
|
569
|
+
none` only when disabling it.
|
|
570
|
+
serving: "Ornith-1.5-35B-A3B follows the 1.0 card: 262K context, qwen3 reasoning parsing, qwen3_coder tool parsing, sampler temperature=0.6, top_p=0.95, top_k=20."
|
|
571
|
+
|
|
361
572
|
- family: qwen3.6-27b
|
|
362
573
|
matchPatterns:
|
|
363
574
|
- qwen3.6-27b
|
|
@@ -376,7 +587,8 @@
|
|
|
376
587
|
rerank: false
|
|
377
588
|
fim: false
|
|
378
589
|
contextWindow: 262144
|
|
379
|
-
|
|
590
|
+
# Vendor maximum for complex problems. Evidence: scratchpad/catalog/models/Qwen-Qwen3.6-27B.md.
|
|
591
|
+
maxTokens: 81920
|
|
380
592
|
quirks:
|
|
381
593
|
measuredUnder:
|
|
382
594
|
hardware: unknown
|
|
@@ -400,7 +612,6 @@
|
|
|
400
612
|
minP: 0.0
|
|
401
613
|
presencePenalty: 0.0
|
|
402
614
|
repetitionPenalty: 1.0
|
|
403
|
-
maxTokens: 32768
|
|
404
615
|
instruct:
|
|
405
616
|
temperature: 0.7
|
|
406
617
|
topP: 0.80
|
|
@@ -408,7 +619,6 @@
|
|
|
408
619
|
minP: 0.0
|
|
409
620
|
presencePenalty: 1.5
|
|
410
621
|
repetitionPenalty: 1.0
|
|
411
|
-
maxTokens: 32768
|
|
412
622
|
gpuTiers:
|
|
413
623
|
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 (Q4_K_XL weights ~19.5 GB plus ~32 GB KV)."
|
|
414
624
|
runtimePreference:
|
|
@@ -444,6 +654,7 @@
|
|
|
444
654
|
rerank: false
|
|
445
655
|
fim: false
|
|
446
656
|
contextWindow: 262144
|
|
657
|
+
# Vendor maximum for final responses. Evidence: scratchpad/catalog/models/Qwen-Qwen3.8-27B.md.
|
|
447
658
|
maxTokens: 131072
|
|
448
659
|
quirks:
|
|
449
660
|
measuredUnder:
|
|
@@ -468,9 +679,6 @@
|
|
|
468
679
|
control were established earlier, on 2026-08-18, from a llama.cpp host
|
|
469
680
|
and an LM Studio host whose build strings were not recorded; treat those
|
|
470
681
|
two findings as unknown-build.
|
|
471
|
-
kvCache:
|
|
472
|
-
kQuant: q8_0
|
|
473
|
-
vQuant: q8_0
|
|
474
682
|
sampling:
|
|
475
683
|
thinking:
|
|
476
684
|
temperature: 1.0
|
|
@@ -479,7 +687,6 @@
|
|
|
479
687
|
minP: 0.0
|
|
480
688
|
presencePenalty: 0.0
|
|
481
689
|
repetitionPenalty: 1.0
|
|
482
|
-
maxTokens: 32768
|
|
483
690
|
instruct:
|
|
484
691
|
temperature: 0.7
|
|
485
692
|
topP: 0.80
|
|
@@ -487,14 +694,13 @@
|
|
|
487
694
|
minP: 0.0
|
|
488
695
|
presencePenalty: 1.5
|
|
489
696
|
repetitionPenalty: 1.0
|
|
490
|
-
maxTokens: 32768
|
|
491
697
|
gpuTiers:
|
|
492
698
|
"24gb-rocm": "24 GB ROCm class: IQ4_NL uses about 17.3 GB and UD-Q4_K_XL uses about 19.5 GB. Serve ctx=131072 with q8_0/q8_0 KV, parallel=2, and draft-mtp speculative decoding."
|
|
493
699
|
"32gb-cuda": "32 GB CUDA class: Q6_K uses about 23.8 GB. Serve ctx=131072 with parallel=4 because the full 262144-token context is VRAM-bound."
|
|
494
700
|
"128gb-unified-vulkan": "128 GB unified-memory Vulkan class: IQ4_NL uses about 17.3 GB and fits ctx=262144 with parallel=4. Vulkan backends show much slower prefill than ROCm/CUDA on this family; prefer ROCm where available."
|
|
495
701
|
runtimePreference:
|
|
496
702
|
llamaCpp: "Primary verified path. OpenAI-compatible chat completions with --jinja, --reasoning on, --reasoning-effort medium as server defaults; reasoning_content and tool_calls both confirmed to parse correctly. A request can override to off with chat_template_kwargs.enable_thinking:false, or to a different active level with reasoning_effort in {low, medium, xhigh} (never minimal/high/max on the wire; the stock template raises a fatal jinja exception on those and Clio's landed capability resolver already clamps to this set)."
|
|
497
|
-
lmstudioOpenaiCompat: "Clio sends reasoning_effort over LM Studio's OpenAI-compatible HTTP port and omits chat_template_kwargs because LM Studio does not use that field for this model.
|
|
703
|
+
lmstudioOpenaiCompat: "Clio sends reasoning_effort over LM Studio's OpenAI-compatible HTTP port and omits chat_template_kwargs because LM Studio does not use that field for this model. The upstream template defaults to thinking on/xhigh; an LM Studio deployment may configure another default. reasoning_effort:'none' explicitly disables it. Through LiteLLM, this control also requires allowed_openai_params:['reasoning_effort']; Clio selects the LM Studio spelling only when every deployment of that alias explicitly declares model_info.runtime: lm-studio."
|
|
498
704
|
lmstudioNative: "Clio's LM Studio runtime uses the OpenAI-compatible HTTP chat adapter with native REST model discovery and residency management."
|
|
499
705
|
llamaCpp:
|
|
500
706
|
ctxSize: 131072
|
|
@@ -516,7 +722,6 @@
|
|
|
516
722
|
effortByLevel:
|
|
517
723
|
low: low
|
|
518
724
|
medium: medium
|
|
519
|
-
high: xhigh
|
|
520
725
|
xhigh: xhigh
|
|
521
726
|
guidance: |
|
|
522
727
|
Official Qwen3.8-27B chat template (chat_template.jinja) validates
|
|
@@ -524,19 +729,32 @@
|
|
|
524
729
|
fatal jinja exception (hard 500, not a graceful fallback) on anything
|
|
525
730
|
else, including minimal/high/max. The 27B card gives no /think or
|
|
526
731
|
/no_think soft switch for this model; unlike Qwen3, that idiom has no
|
|
527
|
-
effect here.
|
|
528
|
-
|
|
732
|
+
effect here. The template defaults to xhigh with thinking enabled.
|
|
733
|
+
The picker offers off/low/medium/xhigh; released high and max choices
|
|
734
|
+
normalize to xhigh rather than reaching the strict template unchanged.
|
|
735
|
+
Template-level off uses chat_template_kwargs.enable_thinking:false.
|
|
736
|
+
preserve_thinking should stay
|
|
529
737
|
on for an agentic harness (the template's own recommendation for
|
|
530
738
|
agent scenarios; improves KV cache reuse). llama.cpp's own
|
|
531
739
|
reasoning_effort:"none" also disables reasoning at the server layer
|
|
532
740
|
and is a separate mechanism from the template's enable_thinking flag;
|
|
533
|
-
|
|
534
|
-
port.
|
|
741
|
+
the none sentinel is consumed before strict template effort validation.
|
|
742
|
+
LM Studio's HTTP port also accepts none. A 2026-09-05 same-route
|
|
743
|
+
LiteLLM probe showed none and the template flag alone still reasoned
|
|
744
|
+
because the generic OpenAI route filtered the effort parameter. Adding
|
|
745
|
+
allowed_openai_params:["reasoning_effort"] produced zero reasoning
|
|
746
|
+
twice; an allowed low control still reasoned. Clio forwards only an
|
|
747
|
+
effort resolved from the model/runtime contract, and never infers the
|
|
748
|
+
upstream control runtime from a host name, port, or model alias.
|
|
535
749
|
|
|
536
750
|
Thinking off is verified for this family, and no floor is imposed on
|
|
537
|
-
it.
|
|
751
|
+
it. The official card and template were rechecked at revision
|
|
752
|
+
1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0:
|
|
753
|
+
https://huggingface.co/Qwen/Qwen3.8-27B/blob/1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0/chat_template.jinja
|
|
754
|
+
JetBrains report that Qwen3.8 without reasoning gets stuck in a
|
|
538
755
|
loop repeating the same tool call indefinitely under Junie. Clio's
|
|
539
|
-
shipped default is
|
|
756
|
+
shipped default is chat.thinkingLevel: off, an explicit Clio preference
|
|
757
|
+
rather than the vendor's template default. On llama.cpp it
|
|
540
758
|
sends chat_template_kwargs.enable_thinking:false and on LM Studio's
|
|
541
759
|
OpenAI-compatible port sends reasoning_effort:"none". Clio 0.3.8 ran
|
|
542
760
|
that configuration on both runtimes (llama.cpp build b226-2115b73d8 on
|
|
@@ -544,8 +762,12 @@
|
|
|
544
762
|
2026-08-29 and 2026-08-30: an interactive build-and-resume session, a
|
|
545
763
|
three-worker fleet step, and a ten-task eval suite totalling 116 model
|
|
546
764
|
calls, none of which repeated a tool call into a loop. The catalog
|
|
547
|
-
therefore imposes no thinking floor
|
|
548
|
-
|
|
765
|
+
therefore imposes no thinking floor for this family. A provisional 2026-09-03 LM Studio A/B on the heavily quantized
|
|
766
|
+
`qwen3.8-27b-gsq-rco` alias corroborated the switch: four thinking-off
|
|
767
|
+
scenarios reported zero reasoning tokens and no repeated-tool loop,
|
|
768
|
+
versus visible reasoning at the active level. The host had LM Link
|
|
769
|
+
enabled, so this records behavior only—not hardware throughput or a
|
|
770
|
+
controlled quality claim. If a future build or quantization regresses into the loop
|
|
549
771
|
JetBrains describe, the guards that catch it are Clio's own, not a
|
|
550
772
|
catalog setting: the tool-prose-loop detector
|
|
551
773
|
(src/interactive/tool-prose-loop.ts), which is armed on the
|
|
@@ -590,9 +812,6 @@
|
|
|
590
812
|
exact NVFP4 build it was seen on. The KV-budget arithmetic in the
|
|
591
813
|
gpuTiers and the serving note is derived from the official Google
|
|
592
814
|
Gemma 4 31B card.
|
|
593
|
-
kvCache:
|
|
594
|
-
kQuant: q8_0
|
|
595
|
-
vQuant: q8_0
|
|
596
815
|
sampling:
|
|
597
816
|
thinking:
|
|
598
817
|
temperature: 1.0
|
|
@@ -601,13 +820,11 @@
|
|
|
601
820
|
minP: 0.0
|
|
602
821
|
presencePenalty: 0.0
|
|
603
822
|
repetitionPenalty: 1.0
|
|
604
|
-
maxTokens: 32768
|
|
605
823
|
instruct:
|
|
606
824
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
607
825
|
topP: 0.95
|
|
608
826
|
topK: 64
|
|
609
827
|
minP: 0.0
|
|
610
|
-
maxTokens: 32768
|
|
611
828
|
chatTemplate: "gemma-channel"
|
|
612
829
|
leakageNote: |
|
|
613
830
|
LM Studio's gemma4 chat template emits chain-of-thought as plain text
|
|
@@ -643,6 +860,11 @@
|
|
|
643
860
|
- gemma-4-31b-it-qat-ud-q4_k_xl-mtp
|
|
644
861
|
- gemma-4-31b-it-qat-ud
|
|
645
862
|
- gemma-4-31b-it-qat
|
|
863
|
+
# The mini router's short id for this exact build; its alias is
|
|
864
|
+
# Gemma-4-31B-it-qat-UD-Q4_K_XL-MTP-262K, the first pattern above. A bare
|
|
865
|
+
# `gemma4-31b` would also swallow a future `gemma4-31b-nvfp4` id that
|
|
866
|
+
# belongs to the turbo family above, so the full router id is the pattern.
|
|
867
|
+
- gemma4-31b-dense
|
|
646
868
|
capabilities:
|
|
647
869
|
chat: true
|
|
648
870
|
tools: true
|
|
@@ -658,9 +880,35 @@
|
|
|
658
880
|
contextWindow: 262144
|
|
659
881
|
maxTokens: 32768
|
|
660
882
|
quirks:
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
883
|
+
measuredUnder:
|
|
884
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
885
|
+
runtime: llamacpp
|
|
886
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
887
|
+
model: "Gemma-4-31B-it-qat UD-Q4_K_XL with mtp-gemma-4-31B-it draft and mmproj-F16 (arch gemma4)"
|
|
888
|
+
llamaCpp: >-
|
|
889
|
+
--ctx-size 139264 --parallel 1 --kv-unified --cache-type-k f16
|
|
890
|
+
--cache-type-v f16 --batch-size 1024 --ubatch-size 256 --flash-attn true
|
|
891
|
+
--fit off --n-gpu-layers 99 --spec-type draft-mtp --spec-draft-n-max 4
|
|
892
|
+
--n-gpu-layers-draft 99 --jinja --reasoning off --temperature 0.8
|
|
893
|
+
--top-k 64 --top-p 0.95 --repeat-penalty 1.05
|
|
894
|
+
date: "2026-09-02"
|
|
895
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
896
|
+
note: |
|
|
897
|
+
The on-off mechanism this entry already claimed is now measured on
|
|
898
|
+
llama.cpp rather than inherited from the gemma-4 template. With no
|
|
899
|
+
thinking field, and with chat_template_kwargs.enable_thinking false, the
|
|
900
|
+
model answered in 3 completion tokens with no reasoning_content; with it
|
|
901
|
+
true the same prompt spent 68 tokens, 128 characters of them reasoning.
|
|
902
|
+
reasoning_effort is inert: none, low, and high all stayed at 3 tokens
|
|
903
|
+
with no trace, so this family is on-off and not effort-levels on this
|
|
904
|
+
runtime. The router serves it with --reasoning off, which is why the
|
|
905
|
+
no-field baseline is silent.
|
|
906
|
+
|
|
907
|
+
The measuring server is not the 32gb profile recommended below. It ran
|
|
908
|
+
ROCm at ctx-size 139264 with f16 KV, batch 1024 and ubatch 256, where the
|
|
909
|
+
block below recommends Vulkan at 262144 with q8_0 KV. The serving note
|
|
910
|
+
and gpuTiers keep the 262k recommendation; treat the argv above as the
|
|
911
|
+
configuration the token counts came from.
|
|
664
912
|
sampling:
|
|
665
913
|
thinking:
|
|
666
914
|
temperature: 1.0
|
|
@@ -669,14 +917,12 @@
|
|
|
669
917
|
minP: 0.0
|
|
670
918
|
presencePenalty: 0.0
|
|
671
919
|
repetitionPenalty: 1.0
|
|
672
|
-
maxTokens: 32768
|
|
673
920
|
instruct:
|
|
674
921
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; router profiles may serve this qat build at 0.8
|
|
675
922
|
topP: 0.95
|
|
676
923
|
topK: 64
|
|
677
924
|
minP: 0.0
|
|
678
925
|
repeatPenalty: 1.05
|
|
679
|
-
maxTokens: 32768
|
|
680
926
|
chatTemplate: "gemma-channel"
|
|
681
927
|
gpuTiers:
|
|
682
928
|
"32gb": "32 GB Vulkan llama.cpp profile with UD-Q4_K_XL weights, ctx=262144, q8 KV, flash attention, parallel=1, F16 gemma4v mmproj (vision), and MTP draft (spec-type draft-mtp, n_max 4) both active."
|
|
@@ -726,9 +972,6 @@
|
|
|
726
972
|
contextWindow: 122880
|
|
727
973
|
maxTokens: 32768
|
|
728
974
|
quirks:
|
|
729
|
-
kvCache:
|
|
730
|
-
kQuant: q8_0
|
|
731
|
-
vQuant: q8_0
|
|
732
975
|
sampling:
|
|
733
976
|
thinking:
|
|
734
977
|
temperature: 1.0
|
|
@@ -737,13 +980,11 @@
|
|
|
737
980
|
minP: 0.0
|
|
738
981
|
presencePenalty: 0.0
|
|
739
982
|
repetitionPenalty: 1.0
|
|
740
|
-
maxTokens: 32768
|
|
741
983
|
instruct:
|
|
742
984
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
743
985
|
topP: 0.95
|
|
744
986
|
topK: 64
|
|
745
987
|
minP: 0.0
|
|
746
|
-
maxTokens: 32768
|
|
747
988
|
chatTemplate: "gemma-channel"
|
|
748
989
|
measuredUnder:
|
|
749
990
|
hardware: unknown
|
|
@@ -779,10 +1020,26 @@
|
|
|
779
1020
|
serving: "Distilled from Claude 4.6 Opus reasoning traces onto Google Gemma 4 31B; sampler matches the gemopus card values."
|
|
780
1021
|
|
|
781
1022
|
- family: qwopus3.6-27b-v1-preview
|
|
1023
|
+
# A bare `qwopus3.6` pattern used to sit here and swallowed every Qwopus3.6
|
|
1024
|
+
# router id that was not spelled 27b. The mini router's `qwopus3.6-35b-moe`
|
|
1025
|
+
# landed on this 27B distill and inherited its budget-tokens dial, which emits
|
|
1026
|
+
# nothing on llama.cpp, so that model ran with a four-level dial that never
|
|
1027
|
+
# reached the wire (issue #263). The matcher in knowledge-base.ts ranks by
|
|
1028
|
+
# pattern length and ignores file order, so the only durable guard against the
|
|
1029
|
+
# next `qwopus3.6-*` id landing here is not to declare the short pattern.
|
|
1030
|
+
#
|
|
1031
|
+
# The cost is real and is chosen deliberately, here and at the other four
|
|
1032
|
+
# families that keep only their full spellings. An unlisted `qwopus3.6-*` id
|
|
1033
|
+
# now matches nothing, and a model with no family gets whatever
|
|
1034
|
+
# `inferThinkingMechanism` reads off the live probe: `none` with the dial
|
|
1035
|
+
# pinned to off and every thinking field stripped when the server reports no
|
|
1036
|
+
# reasoning, `on-off` when it reports some. Both are honest about knowing
|
|
1037
|
+
# nothing. Inheriting a sibling's measured dial is not, and it is the failure
|
|
1038
|
+
# mode that costs a session its reasoning silently. Adding the id is a
|
|
1039
|
+
# one-line catalog edit; noticing that a dial reaches nothing takes a sweep.
|
|
782
1040
|
matchPatterns:
|
|
783
1041
|
- qwopus3.6-27b-v1-preview
|
|
784
1042
|
- qwopus3.6-27b
|
|
785
|
-
- qwopus3.6
|
|
786
1043
|
capabilities:
|
|
787
1044
|
chat: true
|
|
788
1045
|
tools: true
|
|
@@ -818,7 +1075,6 @@
|
|
|
818
1075
|
minP: 0.0
|
|
819
1076
|
presencePenalty: 0.0
|
|
820
1077
|
repetitionPenalty: 1.0
|
|
821
|
-
maxTokens: 32768
|
|
822
1078
|
instruct:
|
|
823
1079
|
temperature: 0.7
|
|
824
1080
|
topP: 0.80
|
|
@@ -826,7 +1082,6 @@
|
|
|
826
1082
|
minP: 0.0
|
|
827
1083
|
presencePenalty: 1.5
|
|
828
1084
|
repetitionPenalty: 1.0
|
|
829
|
-
maxTokens: 32768
|
|
830
1085
|
gpuTiers:
|
|
831
1086
|
"32gb": "Fits at f16 KV with parallel=1 and ctx=262144 thanks to the Q4_K_M weights (~16.5 GB)."
|
|
832
1087
|
runtimePreference:
|
|
@@ -850,21 +1105,31 @@
|
|
|
850
1105
|
- qwopus3.6-35b-a3b-coder-mtp
|
|
851
1106
|
- qwopus3.6-35b-a3b-coder
|
|
852
1107
|
- qwopus3.6-35b-a3b
|
|
1108
|
+
# The mini router's short id for this same gguf. At 17 characters it outbids
|
|
1109
|
+
# the 13-character `qwopus3.6-27b` of the preview family, whichever order
|
|
1110
|
+
# the two entries load in.
|
|
1111
|
+
- qwopus3.6-35b-moe
|
|
853
1112
|
capabilities:
|
|
854
1113
|
chat: true
|
|
855
1114
|
tools: true
|
|
856
1115
|
toolCallFormat: qwen
|
|
857
|
-
# Reasoning class:
|
|
858
|
-
#
|
|
859
|
-
#
|
|
860
|
-
#
|
|
861
|
-
#
|
|
862
|
-
#
|
|
1116
|
+
# Reasoning class: on-off, and which field carries the switch depends on the
|
|
1117
|
+
# runtime. The card frames the Coder-MTP line as thinking-off execution, but
|
|
1118
|
+
# the wire disagrees on both hosts measured: the model reasons unless it is
|
|
1119
|
+
# told not to, or stays silent unless it is told to, and never ignores the
|
|
1120
|
+
# control that runtime honors.
|
|
1121
|
+
#
|
|
1122
|
+
# llama.cpp, 2026-09-02: chat_template_kwargs.enable_thinking is the switch.
|
|
1123
|
+
# true took a one-line arithmetic prompt from 3 completion tokens to 130
|
|
1124
|
+
# with 338 characters of reasoning_content, while reasoning_effort none, low
|
|
1125
|
+
# and high all left it at 3 tokens with no trace.
|
|
863
1126
|
#
|
|
864
|
-
#
|
|
865
|
-
# prompts to rule out cache hits)
|
|
866
|
-
# that
|
|
867
|
-
#
|
|
1127
|
+
# LM Studio, 2026-08-08: exactly the reverse. enable_thinking was inert
|
|
1128
|
+
# (verified with unique prompts to rule out cache hits) and reasoning_effort
|
|
1129
|
+
# "none" was the control that worked, taking a prompt from 98 reasoning
|
|
1130
|
+
# tokens to 0. Clio's on-off mechanism sends the template flag on every
|
|
1131
|
+
# runtime and adds LM Studio's reasoning_effort spelling on that one, so
|
|
1132
|
+
# both surfaces get a dial that reaches the model.
|
|
868
1133
|
#
|
|
869
1134
|
# reasoning=false previously resolved to mechanism "none", and mechanism
|
|
870
1135
|
# "none" makes openai-completions strip reasoning_effort from the payload.
|
|
@@ -882,25 +1147,75 @@
|
|
|
882
1147
|
maxTokens: 32768
|
|
883
1148
|
quirks:
|
|
884
1149
|
measuredUnder:
|
|
885
|
-
hardware: unknown
|
|
886
|
-
runtime: "lmstudio and its OpenAI-compatible port for the reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
|
|
887
|
-
build: "
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
1150
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 4 request slots, for the 2026-09-02 run; unknown for both earlier dates"
|
|
1151
|
+
runtime: "llamacpp for the 2026-09-02 on-off measurement; lmstudio and its OpenAI-compatible port for the 2026-08-08 reasoning findings; an unrecorded llama.cpp host for the presence-penalty finding"
|
|
1152
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin behind a router reporting build_info b1-c841aee, for the 2026-09-02 run. Unknown for both earlier dates; the Windows LM Studio measurement host's bundled build string was first recorded on 2026-08-29 as llama.cpp-win-x86_64-nvidia-cuda12-avx2, which is three weeks after the reasoning measurement and cannot be backdated to it"
|
|
1153
|
+
model: "Qwopus3.6-35B-A3B-Coder-MTP Q4_K_M (arch qwen35moe)"
|
|
1154
|
+
llamaCpp: >-
|
|
1155
|
+
2026-09-02 only: --ctx-size 786432 --parallel 4 --no-kv-unified
|
|
1156
|
+
--cache-type-k q8_0 --cache-type-v q8_0 --batch-size 16384 --ubatch-size
|
|
1157
|
+
2048 --flash-attn true --fit off --n-gpu-layers 99 --jinja --reasoning
|
|
1158
|
+
off --chat-template-kwargs {"enable_thinking": false} --temperature 0.6
|
|
1159
|
+
--top-k 20 --top-p 0.95 --min-p 0.0 --presence-penalty 0.0
|
|
1160
|
+
--repeat-last-n 256 --repeat-penalty 1.0. Unknown for both earlier
|
|
1161
|
+
dates; the reasoning measurement went through LM Studio's HTTP port,
|
|
1162
|
+
which does not expose the underlying server argv, and no argv survives
|
|
1163
|
+
for the dispatch that produced the presence-penalty value
|
|
1164
|
+
date: "2026-09-02 (llama.cpp on-off switch), 2026-08-08 (reasoning_effort and enable_thinking on LM Studio), 2026-07-06 (presence penalty)"
|
|
1165
|
+
source: "issue #263 thinking sweep for the llama.cpp date; commits b3db3b82 and f7d76bf4 for the two earlier ones"
|
|
891
1166
|
note: |
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
1167
|
+
Measured 2026-09-02 on llama.cpp: with no thinking field, and with
|
|
1168
|
+
chat_template_kwargs.enable_thinking false, the model answered a
|
|
1169
|
+
one-line arithmetic prompt in 3 completion tokens with no
|
|
1170
|
+
reasoning_content; with enable_thinking true the same prompt spent 130
|
|
1171
|
+
tokens, 338 characters of them reasoning. reasoning_effort none, low and
|
|
1172
|
+
high all left it at 3 tokens with no trace. The silent baseline is the
|
|
1173
|
+
router's doing rather than the checkpoint's: the preset launches this
|
|
1174
|
+
model with --reasoning off and chat_template_kwargs enable_thinking
|
|
1175
|
+
false, which is the opposite default from the LM Studio host below.
|
|
1176
|
+
|
|
1177
|
+
Three numbers from the earlier dates are also measurements. On
|
|
1178
|
+
2026-08-08 on that LM Studio host this model spent 98 of 103 completion
|
|
1179
|
+
tokens reasoning on "what is 17+25" with no thinking field set;
|
|
1180
|
+
chat_template_kwargs enable_thinking was inert, re-checked with unique
|
|
1181
|
+
prompts to rule out prompt-cache hits; and reasoning_effort "none" took
|
|
1182
|
+
the same prompt to 0. The same session recorded a wiki planning dispatch
|
|
1183
|
+
over a 1007-file repository going from 89,501 reasoning tokens, 47 tool
|
|
1184
|
+
calls and exit 1 at 459 s to 0 reasoning tokens, 8 tool calls and exit 0
|
|
1185
|
+
at 218 s. On 2026-07-06 a live dispatch measured presence_penalty 1.5:
|
|
1186
|
+
without it a coder worker repeated one code_nav call into the loop-guard
|
|
1187
|
+
abort on 3 of 3 runs, and with it the same task passed 3 of 3. Neither
|
|
1188
|
+
commit records the hardware or the server configuration, so both stay
|
|
1189
|
+
unknown.
|
|
1190
|
+
|
|
1191
|
+
The 2026-09-02 server also differs from the profile recommended below.
|
|
1192
|
+
It ran ctx-size 786432 with --parallel 4 and --no-kv-unified, giving each
|
|
1193
|
+
slot 196608 tokens, and served text-only with no mmproj and no MTP draft
|
|
1194
|
+
where the block below recommends 262144 at parallel 1 with both active.
|
|
1195
|
+
The vision capability stays the checkpoint's; that run simply did not
|
|
1196
|
+
load the projector. That has a consequence for any projector-less
|
|
1197
|
+
deployment, because the model Clio builds takes vision from this entry
|
|
1198
|
+
and from a target-level override only: synthLocalModel merges the
|
|
1199
|
+
runtime defaults, this entry and target.capabilities and passes no probe
|
|
1200
|
+
layer, then sets input to ["text","image"] whenever vision is true. Once
|
|
1201
|
+
this id resolves here rather than to the 27B preview it did before issue
|
|
1202
|
+
#263, mini's text-only preset would be offered image input it rejects,
|
|
1203
|
+
so a target serving this checkpoint without --mmproj needs
|
|
1204
|
+
capabilities.vision false on the target entry.
|
|
1205
|
+
|
|
1206
|
+
Why on-off replaced the four-level effort map, and what it costs. The
|
|
1207
|
+
map gave LM Studio a low/medium/high granularity nothing had measured,
|
|
1208
|
+
and gave llama.cpp an active level that put reasoning_effort on the wire
|
|
1209
|
+
and nothing else, which this template does not read. on-off is what both
|
|
1210
|
+
measurements support. The cost is that resolveRequestCapability adds the
|
|
1211
|
+
reasoning_effort spelling for an on-off family only when runtimeId is
|
|
1212
|
+
literally `lmstudio` (REASONING_EFFORT_ONLY_RUNTIMES), while an
|
|
1213
|
+
effort-levels family emitted it on every runtime. An LM Studio server
|
|
1214
|
+
reached through a generic openai-compat, litellm or vllm target
|
|
1215
|
+
therefore now receives only the template flag, which that host ignores,
|
|
1216
|
+
and the model reasons at every level. Point such a target at the
|
|
1217
|
+
`lmstudio` runtime, which is what the 2026-08-08 finding above was taken
|
|
1218
|
+
on and what carries the spelling that reaches it.
|
|
904
1219
|
sampling:
|
|
905
1220
|
instruct:
|
|
906
1221
|
temperature: 0.2
|
|
@@ -913,7 +1228,6 @@
|
|
|
913
1228
|
# code_nav call into the loop-guard abort on 3 of 3 runs; with it the
|
|
914
1229
|
# same task passed 3 of 3 with clean edit-then-validate trajectories.
|
|
915
1230
|
presencePenalty: 1.5
|
|
916
|
-
maxTokens: 32768
|
|
917
1231
|
gpuTiers:
|
|
918
1232
|
"32gb": "Default 32 GB local profile: Q4_K_M, ctx=262144, q8 KV, flash attention, parallel=1, with F32 mmproj (vision) and draft-mtp speculative decoding both active. Leaves several GiB free at 262k on a 32 GiB class card."
|
|
919
1233
|
runtimePreference:
|
|
@@ -933,21 +1247,13 @@
|
|
|
933
1247
|
chatTemplateKwargs:
|
|
934
1248
|
enable_thinking: false
|
|
935
1249
|
thinking:
|
|
936
|
-
mechanism:
|
|
937
|
-
effortByLevel:
|
|
938
|
-
# "none" is the only value that silences this family. It is carried on
|
|
939
|
-
# the wire because the model reasons by default, so sending nothing is
|
|
940
|
-
# in effect a request to keep reasoning.
|
|
941
|
-
off: none
|
|
942
|
-
low: low
|
|
943
|
-
medium: medium
|
|
944
|
-
high: high
|
|
1250
|
+
mechanism: on-off
|
|
945
1251
|
guidance: |
|
|
946
|
-
|
|
947
|
-
"none",
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
1252
|
+
One switch, two spellings: chat_template_kwargs.enable_thinking on
|
|
1253
|
+
llama.cpp, reasoning_effort ("none" off, "low" on) on LM Studio, each
|
|
1254
|
+
inert on the other host. Intermediate levels coerce to on, and off has
|
|
1255
|
+
to be carried on the wire because the checkpoint reasons unless told
|
|
1256
|
+
not to, whatever its card says about thinking-off execution.
|
|
951
1257
|
serving: "Jackrong Qwopus3.6 35B-A3B Coder MTP, Q4_K_M. Thinking-off coder/agent tuned for tool-use loops. Served as a local router default with MTP speculative decoding (spec-type draft-mtp, n_max 2) and F32 mmproj vision both active at 262k."
|
|
952
1258
|
|
|
953
1259
|
- family: qwopus3.6-27b-coder
|
|
@@ -994,7 +1300,6 @@
|
|
|
994
1300
|
# Same Coder-MTP non-thinking design as the 35B; see the measured
|
|
995
1301
|
# presence-penalty note there.
|
|
996
1302
|
presencePenalty: 1.5
|
|
997
|
-
maxTokens: 32768
|
|
998
1303
|
runtimePreference:
|
|
999
1304
|
llamaCpp: "A local llama.cpp router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
1000
1305
|
lmstudio: "Dense 27B Coder-MTP; configure MTP in LM Studio's per-model load settings."
|
|
@@ -1044,7 +1349,6 @@
|
|
|
1044
1349
|
topP: 0.9
|
|
1045
1350
|
topK: 20
|
|
1046
1351
|
repeatPenalty: 1.05
|
|
1047
|
-
maxTokens: 32768
|
|
1048
1352
|
runtimePreference:
|
|
1049
1353
|
llamaCpp: "A local llama.cpp router serves it with --jinja and enable_thinking false; coder sampler temperature 0.2, top_p 0.9, top_k 20."
|
|
1050
1354
|
thinking:
|
|
@@ -1062,6 +1366,13 @@
|
|
|
1062
1366
|
- gemma4-26b-a4b-it
|
|
1063
1367
|
- gemma-4-26b-a4b-it-q4
|
|
1064
1368
|
- gemma-4-26b-a4b-it-q4_k_m
|
|
1369
|
+
# The mini router's short id for the same checkpoint; its alias is
|
|
1370
|
+
# Gemma-4-26B-A4B-it-Q4_K_M-262K and it loads
|
|
1371
|
+
# gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf. QAT changes how the weights are
|
|
1372
|
+
# represented, not the chat template, so the on-off mechanism below carries
|
|
1373
|
+
# over and there is no reason to split the family the way the 31B nvfp4 and
|
|
1374
|
+
# qat builds are split (those declare different context windows).
|
|
1375
|
+
- gemma4-26b-moe
|
|
1065
1376
|
capabilities:
|
|
1066
1377
|
chat: true
|
|
1067
1378
|
tools: true
|
|
@@ -1077,12 +1388,48 @@
|
|
|
1077
1388
|
contextWindow: 262144
|
|
1078
1389
|
maxTokens: 65536
|
|
1079
1390
|
quirks:
|
|
1391
|
+
measuredUnder:
|
|
1392
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, Vulkan backend, 1 request slot"
|
|
1393
|
+
runtime: llamacpp
|
|
1394
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-vulkan/bin, behind a router reporting build_info b1-c841aee"
|
|
1395
|
+
model: "Gemma-4-26B-A4B-it-qat UD-Q4_K_XL with mmproj-F16, no draft model (arch gemma4)"
|
|
1396
|
+
llamaCpp: >-
|
|
1397
|
+
--ctx-size 262144 --parallel 1 --kv-unified --cache-type-k f16
|
|
1398
|
+
--cache-type-v f16 --batch-size 16384 --ubatch-size 2048 --flash-attn
|
|
1399
|
+
true --n-gpu-layers 99 --jinja --reasoning off --temperature 0.8
|
|
1400
|
+
--top-k 64 --top-p 0.95 --repeat-penalty 1.05
|
|
1401
|
+
date: "2026-09-02"
|
|
1402
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1403
|
+
note: |
|
|
1404
|
+
The on-off mechanism this entry already claimed is now measured on
|
|
1405
|
+
llama.cpp rather than inherited from the gemma-4 template. With no
|
|
1406
|
+
thinking field, and with chat_template_kwargs.enable_thinking false, the
|
|
1407
|
+
model answered in 3 completion tokens with no reasoning_content; with it
|
|
1408
|
+
true the same prompt spent 99 tokens, 227 characters of them reasoning.
|
|
1409
|
+
reasoning_effort is inert: none, low, and high all stayed at 3 tokens
|
|
1410
|
+
with no trace, so this family is on-off and not effort-levels on this
|
|
1411
|
+
runtime. The router serves it with --reasoning off, which is why the
|
|
1412
|
+
no-field baseline is silent.
|
|
1413
|
+
|
|
1414
|
+
Three serving facts differ from the recommended profile below. The
|
|
1415
|
+
measuring server ran f16 KV where the block below recommends q8_0, and
|
|
1416
|
+
it ran at batch 16384 and ubatch 2048. It also ran without speculative
|
|
1417
|
+
decoding: the preset's --tags string advertises
|
|
1418
|
+
spec:mtp,spec_draft_n:4,draft:mtp-gemma-4-26B-A4B-it, but the argv
|
|
1419
|
+
carries no --model-draft, --spec-type or --spec-draft-n-max, so the
|
|
1420
|
+
token counts above come from a plain decode. The 31B qat sibling above
|
|
1421
|
+
does pass those flags, which is where the tags were copied from.
|
|
1422
|
+
|
|
1423
|
+
Separately, the round-5 cache sweep on this same date measured the
|
|
1424
|
+
Gemma 4 tokenizer rendering Clio's full-capability first prompt at 7,353
|
|
1425
|
+
tokens against roughly 8.4k for the Qwen tokenizer on the identical
|
|
1426
|
+
prompt, so this family buys about a thousand tokens of headroom per turn
|
|
1427
|
+
over the Qwen-tokenized families.
|
|
1080
1428
|
sampling:
|
|
1081
1429
|
thinking:
|
|
1082
1430
|
temperature: 0.3
|
|
1083
1431
|
topP: 0.9
|
|
1084
1432
|
topK: 20
|
|
1085
|
-
maxTokens: 8192
|
|
1086
1433
|
reasoningBudget: 4096
|
|
1087
1434
|
instruct:
|
|
1088
1435
|
temperature: 0.7 # 0.7 keeps tool-call JSON parseable; gemma card's 1.0 is for chat
|
|
@@ -1153,7 +1500,6 @@
|
|
|
1153
1500
|
temperature: 0.3
|
|
1154
1501
|
topP: 0.9
|
|
1155
1502
|
topK: 20
|
|
1156
|
-
maxTokens: 8192
|
|
1157
1503
|
gpuTiers:
|
|
1158
1504
|
"32gb": "Preferred local worker model when a text-only worker is enough."
|
|
1159
1505
|
runtimePreference:
|
|
@@ -1221,13 +1567,11 @@
|
|
|
1221
1567
|
temperature: 0.6
|
|
1222
1568
|
topP: 0.95
|
|
1223
1569
|
topK: 20
|
|
1224
|
-
maxTokens: 16384
|
|
1225
1570
|
reasoningBudget: 4096
|
|
1226
1571
|
instruct:
|
|
1227
1572
|
temperature: 0.2
|
|
1228
1573
|
topP: 0.9
|
|
1229
1574
|
topK: 20
|
|
1230
|
-
maxTokens: 4096
|
|
1231
1575
|
gpuTiers:
|
|
1232
1576
|
"32gb": "Distilled Opus-style reasoning at Qwen3.5 35B A3B scale. Use as a strong local main agent when GPU room is available."
|
|
1233
1577
|
runtimePreference:
|
|
@@ -1288,12 +1632,10 @@
|
|
|
1288
1632
|
temperature: 0.2
|
|
1289
1633
|
topP: 0.9
|
|
1290
1634
|
topK: 20
|
|
1291
|
-
maxTokens: 4096
|
|
1292
1635
|
thinking:
|
|
1293
1636
|
temperature: 0.6
|
|
1294
1637
|
topP: 0.95
|
|
1295
1638
|
topK: 20
|
|
1296
|
-
maxTokens: 8192
|
|
1297
1639
|
reasoningBudget: 2048
|
|
1298
1640
|
gpuTiers:
|
|
1299
1641
|
"16gb": "Use this as the local 16GB emulation target. It keeps the output cap below the 32GB-class MoE defaults."
|
|
@@ -1316,3 +1658,246 @@
|
|
|
1316
1658
|
32GB-class MoE families because this is the 16GB emulation target;
|
|
1317
1659
|
budget for a reasoning preamble ahead of every answer.
|
|
1318
1660
|
serving: "LM Studio reports trained_for_tool_use for the local qwopus3.5-9b-v3 target. Configure residency through LM Studio or explicit lmstudio.load settings."
|
|
1661
|
+
|
|
1662
|
+
# Meta's Muse Glimmer is a new vendor, architecture and template for this
|
|
1663
|
+
# catalog: no existing family shares its perception encoder, its DFlash drafter,
|
|
1664
|
+
# or its reasoning control. It is listed under always-on because its dial is not
|
|
1665
|
+
# representable, not because the model reasons unconditionally by design; the
|
|
1666
|
+
# measuredUnder note below records what was tried.
|
|
1667
|
+
- family: muse-glimmer-30b
|
|
1668
|
+
matchPatterns:
|
|
1669
|
+
# No bare `muse-glimmer`: it would swallow a future Muse Glimmer of another
|
|
1670
|
+
# size, which shares neither this 131072 window nor a measured dial.
|
|
1671
|
+
- unsloth/muse-glimmer-30b-gguf
|
|
1672
|
+
- muse-glimmer-30b
|
|
1673
|
+
# The mini router serves the same weights under both this short id and the
|
|
1674
|
+
# unsloth repo name above.
|
|
1675
|
+
- muse-30b-dense
|
|
1676
|
+
capabilities:
|
|
1677
|
+
chat: true
|
|
1678
|
+
tools: true
|
|
1679
|
+
toolCallFormat: openai
|
|
1680
|
+
reasoning: true
|
|
1681
|
+
thinkingFormat: qwen-chat-template
|
|
1682
|
+
structuredOutputs: json-schema
|
|
1683
|
+
vision: true
|
|
1684
|
+
audio: false
|
|
1685
|
+
embeddings: false
|
|
1686
|
+
rerank: false
|
|
1687
|
+
fim: false
|
|
1688
|
+
contextWindow: 131072
|
|
1689
|
+
maxTokens: 32768
|
|
1690
|
+
quirks:
|
|
1691
|
+
measuredUnder:
|
|
1692
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
1693
|
+
runtime: llamacpp
|
|
1694
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
1695
|
+
model: "Muse-Glimmer-30B UD-Q4_K_XL with dflash-kquant draft and mmproj-Muse-Glimmer-30B-Q8_0 (arch muse-glimmer)"
|
|
1696
|
+
llamaCpp: >-
|
|
1697
|
+
--ctx-size 131072 --parallel 1 --kv-unified --cache-type-k q8_0
|
|
1698
|
+
--cache-type-v q8_0 --batch-size 2048 --ubatch-size 512 --flash-attn
|
|
1699
|
+
true --fit off --n-gpu-layers 99 --spec-type draft-dflash
|
|
1700
|
+
--spec-draft-n-max 16 --n-gpu-layers-draft 99 --jinja
|
|
1701
|
+
--chat-template-kwargs {"reasoning_strength": "xhigh"} --temperature 1.0
|
|
1702
|
+
--top-k 64 --top-p 0.95
|
|
1703
|
+
date: "2026-09-02"
|
|
1704
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1705
|
+
note: |
|
|
1706
|
+
always-on is the measured classification, and it is measured against
|
|
1707
|
+
both controls Clio can send. With no thinking field the model spent 99
|
|
1708
|
+
completion tokens with 358 characters of reasoning_content;
|
|
1709
|
+
chat_template_kwargs.enable_thinking false still produced 91 tokens with
|
|
1710
|
+
332 characters, and true produced 90 with 312; reasoning_effort none,
|
|
1711
|
+
low, and high produced 81, 83, and 61 tokens, every one of them with a
|
|
1712
|
+
trace. Nothing Clio can put on the wire silences it.
|
|
1713
|
+
|
|
1714
|
+
The reason is the model's own design rather than a template bug. The
|
|
1715
|
+
card's dial is reasoning strength with levels low, medium, high, and
|
|
1716
|
+
xhigh, set through the system prompt as "Reasoning strength: <value>",
|
|
1717
|
+
and the router passes it as chat_template_kwargs.reasoning_strength
|
|
1718
|
+
pinned to xhigh. There is no off level on that scale, and no
|
|
1719
|
+
ThinkingMechanism in this engine emits that key:
|
|
1720
|
+
resolveRequestCapability populates chat_template_kwargs only with
|
|
1721
|
+
enable_thinking or, on the harmony path, reasoning_effort. Encoding this
|
|
1722
|
+
family as on-off would put a flag on the wire that this template does
|
|
1723
|
+
not read while the TUI showed a working off/on dial, which is the
|
|
1724
|
+
failure qwopus3.5-9b-v3 above was reclassified to stop. A real dial for
|
|
1725
|
+
this family needs a per-family kwarg key in the engine, not a catalog
|
|
1726
|
+
edit.
|
|
1727
|
+
|
|
1728
|
+
Sampler provenance: temperature 1.0, top_p 0.95 and top_k 64 are the
|
|
1729
|
+
card's Best Practices values and the router preset uses the same three.
|
|
1730
|
+
The preset sets no repeat penalty and the card recommends none, so this
|
|
1731
|
+
entry declares none. contextWindow is the card's 131072, which the
|
|
1732
|
+
router matches at ctx-size 131072. maxTokens has no card value and
|
|
1733
|
+
follows the other 131072-context families here.
|
|
1734
|
+
sampling:
|
|
1735
|
+
thinking:
|
|
1736
|
+
temperature: 1.0
|
|
1737
|
+
topP: 0.95
|
|
1738
|
+
topK: 64
|
|
1739
|
+
instruct:
|
|
1740
|
+
temperature: 1.0
|
|
1741
|
+
topP: 0.95
|
|
1742
|
+
topK: 64
|
|
1743
|
+
gpuTiers:
|
|
1744
|
+
"32gb": "Agent-role target on a 32 GB card: UD-Q4_K_XL weights, q8_0 KV at ctx=131072, parallel=1, with the Q8_0 perception encoder and the DFlash drafter both resident."
|
|
1745
|
+
runtimePreference:
|
|
1746
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja. Set the reasoning budget at launch with --chat-template-kwargs {\"reasoning_strength\": \"...\"}; there is no per-request field Clio can send for it."
|
|
1747
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface. Expect a reasoning trace on every response whatever the configured thinking level says."
|
|
1748
|
+
llamaCpp:
|
|
1749
|
+
ctxSize: 131072
|
|
1750
|
+
cacheTypeK: q8_0
|
|
1751
|
+
cacheTypeV: q8_0
|
|
1752
|
+
flashAttn: true
|
|
1753
|
+
nGpuLayers: 99
|
|
1754
|
+
parallel: 1
|
|
1755
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 131072-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
1756
|
+
batchSize: 2048
|
|
1757
|
+
ubatchSize: 512
|
|
1758
|
+
mmproj: true
|
|
1759
|
+
specType: draft-dflash
|
|
1760
|
+
specDraftNMax: 16
|
|
1761
|
+
thinking:
|
|
1762
|
+
mechanism: always-on
|
|
1763
|
+
chatTemplateKwargs:
|
|
1764
|
+
byLevel:
|
|
1765
|
+
# Card: reasoning strength low|medium|high|xhigh (scratchpad/catalog/models/unsloth-Muse-Glimmer-30B.md:35). The per-family kwarg key the note above asked for.
|
|
1766
|
+
key: reasoning_strength
|
|
1767
|
+
values:
|
|
1768
|
+
low: low
|
|
1769
|
+
medium: medium
|
|
1770
|
+
high: high
|
|
1771
|
+
xhigh: xhigh
|
|
1772
|
+
lmstudio: unsupported
|
|
1773
|
+
lmstudio: unsupported
|
|
1774
|
+
guidance: |
|
|
1775
|
+
Muse Glimmer reasons on every turn and its strength scale has no off
|
|
1776
|
+
level, so Clio shows the dial as forced and budgets for a reasoning
|
|
1777
|
+
preamble ahead of every answer. Change the amount of thinking at the
|
|
1778
|
+
server, with the reasoning_strength chat-template kwarg; a per-request
|
|
1779
|
+
thinking level cannot reach this model.
|
|
1780
|
+
serving: "Meta Superintelligence Lab Muse-Glimmer-30B, a dense 29.6B causal transformer with a ~1.8B ViT-G/14 perception encoder, interleaved local and global attention with a 2048 sliding window, and a DFlash block-diffusion drafter that proposes 16 tokens per forward pass. Card context length is 131072. Input is text plus image; the mini router additionally tags video, which the card does not claim, so the catalog follows the card."
|
|
1781
|
+
|
|
1782
|
+
# ThinkingCap is a brevity finetune of Qwen3.6-27B and not the base model, so it
|
|
1783
|
+
# keeps its own family rather than borrowing qwen3.6-27b's. No pattern here may
|
|
1784
|
+
# be a bare `qwen3.6` spelling: `lookup` is a substring test, so `qwen3.6` alone
|
|
1785
|
+
# would match every base-model id too and win or lose on length rather than on
|
|
1786
|
+
# which checkpoint the id names. The `thinkingcap-` prefixed spellings below all
|
|
1787
|
+
# contain `qwen3.6` harmlessly, because a longer pattern only matches ids that
|
|
1788
|
+
# contain the whole longer string.
|
|
1789
|
+
- family: thinkingcap-qwen3.6-27b
|
|
1790
|
+
matchPatterns:
|
|
1791
|
+
# No bare `thinkingcap`: it would swallow a ThinkingCap built on another
|
|
1792
|
+
# base, such as a Qwen3.8 32B, and hand it this q4's sampler, its 131072
|
|
1793
|
+
# window and a dial measured on a different checkpoint.
|
|
1794
|
+
- thinkingcap-qwen3.6-27b
|
|
1795
|
+
- thinkingcap-27b-dense-q4
|
|
1796
|
+
- thinkingcap-27b-dense
|
|
1797
|
+
- thinkingcap-27b
|
|
1798
|
+
capabilities:
|
|
1799
|
+
chat: true
|
|
1800
|
+
tools: true
|
|
1801
|
+
toolCallFormat: qwen
|
|
1802
|
+
reasoning: true
|
|
1803
|
+
thinkingFormat: qwen-chat-template
|
|
1804
|
+
structuredOutputs: json-schema
|
|
1805
|
+
vision: false
|
|
1806
|
+
audio: false
|
|
1807
|
+
embeddings: false
|
|
1808
|
+
rerank: false
|
|
1809
|
+
fim: false
|
|
1810
|
+
contextWindow: 131072
|
|
1811
|
+
maxTokens: 32768
|
|
1812
|
+
quirks:
|
|
1813
|
+
measuredUnder:
|
|
1814
|
+
hardware: "mini fleet node, AMD Radeon AI PRO R9700 32 GB, ROCm backend, 1 request slot"
|
|
1815
|
+
runtime: llamacpp
|
|
1816
|
+
build: "worker llama-server from /home/akougkas/src/llama.cpp-b10687/build-rocm/bin, behind a router reporting build_info b1-c841aee"
|
|
1817
|
+
model: "ThinkingCap-27B Q4_K_M with the MTP draft head baked into the gguf (arch qwen36)"
|
|
1818
|
+
llamaCpp: >-
|
|
1819
|
+
--ctx-size 131072 --parallel 1 --kv-unified --cache-type-k f16
|
|
1820
|
+
--cache-type-v f16 --batch-size 2048 --ubatch-size 512 --flash-attn true
|
|
1821
|
+
--fit off --n-gpu-layers 99 --spec-type draft-mtp --spec-draft-n-max 4
|
|
1822
|
+
--n-gpu-layers-draft 99 --jinja --reasoning on --chat-template-kwargs
|
|
1823
|
+
{"enable_thinking": true} --temperature 1.0 --top-k 20 --top-p 0.95
|
|
1824
|
+
--min-p 0.0 --repeat-penalty 1.0
|
|
1825
|
+
date: "2026-09-02"
|
|
1826
|
+
source: "issue #263 thinking sweep: one non-streaming POST /v1/chat/completions per variant on a one-line arithmetic prompt, max_tokens 4096"
|
|
1827
|
+
note: |
|
|
1828
|
+
The router tag is wrong for this model and the measurement wins. It is
|
|
1829
|
+
the only reasoning model on this router tagged reasoning:on rather than
|
|
1830
|
+
reasoning:on_off, and it carries no default_reasoning tag at all, which
|
|
1831
|
+
reads as always-on. It is not. On the q4 that this family is keyed to,
|
|
1832
|
+
the no-field baseline spent 125 completion tokens with 354 characters of
|
|
1833
|
+
reasoning_content, chat_template_kwargs.enable_thinking false answered
|
|
1834
|
+
in 3 tokens with none, and true spent 130 tokens with 388 characters.
|
|
1835
|
+
reasoning_effort is inert: none, low, and high produced 115, 140, and
|
|
1836
|
+
131 tokens with a trace every time. The default-on behavior comes from
|
|
1837
|
+
the preset rather than the checkpoint, because the
|
|
1838
|
+
[thinkingcap-27b-dense-q4] block launches the server with reasoning = on
|
|
1839
|
+
and chat-template-kwargs {"enable_thinking": true}.
|
|
1840
|
+
|
|
1841
|
+
Issue #263 names thinkingcap-27b-dense-q6, which no longer exists. That
|
|
1842
|
+
quant was measured on this same date (baseline 99 tokens with 246
|
|
1843
|
+
characters of reasoning, enable_thinking false 3 tokens with none, true
|
|
1844
|
+
98 tokens with 291 characters, the same on-off shape as the q4) and then
|
|
1845
|
+
deleted from mini: the weights are gone and the preset block is removed
|
|
1846
|
+
from the router, which now serves the q4 as the only ThinkingCap id.
|
|
1847
|
+
This family therefore names no q6 quant. The short `thinkingcap-27b`
|
|
1848
|
+
pattern still covers that spelling if the operator ever restores it.
|
|
1849
|
+
|
|
1850
|
+
One serving observation was taken on the q6 before it was removed and
|
|
1851
|
+
has not been repeated on the q4: in the round-5 cache sweep this model
|
|
1852
|
+
re-prefilled 661 tokens on a `--continue` turn where the other eight
|
|
1853
|
+
router models kept their prefix. Read it as a lead rather than a fact
|
|
1854
|
+
about the q4.
|
|
1855
|
+
|
|
1856
|
+
Sampler provenance: temperature 0.6, top_p 0.95 and top_k 20 are the
|
|
1857
|
+
quantizer's stated intended sampling for this finetune, which also warns
|
|
1858
|
+
that greedy decoding makes it loop without closing the think block. The
|
|
1859
|
+
router preset instead runs temperature 1.0 with the same top_p and
|
|
1860
|
+
top_k, plus min_p 0.0 and repeat_penalty 1.0. The catalog follows the
|
|
1861
|
+
card, as it does for every other family here, and records the preset
|
|
1862
|
+
divergence above.
|
|
1863
|
+
sampling:
|
|
1864
|
+
thinking:
|
|
1865
|
+
temperature: 0.6
|
|
1866
|
+
topP: 0.95
|
|
1867
|
+
topK: 20
|
|
1868
|
+
minP: 0.0
|
|
1869
|
+
repeatPenalty: 1.0
|
|
1870
|
+
instruct:
|
|
1871
|
+
temperature: 0.6
|
|
1872
|
+
topP: 0.95
|
|
1873
|
+
topK: 20
|
|
1874
|
+
minP: 0.0
|
|
1875
|
+
repeatPenalty: 1.0
|
|
1876
|
+
gpuTiers:
|
|
1877
|
+
"32gb": "Reasoning-role target on a 32 GB card: Q4_K_M weights (about 17 GB) with f16 KV at ctx=131072, parallel=1, and the baked-in MTP draft head active."
|
|
1878
|
+
runtimePreference:
|
|
1879
|
+
llamaCpp: "OpenAI-compatible chat completions with --jinja and --reasoning on. The MTP head ships inside the gguf, so --spec-type draft-mtp needs no second model. Never decode greedy: temperature 0 makes this finetune loop without closing its think block."
|
|
1880
|
+
openaiCompat: "Use when a gateway fronts the same llama.cpp chat-completions surface with qwen reasoning and tool parsing."
|
|
1881
|
+
llamaCpp:
|
|
1882
|
+
ctxSize: 131072
|
|
1883
|
+
cacheTypeK: f16
|
|
1884
|
+
cacheTypeV: f16
|
|
1885
|
+
flashAttn: true
|
|
1886
|
+
nGpuLayers: 99
|
|
1887
|
+
parallel: 1
|
|
1888
|
+
parallelSlots: "Start the server with --parallel 1 so one agent gets the whole 131072-token context. Clio discovers total_slots 1 and admits one request at a time on this endpoint, so a dispatch raised while the orchestrator streams is refused rather than queued."
|
|
1889
|
+
batchSize: 2048
|
|
1890
|
+
ubatchSize: 512
|
|
1891
|
+
specType: draft-mtp
|
|
1892
|
+
specDraftNMax: 4
|
|
1893
|
+
chatTemplateKwargs:
|
|
1894
|
+
enable_thinking: true
|
|
1895
|
+
thinking:
|
|
1896
|
+
mechanism: on-off
|
|
1897
|
+
guidance: |
|
|
1898
|
+
ThinkingCap keeps the Qwen3.6 chat template's enable_thinking switch;
|
|
1899
|
+
intermediate levels coerce to on. The router serves it with thinking on
|
|
1900
|
+
by default, so off has to be carried on the wire rather than omitted,
|
|
1901
|
+
and reasoning_effort is inert on llama.cpp. The finetune's point is a
|
|
1902
|
+
shorter chain rather than no chain, so an active level is cheap.
|
|
1903
|
+
serving: "BottleCapAI ThinkingCap-Qwen3.6-27B, a brevity finetune of Qwen3.6-27B that holds accuracy at roughly 40 to 46 percent fewer thinking tokens (the quantizer measured a 15-prompt reasoning set at mean 675 tokens against 1401 for base Qwen3.6-27B). Served on mini as the Q4_K_M gguf with the MTP draft head embedded in the file, so speculative decoding needs no second model."
|