@iowarp/clio-coder 0.3.6 → 0.3.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +59 -0
- package/README.md +22 -6
- package/dist/{acp-2BEHC4DL.js → acp-U67UHUK2.js} +14 -14
- package/dist/{agents-LNNFTM53.js → agents-YU6SGALZ.js} +42 -35
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-KXXFI2VS.js → auth-5ZPJOIVG.js} +32 -24
- package/dist/builtins-C6JMZVV6.js +17 -0
- package/dist/{chunk-E25LMLRW.js → chunk-26LEYJZH.js} +2 -2
- package/dist/{chunk-TSHXZTOQ.js → chunk-2HEJ2F35.js} +5 -5
- package/dist/{chunk-PBTHKCPN.js → chunk-2HFZQUHL.js} +7 -7
- package/dist/{chunk-FO5ZOVUY.js → chunk-3BINW3FP.js} +5 -5
- package/dist/{chunk-4OC57DA6.js → chunk-4DGYLA73.js} +53 -2
- package/dist/{chunk-CKXWIANG.js → chunk-4SPRNWDE.js} +18 -16
- package/dist/{chunk-E2ER4LJF.js → chunk-5C3AQNDW.js} +25 -1
- package/dist/{chunk-XF5N4U5A.js → chunk-5DHKRSMQ.js} +9 -8
- package/dist/{chunk-43AOLP7E.js → chunk-5FR74PWO.js} +2 -1
- package/dist/{chunk-EKY57CSP.js → chunk-5H3GB5BO.js} +68 -772
- package/dist/{chunk-6XXKFVSN.js → chunk-5Q2VVUKB.js} +4 -4
- package/dist/{chunk-DJVECN66.js → chunk-7RFXX52T.js} +295 -3705
- package/dist/{chunk-ZXF4XRKW.js → chunk-7RGZWPB6.js} +158 -8
- package/dist/{chunk-FYYLNIL5.js → chunk-A2NJGIB3.js} +2 -2
- package/dist/{chunk-XXQNGV4M.js → chunk-A3WNZD3P.js} +548 -72
- package/dist/{chunk-VEZEGCGW.js → chunk-B5CSFE7B.js} +23 -21
- package/dist/{chunk-2SFS6XQE.js → chunk-DGSYXYMX.js} +3 -2
- package/dist/chunk-DR52UMZW.js +21 -0
- package/dist/{chunk-ZWLZP4ZT.js → chunk-DYIM5TJT.js} +98 -12
- package/dist/{chunk-QKMUKYO7.js → chunk-E77JEWSD.js} +270 -125
- package/dist/{chunk-LYF7OHWH.js → chunk-EMYUUSFG.js} +12 -467
- package/dist/{chunk-4VP4KH3K.js → chunk-EQ63NRB7.js} +8 -8
- package/dist/chunk-FBVTI2TJ.js +518 -0
- package/dist/{chunk-R46L2BIR.js → chunk-FHJEP5SW.js} +19 -25
- package/dist/{chunk-WR67VIZY.js → chunk-GN57SG4G.js} +66 -8
- package/dist/{chunk-RD5U66HV.js → chunk-GPIEI3LY.js} +9 -9
- package/dist/{chunk-PCZJO5TI.js → chunk-GU2UIAFZ.js} +13 -178
- package/dist/chunk-GWS3VEIW.js +195 -0
- package/dist/chunk-H7IXIC72.js +103 -0
- package/dist/{chunk-3BPUFZDL.js → chunk-HLE42MG7.js} +3 -3
- package/dist/{chunk-24I7BN55.js → chunk-IFBNV6H6.js} +3 -3
- package/dist/{chunk-QNQHSOLF.js → chunk-IGWKHNIQ.js} +114 -55
- package/dist/chunk-IIZWH4XA.js +172 -0
- package/dist/{chunk-AD2SYQYC.js → chunk-IJ7RPIYJ.js} +124 -6
- package/dist/chunk-J3YUBZWY.js +382 -0
- package/dist/{chunk-5JGRAMKL.js → chunk-JOZYP4GM.js} +8 -6
- package/dist/{chunk-OH3TOQTB.js → chunk-K4XHGFR5.js} +751 -15
- package/dist/{chunk-EYPA3EGJ.js → chunk-KTYTFRMB.js} +178 -14
- package/dist/chunk-LU7P4LHA.js +33 -0
- package/dist/chunk-ME6CCNFO.js +108 -0
- package/dist/chunk-MXKJU4JB.js +1100 -0
- package/dist/{chunk-RY3LY4J5.js → chunk-N22QMJKY.js} +21 -18
- package/dist/chunk-NMPKI6XL.js +3006 -0
- package/dist/chunk-NUGM5KR6.js +165 -0
- package/dist/{chunk-22NAGB7X.js → chunk-P43ETTHK.js} +5 -94
- package/dist/{chunk-NILBFAPG.js → chunk-PMDBGQSJ.js} +2 -2
- package/dist/chunk-PT7HYKEM.js +165 -0
- package/dist/chunk-RVG5JXAL.js +41 -0
- package/dist/{chunk-GEYXPTRF.js → chunk-RWSI4YD7.js} +2 -2
- package/dist/{chunk-4BPJXDWC.js → chunk-TANS5ZJS.js} +35 -19
- package/dist/{verifiers-NCBTHHN2.js → chunk-TB5666IT.js} +67 -324
- package/dist/{chunk-QM3F2GKX.js → chunk-TLQJPP24.js} +7927 -8088
- package/dist/{chunk-G7MUEIGA.js → chunk-TT36MB5S.js} +2 -1
- package/dist/chunk-TTHACPOM.js +961 -0
- package/dist/{chunk-6US73PDB.js → chunk-TYPGUK6W.js} +7 -7
- package/dist/{chunk-MFFY33HR.js → chunk-U6MBIEMB.js} +554 -209
- package/dist/chunk-VAWNZU7Z.js +242 -0
- package/dist/{chunk-IR4CFBFN.js → chunk-VCBR6CU7.js} +12 -12
- package/dist/{chunk-WHJYKASB.js → chunk-VHN4MY6O.js} +2 -2
- package/dist/{chunk-XYDYPRZI.js → chunk-VWZOAB7K.js} +10 -10
- package/dist/{chunk-ZRGEBJ4T.js → chunk-WLFILSD5.js} +48 -48
- package/dist/{chunk-K7T3E2SR.js → chunk-WNIJTQQK.js} +12 -11
- package/dist/{chunk-CYQKWTG3.js → chunk-WSB3FPX7.js} +68 -20
- package/dist/{chunk-KHSFENX2.js → chunk-WWCZ5F23.js} +116 -14
- package/dist/{chunk-XE2VEJHX.js → chunk-WXY7KU3G.js} +2 -2
- package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
- package/dist/{chunk-KOHPCX4K.js → chunk-XWSF374K.js} +5 -5
- package/dist/{chunk-CJUB2JJ2.js → chunk-YS5VLNH5.js} +10 -10
- package/dist/{chunk-ZZMN5OM4.js → chunk-ZNLWCMVZ.js} +2 -2
- package/dist/{chunk-WHGPSPT5.js → chunk-ZVJ5BLO2.js} +2 -2
- package/dist/cli/index.js +33 -31
- package/dist/{clio-M2KGYUFZ.js → clio-QVTYJ57A.js} +10 -10
- package/dist/{code-nav-GQNL7XA6.js → code-nav-FGGFIE7L.js} +5 -5
- package/dist/codewiki/build-worker.js +4 -4
- package/dist/{components-5TTYYX6G.js → components-ZFA3SAER.js} +9 -9
- package/dist/{config-XUUYQIWO.js → config-LW5IJFQN.js} +65 -55
- package/dist/{configure-IHJ7YOMV.js → configure-7XIZCOU4.js} +29 -24
- package/dist/{context-75MIWW3U.js → context-L3WL3X7K.js} +54 -43
- package/dist/{context-ZQ7SIFJV.js → context-N52ZA626.js} +30 -14
- package/dist/{context-74JLXAWD.js → context-Y6Y7QPR6.js} +12 -12
- package/dist/{context-clear-GYKWNUML.js → context-clear-MBQRLSDQ.js} +54 -43
- package/dist/{context-index-SSR5ECNE.js → context-index-HVMFQHK3.js} +5 -5
- package/dist/{context-working-set-UX5KEP4J.js → context-working-set-GS6DSO7F.js} +18 -18
- package/dist/{dispatch-runner-GIJBHNFL.js → dispatch-runner-22ZCNOM3.js} +367 -79
- package/dist/{docs-6FZSCG5B.js → docs-7LQ23DLM.js} +9 -9
- package/dist/{doctor-SVJ5BZCW.js → doctor-M7YEDGAE.js} +27 -23
- package/dist/{eval-CG6LLBLD.js → eval-BEC2WHDA.js} +73 -20
- package/dist/{evidence-ZYFIEN42.js → evidence-REJUMSKM.js} +70 -59
- package/dist/{evolve-QGEXEMDW.js → evolve-PY5ZBA5K.js} +50 -39
- package/dist/{extensions-ADGNCJJD.js → extensions-HVKU65YU.js} +7 -7
- package/dist/{fleet-S5R4ZOQY.js → fleet-7WZEWRFA.js} +236 -377
- package/dist/fleet-commands-UVHWM76J.js +70 -0
- package/dist/fleet-graph-6ULH7PES.js +125 -0
- package/dist/fleet-new-RDVJLHHH.js +48 -0
- package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-J53T6CCE.js} +5 -5
- package/dist/fleet-validate-72PC4SLA.js +79 -0
- package/dist/{init-5DRU55YR.js → init-OG3TPGQG.js} +69 -57
- package/dist/library-CNTMPLRF.js +217 -0
- package/dist/{memory-7YKKR6UC.js → memory-6IS7F275.js} +52 -41
- package/dist/{models-ZPOLRU2C.js → models-ENRJDA5W.js} +37 -31
- package/dist/{monitor-US5F5YGZ.js → monitor-XLDVO7TN.js} +79 -42
- package/dist/{orchestrator-E2AL4T5N.js → orchestrator-6KSPYRHA.js} +4126 -762
- package/dist/{paths-E7KYAQWE.js → paths-DBXMZMDU.js} +6 -6
- package/dist/registry-LG64LTF4.js +11 -0
- package/dist/{reset-KZ652EK6.js → reset-RZ4ER727.js} +12 -12
- package/dist/{run-SRNBKDWD.js → run-Y2CNK5RU.js} +89 -76
- package/dist/{share-CGZE33UP.js → share-A55GYP6Z.js} +37 -13
- package/dist/{skills-S2X4DLY5.js → skills-ALC5J6AT.js} +29 -13
- package/dist/{skills-eval-W2GGIC4R.js → skills-eval-JPBEBYQU.js} +60 -48
- package/dist/support-MIETYA5E.js +38 -0
- package/dist/{targets-54SWINWB.js → targets-VGNXIR3S.js} +46 -36
- package/dist/{terminal-lease-SAIF2OGY.js → terminal-lease-WOBR64YA.js} +4 -4
- package/dist/{uninstall-BVLWXKBT.js → uninstall-ZJF5H5ZN.js} +9 -9
- package/dist/{upgrade-JKAR27XC.js → upgrade-FUSUAGHR.js} +31 -28
- package/dist/{usage-MSAWCLX4.js → usage-N4MKVHKD.js} +140 -68
- package/dist/verifiers-YAWOJ3H2.js +336 -0
- package/dist/{verify-X5HDROLA.js → verify-LTDHYBGY.js} +10 -9
- package/dist/{wiki-generate-GUSOQ6ZP.js → wiki-generate-6M7GHTBJ.js} +75 -62
- package/dist/worker/entry.js +67 -57
- package/docs/README.md +3 -2
- package/docs/acp.md +1 -1
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +11 -6
- package/docs/built-in-agents.md +26 -2
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +83 -3
- package/docs/configuration-and-targets.md +86 -4
- package/docs/context-engine.md +1 -1
- package/docs/development-pipeline.md +1 -1
- package/docs/dispatch-architecture-rationale.md +1 -1
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +3 -3
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +1 -1
- package/docs/evidence-and-memory.md +74 -10
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +4 -1
- package/docs/extensions-and-sharing.md +6 -2
- package/docs/fleet-demo-runbook.md +2 -2
- package/docs/fleet-dispatch.md +228 -16
- package/docs/git-commit-provenance.md +2 -2
- package/docs/glossary.md +22 -2
- package/docs/installation-and-lifecycle.md +2 -2
- package/docs/middleware-and-components.md +2 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +56 -9
- package/docs/proactive-memory.md +1 -1
- package/docs/prompt-envelope-and-tools.md +4 -2
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +83 -65
- package/docs/resource-library.md +59 -0
- package/docs/safety-model.md +2 -2
- package/docs/scientific-validation.md +3 -3
- package/docs/session-lifecycle.md +37 -1
- package/docs/skills-marketplace.md +16 -3
- package/docs/tool-usage.md +14 -7
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +1 -1
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +3 -3
- package/package.json +1 -2
- package/src/cli/argv.ts +5 -0
- package/src/cli/configure.ts +107 -23
- package/src/cli/doctor.ts +5 -1
- package/src/cli/evidence.ts +30 -25
- package/src/cli/fleet-commands.ts +37 -0
- package/src/cli/fleet-graph.ts +102 -0
- package/src/cli/fleet-new.ts +36 -0
- package/src/cli/fleet-preflight.ts +111 -0
- package/src/cli/fleet-validate.ts +30 -0
- package/src/cli/fleet.ts +173 -335
- package/src/cli/index.ts +3 -1
- package/src/cli/library.ts +190 -0
- package/src/cli/share.ts +13 -1
- package/src/cli/shared.ts +1 -0
- package/src/cli/targets.ts +4 -1
- package/src/cli/usage.ts +111 -19
- package/src/cli/validate-model.ts +60 -5
- package/src/core/bus-events.ts +29 -0
- package/src/core/commit-attribution.ts +4 -4
- package/src/core/config.ts +130 -0
- package/src/core/defaults.ts +81 -0
- package/src/core/path-boundary.ts +100 -0
- package/src/domains/agents/builtins/architect.md +1 -0
- package/src/domains/agents/builtins/oracle.md +33 -0
- package/src/domains/agents/catalog.ts +13 -1
- package/src/domains/agents/extension.ts +2 -11
- package/src/domains/agents/fleet-contract.ts +304 -24
- package/src/domains/agents/index.ts +14 -0
- package/src/domains/agents/recipe.ts +7 -1
- package/src/domains/agents/registry.ts +73 -5
- package/src/domains/agents/result-contract.ts +360 -15
- package/src/domains/agents/write-boundary.ts +15 -50
- package/src/domains/config/classify.ts +4 -0
- package/src/domains/context/project-rules.ts +51 -1
- package/src/domains/dispatch/active-route-planner.ts +14 -0
- package/src/domains/dispatch/assignment-reconcile.ts +22 -5
- package/src/domains/dispatch/assignment-store.ts +151 -14
- package/src/domains/dispatch/backoff.ts +2 -1
- package/src/domains/dispatch/capability-match.ts +1 -0
- package/src/domains/dispatch/checkout-writer-lease.ts +175 -0
- package/src/domains/dispatch/contract.ts +48 -0
- package/src/domains/dispatch/delegation-plan.ts +164 -0
- package/src/domains/dispatch/execution-plan.ts +76 -5
- package/src/domains/dispatch/execution-role.ts +12 -2
- package/src/domains/dispatch/execution-scheduler.ts +183 -67
- package/src/domains/dispatch/extension.ts +412 -74
- package/src/domains/dispatch/fleet-gate.ts +14 -0
- package/src/domains/dispatch/fleet-plan.ts +63 -3
- package/src/domains/dispatch/fleet-run.ts +791 -0
- package/src/domains/dispatch/gate-role-prompts.ts +47 -0
- package/src/domains/dispatch/host-verification.ts +178 -0
- package/src/domains/dispatch/index.ts +41 -1
- package/src/domains/dispatch/intent-requirements.ts +40 -0
- package/src/domains/dispatch/intent.ts +235 -0
- package/src/domains/dispatch/path-scope.ts +370 -0
- package/src/domains/dispatch/receipt-integrity.ts +9 -4
- package/src/domains/dispatch/state.ts +36 -3
- package/src/domains/dispatch/types.ts +64 -12
- package/src/domains/dispatch/validation.ts +69 -6
- package/src/domains/dispatch/write-boundary-enforcer.ts +45 -0
- package/src/domains/dispatch/write-boundary.ts +201 -22
- package/src/domains/eval/metrics/evidence.ts +79 -2
- package/src/domains/eval/runners/clio-run.ts +12 -2
- package/src/domains/evidence/build.ts +69 -11
- package/src/domains/evidence/index.ts +21 -0
- package/src/domains/evidence/provenance.ts +46 -11
- package/src/domains/evidence/trust-projection.ts +274 -0
- package/src/domains/evidence/trust-status.ts +155 -18
- package/src/domains/evidence/types.ts +4 -0
- package/src/domains/extensions/discovery.ts +88 -1
- package/src/domains/extensions/resources.ts +20 -8
- package/src/domains/extensions/state.ts +6 -2
- package/src/domains/extensions/types.ts +4 -1
- package/src/domains/lifecycle/doctor.ts +140 -1
- package/src/domains/middleware/index.ts +15 -0
- package/src/domains/middleware/watchdog.ts +281 -0
- package/src/domains/observability/contract.ts +3 -1
- package/src/domains/observability/cost.ts +12 -1
- package/src/domains/observability/extension.ts +2 -2
- package/src/domains/observability/index.ts +10 -0
- package/src/domains/observability/out-of-turn-usage.ts +223 -0
- package/src/domains/prompts/contract.ts +3 -5
- package/src/domains/providers/extension.ts +30 -2
- package/src/domains/resources/common-loader.ts +3 -0
- package/src/domains/resources/index.ts +20 -0
- package/src/domains/resources/library.ts +326 -0
- package/src/domains/resources/prompts/loader.ts +119 -14
- package/src/domains/resources/skills/marketplace.ts +37 -12
- package/src/domains/safety/policy-engine.ts +5 -5
- package/src/domains/safety/run-effects.ts +64 -1
- package/src/domains/safety/scope.ts +7 -12
- package/src/domains/session/handoff.ts +629 -0
- package/src/domains/share/archive.ts +67 -2
- package/src/engine/acp/server.ts +4 -1
- package/src/engine/prompt-templates.ts +18 -1
- package/src/engine/worker-runtime.ts +6 -3
- package/src/entry/orchestrator.ts +37 -0
- package/src/interactive/bus-notices.ts +26 -0
- package/src/interactive/chat-loop.ts +235 -1
- package/src/interactive/chat-renderer.ts +22 -0
- package/src/interactive/cost-overlay.ts +31 -3
- package/src/interactive/council-dispatch.ts +30 -0
- package/src/interactive/council-grid.ts +213 -0
- package/src/interactive/council.ts +99 -0
- package/src/interactive/dispatch-board.ts +311 -17
- package/src/interactive/fleet-run-preview.ts +307 -0
- package/src/interactive/footer/notifications.ts +219 -0
- package/src/interactive/handoff-round.ts +56 -0
- package/src/interactive/interactive-application.ts +43 -1
- package/src/interactive/interactive-event-projection.ts +23 -1
- package/src/interactive/interactive-slash-runtime.ts +52 -2
- package/src/interactive/interactive-subscriptions.ts +14 -2
- package/src/interactive/oracle.ts +179 -0
- package/src/interactive/overlay-ask-user-lifecycle.ts +6 -0
- package/src/interactive/overlay-general-openers.ts +190 -1
- package/src/interactive/overlay-key-routing.ts +17 -1
- package/src/interactive/overlay-lifecycle.ts +41 -1
- package/src/interactive/overlay-permission-lifecycle.ts +10 -0
- package/src/interactive/overlay-resource-openers.ts +11 -3
- package/src/interactive/overlay-session-lifecycle.ts +234 -2
- package/src/interactive/overlays/fleet-run-approval.ts +208 -0
- package/src/interactive/overlays/handoff-review.ts +185 -0
- package/src/interactive/overlays/library-install-confirm.ts +151 -0
- package/src/interactive/overlays/list-overlay.ts +168 -2
- package/src/interactive/overlays/settings.ts +221 -30
- package/src/interactive/overlays/side-question.ts +139 -0
- package/src/interactive/overlays/skills-hub.ts +401 -15
- package/src/interactive/side-question.ts +171 -0
- package/src/interactive/slash-commands.ts +439 -7
- package/src/interactive/slash-spec.ts +19 -6
- package/src/interactive/theme/tokens.ts +30 -0
- package/src/interactive/turn-middleware.ts +15 -1
- package/src/interactive/view/artifacts.ts +42 -9
- package/src/interactive/view/view-overlay.ts +15 -3
- package/src/interactive/watchdog-run.ts +75 -0
- package/src/interactive/worker-receipts.ts +14 -2
- package/src/interactive/worker-share.ts +56 -1
- package/src/interactive/worker-stream.ts +15 -0
- package/src/tools/bootstrap.ts +3 -0
- package/src/tools/compete-worktrees.ts +13 -79
- package/src/tools/dispatch-admission.ts +251 -18
- package/src/tools/dispatch-arguments.ts +84 -1
- package/src/tools/dispatch-plan.ts +165 -6
- package/src/tools/dispatch-runner.ts +364 -23
- package/src/tools/dispatch-types.ts +20 -1
- package/src/tools/dispatch.ts +72 -2
- package/src/tools/monitor.ts +29 -0
- package/src/tools/profiles.ts +18 -4
- package/src/tools/task-worktree.ts +238 -0
- package/src/tools/verify/authoring.ts +61 -1
- package/src/tools/verify/scripts.ts +62 -0
- package/src/tools/worker-evidence.ts +21 -14
- package/src/worker/spec-contract.ts +3 -1
- package/dist/chunk-HC4CLZ2Y.js +0 -68
|
@@ -3,39 +3,51 @@ import {
|
|
|
3
3
|
createSentinelStripper,
|
|
4
4
|
stripTokenizerSentinels
|
|
5
5
|
} from "./chunk-CFGTUFWB.js";
|
|
6
|
+
import {
|
|
7
|
+
credentialsPresent
|
|
8
|
+
} from "./chunk-LU7P4LHA.js";
|
|
9
|
+
import {
|
|
10
|
+
ceilChars,
|
|
11
|
+
estimateAgentMessageTokens,
|
|
12
|
+
toolSchemaChars
|
|
13
|
+
} from "./chunk-RWSI4YD7.js";
|
|
6
14
|
import {
|
|
7
15
|
readSettings
|
|
8
|
-
} from "./chunk-
|
|
16
|
+
} from "./chunk-IJ7RPIYJ.js";
|
|
17
|
+
import {
|
|
18
|
+
sleep
|
|
19
|
+
} from "./chunk-FQ4SKYE4.js";
|
|
9
20
|
import {
|
|
10
|
-
|
|
11
|
-
} from "./chunk-
|
|
21
|
+
getSharedBus
|
|
22
|
+
} from "./chunk-ZGVHUX3M.js";
|
|
23
|
+
import {
|
|
24
|
+
BusChannels
|
|
25
|
+
} from "./chunk-TT36MB5S.js";
|
|
26
|
+
import {
|
|
27
|
+
CLIO_CONTEXT_WINDOW_WARN_BELOW,
|
|
28
|
+
CLIO_MIN_CONTEXT_WINDOW,
|
|
29
|
+
CLIO_MIN_MAX_OUTPUT_TOKENS,
|
|
30
|
+
extractLocalModelQuirks,
|
|
31
|
+
listLmStudioModels,
|
|
32
|
+
lmStudioReasoningEffort,
|
|
33
|
+
lmStudioRootUrl,
|
|
34
|
+
registerBuiltinRuntimes,
|
|
35
|
+
requestLmStudioJson,
|
|
36
|
+
resolveLmStudioInstance
|
|
37
|
+
} from "./chunk-NMPKI6XL.js";
|
|
38
|
+
import {
|
|
39
|
+
listKnownModelsForRuntime
|
|
40
|
+
} from "./chunk-IIZWH4XA.js";
|
|
12
41
|
import {
|
|
13
42
|
authNotRequiredStatus,
|
|
14
43
|
openAuthStorage,
|
|
15
44
|
registerClioOAuthProviders,
|
|
16
45
|
resolveAuthTarget,
|
|
17
|
-
resolveRuntimeAuthTarget,
|
|
18
46
|
targetRequiresAuth
|
|
19
|
-
} from "./chunk-
|
|
20
|
-
import {
|
|
21
|
-
sleep
|
|
22
|
-
} from "./chunk-FQ4SKYE4.js";
|
|
47
|
+
} from "./chunk-EQ63NRB7.js";
|
|
23
48
|
import {
|
|
24
49
|
withStateFileLock
|
|
25
50
|
} from "./chunk-IKCO5N3L.js";
|
|
26
|
-
import {
|
|
27
|
-
ceilChars,
|
|
28
|
-
estimateAgentMessageTokens,
|
|
29
|
-
toolSchemaChars
|
|
30
|
-
} from "./chunk-GEYXPTRF.js";
|
|
31
|
-
import {
|
|
32
|
-
activateExternalPluginApiBridge,
|
|
33
|
-
calculateEngineCost,
|
|
34
|
-
createEngineAi,
|
|
35
|
-
ensurePiAiRegistered,
|
|
36
|
-
getEngineSupportedThinkingLevels,
|
|
37
|
-
registerEngineApiProvider
|
|
38
|
-
} from "./chunk-FCSXB6T2.js";
|
|
39
51
|
import {
|
|
40
52
|
require_dist
|
|
41
53
|
} from "./chunk-APJ265NV.js";
|
|
@@ -45,340 +57,35 @@ import {
|
|
|
45
57
|
resolveClioDirs
|
|
46
58
|
} from "./chunk-BNAZZHFG.js";
|
|
47
59
|
import {
|
|
48
|
-
|
|
49
|
-
} from "./chunk-
|
|
60
|
+
getRuntimeRegistry
|
|
61
|
+
} from "./chunk-NUGM5KR6.js";
|
|
50
62
|
import {
|
|
51
|
-
|
|
52
|
-
|
|
63
|
+
capabilitiesFromCatalogModel,
|
|
64
|
+
catalogThinkingLevelsForRuntime,
|
|
65
|
+
getCatalogModelForRuntime,
|
|
66
|
+
mergeCapabilities,
|
|
67
|
+
resolveCostProvenance
|
|
68
|
+
} from "./chunk-PT7HYKEM.js";
|
|
69
|
+
import {
|
|
70
|
+
activateExternalPluginApiBridge,
|
|
71
|
+
calculateEngineCost,
|
|
72
|
+
ensurePiAiRegistered,
|
|
73
|
+
registerEngineApiProvider
|
|
74
|
+
} from "./chunk-FCSXB6T2.js";
|
|
53
75
|
import {
|
|
54
76
|
__toESM,
|
|
55
77
|
init_esm_shims
|
|
56
78
|
} from "./chunk-3R73A4XB.js";
|
|
57
79
|
|
|
58
|
-
// src/domains/providers/capabilities.ts
|
|
59
|
-
init_esm_shims();
|
|
60
|
-
function mergeCapabilities(base, kb, probe, userOverride) {
|
|
61
|
-
const merged = { ...base };
|
|
62
|
-
applyLayer(merged, probe);
|
|
63
|
-
applyLayer(merged, kb);
|
|
64
|
-
applyDeploymentLimits(merged, probe);
|
|
65
|
-
applyLayer(merged, userOverride);
|
|
66
|
-
return merged;
|
|
67
|
-
}
|
|
68
|
-
function applyDeploymentLimits(target, probe) {
|
|
69
|
-
if (!probe) return;
|
|
70
|
-
if (probe.contextWindow !== void 0) target.contextWindow = probe.contextWindow;
|
|
71
|
-
if (probe.maxTokens !== void 0) target.maxTokens = probe.maxTokens;
|
|
72
|
-
}
|
|
73
|
-
function supportsAgentRoleTools(capabilities2) {
|
|
74
|
-
return capabilities2.tools === true;
|
|
75
|
-
}
|
|
76
|
-
var AGENT_ROLE_TOOLS_REQUIRED_REASON = "reports no tool support; Clio drives every agent role through typed tools";
|
|
77
|
-
function applyLayer(target, layer) {
|
|
78
|
-
if (!layer) return;
|
|
79
|
-
for (const key of Object.keys(layer)) {
|
|
80
|
-
const value = layer[key];
|
|
81
|
-
if (value !== void 0) target[key] = value;
|
|
82
|
-
}
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
// src/domains/providers/catalog.ts
|
|
86
|
-
init_esm_shims();
|
|
87
|
-
var engineAi = createEngineAi();
|
|
88
|
-
var CATALOG_PROVIDER_BY_RUNTIME_ID = /* @__PURE__ */ new Map([
|
|
89
|
-
["anthropic", "anthropic"],
|
|
90
|
-
["anthropic-max", "anthropic"],
|
|
91
|
-
["claude-code", "anthropic"],
|
|
92
|
-
["claude-sdk", "anthropic"],
|
|
93
|
-
["bedrock", "amazon-bedrock"],
|
|
94
|
-
["deepseek", "deepseek"],
|
|
95
|
-
["google", "google"],
|
|
96
|
-
["groq", "groq"],
|
|
97
|
-
["mistral", "mistral"],
|
|
98
|
-
["openai", "openai"],
|
|
99
|
-
["openai-codex", "openai-codex"],
|
|
100
|
-
["openrouter", "openrouter"]
|
|
101
|
-
]);
|
|
102
|
-
function catalogProviderForRuntime(runtimeId) {
|
|
103
|
-
return CATALOG_PROVIDER_BY_RUNTIME_ID.get(runtimeId);
|
|
104
|
-
}
|
|
105
|
-
function listCatalogModelsForRuntime(runtimeId) {
|
|
106
|
-
const provider = catalogProviderForRuntime(runtimeId);
|
|
107
|
-
if (!provider) return [];
|
|
108
|
-
try {
|
|
109
|
-
return engineAi.listModels(provider);
|
|
110
|
-
} catch {
|
|
111
|
-
return [];
|
|
112
|
-
}
|
|
113
|
-
}
|
|
114
|
-
function resolveEffectivePricing(target, runtimeId, wireModelId) {
|
|
115
|
-
if (target.pricing) {
|
|
116
|
-
const rates = {
|
|
117
|
-
input: target.pricing.input,
|
|
118
|
-
output: target.pricing.output,
|
|
119
|
-
cacheRead: target.pricing.cacheRead ?? 0,
|
|
120
|
-
cacheWrite: target.pricing.cacheWrite ?? 0
|
|
121
|
-
};
|
|
122
|
-
return {
|
|
123
|
-
rates,
|
|
124
|
-
provenance: Object.values(rates).every((rate) => rate === 0) ? "known_free" : "known"
|
|
125
|
-
};
|
|
126
|
-
}
|
|
127
|
-
const catalogModel = getCatalogModelForRuntime(runtimeId, wireModelId);
|
|
128
|
-
if (!catalogModel) return { rates: null, provenance: "unknown" };
|
|
129
|
-
return {
|
|
130
|
-
rates: {
|
|
131
|
-
input: catalogModel.cost.input,
|
|
132
|
-
output: catalogModel.cost.output,
|
|
133
|
-
cacheRead: catalogModel.cost.cacheRead,
|
|
134
|
-
cacheWrite: catalogModel.cost.cacheWrite
|
|
135
|
-
},
|
|
136
|
-
provenance: "estimated"
|
|
137
|
-
};
|
|
138
|
-
}
|
|
139
|
-
function resolveCostProvenance(target, runtimeId, wireModelId) {
|
|
140
|
-
return resolveEffectivePricing(target, runtimeId, wireModelId).provenance;
|
|
141
|
-
}
|
|
142
|
-
function getCatalogModelForRuntime(runtimeId, wireModelId) {
|
|
143
|
-
const provider = catalogProviderForRuntime(runtimeId);
|
|
144
|
-
if (!provider) return void 0;
|
|
145
|
-
try {
|
|
146
|
-
return engineAi.getModel(provider, wireModelId);
|
|
147
|
-
} catch {
|
|
148
|
-
return void 0;
|
|
149
|
-
}
|
|
150
|
-
}
|
|
151
|
-
function capabilitiesFromCatalogModel(defaultCapabilities25, model) {
|
|
152
|
-
if (!model) return defaultCapabilities25;
|
|
153
|
-
return {
|
|
154
|
-
...defaultCapabilities25,
|
|
155
|
-
reasoning: model.reasoning,
|
|
156
|
-
vision: model.input.includes("image"),
|
|
157
|
-
contextWindow: model.contextWindow,
|
|
158
|
-
maxTokens: model.maxTokens
|
|
159
|
-
};
|
|
160
|
-
}
|
|
161
|
-
function catalogThinkingLevelsForRuntime(runtimeId, wireModelId) {
|
|
162
|
-
const model = getCatalogModelForRuntime(runtimeId, wireModelId);
|
|
163
|
-
return model ? getEngineSupportedThinkingLevels(model) : void 0;
|
|
164
|
-
}
|
|
165
|
-
function synthesizeCatalogBackedModel(input) {
|
|
166
|
-
const builtin = getCatalogModelForRuntime(input.runtimeId, input.wireModelId);
|
|
167
|
-
const caps = mergeCapabilities(
|
|
168
|
-
capabilitiesFromCatalogModel(input.defaultCapabilities, builtin),
|
|
169
|
-
input.kb?.entry.capabilities ?? null,
|
|
170
|
-
null,
|
|
171
|
-
input.target.capabilities ?? null
|
|
172
|
-
);
|
|
173
|
-
const pricing = input.target.pricing;
|
|
174
|
-
const targetHeaders = input.target.auth?.headers;
|
|
175
|
-
const model = {
|
|
176
|
-
...builtin ?? {},
|
|
177
|
-
id: input.wireModelId,
|
|
178
|
-
name: `${input.wireModelId} (${input.target.id})`,
|
|
179
|
-
api: input.api,
|
|
180
|
-
provider: input.provider,
|
|
181
|
-
baseUrl: input.target.url ?? builtin?.baseUrl ?? input.defaultBaseUrl,
|
|
182
|
-
reasoning: caps.reasoning,
|
|
183
|
-
input: caps.vision ? builtin?.input.includes("image") ? builtin.input : ["text", "image"] : ["text"],
|
|
184
|
-
cost: {
|
|
185
|
-
input: pricing?.input ?? builtin?.cost.input ?? 0,
|
|
186
|
-
output: pricing?.output ?? builtin?.cost.output ?? 0,
|
|
187
|
-
cacheRead: pricing?.cacheRead ?? builtin?.cost.cacheRead ?? 0,
|
|
188
|
-
cacheWrite: pricing?.cacheWrite ?? builtin?.cost.cacheWrite ?? 0
|
|
189
|
-
},
|
|
190
|
-
contextWindow: caps.contextWindow,
|
|
191
|
-
maxTokens: caps.maxTokens
|
|
192
|
-
};
|
|
193
|
-
const headers = { ...input.defaultHeaders ?? {}, ...builtin?.headers ?? {}, ...targetHeaders ?? {} };
|
|
194
|
-
if (Object.keys(headers).length > 0) {
|
|
195
|
-
model.headers = headers;
|
|
196
|
-
}
|
|
197
|
-
return model;
|
|
198
|
-
}
|
|
199
|
-
|
|
200
|
-
// src/domains/providers/registry.ts
|
|
201
|
-
init_esm_shims();
|
|
202
|
-
import { readdirSync, statSync } from "node:fs";
|
|
203
|
-
import { join } from "node:path";
|
|
204
|
-
import { pathToFileURL } from "node:url";
|
|
205
|
-
var RUNTIME_KINDS = ["http", "sdk", "subprocess"];
|
|
206
|
-
var RUNTIME_AUTHS = ["api-key", "oauth", "aws-sdk", "vertex-adc", "claude-cli", "none"];
|
|
207
|
-
function createRuntimeRegistry() {
|
|
208
|
-
const byId = /* @__PURE__ */ new Map();
|
|
209
|
-
const canonical = /* @__PURE__ */ new Set();
|
|
210
|
-
const register = (desc) => {
|
|
211
|
-
const ids = [desc.id, ...desc.aliases ?? []];
|
|
212
|
-
const conflict = ids.find((id) => byId.has(id));
|
|
213
|
-
if (conflict) {
|
|
214
|
-
throw new Error(`runtime id '${conflict}' already registered`);
|
|
215
|
-
}
|
|
216
|
-
for (const id of ids) byId.set(id, desc);
|
|
217
|
-
canonical.add(desc.id);
|
|
218
|
-
};
|
|
219
|
-
const get = (id) => byId.get(id) ?? null;
|
|
220
|
-
const list = () => Array.from(canonical, (id) => byId.get(id)).filter((entry) => entry !== void 0);
|
|
221
|
-
const clear = () => {
|
|
222
|
-
byId.clear();
|
|
223
|
-
canonical.clear();
|
|
224
|
-
};
|
|
225
|
-
const loadFromDir = async (dir, beforeImport) => {
|
|
226
|
-
let entries;
|
|
227
|
-
try {
|
|
228
|
-
const stat = statSync(dir);
|
|
229
|
-
if (!stat.isDirectory()) return [];
|
|
230
|
-
entries = readdirSync(dir);
|
|
231
|
-
} catch {
|
|
232
|
-
return [];
|
|
233
|
-
}
|
|
234
|
-
const loaded = [];
|
|
235
|
-
for (const name of entries) {
|
|
236
|
-
if (!name.endsWith(".js")) continue;
|
|
237
|
-
const full = join(dir, name);
|
|
238
|
-
const desc = await importDescriptor(full, pathToFileURL(full).href, beforeImport);
|
|
239
|
-
if (desc === null) continue;
|
|
240
|
-
try {
|
|
241
|
-
register(desc);
|
|
242
|
-
loaded.push(desc.id);
|
|
243
|
-
} catch (err) {
|
|
244
|
-
process.stderr.write(`[providers] runtime plugin ${full} rejected: ${describeError(err)}
|
|
245
|
-
`);
|
|
246
|
-
}
|
|
247
|
-
}
|
|
248
|
-
return loaded;
|
|
249
|
-
};
|
|
250
|
-
const loadFromPackage = async (packageName, beforeImport) => {
|
|
251
|
-
let mod;
|
|
252
|
-
try {
|
|
253
|
-
await beforeImport?.();
|
|
254
|
-
mod = await import(packageName);
|
|
255
|
-
} catch (err) {
|
|
256
|
-
process.stderr.write(`[providers] runtime package ${packageName} failed to import: ${describeError(err)}
|
|
257
|
-
`);
|
|
258
|
-
return [];
|
|
259
|
-
}
|
|
260
|
-
const exported = mod.clioRuntimes;
|
|
261
|
-
if (!Array.isArray(exported)) {
|
|
262
|
-
process.stderr.write(`[providers] runtime package ${packageName} has no 'clioRuntimes' array export
|
|
263
|
-
`);
|
|
264
|
-
return [];
|
|
265
|
-
}
|
|
266
|
-
const loaded = [];
|
|
267
|
-
for (const candidate of exported) {
|
|
268
|
-
const validation = validateRuntimeDescriptor(candidate);
|
|
269
|
-
if (!validation.ok) {
|
|
270
|
-
process.stderr.write(
|
|
271
|
-
`[providers] runtime package ${packageName} exported an invalid descriptor: ${validation.reason}
|
|
272
|
-
`
|
|
273
|
-
);
|
|
274
|
-
continue;
|
|
275
|
-
}
|
|
276
|
-
try {
|
|
277
|
-
register(validation.descriptor);
|
|
278
|
-
loaded.push(validation.descriptor.id);
|
|
279
|
-
} catch (err) {
|
|
280
|
-
process.stderr.write(`[providers] runtime package ${packageName} id conflict: ${describeError(err)}
|
|
281
|
-
`);
|
|
282
|
-
}
|
|
283
|
-
}
|
|
284
|
-
return loaded;
|
|
285
|
-
};
|
|
286
|
-
return { register, get, list, clear, loadFromDir, loadFromPackage };
|
|
287
|
-
}
|
|
288
|
-
var singleton = null;
|
|
289
|
-
function getRuntimeRegistry() {
|
|
290
|
-
if (singleton === null) singleton = createRuntimeRegistry();
|
|
291
|
-
return singleton;
|
|
292
|
-
}
|
|
293
|
-
async function importDescriptor(file, href, beforeImport) {
|
|
294
|
-
let mod;
|
|
295
|
-
try {
|
|
296
|
-
await beforeImport?.();
|
|
297
|
-
mod = await import(href);
|
|
298
|
-
} catch (err) {
|
|
299
|
-
process.stderr.write(`[providers] runtime plugin ${file} failed to import: ${describeError(err)}
|
|
300
|
-
`);
|
|
301
|
-
return null;
|
|
302
|
-
}
|
|
303
|
-
const candidate = mod.default;
|
|
304
|
-
const validation = validateRuntimeDescriptor(candidate);
|
|
305
|
-
if (!validation.ok) {
|
|
306
|
-
process.stderr.write(
|
|
307
|
-
`[providers] runtime plugin ${file} has invalid default-export RuntimeDescriptor: ${validation.reason}
|
|
308
|
-
`
|
|
309
|
-
);
|
|
310
|
-
return null;
|
|
311
|
-
}
|
|
312
|
-
return validation.descriptor;
|
|
313
|
-
}
|
|
314
|
-
function validateRuntimeDescriptor(value) {
|
|
315
|
-
if (typeof value !== "object" || value === null || Array.isArray(value)) {
|
|
316
|
-
return { ok: false, reason: "descriptor must be an object" };
|
|
317
|
-
}
|
|
318
|
-
const v = value;
|
|
319
|
-
if (typeof v.id !== "string" || v.id.trim().length === 0) {
|
|
320
|
-
return { ok: false, reason: "id must be a non-empty string" };
|
|
321
|
-
}
|
|
322
|
-
if (v.aliases !== void 0 && (!Array.isArray(v.aliases) || v.aliases.some((alias) => typeof alias !== "string" || alias.trim().length === 0))) {
|
|
323
|
-
return { ok: false, reason: "aliases must be non-empty strings when present" };
|
|
324
|
-
}
|
|
325
|
-
if (typeof v.displayName !== "string" || v.displayName.trim().length === 0) {
|
|
326
|
-
return { ok: false, reason: "displayName must be a non-empty string" };
|
|
327
|
-
}
|
|
328
|
-
if (typeof v.kind !== "string" || !RUNTIME_KINDS.includes(v.kind)) {
|
|
329
|
-
return { ok: false, reason: `kind must be one of: ${RUNTIME_KINDS.join(", ")}` };
|
|
330
|
-
}
|
|
331
|
-
if (typeof v.apiFamily !== "string" || v.apiFamily.trim().length === 0) {
|
|
332
|
-
return { ok: false, reason: "apiFamily must be a non-empty string" };
|
|
333
|
-
}
|
|
334
|
-
if (typeof v.auth !== "string" || !RUNTIME_AUTHS.includes(v.auth)) {
|
|
335
|
-
return { ok: false, reason: `auth must be one of: ${RUNTIME_AUTHS.join(", ")}` };
|
|
336
|
-
}
|
|
337
|
-
if (typeof v.defaultCapabilities !== "object" || v.defaultCapabilities === null || Array.isArray(v.defaultCapabilities)) {
|
|
338
|
-
return { ok: false, reason: "defaultCapabilities must be an object" };
|
|
339
|
-
}
|
|
340
|
-
if (typeof v.synthesizeModel !== "function") {
|
|
341
|
-
return { ok: false, reason: "synthesizeModel must be a function" };
|
|
342
|
-
}
|
|
343
|
-
for (const field of ["probe", "probeModels", "complete", "infill", "embed", "rerank"]) {
|
|
344
|
-
if (v[field] !== void 0 && typeof v[field] !== "function") {
|
|
345
|
-
return { ok: false, reason: `${field} must be a function when present` };
|
|
346
|
-
}
|
|
347
|
-
}
|
|
348
|
-
return { ok: true, descriptor: value };
|
|
349
|
-
}
|
|
350
|
-
function describeError(err) {
|
|
351
|
-
if (err instanceof Error) return err.message;
|
|
352
|
-
return String(err);
|
|
353
|
-
}
|
|
354
|
-
|
|
355
|
-
// src/domains/providers/credentials.ts
|
|
356
|
-
init_esm_shims();
|
|
357
|
-
function credentialsPresent() {
|
|
358
|
-
const present = /* @__PURE__ */ new Set();
|
|
359
|
-
const registry = getRuntimeRegistry();
|
|
360
|
-
const auth = openAuthStorage();
|
|
361
|
-
for (const desc of registry.list()) {
|
|
362
|
-
const envVar = desc.credentialsEnvVar;
|
|
363
|
-
if (!envVar) continue;
|
|
364
|
-
const providerId = desc.id;
|
|
365
|
-
const status = auth.status(providerId, { explicitEnvVar: envVar, includeFallback: false });
|
|
366
|
-
if (status.available) {
|
|
367
|
-
present.add(envVar);
|
|
368
|
-
}
|
|
369
|
-
}
|
|
370
|
-
return present;
|
|
371
|
-
}
|
|
372
|
-
|
|
373
80
|
// src/engine/apis/residency.ts
|
|
374
81
|
init_esm_shims();
|
|
375
82
|
|
|
376
83
|
// src/engine/apis/residency-lock.ts
|
|
377
84
|
init_esm_shims();
|
|
378
|
-
import { join
|
|
85
|
+
import { join } from "node:path";
|
|
379
86
|
var LOCK_WAIT_MS = 12e4;
|
|
380
87
|
function lockTargetFor(targetKey) {
|
|
381
|
-
return
|
|
88
|
+
return join(clioStatePath(), "residency-locks", targetKey.replace(/[^a-zA-Z0-9._-]+/g, "_"));
|
|
382
89
|
}
|
|
383
90
|
async function withResidencyLock(targetKey, fn) {
|
|
384
91
|
return withStateFileLock(lockTargetFor(targetKey), fn, {
|
|
@@ -927,130 +634,6 @@ function availableThinkingLevels(caps, options) {
|
|
|
927
634
|
|
|
928
635
|
// src/domains/providers/model-capabilities.ts
|
|
929
636
|
init_esm_shims();
|
|
930
|
-
|
|
931
|
-
// src/domains/providers/types/local-model-quirks.ts
|
|
932
|
-
init_esm_shims();
|
|
933
|
-
var KV_CACHE_QUANTS = ["f32", "f16", "q8_0", "q4_0", "q4_1", "iq4_nl", "q5_0", "q5_1"];
|
|
934
|
-
function isRecord(value) {
|
|
935
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
936
|
-
}
|
|
937
|
-
function asKvCacheQuant(value) {
|
|
938
|
-
if (value === false) return false;
|
|
939
|
-
if (typeof value !== "string") return void 0;
|
|
940
|
-
return KV_CACHE_QUANTS.includes(value) ? value : void 0;
|
|
941
|
-
}
|
|
942
|
-
function asPositive(value) {
|
|
943
|
-
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : void 0;
|
|
944
|
-
}
|
|
945
|
-
function asInteger(value) {
|
|
946
|
-
const n = asPositive(value);
|
|
947
|
-
return n !== void 0 && Number.isInteger(n) ? n : void 0;
|
|
948
|
-
}
|
|
949
|
-
function asPenalty(value) {
|
|
950
|
-
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
951
|
-
}
|
|
952
|
-
function extractKvCache(raw) {
|
|
953
|
-
if (!isRecord(raw)) return void 0;
|
|
954
|
-
const out = {};
|
|
955
|
-
const k = asKvCacheQuant(raw.kQuant);
|
|
956
|
-
if (k !== void 0) out.kQuant = k;
|
|
957
|
-
const v = asKvCacheQuant(raw.vQuant);
|
|
958
|
-
if (v !== void 0) out.vQuant = v;
|
|
959
|
-
if (typeof raw.useFp16 === "boolean") out.useFp16 = raw.useFp16;
|
|
960
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
961
|
-
}
|
|
962
|
-
function extractSamplingProfile(raw) {
|
|
963
|
-
if (!isRecord(raw)) return void 0;
|
|
964
|
-
const out = {};
|
|
965
|
-
const temperature = asPositive(raw.temperature);
|
|
966
|
-
if (temperature !== void 0) out.temperature = temperature;
|
|
967
|
-
const topP = asPositive(raw.topP);
|
|
968
|
-
if (topP !== void 0) out.topP = topP;
|
|
969
|
-
const topK = asInteger(raw.topK);
|
|
970
|
-
if (topK !== void 0) out.topK = topK;
|
|
971
|
-
const minP = asPositive(raw.minP);
|
|
972
|
-
if (minP !== void 0) out.minP = minP;
|
|
973
|
-
const rp = asPenalty(raw.repeatPenalty) ?? asPenalty(raw.repetitionPenalty);
|
|
974
|
-
if (rp !== void 0) out.repeatPenalty = rp;
|
|
975
|
-
const pp = asPenalty(raw.presencePenalty);
|
|
976
|
-
if (pp !== void 0) out.presencePenalty = pp;
|
|
977
|
-
const fp = asPenalty(raw.frequencyPenalty);
|
|
978
|
-
if (fp !== void 0) out.frequencyPenalty = fp;
|
|
979
|
-
const maxTokens = asInteger(raw.maxTokens);
|
|
980
|
-
if (maxTokens !== void 0) out.maxTokens = maxTokens;
|
|
981
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
982
|
-
}
|
|
983
|
-
function extractSampling(raw) {
|
|
984
|
-
if (!isRecord(raw)) return void 0;
|
|
985
|
-
const out = {};
|
|
986
|
-
const thinking = extractSamplingProfile(raw.thinking);
|
|
987
|
-
if (thinking) out.thinking = thinking;
|
|
988
|
-
const instruct = extractSamplingProfile(raw.instruct);
|
|
989
|
-
if (instruct) out.instruct = instruct;
|
|
990
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
991
|
-
}
|
|
992
|
-
var THINKING_MECHANISMS = [
|
|
993
|
-
"effort-levels",
|
|
994
|
-
"budget-tokens",
|
|
995
|
-
"on-off",
|
|
996
|
-
"always-on",
|
|
997
|
-
"none"
|
|
998
|
-
];
|
|
999
|
-
function asThinkingMechanism(value) {
|
|
1000
|
-
if (typeof value !== "string") return void 0;
|
|
1001
|
-
return THINKING_MECHANISMS.includes(value) ? value : void 0;
|
|
1002
|
-
}
|
|
1003
|
-
function extractBudgetByLevel(raw) {
|
|
1004
|
-
if (!isRecord(raw)) return void 0;
|
|
1005
|
-
const out = {};
|
|
1006
|
-
const minimal = asInteger(raw.minimal);
|
|
1007
|
-
if (minimal !== void 0) out.minimal = minimal;
|
|
1008
|
-
const low = asInteger(raw.low);
|
|
1009
|
-
if (low !== void 0) out.low = low;
|
|
1010
|
-
const medium = asInteger(raw.medium);
|
|
1011
|
-
if (medium !== void 0) out.medium = medium;
|
|
1012
|
-
const high = asInteger(raw.high);
|
|
1013
|
-
if (high !== void 0) out.high = high;
|
|
1014
|
-
const xhigh = asInteger(raw.xhigh);
|
|
1015
|
-
if (xhigh !== void 0) out.xhigh = xhigh;
|
|
1016
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1017
|
-
}
|
|
1018
|
-
function extractEffortByLevel(raw) {
|
|
1019
|
-
if (!isRecord(raw)) return void 0;
|
|
1020
|
-
const out = {};
|
|
1021
|
-
if (typeof raw.off === "string" && raw.off.length > 0) out.off = raw.off;
|
|
1022
|
-
if (typeof raw.minimal === "string" && raw.minimal.length > 0) out.minimal = raw.minimal;
|
|
1023
|
-
if (typeof raw.low === "string" && raw.low.length > 0) out.low = raw.low;
|
|
1024
|
-
if (typeof raw.medium === "string" && raw.medium.length > 0) out.medium = raw.medium;
|
|
1025
|
-
if (typeof raw.high === "string" && raw.high.length > 0) out.high = raw.high;
|
|
1026
|
-
if (typeof raw.xhigh === "string" && raw.xhigh.length > 0) out.xhigh = raw.xhigh;
|
|
1027
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1028
|
-
}
|
|
1029
|
-
function extractThinkingQuirks(raw) {
|
|
1030
|
-
if (!isRecord(raw)) return void 0;
|
|
1031
|
-
const mechanism = asThinkingMechanism(raw.mechanism);
|
|
1032
|
-
if (!mechanism) return void 0;
|
|
1033
|
-
const out = { mechanism };
|
|
1034
|
-
const budgetByLevel = extractBudgetByLevel(raw.budgetByLevel);
|
|
1035
|
-
if (budgetByLevel) out.budgetByLevel = budgetByLevel;
|
|
1036
|
-
const effortByLevel = extractEffortByLevel(raw.effortByLevel);
|
|
1037
|
-
if (effortByLevel) out.effortByLevel = effortByLevel;
|
|
1038
|
-
if (typeof raw.guidance === "string" && raw.guidance.length > 0) out.guidance = raw.guidance;
|
|
1039
|
-
return out;
|
|
1040
|
-
}
|
|
1041
|
-
function extractLocalModelQuirks(raw) {
|
|
1042
|
-
if (!isRecord(raw)) return void 0;
|
|
1043
|
-
const out = {};
|
|
1044
|
-
const kvCache = extractKvCache(raw.kvCache);
|
|
1045
|
-
if (kvCache) out.kvCache = kvCache;
|
|
1046
|
-
const sampling = extractSampling(raw.sampling);
|
|
1047
|
-
if (sampling) out.sampling = sampling;
|
|
1048
|
-
const thinking = extractThinkingQuirks(raw.thinking);
|
|
1049
|
-
if (thinking) out.thinking = thinking;
|
|
1050
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1051
|
-
}
|
|
1052
|
-
|
|
1053
|
-
// src/domains/providers/model-capabilities.ts
|
|
1054
637
|
function normalizedModelId(wireModelId) {
|
|
1055
638
|
const trimmed = wireModelId?.trim();
|
|
1056
639
|
return trimmed ? trimmed : null;
|
|
@@ -1422,3038 +1005,185 @@ function resolveThinkingCapability(input, quirks, parser) {
|
|
|
1422
1005
|
`model has on/off thinking; ${configuredLevel} was coerced to ${thinkingLevelDisplayWord(mechanism, effectiveLevel)}`,
|
|
1423
1006
|
"ignored-on-off"
|
|
1424
1007
|
);
|
|
1425
|
-
} else if (mechanism === "always-on" && configuredLevel !== effectiveLevel) {
|
|
1426
|
-
applied = appendNotice(applied, `${configuredLevel} was ignored because thinking is always on`, "always-on");
|
|
1427
|
-
} else if (mechanism === "none" && configuredLevel !== effectiveLevel) {
|
|
1428
|
-
applied = appendNotice(applied, `${configuredLevel} was ignored because thinking is unsupported`, "unsupported");
|
|
1429
|
-
}
|
|
1430
|
-
const budgetEnforcement = resolveBudgetEnforcement(mechanism, input);
|
|
1431
|
-
if (applied.thinkingActive && mechanism === "budget-tokens" && budgetEnforcement === "informational") {
|
|
1432
|
-
applied = appendNotice(
|
|
1433
|
-
applied,
|
|
1434
|
-
"target does not expose an enforceable per-request thinking budget; level is advisory",
|
|
1435
|
-
"applied"
|
|
1436
|
-
);
|
|
1437
|
-
}
|
|
1438
|
-
return {
|
|
1439
|
-
...applied,
|
|
1440
|
-
configuredLevel,
|
|
1441
|
-
effectiveLevel,
|
|
1442
|
-
supportedLevels,
|
|
1443
|
-
display: thinkingLevelDisplayWord(applied.mechanism, effectiveLevel),
|
|
1444
|
-
budgetEnforcement
|
|
1445
|
-
};
|
|
1446
|
-
}
|
|
1447
|
-
var REASONING_EFFORT_ON_OFF_RUNTIMES = /* @__PURE__ */ new Set(["lmstudio"]);
|
|
1448
|
-
function onOffReasoningEffort(thinkingActive) {
|
|
1449
|
-
return thinkingActive ? "low" : "none";
|
|
1450
|
-
}
|
|
1451
|
-
function resolveRequestCapability(thinking, parser, runtimeId) {
|
|
1452
|
-
const request = { budgetEnforcement: thinking.budgetEnforcement };
|
|
1453
|
-
if (thinking.mechanism === "effort-levels" && thinking.effort) {
|
|
1454
|
-
request.reasoningEffort = thinking.effort;
|
|
1455
|
-
}
|
|
1456
|
-
if (thinking.mechanism === "effort-levels" && !thinking.thinkingActive) {
|
|
1457
|
-
request.chatTemplateKwargs = { ...request.chatTemplateKwargs ?? {}, enable_thinking: false };
|
|
1458
|
-
}
|
|
1459
|
-
if (thinking.mechanism === "budget-tokens" && thinking.budgetTokens !== void 0) {
|
|
1460
|
-
request.budgetTokens = thinking.budgetTokens;
|
|
1461
|
-
}
|
|
1462
|
-
if (thinking.mechanism === "on-off" && thinking.chatTemplateKwargs) {
|
|
1463
|
-
request.chatTemplateKwargs = { ...thinking.chatTemplateKwargs };
|
|
1464
|
-
if (REASONING_EFFORT_ON_OFF_RUNTIMES.has(runtimeId)) {
|
|
1465
|
-
request.reasoningEffort = onOffReasoningEffort(thinking.thinkingActive);
|
|
1466
|
-
}
|
|
1467
|
-
}
|
|
1468
|
-
if (parser === "harmony" && thinking.effort) {
|
|
1469
|
-
request.reasoningEffort = thinking.effort;
|
|
1470
|
-
request.chatTemplateKwargs = { ...request.chatTemplateKwargs ?? {}, reasoning_effort: thinking.effort };
|
|
1471
|
-
}
|
|
1472
|
-
return request;
|
|
1473
|
-
}
|
|
1474
|
-
function resolveModelRuntimeCapabilities(input) {
|
|
1475
|
-
const family = capabilityFamily(input);
|
|
1476
|
-
const quirks = resolveQuirks(input);
|
|
1477
|
-
const parser = resolveResponseParser(input, family);
|
|
1478
|
-
const thinking = resolveThinkingCapability(input, quirks, parser);
|
|
1479
|
-
const result = {
|
|
1480
|
-
targetId: input.targetId ?? null,
|
|
1481
|
-
runtimeId: input.runtimeId,
|
|
1482
|
-
apiFamily: input.apiFamily ?? null,
|
|
1483
|
-
modelId: input.modelId,
|
|
1484
|
-
family,
|
|
1485
|
-
capabilities: input.capabilities,
|
|
1486
|
-
thinking,
|
|
1487
|
-
request: resolveRequestCapability(thinking, parser, input.runtimeId),
|
|
1488
|
-
response: {
|
|
1489
|
-
parser,
|
|
1490
|
-
stripTokenizerSentinels: true
|
|
1491
|
-
}
|
|
1492
|
-
};
|
|
1493
|
-
if (quirks) result.quirks = quirks;
|
|
1494
|
-
return result;
|
|
1495
|
-
}
|
|
1496
|
-
function resolveModelRuntimeCapabilitiesForStatus(status, wireModelId, knowledgeBase, options) {
|
|
1497
|
-
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1498
|
-
const kbHit = modelId ? knowledgeBase?.lookup(modelId) ?? null : null;
|
|
1499
|
-
const capabilities2 = resolveModelCapabilities(status, modelId, knowledgeBase, {
|
|
1500
|
-
detectedReasoning: options?.detectedReasoning ?? null
|
|
1501
|
-
});
|
|
1502
|
-
const runtimeId = status.runtime?.id ?? status.target.runtime;
|
|
1503
|
-
return resolveModelRuntimeCapabilities({
|
|
1504
|
-
targetId: status.target.id,
|
|
1505
|
-
runtimeId,
|
|
1506
|
-
apiFamily: status.runtime?.apiFamily ?? null,
|
|
1507
|
-
modelId,
|
|
1508
|
-
capabilities: capabilities2,
|
|
1509
|
-
kbHit,
|
|
1510
|
-
...thinkingHintsForCatalogModel(runtimeId, modelId),
|
|
1511
|
-
...options?.configuredThinkingLevel ? { configuredThinkingLevel: options.configuredThinkingLevel } : {}
|
|
1512
|
-
});
|
|
1513
|
-
}
|
|
1514
|
-
function resolveModelRuntimeCapabilitiesForProviders(providers, targetId, wireModelId, configuredThinkingLevel) {
|
|
1515
|
-
const id = targetId?.trim();
|
|
1516
|
-
if (!id) return null;
|
|
1517
|
-
const status = providers.list().find((entry) => entry.target.id === id);
|
|
1518
|
-
if (!status) return null;
|
|
1519
|
-
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1520
|
-
const detectedReasoning = modelId && typeof providers.getDetectedReasoning === "function" ? providers.getDetectedReasoning(id, modelId) : null;
|
|
1521
|
-
return resolveModelRuntimeCapabilitiesForStatus(status, modelId, providers.knowledgeBase, {
|
|
1522
|
-
detectedReasoning,
|
|
1523
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1524
|
-
});
|
|
1525
|
-
}
|
|
1526
|
-
function thinkingHintsForModel(model) {
|
|
1527
|
-
const out = {};
|
|
1528
|
-
if (!model) return out;
|
|
1529
|
-
const compat = model.compat;
|
|
1530
|
-
if (compat?.forceAdaptiveThinking !== void 0) out.adaptiveThinking = compat.forceAdaptiveThinking;
|
|
1531
|
-
if (model.thinkingLevelMap) out.thinkingLevelMap = model.thinkingLevelMap;
|
|
1532
|
-
return out;
|
|
1533
|
-
}
|
|
1534
|
-
function thinkingHintsForCatalogModel(runtimeId, modelId) {
|
|
1535
|
-
if (!runtimeId || modelId.length === 0) return {};
|
|
1536
|
-
return thinkingHintsForModel(getCatalogModelForRuntime(runtimeId, modelId));
|
|
1537
|
-
}
|
|
1538
|
-
function thinkingFormatFromModelApi(api) {
|
|
1539
|
-
switch (api) {
|
|
1540
|
-
case "anthropic-messages":
|
|
1541
|
-
case "bedrock-converse-stream":
|
|
1542
|
-
case "claude-agent-sdk":
|
|
1543
|
-
case "claude-code-subprocess":
|
|
1544
|
-
return "anthropic-extended";
|
|
1545
|
-
case "openai-codex-responses":
|
|
1546
|
-
return "openai-codex";
|
|
1547
|
-
default:
|
|
1548
|
-
return void 0;
|
|
1549
|
-
}
|
|
1550
|
-
}
|
|
1551
|
-
function capabilitiesFromModel(model) {
|
|
1552
|
-
const format = model.compat?.thinkingFormat ?? thinkingFormatFromModelApi(model.api);
|
|
1553
|
-
const caps = {
|
|
1554
|
-
chat: true,
|
|
1555
|
-
tools: true,
|
|
1556
|
-
reasoning: model.reasoning === true,
|
|
1557
|
-
vision: Array.isArray(model.input) && model.input.includes("image"),
|
|
1558
|
-
audio: false,
|
|
1559
|
-
embeddings: false,
|
|
1560
|
-
rerank: false,
|
|
1561
|
-
fim: false,
|
|
1562
|
-
contextWindow: model.contextWindow,
|
|
1563
|
-
maxTokens: model.maxTokens
|
|
1564
|
-
};
|
|
1565
|
-
if (format === "qwen-chat-template" || format === "openrouter" || format === "zai" || format === "anthropic-extended" || format === "deepseek-r1" || format === "openai-codex" || format === "harmony") {
|
|
1566
|
-
caps.thinkingFormat = format;
|
|
1567
|
-
}
|
|
1568
|
-
return caps;
|
|
1569
|
-
}
|
|
1570
|
-
function resolveModelRuntimeCapabilitiesForModel(model, configuredThinkingLevel) {
|
|
1571
|
-
const metadata2 = model.clio;
|
|
1572
|
-
const caps = capabilitiesFromModel(model);
|
|
1573
|
-
return resolveModelRuntimeCapabilities({
|
|
1574
|
-
targetId: metadata2?.targetId ?? null,
|
|
1575
|
-
runtimeId: metadata2?.runtimeId ?? model.provider,
|
|
1576
|
-
apiFamily: model.api,
|
|
1577
|
-
modelId: model.id,
|
|
1578
|
-
capabilities: caps,
|
|
1579
|
-
...thinkingHintsForModel(model),
|
|
1580
|
-
...metadata2?.quirks ? { quirks: metadata2.quirks } : {},
|
|
1581
|
-
kbHit: metadata2?.family ? {
|
|
1582
|
-
matchKind: "family",
|
|
1583
|
-
entry: {
|
|
1584
|
-
family: metadata2.family,
|
|
1585
|
-
matchPatterns: [metadata2.family],
|
|
1586
|
-
capabilities: {}
|
|
1587
|
-
}
|
|
1588
|
-
} : null,
|
|
1589
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1590
|
-
});
|
|
1591
|
-
}
|
|
1592
|
-
function resolveTargetRuntimeCapabilities(target, runtime, wireModelId, capabilities2, knowledgeBase, configuredThinkingLevel) {
|
|
1593
|
-
const kbHit = knowledgeBase?.lookup(wireModelId) ?? null;
|
|
1594
|
-
return resolveModelRuntimeCapabilities({
|
|
1595
|
-
targetId: target.id,
|
|
1596
|
-
runtimeId: runtime.id,
|
|
1597
|
-
apiFamily: runtime.apiFamily,
|
|
1598
|
-
modelId: wireModelId,
|
|
1599
|
-
capabilities: capabilities2,
|
|
1600
|
-
kbHit,
|
|
1601
|
-
...thinkingHintsForCatalogModel(runtime.id, wireModelId),
|
|
1602
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1603
|
-
});
|
|
1604
|
-
}
|
|
1605
|
-
|
|
1606
|
-
// src/domains/providers/runtimes/common/lmstudio-http.ts
|
|
1607
|
-
init_esm_shims();
|
|
1608
|
-
import { performance } from "node:perf_hooks";
|
|
1609
|
-
|
|
1610
|
-
// src/domains/providers/runtimes/common/local-synth.ts
|
|
1611
|
-
init_esm_shims();
|
|
1612
|
-
function openAIThinkingFormat(caps) {
|
|
1613
|
-
switch (caps.thinkingFormat) {
|
|
1614
|
-
case "qwen-chat-template":
|
|
1615
|
-
case "openrouter":
|
|
1616
|
-
case "zai":
|
|
1617
|
-
return caps.thinkingFormat;
|
|
1618
|
-
case "deepseek-r1":
|
|
1619
|
-
return "deepseek";
|
|
1620
|
-
case "harmony":
|
|
1621
|
-
return "harmony";
|
|
1622
|
-
default:
|
|
1623
|
-
return void 0;
|
|
1624
|
-
}
|
|
1625
|
-
}
|
|
1626
|
-
function localOpenAICompat(caps, runtimeId) {
|
|
1627
|
-
const compat = {
|
|
1628
|
-
supportsStore: false,
|
|
1629
|
-
supportsDeveloperRole: false,
|
|
1630
|
-
supportsReasoningEffort: false,
|
|
1631
|
-
supportsUsageInStreaming: true,
|
|
1632
|
-
supportsFinishReason: false,
|
|
1633
|
-
maxTokensField: "max_tokens",
|
|
1634
|
-
supportsThinkingTokenBudget: runtimeId === "vllm",
|
|
1635
|
-
supportsStrictMode: false
|
|
1636
|
-
};
|
|
1637
|
-
const thinkingFormat = openAIThinkingFormat(caps);
|
|
1638
|
-
if (thinkingFormat) compat.thinkingFormat = thinkingFormat;
|
|
1639
|
-
return compat;
|
|
1640
|
-
}
|
|
1641
|
-
function localAnthropicCompat() {
|
|
1642
|
-
return {
|
|
1643
|
-
supportsEagerToolInputStreaming: false,
|
|
1644
|
-
supportsLongCacheRetention: false
|
|
1645
|
-
};
|
|
1646
|
-
}
|
|
1647
|
-
function synthLocalModel(input) {
|
|
1648
|
-
const { target, wireModelId, kb, defaultCapabilities: defaultCapabilities25, apiFamily, provider } = input;
|
|
1649
|
-
const caps = mergeCapabilities(defaultCapabilities25, kb?.entry.capabilities ?? null, null, target.capabilities ?? null);
|
|
1650
|
-
const rawUrl = target.url ?? "";
|
|
1651
|
-
const baseUrl = rawUrl.length > 0 ? input.baseUrlForTarget(rawUrl) : "";
|
|
1652
|
-
const pricing = target.pricing;
|
|
1653
|
-
const headers = target.auth?.headers;
|
|
1654
|
-
const quirks = extractLocalModelQuirks(kb?.entry.quirks);
|
|
1655
|
-
const model = {
|
|
1656
|
-
id: wireModelId,
|
|
1657
|
-
name: `${wireModelId} (${target.id})`,
|
|
1658
|
-
api: apiFamily,
|
|
1659
|
-
provider,
|
|
1660
|
-
baseUrl,
|
|
1661
|
-
reasoning: caps.reasoning,
|
|
1662
|
-
input: caps.vision ? ["text", "image"] : ["text"],
|
|
1663
|
-
cost: {
|
|
1664
|
-
input: pricing?.input ?? 0,
|
|
1665
|
-
output: pricing?.output ?? 0,
|
|
1666
|
-
cacheRead: pricing?.cacheRead ?? 0,
|
|
1667
|
-
cacheWrite: pricing?.cacheWrite ?? 0
|
|
1668
|
-
},
|
|
1669
|
-
contextWindow: caps.contextWindow,
|
|
1670
|
-
maxTokens: caps.maxTokens,
|
|
1671
|
-
clio: {
|
|
1672
|
-
targetId: target.id,
|
|
1673
|
-
runtimeId: target.runtime,
|
|
1674
|
-
...target.lifecycle ? { lifecycle: target.lifecycle } : {},
|
|
1675
|
-
...target.gateway === true ? { gateway: true } : {},
|
|
1676
|
-
...kb?.entry.family ? { family: kb.entry.family } : {},
|
|
1677
|
-
...quirks ? { quirks } : {},
|
|
1678
|
-
...target.lmstudio ? { lmstudio: target.lmstudio } : {}
|
|
1679
|
-
}
|
|
1680
|
-
};
|
|
1681
|
-
if (headers) model.headers = headers;
|
|
1682
|
-
if (apiFamily === "openai-completions") {
|
|
1683
|
-
model.compat = localOpenAICompat(caps, target.runtime);
|
|
1684
|
-
}
|
|
1685
|
-
if (apiFamily === "anthropic-messages") {
|
|
1686
|
-
model.compat = localAnthropicCompat();
|
|
1687
|
-
}
|
|
1688
|
-
return model;
|
|
1689
|
-
}
|
|
1690
|
-
function stripTrailingSlash(url2) {
|
|
1691
|
-
return url2.endsWith("/") ? url2.slice(0, -1) : url2;
|
|
1692
|
-
}
|
|
1693
|
-
function stripRedundantV1(url2) {
|
|
1694
|
-
return url2.endsWith("/v1") ? url2.slice(0, -"/v1".length) : url2;
|
|
1695
|
-
}
|
|
1696
|
-
var withV1 = (url2) => `${stripRedundantV1(stripTrailingSlash(url2))}/v1`;
|
|
1697
|
-
var withAsIs = (url2) => stripTrailingSlash(url2);
|
|
1698
|
-
function targetBaseUrl(target) {
|
|
1699
|
-
return target.url ? stripTrailingSlash(target.url) : null;
|
|
1700
|
-
}
|
|
1701
|
-
function targetRootUrl(target) {
|
|
1702
|
-
return target.url ? stripRedundantV1(stripTrailingSlash(target.url)) : null;
|
|
1703
|
-
}
|
|
1704
|
-
|
|
1705
|
-
// src/domains/providers/runtimes/common/lmstudio-http.ts
|
|
1706
|
-
var catalogsByHost = /* @__PURE__ */ new Map();
|
|
1707
|
-
function isRecord2(value) {
|
|
1708
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1709
|
-
}
|
|
1710
|
-
function positiveNumber(value) {
|
|
1711
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
1712
|
-
}
|
|
1713
|
-
function nonEmptyString(value) {
|
|
1714
|
-
return typeof value === "string" && value.trim().length > 0 ? value.trim() : void 0;
|
|
1715
|
-
}
|
|
1716
|
-
function lmStudioRootUrl(url2) {
|
|
1717
|
-
const normalized = targetRootUrl({ id: "lmstudio", runtime: "lmstudio", url: url2 });
|
|
1718
|
-
if (!normalized) return "";
|
|
1719
|
-
if (normalized.startsWith("ws://")) return `http://${normalized.slice("ws://".length)}`;
|
|
1720
|
-
if (normalized.startsWith("wss://")) return `https://${normalized.slice("wss://".length)}`;
|
|
1721
|
-
return normalized;
|
|
1722
|
-
}
|
|
1723
|
-
function lmStudioRequestHeaders(base, apiKey) {
|
|
1724
|
-
const headers = { ...base ?? {} };
|
|
1725
|
-
const token = apiKey?.trim();
|
|
1726
|
-
if (token) headers.authorization = `Bearer ${token}`;
|
|
1727
|
-
return headers;
|
|
1728
|
-
}
|
|
1729
|
-
function lmStudioProbeHeaders(target, ctx) {
|
|
1730
|
-
let token = ctx.authToken?.trim();
|
|
1731
|
-
const envName = target.auth?.apiKeyEnvVar;
|
|
1732
|
-
if (!token && envName && ctx.credentialsPresent.has(envName)) token = process.env[envName]?.trim();
|
|
1733
|
-
return lmStudioRequestHeaders(target.auth?.headers, token);
|
|
1734
|
-
}
|
|
1735
|
-
function combinedSignal(timeoutMs, signal) {
|
|
1736
|
-
const controller = new AbortController();
|
|
1737
|
-
const timer = setTimeout(() => controller.abort(new Error(`timeout after ${timeoutMs}ms`)), timeoutMs);
|
|
1738
|
-
const onAbort = () => controller.abort(signal?.reason);
|
|
1739
|
-
if (signal?.aborted) controller.abort(signal.reason);
|
|
1740
|
-
else signal?.addEventListener("abort", onAbort, { once: true });
|
|
1741
|
-
return {
|
|
1742
|
-
signal: controller.signal,
|
|
1743
|
-
close() {
|
|
1744
|
-
clearTimeout(timer);
|
|
1745
|
-
signal?.removeEventListener("abort", onAbort);
|
|
1746
|
-
}
|
|
1747
|
-
};
|
|
1748
|
-
}
|
|
1749
|
-
async function requestLmStudioJson(url2, init, timeoutMs, signal) {
|
|
1750
|
-
const started = performance.now();
|
|
1751
|
-
const bounded = combinedSignal(timeoutMs, signal);
|
|
1752
|
-
try {
|
|
1753
|
-
const response = await fetch(url2, { ...init, signal: bounded.signal });
|
|
1754
|
-
let data;
|
|
1755
|
-
try {
|
|
1756
|
-
data = await response.json();
|
|
1757
|
-
} catch {
|
|
1758
|
-
data = void 0;
|
|
1759
|
-
}
|
|
1760
|
-
const result = {
|
|
1761
|
-
ok: response.ok,
|
|
1762
|
-
status: response.status,
|
|
1763
|
-
latencyMs: Math.round(performance.now() - started)
|
|
1764
|
-
};
|
|
1765
|
-
if (data !== void 0) result.data = data;
|
|
1766
|
-
if (!response.ok) {
|
|
1767
|
-
const message = isRecord2(data) ? nonEmptyString(data.error) : void 0;
|
|
1768
|
-
result.error = message ?? `HTTP ${response.status}`;
|
|
1769
|
-
}
|
|
1770
|
-
return result;
|
|
1771
|
-
} catch (error) {
|
|
1772
|
-
return {
|
|
1773
|
-
ok: false,
|
|
1774
|
-
status: 0,
|
|
1775
|
-
latencyMs: Math.round(performance.now() - started),
|
|
1776
|
-
error: error instanceof Error ? error.message : String(error)
|
|
1777
|
-
};
|
|
1778
|
-
} finally {
|
|
1779
|
-
bounded.close();
|
|
1780
|
-
}
|
|
1781
|
-
}
|
|
1782
|
-
function loadedInstances(value) {
|
|
1783
|
-
if (!Array.isArray(value)) return [];
|
|
1784
|
-
const instances = [];
|
|
1785
|
-
for (const raw of value) {
|
|
1786
|
-
if (!isRecord2(raw)) continue;
|
|
1787
|
-
const id = nonEmptyString(raw.id);
|
|
1788
|
-
if (!id) continue;
|
|
1789
|
-
instances.push({ id, config: isRecord2(raw.config) ? { ...raw.config } : {} });
|
|
1790
|
-
}
|
|
1791
|
-
return instances;
|
|
1792
|
-
}
|
|
1793
|
-
function reasoningOptions(capabilities2) {
|
|
1794
|
-
if (!isRecord2(capabilities2) || !isRecord2(capabilities2.reasoning)) return void 0;
|
|
1795
|
-
const allowed = capabilities2.reasoning.allowed_options;
|
|
1796
|
-
if (!Array.isArray(allowed)) return void 0;
|
|
1797
|
-
const options = allowed.filter((entry) => typeof entry === "string");
|
|
1798
|
-
return options.length > 0 ? options : void 0;
|
|
1799
|
-
}
|
|
1800
|
-
function parseLmStudioV1Models(data) {
|
|
1801
|
-
if (!isRecord2(data) || !Array.isArray(data.models)) return null;
|
|
1802
|
-
const models = [];
|
|
1803
|
-
for (const raw of data.models) {
|
|
1804
|
-
if (!isRecord2(raw)) continue;
|
|
1805
|
-
const key = nonEmptyString(raw.key);
|
|
1806
|
-
if (!key) continue;
|
|
1807
|
-
const capabilities2 = isRecord2(raw.capabilities) ? raw.capabilities : void 0;
|
|
1808
|
-
const info = {
|
|
1809
|
-
key,
|
|
1810
|
-
loadedInstances: loadedInstances(raw.loaded_instances),
|
|
1811
|
-
metadata: { ...raw }
|
|
1812
|
-
};
|
|
1813
|
-
const type = nonEmptyString(raw.type);
|
|
1814
|
-
if (type) info.type = type;
|
|
1815
|
-
const maxContextLength = positiveNumber(raw.max_context_length);
|
|
1816
|
-
if (maxContextLength !== void 0) info.maxContextLength = maxContextLength;
|
|
1817
|
-
if (capabilities2 && typeof capabilities2.vision === "boolean") info.vision = capabilities2.vision;
|
|
1818
|
-
else if (type) info.vision = type === "vlm";
|
|
1819
|
-
if (capabilities2 && typeof capabilities2.trained_for_tool_use === "boolean") {
|
|
1820
|
-
info.tools = capabilities2.trained_for_tool_use;
|
|
1821
|
-
}
|
|
1822
|
-
const options = reasoningOptions(capabilities2);
|
|
1823
|
-
if (options) info.reasoningOptions = options;
|
|
1824
|
-
if (capabilities2 && "reasoning" in capabilities2) {
|
|
1825
|
-
info.reasoning = capabilities2.reasoning === true || isRecord2(capabilities2.reasoning);
|
|
1826
|
-
}
|
|
1827
|
-
models.push(info);
|
|
1828
|
-
}
|
|
1829
|
-
return models;
|
|
1830
|
-
}
|
|
1831
|
-
function foldLmStudioModels(models) {
|
|
1832
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
1833
|
-
for (const model of models) {
|
|
1834
|
-
const previous = byKey.get(model.key);
|
|
1835
|
-
if (!previous) {
|
|
1836
|
-
byKey.set(model.key, { ...model, loadedInstances: [...model.loadedInstances] });
|
|
1837
|
-
continue;
|
|
1838
|
-
}
|
|
1839
|
-
const knownIds = new Set(previous.loadedInstances.map((instance) => instance.id));
|
|
1840
|
-
for (const instance of model.loadedInstances) {
|
|
1841
|
-
if (!knownIds.has(instance.id)) previous.loadedInstances.push(instance);
|
|
1842
|
-
}
|
|
1843
|
-
if (previous.maxContextLength === void 0 && model.maxContextLength !== void 0) {
|
|
1844
|
-
previous.maxContextLength = model.maxContextLength;
|
|
1845
|
-
}
|
|
1846
|
-
if (previous.vision === void 0 && model.vision !== void 0) previous.vision = model.vision;
|
|
1847
|
-
if (previous.tools === void 0 && model.tools !== void 0) previous.tools = model.tools;
|
|
1848
|
-
if (previous.reasoningOptions === void 0 && model.reasoningOptions !== void 0) {
|
|
1849
|
-
previous.reasoningOptions = model.reasoningOptions;
|
|
1850
|
-
}
|
|
1851
|
-
if (previous.reasoning === void 0 && model.reasoning !== void 0) previous.reasoning = model.reasoning;
|
|
1852
|
-
}
|
|
1853
|
-
return [...byKey.values()];
|
|
1854
|
-
}
|
|
1855
|
-
function rememberHostCatalog(target, models) {
|
|
1856
|
-
const rootUrl2 = lmStudioRootUrl(target.url ?? "");
|
|
1857
|
-
if (!rootUrl2) return;
|
|
1858
|
-
catalogsByHost.set(rootUrl2, { targetId: target.id, rootUrl: rootUrl2, models: foldLmStudioModels(models) });
|
|
1859
|
-
}
|
|
1860
|
-
function peerTargetsFor(target, instanceId) {
|
|
1861
|
-
const rootUrl2 = lmStudioRootUrl(target.url ?? "");
|
|
1862
|
-
const peers = [];
|
|
1863
|
-
for (const catalog of catalogsByHost.values()) {
|
|
1864
|
-
if (catalog.rootUrl === rootUrl2) continue;
|
|
1865
|
-
if (catalog.models.some((model) => model.loadedInstances.some((instance) => instance.id === instanceId))) {
|
|
1866
|
-
peers.push(catalog.targetId);
|
|
1867
|
-
}
|
|
1868
|
-
}
|
|
1869
|
-
return [...new Set(peers)];
|
|
1870
|
-
}
|
|
1871
|
-
function resolveLmStudioInstance(target, models, requestedId, configuredDefault) {
|
|
1872
|
-
const folded = foldLmStudioModels(models);
|
|
1873
|
-
for (const model of folded) {
|
|
1874
|
-
const explicit = model.loadedInstances.find((instance2) => instance2.id === requestedId);
|
|
1875
|
-
if (explicit) {
|
|
1876
|
-
return {
|
|
1877
|
-
requestedId,
|
|
1878
|
-
wireModelId: explicit.id,
|
|
1879
|
-
model,
|
|
1880
|
-
instance: explicit,
|
|
1881
|
-
peerTargets: peerTargetsFor(target, explicit.id),
|
|
1882
|
-
state: "instance"
|
|
1883
|
-
};
|
|
1884
|
-
}
|
|
1885
|
-
if (model.key !== requestedId) continue;
|
|
1886
|
-
if (model.loadedInstances.length === 0) {
|
|
1887
|
-
return { requestedId, wireModelId: requestedId, model, peerTargets: [], state: "jit" };
|
|
1888
|
-
}
|
|
1889
|
-
const preferred = configuredDefault ? model.loadedInstances.find((instance2) => instance2.id === configuredDefault) : void 0;
|
|
1890
|
-
const local = model.loadedInstances.find((instance2) => peerTargetsFor(target, instance2.id).length === 0);
|
|
1891
|
-
const instance = preferred ?? local ?? model.loadedInstances[0];
|
|
1892
|
-
if (!instance) return { requestedId, wireModelId: requestedId, model, peerTargets: [], state: "jit" };
|
|
1893
|
-
return {
|
|
1894
|
-
requestedId,
|
|
1895
|
-
wireModelId: instance.id,
|
|
1896
|
-
model,
|
|
1897
|
-
instance,
|
|
1898
|
-
peerTargets: peerTargetsFor(target, instance.id),
|
|
1899
|
-
state: "instance"
|
|
1900
|
-
};
|
|
1901
|
-
}
|
|
1902
|
-
return { requestedId, wireModelId: requestedId, peerTargets: [], state: "unknown" };
|
|
1903
|
-
}
|
|
1904
|
-
function lmStudioReasoningLevels(options) {
|
|
1905
|
-
if (!options || options.length === 0) return [...THINKING_LEVELS];
|
|
1906
|
-
if (options.includes("on") && !options.includes("low") && !options.includes("medium") && !options.includes("high")) {
|
|
1907
|
-
return ["off", "low"];
|
|
1908
|
-
}
|
|
1909
|
-
const levels = ["off"];
|
|
1910
|
-
if (options.includes("low")) levels.push("minimal", "low");
|
|
1911
|
-
if (options.includes("medium")) levels.push("medium");
|
|
1912
|
-
if (options.includes("high")) levels.push("high", "xhigh", "max");
|
|
1913
|
-
return [...new Set(levels)];
|
|
1914
|
-
}
|
|
1915
|
-
function lmStudioReasoningEffort(level, options) {
|
|
1916
|
-
if (level === "off") return "none";
|
|
1917
|
-
if (options?.includes("on") && !options.includes("medium") && !options.includes("high")) return "low";
|
|
1918
|
-
if (level === "medium" && options?.includes("medium") !== false) return "medium";
|
|
1919
|
-
if ((level === "high" || level === "xhigh" || level === "max") && options?.includes("high") !== false) return "high";
|
|
1920
|
-
return "low";
|
|
1921
|
-
}
|
|
1922
|
-
function v0Models(data) {
|
|
1923
|
-
if (!isRecord2(data) || !Array.isArray(data.data)) return null;
|
|
1924
|
-
const models = [];
|
|
1925
|
-
for (const raw of data.data) {
|
|
1926
|
-
if (!isRecord2(raw)) continue;
|
|
1927
|
-
const key = nonEmptyString(raw.id);
|
|
1928
|
-
if (!key) continue;
|
|
1929
|
-
const state = nonEmptyString(raw.state);
|
|
1930
|
-
const loadedContext = positiveNumber(raw.loaded_context_length);
|
|
1931
|
-
const config = loadedContext === void 0 ? {} : { context_length: loadedContext };
|
|
1932
|
-
const info = {
|
|
1933
|
-
key,
|
|
1934
|
-
loadedInstances: state === "loaded" || loadedContext !== void 0 ? [{ id: key, config }] : [],
|
|
1935
|
-
metadata: { ...raw }
|
|
1936
|
-
};
|
|
1937
|
-
const type = nonEmptyString(raw.type);
|
|
1938
|
-
if (type) {
|
|
1939
|
-
info.type = type;
|
|
1940
|
-
info.vision = type === "vlm";
|
|
1941
|
-
}
|
|
1942
|
-
const maxContextLength = positiveNumber(raw.max_context_length);
|
|
1943
|
-
if (maxContextLength !== void 0) info.maxContextLength = maxContextLength;
|
|
1944
|
-
if (Array.isArray(raw.capabilities)) info.tools = raw.capabilities.includes("tool_use");
|
|
1945
|
-
models.push(info);
|
|
1946
|
-
}
|
|
1947
|
-
return models;
|
|
1948
|
-
}
|
|
1949
|
-
function openAIModels(data) {
|
|
1950
|
-
if (!isRecord2(data) || !Array.isArray(data.data)) return null;
|
|
1951
|
-
const models = [];
|
|
1952
|
-
for (const raw of data.data) {
|
|
1953
|
-
if (!isRecord2(raw)) continue;
|
|
1954
|
-
const key = nonEmptyString(raw.id);
|
|
1955
|
-
if (key) models.push({ key, loadedInstances: [], metadata: { ...raw } });
|
|
1956
|
-
}
|
|
1957
|
-
return models;
|
|
1958
|
-
}
|
|
1959
|
-
function authFailure(result) {
|
|
1960
|
-
if (result.status !== 401 && result.status !== 403) return null;
|
|
1961
|
-
return {
|
|
1962
|
-
ok: false,
|
|
1963
|
-
models: [],
|
|
1964
|
-
latencyMs: result.latencyMs,
|
|
1965
|
-
error: "LM Studio authentication required",
|
|
1966
|
-
authRequired: true
|
|
1967
|
-
};
|
|
1968
|
-
}
|
|
1969
|
-
async function listLmStudioModels(target, ctx) {
|
|
1970
|
-
if (!target.url) return { ok: false, models: [], error: "target has no url" };
|
|
1971
|
-
const root = lmStudioRootUrl(target.url);
|
|
1972
|
-
const headers = lmStudioProbeHeaders(target, ctx);
|
|
1973
|
-
const init = { headers };
|
|
1974
|
-
const v1 = await requestLmStudioJson(`${root}/api/v1/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1975
|
-
const v1Auth = authFailure(v1);
|
|
1976
|
-
if (v1Auth) return v1Auth;
|
|
1977
|
-
const parsedV1 = parseLmStudioV1Models(v1.data);
|
|
1978
|
-
if (v1.ok && parsedV1) {
|
|
1979
|
-
rememberHostCatalog(target, parsedV1);
|
|
1980
|
-
return { ok: true, models: parsedV1, tier: "0.4+", latencyMs: v1.latencyMs };
|
|
1981
|
-
}
|
|
1982
|
-
const v0 = await requestLmStudioJson(`${root}/api/v0/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1983
|
-
const v0Auth = authFailure(v0);
|
|
1984
|
-
if (v0Auth) return v0Auth;
|
|
1985
|
-
const parsedV0 = v0Models(v0.data);
|
|
1986
|
-
if (v0.ok && parsedV0) {
|
|
1987
|
-
rememberHostCatalog(target, parsedV0);
|
|
1988
|
-
return { ok: true, models: parsedV0, tier: "0.3.x", latencyMs: v0.latencyMs };
|
|
1989
|
-
}
|
|
1990
|
-
const openAI = await requestLmStudioJson(`${root}/v1/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1991
|
-
const openAIAuth = authFailure(openAI);
|
|
1992
|
-
if (openAIAuth) return openAIAuth;
|
|
1993
|
-
const parsedOpenAI = openAIModels(openAI.data);
|
|
1994
|
-
if (openAI.ok && parsedOpenAI) {
|
|
1995
|
-
rememberHostCatalog(target, parsedOpenAI);
|
|
1996
|
-
return { ok: true, models: parsedOpenAI, tier: "openai-compat", latencyMs: openAI.latencyMs };
|
|
1997
|
-
}
|
|
1998
|
-
return {
|
|
1999
|
-
ok: false,
|
|
2000
|
-
models: [],
|
|
2001
|
-
latencyMs: v1.latencyMs + v0.latencyMs + openAI.latencyMs,
|
|
2002
|
-
error: v1.error ?? v0.error ?? openAI.error ?? "LM Studio returned no recognized model catalog"
|
|
2003
|
-
};
|
|
2004
|
-
}
|
|
2005
|
-
async function greetLmStudio(target, ctx) {
|
|
2006
|
-
if (!target.url) return { ok: false, error: "target has no url" };
|
|
2007
|
-
const result = await requestLmStudioJson(
|
|
2008
|
-
`${lmStudioRootUrl(target.url)}/lmstudio-greeting`,
|
|
2009
|
-
{ headers: lmStudioProbeHeaders(target, ctx) },
|
|
2010
|
-
ctx.httpTimeoutMs,
|
|
2011
|
-
ctx.signal
|
|
2012
|
-
);
|
|
2013
|
-
if (result.ok && isRecord2(result.data) && result.data.lmstudio === true) {
|
|
2014
|
-
return { ok: true, latencyMs: result.latencyMs };
|
|
2015
|
-
}
|
|
2016
|
-
return { ok: false, latencyMs: result.latencyMs, error: result.error ?? "endpoint did not return {lmstudio:true}" };
|
|
2017
|
-
}
|
|
2018
|
-
function loadedContextLength(instance) {
|
|
2019
|
-
return positiveNumber(instance?.config.context_length);
|
|
2020
|
-
}
|
|
2021
|
-
|
|
2022
|
-
// src/domains/providers/runtimes/builtins.ts
|
|
2023
|
-
init_esm_shims();
|
|
2024
|
-
|
|
2025
|
-
// src/domains/providers/runtimes/antigravity/antigravity-code.ts
|
|
2026
|
-
init_esm_shims();
|
|
2027
|
-
var ANTIGRAVITY_AUTH_NOTICE = "Uses your existing Antigravity (`agy`) login. Clio stores no Antigravity credentials. agy runs its own agent harness as a subprocess; Clio maps autonomy levels onto its CLI flags but cannot mediate individual agy tool calls.";
|
|
2028
|
-
var ANTIGRAVITY_MODELS = [
|
|
2029
|
-
"Gemini 3.5 Flash (High)",
|
|
2030
|
-
"Gemini 3.5 Flash (Medium)",
|
|
2031
|
-
"Gemini 3.5 Flash (Low)",
|
|
2032
|
-
"Gemini 3.1 Pro (High)",
|
|
2033
|
-
"Gemini 3.1 Pro (Low)",
|
|
2034
|
-
"Claude Sonnet 4.6 (Thinking)",
|
|
2035
|
-
"Claude Opus 4.6 (Thinking)",
|
|
2036
|
-
"GPT-OSS 120B (Medium)"
|
|
2037
|
-
];
|
|
2038
|
-
var antigravityCapabilities = {
|
|
2039
|
-
chat: true,
|
|
2040
|
-
tools: true,
|
|
2041
|
-
toolCallFormat: "openai",
|
|
2042
|
-
reasoning: true,
|
|
2043
|
-
vision: true,
|
|
2044
|
-
audio: false,
|
|
2045
|
-
embeddings: false,
|
|
2046
|
-
rerank: false,
|
|
2047
|
-
fim: false,
|
|
2048
|
-
contextWindow: 1e6,
|
|
2049
|
-
maxTokens: 8192
|
|
2050
|
-
};
|
|
2051
|
-
var antigravityCodeRuntime = {
|
|
2052
|
-
id: "antigravity-code",
|
|
2053
|
-
displayName: "Antigravity CLI",
|
|
2054
|
-
kind: "subprocess",
|
|
2055
|
-
tier: "subscription",
|
|
2056
|
-
// Reuses the existing `google-generative-ai` api family: the worker branches
|
|
2057
|
-
// to its own runner before any pi-ai inference, so this only needs to be a
|
|
2058
|
-
// valid, google-backed family. The runner is selected by runtime id, and the
|
|
2059
|
-
// plain-text parser is named by `outputParser`.
|
|
2060
|
-
apiFamily: "google-generative-ai",
|
|
2061
|
-
auth: "none",
|
|
2062
|
-
authNotice: ANTIGRAVITY_AUTH_NOTICE,
|
|
2063
|
-
knownModels: [...ANTIGRAVITY_MODELS],
|
|
2064
|
-
binaryName: "agy",
|
|
2065
|
-
headlessCommand: "agy --print",
|
|
2066
|
-
outputParser: "antigravity-print-text",
|
|
2067
|
-
defaultCapabilities: antigravityCapabilities,
|
|
2068
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2069
|
-
return synthesizeCatalogBackedModel({
|
|
2070
|
-
target,
|
|
2071
|
-
wireModelId,
|
|
2072
|
-
kb,
|
|
2073
|
-
defaultCapabilities: antigravityCapabilities,
|
|
2074
|
-
runtimeId: "antigravity-code",
|
|
2075
|
-
api: "google-generative-ai",
|
|
2076
|
-
provider: "google",
|
|
2077
|
-
defaultBaseUrl: "antigravity://local"
|
|
2078
|
-
});
|
|
2079
|
-
}
|
|
2080
|
-
};
|
|
2081
|
-
var antigravity_code_default = antigravityCodeRuntime;
|
|
2082
|
-
|
|
2083
|
-
// src/domains/providers/runtimes/claude/claude-code.ts
|
|
2084
|
-
init_esm_shims();
|
|
2085
|
-
|
|
2086
|
-
// src/domains/providers/runtimes/claude/common.ts
|
|
2087
|
-
init_esm_shims();
|
|
2088
|
-
var CLAUDE_CODE_AUTH_NOTICE = "Uses your existing Claude Code login from the installed `claude` command. Clio stores no Claude Code credentials.";
|
|
2089
|
-
var CLAUDE_CODE_MODELS = [
|
|
2090
|
-
"sonnet",
|
|
2091
|
-
"opus",
|
|
2092
|
-
"haiku",
|
|
2093
|
-
"claude-sonnet-4-5",
|
|
2094
|
-
"claude-opus-4-5",
|
|
2095
|
-
"claude-haiku-4-5",
|
|
2096
|
-
"claude-3-7-sonnet-latest"
|
|
2097
|
-
];
|
|
2098
|
-
var claudeCodeCapabilities = {
|
|
2099
|
-
chat: true,
|
|
2100
|
-
tools: true,
|
|
2101
|
-
toolCallFormat: "anthropic",
|
|
2102
|
-
reasoning: true,
|
|
2103
|
-
thinkingFormat: "anthropic-extended",
|
|
2104
|
-
vision: true,
|
|
2105
|
-
audio: false,
|
|
2106
|
-
embeddings: false,
|
|
2107
|
-
rerank: false,
|
|
2108
|
-
fim: false,
|
|
2109
|
-
contextWindow: 2e5,
|
|
2110
|
-
maxTokens: 8192
|
|
2111
|
-
};
|
|
2112
|
-
function synthesizeClaudeDelegatedModel(input) {
|
|
2113
|
-
return synthesizeCatalogBackedModel({
|
|
2114
|
-
target: input.target,
|
|
2115
|
-
wireModelId: input.wireModelId,
|
|
2116
|
-
kb: input.kb,
|
|
2117
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2118
|
-
runtimeId: input.runtimeId,
|
|
2119
|
-
api: input.apiFamily,
|
|
2120
|
-
provider: "anthropic",
|
|
2121
|
-
defaultBaseUrl: "claude-code://local"
|
|
2122
|
-
});
|
|
2123
|
-
}
|
|
2124
|
-
|
|
2125
|
-
// src/domains/providers/runtimes/claude/claude-code.ts
|
|
2126
|
-
var claudeCodeRuntime = {
|
|
2127
|
-
id: "claude-code",
|
|
2128
|
-
displayName: "Claude Code CLI",
|
|
2129
|
-
kind: "subprocess",
|
|
2130
|
-
tier: "subscription",
|
|
2131
|
-
apiFamily: "claude-code-subprocess",
|
|
2132
|
-
auth: "claude-cli",
|
|
2133
|
-
authNotice: CLAUDE_CODE_AUTH_NOTICE,
|
|
2134
|
-
knownModels: [...CLAUDE_CODE_MODELS],
|
|
2135
|
-
binaryName: "claude",
|
|
2136
|
-
headlessCommand: "claude -p --output-format stream-json",
|
|
2137
|
-
outputParser: "claude-code-stream-json",
|
|
2138
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2139
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2140
|
-
return synthesizeClaudeDelegatedModel({
|
|
2141
|
-
target,
|
|
2142
|
-
wireModelId,
|
|
2143
|
-
kb,
|
|
2144
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2145
|
-
runtimeId: "claude-code",
|
|
2146
|
-
apiFamily: "claude-code-subprocess"
|
|
2147
|
-
});
|
|
2148
|
-
}
|
|
2149
|
-
};
|
|
2150
|
-
var claude_code_default = claudeCodeRuntime;
|
|
2151
|
-
|
|
2152
|
-
// src/domains/providers/runtimes/claude/claude-sdk.ts
|
|
2153
|
-
init_esm_shims();
|
|
2154
|
-
var claudeSdkRuntime = {
|
|
2155
|
-
id: "claude-sdk",
|
|
2156
|
-
displayName: "Claude Agent SDK",
|
|
2157
|
-
kind: "sdk",
|
|
2158
|
-
tier: "subscription",
|
|
2159
|
-
apiFamily: "claude-agent-sdk",
|
|
2160
|
-
auth: "claude-cli",
|
|
2161
|
-
authNotice: CLAUDE_CODE_AUTH_NOTICE,
|
|
2162
|
-
knownModels: [...CLAUDE_CODE_MODELS],
|
|
2163
|
-
binaryName: "claude",
|
|
2164
|
-
headlessCommand: "@anthropic-ai/claude-agent-sdk query()",
|
|
2165
|
-
outputParser: "claude-agent-sdk-messages",
|
|
2166
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2167
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2168
|
-
return synthesizeClaudeDelegatedModel({
|
|
2169
|
-
target,
|
|
2170
|
-
wireModelId,
|
|
2171
|
-
kb,
|
|
2172
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2173
|
-
runtimeId: "claude-sdk",
|
|
2174
|
-
apiFamily: "claude-agent-sdk"
|
|
2175
|
-
});
|
|
2176
|
-
}
|
|
2177
|
-
};
|
|
2178
|
-
var claude_sdk_default = claudeSdkRuntime;
|
|
2179
|
-
|
|
2180
|
-
// src/domains/providers/runtimes/cloud/alcf.ts
|
|
2181
|
-
init_esm_shims();
|
|
2182
|
-
|
|
2183
|
-
// src/domains/providers/probe/http.ts
|
|
2184
|
-
init_esm_shims();
|
|
2185
|
-
import { performance as performance2 } from "node:perf_hooks";
|
|
2186
|
-
async function probeHttp(opts) {
|
|
2187
|
-
const { response, latencyMs, error } = await runFetch(opts);
|
|
2188
|
-
if (response === null) return { ok: false, error: error ?? "unknown transport error", latencyMs };
|
|
2189
|
-
const method = opts.method ?? "GET";
|
|
2190
|
-
if (response.ok) return { ok: true, latencyMs };
|
|
2191
|
-
if (method === "HEAD" && response.status === 405) return { ok: true, latencyMs };
|
|
2192
|
-
return {
|
|
2193
|
-
ok: false,
|
|
2194
|
-
latencyMs,
|
|
2195
|
-
error: `HTTP ${response.status}: ${response.statusText}`
|
|
2196
|
-
};
|
|
2197
|
-
}
|
|
2198
|
-
async function probeJson(opts) {
|
|
2199
|
-
const { response, latencyMs, error } = await runFetch(opts);
|
|
2200
|
-
if (response === null) return { ok: false, error: error ?? "unknown transport error", latencyMs };
|
|
2201
|
-
const method = opts.method ?? "GET";
|
|
2202
|
-
if (!response.ok && !(method === "HEAD" && response.status === 405)) {
|
|
2203
|
-
return {
|
|
2204
|
-
ok: false,
|
|
2205
|
-
latencyMs,
|
|
2206
|
-
error: `HTTP ${response.status}: ${response.statusText}`
|
|
2207
|
-
};
|
|
2208
|
-
}
|
|
2209
|
-
let data;
|
|
2210
|
-
try {
|
|
2211
|
-
data = await response.json();
|
|
2212
|
-
} catch (err) {
|
|
2213
|
-
return { ok: false, latencyMs, error: `JSON parse: ${describeError2(err)}` };
|
|
2214
|
-
}
|
|
2215
|
-
return { ok: true, latencyMs, data };
|
|
2216
|
-
}
|
|
2217
|
-
async function runFetch(opts) {
|
|
2218
|
-
const controller = new AbortController();
|
|
2219
|
-
let timedOut = false;
|
|
2220
|
-
const timer = setTimeout(() => {
|
|
2221
|
-
timedOut = true;
|
|
2222
|
-
controller.abort();
|
|
2223
|
-
}, opts.timeoutMs);
|
|
2224
|
-
const onExternalAbort = () => controller.abort();
|
|
2225
|
-
if (opts.signal) {
|
|
2226
|
-
if (opts.signal.aborted) controller.abort();
|
|
2227
|
-
else opts.signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
2228
|
-
}
|
|
2229
|
-
const init = {
|
|
2230
|
-
method: opts.method ?? "GET",
|
|
2231
|
-
signal: controller.signal
|
|
2232
|
-
};
|
|
2233
|
-
if (opts.headers) init.headers = opts.headers;
|
|
2234
|
-
if (opts.body !== void 0) init.body = opts.body;
|
|
2235
|
-
const started = performance2.now();
|
|
2236
|
-
try {
|
|
2237
|
-
const response = await fetch(opts.url, init);
|
|
2238
|
-
return { response, latencyMs: Math.round(performance2.now() - started) };
|
|
2239
|
-
} catch (err) {
|
|
2240
|
-
const latencyMs = Math.round(performance2.now() - started);
|
|
2241
|
-
if (timedOut) {
|
|
2242
|
-
return { response: null, latencyMs, error: `timeout after ${opts.timeoutMs}ms` };
|
|
2243
|
-
}
|
|
2244
|
-
if (opts.signal?.aborted) {
|
|
2245
|
-
return { response: null, latencyMs, error: "aborted by caller" };
|
|
2246
|
-
}
|
|
2247
|
-
return { response: null, latencyMs, error: describeError2(err) };
|
|
2248
|
-
} finally {
|
|
2249
|
-
clearTimeout(timer);
|
|
2250
|
-
if (opts.signal) opts.signal.removeEventListener("abort", onExternalAbort);
|
|
2251
|
-
}
|
|
2252
|
-
}
|
|
2253
|
-
function describeError2(err) {
|
|
2254
|
-
if (!(err instanceof Error)) return String(err);
|
|
2255
|
-
const cause = err.cause;
|
|
2256
|
-
if (cause instanceof Error && cause.message.length > 0) {
|
|
2257
|
-
const code = cause.code;
|
|
2258
|
-
return typeof code === "string" && !cause.message.includes(code) ? `${cause.message} (${code})` : cause.message;
|
|
2259
|
-
}
|
|
2260
|
-
return err.message;
|
|
2261
|
-
}
|
|
2262
|
-
|
|
2263
|
-
// src/domains/providers/runtimes/protocol/openai-compat.ts
|
|
2264
|
-
init_esm_shims();
|
|
2265
|
-
|
|
2266
|
-
// src/core/context-floor.ts
|
|
2267
|
-
init_esm_shims();
|
|
2268
|
-
var CLIO_MIN_CONTEXT_WINDOW = 131072;
|
|
2269
|
-
var CLIO_MIN_MAX_OUTPUT_TOKENS = 32768;
|
|
2270
|
-
var CLIO_CONTEXT_WINDOW_WARN_BELOW = 128e3;
|
|
2271
|
-
|
|
2272
|
-
// src/domains/providers/probe/reasoning.ts
|
|
2273
|
-
init_esm_shims();
|
|
2274
|
-
import { performance as performance3 } from "node:perf_hooks";
|
|
2275
|
-
var PROMPT = "What is 2+2? Think briefly, then answer.";
|
|
2276
|
-
function nonEmptyString2(value) {
|
|
2277
|
-
return typeof value === "string" && value.trim().length > 0;
|
|
2278
|
-
}
|
|
2279
|
-
function detectReasoningField(data) {
|
|
2280
|
-
const message = data.choices?.[0]?.message;
|
|
2281
|
-
if (!message) return null;
|
|
2282
|
-
if (nonEmptyString2(message.reasoning_content)) return "reasoning_content";
|
|
2283
|
-
if (nonEmptyString2(message.reasoning)) return "reasoning";
|
|
2284
|
-
if (nonEmptyString2(message.reasoning_text)) return "reasoning_text";
|
|
2285
|
-
return null;
|
|
2286
|
-
}
|
|
2287
|
-
function trimTrailingSlash(url2) {
|
|
2288
|
-
return url2.endsWith("/") ? url2.slice(0, -1) : url2;
|
|
2289
|
-
}
|
|
2290
|
-
async function probeOpenAICompatReasoning(opts) {
|
|
2291
|
-
const controller = new AbortController();
|
|
2292
|
-
let timedOut = false;
|
|
2293
|
-
const timer = setTimeout(() => {
|
|
2294
|
-
timedOut = true;
|
|
2295
|
-
controller.abort();
|
|
2296
|
-
}, opts.timeoutMs);
|
|
2297
|
-
const onExternalAbort = () => controller.abort();
|
|
2298
|
-
if (opts.signal) {
|
|
2299
|
-
if (opts.signal.aborted) controller.abort();
|
|
2300
|
-
else opts.signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
2301
|
-
}
|
|
2302
|
-
const headers = { "content-type": "application/json" };
|
|
2303
|
-
if (opts.apiKey && opts.apiKey.length > 0) headers.authorization = `Bearer ${opts.apiKey}`;
|
|
2304
|
-
const body = JSON.stringify({
|
|
2305
|
-
model: opts.modelId,
|
|
2306
|
-
messages: [{ role: "user", content: PROMPT }],
|
|
2307
|
-
max_tokens: 200,
|
|
2308
|
-
temperature: 0,
|
|
2309
|
-
stream: false,
|
|
2310
|
-
reasoning_effort: "low"
|
|
2311
|
-
});
|
|
2312
|
-
const url2 = `${trimTrailingSlash(opts.baseUrl)}/v1/chat/completions`;
|
|
2313
|
-
const started = performance3.now();
|
|
2314
|
-
try {
|
|
2315
|
-
const response = await fetch(url2, {
|
|
2316
|
-
method: "POST",
|
|
2317
|
-
headers,
|
|
2318
|
-
body,
|
|
2319
|
-
signal: controller.signal
|
|
2320
|
-
});
|
|
2321
|
-
const latencyMs = Math.round(performance3.now() - started);
|
|
2322
|
-
if (!response.ok) {
|
|
2323
|
-
return { reasoning: false, latencyMs, error: `HTTP ${response.status}: ${response.statusText}` };
|
|
2324
|
-
}
|
|
2325
|
-
const data = await response.json();
|
|
2326
|
-
const field = detectReasoningField(data);
|
|
2327
|
-
if (field) return { reasoning: true, field, latencyMs };
|
|
2328
|
-
return { reasoning: false, latencyMs };
|
|
2329
|
-
} catch (err) {
|
|
2330
|
-
const latencyMs = Math.round(performance3.now() - started);
|
|
2331
|
-
if (timedOut) return { reasoning: false, latencyMs, error: `timeout after ${opts.timeoutMs}ms` };
|
|
2332
|
-
if (opts.signal?.aborted) return { reasoning: false, latencyMs, error: "aborted by caller" };
|
|
2333
|
-
return { reasoning: false, latencyMs, error: err instanceof Error ? err.message : String(err) };
|
|
2334
|
-
} finally {
|
|
2335
|
-
clearTimeout(timer);
|
|
2336
|
-
if (opts.signal) opts.signal.removeEventListener("abort", onExternalAbort);
|
|
2337
|
-
}
|
|
2338
|
-
}
|
|
2339
|
-
|
|
2340
|
-
// src/domains/providers/runtimes/common/probe-helpers.ts
|
|
2341
|
-
init_esm_shims();
|
|
2342
|
-
|
|
2343
|
-
// src/domains/providers/types/context-window-slots.ts
|
|
2344
|
-
init_esm_shims();
|
|
2345
|
-
function formatContextWindowSlots(contextWindow, slots) {
|
|
2346
|
-
const format = (n) => Math.round(n).toLocaleString("en-US");
|
|
2347
|
-
return `${format(contextWindow)} (${format(slots.totalContextSize)} / ${slots.slots} slots)`;
|
|
2348
|
-
}
|
|
2349
|
-
|
|
2350
|
-
// src/domains/providers/runtimes/common/probe-helpers.ts
|
|
2351
|
-
async function probeUrl(url2, ctx, method = "GET") {
|
|
2352
|
-
const base = { url: url2, timeoutMs: ctx.httpTimeoutMs, method };
|
|
2353
|
-
return ctx.signal ? probeHttp({ ...base, signal: ctx.signal }) : probeHttp(base);
|
|
2354
|
-
}
|
|
2355
|
-
async function probeOpenAIModels(base, ctx, modelsPath = "/v1/models") {
|
|
2356
|
-
return (await probeOpenAIModelCatalog(base, ctx, modelsPath)).models;
|
|
2357
|
-
}
|
|
2358
|
-
var OPENAI_COMPAT_DETAIL_PATHS = ["/api/v0/models"];
|
|
2359
|
-
async function probeModelDetailRows(base, ctx, modelsPath) {
|
|
2360
|
-
const rows = /* @__PURE__ */ new Map();
|
|
2361
|
-
for (const detailPath of OPENAI_COMPAT_DETAIL_PATHS) {
|
|
2362
|
-
if (detailPath === modelsPath) continue;
|
|
2363
|
-
const opts = { url: `${base}${detailPath}`, timeoutMs: ctx.httpTimeoutMs };
|
|
2364
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2365
|
-
if (!result.ok || !Array.isArray(result.data?.data)) continue;
|
|
2366
|
-
for (const row of result.data.data) {
|
|
2367
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2368
|
-
if (!rows.has(row.id)) rows.set(row.id, row);
|
|
2369
|
-
}
|
|
2370
|
-
if (rows.size > 0) break;
|
|
2371
|
-
}
|
|
2372
|
-
return rows;
|
|
2373
|
-
}
|
|
2374
|
-
async function probeOpenAIModelCatalog(base, ctx, modelsPath = "/v1/models") {
|
|
2375
|
-
const opts = { url: `${base}${modelsPath}`, timeoutMs: ctx.httpTimeoutMs };
|
|
2376
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2377
|
-
if (!result.ok || !result.data?.data) return { models: [], modelCapabilities: {}, modelStates: {} };
|
|
2378
|
-
const detail = await probeModelDetailRows(base, ctx, modelsPath);
|
|
2379
|
-
const models = [];
|
|
2380
|
-
const modelCapabilities = {};
|
|
2381
|
-
const modelStates = {};
|
|
2382
|
-
for (const row of result.data.data) {
|
|
2383
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2384
|
-
models.push(row.id);
|
|
2385
|
-
const detailRow = detail.get(row.id);
|
|
2386
|
-
const caps = {
|
|
2387
|
-
...detailRow ? capabilitiesFromOpenAIModelEntry(detailRow) : {},
|
|
2388
|
-
...capabilitiesFromOpenAIModelEntry(row)
|
|
2389
|
-
};
|
|
2390
|
-
if (Object.keys(caps).length > 0) modelCapabilities[row.id] = caps;
|
|
2391
|
-
const loadedContext = loadedContextFromEntry(row) ?? (detailRow ? loadedContextFromEntry(detailRow) : void 0);
|
|
2392
|
-
const state = modelStateFromOpenAIModelEntry(row) ?? (detailRow ? modelStateFromOpenAIModelEntry(detailRow) : void 0) ?? // A reported loaded context is itself the residency answer: nothing
|
|
2393
|
-
// serves a window for a model it has not loaded.
|
|
2394
|
-
(loadedContext !== void 0 ? { state: "loaded" } : void 0);
|
|
2395
|
-
const contextSlots = contextSlotsFromEntry(row) ?? (detailRow ? contextSlotsFromEntry(detailRow) : void 0);
|
|
2396
|
-
const withSlots = contextSlots ? { ...state ?? { state: "unknown" }, contextSlots } : state;
|
|
2397
|
-
if (withSlots) {
|
|
2398
|
-
modelStates[row.id] = loadedContext === void 0 ? withSlots : { ...withSlots, contextLength: loadedContext };
|
|
2399
|
-
}
|
|
2400
|
-
}
|
|
2401
|
-
return { models, modelCapabilities, modelStates };
|
|
2402
|
-
}
|
|
2403
|
-
function isRecord3(value) {
|
|
2404
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
2405
|
-
}
|
|
2406
|
-
function positiveNumber2(value) {
|
|
2407
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
2408
|
-
}
|
|
2409
|
-
function firstPositiveNumber(record, keys) {
|
|
2410
|
-
for (const key of keys) {
|
|
2411
|
-
const value = positiveNumber2(record[key]);
|
|
2412
|
-
if (value !== void 0) return value;
|
|
2413
|
-
}
|
|
2414
|
-
return void 0;
|
|
2415
|
-
}
|
|
2416
|
-
function capabilityListFlag(record, name) {
|
|
2417
|
-
const list = record.capabilities;
|
|
2418
|
-
if (!Array.isArray(list)) return void 0;
|
|
2419
|
-
return list.some((entry) => entry === name) ? true : void 0;
|
|
2420
|
-
}
|
|
2421
|
-
function booleanFromAny(record, keys) {
|
|
2422
|
-
for (const key of keys) {
|
|
2423
|
-
const value = record[key];
|
|
2424
|
-
if (typeof value === "boolean") return value;
|
|
2425
|
-
}
|
|
2426
|
-
return void 0;
|
|
2427
|
-
}
|
|
2428
|
-
function nestedRecord(record, key) {
|
|
2429
|
-
const value = record[key];
|
|
2430
|
-
return isRecord3(value) ? value : null;
|
|
2431
|
-
}
|
|
2432
|
-
function firstString(record, keys) {
|
|
2433
|
-
if (!record) return void 0;
|
|
2434
|
-
for (const key of keys) {
|
|
2435
|
-
const value = record[key];
|
|
2436
|
-
if (typeof value === "string" && value.trim().length > 0) return value.trim();
|
|
2437
|
-
}
|
|
2438
|
-
return void 0;
|
|
2439
|
-
}
|
|
2440
|
-
function firstBoolean(record, keys) {
|
|
2441
|
-
if (!record) return void 0;
|
|
2442
|
-
for (const key of keys) {
|
|
2443
|
-
const value = record[key];
|
|
2444
|
-
if (typeof value === "boolean") return value;
|
|
2445
|
-
}
|
|
2446
|
-
return void 0;
|
|
2447
|
-
}
|
|
2448
|
-
function normalizeModelState(raw) {
|
|
2449
|
-
const value = raw?.trim().toLowerCase().replace(/[\s_]+/g, "-");
|
|
2450
|
-
if (!value) return void 0;
|
|
2451
|
-
if (value === "loaded" || value === "ready" || value === "running" || value === "active") return "loaded";
|
|
2452
|
-
if (value === "loading" || value === "pending" || value === "queued" || value === "starting") return "loading";
|
|
2453
|
-
if (value === "unloaded" || value === "not-loaded" || value === "idle" || value === "sleeping" || value === "stopped") {
|
|
2454
|
-
return "unloaded";
|
|
2455
|
-
}
|
|
2456
|
-
if (value === "failed" || value === "error" || value === "errored") return "failed";
|
|
2457
|
-
if (value === "unknown") return "unknown";
|
|
2458
|
-
return void 0;
|
|
2459
|
-
}
|
|
2460
|
-
function modelStateFromOpenAIModelEntry(row) {
|
|
2461
|
-
const status = nestedRecord(row, "status");
|
|
2462
|
-
const failed = firstBoolean(status, ["failed"]) ?? firstBoolean(row, ["failed"]);
|
|
2463
|
-
if (failed === true) {
|
|
2464
|
-
const detail = firstString(status, ["error", "reason", "message"]) ?? firstString(row, ["error", "reason", "message"]);
|
|
2465
|
-
return detail ? { state: "failed", detail } : { state: "failed" };
|
|
2466
|
-
}
|
|
2467
|
-
const raw = typeof row.status === "string" ? row.status : firstString(status, ["value", "state", "status"]) ?? firstString(row, ["state"]);
|
|
2468
|
-
const normalized = normalizeModelState(raw);
|
|
2469
|
-
if (normalized) {
|
|
2470
|
-
const detail = firstString(status, ["detail", "message", "reason"]);
|
|
2471
|
-
return detail ? { state: normalized, detail } : { state: normalized };
|
|
2472
|
-
}
|
|
2473
|
-
const loaded = firstBoolean(status, ["loaded"]) ?? firstBoolean(row, ["loaded"]);
|
|
2474
|
-
if (loaded === true) return { state: "loaded" };
|
|
2475
|
-
if (loaded === false) return { state: "unloaded" };
|
|
2476
|
-
return void 0;
|
|
2477
|
-
}
|
|
2478
|
-
function loadedContextFromEntry(row) {
|
|
2479
|
-
const reported = firstPositiveNumber(row, ["loaded_context_length", "loadedContextLength"]);
|
|
2480
|
-
return reported === void 0 ? void 0 : Math.floor(reported);
|
|
2481
|
-
}
|
|
2482
|
-
function statusArgsFromEntry(row) {
|
|
2483
|
-
const status = nestedRecord(row, "status");
|
|
2484
|
-
return argsFromStatus(status);
|
|
2485
|
-
}
|
|
2486
|
-
function contextSlotsFromEntry(row) {
|
|
2487
|
-
return llamaCppRequestContextWindow(parseLlamaCppServerFlags(statusArgsFromEntry(row)))?.slots;
|
|
2488
|
-
}
|
|
2489
|
-
function capabilitiesFromOpenAIModelEntry(row) {
|
|
2490
|
-
const caps = {};
|
|
2491
|
-
const meta = nestedRecord(row, "meta");
|
|
2492
|
-
const flags = parseLlamaCppServerFlags(statusArgsFromEntry(row));
|
|
2493
|
-
const contextWindow = llamaCppRequestContextWindow(flags)?.contextWindow ?? firstPositiveNumber(row, [
|
|
2494
|
-
// What is actually loaded outranks what the model could support: a
|
|
2495
|
-
// model served at 8k out of a possible 262k has an 8k window today,
|
|
2496
|
-
// and the run has to be planned against the real one.
|
|
2497
|
-
"loaded_context_length",
|
|
2498
|
-
"loadedContextLength",
|
|
2499
|
-
"context_window",
|
|
2500
|
-
"contextWindow",
|
|
2501
|
-
"context_length",
|
|
2502
|
-
"contextLength",
|
|
2503
|
-
"max_context_length",
|
|
2504
|
-
"maxContextLength",
|
|
2505
|
-
"n_ctx"
|
|
2506
|
-
]) ?? (meta ? firstPositiveNumber(meta, ["n_ctx", "n_ctx_train", "context_length", "contextWindow"]) : void 0);
|
|
2507
|
-
if (contextWindow !== void 0) caps.contextWindow = Math.floor(contextWindow);
|
|
2508
|
-
const maxTokens = positiveNumber2(flags.maxTokens) ?? firstPositiveNumber(row, [
|
|
2509
|
-
"max_output_tokens",
|
|
2510
|
-
"maxOutputTokens",
|
|
2511
|
-
"max_completion_tokens",
|
|
2512
|
-
"maxCompletionTokens",
|
|
2513
|
-
"max_tokens",
|
|
2514
|
-
"maxTokens",
|
|
2515
|
-
"n_predict"
|
|
2516
|
-
]);
|
|
2517
|
-
if (maxTokens !== void 0) caps.maxTokens = Math.floor(maxTokens);
|
|
2518
|
-
const tools = flags.jinja ?? booleanFromAny(row, ["tools", "tool_use", "toolUse", "trained_for_tool_use"]) ?? capabilityListFlag(row, "tool_use");
|
|
2519
|
-
if (tools !== void 0) caps.tools = tools;
|
|
2520
|
-
const reasoning = flags.reasoning ?? (flags.reasoningBudget !== void 0 ? true : void 0) ?? booleanFromAny(row, ["reasoning", "thinking"]);
|
|
2521
|
-
if (reasoning !== void 0) caps.reasoning = reasoning;
|
|
2522
|
-
const architecture = nestedRecord(row, "architecture");
|
|
2523
|
-
const architectureInput = architecture?.input_modalities;
|
|
2524
|
-
const modalities = Array.isArray(row.modalities) ? row.modalities : Array.isArray(row.input) ? row.input : Array.isArray(architectureInput) ? architectureInput : null;
|
|
2525
|
-
if (modalities) {
|
|
2526
|
-
caps.vision = modalities.some((entry) => entry === "image" || entry === "vision");
|
|
2527
|
-
if (modalities.some((entry) => entry === "audio")) caps.audio = true;
|
|
2528
|
-
}
|
|
2529
|
-
return caps;
|
|
2530
|
-
}
|
|
2531
|
-
function modelEntries(payload) {
|
|
2532
|
-
if (!Array.isArray(payload?.data)) return [];
|
|
2533
|
-
const out = [];
|
|
2534
|
-
for (const row of payload.data) {
|
|
2535
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2536
|
-
out.push({ id: row.id, status: row.status });
|
|
2537
|
-
}
|
|
2538
|
-
return out;
|
|
2539
|
-
}
|
|
2540
|
-
async function probeOpenAIModelEntries(base, ctx) {
|
|
2541
|
-
const opts = { url: `${base}/v1/models`, timeoutMs: ctx.httpTimeoutMs };
|
|
2542
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2543
|
-
if (!result.ok) return [];
|
|
2544
|
-
return modelEntries(result.data);
|
|
2545
|
-
}
|
|
2546
|
-
function llamaCppRequestContextWindow(flags) {
|
|
2547
|
-
const total = positiveNumber2(flags.contextSize);
|
|
2548
|
-
if (total === void 0) return void 0;
|
|
2549
|
-
const parallel = positiveNumber2(flags.parallel);
|
|
2550
|
-
if (parallel === void 0 || parallel <= 1 || flags.kvUnified === true) return { contextWindow: Math.floor(total) };
|
|
2551
|
-
const slots = Math.floor(parallel);
|
|
2552
|
-
return {
|
|
2553
|
-
contextWindow: Math.floor(total / slots),
|
|
2554
|
-
slots: { totalContextSize: Math.floor(total), slots }
|
|
2555
|
-
};
|
|
2556
|
-
}
|
|
2557
|
-
function argsFromStatus(status) {
|
|
2558
|
-
if (!isRecord3(status)) return [];
|
|
2559
|
-
const args = status.args;
|
|
2560
|
-
if (Array.isArray(args)) return args.filter((entry) => typeof entry === "string");
|
|
2561
|
-
if (typeof args === "string") return args.trim().split(/\s+/).filter(Boolean);
|
|
2562
|
-
return [];
|
|
2563
|
-
}
|
|
2564
|
-
function looksLikeFlag(token) {
|
|
2565
|
-
return token.startsWith("-") && Number.isNaN(Number(token));
|
|
2566
|
-
}
|
|
2567
|
-
function valueAfter(args, ...flags) {
|
|
2568
|
-
for (const flag of flags) {
|
|
2569
|
-
const index = args.indexOf(flag);
|
|
2570
|
-
if (index < 0) continue;
|
|
2571
|
-
const value = args[index + 1];
|
|
2572
|
-
return value && !looksLikeFlag(value) ? value : void 0;
|
|
2573
|
-
}
|
|
2574
|
-
return void 0;
|
|
2575
|
-
}
|
|
2576
|
-
function numberFlag(args, ...flags) {
|
|
2577
|
-
const value = valueAfter(args, ...flags);
|
|
2578
|
-
if (value === void 0) return void 0;
|
|
2579
|
-
const parsed = Number(value);
|
|
2580
|
-
return Number.isFinite(parsed) ? parsed : void 0;
|
|
2581
|
-
}
|
|
2582
|
-
function booleanFlag(args, ...flags) {
|
|
2583
|
-
const flag = flags.find((candidate) => args.includes(candidate));
|
|
2584
|
-
if (flag === void 0) return void 0;
|
|
2585
|
-
const value = valueAfter(args, flag);
|
|
2586
|
-
if (value === void 0) return true;
|
|
2587
|
-
const normalized = value.toLowerCase();
|
|
2588
|
-
if (normalized === "true" || normalized === "on" || normalized === "1") return true;
|
|
2589
|
-
if (normalized === "false" || normalized === "off" || normalized === "0") return false;
|
|
2590
|
-
return void 0;
|
|
2591
|
-
}
|
|
2592
|
-
function parseLlamaCppServerFlags(args) {
|
|
2593
|
-
const flags = {};
|
|
2594
|
-
const ctxSize = numberFlag(args, "--ctx-size", "-c");
|
|
2595
|
-
if (ctxSize !== void 0) flags.contextSize = ctxSize;
|
|
2596
|
-
const maxTokens = numberFlag(args, "--n-predict");
|
|
2597
|
-
if (maxTokens !== void 0) flags.maxTokens = maxTokens;
|
|
2598
|
-
const flashAttention = booleanFlag(args, "--flash-attn");
|
|
2599
|
-
if (flashAttention !== void 0) flags.flashAttention = flashAttention;
|
|
2600
|
-
const jinja = booleanFlag(args, "--jinja");
|
|
2601
|
-
if (jinja !== void 0) flags.jinja = jinja;
|
|
2602
|
-
const reasoningRaw = valueAfter(args, "--reasoning");
|
|
2603
|
-
if (reasoningRaw) flags.reasoning = reasoningRaw === "on" || reasoningRaw === "true" || reasoningRaw === "1";
|
|
2604
|
-
const reasoningBudget = numberFlag(args, "--reasoning-budget");
|
|
2605
|
-
if (reasoningBudget !== void 0) flags.reasoningBudget = reasoningBudget;
|
|
2606
|
-
const temperature = numberFlag(args, "--temperature");
|
|
2607
|
-
if (temperature !== void 0) flags.temperature = temperature;
|
|
2608
|
-
const topP = numberFlag(args, "--top-p");
|
|
2609
|
-
if (topP !== void 0) flags.topP = topP;
|
|
2610
|
-
const topK = numberFlag(args, "--top-k");
|
|
2611
|
-
if (topK !== void 0) flags.topK = topK;
|
|
2612
|
-
const nGpuLayers = numberFlag(args, "--n-gpu-layers");
|
|
2613
|
-
if (nGpuLayers !== void 0) flags.nGpuLayers = nGpuLayers;
|
|
2614
|
-
const parallel = numberFlag(args, "--parallel", "-np");
|
|
2615
|
-
if (parallel !== void 0) flags.parallel = parallel;
|
|
2616
|
-
const kvUnifiedAt = Math.max(args.lastIndexOf("--kv-unified"), args.lastIndexOf("-kvu"));
|
|
2617
|
-
const noKvUnifiedAt = args.lastIndexOf("--no-kv-unified");
|
|
2618
|
-
if (noKvUnifiedAt > kvUnifiedAt) flags.kvUnified = false;
|
|
2619
|
-
else if (kvUnifiedAt >= 0) flags.kvUnified = booleanFlag(args, "--kv-unified", "-kvu") ?? true;
|
|
2620
|
-
const cacheTypeK = valueAfter(args, "--cache-type-k");
|
|
2621
|
-
if (cacheTypeK) flags.cacheTypeK = cacheTypeK;
|
|
2622
|
-
const cacheTypeV = valueAfter(args, "--cache-type-v");
|
|
2623
|
-
if (cacheTypeV) flags.cacheTypeV = cacheTypeV;
|
|
2624
|
-
const mmproj = valueAfter(args, "--mmproj");
|
|
2625
|
-
if (mmproj) flags.mmproj = mmproj;
|
|
2626
|
-
const chatTemplateKwargs = valueAfter(args, "--chat-template-kwargs");
|
|
2627
|
-
if (chatTemplateKwargs) flags.chatTemplateKwargs = chatTemplateKwargs;
|
|
2628
|
-
return flags;
|
|
2629
|
-
}
|
|
2630
|
-
function selectedModelEntry(entries, target) {
|
|
2631
|
-
const expected = target.defaultModel?.trim();
|
|
2632
|
-
if (expected) return entries.find((entry) => entry.id === expected) ?? null;
|
|
2633
|
-
return entries[0] ?? null;
|
|
2634
|
-
}
|
|
2635
|
-
function statusNotes(id, status) {
|
|
2636
|
-
if (!isRecord3(status)) return [];
|
|
2637
|
-
const notes = [];
|
|
2638
|
-
if (status.failed === true) notes.push(`llama.cpp router marks ${id} as failed`);
|
|
2639
|
-
const state = typeof status.state === "string" ? status.state : void 0;
|
|
2640
|
-
if (state === "loading") notes.push(`llama.cpp router reports ${id} is still loading`);
|
|
2641
|
-
return notes;
|
|
2642
|
-
}
|
|
2643
|
-
async function probeLlamaCppModelStatus(base, target, ctx) {
|
|
2644
|
-
const entries = await probeOpenAIModelEntries(base, ctx);
|
|
2645
|
-
const selected = selectedModelEntry(entries, target);
|
|
2646
|
-
if (!selected) return {};
|
|
2647
|
-
const args = argsFromStatus(selected.status);
|
|
2648
|
-
if (args.length === 0) return { notes: statusNotes(selected.id, selected.status) };
|
|
2649
|
-
const flags = parseLlamaCppServerFlags(args);
|
|
2650
|
-
const caps = {};
|
|
2651
|
-
const window = llamaCppRequestContextWindow(flags);
|
|
2652
|
-
if (window !== void 0) caps.contextWindow = window.contextWindow;
|
|
2653
|
-
if (flags.maxTokens !== void 0 && flags.maxTokens > 0) caps.maxTokens = flags.maxTokens;
|
|
2654
|
-
if (flags.reasoning === true || flags.reasoningBudget !== void 0) caps.reasoning = true;
|
|
2655
|
-
if (flags.mmproj) caps.vision = true;
|
|
2656
|
-
if (flags.jinja === true) caps.tools = true;
|
|
2657
|
-
const enrichment = { modelId: selected.id, serverFlags: flags };
|
|
2658
|
-
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
2659
|
-
const notes = statusNotes(selected.id, selected.status);
|
|
2660
|
-
if (window?.slots) {
|
|
2661
|
-
notes.push(
|
|
2662
|
-
`${selected.id} context window ${formatContextWindowSlots(window.contextWindow, window.slots)}: --ctx-size is split across --parallel slots without --kv-unified`
|
|
2663
|
-
);
|
|
2664
|
-
}
|
|
2665
|
-
if (notes.length > 0) enrichment.notes = notes;
|
|
2666
|
-
return enrichment;
|
|
2667
|
-
}
|
|
2668
|
-
async function detectModelMismatch(base, target, ctx) {
|
|
2669
|
-
const expected = target.defaultModel?.trim();
|
|
2670
|
-
if (!expected) return null;
|
|
2671
|
-
const ids = await probeOpenAIModels(base, ctx);
|
|
2672
|
-
if (ids.length === 0) return null;
|
|
2673
|
-
if (ids.includes(expected)) return null;
|
|
2674
|
-
const loaded = ids[0] ?? "(unknown)";
|
|
2675
|
-
return `wire model id ${expected} does not match server's loaded model ${loaded}; llama.cpp serves a single fixed model`;
|
|
2676
|
-
}
|
|
2677
|
-
async function probeLlamaCppProps(base, ctx) {
|
|
2678
|
-
const opts = { url: `${base}/props`, timeoutMs: ctx.httpTimeoutMs };
|
|
2679
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2680
|
-
if (!result.ok || !result.data) return {};
|
|
2681
|
-
const data = result.data;
|
|
2682
|
-
const enrichment = {};
|
|
2683
|
-
const caps = {};
|
|
2684
|
-
const nCtx = data.default_generation_settings?.n_ctx;
|
|
2685
|
-
if (typeof nCtx === "number" && nCtx > 0) caps.contextWindow = nCtx;
|
|
2686
|
-
const nPredict = data.default_generation_settings?.n_predict;
|
|
2687
|
-
if (typeof nPredict === "number" && nPredict > 0) caps.maxTokens = nPredict;
|
|
2688
|
-
const vision = data.modalities?.vision;
|
|
2689
|
-
if (typeof vision === "boolean") caps.vision = vision;
|
|
2690
|
-
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
2691
|
-
if (typeof data.build_info === "string" && data.build_info.length > 0) {
|
|
2692
|
-
enrichment.serverVersion = data.build_info;
|
|
2693
|
-
}
|
|
2694
|
-
return enrichment;
|
|
2695
|
-
}
|
|
2696
|
-
|
|
2697
|
-
// src/domains/providers/runtimes/protocol/openai-compat.ts
|
|
2698
|
-
function synthesizeOpenAICompatModel(input) {
|
|
2699
|
-
return synthLocalModel({
|
|
2700
|
-
target: input.target,
|
|
2701
|
-
wireModelId: input.wireModelId,
|
|
2702
|
-
kb: input.kb,
|
|
2703
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2704
|
-
apiFamily: input.apiFamily ?? "openai-completions",
|
|
2705
|
-
provider: input.provider,
|
|
2706
|
-
baseUrlForTarget: input.baseUrlForTarget ?? withV1
|
|
2707
|
-
});
|
|
2708
|
-
}
|
|
2709
|
-
function makeOpenAICompatRuntime(spec) {
|
|
2710
|
-
const apiFamily = spec.apiFamily ?? "openai-completions";
|
|
2711
|
-
const asIs = spec.baseUrlStyle === "asIs";
|
|
2712
|
-
const baseUrlForTarget = asIs ? withAsIs : withV1;
|
|
2713
|
-
const healthPath = spec.healthPath ?? "/v1/models";
|
|
2714
|
-
const modelsPath = spec.modelsPath ?? "/v1/models";
|
|
2715
|
-
const probeBase = asIs ? targetBaseUrl : targetRootUrl;
|
|
2716
|
-
return {
|
|
2717
|
-
id: spec.id,
|
|
2718
|
-
displayName: spec.displayName,
|
|
2719
|
-
kind: "http",
|
|
2720
|
-
tier: spec.tier,
|
|
2721
|
-
apiFamily,
|
|
2722
|
-
auth: spec.auth,
|
|
2723
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
2724
|
-
async probe(target, ctx) {
|
|
2725
|
-
const base = probeBase(target);
|
|
2726
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
2727
|
-
const health = await probeUrl(`${base}${healthPath}`, ctx);
|
|
2728
|
-
if (!health.ok) return health;
|
|
2729
|
-
const catalog = await probeOpenAIModelCatalog(base, ctx, modelsPath);
|
|
2730
|
-
const result = { ...health };
|
|
2731
|
-
if (catalog.models.length > 0) result.models = catalog.models;
|
|
2732
|
-
if (Object.keys(catalog.modelStates).length > 0) result.modelStates = catalog.modelStates;
|
|
2733
|
-
if (Object.keys(catalog.modelCapabilities).length > 0) {
|
|
2734
|
-
result.modelCapabilities = catalog.modelCapabilities;
|
|
2735
|
-
const selected = target.defaultModel?.trim();
|
|
2736
|
-
const selectedCaps = selected ? catalog.modelCapabilities[selected] : void 0;
|
|
2737
|
-
if (selected && selectedCaps) {
|
|
2738
|
-
result.discoveredCapabilities = selectedCaps;
|
|
2739
|
-
result.capabilityModelId = selected;
|
|
2740
|
-
}
|
|
2741
|
-
}
|
|
2742
|
-
return result;
|
|
2743
|
-
},
|
|
2744
|
-
async probeModels(target, ctx) {
|
|
2745
|
-
const base = probeBase(target);
|
|
2746
|
-
if (!base) return [];
|
|
2747
|
-
return probeOpenAIModels(base, ctx, modelsPath);
|
|
2748
|
-
},
|
|
2749
|
-
async probeReasoning(target, modelId, ctx) {
|
|
2750
|
-
const base = probeBase(target);
|
|
2751
|
-
if (!base) return { reasoning: false, latencyMs: 0, error: "target has no url" };
|
|
2752
|
-
const apiKeyEnv = target.auth?.apiKeyEnvVar;
|
|
2753
|
-
const apiKey = apiKeyEnv && ctx.credentialsPresent.has(apiKeyEnv) ? process.env[apiKeyEnv] : void 0;
|
|
2754
|
-
const probeOpts = {
|
|
2755
|
-
baseUrl: base,
|
|
2756
|
-
modelId,
|
|
2757
|
-
timeoutMs: Math.max(ctx.httpTimeoutMs, 8e3)
|
|
2758
|
-
};
|
|
2759
|
-
if (apiKey) probeOpts.apiKey = apiKey;
|
|
2760
|
-
if (ctx.signal) probeOpts.signal = ctx.signal;
|
|
2761
|
-
return probeOpenAICompatReasoning(probeOpts);
|
|
2762
|
-
},
|
|
2763
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2764
|
-
return synthesizeOpenAICompatModel({
|
|
2765
|
-
target,
|
|
2766
|
-
wireModelId,
|
|
2767
|
-
kb,
|
|
2768
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
2769
|
-
apiFamily,
|
|
2770
|
-
provider: spec.provider,
|
|
2771
|
-
baseUrlForTarget
|
|
2772
|
-
});
|
|
2773
|
-
}
|
|
2774
|
-
};
|
|
2775
|
-
}
|
|
2776
|
-
var defaultCapabilities = {
|
|
2777
|
-
chat: true,
|
|
2778
|
-
tools: false,
|
|
2779
|
-
reasoning: false,
|
|
2780
|
-
vision: false,
|
|
2781
|
-
audio: false,
|
|
2782
|
-
embeddings: false,
|
|
2783
|
-
rerank: false,
|
|
2784
|
-
fim: false,
|
|
2785
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
2786
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
2787
|
-
};
|
|
2788
|
-
var openai_compat_default = makeOpenAICompatRuntime({
|
|
2789
|
-
id: "openai-compat",
|
|
2790
|
-
displayName: "Generic OpenAI-compatible",
|
|
2791
|
-
provider: "openai-compat",
|
|
2792
|
-
auth: "api-key",
|
|
2793
|
-
tier: "protocol",
|
|
2794
|
-
defaultCapabilities
|
|
2795
|
-
});
|
|
2796
|
-
|
|
2797
|
-
// src/domains/providers/runtimes/cloud/alcf.ts
|
|
2798
|
-
var CATALOG_URL = "https://inference-api.alcf.anl.gov/resource_server/list-endpoints";
|
|
2799
|
-
var GATEWAY_ORIGIN = "https://inference-api.alcf.anl.gov/resource_server";
|
|
2800
|
-
var defaultCapabilities2 = {
|
|
2801
|
-
chat: true,
|
|
2802
|
-
tools: true,
|
|
2803
|
-
toolCallFormat: "openai",
|
|
2804
|
-
reasoning: true,
|
|
2805
|
-
vision: false,
|
|
2806
|
-
audio: false,
|
|
2807
|
-
embeddings: false,
|
|
2808
|
-
rerank: false,
|
|
2809
|
-
fim: false,
|
|
2810
|
-
contextWindow: 32768,
|
|
2811
|
-
maxTokens: 4096
|
|
2812
|
-
};
|
|
2813
|
-
var KNOWN_MODELS = [
|
|
2814
|
-
"openai/gpt-oss-120b",
|
|
2815
|
-
"openai/gpt-oss-20b",
|
|
2816
|
-
"gpt-oss-120b",
|
|
2817
|
-
"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
|
2818
|
-
"meta-llama/Llama-4-Scout-17B-16E-Instruct"
|
|
2819
|
-
];
|
|
2820
|
-
function clusterFromUrl(url2) {
|
|
2821
|
-
if (!url2) return null;
|
|
2822
|
-
const match = /\/resource_server\/([^/]+)\//.exec(url2);
|
|
2823
|
-
return match ? match[1] ?? null : null;
|
|
2824
|
-
}
|
|
2825
|
-
function frameworkForCluster(cluster) {
|
|
2826
|
-
return cluster === "metis" ? "api" : "vllm";
|
|
2827
|
-
}
|
|
2828
|
-
function catalogModels(payload, cluster, framework) {
|
|
2829
|
-
const models = payload.clusters?.[cluster]?.frameworks?.[framework]?.models;
|
|
2830
|
-
if (!Array.isArray(models)) return [];
|
|
2831
|
-
return models.map((model) => String(model).trim()).filter((model) => model.length > 0);
|
|
2832
|
-
}
|
|
2833
|
-
function runningModels(payload) {
|
|
2834
|
-
const out = [];
|
|
2835
|
-
for (const job of payload.running ?? []) {
|
|
2836
|
-
for (const raw of String(job.Models ?? "").split(",")) {
|
|
2837
|
-
const id = raw.trim();
|
|
2838
|
-
if (id.length > 0) out.push(id);
|
|
2839
|
-
}
|
|
2840
|
-
}
|
|
2841
|
-
return out;
|
|
2842
|
-
}
|
|
2843
|
-
function dedupe(ids) {
|
|
2844
|
-
const seen = /* @__PURE__ */ new Set();
|
|
2845
|
-
const out = [];
|
|
2846
|
-
for (const id of ids) {
|
|
2847
|
-
if (seen.has(id)) continue;
|
|
2848
|
-
seen.add(id);
|
|
2849
|
-
out.push(id);
|
|
2850
|
-
}
|
|
2851
|
-
return out;
|
|
2852
|
-
}
|
|
2853
|
-
function authHeaders(ctx) {
|
|
2854
|
-
return { Authorization: `Bearer ${ctx.authToken}`, Accept: "application/json" };
|
|
2855
|
-
}
|
|
2856
|
-
async function discover(target, ctx) {
|
|
2857
|
-
if (!target.url) return { ok: false, error: "ALCF target has no url" };
|
|
2858
|
-
const cluster = clusterFromUrl(target.url);
|
|
2859
|
-
if (!cluster) {
|
|
2860
|
-
return { ok: false, error: `cannot determine ALCF cluster from url ${target.url}` };
|
|
2861
|
-
}
|
|
2862
|
-
if (!ctx.authToken) {
|
|
2863
|
-
return { ok: false, error: "ALCF requires Globus auth; run `clio-coder auth login alcf`." };
|
|
2864
|
-
}
|
|
2865
|
-
const framework = frameworkForCluster(cluster);
|
|
2866
|
-
const headers = authHeaders(ctx);
|
|
2867
|
-
const catalog = await probeJson({
|
|
2868
|
-
url: CATALOG_URL,
|
|
2869
|
-
headers,
|
|
2870
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
2871
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
2872
|
-
});
|
|
2873
|
-
if (!catalog.ok || !catalog.data) {
|
|
2874
|
-
const reason = catalog.error ?? "unknown error";
|
|
2875
|
-
const hint = reason.includes("401") ? " (token expired or rejected; run `clio-coder auth login alcf` again)" : "";
|
|
2876
|
-
return {
|
|
2877
|
-
ok: false,
|
|
2878
|
-
error: `ALCF endpoint catalog unreachable: ${reason}${hint}`,
|
|
2879
|
-
...catalog.latencyMs !== void 0 ? { latencyMs: catalog.latencyMs } : {}
|
|
2880
|
-
};
|
|
2881
|
-
}
|
|
2882
|
-
const fromCatalog = catalogModels(catalog.data, cluster, framework);
|
|
2883
|
-
let live = [];
|
|
2884
|
-
try {
|
|
2885
|
-
const jobs = await probeJson({
|
|
2886
|
-
url: `${GATEWAY_ORIGIN}/${cluster}/jobs`,
|
|
2887
|
-
headers,
|
|
2888
|
-
timeoutMs: Math.min(ctx.httpTimeoutMs, 12e3),
|
|
2889
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
2890
|
-
});
|
|
2891
|
-
if (jobs.ok && jobs.data) live = runningModels(jobs.data);
|
|
2892
|
-
} catch {
|
|
2893
|
-
}
|
|
2894
|
-
const models = dedupe([...fromCatalog, ...live]);
|
|
2895
|
-
const result = { ok: true, models };
|
|
2896
|
-
if (catalog.latencyMs !== void 0) result.latencyMs = catalog.latencyMs;
|
|
2897
|
-
if (models.length === 0) {
|
|
2898
|
-
result.notes = [`ALCF ${cluster}/${framework} reported no models; the gateway may have no running jobs.`];
|
|
2899
|
-
}
|
|
2900
|
-
return result;
|
|
2901
|
-
}
|
|
2902
|
-
var alcfRuntime = {
|
|
2903
|
-
id: "alcf",
|
|
2904
|
-
displayName: "ALCF Inference (Globus)",
|
|
2905
|
-
kind: "http",
|
|
2906
|
-
tier: "cloud",
|
|
2907
|
-
apiFamily: "openai-completions",
|
|
2908
|
-
auth: "oauth",
|
|
2909
|
-
knownModels: KNOWN_MODELS,
|
|
2910
|
-
defaultCapabilities: defaultCapabilities2,
|
|
2911
|
-
probe(target, ctx) {
|
|
2912
|
-
return discover(target, ctx);
|
|
2913
|
-
},
|
|
2914
|
-
async probeModels(target, ctx) {
|
|
2915
|
-
const result = await discover(target, ctx);
|
|
2916
|
-
return result.models ?? [];
|
|
2917
|
-
},
|
|
2918
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2919
|
-
const model = synthesizeOpenAICompatModel({
|
|
2920
|
-
target,
|
|
2921
|
-
wireModelId,
|
|
2922
|
-
kb,
|
|
2923
|
-
defaultCapabilities: defaultCapabilities2,
|
|
2924
|
-
apiFamily: "openai-completions",
|
|
2925
|
-
provider: "alcf",
|
|
2926
|
-
baseUrlForTarget: withAsIs
|
|
2927
|
-
});
|
|
2928
|
-
const meta = model.clio;
|
|
2929
|
-
if (meta) meta.chatTemplateKwargsUnsupported = true;
|
|
2930
|
-
return model;
|
|
2931
|
-
}
|
|
2932
|
-
};
|
|
2933
|
-
var alcf_default = alcfRuntime;
|
|
2934
|
-
|
|
2935
|
-
// src/domains/providers/runtimes/cloud/anthropic.ts
|
|
2936
|
-
init_esm_shims();
|
|
2937
|
-
|
|
2938
|
-
// src/domains/providers/runtimes/protocol/anthropic-messages.ts
|
|
2939
|
-
init_esm_shims();
|
|
2940
|
-
function synthesizeAnthropicMessagesModel(input) {
|
|
2941
|
-
return synthesizeCatalogBackedModel({
|
|
2942
|
-
target: input.target,
|
|
2943
|
-
wireModelId: input.wireModelId,
|
|
2944
|
-
kb: input.kb,
|
|
2945
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2946
|
-
runtimeId: "anthropic",
|
|
2947
|
-
api: "anthropic-messages",
|
|
2948
|
-
provider: input.provider,
|
|
2949
|
-
defaultBaseUrl: input.defaultBaseUrl
|
|
2950
|
-
});
|
|
2951
|
-
}
|
|
2952
|
-
|
|
2953
|
-
// src/domains/providers/runtimes/cloud/anthropic.ts
|
|
2954
|
-
var defaultCapabilities3 = {
|
|
2955
|
-
chat: true,
|
|
2956
|
-
tools: true,
|
|
2957
|
-
toolCallFormat: "anthropic",
|
|
2958
|
-
reasoning: true,
|
|
2959
|
-
thinkingFormat: "anthropic-extended",
|
|
2960
|
-
vision: true,
|
|
2961
|
-
audio: false,
|
|
2962
|
-
embeddings: false,
|
|
2963
|
-
rerank: false,
|
|
2964
|
-
fim: false,
|
|
2965
|
-
contextWindow: 2e5,
|
|
2966
|
-
maxTokens: 8192
|
|
2967
|
-
};
|
|
2968
|
-
var anthropicRuntime = {
|
|
2969
|
-
id: "anthropic",
|
|
2970
|
-
displayName: "Anthropic",
|
|
2971
|
-
kind: "http",
|
|
2972
|
-
tier: "cloud",
|
|
2973
|
-
apiFamily: "anthropic-messages",
|
|
2974
|
-
auth: "api-key",
|
|
2975
|
-
credentialsEnvVar: "ANTHROPIC_API_KEY",
|
|
2976
|
-
defaultCapabilities: defaultCapabilities3,
|
|
2977
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2978
|
-
return synthesizeAnthropicMessagesModel({
|
|
2979
|
-
target,
|
|
2980
|
-
wireModelId,
|
|
2981
|
-
kb,
|
|
2982
|
-
defaultCapabilities: defaultCapabilities3,
|
|
2983
|
-
provider: "anthropic",
|
|
2984
|
-
defaultBaseUrl: "https://api.anthropic.com"
|
|
2985
|
-
});
|
|
2986
|
-
}
|
|
2987
|
-
};
|
|
2988
|
-
var anthropic_default = anthropicRuntime;
|
|
2989
|
-
|
|
2990
|
-
// src/domains/providers/runtimes/cloud/anthropic-max.ts
|
|
2991
|
-
init_esm_shims();
|
|
2992
|
-
var defaultCapabilities4 = {
|
|
2993
|
-
chat: true,
|
|
2994
|
-
tools: true,
|
|
2995
|
-
toolCallFormat: "anthropic",
|
|
2996
|
-
reasoning: true,
|
|
2997
|
-
thinkingFormat: "anthropic-extended",
|
|
2998
|
-
vision: true,
|
|
2999
|
-
audio: false,
|
|
3000
|
-
embeddings: false,
|
|
3001
|
-
rerank: false,
|
|
3002
|
-
fim: false,
|
|
3003
|
-
contextWindow: 2e5,
|
|
3004
|
-
maxTokens: 8192
|
|
3005
|
-
};
|
|
3006
|
-
var anthropicMaxRuntime = {
|
|
3007
|
-
id: "anthropic-max",
|
|
3008
|
-
displayName: "Anthropic (Claude Pro/Max)",
|
|
3009
|
-
kind: "http",
|
|
3010
|
-
tier: "cloud",
|
|
3011
|
-
apiFamily: "anthropic-messages",
|
|
3012
|
-
auth: "oauth",
|
|
3013
|
-
oauthProviderId: "anthropic",
|
|
3014
|
-
authNotice: "Connects with your Claude Pro/Max subscription via OAuth (the same path Claude Code uses). Using subscription credentials outside Anthropic's first-party apps may not align with their terms of service; enable at your own discretion.",
|
|
3015
|
-
defaultCapabilities: defaultCapabilities4,
|
|
3016
|
-
async probeModels(_target, _ctx) {
|
|
3017
|
-
return listCatalogModelsForRuntime("anthropic-max").map((model) => model.id);
|
|
3018
|
-
},
|
|
3019
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3020
|
-
return synthesizeCatalogBackedModel({
|
|
3021
|
-
target,
|
|
3022
|
-
wireModelId,
|
|
3023
|
-
kb,
|
|
3024
|
-
defaultCapabilities: defaultCapabilities4,
|
|
3025
|
-
runtimeId: "anthropic-max",
|
|
3026
|
-
api: "anthropic-messages",
|
|
3027
|
-
provider: "anthropic",
|
|
3028
|
-
defaultBaseUrl: "https://api.anthropic.com"
|
|
3029
|
-
});
|
|
3030
|
-
}
|
|
3031
|
-
};
|
|
3032
|
-
var anthropic_max_default = anthropicMaxRuntime;
|
|
3033
|
-
|
|
3034
|
-
// src/domains/providers/runtimes/cloud/bedrock.ts
|
|
3035
|
-
init_esm_shims();
|
|
3036
|
-
|
|
3037
|
-
// src/domains/providers/runtimes/protocol/bedrock.ts
|
|
3038
|
-
init_esm_shims();
|
|
3039
|
-
function synthesizeBedrockModel(input) {
|
|
3040
|
-
return synthesizeCatalogBackedModel({
|
|
3041
|
-
target: input.target,
|
|
3042
|
-
wireModelId: input.wireModelId,
|
|
3043
|
-
kb: input.kb,
|
|
3044
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3045
|
-
runtimeId: "bedrock",
|
|
3046
|
-
api: "bedrock-converse-stream",
|
|
3047
|
-
provider: "amazon-bedrock",
|
|
3048
|
-
defaultBaseUrl: ""
|
|
3049
|
-
});
|
|
3050
|
-
}
|
|
3051
|
-
|
|
3052
|
-
// src/domains/providers/runtimes/cloud/bedrock.ts
|
|
3053
|
-
var defaultCapabilities5 = {
|
|
3054
|
-
chat: true,
|
|
3055
|
-
tools: true,
|
|
3056
|
-
toolCallFormat: "anthropic",
|
|
3057
|
-
reasoning: true,
|
|
3058
|
-
thinkingFormat: "anthropic-extended",
|
|
3059
|
-
vision: false,
|
|
3060
|
-
audio: false,
|
|
3061
|
-
embeddings: false,
|
|
3062
|
-
rerank: false,
|
|
3063
|
-
fim: false,
|
|
3064
|
-
contextWindow: 2e5,
|
|
3065
|
-
maxTokens: 8192
|
|
3066
|
-
};
|
|
3067
|
-
var bedrockRuntime = {
|
|
3068
|
-
id: "bedrock",
|
|
3069
|
-
displayName: "Amazon Bedrock",
|
|
3070
|
-
kind: "http",
|
|
3071
|
-
tier: "cloud",
|
|
3072
|
-
apiFamily: "bedrock-converse-stream",
|
|
3073
|
-
auth: "aws-sdk",
|
|
3074
|
-
defaultCapabilities: defaultCapabilities5,
|
|
3075
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3076
|
-
return synthesizeBedrockModel({ target, wireModelId, kb, defaultCapabilities: defaultCapabilities5 });
|
|
3077
|
-
}
|
|
3078
|
-
};
|
|
3079
|
-
var bedrock_default = bedrockRuntime;
|
|
3080
|
-
|
|
3081
|
-
// src/domains/providers/runtimes/cloud/deepseek.ts
|
|
3082
|
-
init_esm_shims();
|
|
3083
|
-
var defaultCapabilities6 = {
|
|
3084
|
-
chat: true,
|
|
3085
|
-
tools: true,
|
|
3086
|
-
toolCallFormat: "openai",
|
|
3087
|
-
reasoning: true,
|
|
3088
|
-
thinkingFormat: "deepseek-r1",
|
|
3089
|
-
vision: false,
|
|
3090
|
-
audio: false,
|
|
3091
|
-
embeddings: false,
|
|
3092
|
-
rerank: false,
|
|
3093
|
-
fim: false,
|
|
3094
|
-
contextWindow: 128e3,
|
|
3095
|
-
maxTokens: 65536
|
|
3096
|
-
};
|
|
3097
|
-
var deepseekRuntime = {
|
|
3098
|
-
id: "deepseek",
|
|
3099
|
-
displayName: "DeepSeek",
|
|
3100
|
-
kind: "http",
|
|
3101
|
-
tier: "cloud",
|
|
3102
|
-
apiFamily: "openai-completions",
|
|
3103
|
-
auth: "api-key",
|
|
3104
|
-
credentialsEnvVar: "DEEPSEEK_API_KEY",
|
|
3105
|
-
defaultCapabilities: defaultCapabilities6,
|
|
3106
|
-
async probeModels(_target, _ctx) {
|
|
3107
|
-
return listCatalogModelsForRuntime("deepseek").map((model) => model.id);
|
|
3108
|
-
},
|
|
3109
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3110
|
-
return synthesizeCatalogBackedModel({
|
|
3111
|
-
target,
|
|
3112
|
-
wireModelId,
|
|
3113
|
-
kb,
|
|
3114
|
-
defaultCapabilities: defaultCapabilities6,
|
|
3115
|
-
runtimeId: "deepseek",
|
|
3116
|
-
api: "openai-completions",
|
|
3117
|
-
provider: "deepseek",
|
|
3118
|
-
defaultBaseUrl: "https://api.deepseek.com"
|
|
3119
|
-
});
|
|
3120
|
-
}
|
|
3121
|
-
};
|
|
3122
|
-
var deepseek_default = deepseekRuntime;
|
|
3123
|
-
|
|
3124
|
-
// src/domains/providers/runtimes/cloud/google.ts
|
|
3125
|
-
init_esm_shims();
|
|
3126
|
-
|
|
3127
|
-
// src/domains/providers/runtimes/protocol/google.ts
|
|
3128
|
-
init_esm_shims();
|
|
3129
|
-
function synthesizeGoogleModel(input) {
|
|
3130
|
-
return synthesizeCatalogBackedModel({
|
|
3131
|
-
target: input.target,
|
|
3132
|
-
wireModelId: input.wireModelId,
|
|
3133
|
-
kb: input.kb,
|
|
3134
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3135
|
-
runtimeId: "google",
|
|
3136
|
-
api: "google-generative-ai",
|
|
3137
|
-
provider: "google",
|
|
3138
|
-
defaultBaseUrl: input.defaultBaseUrl
|
|
3139
|
-
});
|
|
3140
|
-
}
|
|
3141
|
-
|
|
3142
|
-
// src/domains/providers/runtimes/cloud/google.ts
|
|
3143
|
-
var defaultCapabilities7 = {
|
|
3144
|
-
chat: true,
|
|
3145
|
-
tools: true,
|
|
3146
|
-
toolCallFormat: "openai",
|
|
3147
|
-
reasoning: true,
|
|
3148
|
-
vision: true,
|
|
3149
|
-
audio: false,
|
|
3150
|
-
embeddings: false,
|
|
3151
|
-
rerank: false,
|
|
3152
|
-
fim: false,
|
|
3153
|
-
contextWindow: 2e6,
|
|
3154
|
-
maxTokens: 8192
|
|
3155
|
-
};
|
|
3156
|
-
var googleRuntime = {
|
|
3157
|
-
id: "google",
|
|
3158
|
-
displayName: "Google Generative AI",
|
|
3159
|
-
kind: "http",
|
|
3160
|
-
tier: "cloud",
|
|
3161
|
-
apiFamily: "google-generative-ai",
|
|
3162
|
-
auth: "api-key",
|
|
3163
|
-
credentialsEnvVar: "GOOGLE_API_KEY",
|
|
3164
|
-
defaultCapabilities: defaultCapabilities7,
|
|
3165
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3166
|
-
return synthesizeGoogleModel({
|
|
3167
|
-
target,
|
|
3168
|
-
wireModelId,
|
|
3169
|
-
kb,
|
|
3170
|
-
defaultCapabilities: defaultCapabilities7,
|
|
3171
|
-
defaultBaseUrl: "https://generativelanguage.googleapis.com/v1beta"
|
|
3172
|
-
});
|
|
3173
|
-
}
|
|
3174
|
-
};
|
|
3175
|
-
var google_default = googleRuntime;
|
|
3176
|
-
|
|
3177
|
-
// src/domains/providers/runtimes/cloud/groq.ts
|
|
3178
|
-
init_esm_shims();
|
|
3179
|
-
var defaultCapabilities8 = {
|
|
3180
|
-
chat: true,
|
|
3181
|
-
tools: true,
|
|
3182
|
-
toolCallFormat: "openai",
|
|
3183
|
-
reasoning: false,
|
|
3184
|
-
vision: false,
|
|
3185
|
-
audio: false,
|
|
3186
|
-
embeddings: false,
|
|
3187
|
-
rerank: false,
|
|
3188
|
-
fim: false,
|
|
3189
|
-
contextWindow: 128e3,
|
|
3190
|
-
maxTokens: 8192
|
|
3191
|
-
};
|
|
3192
|
-
var groqRuntime = {
|
|
3193
|
-
id: "groq",
|
|
3194
|
-
displayName: "Groq",
|
|
3195
|
-
kind: "http",
|
|
3196
|
-
tier: "cloud",
|
|
3197
|
-
apiFamily: "openai-completions",
|
|
3198
|
-
auth: "api-key",
|
|
3199
|
-
credentialsEnvVar: "GROQ_API_KEY",
|
|
3200
|
-
defaultCapabilities: defaultCapabilities8,
|
|
3201
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3202
|
-
return synthesizeCatalogBackedModel({
|
|
3203
|
-
target,
|
|
3204
|
-
wireModelId,
|
|
3205
|
-
kb,
|
|
3206
|
-
defaultCapabilities: defaultCapabilities8,
|
|
3207
|
-
runtimeId: "groq",
|
|
3208
|
-
api: "openai-completions",
|
|
3209
|
-
provider: "groq",
|
|
3210
|
-
defaultBaseUrl: "https://api.groq.com/openai/v1"
|
|
3211
|
-
});
|
|
3212
|
-
}
|
|
3213
|
-
};
|
|
3214
|
-
var groq_default = groqRuntime;
|
|
3215
|
-
|
|
3216
|
-
// src/domains/providers/runtimes/cloud/mistral.ts
|
|
3217
|
-
init_esm_shims();
|
|
3218
|
-
var defaultCapabilities9 = {
|
|
3219
|
-
chat: true,
|
|
3220
|
-
tools: true,
|
|
3221
|
-
toolCallFormat: "mistral",
|
|
3222
|
-
reasoning: false,
|
|
3223
|
-
vision: false,
|
|
3224
|
-
audio: false,
|
|
3225
|
-
embeddings: false,
|
|
3226
|
-
rerank: false,
|
|
3227
|
-
fim: false,
|
|
3228
|
-
contextWindow: 128e3,
|
|
3229
|
-
maxTokens: 8192
|
|
3230
|
-
};
|
|
3231
|
-
var mistralRuntime = {
|
|
3232
|
-
id: "mistral",
|
|
3233
|
-
displayName: "Mistral AI",
|
|
3234
|
-
kind: "http",
|
|
3235
|
-
tier: "cloud",
|
|
3236
|
-
apiFamily: "mistral-conversations",
|
|
3237
|
-
auth: "api-key",
|
|
3238
|
-
credentialsEnvVar: "MISTRAL_API_KEY",
|
|
3239
|
-
defaultCapabilities: defaultCapabilities9,
|
|
3240
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3241
|
-
return synthesizeCatalogBackedModel({
|
|
3242
|
-
target,
|
|
3243
|
-
wireModelId,
|
|
3244
|
-
kb,
|
|
3245
|
-
defaultCapabilities: defaultCapabilities9,
|
|
3246
|
-
runtimeId: "mistral",
|
|
3247
|
-
api: "mistral-conversations",
|
|
3248
|
-
provider: "mistral",
|
|
3249
|
-
defaultBaseUrl: "https://api.mistral.ai"
|
|
3250
|
-
});
|
|
3251
|
-
}
|
|
3252
|
-
};
|
|
3253
|
-
var mistral_default = mistralRuntime;
|
|
3254
|
-
|
|
3255
|
-
// src/domains/providers/runtimes/cloud/openai.ts
|
|
3256
|
-
init_esm_shims();
|
|
3257
|
-
var defaultCapabilities10 = {
|
|
3258
|
-
chat: true,
|
|
3259
|
-
tools: true,
|
|
3260
|
-
toolCallFormat: "openai",
|
|
3261
|
-
reasoning: true,
|
|
3262
|
-
vision: true,
|
|
3263
|
-
audio: false,
|
|
3264
|
-
embeddings: false,
|
|
3265
|
-
rerank: false,
|
|
3266
|
-
fim: false,
|
|
3267
|
-
contextWindow: 272e3,
|
|
3268
|
-
maxTokens: 16384
|
|
3269
|
-
};
|
|
3270
|
-
var openaiRuntime = {
|
|
3271
|
-
id: "openai",
|
|
3272
|
-
displayName: "OpenAI",
|
|
3273
|
-
kind: "http",
|
|
3274
|
-
tier: "cloud",
|
|
3275
|
-
apiFamily: "openai-responses",
|
|
3276
|
-
auth: "api-key",
|
|
3277
|
-
credentialsEnvVar: "OPENAI_API_KEY",
|
|
3278
|
-
defaultCapabilities: defaultCapabilities10,
|
|
3279
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3280
|
-
return synthesizeCatalogBackedModel({
|
|
3281
|
-
target,
|
|
3282
|
-
wireModelId,
|
|
3283
|
-
kb,
|
|
3284
|
-
defaultCapabilities: defaultCapabilities10,
|
|
3285
|
-
runtimeId: "openai",
|
|
3286
|
-
api: "openai-responses",
|
|
3287
|
-
provider: "openai",
|
|
3288
|
-
defaultBaseUrl: "https://api.openai.com/v1"
|
|
3289
|
-
});
|
|
3290
|
-
}
|
|
3291
|
-
};
|
|
3292
|
-
var openai_default = openaiRuntime;
|
|
3293
|
-
|
|
3294
|
-
// src/domains/providers/runtimes/cloud/openai-codex.ts
|
|
3295
|
-
init_esm_shims();
|
|
3296
|
-
var defaultCapabilities11 = {
|
|
3297
|
-
chat: true,
|
|
3298
|
-
tools: true,
|
|
3299
|
-
toolCallFormat: "openai",
|
|
3300
|
-
reasoning: true,
|
|
3301
|
-
thinkingFormat: "openai-codex",
|
|
3302
|
-
vision: true,
|
|
3303
|
-
audio: false,
|
|
3304
|
-
embeddings: false,
|
|
3305
|
-
rerank: false,
|
|
3306
|
-
fim: false,
|
|
3307
|
-
contextWindow: 272e3,
|
|
3308
|
-
maxTokens: 16384
|
|
3309
|
-
};
|
|
3310
|
-
var openaiCodexRuntime = {
|
|
3311
|
-
id: "openai-codex",
|
|
3312
|
-
displayName: "OpenAI Codex",
|
|
3313
|
-
kind: "http",
|
|
3314
|
-
tier: "cloud",
|
|
3315
|
-
apiFamily: "openai-codex-responses",
|
|
3316
|
-
auth: "oauth",
|
|
3317
|
-
defaultCapabilities: defaultCapabilities11,
|
|
3318
|
-
async probeModels(_target, _ctx) {
|
|
3319
|
-
return listCatalogModelsForRuntime("openai-codex").map((model) => model.id);
|
|
3320
|
-
},
|
|
3321
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3322
|
-
return synthesizeCatalogBackedModel({
|
|
3323
|
-
target,
|
|
3324
|
-
wireModelId,
|
|
3325
|
-
kb,
|
|
3326
|
-
defaultCapabilities: defaultCapabilities11,
|
|
3327
|
-
runtimeId: "openai-codex",
|
|
3328
|
-
api: "openai-codex-responses",
|
|
3329
|
-
provider: "openai-codex",
|
|
3330
|
-
defaultBaseUrl: "https://chatgpt.com/backend-api"
|
|
3331
|
-
});
|
|
3332
|
-
}
|
|
3333
|
-
};
|
|
3334
|
-
var openai_codex_default = openaiCodexRuntime;
|
|
3335
|
-
|
|
3336
|
-
// src/domains/providers/runtimes/cloud/openrouter.ts
|
|
3337
|
-
init_esm_shims();
|
|
3338
|
-
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
3339
|
-
var OPENROUTER_HEADERS = {
|
|
3340
|
-
"HTTP-Referer": "https://github.com/iowarp/clio-coder",
|
|
3341
|
-
"X-OpenRouter-Title": "Clio Coder"
|
|
3342
|
-
};
|
|
3343
|
-
var defaultCapabilities12 = {
|
|
3344
|
-
chat: true,
|
|
3345
|
-
tools: true,
|
|
3346
|
-
toolCallFormat: "openai",
|
|
3347
|
-
reasoning: false,
|
|
3348
|
-
thinkingFormat: "openrouter",
|
|
3349
|
-
vision: false,
|
|
3350
|
-
audio: false,
|
|
3351
|
-
embeddings: false,
|
|
3352
|
-
rerank: false,
|
|
3353
|
-
fim: false,
|
|
3354
|
-
contextWindow: 128e3,
|
|
3355
|
-
maxTokens: 8192
|
|
3356
|
-
};
|
|
3357
|
-
function trimTrailingSlash2(value) {
|
|
3358
|
-
return value.endsWith("/") && value.length > 1 ? value.slice(0, -1) : value;
|
|
3359
|
-
}
|
|
3360
|
-
function targetBaseUrl2(target) {
|
|
3361
|
-
return trimTrailingSlash2(target.url ?? OPENROUTER_BASE_URL);
|
|
3362
|
-
}
|
|
3363
|
-
function modelsUrl(target) {
|
|
3364
|
-
return `${targetBaseUrl2(target)}/models`;
|
|
3365
|
-
}
|
|
3366
|
-
function probeHeaders(target, ctx) {
|
|
3367
|
-
const headers = { ...OPENROUTER_HEADERS, ...target.auth?.headers ?? {} };
|
|
3368
|
-
const envName = target.auth?.apiKeyEnvVar ?? "OPENROUTER_API_KEY";
|
|
3369
|
-
if (ctx.credentialsPresent.has(envName)) {
|
|
3370
|
-
const key = process.env[envName]?.trim();
|
|
3371
|
-
if (key) headers.authorization = `Bearer ${key}`;
|
|
3372
|
-
}
|
|
3373
|
-
return headers;
|
|
3374
|
-
}
|
|
3375
|
-
async function fetchModels(target, ctx) {
|
|
3376
|
-
const opts = {
|
|
3377
|
-
url: modelsUrl(target),
|
|
3378
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
3379
|
-
headers: probeHeaders(target, ctx)
|
|
3380
|
-
};
|
|
3381
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
3382
|
-
if (!result.ok) return result;
|
|
3383
|
-
const models = (result.data?.data ?? []).map((row) => typeof row?.id === "string" ? row.id : null).filter((id) => id !== null);
|
|
3384
|
-
const out = { ok: true, models };
|
|
3385
|
-
if (result.latencyMs !== void 0) out.latencyMs = result.latencyMs;
|
|
3386
|
-
const configured = target.defaultModel?.trim();
|
|
3387
|
-
if (configured && models.length > 0 && !models.includes(configured)) {
|
|
3388
|
-
out.ok = false;
|
|
3389
|
-
out.error = `configured model '${configured}' was not returned by OpenRouter`;
|
|
3390
|
-
}
|
|
3391
|
-
return out;
|
|
3392
|
-
}
|
|
3393
|
-
var openrouterRuntime = {
|
|
3394
|
-
id: "openrouter",
|
|
3395
|
-
displayName: "OpenRouter",
|
|
3396
|
-
kind: "http",
|
|
3397
|
-
tier: "cloud",
|
|
3398
|
-
apiFamily: "openai-completions",
|
|
3399
|
-
auth: "api-key",
|
|
3400
|
-
credentialsEnvVar: "OPENROUTER_API_KEY",
|
|
3401
|
-
defaultCapabilities: defaultCapabilities12,
|
|
3402
|
-
probe(target, ctx) {
|
|
3403
|
-
return fetchModels(target, ctx);
|
|
3404
|
-
},
|
|
3405
|
-
async probeModels(target, ctx) {
|
|
3406
|
-
const result = await fetchModels(target, ctx);
|
|
3407
|
-
return result.ok && result.models ? [...result.models] : [];
|
|
3408
|
-
},
|
|
3409
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3410
|
-
return synthesizeCatalogBackedModel({
|
|
3411
|
-
target,
|
|
3412
|
-
wireModelId,
|
|
3413
|
-
kb,
|
|
3414
|
-
defaultCapabilities: defaultCapabilities12,
|
|
3415
|
-
runtimeId: "openrouter",
|
|
3416
|
-
api: "openai-completions",
|
|
3417
|
-
provider: "openrouter",
|
|
3418
|
-
defaultBaseUrl: OPENROUTER_BASE_URL,
|
|
3419
|
-
defaultHeaders: OPENROUTER_HEADERS
|
|
3420
|
-
});
|
|
3421
|
-
}
|
|
3422
|
-
};
|
|
3423
|
-
var openrouter_default = openrouterRuntime;
|
|
3424
|
-
|
|
3425
|
-
// src/domains/providers/runtimes/local-native/lemonade-anthropic.ts
|
|
3426
|
-
init_esm_shims();
|
|
3427
|
-
|
|
3428
|
-
// src/domains/providers/runtimes/protocol/anthropic-compat.ts
|
|
3429
|
-
init_esm_shims();
|
|
3430
|
-
function synthesizeAnthropicCompatModel(input) {
|
|
3431
|
-
return synthLocalModel({
|
|
3432
|
-
target: input.target,
|
|
3433
|
-
wireModelId: input.wireModelId,
|
|
3434
|
-
kb: input.kb,
|
|
3435
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3436
|
-
apiFamily: "anthropic-messages",
|
|
3437
|
-
provider: input.provider,
|
|
3438
|
-
baseUrlForTarget: input.baseUrlForTarget ?? withAsIs
|
|
3439
|
-
});
|
|
3440
|
-
}
|
|
3441
|
-
function makeAnthropicCompatRuntime(spec) {
|
|
3442
|
-
const messagesPath = spec.messagesPath ?? "/v1/messages";
|
|
3443
|
-
const modelsPath = spec.modelsPath ?? "/v1/models";
|
|
3444
|
-
return {
|
|
3445
|
-
id: spec.id,
|
|
3446
|
-
displayName: spec.displayName,
|
|
3447
|
-
kind: "http",
|
|
3448
|
-
tier: spec.tier,
|
|
3449
|
-
apiFamily: "anthropic-messages",
|
|
3450
|
-
auth: spec.auth,
|
|
3451
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
3452
|
-
...spec.hidden === true ? { hidden: true } : {},
|
|
3453
|
-
async probe(target, ctx) {
|
|
3454
|
-
const base = targetBaseUrl(target);
|
|
3455
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3456
|
-
return probeUrl(`${base}${messagesPath}`, ctx, "HEAD");
|
|
3457
|
-
},
|
|
3458
|
-
async probeModels(target, ctx) {
|
|
3459
|
-
const base = targetBaseUrl(target);
|
|
3460
|
-
if (!base) return [];
|
|
3461
|
-
return probeOpenAIModels(base, ctx, modelsPath);
|
|
3462
|
-
},
|
|
3463
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3464
|
-
return synthesizeAnthropicCompatModel({
|
|
3465
|
-
target,
|
|
3466
|
-
wireModelId,
|
|
3467
|
-
kb,
|
|
3468
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
3469
|
-
provider: spec.provider
|
|
3470
|
-
});
|
|
3471
|
-
}
|
|
3472
|
-
};
|
|
3473
|
-
}
|
|
3474
|
-
var defaultCapabilities13 = {
|
|
3475
|
-
chat: true,
|
|
3476
|
-
tools: true,
|
|
3477
|
-
toolCallFormat: "anthropic",
|
|
3478
|
-
reasoning: false,
|
|
3479
|
-
vision: false,
|
|
3480
|
-
audio: false,
|
|
3481
|
-
embeddings: false,
|
|
3482
|
-
rerank: false,
|
|
3483
|
-
fim: false,
|
|
3484
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3485
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3486
|
-
};
|
|
3487
|
-
var anthropic_compat_default = makeAnthropicCompatRuntime({
|
|
3488
|
-
id: "anthropic-compat",
|
|
3489
|
-
displayName: "Generic Anthropic-compatible",
|
|
3490
|
-
provider: "anthropic-compat",
|
|
3491
|
-
auth: "api-key",
|
|
3492
|
-
tier: "protocol",
|
|
3493
|
-
defaultCapabilities: defaultCapabilities13
|
|
3494
|
-
});
|
|
3495
|
-
|
|
3496
|
-
// src/domains/providers/runtimes/local-native/lemonade-anthropic.ts
|
|
3497
|
-
var defaultCapabilities14 = {
|
|
3498
|
-
chat: true,
|
|
3499
|
-
tools: true,
|
|
3500
|
-
toolCallFormat: "anthropic",
|
|
3501
|
-
reasoning: false,
|
|
3502
|
-
vision: false,
|
|
3503
|
-
audio: false,
|
|
3504
|
-
embeddings: false,
|
|
3505
|
-
rerank: false,
|
|
3506
|
-
fim: false,
|
|
3507
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3508
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3509
|
-
};
|
|
3510
|
-
var lemonade_anthropic_default = makeAnthropicCompatRuntime({
|
|
3511
|
-
id: "lemonade-anthropic",
|
|
3512
|
-
displayName: "Lemonade (Anthropic-compat)",
|
|
3513
|
-
provider: "lemonade",
|
|
3514
|
-
auth: "api-key",
|
|
3515
|
-
tier: "local-native",
|
|
3516
|
-
defaultCapabilities: defaultCapabilities14,
|
|
3517
|
-
hidden: true
|
|
3518
|
-
});
|
|
3519
|
-
|
|
3520
|
-
// src/domains/providers/runtimes/local-native/lemonade-openai.ts
|
|
3521
|
-
init_esm_shims();
|
|
3522
|
-
var defaultCapabilities15 = {
|
|
3523
|
-
chat: true,
|
|
3524
|
-
tools: true,
|
|
3525
|
-
toolCallFormat: "openai",
|
|
3526
|
-
reasoning: false,
|
|
3527
|
-
vision: false,
|
|
3528
|
-
audio: false,
|
|
3529
|
-
embeddings: false,
|
|
3530
|
-
rerank: false,
|
|
3531
|
-
fim: false,
|
|
3532
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3533
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3534
|
-
};
|
|
3535
|
-
var lemonade_openai_default = makeOpenAICompatRuntime({
|
|
3536
|
-
id: "lemonade",
|
|
3537
|
-
displayName: "Lemonade (OpenAI-compat)",
|
|
3538
|
-
provider: "lemonade",
|
|
3539
|
-
auth: "api-key",
|
|
3540
|
-
tier: "local-native",
|
|
3541
|
-
defaultCapabilities: defaultCapabilities15
|
|
3542
|
-
});
|
|
3543
|
-
|
|
3544
|
-
// src/domains/providers/runtimes/local-native/llamacpp.ts
|
|
3545
|
-
init_esm_shims();
|
|
3546
|
-
var defaultCapabilities16 = {
|
|
3547
|
-
chat: true,
|
|
3548
|
-
tools: true,
|
|
3549
|
-
toolCallFormat: "openai",
|
|
3550
|
-
structuredOutputs: "json-schema",
|
|
3551
|
-
reasoning: false,
|
|
3552
|
-
vision: false,
|
|
3553
|
-
audio: false,
|
|
3554
|
-
embeddings: false,
|
|
3555
|
-
rerank: false,
|
|
3556
|
-
fim: false,
|
|
3557
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3558
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3559
|
-
};
|
|
3560
|
-
function targetUrl(target) {
|
|
3561
|
-
return targetRootUrl(target);
|
|
3562
|
-
}
|
|
3563
|
-
var llamacppRuntime = {
|
|
3564
|
-
id: "llamacpp",
|
|
3565
|
-
displayName: "llama.cpp",
|
|
3566
|
-
kind: "http",
|
|
3567
|
-
tier: "local-native",
|
|
3568
|
-
apiFamily: "openai-completions",
|
|
3569
|
-
auth: "api-key",
|
|
3570
|
-
defaultCapabilities: defaultCapabilities16,
|
|
3571
|
-
async probe(target, ctx) {
|
|
3572
|
-
const base = targetUrl(target);
|
|
3573
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3574
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3575
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3576
|
-
if (!health.ok) return health;
|
|
3577
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3578
|
-
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
3579
|
-
const catalog = await probeOpenAIModelCatalog(base, ctx);
|
|
3580
|
-
const result = { ok: true };
|
|
3581
|
-
if (catalog.models.length > 0) result.models = catalog.models;
|
|
3582
|
-
if (Object.keys(catalog.modelCapabilities).length > 0) result.modelCapabilities = catalog.modelCapabilities;
|
|
3583
|
-
if (Object.keys(catalog.modelStates).length > 0) result.modelStates = catalog.modelStates;
|
|
3584
|
-
if (typeof health.latencyMs === "number") result.latencyMs = health.latencyMs;
|
|
3585
|
-
const discoveredCapabilities = {
|
|
3586
|
-
...props.discoveredCapabilities ?? {},
|
|
3587
|
-
...status.discoveredCapabilities ?? {}
|
|
3588
|
-
};
|
|
3589
|
-
if (Object.keys(discoveredCapabilities).length > 0) {
|
|
3590
|
-
result.discoveredCapabilities = discoveredCapabilities;
|
|
3591
|
-
if (status.modelId) result.capabilityModelId = status.modelId;
|
|
3592
|
-
}
|
|
3593
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
3594
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3595
|
-
const notes = [...status.notes ?? [], ...note ? [note] : []];
|
|
3596
|
-
if (notes.length > 0) result.notes = notes;
|
|
3597
|
-
return result;
|
|
3598
|
-
},
|
|
3599
|
-
async probeModels(target, ctx) {
|
|
3600
|
-
const base = targetUrl(target);
|
|
3601
|
-
if (!base) return [];
|
|
3602
|
-
return probeOpenAIModels(base, ctx);
|
|
3603
|
-
},
|
|
3604
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3605
|
-
return synthLocalModel({
|
|
3606
|
-
target,
|
|
3607
|
-
wireModelId,
|
|
3608
|
-
kb,
|
|
3609
|
-
defaultCapabilities: defaultCapabilities16,
|
|
3610
|
-
apiFamily: "openai-completions",
|
|
3611
|
-
provider: "llamacpp",
|
|
3612
|
-
baseUrlForTarget: withV1
|
|
3613
|
-
});
|
|
3614
|
-
}
|
|
3615
|
-
};
|
|
3616
|
-
var llamacpp_default = llamacppRuntime;
|
|
3617
|
-
|
|
3618
|
-
// src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts
|
|
3619
|
-
init_esm_shims();
|
|
3620
|
-
var defaultCapabilities17 = {
|
|
3621
|
-
chat: true,
|
|
3622
|
-
tools: true,
|
|
3623
|
-
toolCallFormat: "anthropic",
|
|
3624
|
-
reasoning: false,
|
|
3625
|
-
vision: false,
|
|
3626
|
-
audio: false,
|
|
3627
|
-
embeddings: false,
|
|
3628
|
-
rerank: false,
|
|
3629
|
-
fim: true,
|
|
3630
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3631
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3632
|
-
};
|
|
3633
|
-
function url(target) {
|
|
3634
|
-
return target.url ? stripTrailingSlash(target.url) : null;
|
|
3635
|
-
}
|
|
3636
|
-
var llamacppAnthropicRuntime = {
|
|
3637
|
-
id: "llamacpp-anthropic",
|
|
3638
|
-
displayName: "llama.cpp (Anthropic-compat)",
|
|
3639
|
-
kind: "http",
|
|
3640
|
-
tier: "local-native",
|
|
3641
|
-
apiFamily: "anthropic-messages",
|
|
3642
|
-
auth: "api-key",
|
|
3643
|
-
defaultCapabilities: defaultCapabilities17,
|
|
3644
|
-
hidden: true,
|
|
3645
|
-
async probe(target, ctx) {
|
|
3646
|
-
const base = url(target);
|
|
3647
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3648
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3649
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3650
|
-
if (!health.ok) return health;
|
|
3651
|
-
const headOpts = {
|
|
3652
|
-
url: `${base}/v1/messages`,
|
|
3653
|
-
method: "HEAD",
|
|
3654
|
-
timeoutMs: ctx.httpTimeoutMs
|
|
3655
|
-
};
|
|
3656
|
-
const head = await (ctx.signal ? probeHttp({ ...headOpts, signal: ctx.signal }) : probeHttp(headOpts));
|
|
3657
|
-
if (!head.ok) return head;
|
|
3658
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3659
|
-
const enriched = { ...head };
|
|
3660
|
-
if (props.discoveredCapabilities) enriched.discoveredCapabilities = props.discoveredCapabilities;
|
|
3661
|
-
if (props.serverVersion) enriched.serverVersion = props.serverVersion;
|
|
3662
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3663
|
-
if (note) enriched.notes = [note];
|
|
3664
|
-
return enriched;
|
|
3665
|
-
},
|
|
3666
|
-
async probeModels(target, ctx) {
|
|
3667
|
-
const base = url(target);
|
|
3668
|
-
if (!base) return [];
|
|
3669
|
-
const opts = { url: `${base}/v1/models`, timeoutMs: ctx.httpTimeoutMs };
|
|
3670
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
3671
|
-
if (!result.ok || !result.data?.data) return [];
|
|
3672
|
-
return result.data.data.map((row) => typeof row?.id === "string" ? row.id : null).filter((id) => id !== null);
|
|
3673
|
-
},
|
|
3674
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3675
|
-
return synthLocalModel({
|
|
3676
|
-
target,
|
|
3677
|
-
wireModelId,
|
|
3678
|
-
kb,
|
|
3679
|
-
defaultCapabilities: defaultCapabilities17,
|
|
3680
|
-
apiFamily: "anthropic-messages",
|
|
3681
|
-
provider: "llamacpp",
|
|
3682
|
-
baseUrlForTarget: withAsIs
|
|
3683
|
-
});
|
|
3684
|
-
}
|
|
3685
|
-
};
|
|
3686
|
-
var llamacpp_anthropic_default = llamacppAnthropicRuntime;
|
|
3687
|
-
|
|
3688
|
-
// src/domains/providers/runtimes/local-native/llamacpp-completion.ts
|
|
3689
|
-
init_esm_shims();
|
|
3690
|
-
var defaultCapabilities18 = {
|
|
3691
|
-
chat: false,
|
|
3692
|
-
tools: false,
|
|
3693
|
-
reasoning: false,
|
|
3694
|
-
vision: false,
|
|
3695
|
-
audio: false,
|
|
3696
|
-
embeddings: false,
|
|
3697
|
-
rerank: false,
|
|
3698
|
-
fim: true,
|
|
3699
|
-
structuredOutputs: "gbnf",
|
|
3700
|
-
contextWindow: 8192,
|
|
3701
|
-
maxTokens: 4096
|
|
3702
|
-
};
|
|
3703
|
-
function targetUrl2(target) {
|
|
3704
|
-
return targetRootUrl(target);
|
|
3705
|
-
}
|
|
3706
|
-
function parseChunk(raw) {
|
|
3707
|
-
const chunk = {
|
|
3708
|
-
content: typeof raw.content === "string" ? raw.content : "",
|
|
3709
|
-
stop: raw.stop === true
|
|
3710
|
-
};
|
|
3711
|
-
if (raw.stop_type === "eos" || raw.stop_type === "limit" || raw.stop_type === "word" || raw.stop_type === "none") {
|
|
3712
|
-
chunk.stop_type = raw.stop_type;
|
|
3713
|
-
}
|
|
3714
|
-
if (typeof raw.tokens_predicted === "number") chunk.tokens_predicted = raw.tokens_predicted;
|
|
3715
|
-
if (typeof raw.tokens_evaluated === "number") chunk.tokens_evaluated = raw.tokens_evaluated;
|
|
3716
|
-
return chunk;
|
|
3717
|
-
}
|
|
3718
|
-
async function* streamSse(body) {
|
|
3719
|
-
const reader = body.getReader();
|
|
3720
|
-
const decoder = new TextDecoder("utf-8");
|
|
3721
|
-
let buffered = "";
|
|
3722
|
-
let droppedFrames = 0;
|
|
3723
|
-
try {
|
|
3724
|
-
while (true) {
|
|
3725
|
-
const { done, value } = await reader.read();
|
|
3726
|
-
if (done) break;
|
|
3727
|
-
buffered += decoder.decode(value, { stream: true });
|
|
3728
|
-
let nl = buffered.indexOf("\n");
|
|
3729
|
-
while (nl !== -1) {
|
|
3730
|
-
const line = buffered.slice(0, nl).trimEnd();
|
|
3731
|
-
buffered = buffered.slice(nl + 1);
|
|
3732
|
-
nl = buffered.indexOf("\n");
|
|
3733
|
-
if (line.length === 0) continue;
|
|
3734
|
-
const payload = line.startsWith("data:") ? line.slice(5).trim() : line;
|
|
3735
|
-
if (payload.length === 0 || payload === "[DONE]") continue;
|
|
3736
|
-
try {
|
|
3737
|
-
const parsed = JSON.parse(payload);
|
|
3738
|
-
const chunk = parseChunk(parsed);
|
|
3739
|
-
yield chunk;
|
|
3740
|
-
if (chunk.stop) return;
|
|
3741
|
-
} catch {
|
|
3742
|
-
droppedFrames += 1;
|
|
3743
|
-
}
|
|
3744
|
-
}
|
|
3745
|
-
}
|
|
3746
|
-
const tail = buffered.trim();
|
|
3747
|
-
if (tail.length > 0) {
|
|
3748
|
-
const payload = tail.startsWith("data:") ? tail.slice(5).trim() : tail;
|
|
3749
|
-
if (payload !== "[DONE]") {
|
|
3750
|
-
try {
|
|
3751
|
-
const parsed = JSON.parse(payload);
|
|
3752
|
-
yield parseChunk(parsed);
|
|
3753
|
-
} catch {
|
|
3754
|
-
droppedFrames += 1;
|
|
3755
|
-
}
|
|
3756
|
-
}
|
|
3757
|
-
}
|
|
3758
|
-
} finally {
|
|
3759
|
-
if (droppedFrames > 0) {
|
|
3760
|
-
process.stderr.write(`[clio:llamacpp] dropped ${droppedFrames} malformed stream frame(s)
|
|
3761
|
-
`);
|
|
3762
|
-
}
|
|
3763
|
-
reader.releaseLock();
|
|
3764
|
-
}
|
|
3765
|
-
}
|
|
3766
|
-
async function postStream(url2, body, signal) {
|
|
3767
|
-
const init = {
|
|
3768
|
-
method: "POST",
|
|
3769
|
-
headers: { "content-type": "application/json", accept: "text/event-stream" },
|
|
3770
|
-
body: JSON.stringify(body)
|
|
3771
|
-
};
|
|
3772
|
-
if (signal) init.signal = signal;
|
|
3773
|
-
const response = await fetch(url2, init);
|
|
3774
|
-
if (!response.ok || !response.body) {
|
|
3775
|
-
throw new Error(`HTTP ${response.status}: ${response.statusText}`);
|
|
3776
|
-
}
|
|
3777
|
-
return response.body;
|
|
3778
|
-
}
|
|
3779
|
-
function buildCompleteBody(opts) {
|
|
3780
|
-
const body = { prompt: opts.prompt, stream: true };
|
|
3781
|
-
if (opts.n_predict !== void 0) body.n_predict = opts.n_predict;
|
|
3782
|
-
if (opts.stop && opts.stop.length > 0) body.stop = opts.stop;
|
|
3783
|
-
if (opts.grammar) body.grammar = opts.grammar;
|
|
3784
|
-
if (opts.json_schema) body.json_schema = opts.json_schema;
|
|
3785
|
-
if (opts.cache_prompt !== void 0) body.cache_prompt = opts.cache_prompt;
|
|
3786
|
-
return body;
|
|
3787
|
-
}
|
|
3788
|
-
function buildInfillBody(opts) {
|
|
3789
|
-
const body = buildCompleteBody(opts);
|
|
3790
|
-
body.input_prefix = opts.input_prefix;
|
|
3791
|
-
body.input_suffix = opts.input_suffix;
|
|
3792
|
-
if (opts.input_extra) body.input_extra = opts.input_extra;
|
|
3793
|
-
return body;
|
|
3794
|
-
}
|
|
3795
|
-
var llamacppCompletionRuntime = {
|
|
3796
|
-
id: "llamacpp-completion",
|
|
3797
|
-
displayName: "llama.cpp (completion / infill)",
|
|
3798
|
-
kind: "http",
|
|
3799
|
-
tier: "local-native",
|
|
3800
|
-
apiFamily: "openai-completions",
|
|
3801
|
-
auth: "api-key",
|
|
3802
|
-
defaultCapabilities: defaultCapabilities18,
|
|
3803
|
-
hidden: true,
|
|
3804
|
-
async probe(target, ctx) {
|
|
3805
|
-
const base = targetUrl2(target);
|
|
3806
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3807
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3808
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3809
|
-
if (!health.ok) return health;
|
|
3810
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3811
|
-
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
3812
|
-
const enriched = { ...health };
|
|
3813
|
-
const discoveredCapabilities = {
|
|
3814
|
-
...props.discoveredCapabilities ?? {},
|
|
3815
|
-
...status.discoveredCapabilities ?? {}
|
|
3816
|
-
};
|
|
3817
|
-
if (Object.keys(discoveredCapabilities).length > 0) {
|
|
3818
|
-
enriched.discoveredCapabilities = discoveredCapabilities;
|
|
3819
|
-
if (status.modelId) enriched.capabilityModelId = status.modelId;
|
|
3820
|
-
}
|
|
3821
|
-
if (props.serverVersion) enriched.serverVersion = props.serverVersion;
|
|
3822
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3823
|
-
const notes = [...status.notes ?? [], ...note ? [note] : []];
|
|
3824
|
-
if (notes.length > 0) enriched.notes = notes;
|
|
3825
|
-
return enriched;
|
|
3826
|
-
},
|
|
3827
|
-
async probeModels(target, ctx) {
|
|
3828
|
-
const base = targetUrl2(target);
|
|
3829
|
-
if (!base) return [];
|
|
3830
|
-
return probeOpenAIModels(base, ctx);
|
|
3831
|
-
},
|
|
3832
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3833
|
-
return synthLocalModel({
|
|
3834
|
-
target,
|
|
3835
|
-
wireModelId,
|
|
3836
|
-
kb,
|
|
3837
|
-
defaultCapabilities: defaultCapabilities18,
|
|
3838
|
-
apiFamily: "openai-completions",
|
|
3839
|
-
provider: "llamacpp",
|
|
3840
|
-
baseUrlForTarget: withV1
|
|
3841
|
-
});
|
|
3842
|
-
},
|
|
3843
|
-
async *complete(target, opts) {
|
|
3844
|
-
const base = targetUrl2(target);
|
|
3845
|
-
if (!base) throw new Error("target has no url");
|
|
3846
|
-
const body = await postStream(`${base}/completion`, buildCompleteBody(opts), opts.signal);
|
|
3847
|
-
for await (const chunk of streamSse(body)) yield chunk;
|
|
3848
|
-
},
|
|
3849
|
-
async *infill(target, opts) {
|
|
3850
|
-
const base = targetUrl2(target);
|
|
3851
|
-
if (!base) throw new Error("target has no url");
|
|
3852
|
-
const body = await postStream(`${base}/infill`, buildInfillBody(opts), opts.signal);
|
|
3853
|
-
for await (const chunk of streamSse(body)) yield chunk;
|
|
3854
|
-
}
|
|
3855
|
-
};
|
|
3856
|
-
var llamacpp_completion_default = llamacppCompletionRuntime;
|
|
3857
|
-
|
|
3858
|
-
// src/domains/providers/runtimes/local-native/llamacpp-embed.ts
|
|
3859
|
-
init_esm_shims();
|
|
3860
|
-
var defaultCapabilities19 = {
|
|
3861
|
-
chat: false,
|
|
3862
|
-
tools: false,
|
|
3863
|
-
reasoning: false,
|
|
3864
|
-
vision: false,
|
|
3865
|
-
audio: false,
|
|
3866
|
-
embeddings: true,
|
|
3867
|
-
rerank: false,
|
|
3868
|
-
fim: false,
|
|
3869
|
-
contextWindow: 8192,
|
|
3870
|
-
maxTokens: 0
|
|
3871
|
-
};
|
|
3872
|
-
function targetUrl3(target) {
|
|
3873
|
-
return targetRootUrl(target);
|
|
3874
|
-
}
|
|
3875
|
-
function meanPool(matrix) {
|
|
3876
|
-
if (matrix.length === 0) return [];
|
|
3877
|
-
const dim = matrix[0]?.length ?? 0;
|
|
3878
|
-
const sum = new Array(dim).fill(0);
|
|
3879
|
-
for (const row of matrix) {
|
|
3880
|
-
for (let i = 0; i < dim; i++) sum[i] = (sum[i] ?? 0) + (row[i] ?? 0);
|
|
3881
|
-
}
|
|
3882
|
-
return sum.map((v) => v / matrix.length);
|
|
3883
|
-
}
|
|
3884
|
-
function flattenNativeEmbedding(entry) {
|
|
3885
|
-
const value = entry.embedding;
|
|
3886
|
-
if (!value || value.length === 0) return [];
|
|
3887
|
-
if (Array.isArray(value[0])) return meanPool(value);
|
|
3888
|
-
return value;
|
|
3889
|
-
}
|
|
3890
|
-
async function postJson(url2, body, signal) {
|
|
3891
|
-
const init = {
|
|
3892
|
-
method: "POST",
|
|
3893
|
-
headers: { "content-type": "application/json" },
|
|
3894
|
-
body: JSON.stringify(body)
|
|
3895
|
-
};
|
|
3896
|
-
if (signal) init.signal = signal;
|
|
3897
|
-
const response = await fetch(url2, init);
|
|
3898
|
-
if (!response.ok) return { status: response.status, data: null };
|
|
3899
|
-
try {
|
|
3900
|
-
const data = await response.json();
|
|
3901
|
-
return { status: response.status, data };
|
|
3902
|
-
} catch {
|
|
3903
|
-
return { status: response.status, data: null };
|
|
3904
|
-
}
|
|
3905
|
-
}
|
|
3906
|
-
var llamacppEmbedRuntime = {
|
|
3907
|
-
id: "llamacpp-embed",
|
|
3908
|
-
displayName: "llama.cpp (embeddings)",
|
|
3909
|
-
kind: "http",
|
|
3910
|
-
tier: "local-native",
|
|
3911
|
-
apiFamily: "openai-completions",
|
|
3912
|
-
auth: "api-key",
|
|
3913
|
-
defaultCapabilities: defaultCapabilities19,
|
|
3914
|
-
hidden: true,
|
|
3915
|
-
async probe(target, ctx) {
|
|
3916
|
-
const base = targetUrl3(target);
|
|
3917
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3918
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3919
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3920
|
-
if (!health.ok) return health;
|
|
3921
|
-
const probeResponse = await fetch(`${base}/embedding`, {
|
|
3922
|
-
method: "POST",
|
|
3923
|
-
headers: { "content-type": "application/json" },
|
|
3924
|
-
body: JSON.stringify({ content: "probe" }),
|
|
3925
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
3926
|
-
}).catch((err) => {
|
|
3927
|
-
return new Response(null, { status: 599, statusText: String(err) });
|
|
3928
|
-
});
|
|
3929
|
-
if (!probeResponse.ok) {
|
|
3930
|
-
return { ok: false, error: `/embedding not available: HTTP ${probeResponse.status}` };
|
|
3931
|
-
}
|
|
3932
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3933
|
-
const result = { ok: true };
|
|
3934
|
-
if (health.latencyMs !== void 0) result.latencyMs = health.latencyMs;
|
|
3935
|
-
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
3936
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
3937
|
-
return result;
|
|
3938
|
-
},
|
|
3939
|
-
async probeModels(target, ctx) {
|
|
3940
|
-
const base = targetUrl3(target);
|
|
3941
|
-
if (!base) return [];
|
|
3942
|
-
return probeOpenAIModels(base, ctx);
|
|
3943
|
-
},
|
|
3944
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3945
|
-
return synthLocalModel({
|
|
3946
|
-
target,
|
|
3947
|
-
wireModelId,
|
|
3948
|
-
kb,
|
|
3949
|
-
defaultCapabilities: defaultCapabilities19,
|
|
3950
|
-
apiFamily: "openai-completions",
|
|
3951
|
-
provider: "llamacpp",
|
|
3952
|
-
baseUrlForTarget: withV1
|
|
3953
|
-
});
|
|
3954
|
-
},
|
|
3955
|
-
async embed(target, input, ctx) {
|
|
3956
|
-
const base = targetUrl3(target);
|
|
3957
|
-
if (!base) throw new Error("target has no url");
|
|
3958
|
-
const modelId = target.defaultModel ?? "default";
|
|
3959
|
-
const inputs = Array.isArray(input) ? input : [input];
|
|
3960
|
-
const oai = await postJson(
|
|
3961
|
-
`${base}/v1/embeddings`,
|
|
3962
|
-
{ input: inputs, model: modelId, encoding_format: "float" },
|
|
3963
|
-
ctx.signal
|
|
3964
|
-
);
|
|
3965
|
-
if (oai.data && Array.isArray(oai.data.data)) {
|
|
3966
|
-
const rows = oai.data.data;
|
|
3967
|
-
const sorted = [...rows].sort((a, b) => (a.index ?? 0) - (b.index ?? 0));
|
|
3968
|
-
const vectors2 = sorted.map((row) => row.embedding ?? []);
|
|
3969
|
-
const tokens = oai.data.usage?.total_tokens ?? oai.data.usage?.prompt_tokens ?? void 0;
|
|
3970
|
-
const dim = vectors2[0]?.length ?? 0;
|
|
3971
|
-
const result = {
|
|
3972
|
-
vectors: vectors2,
|
|
3973
|
-
model: oai.data.model ?? modelId,
|
|
3974
|
-
dimensions: dim
|
|
3975
|
-
};
|
|
3976
|
-
if (tokens !== void 0) result.tokensUsed = tokens;
|
|
3977
|
-
return result;
|
|
3978
|
-
}
|
|
3979
|
-
const probeOpts = { url: `${base}/embedding`, timeoutMs: ctx.httpTimeoutMs };
|
|
3980
|
-
const native = await (ctx.signal ? probeJson({
|
|
3981
|
-
...probeOpts,
|
|
3982
|
-
method: "POST",
|
|
3983
|
-
body: JSON.stringify({ content: inputs }),
|
|
3984
|
-
headers: { "content-type": "application/json" },
|
|
3985
|
-
signal: ctx.signal
|
|
3986
|
-
}) : probeJson({
|
|
3987
|
-
...probeOpts,
|
|
3988
|
-
method: "POST",
|
|
3989
|
-
body: JSON.stringify({ content: inputs }),
|
|
3990
|
-
headers: { "content-type": "application/json" }
|
|
3991
|
-
}));
|
|
3992
|
-
if (!native.ok || !native.data) {
|
|
3993
|
-
throw new Error(`llama.cpp embedding failed: ${native.error ?? "unknown"}`);
|
|
3994
|
-
}
|
|
3995
|
-
const items = native.data;
|
|
3996
|
-
items.sort((a, b) => (a.index ?? 0) - (b.index ?? 0));
|
|
3997
|
-
const vectors = items.map(flattenNativeEmbedding);
|
|
3998
|
-
return {
|
|
3999
|
-
vectors,
|
|
4000
|
-
model: modelId,
|
|
4001
|
-
dimensions: vectors[0]?.length ?? 0
|
|
4002
|
-
};
|
|
4003
|
-
}
|
|
4004
|
-
};
|
|
4005
|
-
var llamacpp_embed_default = llamacppEmbedRuntime;
|
|
4006
|
-
|
|
4007
|
-
// src/domains/providers/runtimes/local-native/llamacpp-rerank.ts
|
|
4008
|
-
init_esm_shims();
|
|
4009
|
-
var defaultCapabilities20 = {
|
|
4010
|
-
chat: false,
|
|
4011
|
-
tools: false,
|
|
4012
|
-
reasoning: false,
|
|
4013
|
-
vision: false,
|
|
4014
|
-
audio: false,
|
|
4015
|
-
embeddings: false,
|
|
4016
|
-
rerank: true,
|
|
4017
|
-
fim: false,
|
|
4018
|
-
contextWindow: 8192,
|
|
4019
|
-
maxTokens: 0
|
|
4020
|
-
};
|
|
4021
|
-
function targetUrl4(target) {
|
|
4022
|
-
return targetRootUrl(target);
|
|
4023
|
-
}
|
|
4024
|
-
var llamacppRerankRuntime = {
|
|
4025
|
-
id: "llamacpp-rerank",
|
|
4026
|
-
displayName: "llama.cpp (rerank)",
|
|
4027
|
-
kind: "http",
|
|
4028
|
-
tier: "local-native",
|
|
4029
|
-
apiFamily: "openai-completions",
|
|
4030
|
-
auth: "api-key",
|
|
4031
|
-
defaultCapabilities: defaultCapabilities20,
|
|
4032
|
-
hidden: true,
|
|
4033
|
-
async probe(target, ctx) {
|
|
4034
|
-
const base = targetUrl4(target);
|
|
4035
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
4036
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
4037
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
4038
|
-
if (!health.ok) return health;
|
|
4039
|
-
const modelId = target.defaultModel ?? "default";
|
|
4040
|
-
const probeResponse = await fetch(`${base}/reranking`, {
|
|
4041
|
-
method: "POST",
|
|
4042
|
-
headers: { "content-type": "application/json" },
|
|
4043
|
-
body: JSON.stringify({ query: "probe", documents: ["a"], model: modelId }),
|
|
4044
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
4045
|
-
}).catch((err) => new Response(null, { status: 599, statusText: String(err) }));
|
|
4046
|
-
if (!(probeResponse.status === 200 || probeResponse.status === 202)) {
|
|
4047
|
-
return { ok: false, error: `/reranking not available: HTTP ${probeResponse.status}` };
|
|
4048
|
-
}
|
|
4049
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
4050
|
-
const result = { ok: true };
|
|
4051
|
-
if (health.latencyMs !== void 0) result.latencyMs = health.latencyMs;
|
|
4052
|
-
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
4053
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
4054
|
-
return result;
|
|
4055
|
-
},
|
|
4056
|
-
async probeModels(target, ctx) {
|
|
4057
|
-
const base = targetUrl4(target);
|
|
4058
|
-
if (!base) return [];
|
|
4059
|
-
return probeOpenAIModels(base, ctx);
|
|
4060
|
-
},
|
|
4061
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4062
|
-
return synthLocalModel({
|
|
4063
|
-
target,
|
|
4064
|
-
wireModelId,
|
|
4065
|
-
kb,
|
|
4066
|
-
defaultCapabilities: defaultCapabilities20,
|
|
4067
|
-
apiFamily: "openai-completions",
|
|
4068
|
-
provider: "llamacpp",
|
|
4069
|
-
baseUrlForTarget: withV1
|
|
4070
|
-
});
|
|
4071
|
-
},
|
|
4072
|
-
async rerank(target, query, documents, ctx) {
|
|
4073
|
-
const base = targetUrl4(target);
|
|
4074
|
-
if (!base) throw new Error("target has no url");
|
|
4075
|
-
const modelId = target.defaultModel ?? "default";
|
|
4076
|
-
const req = {
|
|
4077
|
-
url: `${base}/reranking`,
|
|
4078
|
-
method: "POST",
|
|
4079
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
4080
|
-
headers: { "content-type": "application/json" },
|
|
4081
|
-
body: JSON.stringify({
|
|
4082
|
-
query,
|
|
4083
|
-
documents,
|
|
4084
|
-
top_n: documents.length,
|
|
4085
|
-
model: modelId
|
|
4086
|
-
})
|
|
4087
|
-
};
|
|
4088
|
-
const result = await (ctx.signal ? probeJson({ ...req, signal: ctx.signal }) : probeJson(req));
|
|
4089
|
-
if (!result.ok || !result.data) {
|
|
4090
|
-
throw new Error(`llama.cpp rerank failed: ${result.error ?? "unknown"}`);
|
|
4091
|
-
}
|
|
4092
|
-
const rows = result.data.results ?? [];
|
|
4093
|
-
const items = rows.map((row) => {
|
|
4094
|
-
const idx = typeof row.index === "number" ? row.index : 0;
|
|
4095
|
-
const score = typeof row.relevance_score === "number" ? row.relevance_score : 0;
|
|
4096
|
-
const doc = typeof row.document === "string" ? row.document : typeof row.document?.text === "string" ? row.document.text : void 0;
|
|
4097
|
-
const item = { index: idx, score };
|
|
4098
|
-
if (doc !== void 0) item.document = doc;
|
|
4099
|
-
return item;
|
|
4100
|
-
});
|
|
4101
|
-
return { items, model: result.data.model ?? modelId };
|
|
4102
|
-
}
|
|
4103
|
-
};
|
|
4104
|
-
var llamacpp_rerank_default = llamacppRerankRuntime;
|
|
4105
|
-
|
|
4106
|
-
// src/domains/providers/runtimes/local-native/lmstudio.ts
|
|
4107
|
-
init_esm_shims();
|
|
4108
|
-
var defaultCapabilities21 = {
|
|
4109
|
-
chat: true,
|
|
4110
|
-
tools: true,
|
|
4111
|
-
toolCallFormat: "openai",
|
|
4112
|
-
structuredOutputs: "json-schema",
|
|
4113
|
-
reasoning: true,
|
|
4114
|
-
vision: true,
|
|
4115
|
-
audio: false,
|
|
4116
|
-
embeddings: false,
|
|
4117
|
-
rerank: false,
|
|
4118
|
-
fim: false,
|
|
4119
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4120
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4121
|
-
};
|
|
4122
|
-
var reasoningOptionsByTargetModel = /* @__PURE__ */ new Map();
|
|
4123
|
-
function reasoningOptionsKey(target, modelId) {
|
|
4124
|
-
const url2 = (target.url ?? "").replace(/^ws:/u, "http:").replace(/^wss:/u, "https:").replace(/\/$/u, "");
|
|
4125
|
-
return `${target.id}|${url2}|${modelId}`;
|
|
4126
|
-
}
|
|
4127
|
-
function rememberReasoningOptions(target, models) {
|
|
4128
|
-
const prefix = reasoningOptionsKey(target, "");
|
|
4129
|
-
for (const key of reasoningOptionsByTargetModel.keys()) {
|
|
4130
|
-
if (key.startsWith(prefix)) reasoningOptionsByTargetModel.delete(key);
|
|
4131
|
-
}
|
|
4132
|
-
for (const model of models) {
|
|
4133
|
-
if (!model.reasoningOptions) continue;
|
|
4134
|
-
reasoningOptionsByTargetModel.set(reasoningOptionsKey(target, model.key), model.reasoningOptions);
|
|
4135
|
-
for (const instance of model.loadedInstances) {
|
|
4136
|
-
reasoningOptionsByTargetModel.set(reasoningOptionsKey(target, instance.id), model.reasoningOptions);
|
|
4137
|
-
}
|
|
1008
|
+
} else if (mechanism === "always-on" && configuredLevel !== effectiveLevel) {
|
|
1009
|
+
applied = appendNotice(applied, `${configuredLevel} was ignored because thinking is always on`, "always-on");
|
|
1010
|
+
} else if (mechanism === "none" && configuredLevel !== effectiveLevel) {
|
|
1011
|
+
applied = appendNotice(applied, `${configuredLevel} was ignored because thinking is unsupported`, "unsupported");
|
|
4138
1012
|
}
|
|
4139
|
-
|
|
4140
|
-
|
|
4141
|
-
|
|
4142
|
-
|
|
4143
|
-
|
|
4144
|
-
|
|
4145
|
-
|
|
4146
|
-
for (const model2 of models) {
|
|
4147
|
-
if (model2.key === id) return { model: model2, ...model2.loadedInstances[0] ? { instance: model2.loadedInstances[0] } : {} };
|
|
4148
|
-
const instance = model2.loadedInstances.find((entry) => entry.id === id);
|
|
4149
|
-
if (instance) return { model: model2, instance };
|
|
4150
|
-
}
|
|
4151
|
-
return null;
|
|
1013
|
+
const budgetEnforcement = resolveBudgetEnforcement(mechanism, input);
|
|
1014
|
+
if (applied.thinkingActive && mechanism === "budget-tokens" && budgetEnforcement === "informational") {
|
|
1015
|
+
applied = appendNotice(
|
|
1016
|
+
applied,
|
|
1017
|
+
"target does not expose an enforceable per-request thinking budget; level is advisory",
|
|
1018
|
+
"applied"
|
|
1019
|
+
);
|
|
4152
1020
|
}
|
|
4153
|
-
|
|
4154
|
-
|
|
1021
|
+
return {
|
|
1022
|
+
...applied,
|
|
1023
|
+
configuredLevel,
|
|
1024
|
+
effectiveLevel,
|
|
1025
|
+
supportedLevels,
|
|
1026
|
+
display: thinkingLevelDisplayWord(applied.mechanism, effectiveLevel),
|
|
1027
|
+
budgetEnforcement
|
|
1028
|
+
};
|
|
4155
1029
|
}
|
|
4156
|
-
|
|
4157
|
-
|
|
4158
|
-
|
|
4159
|
-
if (model.tools !== void 0) out.tools = model.tools;
|
|
4160
|
-
if (model.reasoning !== void 0) out.reasoning = model.reasoning;
|
|
4161
|
-
else if (model.reasoningOptions !== void 0)
|
|
4162
|
-
out.reasoning = model.reasoningOptions.some((option) => option !== "off");
|
|
4163
|
-
const contextWindow = loadedContextLength(instance) ?? model.maxContextLength;
|
|
4164
|
-
if (contextWindow !== void 0) out.contextWindow = contextWindow;
|
|
4165
|
-
return out;
|
|
1030
|
+
var REASONING_EFFORT_ON_OFF_RUNTIMES = /* @__PURE__ */ new Set(["lmstudio"]);
|
|
1031
|
+
function onOffReasoningEffort(thinkingActive) {
|
|
1032
|
+
return thinkingActive ? "low" : "none";
|
|
4166
1033
|
}
|
|
4167
|
-
function
|
|
4168
|
-
const
|
|
4169
|
-
|
|
4170
|
-
|
|
4171
|
-
reasoningLevels
|
|
4172
|
-
};
|
|
4173
|
-
if (instance) {
|
|
4174
|
-
status.instanceId = instance.id;
|
|
4175
|
-
status.loadConfig = instance.config;
|
|
4176
|
-
const contextLength = loadedContextLength(instance);
|
|
4177
|
-
if (contextLength !== void 0) status.contextLength = contextLength;
|
|
1034
|
+
function resolveRequestCapability(thinking, parser, runtimeId) {
|
|
1035
|
+
const request = { budgetEnforcement: thinking.budgetEnforcement };
|
|
1036
|
+
if (thinking.mechanism === "effort-levels" && thinking.effort) {
|
|
1037
|
+
request.reasoningEffort = thinking.effort;
|
|
4178
1038
|
}
|
|
4179
|
-
|
|
4180
|
-
}
|
|
4181
|
-
function probeFromCatalog(catalog, target) {
|
|
4182
|
-
if (!catalog.ok) {
|
|
4183
|
-
return {
|
|
4184
|
-
ok: false,
|
|
4185
|
-
...catalog.latencyMs !== void 0 ? { latencyMs: catalog.latencyMs } : {},
|
|
4186
|
-
error: catalog.error ?? "LM Studio model listing failed"
|
|
4187
|
-
};
|
|
1039
|
+
if (thinking.mechanism === "effort-levels" && !thinking.thinkingActive) {
|
|
1040
|
+
request.chatTemplateKwargs = { ...request.chatTemplateKwargs ?? {}, enable_thinking: false };
|
|
4188
1041
|
}
|
|
4189
|
-
|
|
4190
|
-
|
|
4191
|
-
|
|
4192
|
-
|
|
4193
|
-
|
|
4194
|
-
|
|
4195
|
-
|
|
4196
|
-
const keyInstance = model.loadedInstances[0];
|
|
4197
|
-
modelCapabilities[model.key] = capabilities(model, keyInstance);
|
|
4198
|
-
modelStates[model.key] = statusFor(model, keyInstance, levels);
|
|
4199
|
-
for (const instance of model.loadedInstances) {
|
|
4200
|
-
if (!ids.includes(instance.id)) ids.push(instance.id);
|
|
4201
|
-
modelCapabilities[instance.id] = capabilities(model, instance);
|
|
4202
|
-
const instanceStatus = statusFor(model, instance, levels);
|
|
4203
|
-
modelStates[instance.id] = instanceStatus;
|
|
4204
|
-
const resolution = resolveLmStudioInstance(target, models, instance.id, configuredModel(target));
|
|
4205
|
-
instanceStatus.detail = resolution.peerTargets.length > 0 ? `loaded on ${target.id}; also loaded on ${resolution.peerTargets.join(", ")}` : `loaded on ${target.id}`;
|
|
4206
|
-
}
|
|
4207
|
-
if (model.loadedInstances.length === 0) {
|
|
4208
|
-
ids.push(model.key);
|
|
4209
|
-
const keyStatus = modelStates[model.key];
|
|
4210
|
-
if (keyStatus) keyStatus.detail = "not loaded (LM Studio will load it on first use)";
|
|
1042
|
+
if (thinking.mechanism === "budget-tokens" && thinking.budgetTokens !== void 0) {
|
|
1043
|
+
request.budgetTokens = thinking.budgetTokens;
|
|
1044
|
+
}
|
|
1045
|
+
if (thinking.mechanism === "on-off" && thinking.chatTemplateKwargs) {
|
|
1046
|
+
request.chatTemplateKwargs = { ...thinking.chatTemplateKwargs };
|
|
1047
|
+
if (REASONING_EFFORT_ON_OFF_RUNTIMES.has(runtimeId)) {
|
|
1048
|
+
request.reasoningEffort = onOffReasoningEffort(thinking.thinkingActive);
|
|
4211
1049
|
}
|
|
4212
1050
|
}
|
|
4213
|
-
|
|
1051
|
+
if (parser === "harmony" && thinking.effort) {
|
|
1052
|
+
request.reasoningEffort = thinking.effort;
|
|
1053
|
+
request.chatTemplateKwargs = { ...request.chatTemplateKwargs ?? {}, reasoning_effort: thinking.effort };
|
|
1054
|
+
}
|
|
1055
|
+
return request;
|
|
1056
|
+
}
|
|
1057
|
+
function resolveModelRuntimeCapabilities(input) {
|
|
1058
|
+
const family = capabilityFamily(input);
|
|
1059
|
+
const quirks = resolveQuirks(input);
|
|
1060
|
+
const parser = resolveResponseParser(input, family);
|
|
1061
|
+
const thinking = resolveThinkingCapability(input, quirks, parser);
|
|
4214
1062
|
const result = {
|
|
4215
|
-
|
|
4216
|
-
|
|
4217
|
-
|
|
4218
|
-
|
|
4219
|
-
|
|
4220
|
-
|
|
4221
|
-
|
|
4222
|
-
|
|
4223
|
-
|
|
1063
|
+
targetId: input.targetId ?? null,
|
|
1064
|
+
runtimeId: input.runtimeId,
|
|
1065
|
+
apiFamily: input.apiFamily ?? null,
|
|
1066
|
+
modelId: input.modelId,
|
|
1067
|
+
family,
|
|
1068
|
+
capabilities: input.capabilities,
|
|
1069
|
+
thinking,
|
|
1070
|
+
request: resolveRequestCapability(thinking, parser, input.runtimeId),
|
|
1071
|
+
response: {
|
|
1072
|
+
parser,
|
|
1073
|
+
stripTokenizerSentinels: true
|
|
4224
1074
|
}
|
|
4225
1075
|
};
|
|
4226
|
-
if (
|
|
4227
|
-
if (selected) {
|
|
4228
|
-
result.discoveredCapabilities = capabilities(selected.model, selected.instance);
|
|
4229
|
-
result.capabilityModelId = configuredModel(target) ?? selected.model.key;
|
|
4230
|
-
}
|
|
4231
|
-
const loaded = models.flatMap(
|
|
4232
|
-
(model) => model.loadedInstances.map((instance) => {
|
|
4233
|
-
const context = loadedContextLength(instance);
|
|
4234
|
-
return `${model.key} as ${instance.id}${context ? ` at ${context} tokens` : ""}`;
|
|
4235
|
-
})
|
|
4236
|
-
);
|
|
4237
|
-
result.notes = [
|
|
4238
|
-
`LM Studio surface tier ${catalog.tier ?? "unknown"}`,
|
|
4239
|
-
`Surfaces: ${Object.values(result.surfaces ?? {}).join(", ")}`,
|
|
4240
|
-
...loaded.length > 0 ? [`Loaded instances: ${loaded.join(", ")}`] : []
|
|
4241
|
-
];
|
|
1076
|
+
if (quirks) result.quirks = quirks;
|
|
4242
1077
|
return result;
|
|
4243
1078
|
}
|
|
4244
|
-
|
|
4245
|
-
|
|
4246
|
-
|
|
4247
|
-
|
|
4248
|
-
|
|
4249
|
-
|
|
4250
|
-
|
|
4251
|
-
|
|
4252
|
-
|
|
4253
|
-
|
|
4254
|
-
|
|
4255
|
-
|
|
4256
|
-
|
|
4257
|
-
|
|
4258
|
-
|
|
4259
|
-
|
|
4260
|
-
|
|
4261
|
-
}
|
|
4262
|
-
return probeFromCatalog(catalog, target);
|
|
4263
|
-
},
|
|
4264
|
-
async probeModels(target, ctx) {
|
|
4265
|
-
const catalog = await listLmStudioModels(target, ctx);
|
|
4266
|
-
return probeFromCatalog(catalog, target).models ?? [];
|
|
4267
|
-
},
|
|
4268
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4269
|
-
const canonicalTarget = target.runtime === "lmstudio" ? target : { ...target, runtime: "lmstudio" };
|
|
4270
|
-
const model = synthLocalModel({
|
|
4271
|
-
target: canonicalTarget,
|
|
4272
|
-
wireModelId,
|
|
4273
|
-
kb,
|
|
4274
|
-
defaultCapabilities: defaultCapabilities21,
|
|
4275
|
-
apiFamily: "openai-completions",
|
|
4276
|
-
provider: "lmstudio",
|
|
4277
|
-
baseUrlForTarget: (url2) => withV1(url2.replace(/^ws:/u, "http:").replace(/^wss:/u, "https:"))
|
|
4278
|
-
});
|
|
4279
|
-
const metadata2 = model.clio;
|
|
4280
|
-
if (metadata2) {
|
|
4281
|
-
metadata2.chatTemplateKwargsUnsupported = true;
|
|
4282
|
-
if (target.defaultModel) metadata2.lmstudioDefaultModel = target.defaultModel;
|
|
4283
|
-
const options = reasoningOptionsByTargetModel.get(reasoningOptionsKey(canonicalTarget, wireModelId));
|
|
4284
|
-
if (options) metadata2.lmstudioReasoningOptions = options;
|
|
4285
|
-
}
|
|
4286
|
-
return model;
|
|
4287
|
-
}
|
|
4288
|
-
};
|
|
4289
|
-
var lmstudio_default = lmstudioRuntime;
|
|
4290
|
-
|
|
4291
|
-
// src/domains/providers/runtimes/local-native/ollama-native.ts
|
|
4292
|
-
init_esm_shims();
|
|
4293
|
-
var defaultCapabilities22 = {
|
|
4294
|
-
chat: true,
|
|
4295
|
-
tools: true,
|
|
4296
|
-
toolCallFormat: "openai",
|
|
4297
|
-
reasoning: false,
|
|
4298
|
-
vision: false,
|
|
4299
|
-
audio: false,
|
|
4300
|
-
embeddings: false,
|
|
4301
|
-
rerank: false,
|
|
4302
|
-
fim: false,
|
|
4303
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4304
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4305
|
-
};
|
|
4306
|
-
function positiveNumber3(value) {
|
|
4307
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
1079
|
+
function resolveModelRuntimeCapabilitiesForStatus(status, wireModelId, knowledgeBase, options) {
|
|
1080
|
+
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1081
|
+
const kbHit = modelId ? knowledgeBase?.lookup(modelId) ?? null : null;
|
|
1082
|
+
const capabilities = resolveModelCapabilities(status, modelId, knowledgeBase, {
|
|
1083
|
+
detectedReasoning: options?.detectedReasoning ?? null
|
|
1084
|
+
});
|
|
1085
|
+
const runtimeId = status.runtime?.id ?? status.target.runtime;
|
|
1086
|
+
return resolveModelRuntimeCapabilities({
|
|
1087
|
+
targetId: status.target.id,
|
|
1088
|
+
runtimeId,
|
|
1089
|
+
apiFamily: status.runtime?.apiFamily ?? null,
|
|
1090
|
+
modelId,
|
|
1091
|
+
capabilities,
|
|
1092
|
+
kbHit,
|
|
1093
|
+
...thinkingHintsForCatalogModel(runtimeId, modelId),
|
|
1094
|
+
...options?.configuredThinkingLevel ? { configuredThinkingLevel: options.configuredThinkingLevel } : {}
|
|
1095
|
+
});
|
|
4308
1096
|
}
|
|
4309
|
-
|
|
4310
|
-
const
|
|
4311
|
-
|
|
4312
|
-
|
|
4313
|
-
|
|
4314
|
-
|
|
4315
|
-
|
|
4316
|
-
|
|
4317
|
-
|
|
4318
|
-
|
|
4319
|
-
|
|
4320
|
-
|
|
4321
|
-
|
|
4322
|
-
|
|
4323
|
-
|
|
4324
|
-
|
|
4325
|
-
|
|
4326
|
-
|
|
4327
|
-
|
|
4328
|
-
|
|
4329
|
-
|
|
4330
|
-
|
|
4331
|
-
|
|
4332
|
-
|
|
4333
|
-
|
|
4334
|
-
|
|
4335
|
-
|
|
4336
|
-
|
|
4337
|
-
|
|
4338
|
-
|
|
4339
|
-
|
|
4340
|
-
|
|
4341
|
-
|
|
4342
|
-
|
|
4343
|
-
return
|
|
4344
|
-
}
|
|
4345
|
-
const out = { ok: true };
|
|
4346
|
-
if (result.latencyMs !== void 0) out.latencyMs = result.latencyMs;
|
|
4347
|
-
const modelStates = await probeResidentModelStates(base, ctx);
|
|
4348
|
-
if (modelStates) out.modelStates = modelStates;
|
|
4349
|
-
return out;
|
|
4350
|
-
},
|
|
4351
|
-
async probeModels(target, ctx) {
|
|
4352
|
-
const base = targetBaseUrl(target);
|
|
4353
|
-
if (!base) return [];
|
|
4354
|
-
const opts = { url: `${base}/api/tags`, timeoutMs: ctx.httpTimeoutMs };
|
|
4355
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
4356
|
-
if (!result.ok || !result.data?.models) return [];
|
|
4357
|
-
return result.data.models.map((row) => typeof row?.name === "string" ? row.name : null).filter((name) => name !== null);
|
|
4358
|
-
},
|
|
4359
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4360
|
-
return synthLocalModel({
|
|
4361
|
-
target,
|
|
4362
|
-
wireModelId,
|
|
4363
|
-
kb,
|
|
4364
|
-
defaultCapabilities: defaultCapabilities22,
|
|
4365
|
-
apiFamily: "ollama-native",
|
|
4366
|
-
provider: "ollama",
|
|
4367
|
-
baseUrlForTarget: withAsIs
|
|
4368
|
-
});
|
|
1097
|
+
function resolveModelRuntimeCapabilitiesForProviders(providers, targetId, wireModelId, configuredThinkingLevel) {
|
|
1098
|
+
const id = targetId?.trim();
|
|
1099
|
+
if (!id) return null;
|
|
1100
|
+
const status = providers.list().find((entry) => entry.target.id === id);
|
|
1101
|
+
if (!status) return null;
|
|
1102
|
+
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1103
|
+
const detectedReasoning = modelId && typeof providers.getDetectedReasoning === "function" ? providers.getDetectedReasoning(id, modelId) : null;
|
|
1104
|
+
return resolveModelRuntimeCapabilitiesForStatus(status, modelId, providers.knowledgeBase, {
|
|
1105
|
+
detectedReasoning,
|
|
1106
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1107
|
+
});
|
|
1108
|
+
}
|
|
1109
|
+
function thinkingHintsForModel(model) {
|
|
1110
|
+
const out = {};
|
|
1111
|
+
if (!model) return out;
|
|
1112
|
+
const compat = model.compat;
|
|
1113
|
+
if (compat?.forceAdaptiveThinking !== void 0) out.adaptiveThinking = compat.forceAdaptiveThinking;
|
|
1114
|
+
if (model.thinkingLevelMap) out.thinkingLevelMap = model.thinkingLevelMap;
|
|
1115
|
+
return out;
|
|
1116
|
+
}
|
|
1117
|
+
function thinkingHintsForCatalogModel(runtimeId, modelId) {
|
|
1118
|
+
if (!runtimeId || modelId.length === 0) return {};
|
|
1119
|
+
return thinkingHintsForModel(getCatalogModelForRuntime(runtimeId, modelId));
|
|
1120
|
+
}
|
|
1121
|
+
function thinkingFormatFromModelApi(api) {
|
|
1122
|
+
switch (api) {
|
|
1123
|
+
case "anthropic-messages":
|
|
1124
|
+
case "bedrock-converse-stream":
|
|
1125
|
+
case "claude-agent-sdk":
|
|
1126
|
+
case "claude-code-subprocess":
|
|
1127
|
+
return "anthropic-extended";
|
|
1128
|
+
case "openai-codex-responses":
|
|
1129
|
+
return "openai-codex";
|
|
1130
|
+
default:
|
|
1131
|
+
return void 0;
|
|
4369
1132
|
}
|
|
4370
|
-
}
|
|
4371
|
-
|
|
4372
|
-
|
|
4373
|
-
|
|
4374
|
-
|
|
4375
|
-
|
|
4376
|
-
|
|
4377
|
-
|
|
4378
|
-
|
|
4379
|
-
|
|
4380
|
-
|
|
4381
|
-
|
|
4382
|
-
|
|
4383
|
-
|
|
4384
|
-
|
|
4385
|
-
|
|
4386
|
-
|
|
4387
|
-
};
|
|
4388
|
-
var sglang_default = makeOpenAICompatRuntime({
|
|
4389
|
-
id: "sglang",
|
|
4390
|
-
displayName: "SGLang",
|
|
4391
|
-
provider: "sglang",
|
|
4392
|
-
auth: "api-key",
|
|
4393
|
-
tier: "local-native",
|
|
4394
|
-
defaultCapabilities: defaultCapabilities23
|
|
4395
|
-
});
|
|
4396
|
-
|
|
4397
|
-
// src/domains/providers/runtimes/local-native/vllm.ts
|
|
4398
|
-
init_esm_shims();
|
|
4399
|
-
var defaultCapabilities24 = {
|
|
4400
|
-
chat: true,
|
|
4401
|
-
tools: true,
|
|
4402
|
-
toolCallFormat: "openai",
|
|
4403
|
-
reasoning: false,
|
|
4404
|
-
vision: false,
|
|
4405
|
-
audio: false,
|
|
4406
|
-
embeddings: false,
|
|
4407
|
-
rerank: false,
|
|
4408
|
-
fim: false,
|
|
4409
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4410
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4411
|
-
};
|
|
4412
|
-
var vllm_default = makeOpenAICompatRuntime({
|
|
4413
|
-
id: "vllm",
|
|
4414
|
-
displayName: "vLLM",
|
|
4415
|
-
provider: "vllm",
|
|
4416
|
-
auth: "api-key",
|
|
4417
|
-
tier: "local-native",
|
|
4418
|
-
defaultCapabilities: defaultCapabilities24,
|
|
4419
|
-
healthPath: "/health"
|
|
4420
|
-
});
|
|
4421
|
-
|
|
4422
|
-
// src/domains/providers/runtimes/builtins.ts
|
|
4423
|
-
var BUILTIN_RUNTIMES = [
|
|
4424
|
-
alcf_default,
|
|
4425
|
-
anthropic_default,
|
|
4426
|
-
anthropic_max_default,
|
|
4427
|
-
bedrock_default,
|
|
4428
|
-
deepseek_default,
|
|
4429
|
-
google_default,
|
|
4430
|
-
groq_default,
|
|
4431
|
-
mistral_default,
|
|
4432
|
-
openai_default,
|
|
4433
|
-
openai_codex_default,
|
|
4434
|
-
openrouter_default,
|
|
4435
|
-
lemonade_anthropic_default,
|
|
4436
|
-
lemonade_openai_default,
|
|
4437
|
-
llamacpp_default,
|
|
4438
|
-
llamacpp_anthropic_default,
|
|
4439
|
-
llamacpp_completion_default,
|
|
4440
|
-
llamacpp_embed_default,
|
|
4441
|
-
llamacpp_rerank_default,
|
|
4442
|
-
lmstudio_default,
|
|
4443
|
-
ollama_native_default,
|
|
4444
|
-
anthropic_compat_default,
|
|
4445
|
-
openai_compat_default,
|
|
4446
|
-
sglang_default,
|
|
4447
|
-
vllm_default,
|
|
4448
|
-
claude_code_default,
|
|
4449
|
-
claude_sdk_default,
|
|
4450
|
-
antigravity_code_default
|
|
4451
|
-
];
|
|
4452
|
-
function registerBuiltinRuntimes(registry) {
|
|
4453
|
-
for (const desc of BUILTIN_RUNTIMES) {
|
|
4454
|
-
if (registry.get(desc.id) !== null) continue;
|
|
4455
|
-
registry.register(desc);
|
|
1133
|
+
}
|
|
1134
|
+
function capabilitiesFromModel(model) {
|
|
1135
|
+
const format = model.compat?.thinkingFormat ?? thinkingFormatFromModelApi(model.api);
|
|
1136
|
+
const caps = {
|
|
1137
|
+
chat: true,
|
|
1138
|
+
tools: true,
|
|
1139
|
+
reasoning: model.reasoning === true,
|
|
1140
|
+
vision: Array.isArray(model.input) && model.input.includes("image"),
|
|
1141
|
+
audio: false,
|
|
1142
|
+
embeddings: false,
|
|
1143
|
+
rerank: false,
|
|
1144
|
+
fim: false,
|
|
1145
|
+
contextWindow: model.contextWindow,
|
|
1146
|
+
maxTokens: model.maxTokens
|
|
1147
|
+
};
|
|
1148
|
+
if (format === "qwen-chat-template" || format === "openrouter" || format === "zai" || format === "anthropic-extended" || format === "deepseek-r1" || format === "openai-codex" || format === "harmony") {
|
|
1149
|
+
caps.thinkingFormat = format;
|
|
4456
1150
|
}
|
|
1151
|
+
return caps;
|
|
1152
|
+
}
|
|
1153
|
+
function resolveModelRuntimeCapabilitiesForModel(model, configuredThinkingLevel) {
|
|
1154
|
+
const metadata2 = model.clio;
|
|
1155
|
+
const caps = capabilitiesFromModel(model);
|
|
1156
|
+
return resolveModelRuntimeCapabilities({
|
|
1157
|
+
targetId: metadata2?.targetId ?? null,
|
|
1158
|
+
runtimeId: metadata2?.runtimeId ?? model.provider,
|
|
1159
|
+
apiFamily: model.api,
|
|
1160
|
+
modelId: model.id,
|
|
1161
|
+
capabilities: caps,
|
|
1162
|
+
...thinkingHintsForModel(model),
|
|
1163
|
+
...metadata2?.quirks ? { quirks: metadata2.quirks } : {},
|
|
1164
|
+
kbHit: metadata2?.family ? {
|
|
1165
|
+
matchKind: "family",
|
|
1166
|
+
entry: {
|
|
1167
|
+
family: metadata2.family,
|
|
1168
|
+
matchPatterns: [metadata2.family],
|
|
1169
|
+
capabilities: {}
|
|
1170
|
+
}
|
|
1171
|
+
} : null,
|
|
1172
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1173
|
+
});
|
|
1174
|
+
}
|
|
1175
|
+
function resolveTargetRuntimeCapabilities(target, runtime, wireModelId, capabilities, knowledgeBase, configuredThinkingLevel) {
|
|
1176
|
+
const kbHit = knowledgeBase?.lookup(wireModelId) ?? null;
|
|
1177
|
+
return resolveModelRuntimeCapabilities({
|
|
1178
|
+
targetId: target.id,
|
|
1179
|
+
runtimeId: runtime.id,
|
|
1180
|
+
apiFamily: runtime.apiFamily,
|
|
1181
|
+
modelId: wireModelId,
|
|
1182
|
+
capabilities,
|
|
1183
|
+
kbHit,
|
|
1184
|
+
...thinkingHintsForCatalogModel(runtime.id, wireModelId),
|
|
1185
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1186
|
+
});
|
|
4457
1187
|
}
|
|
4458
1188
|
|
|
4459
1189
|
// src/domains/providers/eligibility.ts
|
|
@@ -4468,140 +1198,6 @@ function isDispatchEligibleRuntime(runtime) {
|
|
|
4468
1198
|
return isTargetEligibleRuntime(runtime);
|
|
4469
1199
|
}
|
|
4470
1200
|
|
|
4471
|
-
// src/domains/providers/support.ts
|
|
4472
|
-
init_esm_shims();
|
|
4473
|
-
var SUMMARY_BY_RUNTIME_ID = {
|
|
4474
|
-
alcf: "ALCF inference gateway (Sophia/Metis) via Globus",
|
|
4475
|
-
anthropic: "Anthropic API",
|
|
4476
|
-
"anthropic-max": "Claude Pro/Max subscription via Anthropic OAuth",
|
|
4477
|
-
bedrock: "Amazon Bedrock",
|
|
4478
|
-
"claude-code": "Claude Code subscription via installed claude CLI",
|
|
4479
|
-
"claude-sdk": "Claude Code subscription via Claude Agent SDK",
|
|
4480
|
-
deepseek: "DeepSeek API",
|
|
4481
|
-
google: "Google Gemini API",
|
|
4482
|
-
groq: "Groq API",
|
|
4483
|
-
mistral: "Mistral API",
|
|
4484
|
-
openai: "OpenAI Platform API",
|
|
4485
|
-
"openai-codex": "ChatGPT Plus/Pro via Codex OAuth",
|
|
4486
|
-
openrouter: "OpenRouter API",
|
|
4487
|
-
"ollama-native": "Ollama native API",
|
|
4488
|
-
lmstudio: "LM Studio chat over OpenAI-compatible REST with native REST model management",
|
|
4489
|
-
llamacpp: "llama.cpp server (auto-detect surface)",
|
|
4490
|
-
"anthropic-compat": "Generic Anthropic-compatible REST",
|
|
4491
|
-
"openai-compat": "Generic OpenAI-compatible REST"
|
|
4492
|
-
};
|
|
4493
|
-
function groupPriority(group) {
|
|
4494
|
-
switch (group) {
|
|
4495
|
-
case "featured":
|
|
4496
|
-
return 0;
|
|
4497
|
-
case "subscription":
|
|
4498
|
-
return 1;
|
|
4499
|
-
case "cloud-api":
|
|
4500
|
-
return 2;
|
|
4501
|
-
case "local-http":
|
|
4502
|
-
return 3;
|
|
4503
|
-
}
|
|
4504
|
-
}
|
|
4505
|
-
function supportGroupLabel(group) {
|
|
4506
|
-
switch (group) {
|
|
4507
|
-
case "featured":
|
|
4508
|
-
return "Featured";
|
|
4509
|
-
case "subscription":
|
|
4510
|
-
return "Subscriptions";
|
|
4511
|
-
case "cloud-api":
|
|
4512
|
-
return "Cloud APIs";
|
|
4513
|
-
case "local-http":
|
|
4514
|
-
return "Local HTTP";
|
|
4515
|
-
}
|
|
4516
|
-
}
|
|
4517
|
-
function classifyGroup(runtime) {
|
|
4518
|
-
if (runtime.id === "openai-codex") return "featured";
|
|
4519
|
-
if (runtime.id === "alcf") return "cloud-api";
|
|
4520
|
-
if (runtime.auth === "oauth" || runtime.auth === "claude-cli") return "subscription";
|
|
4521
|
-
if (catalogProviderForRuntime(runtime.id) || runtime.auth === "api-key" && !runtime.probe) {
|
|
4522
|
-
return "cloud-api";
|
|
4523
|
-
}
|
|
4524
|
-
return "local-http";
|
|
4525
|
-
}
|
|
4526
|
-
function knownModelsFor(runtimeId, runtime) {
|
|
4527
|
-
const catalogModels2 = listCatalogModelsForRuntime(runtimeId);
|
|
4528
|
-
if (catalogModels2.length === 0) return runtime?.knownModels ? [...runtime.knownModels] : [];
|
|
4529
|
-
return catalogModels2.map((model) => model.id);
|
|
4530
|
-
}
|
|
4531
|
-
function listKnownModelsForRuntime(runtimeId) {
|
|
4532
|
-
return knownModelsFor(runtimeId, getRuntimeIfRegistered(runtimeId));
|
|
4533
|
-
}
|
|
4534
|
-
function getRuntimeIfRegistered(runtimeId) {
|
|
4535
|
-
try {
|
|
4536
|
-
return getRuntimeRegistry().get(runtimeId);
|
|
4537
|
-
} catch {
|
|
4538
|
-
return null;
|
|
4539
|
-
}
|
|
4540
|
-
}
|
|
4541
|
-
function runtimeModelListSource(runtime) {
|
|
4542
|
-
if (runtime.knownModels && runtime.knownModels.length > 0) return "runtime";
|
|
4543
|
-
if (listCatalogModelsForRuntime(runtime.id).length > 0) return "catalog";
|
|
4544
|
-
return "none";
|
|
4545
|
-
}
|
|
4546
|
-
function describeRuntimeModels(entry, sample) {
|
|
4547
|
-
if (entry.modelSource === "catalog") return `${entry.modelHints.length} in catalog`;
|
|
4548
|
-
if (entry.modelHints.length === 0) return "-";
|
|
4549
|
-
return entry.modelHints.slice(0, sample).join(", ");
|
|
4550
|
-
}
|
|
4551
|
-
function buildProviderSupportEntry(runtime) {
|
|
4552
|
-
const modelHints = knownModelsFor(runtime.id, runtime);
|
|
4553
|
-
const modelSource = runtimeModelListSource(runtime);
|
|
4554
|
-
const defaultModel = modelSource === "catalog" ? void 0 : modelHints[0];
|
|
4555
|
-
return {
|
|
4556
|
-
runtimeId: runtime.id,
|
|
4557
|
-
label: runtime.displayName,
|
|
4558
|
-
group: classifyGroup(runtime),
|
|
4559
|
-
summary: SUMMARY_BY_RUNTIME_ID[runtime.id] ?? runtime.displayName,
|
|
4560
|
-
...defaultModel ? { defaultModel } : {},
|
|
4561
|
-
modelHints,
|
|
4562
|
-
modelSource,
|
|
4563
|
-
featured: runtime.id === "openai-codex",
|
|
4564
|
-
connectable: runtime.auth === "oauth" || runtime.auth === "api-key",
|
|
4565
|
-
supportsCustomUrl: runtime.kind === "http" && (classifyGroup(runtime) === "local-http" || runtime.id === "openai-compat" || runtime.id === "anthropic-compat" || runtime.id === "alcf")
|
|
4566
|
-
};
|
|
4567
|
-
}
|
|
4568
|
-
function compareProviderSupportEntries(a, b) {
|
|
4569
|
-
return groupPriority(a.group) - groupPriority(b.group) || (a.featured === b.featured ? 0 : a.featured ? -1 : 1) || a.label.localeCompare(b.label) || a.runtimeId.localeCompare(b.runtimeId);
|
|
4570
|
-
}
|
|
4571
|
-
function listProviderSupportEntries(runtimes, options = {}) {
|
|
4572
|
-
const filtered = options.includeHidden ? runtimes : runtimes.filter((runtime) => runtime.hidden !== true);
|
|
4573
|
-
return filtered.map((runtime) => buildProviderSupportEntry(runtime)).sort(compareProviderSupportEntries);
|
|
4574
|
-
}
|
|
4575
|
-
function configuredTargetsForRuntime(settings, runtimeId) {
|
|
4576
|
-
const canonical = getRuntimeIfRegistered(runtimeId)?.id ?? runtimeId;
|
|
4577
|
-
return settings.targets.filter(
|
|
4578
|
-
(target) => (getRuntimeIfRegistered(target.runtime)?.id ?? target.runtime) === canonical
|
|
4579
|
-
);
|
|
4580
|
-
}
|
|
4581
|
-
function resolveProviderReference(input, settings, getRuntime) {
|
|
4582
|
-
const trimmed = input.trim();
|
|
4583
|
-
if (trimmed.length === 0) return null;
|
|
4584
|
-
const target = settings.targets.find((entry) => entry.id === trimmed) ?? null;
|
|
4585
|
-
if (target) {
|
|
4586
|
-
const runtime2 = getRuntime(target.runtime);
|
|
4587
|
-
if (!runtime2) return null;
|
|
4588
|
-
return {
|
|
4589
|
-
input: trimmed,
|
|
4590
|
-
target,
|
|
4591
|
-
runtime: runtime2,
|
|
4592
|
-
authTarget: resolveAuthTarget(target, runtime2)
|
|
4593
|
-
};
|
|
4594
|
-
}
|
|
4595
|
-
const runtime = getRuntime(trimmed);
|
|
4596
|
-
if (!runtime) return null;
|
|
4597
|
-
return {
|
|
4598
|
-
input: trimmed,
|
|
4599
|
-
target: null,
|
|
4600
|
-
runtime,
|
|
4601
|
-
authTarget: resolveRuntimeAuthTarget(runtime)
|
|
4602
|
-
};
|
|
4603
|
-
}
|
|
4604
|
-
|
|
4605
1201
|
// src/domains/providers/model-discovery.ts
|
|
4606
1202
|
init_esm_shims();
|
|
4607
1203
|
function uniqueModels(ids) {
|
|
@@ -4703,24 +1299,24 @@ function diagnostic(severity, code, message) {
|
|
|
4703
1299
|
function hasError(diagnostics) {
|
|
4704
1300
|
return diagnostics.some((entry) => entry.severity === "error");
|
|
4705
1301
|
}
|
|
4706
|
-
function
|
|
1302
|
+
function statusFor(providers, target, runtime, _wireModelId) {
|
|
4707
1303
|
const existing = providers.list().find((entry) => entry.target.id === target.id);
|
|
4708
1304
|
if (existing) return existing;
|
|
4709
|
-
const
|
|
1305
|
+
const capabilities = { ...runtime.defaultCapabilities, ...target.capabilities ?? {} };
|
|
4710
1306
|
return {
|
|
4711
1307
|
target,
|
|
4712
1308
|
runtime,
|
|
4713
1309
|
available: true,
|
|
4714
1310
|
reason: "synthetic-status",
|
|
4715
1311
|
health: { status: "unknown", lastCheckAt: null, lastError: null, latencyMs: null },
|
|
4716
|
-
capabilities
|
|
1312
|
+
capabilities,
|
|
4717
1313
|
probeCapabilities: null,
|
|
4718
1314
|
probeModelId: null,
|
|
4719
1315
|
discoveredModels: runtime.knownModels ?? []
|
|
4720
1316
|
};
|
|
4721
1317
|
}
|
|
4722
|
-
function requiredCapabilitySupported(
|
|
4723
|
-
const value =
|
|
1318
|
+
function requiredCapabilitySupported(capabilities, name) {
|
|
1319
|
+
const value = capabilities[name];
|
|
4724
1320
|
return value !== void 0 && value !== false && value !== 0 && value !== "";
|
|
4725
1321
|
}
|
|
4726
1322
|
function streamingDecision(runtime) {
|
|
@@ -4730,18 +1326,18 @@ function runtimeSupportsUse(runtime, use) {
|
|
|
4730
1326
|
if (use === "dispatch") return isDispatchEligibleRuntime(runtime);
|
|
4731
1327
|
return isOrchestratorEligibleRuntime(runtime);
|
|
4732
1328
|
}
|
|
4733
|
-
function capabilityDecisions(runtime,
|
|
1329
|
+
function capabilityDecisions(runtime, capabilities) {
|
|
4734
1330
|
return {
|
|
4735
|
-
chat:
|
|
4736
|
-
tools:
|
|
4737
|
-
reasoning:
|
|
4738
|
-
vision:
|
|
1331
|
+
chat: capabilities.chat,
|
|
1332
|
+
tools: capabilities.tools,
|
|
1333
|
+
reasoning: capabilities.reasoning,
|
|
1334
|
+
vision: capabilities.vision,
|
|
4739
1335
|
streaming: streamingDecision(runtime),
|
|
4740
|
-
contextWindow:
|
|
4741
|
-
maxTokens:
|
|
1336
|
+
contextWindow: capabilities.contextWindow,
|
|
1337
|
+
maxTokens: capabilities.maxTokens
|
|
4742
1338
|
};
|
|
4743
1339
|
}
|
|
4744
|
-
function appendCapabilityDiagnostics(diagnostics, input,
|
|
1340
|
+
function appendCapabilityDiagnostics(diagnostics, input, capabilities, decisions, targetId) {
|
|
4745
1341
|
if (!decisions.chat) {
|
|
4746
1342
|
diagnostics.push(diagnostic("error", "chat-unsupported", `target '${targetId}' does not advertise chat support`));
|
|
4747
1343
|
}
|
|
@@ -4757,7 +1353,7 @@ function appendCapabilityDiagnostics(diagnostics, input, capabilities2, decision
|
|
|
4757
1353
|
);
|
|
4758
1354
|
}
|
|
4759
1355
|
for (const capability of input.requiredCapabilities ?? []) {
|
|
4760
|
-
if (!requiredCapabilitySupported(
|
|
1356
|
+
if (!requiredCapabilitySupported(capabilities, capability)) {
|
|
4761
1357
|
diagnostics.push(
|
|
4762
1358
|
diagnostic(
|
|
4763
1359
|
"error",
|
|
@@ -4874,12 +1470,12 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4874
1470
|
diagnostics: [diagnostic("error", "model-not-configured", `target '${targetId}' has no model configured`)]
|
|
4875
1471
|
};
|
|
4876
1472
|
}
|
|
4877
|
-
const status =
|
|
1473
|
+
const status = statusFor(providers, target, runtime, wireModelId);
|
|
4878
1474
|
const unknownModel = unknownModelDiagnostic(target, status, wireModelId);
|
|
4879
1475
|
if (unknownModel) diagnostics.push(unknownModel);
|
|
4880
1476
|
const requestedThinkingLevel = input.requestedThinkingLevel ?? "off";
|
|
4881
1477
|
const capabilityResolution = modelCapabilitiesFor(providers, status, wireModelId);
|
|
4882
|
-
const
|
|
1478
|
+
const capabilities = { ...capabilityResolution.capabilities };
|
|
4883
1479
|
const probedContextWindow = probeCapabilitiesForModel(status, wireModelId)?.contextWindow ?? null;
|
|
4884
1480
|
const loadedContextWindow = loadedContextWindowForModel(status, wireModelId);
|
|
4885
1481
|
const contextWindowDetails = resolveContextWindowDetails(
|
|
@@ -4892,7 +1488,7 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4892
1488
|
void 0,
|
|
4893
1489
|
contextSlotsForModel(status, wireModelId)
|
|
4894
1490
|
);
|
|
4895
|
-
|
|
1491
|
+
capabilities.contextWindow = contextWindowDetails.effectiveContextWindow;
|
|
4896
1492
|
if (contextWindowDetails.warning) {
|
|
4897
1493
|
diagnostics.push(diagnostic("warning", "context-window-low", contextWindowDetails.warning));
|
|
4898
1494
|
}
|
|
@@ -4903,12 +1499,12 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4903
1499
|
target,
|
|
4904
1500
|
runtime,
|
|
4905
1501
|
wireModelId,
|
|
4906
|
-
|
|
1502
|
+
capabilities,
|
|
4907
1503
|
providers.knowledgeBase,
|
|
4908
1504
|
requestedThinkingLevel
|
|
4909
1505
|
);
|
|
4910
|
-
const decisions = capabilityDecisions(runtime,
|
|
4911
|
-
appendCapabilityDiagnostics(diagnostics, input,
|
|
1506
|
+
const decisions = capabilityDecisions(runtime, capabilities);
|
|
1507
|
+
appendCapabilityDiagnostics(diagnostics, input, capabilities, decisions, targetId);
|
|
4912
1508
|
appendThinkingDiagnostics(diagnostics, modelRuntime, requestedThinkingLevel);
|
|
4913
1509
|
if (hasError(diagnostics)) return { ok: false, diagnostics };
|
|
4914
1510
|
const resolved = {
|
|
@@ -4924,7 +1520,7 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4924
1520
|
costProvenance: resolveCostProvenance(target, runtime.id, wireModelId),
|
|
4925
1521
|
requestedThinkingLevel,
|
|
4926
1522
|
effectiveThinkingLevel: modelRuntime.thinking.effectiveLevel,
|
|
4927
|
-
capabilities
|
|
1523
|
+
capabilities,
|
|
4928
1524
|
capabilityDecisions: decisions,
|
|
4929
1525
|
modelRuntime,
|
|
4930
1526
|
modelReasoningAuthoritative: capabilityResolution.reasoningAuthoritative,
|
|
@@ -4960,7 +1556,7 @@ function refineRuntimeTargetWithModelHints(target, model, knowledgeBase) {
|
|
|
4960
1556
|
const modelHintContextWindow = nonNegativeFiniteNumber(hintRecord?.contextWindow);
|
|
4961
1557
|
const windowHintDiffers = modelHintContextWindow !== void 0 && modelHintContextWindow > 0 && modelHintContextWindow !== target.contextWindowDetails.effectiveContextWindow;
|
|
4962
1558
|
if (Object.keys(patch).length === 0 && !windowHintDiffers) return target;
|
|
4963
|
-
const
|
|
1559
|
+
const capabilities = { ...target.capabilities, ...patch };
|
|
4964
1560
|
const contextWindowDetails = resolveContextWindowDetails(
|
|
4965
1561
|
target.target,
|
|
4966
1562
|
target.runtime,
|
|
@@ -4974,22 +1570,22 @@ function refineRuntimeTargetWithModelHints(target, model, knowledgeBase) {
|
|
|
4974
1570
|
modelHintContextWindow,
|
|
4975
1571
|
target.contextWindowDetails.contextWindowSlots
|
|
4976
1572
|
);
|
|
4977
|
-
|
|
1573
|
+
capabilities.contextWindow = contextWindowDetails.effectiveContextWindow;
|
|
4978
1574
|
const modelRuntime = resolveModelRuntimeCapabilities({
|
|
4979
1575
|
targetId: target.targetId,
|
|
4980
1576
|
runtimeId: target.runtimeId,
|
|
4981
1577
|
apiFamily: target.apiFamily,
|
|
4982
1578
|
modelId: target.wireModelId,
|
|
4983
|
-
capabilities
|
|
1579
|
+
capabilities,
|
|
4984
1580
|
...target.modelRuntime.quirks ? { quirks: target.modelRuntime.quirks } : {},
|
|
4985
1581
|
configuredThinkingLevel: target.requestedThinkingLevel
|
|
4986
1582
|
});
|
|
4987
|
-
const decisions = capabilityDecisions(target.runtime,
|
|
1583
|
+
const decisions = capabilityDecisions(target.runtime, capabilities);
|
|
4988
1584
|
const diagnostics = withoutStaleRuntimeDiagnostics(target.diagnostics, decisions);
|
|
4989
1585
|
appendThinkingDiagnostics(diagnostics, modelRuntime, target.requestedThinkingLevel);
|
|
4990
1586
|
return {
|
|
4991
1587
|
...target,
|
|
4992
|
-
capabilities
|
|
1588
|
+
capabilities,
|
|
4993
1589
|
capabilityDecisions: decisions,
|
|
4994
1590
|
modelRuntime,
|
|
4995
1591
|
effectiveThinkingLevel: modelRuntime.thinking.effectiveLevel,
|
|
@@ -5766,7 +2362,7 @@ var ollamaNativeApiProvider = {
|
|
|
5766
2362
|
|
|
5767
2363
|
// src/engine/apis/openai-completions.ts
|
|
5768
2364
|
init_esm_shims();
|
|
5769
|
-
import { TextDecoder
|
|
2365
|
+
import { TextDecoder } from "node:util";
|
|
5770
2366
|
import {
|
|
5771
2367
|
createAssistantMessageEventStream as createAssistantMessageEventStream3
|
|
5772
2368
|
} from "@earendil-works/pi-ai";
|
|
@@ -5893,30 +2489,30 @@ function harmonyPrefixTailLength(value) {
|
|
|
5893
2489
|
|
|
5894
2490
|
// src/engine/apis/llamacpp-residency.ts
|
|
5895
2491
|
init_esm_shims();
|
|
5896
|
-
import { performance
|
|
2492
|
+
import { performance } from "node:perf_hooks";
|
|
5897
2493
|
var LOAD_TIMEOUT_MS = 12e4;
|
|
5898
2494
|
var POLL_INTERVAL_MS = 500;
|
|
5899
2495
|
function isResidentState(state) {
|
|
5900
2496
|
return state === "loaded" || state === "loading" || state === "sleeping";
|
|
5901
2497
|
}
|
|
5902
|
-
function
|
|
2498
|
+
function isRecord(value) {
|
|
5903
2499
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
5904
2500
|
}
|
|
5905
2501
|
function routerModelState(entry) {
|
|
5906
2502
|
const status = entry.status;
|
|
5907
|
-
const state =
|
|
2503
|
+
const state = isRecord(status) ? status.value : status;
|
|
5908
2504
|
if (state === "loaded" || state === "loading" || state === "sleeping" || state === "unloaded" || state === "failed") {
|
|
5909
2505
|
return state;
|
|
5910
2506
|
}
|
|
5911
2507
|
return "unknown";
|
|
5912
2508
|
}
|
|
5913
2509
|
function parseRouterModels(payload) {
|
|
5914
|
-
if (!
|
|
2510
|
+
if (!isRecord(payload)) return [];
|
|
5915
2511
|
const data = payload.data;
|
|
5916
2512
|
if (!Array.isArray(data)) return [];
|
|
5917
2513
|
const models = [];
|
|
5918
2514
|
for (const entry of data) {
|
|
5919
|
-
if (!
|
|
2515
|
+
if (!isRecord(entry) || typeof entry.id !== "string") continue;
|
|
5920
2516
|
const tags = Array.isArray(entry.tags) ? entry.tags.filter((tag) => typeof tag === "string") : [];
|
|
5921
2517
|
models.push({ id: entry.id, state: routerModelState(entry), tags });
|
|
5922
2518
|
}
|
|
@@ -5926,14 +2522,14 @@ function residentModel(model) {
|
|
|
5926
2522
|
return isResidentState(model.state);
|
|
5927
2523
|
}
|
|
5928
2524
|
function parseLlamaCppRouterProps(payload) {
|
|
5929
|
-
if (!
|
|
2525
|
+
if (!isRecord(payload)) return {};
|
|
5930
2526
|
const maxInstances = payload.max_instances;
|
|
5931
2527
|
return typeof maxInstances === "number" && Number.isFinite(maxInstances) && maxInstances > 0 ? { maxInstances: Math.floor(maxInstances) } : {};
|
|
5932
2528
|
}
|
|
5933
2529
|
function rootUrl(baseUrl) {
|
|
5934
2530
|
return baseUrl.replace(/\/+$/, "").replace(/\/v1$/, "");
|
|
5935
2531
|
}
|
|
5936
|
-
function
|
|
2532
|
+
function modelsUrl(baseUrl) {
|
|
5937
2533
|
return `${baseUrl.replace(/\/+$/, "")}/models`;
|
|
5938
2534
|
}
|
|
5939
2535
|
function propsUrl(baseUrl) {
|
|
@@ -5957,25 +2553,25 @@ async function fetchRouterProps(input, fetchImpl) {
|
|
|
5957
2553
|
}
|
|
5958
2554
|
}
|
|
5959
2555
|
async function fetchRouterModels(input, fetchImpl) {
|
|
5960
|
-
const response = await fetchImpl(
|
|
2556
|
+
const response = await fetchImpl(modelsUrl(input.baseUrl), {
|
|
5961
2557
|
signal: AbortSignal.timeout(input.timeoutMs ?? 1500)
|
|
5962
2558
|
});
|
|
5963
2559
|
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
|
5964
2560
|
return parseRouterModels(await response.json());
|
|
5965
2561
|
}
|
|
5966
|
-
async function postRouterModel(fetchImpl,
|
|
5967
|
-
const response = await fetchImpl(
|
|
2562
|
+
async function postRouterModel(fetchImpl, url, modelId) {
|
|
2563
|
+
const response = await fetchImpl(url, {
|
|
5968
2564
|
method: "POST",
|
|
5969
2565
|
headers: { "content-type": "application/json" },
|
|
5970
2566
|
body: JSON.stringify({ model: modelId }),
|
|
5971
2567
|
signal: AbortSignal.timeout(LOAD_TIMEOUT_MS)
|
|
5972
2568
|
});
|
|
5973
2569
|
if (response.ok) return;
|
|
5974
|
-
throw new Error(`llama.cpp router rejected ${
|
|
2570
|
+
throw new Error(`llama.cpp router rejected ${url}: HTTP ${response.status}`);
|
|
5975
2571
|
}
|
|
5976
2572
|
async function waitForLoaded(input, fetchImpl, modelId) {
|
|
5977
|
-
const started =
|
|
5978
|
-
while (
|
|
2573
|
+
const started = performance.now();
|
|
2574
|
+
while (performance.now() - started < LOAD_TIMEOUT_MS) {
|
|
5979
2575
|
const models = await fetchRouterModels(input, fetchImpl);
|
|
5980
2576
|
const model = models.find((entry) => entry.id === modelId);
|
|
5981
2577
|
if (model?.state === "loaded" || model?.state === "sleeping") return;
|
|
@@ -6112,7 +2708,7 @@ function targetForModel(model) {
|
|
|
6112
2708
|
...info.lmstudio ? { lmstudio: info.lmstudio } : {}
|
|
6113
2709
|
};
|
|
6114
2710
|
}
|
|
6115
|
-
function
|
|
2711
|
+
function resolveModel(catalog, id) {
|
|
6116
2712
|
for (const model of catalog.models) {
|
|
6117
2713
|
if (model.key === id) return { model, ...model.loadedInstances[0] ? { instance: model.loadedInstances[0] } : {} };
|
|
6118
2714
|
const instance = model.loadedInstances.find((entry) => entry.id === id);
|
|
@@ -6207,7 +2803,7 @@ async function ensureLmStudioResidency(model, options = {}) {
|
|
|
6207
2803
|
}
|
|
6208
2804
|
if (!load || Object.keys(load).length === 0 || resolution.instance) return resolution.wireModelId;
|
|
6209
2805
|
if (catalog.tier !== "0.4+") return resolution.wireModelId;
|
|
6210
|
-
const selected =
|
|
2806
|
+
const selected = resolveModel(catalog, model.id);
|
|
6211
2807
|
const modelKey = selected?.model.key ?? model.id;
|
|
6212
2808
|
let instances = catalog.models.flatMap(
|
|
6213
2809
|
(entry) => entry.loadedInstances.map((instance) => ({ modelKey: entry.key, identifier: instance.id, instance }))
|
|
@@ -6355,7 +2951,7 @@ function withResponseModelIdCapture(options, sourceFactory) {
|
|
|
6355
2951
|
observed: false,
|
|
6356
2952
|
done: false,
|
|
6357
2953
|
buffer: "",
|
|
6358
|
-
decoder: new
|
|
2954
|
+
decoder: new TextDecoder()
|
|
6359
2955
|
};
|
|
6360
2956
|
const fetchImpl = options.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
6361
2957
|
const capturedOptions = {
|
|
@@ -6652,12 +3248,12 @@ function estimateReasoningTokens(content) {
|
|
|
6652
3248
|
if (chars === 0) return 0;
|
|
6653
3249
|
return Math.max(1, Math.round(chars / REASONING_CHARS_PER_TOKEN2));
|
|
6654
3250
|
}
|
|
6655
|
-
function
|
|
3251
|
+
function positiveNumber(value) {
|
|
6656
3252
|
return typeof value === "number" && Number.isFinite(value) && value > 0;
|
|
6657
3253
|
}
|
|
6658
3254
|
function hasReportedReasoningUsage(usage) {
|
|
6659
3255
|
const aliases = usage;
|
|
6660
|
-
return
|
|
3256
|
+
return positiveNumber(aliases.reasoning) || positiveNumber(aliases.reasoningTokens) || positiveNumber(aliases.reasoning_tokens);
|
|
6661
3257
|
}
|
|
6662
3258
|
function applyOpenAICompatReasoningEstimate(message) {
|
|
6663
3259
|
if (hasReportedReasoningUsage(message.usage)) return;
|
|
@@ -6929,14 +3525,14 @@ function registerClioApiProviders() {
|
|
|
6929
3525
|
|
|
6930
3526
|
// src/domains/providers/knowledge-base-path.ts
|
|
6931
3527
|
init_esm_shims();
|
|
6932
|
-
import { existsSync, statSync
|
|
6933
|
-
import { delimiter, dirname, join as
|
|
3528
|
+
import { existsSync, statSync } from "node:fs";
|
|
3529
|
+
import { delimiter, dirname, join as join2 } from "node:path";
|
|
6934
3530
|
import { fileURLToPath } from "node:url";
|
|
6935
3531
|
var MODEL_CATALOG_OVERLAY_DIR = "model-catalog.d";
|
|
6936
3532
|
var MODEL_CATALOG_DIRS_ENV = "CLIO_CODER_MODEL_CATALOG_DIRS";
|
|
6937
3533
|
function isDirectory(path) {
|
|
6938
3534
|
try {
|
|
6939
|
-
return
|
|
3535
|
+
return statSync(path).isDirectory();
|
|
6940
3536
|
} catch {
|
|
6941
3537
|
return false;
|
|
6942
3538
|
}
|
|
@@ -6944,18 +3540,18 @@ function isDirectory(path) {
|
|
|
6944
3540
|
function resolveProvidersModelsDir(importMetaUrl) {
|
|
6945
3541
|
const start = dirname(fileURLToPath(importMetaUrl));
|
|
6946
3542
|
const directCandidates = [
|
|
6947
|
-
|
|
6948
|
-
|
|
6949
|
-
|
|
3543
|
+
join2(start, "models"),
|
|
3544
|
+
join2(start, "..", "domains", "providers", "models"),
|
|
3545
|
+
join2(start, "..", "providers-models")
|
|
6950
3546
|
];
|
|
6951
3547
|
for (const candidate of directCandidates) {
|
|
6952
3548
|
if (isDirectory(candidate)) return candidate;
|
|
6953
3549
|
}
|
|
6954
3550
|
let cursor = start;
|
|
6955
3551
|
for (let i = 0; i < 8; i++) {
|
|
6956
|
-
const packageJson =
|
|
6957
|
-
const sourceModels =
|
|
6958
|
-
const distModels =
|
|
3552
|
+
const packageJson = join2(cursor, "package.json");
|
|
3553
|
+
const sourceModels = join2(cursor, "src", "domains", "providers", "models");
|
|
3554
|
+
const distModels = join2(cursor, "dist", "providers-models");
|
|
6959
3555
|
if (existsSync(packageJson)) {
|
|
6960
3556
|
if (isDirectory(sourceModels)) return sourceModels;
|
|
6961
3557
|
if (isDirectory(distModels)) return distModels;
|
|
@@ -6986,8 +3582,8 @@ function resolveProviderModelCatalogDirs(importMetaUrl, options = {}) {
|
|
|
6986
3582
|
const bundled = resolveProvidersModelsDir(importMetaUrl);
|
|
6987
3583
|
const cwd = options.cwd ?? process.cwd();
|
|
6988
3584
|
const overlays = uniqueExistingDirs([
|
|
6989
|
-
|
|
6990
|
-
|
|
3585
|
+
join2(resolveClioDirs().config, MODEL_CATALOG_OVERLAY_DIR),
|
|
3586
|
+
join2(cwd, ".clio-coder", MODEL_CATALOG_OVERLAY_DIR),
|
|
6991
3587
|
...envOverlayDirs()
|
|
6992
3588
|
]);
|
|
6993
3589
|
return {
|
|
@@ -7006,7 +3602,7 @@ function resolveProviderKnowledgeBaseRoots(importMetaUrl, options = {}) {
|
|
|
7006
3602
|
|
|
7007
3603
|
// src/domains/providers/plugins.ts
|
|
7008
3604
|
init_esm_shims();
|
|
7009
|
-
import { join as
|
|
3605
|
+
import { join as join3 } from "node:path";
|
|
7010
3606
|
function extractPluginPackages(settings) {
|
|
7011
3607
|
if (!settings || typeof settings !== "object") return [];
|
|
7012
3608
|
const raw = settings.runtimePlugins;
|
|
@@ -7015,7 +3611,7 @@ function extractPluginPackages(settings) {
|
|
|
7015
3611
|
}
|
|
7016
3612
|
async function loadPluginRuntimes(registry, settings) {
|
|
7017
3613
|
const loaded = [];
|
|
7018
|
-
const pluginDir =
|
|
3614
|
+
const pluginDir = join3(clioConfigDir(), "runtimes");
|
|
7019
3615
|
const packages = extractPluginPackages(settings);
|
|
7020
3616
|
try {
|
|
7021
3617
|
const ids = await registry.loadFromDir(pluginDir, activateExternalPluginApiBridge);
|
|
@@ -7043,8 +3639,8 @@ async function loadPluginRuntimes(registry, settings) {
|
|
|
7043
3639
|
// src/domains/providers/types/knowledge-base.ts
|
|
7044
3640
|
init_esm_shims();
|
|
7045
3641
|
var import_yaml = __toESM(require_dist(), 1);
|
|
7046
|
-
import { readdirSync
|
|
7047
|
-
import { join as
|
|
3642
|
+
import { readdirSync, readFileSync, statSync as statSync2 } from "node:fs";
|
|
3643
|
+
import { join as join4 } from "node:path";
|
|
7048
3644
|
var FileKnowledgeBase = class {
|
|
7049
3645
|
roots;
|
|
7050
3646
|
loaded = [];
|
|
@@ -7098,7 +3694,7 @@ function normalizeRoots(root) {
|
|
|
7098
3694
|
if (dir.length === 0 || seen.has(dir)) continue;
|
|
7099
3695
|
let isRootDir = false;
|
|
7100
3696
|
try {
|
|
7101
|
-
isRootDir =
|
|
3697
|
+
isRootDir = statSync2(dir).isDirectory();
|
|
7102
3698
|
} catch (err) {
|
|
7103
3699
|
if (raw.optional === true) continue;
|
|
7104
3700
|
throw err;
|
|
@@ -7114,9 +3710,9 @@ function normalizeRoots(root) {
|
|
|
7114
3710
|
}
|
|
7115
3711
|
function collectYamlFiles(dir, prefix = "") {
|
|
7116
3712
|
const out = [];
|
|
7117
|
-
const entries =
|
|
3713
|
+
const entries = readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name));
|
|
7118
3714
|
for (const entry of entries) {
|
|
7119
|
-
const path =
|
|
3715
|
+
const path = join4(dir, entry.name);
|
|
7120
3716
|
const name = prefix ? `${prefix}/${entry.name}` : entry.name;
|
|
7121
3717
|
if (entry.isDirectory()) {
|
|
7122
3718
|
out.push(...collectYamlFiles(path, name));
|
|
@@ -7135,20 +3731,20 @@ function normalizeEntry(raw, file) {
|
|
|
7135
3731
|
const candidate = raw;
|
|
7136
3732
|
const family = candidate.family;
|
|
7137
3733
|
const patterns = candidate.matchPatterns;
|
|
7138
|
-
const
|
|
3734
|
+
const capabilities = candidate.capabilities;
|
|
7139
3735
|
if (typeof family !== "string" || family.length === 0) {
|
|
7140
3736
|
throw new Error(`knowledge base file ${file}: entry is missing 'family' string`);
|
|
7141
3737
|
}
|
|
7142
3738
|
if (!Array.isArray(patterns) || patterns.some((p) => typeof p !== "string")) {
|
|
7143
3739
|
throw new Error(`knowledge base file ${file}: entry '${family}' needs matchPatterns: string[]`);
|
|
7144
3740
|
}
|
|
7145
|
-
if (typeof
|
|
3741
|
+
if (typeof capabilities !== "object" || capabilities === null || Array.isArray(capabilities)) {
|
|
7146
3742
|
throw new Error(`knowledge base file ${file}: entry '${family}' needs capabilities object`);
|
|
7147
3743
|
}
|
|
7148
3744
|
const entry = {
|
|
7149
3745
|
family,
|
|
7150
3746
|
matchPatterns: patterns,
|
|
7151
|
-
capabilities
|
|
3747
|
+
capabilities
|
|
7152
3748
|
};
|
|
7153
3749
|
if (candidate.quirks !== void 0) {
|
|
7154
3750
|
if (typeof candidate.quirks !== "object" || candidate.quirks === null || Array.isArray(candidate.quirks)) {
|
|
@@ -7198,9 +3794,9 @@ function capabilitiesFor(desc, target, probe, kb) {
|
|
|
7198
3794
|
const base = capabilitiesFromCatalogModel(desc.defaultCapabilities, catalogModel);
|
|
7199
3795
|
const probeCaps = probeCapabilitiesForModel({ target, ...probe }, target.defaultModel);
|
|
7200
3796
|
const userOverride = target.capabilities ?? null;
|
|
7201
|
-
const
|
|
3797
|
+
const capabilities = mergeCapabilities(base, kbHit?.entry.capabilities ?? null, probeCaps, userOverride);
|
|
7202
3798
|
return {
|
|
7203
|
-
capabilities
|
|
3799
|
+
capabilities,
|
|
7204
3800
|
contextWindowProvenance: contextWindowProvenanceOf(kbHit, catalogModel, probeCaps, userOverride)
|
|
7205
3801
|
};
|
|
7206
3802
|
}
|
|
@@ -7255,6 +3851,15 @@ function mergeProbeResult(desc, target, probe, previous) {
|
|
|
7255
3851
|
if (probeSurfaces && Object.keys(probeSurfaces).length > 0) merge.probeSurfaces = probeSurfaces;
|
|
7256
3852
|
return merge;
|
|
7257
3853
|
}
|
|
3854
|
+
function unservedDefaultModelReason(desc, target, merge) {
|
|
3855
|
+
const model = target.defaultModel;
|
|
3856
|
+
if (!model) return null;
|
|
3857
|
+
if (merge.discoveredModelsSource !== "probe" || merge.discoveredModels.length === 0) return null;
|
|
3858
|
+
if (listKnownModelsForRuntime(desc.id).length > 0) return null;
|
|
3859
|
+
if (merge.discoveredModels.includes(model)) return null;
|
|
3860
|
+
if (merge.discoveredModelStates && model in merge.discoveredModelStates) return null;
|
|
3861
|
+
return `default model '${model}' is not advertised by the target`;
|
|
3862
|
+
}
|
|
7258
3863
|
function createProvidersBundle(context) {
|
|
7259
3864
|
const registry = getRuntimeRegistry();
|
|
7260
3865
|
const authStore = openAuthStorage();
|
|
@@ -7330,12 +3935,13 @@ function createProvidersBundle(context) {
|
|
|
7330
3935
|
}
|
|
7331
3936
|
const availability = availabilityFor(desc, target, authStatusFor);
|
|
7332
3937
|
const merge = mergeProbeResult(desc, target, probe, previous);
|
|
7333
|
-
const { capabilities
|
|
3938
|
+
const { capabilities, contextWindowProvenance } = capabilitiesFor(desc, target, merge, kb);
|
|
7334
3939
|
const healthy = probe !== null ? probe.ok : null;
|
|
3940
|
+
const unservedDefault = probe?.ok ? unservedDefaultModelReason(desc, target, merge) : null;
|
|
7335
3941
|
const health = probe === null ? previous?.health ?? emptyHealth() : {
|
|
7336
|
-
status: healthy ? "healthy" : "down",
|
|
3942
|
+
status: healthy ? unservedDefault === null ? "healthy" : "degraded" : "down",
|
|
7337
3943
|
lastCheckAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
7338
|
-
lastError: probe.error ??
|
|
3944
|
+
lastError: probe.error ?? unservedDefault,
|
|
7339
3945
|
latencyMs: probe.latencyMs ?? null
|
|
7340
3946
|
};
|
|
7341
3947
|
const available = availability.available && (probe === null || probe.ok);
|
|
@@ -7346,7 +3952,7 @@ function createProvidersBundle(context) {
|
|
|
7346
3952
|
available,
|
|
7347
3953
|
reason,
|
|
7348
3954
|
health,
|
|
7349
|
-
capabilities
|
|
3955
|
+
capabilities,
|
|
7350
3956
|
contextWindowProvenance,
|
|
7351
3957
|
probeCapabilities: merge.probeCapabilities,
|
|
7352
3958
|
probeModelCapabilities: merge.probeModelCapabilities,
|
|
@@ -7690,12 +4296,6 @@ var ProvidersDomainModule = {
|
|
|
7690
4296
|
};
|
|
7691
4297
|
|
|
7692
4298
|
export {
|
|
7693
|
-
supportsAgentRoleTools,
|
|
7694
|
-
AGENT_ROLE_TOOLS_REQUIRED_REASON,
|
|
7695
|
-
resolveEffectivePricing,
|
|
7696
|
-
getCatalogModelForRuntime,
|
|
7697
|
-
getRuntimeRegistry,
|
|
7698
|
-
credentialsPresent,
|
|
7699
4299
|
setGlobalDefaultMaxOutputTokens,
|
|
7700
4300
|
resolveReservedOutputTokens,
|
|
7701
4301
|
setResidencyNoticeSink,
|
|
@@ -7710,22 +4310,12 @@ export {
|
|
|
7710
4310
|
thinkingLevelFromChoiceLabel,
|
|
7711
4311
|
resolveModelRuntimeCapabilitiesForProviders,
|
|
7712
4312
|
resolveModelRuntimeCapabilitiesForModel,
|
|
7713
|
-
greetLmStudio,
|
|
7714
4313
|
registerClioApiProviders,
|
|
7715
4314
|
resolveProviderKnowledgeBaseRoots,
|
|
7716
4315
|
loadPluginRuntimes,
|
|
7717
|
-
formatContextWindowSlots,
|
|
7718
|
-
registerBuiltinRuntimes,
|
|
7719
4316
|
FileKnowledgeBase,
|
|
7720
4317
|
isOrchestratorEligibleRuntime,
|
|
7721
4318
|
isDispatchEligibleRuntime,
|
|
7722
|
-
supportGroupLabel,
|
|
7723
|
-
listKnownModelsForRuntime,
|
|
7724
|
-
describeRuntimeModels,
|
|
7725
|
-
buildProviderSupportEntry,
|
|
7726
|
-
listProviderSupportEntries,
|
|
7727
|
-
configuredTargetsForRuntime,
|
|
7728
|
-
resolveProviderReference,
|
|
7729
4319
|
modelResidencyForStatus,
|
|
7730
4320
|
modelCandidatesForStatus,
|
|
7731
4321
|
modelIdsForStatus,
|
|
@@ -7740,4 +4330,4 @@ export {
|
|
|
7740
4330
|
normalizeCostProvenance,
|
|
7741
4331
|
ProvidersDomainModule
|
|
7742
4332
|
};
|
|
7743
|
-
//# sourceMappingURL=chunk-
|
|
4333
|
+
//# sourceMappingURL=chunk-7RFXX52T.js.map
|