@iowarp/clio-coder 0.3.7 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/README.md +13 -4
- package/dist/{acp-SK4MD6MM.js → acp-7LOELQFP.js} +13 -13
- package/dist/{agents-2FN2K6ME.js → agents-FIBG2SHA.js} +41 -37
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-QIYZWM5I.js → auth-OI4LIH2I.js} +31 -24
- package/dist/builtins-AD25UL3C.js +17 -0
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-EOOQZZDE.js → chunk-3DUR4WUA.js} +19 -19
- package/dist/{chunk-WHJYKASB.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-EBEFWSGL.js → chunk-3UUY7R3Z.js} +14 -10
- package/dist/{chunk-LADCF22A.js → chunk-3V5AYSEQ.js} +113 -54
- package/dist/{chunk-BMWK7ZIZ.js → chunk-465CC7FK.js} +16 -13
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-CEYBNUGC.js → chunk-4H6ULJ3H.js} +378 -36
- package/dist/{chunk-YTYFXUI3.js → chunk-4LJX2PUC.js} +9 -9
- package/dist/{chunk-DOOEX22V.js → chunk-56KB5IJP.js} +5 -5
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-TSHXZTOQ.js → chunk-5HFBWUMU.js} +23 -11
- package/dist/{chunk-5UJ6ECTS.js → chunk-5PVQ4SRS.js} +80 -8
- package/dist/{chunk-ZWLZP4ZT.js → chunk-5QKCQQ3E.js} +359 -17
- package/dist/{chunk-6M7VS3J3.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/chunk-774ILSRL.js +172 -0
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-GH5622CP.js → chunk-A2NJGIB3.js} +2 -2
- package/dist/{chunk-C4JBQ5SR.js → chunk-AD7Y7STJ.js} +6 -6
- package/dist/{chunk-GEYXPTRF.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-2SFS6XQE.js → chunk-AMKHQW3C.js} +3 -2
- package/dist/{chunk-D4MDIG46.js → chunk-B5CSFE7B.js} +7 -7
- package/dist/{chunk-MXI6J5JF.js → chunk-B5XRQOLB.js} +10 -10
- package/dist/{chunk-X2KV5FXT.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-JNXPYBB4.js → chunk-CA42X6KT.js} +3 -3
- package/dist/{chunk-VREKEFLL.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-JTSEDYVQ.js → chunk-DG4M6ZUE.js} +7 -7
- package/dist/{chunk-DQA7QLMD.js → chunk-EBOC7MT3.js} +10 -25
- package/dist/{chunk-KZ2H5X4G.js → chunk-ECUO3KDP.js} +129 -14
- package/dist/{chunk-JRIO5UD2.js → chunk-EQ63NRB7.js} +5 -5
- package/dist/{chunk-YD734TPH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-XEGB6BCN.js → chunk-GAYUJ7LE.js} +68 -14
- package/dist/{chunk-UND3GU2L.js → chunk-H7IXIC72.js} +2 -2
- package/dist/{chunk-IR4CFBFN.js → chunk-HAY4ZE2P.js} +12 -12
- package/dist/{chunk-UVDSQ6LW.js → chunk-HCBCAYZU.js} +74 -147
- package/dist/{chunk-4DWFMQDR.js → chunk-HJB5IUKP.js} +89 -145
- package/dist/{chunk-M4AKACEO.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-KCMKRQX4.js → chunk-HPCTNZM2.js} +45 -82
- package/dist/{chunk-465YSENW.js → chunk-IFBNV6H6.js} +3 -3
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/chunk-JEQQR47K.js +3025 -0
- package/dist/{chunk-FO5ZOVUY.js → chunk-KV2AOLDF.js} +27 -7
- package/dist/chunk-LU7P4LHA.js +33 -0
- package/dist/{chunk-6TUKSZVF.js → chunk-LXPJXFM5.js} +11 -11
- package/dist/{chunk-VQNODYQ4.js → chunk-MIX5N5AC.js} +488 -3668
- package/dist/chunk-MLOK6ZOS.js +2888 -0
- package/dist/{chunk-ZZMN5OM4.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-OBMAI2DP.js → chunk-N3PBVRTZ.js} +12 -388
- package/dist/{chunk-WJHBC77E.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-UFQ3F4FW.js → chunk-NQ6UCCOD.js} +4 -4
- package/dist/chunk-NUGM5KR6.js +165 -0
- package/dist/{chunk-DMD2AGVS.js → chunk-NZU6YDNV.js} +20 -18
- package/dist/{chunk-WHGPSPT5.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-PD3MESLB.js → chunk-P3JGPQFL.js} +4 -4
- package/dist/{chunk-UHXRNZ2J.js → chunk-PNY46YEY.js} +23 -6
- package/dist/{chunk-THKY7CD7.js → chunk-PZ4I4JE2.js} +134 -29
- package/dist/{chunk-SROCI7ZU.js → chunk-QQ7EKM72.js} +5 -5
- package/dist/{chunk-QCTRSGHQ.js → chunk-R7LNVMCS.js} +91 -53
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-OB5HIGJY.js → chunk-RKRLDWD3.js} +4 -1
- package/dist/{chunk-DJNLUABN.js → chunk-S4COXYBG.js} +588 -32
- package/dist/{chunk-3HAPLH5M.js → chunk-T3Z6VAAF.js} +172 -11
- package/dist/{chunk-FOT2FX5J.js → chunk-TD7UE2L5.js} +12 -10
- package/dist/{chunk-UUANF5CR.js → chunk-TEO2TLVN.js} +856 -967
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-EELBMBT6.js → chunk-VKBMFOYV.js} +74 -15
- package/dist/chunk-VO2LKSTM.js +165 -0
- package/dist/{chunk-5C77SEEY.js → chunk-VPTUJU4P.js} +3 -3
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-J7PIKKWC.js → chunk-WXCJ7VME.js} +8 -8
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
- package/dist/{chunk-AB4XIIVB.js → chunk-YKOFT37S.js} +6 -6
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +32 -32
- package/dist/{clio-WBVQEBKO.js → clio-LT5V7SSZ.js} +9 -9
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +5 -5
- package/dist/codewiki/build-worker.js +4 -4
- package/dist/{components-F7OEATSO.js → components-ZFA3SAER.js} +8 -8
- package/dist/{config-TRBL3RCF.js → config-RXS5T3JT.js} +98 -65
- package/dist/{configure-OLCVPHNM.js → configure-2WYWSCSD.js} +26 -22
- package/dist/{context-MJIJ6GOX.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-XEWE3MOJ.js → context-MVOORGMF.js} +54 -47
- package/dist/{context-WFPKQSM6.js → context-PALKKQYL.js} +28 -28
- package/dist/{context-clear-KNOS2JPB.js → context-clear-N2WOYZ2K.js} +53 -46
- package/dist/{context-index-SSR5ECNE.js → context-index-HNG3MOME.js} +6 -6
- package/dist/{context-working-set-EUXAZI6N.js → context-working-set-MIEVECVZ.js} +17 -18
- package/dist/{dispatch-runner-B7MTOVKL.js → dispatch-runner-VVA4SRRH.js} +90 -61
- package/dist/{docs-FLJTIDSE.js → docs-7LQ23DLM.js} +8 -8
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/eval-IJ5VEZDJ.js +4483 -0
- package/dist/{evidence-JZNBUOQZ.js → evidence-L5APPXNV.js} +68 -61
- package/dist/{evolve-FJVC4KKI.js → evolve-RGNKFJ52.js} +47 -40
- package/dist/{extensions-IQL36S7K.js → extensions-7WYWUX5A.js} +13 -7
- package/dist/{fleet-BDKYJFCP.js → fleet-6CNVBZZP.js} +113 -76
- package/dist/{fleet-commands-ZFIWZSB3.js → fleet-commands-L2SXSYEI.js} +10 -10
- package/dist/{fleet-graph-Y6HPXIVF.js → fleet-graph-2J3OOIPO.js} +17 -15
- package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-CZRJ4JP5.js} +5 -6
- package/dist/{fleet-validate-BIYREGIK.js → fleet-validate-C5RI6DP7.js} +20 -19
- package/dist/{init-LQUB5COQ.js → init-VBN2ACVA.js} +70 -63
- package/dist/{library-NJAHIGG4.js → library-JHGUMLY2.js} +22 -20
- package/dist/{memory-OG6HOYKM.js → memory-K4OQIYWG.js} +49 -42
- package/dist/{models-5ZG5XY7J.js → models-2NCZUWDD.js} +35 -29
- package/dist/{monitor-TJ7AMTGB.js → monitor-MMVTJABD.js} +64 -45
- package/dist/{orchestrator-WZYB54DM.js → orchestrator-ZKBPCHW6.js} +1971 -520
- package/dist/{paths-XUC7GS6E.js → paths-DBXMZMDU.js} +5 -5
- package/dist/registry-LG64LTF4.js +11 -0
- package/dist/{reset-PXQT45IY.js → reset-DD5JGOY3.js} +11 -11
- package/dist/{run-FQ74YF62.js → run-QEGNX7FL.js} +89 -83
- package/dist/{share-FW7SVCL3.js → share-JKD3BQMW.js} +20 -18
- package/dist/{skills-7E7IRB3R.js → skills-LMQIKDOZ.js} +23 -21
- package/dist/{skills-eval-LI75W6OK.js → skills-eval-I7X2774U.js} +59 -52
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/support-I7LOJLIF.js +38 -0
- package/dist/{targets-4CIFKCTW.js → targets-RUSR6B5Z.js} +77 -42
- package/dist/{terminal-lease-WUZY7ZV5.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{uninstall-7FV7IP4E.js → uninstall-ZJF5H5ZN.js} +8 -8
- package/dist/{upgrade-K2HVIVMQ.js → upgrade-XANW3FXB.js} +29 -26
- package/dist/{usage-GTZELZQX.js → usage-4H7ZRXQT.js} +110 -61
- package/dist/{verifiers-RLAHT27O.js → verifiers-UZXNBZEB.js} +13 -13
- package/dist/{verify-BX3BRKH5.js → verify-BVKWTNDL.js} +9 -9
- package/dist/{wiki-generate-ASIFASCN.js → wiki-generate-MY7WV2QI.js} +76 -69
- package/dist/worker/entry.js +69 -66
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +11 -5
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +2 -2
- package/docs/configuration-and-targets.md +37 -5
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +77 -12
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +34 -9
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +54 -3
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +22 -2
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +60 -41
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -2
- package/src/cli/agents.ts +1 -1
- package/src/cli/argv.ts +5 -0
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/configure.ts +107 -23
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +7 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/evidence.ts +30 -25
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet-preflight.ts +2 -12
- package/src/cli/fleet.ts +32 -3
- package/src/cli/shared.ts +1 -0
- package/src/cli/targets.ts +45 -11
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/cli/validate-model.ts +60 -5
- package/src/core/bus-events.ts +54 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/commit-attribution.ts +4 -4
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/path-boundary.ts +100 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/agents/extension.ts +2 -11
- package/src/domains/agents/fleet-contract.ts +30 -12
- package/src/domains/agents/recipe.ts +7 -1
- package/src/domains/agents/registry.ts +73 -5
- package/src/domains/agents/result-contract.ts +128 -17
- package/src/domains/agents/write-boundary.ts +15 -50
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/context/project-rules.ts +51 -1
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/assignment-reconcile.ts +22 -5
- package/src/domains/dispatch/assignment-store.ts +151 -14
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +26 -1
- package/src/domains/dispatch/delegation-plan.ts +2 -5
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/execution-role.ts +9 -1
- package/src/domains/dispatch/extension.ts +309 -96
- package/src/domains/dispatch/fleet-run.ts +78 -4
- package/src/domains/dispatch/gate-role-prompts.ts +38 -0
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +6 -1
- package/src/domains/dispatch/intent-requirements.ts +40 -0
- package/src/domains/dispatch/intent.ts +84 -8
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/path-scope.ts +370 -0
- package/src/domains/dispatch/receipt-integrity.ts +2 -1
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/types.ts +14 -7
- package/src/domains/dispatch/validation.ts +6 -3
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +62 -0
- package/src/domains/dispatch/write-boundary.ts +262 -22
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/evidence.ts +79 -2
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +139 -2
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +68 -21
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/index.ts +21 -0
- package/src/domains/evidence/provenance.ts +46 -11
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +274 -0
- package/src/domains/evidence/trust-status.ts +145 -17
- package/src/domains/evidence/types.ts +4 -0
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +126 -4
- package/src/domains/extensions/resources.ts +21 -9
- package/src/domains/extensions/state.ts +18 -5
- package/src/domains/extensions/types.ts +6 -1
- package/src/domains/lifecycle/doctor.ts +209 -2
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/prompts/contract.ts +3 -5
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/extension.ts +30 -2
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/common-loader.ts +3 -0
- package/src/domains/resources/prompts/loader.ts +184 -17
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/policy-engine.ts +5 -5
- package/src/domains/safety/run-effects.ts +96 -2
- package/src/domains/safety/scope.ts +7 -12
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/acp/server.ts +4 -1
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/prompt-templates.ts +18 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/engine/worker-runtime.ts +6 -3
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +127 -5
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-event-projection.ts +14 -0
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +176 -48
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/slash-commands.ts +7 -2
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/artifacts.ts +42 -9
- package/src/interactive/view/view-overlay.ts +43 -6
- package/src/interactive/worker-receipts.ts +14 -2
- package/src/interactive/worker-stream.ts +8 -0
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-admission.ts +12 -13
- package/src/tools/dispatch-arguments.ts +27 -0
- package/src/tools/dispatch-plan.ts +46 -9
- package/src/tools/dispatch-runner.ts +48 -13
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/monitor.ts +13 -0
- package/src/tools/registry.ts +16 -0
- package/src/tools/worker-evidence.ts +19 -13
- package/src/worker/spec-contract.ts +2 -1
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/chunk-ZGH7FGS5.js +0 -1079
- package/dist/doctor-RN4YKO2X.js +0 -87
- package/dist/eval-RUBJVSNQ.js +0 -2557
|
@@ -4,38 +4,50 @@ import {
|
|
|
4
4
|
stripTokenizerSentinels
|
|
5
5
|
} from "./chunk-CFGTUFWB.js";
|
|
6
6
|
import {
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
registerClioOAuthProviders,
|
|
10
|
-
resolveAuthTarget,
|
|
11
|
-
resolveRuntimeAuthTarget,
|
|
12
|
-
targetRequiresAuth
|
|
13
|
-
} from "./chunk-JRIO5UD2.js";
|
|
7
|
+
credentialsPresent
|
|
8
|
+
} from "./chunk-LU7P4LHA.js";
|
|
14
9
|
import {
|
|
15
10
|
ceilChars,
|
|
16
11
|
estimateAgentMessageTokens,
|
|
17
12
|
toolSchemaChars
|
|
18
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-AEYBF3TB.js";
|
|
19
14
|
import {
|
|
20
15
|
readSettings
|
|
21
|
-
} from "./chunk-
|
|
22
|
-
import {
|
|
23
|
-
THINKING_LEVELS
|
|
24
|
-
} from "./chunk-4DGYLA73.js";
|
|
16
|
+
} from "./chunk-PNY46YEY.js";
|
|
25
17
|
import {
|
|
26
18
|
sleep
|
|
27
19
|
} from "./chunk-FQ4SKYE4.js";
|
|
20
|
+
import {
|
|
21
|
+
getSharedBus
|
|
22
|
+
} from "./chunk-ZGVHUX3M.js";
|
|
23
|
+
import {
|
|
24
|
+
BusChannels
|
|
25
|
+
} from "./chunk-RKRLDWD3.js";
|
|
26
|
+
import {
|
|
27
|
+
CLIO_CONTEXT_WINDOW_WARN_BELOW,
|
|
28
|
+
CLIO_MIN_CONTEXT_WINDOW,
|
|
29
|
+
CLIO_MIN_MAX_OUTPUT_TOKENS,
|
|
30
|
+
extractLocalModelQuirks,
|
|
31
|
+
listLmStudioModels,
|
|
32
|
+
lmStudioReasoningEffort,
|
|
33
|
+
lmStudioRootUrl,
|
|
34
|
+
registerBuiltinRuntimes,
|
|
35
|
+
requestLmStudioJson,
|
|
36
|
+
resolveLmStudioInstance
|
|
37
|
+
} from "./chunk-JEQQR47K.js";
|
|
38
|
+
import {
|
|
39
|
+
listKnownModelsForRuntime
|
|
40
|
+
} from "./chunk-774ILSRL.js";
|
|
41
|
+
import {
|
|
42
|
+
authNotRequiredStatus,
|
|
43
|
+
openAuthStorage,
|
|
44
|
+
registerClioOAuthProviders,
|
|
45
|
+
resolveAuthTarget,
|
|
46
|
+
targetRequiresAuth
|
|
47
|
+
} from "./chunk-EQ63NRB7.js";
|
|
28
48
|
import {
|
|
29
49
|
withStateFileLock
|
|
30
50
|
} from "./chunk-IKCO5N3L.js";
|
|
31
|
-
import {
|
|
32
|
-
activateExternalPluginApiBridge,
|
|
33
|
-
calculateEngineCost,
|
|
34
|
-
createEngineAi,
|
|
35
|
-
ensurePiAiRegistered,
|
|
36
|
-
getEngineSupportedThinkingLevels,
|
|
37
|
-
registerEngineApiProvider
|
|
38
|
-
} from "./chunk-FCSXB6T2.js";
|
|
39
51
|
import {
|
|
40
52
|
require_dist
|
|
41
53
|
} from "./chunk-APJ265NV.js";
|
|
@@ -45,329 +57,106 @@ import {
|
|
|
45
57
|
resolveClioDirs
|
|
46
58
|
} from "./chunk-BNAZZHFG.js";
|
|
47
59
|
import {
|
|
48
|
-
|
|
49
|
-
} from "./chunk-
|
|
60
|
+
getRuntimeRegistry
|
|
61
|
+
} from "./chunk-NUGM5KR6.js";
|
|
50
62
|
import {
|
|
51
|
-
|
|
52
|
-
|
|
63
|
+
capabilitiesFromCatalogModel,
|
|
64
|
+
catalogThinkingLevelsForRuntime,
|
|
65
|
+
getCatalogModelForRuntime,
|
|
66
|
+
mergeCapabilities,
|
|
67
|
+
resolveCostProvenance
|
|
68
|
+
} from "./chunk-VO2LKSTM.js";
|
|
69
|
+
import {
|
|
70
|
+
activateExternalPluginApiBridge,
|
|
71
|
+
calculateEngineCost,
|
|
72
|
+
ensurePiAiRegistered,
|
|
73
|
+
registerEngineApiProvider
|
|
74
|
+
} from "./chunk-UOSL25KY.js";
|
|
53
75
|
import {
|
|
54
76
|
__toESM,
|
|
55
77
|
init_esm_shims
|
|
56
78
|
} from "./chunk-3R73A4XB.js";
|
|
57
79
|
|
|
58
|
-
// src/domains/providers/
|
|
80
|
+
// src/domains/providers/endpoint-capacity.ts
|
|
59
81
|
init_esm_shims();
|
|
60
|
-
function mergeCapabilities(base, kb, probe, userOverride) {
|
|
61
|
-
const merged = { ...base };
|
|
62
|
-
applyLayer(merged, probe);
|
|
63
|
-
applyLayer(merged, kb);
|
|
64
|
-
applyDeploymentLimits(merged, probe);
|
|
65
|
-
applyLayer(merged, userOverride);
|
|
66
|
-
return merged;
|
|
67
|
-
}
|
|
68
|
-
function applyDeploymentLimits(target, probe) {
|
|
69
|
-
if (!probe) return;
|
|
70
|
-
if (probe.contextWindow !== void 0) target.contextWindow = probe.contextWindow;
|
|
71
|
-
if (probe.maxTokens !== void 0) target.maxTokens = probe.maxTokens;
|
|
72
|
-
}
|
|
73
|
-
function supportsAgentRoleTools(capabilities2) {
|
|
74
|
-
return capabilities2.tools === true;
|
|
75
|
-
}
|
|
76
|
-
var AGENT_ROLE_TOOLS_REQUIRED_REASON = "reports no tool support; Clio drives every agent role through typed tools";
|
|
77
|
-
function applyLayer(target, layer) {
|
|
78
|
-
if (!layer) return;
|
|
79
|
-
for (const key of Object.keys(layer)) {
|
|
80
|
-
const value = layer[key];
|
|
81
|
-
if (value !== void 0) target[key] = value;
|
|
82
|
-
}
|
|
83
|
-
}
|
|
84
82
|
|
|
85
|
-
// src/
|
|
83
|
+
// src/core/endpoint-key.ts
|
|
86
84
|
init_esm_shims();
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
["anthropic", "anthropic"],
|
|
90
|
-
["anthropic-max", "anthropic"],
|
|
91
|
-
["claude-code", "anthropic"],
|
|
92
|
-
["claude-sdk", "anthropic"],
|
|
93
|
-
["bedrock", "amazon-bedrock"],
|
|
94
|
-
["deepseek", "deepseek"],
|
|
95
|
-
["google", "google"],
|
|
96
|
-
["groq", "groq"],
|
|
97
|
-
["mistral", "mistral"],
|
|
98
|
-
["openai", "openai"],
|
|
99
|
-
["openai-codex", "openai-codex"],
|
|
100
|
-
["openrouter", "openrouter"]
|
|
101
|
-
]);
|
|
102
|
-
function catalogProviderForRuntime(runtimeId) {
|
|
103
|
-
return CATALOG_PROVIDER_BY_RUNTIME_ID.get(runtimeId);
|
|
104
|
-
}
|
|
105
|
-
function listCatalogModelsForRuntime(runtimeId) {
|
|
106
|
-
const provider = catalogProviderForRuntime(runtimeId);
|
|
107
|
-
if (!provider) return [];
|
|
108
|
-
try {
|
|
109
|
-
return engineAi.listModels(provider);
|
|
110
|
-
} catch {
|
|
111
|
-
return [];
|
|
112
|
-
}
|
|
113
|
-
}
|
|
114
|
-
function resolveEffectivePricing(target, runtimeId, wireModelId) {
|
|
115
|
-
if (target.pricing) {
|
|
116
|
-
const rates = {
|
|
117
|
-
input: target.pricing.input,
|
|
118
|
-
output: target.pricing.output,
|
|
119
|
-
cacheRead: target.pricing.cacheRead ?? 0,
|
|
120
|
-
cacheWrite: target.pricing.cacheWrite ?? 0
|
|
121
|
-
};
|
|
122
|
-
return {
|
|
123
|
-
rates,
|
|
124
|
-
provenance: Object.values(rates).every((rate) => rate === 0) ? "known_free" : "known"
|
|
125
|
-
};
|
|
126
|
-
}
|
|
127
|
-
const catalogModel = getCatalogModelForRuntime(runtimeId, wireModelId);
|
|
128
|
-
if (!catalogModel) return { rates: null, provenance: "unknown" };
|
|
129
|
-
return {
|
|
130
|
-
rates: {
|
|
131
|
-
input: catalogModel.cost.input,
|
|
132
|
-
output: catalogModel.cost.output,
|
|
133
|
-
cacheRead: catalogModel.cost.cacheRead,
|
|
134
|
-
cacheWrite: catalogModel.cost.cacheWrite
|
|
135
|
-
},
|
|
136
|
-
provenance: "estimated"
|
|
137
|
-
};
|
|
138
|
-
}
|
|
139
|
-
function resolveCostProvenance(target, runtimeId, wireModelId) {
|
|
140
|
-
return resolveEffectivePricing(target, runtimeId, wireModelId).provenance;
|
|
141
|
-
}
|
|
142
|
-
function getCatalogModelForRuntime(runtimeId, wireModelId) {
|
|
143
|
-
const provider = catalogProviderForRuntime(runtimeId);
|
|
144
|
-
if (!provider) return void 0;
|
|
85
|
+
function canonicalEndpointUrl(raw) {
|
|
86
|
+
if (!raw?.trim()) return null;
|
|
145
87
|
try {
|
|
146
|
-
|
|
88
|
+
const url = new URL(raw.trim().replace(/^ws:/u, "http:").replace(/^wss:/u, "https:"));
|
|
89
|
+
url.protocol = url.protocol.toLowerCase();
|
|
90
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
|
|
91
|
+
url.hostname = url.hostname.toLowerCase();
|
|
92
|
+
url.username = "";
|
|
93
|
+
url.password = "";
|
|
94
|
+
url.hash = "";
|
|
95
|
+
url.search = "";
|
|
96
|
+
let path = url.pathname.replace(/\/{2,}/gu, "/").replace(/\/$/u, "");
|
|
97
|
+
if (path.endsWith("/v1")) path = path.slice(0, -3);
|
|
98
|
+
url.pathname = path || "/";
|
|
99
|
+
return url.toString().replace(/\/$/u, "");
|
|
147
100
|
} catch {
|
|
148
|
-
return
|
|
149
|
-
}
|
|
150
|
-
}
|
|
151
|
-
function capabilitiesFromCatalogModel(defaultCapabilities25, model) {
|
|
152
|
-
if (!model) return defaultCapabilities25;
|
|
153
|
-
return {
|
|
154
|
-
...defaultCapabilities25,
|
|
155
|
-
reasoning: model.reasoning,
|
|
156
|
-
vision: model.input.includes("image"),
|
|
157
|
-
contextWindow: model.contextWindow,
|
|
158
|
-
maxTokens: model.maxTokens
|
|
159
|
-
};
|
|
160
|
-
}
|
|
161
|
-
function catalogThinkingLevelsForRuntime(runtimeId, wireModelId) {
|
|
162
|
-
const model = getCatalogModelForRuntime(runtimeId, wireModelId);
|
|
163
|
-
return model ? getEngineSupportedThinkingLevels(model) : void 0;
|
|
164
|
-
}
|
|
165
|
-
function synthesizeCatalogBackedModel(input) {
|
|
166
|
-
const builtin = getCatalogModelForRuntime(input.runtimeId, input.wireModelId);
|
|
167
|
-
const caps = mergeCapabilities(
|
|
168
|
-
capabilitiesFromCatalogModel(input.defaultCapabilities, builtin),
|
|
169
|
-
input.kb?.entry.capabilities ?? null,
|
|
170
|
-
null,
|
|
171
|
-
input.target.capabilities ?? null
|
|
172
|
-
);
|
|
173
|
-
const pricing = input.target.pricing;
|
|
174
|
-
const targetHeaders = input.target.auth?.headers;
|
|
175
|
-
const model = {
|
|
176
|
-
...builtin ?? {},
|
|
177
|
-
id: input.wireModelId,
|
|
178
|
-
name: `${input.wireModelId} (${input.target.id})`,
|
|
179
|
-
api: input.api,
|
|
180
|
-
provider: input.provider,
|
|
181
|
-
baseUrl: input.target.url ?? builtin?.baseUrl ?? input.defaultBaseUrl,
|
|
182
|
-
reasoning: caps.reasoning,
|
|
183
|
-
input: caps.vision ? builtin?.input.includes("image") ? builtin.input : ["text", "image"] : ["text"],
|
|
184
|
-
cost: {
|
|
185
|
-
input: pricing?.input ?? builtin?.cost.input ?? 0,
|
|
186
|
-
output: pricing?.output ?? builtin?.cost.output ?? 0,
|
|
187
|
-
cacheRead: pricing?.cacheRead ?? builtin?.cost.cacheRead ?? 0,
|
|
188
|
-
cacheWrite: pricing?.cacheWrite ?? builtin?.cost.cacheWrite ?? 0
|
|
189
|
-
},
|
|
190
|
-
contextWindow: caps.contextWindow,
|
|
191
|
-
maxTokens: caps.maxTokens
|
|
192
|
-
};
|
|
193
|
-
const headers = { ...input.defaultHeaders ?? {}, ...builtin?.headers ?? {}, ...targetHeaders ?? {} };
|
|
194
|
-
if (Object.keys(headers).length > 0) {
|
|
195
|
-
model.headers = headers;
|
|
101
|
+
return null;
|
|
196
102
|
}
|
|
197
|
-
return model;
|
|
198
103
|
}
|
|
199
104
|
|
|
200
|
-
// src/domains/providers/
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
import { pathToFileURL } from "node:url";
|
|
205
|
-
var RUNTIME_KINDS = ["http", "sdk", "subprocess"];
|
|
206
|
-
var RUNTIME_AUTHS = ["api-key", "oauth", "aws-sdk", "vertex-adc", "claude-cli", "none"];
|
|
207
|
-
function createRuntimeRegistry() {
|
|
208
|
-
const byId = /* @__PURE__ */ new Map();
|
|
209
|
-
const canonical = /* @__PURE__ */ new Set();
|
|
210
|
-
const register = (desc) => {
|
|
211
|
-
const ids = [desc.id, ...desc.aliases ?? []];
|
|
212
|
-
const conflict = ids.find((id) => byId.has(id));
|
|
213
|
-
if (conflict) {
|
|
214
|
-
throw new Error(`runtime id '${conflict}' already registered`);
|
|
215
|
-
}
|
|
216
|
-
for (const id of ids) byId.set(id, desc);
|
|
217
|
-
canonical.add(desc.id);
|
|
218
|
-
};
|
|
219
|
-
const get = (id) => byId.get(id) ?? null;
|
|
220
|
-
const list = () => Array.from(canonical, (id) => byId.get(id)).filter((entry) => entry !== void 0);
|
|
221
|
-
const clear = () => {
|
|
222
|
-
byId.clear();
|
|
223
|
-
canonical.clear();
|
|
224
|
-
};
|
|
225
|
-
const loadFromDir = async (dir, beforeImport) => {
|
|
226
|
-
let entries;
|
|
227
|
-
try {
|
|
228
|
-
const stat = statSync(dir);
|
|
229
|
-
if (!stat.isDirectory()) return [];
|
|
230
|
-
entries = readdirSync(dir);
|
|
231
|
-
} catch {
|
|
232
|
-
return [];
|
|
233
|
-
}
|
|
234
|
-
const loaded = [];
|
|
235
|
-
for (const name of entries) {
|
|
236
|
-
if (!name.endsWith(".js")) continue;
|
|
237
|
-
const full = join(dir, name);
|
|
238
|
-
const desc = await importDescriptor(full, pathToFileURL(full).href, beforeImport);
|
|
239
|
-
if (desc === null) continue;
|
|
240
|
-
try {
|
|
241
|
-
register(desc);
|
|
242
|
-
loaded.push(desc.id);
|
|
243
|
-
} catch (err) {
|
|
244
|
-
process.stderr.write(`[providers] runtime plugin ${full} rejected: ${describeError(err)}
|
|
245
|
-
`);
|
|
246
|
-
}
|
|
247
|
-
}
|
|
248
|
-
return loaded;
|
|
249
|
-
};
|
|
250
|
-
const loadFromPackage = async (packageName, beforeImport) => {
|
|
251
|
-
let mod;
|
|
252
|
-
try {
|
|
253
|
-
await beforeImport?.();
|
|
254
|
-
mod = await import(packageName);
|
|
255
|
-
} catch (err) {
|
|
256
|
-
process.stderr.write(`[providers] runtime package ${packageName} failed to import: ${describeError(err)}
|
|
257
|
-
`);
|
|
258
|
-
return [];
|
|
259
|
-
}
|
|
260
|
-
const exported = mod.clioRuntimes;
|
|
261
|
-
if (!Array.isArray(exported)) {
|
|
262
|
-
process.stderr.write(`[providers] runtime package ${packageName} has no 'clioRuntimes' array export
|
|
263
|
-
`);
|
|
264
|
-
return [];
|
|
265
|
-
}
|
|
266
|
-
const loaded = [];
|
|
267
|
-
for (const candidate of exported) {
|
|
268
|
-
const validation = validateRuntimeDescriptor(candidate);
|
|
269
|
-
if (!validation.ok) {
|
|
270
|
-
process.stderr.write(
|
|
271
|
-
`[providers] runtime package ${packageName} exported an invalid descriptor: ${validation.reason}
|
|
272
|
-
`
|
|
273
|
-
);
|
|
274
|
-
continue;
|
|
275
|
-
}
|
|
276
|
-
try {
|
|
277
|
-
register(validation.descriptor);
|
|
278
|
-
loaded.push(validation.descriptor.id);
|
|
279
|
-
} catch (err) {
|
|
280
|
-
process.stderr.write(`[providers] runtime package ${packageName} id conflict: ${describeError(err)}
|
|
281
|
-
`);
|
|
282
|
-
}
|
|
283
|
-
}
|
|
284
|
-
return loaded;
|
|
285
|
-
};
|
|
286
|
-
return { register, get, list, clear, loadFromDir, loadFromPackage };
|
|
105
|
+
// src/domains/providers/endpoint-capacity.ts
|
|
106
|
+
var UNBOUNDED_LOCAL_NATIVE_RUNTIMES = /* @__PURE__ */ new Set(["vllm", "sglang"]);
|
|
107
|
+
function canonicalEndpointKey(value) {
|
|
108
|
+
return canonicalEndpointUrl(typeof value === "string" ? value : value.url);
|
|
287
109
|
}
|
|
288
|
-
|
|
289
|
-
function getRuntimeRegistry() {
|
|
290
|
-
if (singleton === null) singleton = createRuntimeRegistry();
|
|
291
|
-
return singleton;
|
|
292
|
-
}
|
|
293
|
-
async function importDescriptor(file, href, beforeImport) {
|
|
294
|
-
let mod;
|
|
110
|
+
function endpointLabel(endpointKey) {
|
|
295
111
|
try {
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
} catch
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
}
|
|
312
|
-
return
|
|
313
|
-
}
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
return { ok: false, reason: "displayName must be a non-empty string" };
|
|
327
|
-
}
|
|
328
|
-
if (typeof v.kind !== "string" || !RUNTIME_KINDS.includes(v.kind)) {
|
|
329
|
-
return { ok: false, reason: `kind must be one of: ${RUNTIME_KINDS.join(", ")}` };
|
|
330
|
-
}
|
|
331
|
-
if (typeof v.apiFamily !== "string" || v.apiFamily.trim().length === 0) {
|
|
332
|
-
return { ok: false, reason: "apiFamily must be a non-empty string" };
|
|
333
|
-
}
|
|
334
|
-
if (typeof v.auth !== "string" || !RUNTIME_AUTHS.includes(v.auth)) {
|
|
335
|
-
return { ok: false, reason: `auth must be one of: ${RUNTIME_AUTHS.join(", ")}` };
|
|
336
|
-
}
|
|
337
|
-
if (typeof v.defaultCapabilities !== "object" || v.defaultCapabilities === null || Array.isArray(v.defaultCapabilities)) {
|
|
338
|
-
return { ok: false, reason: "defaultCapabilities must be an object" };
|
|
339
|
-
}
|
|
340
|
-
if (typeof v.synthesizeModel !== "function") {
|
|
341
|
-
return { ok: false, reason: "synthesizeModel must be a function" };
|
|
342
|
-
}
|
|
343
|
-
for (const field of ["probe", "probeModels", "complete", "infill", "embed", "rerank"]) {
|
|
344
|
-
if (v[field] !== void 0 && typeof v[field] !== "function") {
|
|
345
|
-
return { ok: false, reason: `${field} must be a function when present` };
|
|
112
|
+
const url = new URL(endpointKey);
|
|
113
|
+
return `${url.host}${url.pathname === "/" ? "" : url.pathname}`;
|
|
114
|
+
} catch {
|
|
115
|
+
return endpointKey;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
function positiveInteger(value) {
|
|
119
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : void 0;
|
|
120
|
+
}
|
|
121
|
+
function endpointCapacityForStatus(status) {
|
|
122
|
+
const key = canonicalEndpointKey(status.target);
|
|
123
|
+
if (key === null) return null;
|
|
124
|
+
const override = positiveInteger(status.target.maxConcurrentRequests);
|
|
125
|
+
if (override !== void 0) return { key, label: endpointLabel(key), limit: override, source: "override" };
|
|
126
|
+
const discovered = positiveInteger(status.probeCapabilities?.parallelSlots);
|
|
127
|
+
if (discovered !== void 0) return { key, label: endpointLabel(key), limit: discovered, source: "discovered" };
|
|
128
|
+
if (status.runtime?.tier !== "local-native" || UNBOUNDED_LOCAL_NATIVE_RUNTIMES.has(status.runtime.id)) return null;
|
|
129
|
+
return { key, label: endpointLabel(key), limit: 1, source: "local-native-default" };
|
|
130
|
+
}
|
|
131
|
+
function sourceRank(source) {
|
|
132
|
+
return source === "override" ? 2 : source === "discovered" ? 1 : 0;
|
|
133
|
+
}
|
|
134
|
+
function endpointCapacitiesForStatuses(statuses) {
|
|
135
|
+
const capacities = {};
|
|
136
|
+
for (const status of statuses) {
|
|
137
|
+
const candidate = endpointCapacityForStatus(status);
|
|
138
|
+
if (candidate === null) continue;
|
|
139
|
+
const current = capacities[candidate.key];
|
|
140
|
+
if (current === void 0 || sourceRank(candidate.source) > sourceRank(current.source) || sourceRank(candidate.source) === sourceRank(current.source) && candidate.limit < current.limit) {
|
|
141
|
+
capacities[candidate.key] = candidate;
|
|
346
142
|
}
|
|
347
143
|
}
|
|
348
|
-
return
|
|
349
|
-
}
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
144
|
+
return capacities;
|
|
145
|
+
}
|
|
146
|
+
var foregroundStreams = /* @__PURE__ */ new Map();
|
|
147
|
+
function registerForegroundStream(endpointKey) {
|
|
148
|
+
foregroundStreams.set(endpointKey, (foregroundStreams.get(endpointKey) ?? 0) + 1);
|
|
149
|
+
let held = true;
|
|
150
|
+
return () => {
|
|
151
|
+
if (!held) return;
|
|
152
|
+
held = false;
|
|
153
|
+
const next = (foregroundStreams.get(endpointKey) ?? 1) - 1;
|
|
154
|
+
if (next <= 0) foregroundStreams.delete(endpointKey);
|
|
155
|
+
else foregroundStreams.set(endpointKey, next);
|
|
156
|
+
};
|
|
353
157
|
}
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
init_esm_shims();
|
|
357
|
-
function credentialsPresent() {
|
|
358
|
-
const present = /* @__PURE__ */ new Set();
|
|
359
|
-
const registry = getRuntimeRegistry();
|
|
360
|
-
const auth = openAuthStorage();
|
|
361
|
-
for (const desc of registry.list()) {
|
|
362
|
-
const envVar = desc.credentialsEnvVar;
|
|
363
|
-
if (!envVar) continue;
|
|
364
|
-
const providerId = desc.id;
|
|
365
|
-
const status = auth.status(providerId, { explicitEnvVar: envVar, includeFallback: false });
|
|
366
|
-
if (status.available) {
|
|
367
|
-
present.add(envVar);
|
|
368
|
-
}
|
|
369
|
-
}
|
|
370
|
-
return present;
|
|
158
|
+
function foregroundStreamUsage() {
|
|
159
|
+
return Object.fromEntries(foregroundStreams);
|
|
371
160
|
}
|
|
372
161
|
|
|
373
162
|
// src/engine/apis/residency.ts
|
|
@@ -375,10 +164,11 @@ init_esm_shims();
|
|
|
375
164
|
|
|
376
165
|
// src/engine/apis/residency-lock.ts
|
|
377
166
|
init_esm_shims();
|
|
378
|
-
import { join
|
|
167
|
+
import { join } from "node:path";
|
|
379
168
|
var LOCK_WAIT_MS = 12e4;
|
|
380
169
|
function lockTargetFor(targetKey) {
|
|
381
|
-
|
|
170
|
+
const key = canonicalEndpointKey(targetKey) ?? targetKey;
|
|
171
|
+
return join(clioStatePath(), "residency-locks", key.replace(/[^a-zA-Z0-9._-]+/g, "_"));
|
|
382
172
|
}
|
|
383
173
|
async function withResidencyLock(targetKey, fn) {
|
|
384
174
|
return withStateFileLock(lockTargetFor(targetKey), fn, {
|
|
@@ -407,6 +197,9 @@ function emitResidencyNotice(notice) {
|
|
|
407
197
|
} catch {
|
|
408
198
|
}
|
|
409
199
|
}
|
|
200
|
+
function emitResidencyMutation(mutation) {
|
|
201
|
+
getSharedBus().emit(BusChannels.ResidencyMutation, { ...mutation, at: Date.now() });
|
|
202
|
+
}
|
|
410
203
|
var protectedModelsProvider = null;
|
|
411
204
|
function setProtectedModelsProvider(provider) {
|
|
412
205
|
protectedModelsProvider = provider;
|
|
@@ -637,6 +430,13 @@ async function restoreEvictedModels(adapter, evicted, loadError) {
|
|
|
637
430
|
try {
|
|
638
431
|
await adapter.reloadEvicted(entry.modelId);
|
|
639
432
|
restored.push(entry.modelId);
|
|
433
|
+
emitResidencyMutation({
|
|
434
|
+
targetKey: adapter.targetKey,
|
|
435
|
+
targetId: adapter.targetId,
|
|
436
|
+
runtimeId: adapter.runtimeId,
|
|
437
|
+
model: entry.modelId,
|
|
438
|
+
operation: "load"
|
|
439
|
+
});
|
|
640
440
|
} catch {
|
|
641
441
|
failed.push(entry.modelId);
|
|
642
442
|
}
|
|
@@ -753,12 +553,28 @@ async function reconcileResidency(adapter) {
|
|
|
753
553
|
await adapter.unload(entry.modelId);
|
|
754
554
|
forgetClioLoaded(adapter.targetKey, entry.modelId);
|
|
755
555
|
evicted.push(entry);
|
|
556
|
+
emitResidencyMutation({
|
|
557
|
+
targetKey: adapter.targetKey,
|
|
558
|
+
targetId: adapter.targetId,
|
|
559
|
+
runtimeId: adapter.runtimeId,
|
|
560
|
+
model: entry.modelId,
|
|
561
|
+
operation: "evict"
|
|
562
|
+
});
|
|
756
563
|
} catch {
|
|
757
564
|
}
|
|
758
565
|
}
|
|
759
566
|
if (!adapter.load) return;
|
|
760
567
|
try {
|
|
761
568
|
await adapter.load(adapter.keepModelId);
|
|
569
|
+
if (!plan.keepResident) {
|
|
570
|
+
emitResidencyMutation({
|
|
571
|
+
targetKey: adapter.targetKey,
|
|
572
|
+
targetId: adapter.targetId,
|
|
573
|
+
runtimeId: adapter.runtimeId,
|
|
574
|
+
model: adapter.keepModelId,
|
|
575
|
+
operation: "load"
|
|
576
|
+
});
|
|
577
|
+
}
|
|
762
578
|
} catch (error) {
|
|
763
579
|
await restoreEvictedModels(adapter, evicted, error);
|
|
764
580
|
throw error;
|
|
@@ -927,130 +743,6 @@ function availableThinkingLevels(caps, options) {
|
|
|
927
743
|
|
|
928
744
|
// src/domains/providers/model-capabilities.ts
|
|
929
745
|
init_esm_shims();
|
|
930
|
-
|
|
931
|
-
// src/domains/providers/types/local-model-quirks.ts
|
|
932
|
-
init_esm_shims();
|
|
933
|
-
var KV_CACHE_QUANTS = ["f32", "f16", "q8_0", "q4_0", "q4_1", "iq4_nl", "q5_0", "q5_1"];
|
|
934
|
-
function isRecord(value) {
|
|
935
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
936
|
-
}
|
|
937
|
-
function asKvCacheQuant(value) {
|
|
938
|
-
if (value === false) return false;
|
|
939
|
-
if (typeof value !== "string") return void 0;
|
|
940
|
-
return KV_CACHE_QUANTS.includes(value) ? value : void 0;
|
|
941
|
-
}
|
|
942
|
-
function asPositive(value) {
|
|
943
|
-
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : void 0;
|
|
944
|
-
}
|
|
945
|
-
function asInteger(value) {
|
|
946
|
-
const n = asPositive(value);
|
|
947
|
-
return n !== void 0 && Number.isInteger(n) ? n : void 0;
|
|
948
|
-
}
|
|
949
|
-
function asPenalty(value) {
|
|
950
|
-
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
951
|
-
}
|
|
952
|
-
function extractKvCache(raw) {
|
|
953
|
-
if (!isRecord(raw)) return void 0;
|
|
954
|
-
const out = {};
|
|
955
|
-
const k = asKvCacheQuant(raw.kQuant);
|
|
956
|
-
if (k !== void 0) out.kQuant = k;
|
|
957
|
-
const v = asKvCacheQuant(raw.vQuant);
|
|
958
|
-
if (v !== void 0) out.vQuant = v;
|
|
959
|
-
if (typeof raw.useFp16 === "boolean") out.useFp16 = raw.useFp16;
|
|
960
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
961
|
-
}
|
|
962
|
-
function extractSamplingProfile(raw) {
|
|
963
|
-
if (!isRecord(raw)) return void 0;
|
|
964
|
-
const out = {};
|
|
965
|
-
const temperature = asPositive(raw.temperature);
|
|
966
|
-
if (temperature !== void 0) out.temperature = temperature;
|
|
967
|
-
const topP = asPositive(raw.topP);
|
|
968
|
-
if (topP !== void 0) out.topP = topP;
|
|
969
|
-
const topK = asInteger(raw.topK);
|
|
970
|
-
if (topK !== void 0) out.topK = topK;
|
|
971
|
-
const minP = asPositive(raw.minP);
|
|
972
|
-
if (minP !== void 0) out.minP = minP;
|
|
973
|
-
const rp = asPenalty(raw.repeatPenalty) ?? asPenalty(raw.repetitionPenalty);
|
|
974
|
-
if (rp !== void 0) out.repeatPenalty = rp;
|
|
975
|
-
const pp = asPenalty(raw.presencePenalty);
|
|
976
|
-
if (pp !== void 0) out.presencePenalty = pp;
|
|
977
|
-
const fp = asPenalty(raw.frequencyPenalty);
|
|
978
|
-
if (fp !== void 0) out.frequencyPenalty = fp;
|
|
979
|
-
const maxTokens = asInteger(raw.maxTokens);
|
|
980
|
-
if (maxTokens !== void 0) out.maxTokens = maxTokens;
|
|
981
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
982
|
-
}
|
|
983
|
-
function extractSampling(raw) {
|
|
984
|
-
if (!isRecord(raw)) return void 0;
|
|
985
|
-
const out = {};
|
|
986
|
-
const thinking = extractSamplingProfile(raw.thinking);
|
|
987
|
-
if (thinking) out.thinking = thinking;
|
|
988
|
-
const instruct = extractSamplingProfile(raw.instruct);
|
|
989
|
-
if (instruct) out.instruct = instruct;
|
|
990
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
991
|
-
}
|
|
992
|
-
var THINKING_MECHANISMS = [
|
|
993
|
-
"effort-levels",
|
|
994
|
-
"budget-tokens",
|
|
995
|
-
"on-off",
|
|
996
|
-
"always-on",
|
|
997
|
-
"none"
|
|
998
|
-
];
|
|
999
|
-
function asThinkingMechanism(value) {
|
|
1000
|
-
if (typeof value !== "string") return void 0;
|
|
1001
|
-
return THINKING_MECHANISMS.includes(value) ? value : void 0;
|
|
1002
|
-
}
|
|
1003
|
-
function extractBudgetByLevel(raw) {
|
|
1004
|
-
if (!isRecord(raw)) return void 0;
|
|
1005
|
-
const out = {};
|
|
1006
|
-
const minimal = asInteger(raw.minimal);
|
|
1007
|
-
if (minimal !== void 0) out.minimal = minimal;
|
|
1008
|
-
const low = asInteger(raw.low);
|
|
1009
|
-
if (low !== void 0) out.low = low;
|
|
1010
|
-
const medium = asInteger(raw.medium);
|
|
1011
|
-
if (medium !== void 0) out.medium = medium;
|
|
1012
|
-
const high = asInteger(raw.high);
|
|
1013
|
-
if (high !== void 0) out.high = high;
|
|
1014
|
-
const xhigh = asInteger(raw.xhigh);
|
|
1015
|
-
if (xhigh !== void 0) out.xhigh = xhigh;
|
|
1016
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1017
|
-
}
|
|
1018
|
-
function extractEffortByLevel(raw) {
|
|
1019
|
-
if (!isRecord(raw)) return void 0;
|
|
1020
|
-
const out = {};
|
|
1021
|
-
if (typeof raw.off === "string" && raw.off.length > 0) out.off = raw.off;
|
|
1022
|
-
if (typeof raw.minimal === "string" && raw.minimal.length > 0) out.minimal = raw.minimal;
|
|
1023
|
-
if (typeof raw.low === "string" && raw.low.length > 0) out.low = raw.low;
|
|
1024
|
-
if (typeof raw.medium === "string" && raw.medium.length > 0) out.medium = raw.medium;
|
|
1025
|
-
if (typeof raw.high === "string" && raw.high.length > 0) out.high = raw.high;
|
|
1026
|
-
if (typeof raw.xhigh === "string" && raw.xhigh.length > 0) out.xhigh = raw.xhigh;
|
|
1027
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1028
|
-
}
|
|
1029
|
-
function extractThinkingQuirks(raw) {
|
|
1030
|
-
if (!isRecord(raw)) return void 0;
|
|
1031
|
-
const mechanism = asThinkingMechanism(raw.mechanism);
|
|
1032
|
-
if (!mechanism) return void 0;
|
|
1033
|
-
const out = { mechanism };
|
|
1034
|
-
const budgetByLevel = extractBudgetByLevel(raw.budgetByLevel);
|
|
1035
|
-
if (budgetByLevel) out.budgetByLevel = budgetByLevel;
|
|
1036
|
-
const effortByLevel = extractEffortByLevel(raw.effortByLevel);
|
|
1037
|
-
if (effortByLevel) out.effortByLevel = effortByLevel;
|
|
1038
|
-
if (typeof raw.guidance === "string" && raw.guidance.length > 0) out.guidance = raw.guidance;
|
|
1039
|
-
return out;
|
|
1040
|
-
}
|
|
1041
|
-
function extractLocalModelQuirks(raw) {
|
|
1042
|
-
if (!isRecord(raw)) return void 0;
|
|
1043
|
-
const out = {};
|
|
1044
|
-
const kvCache = extractKvCache(raw.kvCache);
|
|
1045
|
-
if (kvCache) out.kvCache = kvCache;
|
|
1046
|
-
const sampling = extractSampling(raw.sampling);
|
|
1047
|
-
if (sampling) out.sampling = sampling;
|
|
1048
|
-
const thinking = extractThinkingQuirks(raw.thinking);
|
|
1049
|
-
if (thinking) out.thinking = thinking;
|
|
1050
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
1051
|
-
}
|
|
1052
|
-
|
|
1053
|
-
// src/domains/providers/model-capabilities.ts
|
|
1054
746
|
function normalizedModelId(wireModelId) {
|
|
1055
747
|
const trimmed = wireModelId?.trim();
|
|
1056
748
|
return trimmed ? trimmed : null;
|
|
@@ -1478,2982 +1170,129 @@ function resolveModelRuntimeCapabilities(input) {
|
|
|
1478
1170
|
const thinking = resolveThinkingCapability(input, quirks, parser);
|
|
1479
1171
|
const result = {
|
|
1480
1172
|
targetId: input.targetId ?? null,
|
|
1481
|
-
runtimeId: input.runtimeId,
|
|
1482
|
-
apiFamily: input.apiFamily ?? null,
|
|
1483
|
-
modelId: input.modelId,
|
|
1484
|
-
family,
|
|
1485
|
-
capabilities: input.capabilities,
|
|
1486
|
-
thinking,
|
|
1487
|
-
request: resolveRequestCapability(thinking, parser, input.runtimeId),
|
|
1488
|
-
response: {
|
|
1489
|
-
parser,
|
|
1490
|
-
stripTokenizerSentinels: true
|
|
1491
|
-
}
|
|
1492
|
-
};
|
|
1493
|
-
if (quirks) result.quirks = quirks;
|
|
1494
|
-
return result;
|
|
1495
|
-
}
|
|
1496
|
-
function resolveModelRuntimeCapabilitiesForStatus(status, wireModelId, knowledgeBase, options) {
|
|
1497
|
-
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1498
|
-
const kbHit = modelId ? knowledgeBase?.lookup(modelId) ?? null : null;
|
|
1499
|
-
const capabilities2 = resolveModelCapabilities(status, modelId, knowledgeBase, {
|
|
1500
|
-
detectedReasoning: options?.detectedReasoning ?? null
|
|
1501
|
-
});
|
|
1502
|
-
const runtimeId = status.runtime?.id ?? status.target.runtime;
|
|
1503
|
-
return resolveModelRuntimeCapabilities({
|
|
1504
|
-
targetId: status.target.id,
|
|
1505
|
-
runtimeId,
|
|
1506
|
-
apiFamily: status.runtime?.apiFamily ?? null,
|
|
1507
|
-
modelId,
|
|
1508
|
-
capabilities: capabilities2,
|
|
1509
|
-
kbHit,
|
|
1510
|
-
...thinkingHintsForCatalogModel(runtimeId, modelId),
|
|
1511
|
-
...options?.configuredThinkingLevel ? { configuredThinkingLevel: options.configuredThinkingLevel } : {}
|
|
1512
|
-
});
|
|
1513
|
-
}
|
|
1514
|
-
function resolveModelRuntimeCapabilitiesForProviders(providers, targetId, wireModelId, configuredThinkingLevel) {
|
|
1515
|
-
const id = targetId?.trim();
|
|
1516
|
-
if (!id) return null;
|
|
1517
|
-
const status = providers.list().find((entry) => entry.target.id === id);
|
|
1518
|
-
if (!status) return null;
|
|
1519
|
-
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1520
|
-
const detectedReasoning = modelId && typeof providers.getDetectedReasoning === "function" ? providers.getDetectedReasoning(id, modelId) : null;
|
|
1521
|
-
return resolveModelRuntimeCapabilitiesForStatus(status, modelId, providers.knowledgeBase, {
|
|
1522
|
-
detectedReasoning,
|
|
1523
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1524
|
-
});
|
|
1525
|
-
}
|
|
1526
|
-
function thinkingHintsForModel(model) {
|
|
1527
|
-
const out = {};
|
|
1528
|
-
if (!model) return out;
|
|
1529
|
-
const compat = model.compat;
|
|
1530
|
-
if (compat?.forceAdaptiveThinking !== void 0) out.adaptiveThinking = compat.forceAdaptiveThinking;
|
|
1531
|
-
if (model.thinkingLevelMap) out.thinkingLevelMap = model.thinkingLevelMap;
|
|
1532
|
-
return out;
|
|
1533
|
-
}
|
|
1534
|
-
function thinkingHintsForCatalogModel(runtimeId, modelId) {
|
|
1535
|
-
if (!runtimeId || modelId.length === 0) return {};
|
|
1536
|
-
return thinkingHintsForModel(getCatalogModelForRuntime(runtimeId, modelId));
|
|
1537
|
-
}
|
|
1538
|
-
function thinkingFormatFromModelApi(api) {
|
|
1539
|
-
switch (api) {
|
|
1540
|
-
case "anthropic-messages":
|
|
1541
|
-
case "bedrock-converse-stream":
|
|
1542
|
-
case "claude-agent-sdk":
|
|
1543
|
-
case "claude-code-subprocess":
|
|
1544
|
-
return "anthropic-extended";
|
|
1545
|
-
case "openai-codex-responses":
|
|
1546
|
-
return "openai-codex";
|
|
1547
|
-
default:
|
|
1548
|
-
return void 0;
|
|
1549
|
-
}
|
|
1550
|
-
}
|
|
1551
|
-
function capabilitiesFromModel(model) {
|
|
1552
|
-
const format = model.compat?.thinkingFormat ?? thinkingFormatFromModelApi(model.api);
|
|
1553
|
-
const caps = {
|
|
1554
|
-
chat: true,
|
|
1555
|
-
tools: true,
|
|
1556
|
-
reasoning: model.reasoning === true,
|
|
1557
|
-
vision: Array.isArray(model.input) && model.input.includes("image"),
|
|
1558
|
-
audio: false,
|
|
1559
|
-
embeddings: false,
|
|
1560
|
-
rerank: false,
|
|
1561
|
-
fim: false,
|
|
1562
|
-
contextWindow: model.contextWindow,
|
|
1563
|
-
maxTokens: model.maxTokens
|
|
1564
|
-
};
|
|
1565
|
-
if (format === "qwen-chat-template" || format === "openrouter" || format === "zai" || format === "anthropic-extended" || format === "deepseek-r1" || format === "openai-codex" || format === "harmony") {
|
|
1566
|
-
caps.thinkingFormat = format;
|
|
1567
|
-
}
|
|
1568
|
-
return caps;
|
|
1569
|
-
}
|
|
1570
|
-
function resolveModelRuntimeCapabilitiesForModel(model, configuredThinkingLevel) {
|
|
1571
|
-
const metadata2 = model.clio;
|
|
1572
|
-
const caps = capabilitiesFromModel(model);
|
|
1573
|
-
return resolveModelRuntimeCapabilities({
|
|
1574
|
-
targetId: metadata2?.targetId ?? null,
|
|
1575
|
-
runtimeId: metadata2?.runtimeId ?? model.provider,
|
|
1576
|
-
apiFamily: model.api,
|
|
1577
|
-
modelId: model.id,
|
|
1578
|
-
capabilities: caps,
|
|
1579
|
-
...thinkingHintsForModel(model),
|
|
1580
|
-
...metadata2?.quirks ? { quirks: metadata2.quirks } : {},
|
|
1581
|
-
kbHit: metadata2?.family ? {
|
|
1582
|
-
matchKind: "family",
|
|
1583
|
-
entry: {
|
|
1584
|
-
family: metadata2.family,
|
|
1585
|
-
matchPatterns: [metadata2.family],
|
|
1586
|
-
capabilities: {}
|
|
1587
|
-
}
|
|
1588
|
-
} : null,
|
|
1589
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1590
|
-
});
|
|
1591
|
-
}
|
|
1592
|
-
function resolveTargetRuntimeCapabilities(target, runtime, wireModelId, capabilities2, knowledgeBase, configuredThinkingLevel) {
|
|
1593
|
-
const kbHit = knowledgeBase?.lookup(wireModelId) ?? null;
|
|
1594
|
-
return resolveModelRuntimeCapabilities({
|
|
1595
|
-
targetId: target.id,
|
|
1596
|
-
runtimeId: runtime.id,
|
|
1597
|
-
apiFamily: runtime.apiFamily,
|
|
1598
|
-
modelId: wireModelId,
|
|
1599
|
-
capabilities: capabilities2,
|
|
1600
|
-
kbHit,
|
|
1601
|
-
...thinkingHintsForCatalogModel(runtime.id, wireModelId),
|
|
1602
|
-
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1603
|
-
});
|
|
1604
|
-
}
|
|
1605
|
-
|
|
1606
|
-
// src/domains/providers/runtimes/common/lmstudio-http.ts
|
|
1607
|
-
init_esm_shims();
|
|
1608
|
-
import { performance } from "node:perf_hooks";
|
|
1609
|
-
|
|
1610
|
-
// src/domains/providers/runtimes/common/local-synth.ts
|
|
1611
|
-
init_esm_shims();
|
|
1612
|
-
function openAIThinkingFormat(caps) {
|
|
1613
|
-
switch (caps.thinkingFormat) {
|
|
1614
|
-
case "qwen-chat-template":
|
|
1615
|
-
case "openrouter":
|
|
1616
|
-
case "zai":
|
|
1617
|
-
return caps.thinkingFormat;
|
|
1618
|
-
case "deepseek-r1":
|
|
1619
|
-
return "deepseek";
|
|
1620
|
-
case "harmony":
|
|
1621
|
-
return "harmony";
|
|
1622
|
-
default:
|
|
1623
|
-
return void 0;
|
|
1624
|
-
}
|
|
1625
|
-
}
|
|
1626
|
-
function localOpenAICompat(caps, runtimeId) {
|
|
1627
|
-
const compat = {
|
|
1628
|
-
supportsStore: false,
|
|
1629
|
-
supportsDeveloperRole: false,
|
|
1630
|
-
supportsReasoningEffort: false,
|
|
1631
|
-
supportsUsageInStreaming: true,
|
|
1632
|
-
supportsFinishReason: false,
|
|
1633
|
-
maxTokensField: "max_tokens",
|
|
1634
|
-
supportsThinkingTokenBudget: runtimeId === "vllm",
|
|
1635
|
-
supportsStrictMode: false
|
|
1636
|
-
};
|
|
1637
|
-
const thinkingFormat = openAIThinkingFormat(caps);
|
|
1638
|
-
if (thinkingFormat) compat.thinkingFormat = thinkingFormat;
|
|
1639
|
-
return compat;
|
|
1640
|
-
}
|
|
1641
|
-
function localAnthropicCompat() {
|
|
1642
|
-
return {
|
|
1643
|
-
supportsEagerToolInputStreaming: false,
|
|
1644
|
-
supportsLongCacheRetention: false
|
|
1645
|
-
};
|
|
1646
|
-
}
|
|
1647
|
-
function synthLocalModel(input) {
|
|
1648
|
-
const { target, wireModelId, kb, defaultCapabilities: defaultCapabilities25, apiFamily, provider } = input;
|
|
1649
|
-
const caps = mergeCapabilities(defaultCapabilities25, kb?.entry.capabilities ?? null, null, target.capabilities ?? null);
|
|
1650
|
-
const rawUrl = target.url ?? "";
|
|
1651
|
-
const baseUrl = rawUrl.length > 0 ? input.baseUrlForTarget(rawUrl) : "";
|
|
1652
|
-
const pricing = target.pricing;
|
|
1653
|
-
const headers = target.auth?.headers;
|
|
1654
|
-
const quirks = extractLocalModelQuirks(kb?.entry.quirks);
|
|
1655
|
-
const model = {
|
|
1656
|
-
id: wireModelId,
|
|
1657
|
-
name: `${wireModelId} (${target.id})`,
|
|
1658
|
-
api: apiFamily,
|
|
1659
|
-
provider,
|
|
1660
|
-
baseUrl,
|
|
1661
|
-
reasoning: caps.reasoning,
|
|
1662
|
-
input: caps.vision ? ["text", "image"] : ["text"],
|
|
1663
|
-
cost: {
|
|
1664
|
-
input: pricing?.input ?? 0,
|
|
1665
|
-
output: pricing?.output ?? 0,
|
|
1666
|
-
cacheRead: pricing?.cacheRead ?? 0,
|
|
1667
|
-
cacheWrite: pricing?.cacheWrite ?? 0
|
|
1668
|
-
},
|
|
1669
|
-
contextWindow: caps.contextWindow,
|
|
1670
|
-
maxTokens: caps.maxTokens,
|
|
1671
|
-
clio: {
|
|
1672
|
-
targetId: target.id,
|
|
1673
|
-
runtimeId: target.runtime,
|
|
1674
|
-
...target.lifecycle ? { lifecycle: target.lifecycle } : {},
|
|
1675
|
-
...target.gateway === true ? { gateway: true } : {},
|
|
1676
|
-
...kb?.entry.family ? { family: kb.entry.family } : {},
|
|
1677
|
-
...quirks ? { quirks } : {},
|
|
1678
|
-
...target.lmstudio ? { lmstudio: target.lmstudio } : {}
|
|
1679
|
-
}
|
|
1680
|
-
};
|
|
1681
|
-
if (headers) model.headers = headers;
|
|
1682
|
-
if (apiFamily === "openai-completions") {
|
|
1683
|
-
model.compat = localOpenAICompat(caps, target.runtime);
|
|
1684
|
-
}
|
|
1685
|
-
if (apiFamily === "anthropic-messages") {
|
|
1686
|
-
model.compat = localAnthropicCompat();
|
|
1687
|
-
}
|
|
1688
|
-
return model;
|
|
1689
|
-
}
|
|
1690
|
-
function stripTrailingSlash(url2) {
|
|
1691
|
-
return url2.endsWith("/") ? url2.slice(0, -1) : url2;
|
|
1692
|
-
}
|
|
1693
|
-
function stripRedundantV1(url2) {
|
|
1694
|
-
return url2.endsWith("/v1") ? url2.slice(0, -"/v1".length) : url2;
|
|
1695
|
-
}
|
|
1696
|
-
var withV1 = (url2) => `${stripRedundantV1(stripTrailingSlash(url2))}/v1`;
|
|
1697
|
-
var withAsIs = (url2) => stripTrailingSlash(url2);
|
|
1698
|
-
function targetBaseUrl(target) {
|
|
1699
|
-
return target.url ? stripTrailingSlash(target.url) : null;
|
|
1700
|
-
}
|
|
1701
|
-
function targetRootUrl(target) {
|
|
1702
|
-
return target.url ? stripRedundantV1(stripTrailingSlash(target.url)) : null;
|
|
1703
|
-
}
|
|
1704
|
-
|
|
1705
|
-
// src/domains/providers/runtimes/common/lmstudio-http.ts
|
|
1706
|
-
var catalogsByHost = /* @__PURE__ */ new Map();
|
|
1707
|
-
function isRecord2(value) {
|
|
1708
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1709
|
-
}
|
|
1710
|
-
function positiveNumber(value) {
|
|
1711
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
1712
|
-
}
|
|
1713
|
-
function nonEmptyString(value) {
|
|
1714
|
-
return typeof value === "string" && value.trim().length > 0 ? value.trim() : void 0;
|
|
1715
|
-
}
|
|
1716
|
-
function lmStudioRootUrl(url2) {
|
|
1717
|
-
const normalized = targetRootUrl({ id: "lmstudio", runtime: "lmstudio", url: url2 });
|
|
1718
|
-
if (!normalized) return "";
|
|
1719
|
-
if (normalized.startsWith("ws://")) return `http://${normalized.slice("ws://".length)}`;
|
|
1720
|
-
if (normalized.startsWith("wss://")) return `https://${normalized.slice("wss://".length)}`;
|
|
1721
|
-
return normalized;
|
|
1722
|
-
}
|
|
1723
|
-
function lmStudioRequestHeaders(base, apiKey) {
|
|
1724
|
-
const headers = { ...base ?? {} };
|
|
1725
|
-
const token = apiKey?.trim();
|
|
1726
|
-
if (token) headers.authorization = `Bearer ${token}`;
|
|
1727
|
-
return headers;
|
|
1728
|
-
}
|
|
1729
|
-
function lmStudioProbeHeaders(target, ctx) {
|
|
1730
|
-
let token = ctx.authToken?.trim();
|
|
1731
|
-
const envName = target.auth?.apiKeyEnvVar;
|
|
1732
|
-
if (!token && envName && ctx.credentialsPresent.has(envName)) token = process.env[envName]?.trim();
|
|
1733
|
-
return lmStudioRequestHeaders(target.auth?.headers, token);
|
|
1734
|
-
}
|
|
1735
|
-
function combinedSignal(timeoutMs, signal) {
|
|
1736
|
-
const controller = new AbortController();
|
|
1737
|
-
const timer = setTimeout(() => controller.abort(new Error(`timeout after ${timeoutMs}ms`)), timeoutMs);
|
|
1738
|
-
const onAbort = () => controller.abort(signal?.reason);
|
|
1739
|
-
if (signal?.aborted) controller.abort(signal.reason);
|
|
1740
|
-
else signal?.addEventListener("abort", onAbort, { once: true });
|
|
1741
|
-
return {
|
|
1742
|
-
signal: controller.signal,
|
|
1743
|
-
close() {
|
|
1744
|
-
clearTimeout(timer);
|
|
1745
|
-
signal?.removeEventListener("abort", onAbort);
|
|
1746
|
-
}
|
|
1747
|
-
};
|
|
1748
|
-
}
|
|
1749
|
-
async function requestLmStudioJson(url2, init, timeoutMs, signal) {
|
|
1750
|
-
const started = performance.now();
|
|
1751
|
-
const bounded = combinedSignal(timeoutMs, signal);
|
|
1752
|
-
try {
|
|
1753
|
-
const response = await fetch(url2, { ...init, signal: bounded.signal });
|
|
1754
|
-
let data;
|
|
1755
|
-
try {
|
|
1756
|
-
data = await response.json();
|
|
1757
|
-
} catch {
|
|
1758
|
-
data = void 0;
|
|
1759
|
-
}
|
|
1760
|
-
const result = {
|
|
1761
|
-
ok: response.ok,
|
|
1762
|
-
status: response.status,
|
|
1763
|
-
latencyMs: Math.round(performance.now() - started)
|
|
1764
|
-
};
|
|
1765
|
-
if (data !== void 0) result.data = data;
|
|
1766
|
-
if (!response.ok) {
|
|
1767
|
-
const message = isRecord2(data) ? nonEmptyString(data.error) : void 0;
|
|
1768
|
-
result.error = message ?? `HTTP ${response.status}`;
|
|
1769
|
-
}
|
|
1770
|
-
return result;
|
|
1771
|
-
} catch (error) {
|
|
1772
|
-
return {
|
|
1773
|
-
ok: false,
|
|
1774
|
-
status: 0,
|
|
1775
|
-
latencyMs: Math.round(performance.now() - started),
|
|
1776
|
-
error: error instanceof Error ? error.message : String(error)
|
|
1777
|
-
};
|
|
1778
|
-
} finally {
|
|
1779
|
-
bounded.close();
|
|
1780
|
-
}
|
|
1781
|
-
}
|
|
1782
|
-
function loadedInstances(value) {
|
|
1783
|
-
if (!Array.isArray(value)) return [];
|
|
1784
|
-
const instances = [];
|
|
1785
|
-
for (const raw of value) {
|
|
1786
|
-
if (!isRecord2(raw)) continue;
|
|
1787
|
-
const id = nonEmptyString(raw.id);
|
|
1788
|
-
if (!id) continue;
|
|
1789
|
-
instances.push({ id, config: isRecord2(raw.config) ? { ...raw.config } : {} });
|
|
1790
|
-
}
|
|
1791
|
-
return instances;
|
|
1792
|
-
}
|
|
1793
|
-
function reasoningOptions(capabilities2) {
|
|
1794
|
-
if (!isRecord2(capabilities2) || !isRecord2(capabilities2.reasoning)) return void 0;
|
|
1795
|
-
const allowed = capabilities2.reasoning.allowed_options;
|
|
1796
|
-
if (!Array.isArray(allowed)) return void 0;
|
|
1797
|
-
const options = allowed.filter((entry) => typeof entry === "string");
|
|
1798
|
-
return options.length > 0 ? options : void 0;
|
|
1799
|
-
}
|
|
1800
|
-
function parseLmStudioV1Models(data) {
|
|
1801
|
-
if (!isRecord2(data) || !Array.isArray(data.models)) return null;
|
|
1802
|
-
const models = [];
|
|
1803
|
-
for (const raw of data.models) {
|
|
1804
|
-
if (!isRecord2(raw)) continue;
|
|
1805
|
-
const key = nonEmptyString(raw.key);
|
|
1806
|
-
if (!key) continue;
|
|
1807
|
-
const capabilities2 = isRecord2(raw.capabilities) ? raw.capabilities : void 0;
|
|
1808
|
-
const info = {
|
|
1809
|
-
key,
|
|
1810
|
-
loadedInstances: loadedInstances(raw.loaded_instances),
|
|
1811
|
-
metadata: { ...raw }
|
|
1812
|
-
};
|
|
1813
|
-
const type = nonEmptyString(raw.type);
|
|
1814
|
-
if (type) info.type = type;
|
|
1815
|
-
const maxContextLength = positiveNumber(raw.max_context_length);
|
|
1816
|
-
if (maxContextLength !== void 0) info.maxContextLength = maxContextLength;
|
|
1817
|
-
if (capabilities2 && typeof capabilities2.vision === "boolean") info.vision = capabilities2.vision;
|
|
1818
|
-
else if (type) info.vision = type === "vlm";
|
|
1819
|
-
if (capabilities2 && typeof capabilities2.trained_for_tool_use === "boolean") {
|
|
1820
|
-
info.tools = capabilities2.trained_for_tool_use;
|
|
1821
|
-
}
|
|
1822
|
-
const options = reasoningOptions(capabilities2);
|
|
1823
|
-
if (options) info.reasoningOptions = options;
|
|
1824
|
-
if (capabilities2 && "reasoning" in capabilities2) {
|
|
1825
|
-
info.reasoning = capabilities2.reasoning === true || isRecord2(capabilities2.reasoning);
|
|
1826
|
-
}
|
|
1827
|
-
models.push(info);
|
|
1828
|
-
}
|
|
1829
|
-
return models;
|
|
1830
|
-
}
|
|
1831
|
-
function foldLmStudioModels(models) {
|
|
1832
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
1833
|
-
for (const model of models) {
|
|
1834
|
-
const previous = byKey.get(model.key);
|
|
1835
|
-
if (!previous) {
|
|
1836
|
-
byKey.set(model.key, { ...model, loadedInstances: [...model.loadedInstances] });
|
|
1837
|
-
continue;
|
|
1838
|
-
}
|
|
1839
|
-
const knownIds = new Set(previous.loadedInstances.map((instance) => instance.id));
|
|
1840
|
-
for (const instance of model.loadedInstances) {
|
|
1841
|
-
if (!knownIds.has(instance.id)) previous.loadedInstances.push(instance);
|
|
1842
|
-
}
|
|
1843
|
-
if (previous.maxContextLength === void 0 && model.maxContextLength !== void 0) {
|
|
1844
|
-
previous.maxContextLength = model.maxContextLength;
|
|
1845
|
-
}
|
|
1846
|
-
if (previous.vision === void 0 && model.vision !== void 0) previous.vision = model.vision;
|
|
1847
|
-
if (previous.tools === void 0 && model.tools !== void 0) previous.tools = model.tools;
|
|
1848
|
-
if (previous.reasoningOptions === void 0 && model.reasoningOptions !== void 0) {
|
|
1849
|
-
previous.reasoningOptions = model.reasoningOptions;
|
|
1850
|
-
}
|
|
1851
|
-
if (previous.reasoning === void 0 && model.reasoning !== void 0) previous.reasoning = model.reasoning;
|
|
1852
|
-
}
|
|
1853
|
-
return [...byKey.values()];
|
|
1854
|
-
}
|
|
1855
|
-
function rememberHostCatalog(target, models) {
|
|
1856
|
-
const rootUrl2 = lmStudioRootUrl(target.url ?? "");
|
|
1857
|
-
if (!rootUrl2) return;
|
|
1858
|
-
catalogsByHost.set(rootUrl2, { targetId: target.id, rootUrl: rootUrl2, models: foldLmStudioModels(models) });
|
|
1859
|
-
}
|
|
1860
|
-
function peerTargetsFor(target, instanceId) {
|
|
1861
|
-
const rootUrl2 = lmStudioRootUrl(target.url ?? "");
|
|
1862
|
-
const peers = [];
|
|
1863
|
-
for (const catalog of catalogsByHost.values()) {
|
|
1864
|
-
if (catalog.rootUrl === rootUrl2) continue;
|
|
1865
|
-
if (catalog.models.some((model) => model.loadedInstances.some((instance) => instance.id === instanceId))) {
|
|
1866
|
-
peers.push(catalog.targetId);
|
|
1867
|
-
}
|
|
1868
|
-
}
|
|
1869
|
-
return [...new Set(peers)];
|
|
1870
|
-
}
|
|
1871
|
-
function resolveLmStudioInstance(target, models, requestedId, configuredDefault) {
|
|
1872
|
-
const folded = foldLmStudioModels(models);
|
|
1873
|
-
for (const model of folded) {
|
|
1874
|
-
const explicit = model.loadedInstances.find((instance2) => instance2.id === requestedId);
|
|
1875
|
-
if (explicit) {
|
|
1876
|
-
return {
|
|
1877
|
-
requestedId,
|
|
1878
|
-
wireModelId: explicit.id,
|
|
1879
|
-
model,
|
|
1880
|
-
instance: explicit,
|
|
1881
|
-
peerTargets: peerTargetsFor(target, explicit.id),
|
|
1882
|
-
state: "instance"
|
|
1883
|
-
};
|
|
1884
|
-
}
|
|
1885
|
-
if (model.key !== requestedId) continue;
|
|
1886
|
-
if (model.loadedInstances.length === 0) {
|
|
1887
|
-
return { requestedId, wireModelId: requestedId, model, peerTargets: [], state: "jit" };
|
|
1888
|
-
}
|
|
1889
|
-
const preferred = configuredDefault ? model.loadedInstances.find((instance2) => instance2.id === configuredDefault) : void 0;
|
|
1890
|
-
const local = model.loadedInstances.find((instance2) => peerTargetsFor(target, instance2.id).length === 0);
|
|
1891
|
-
const instance = preferred ?? local ?? model.loadedInstances[0];
|
|
1892
|
-
if (!instance) return { requestedId, wireModelId: requestedId, model, peerTargets: [], state: "jit" };
|
|
1893
|
-
return {
|
|
1894
|
-
requestedId,
|
|
1895
|
-
wireModelId: instance.id,
|
|
1896
|
-
model,
|
|
1897
|
-
instance,
|
|
1898
|
-
peerTargets: peerTargetsFor(target, instance.id),
|
|
1899
|
-
state: "instance"
|
|
1900
|
-
};
|
|
1901
|
-
}
|
|
1902
|
-
return { requestedId, wireModelId: requestedId, peerTargets: [], state: "unknown" };
|
|
1903
|
-
}
|
|
1904
|
-
function lmStudioReasoningLevels(options) {
|
|
1905
|
-
if (!options || options.length === 0) return [...THINKING_LEVELS];
|
|
1906
|
-
if (options.includes("on") && !options.includes("low") && !options.includes("medium") && !options.includes("high")) {
|
|
1907
|
-
return ["off", "low"];
|
|
1908
|
-
}
|
|
1909
|
-
const levels = ["off"];
|
|
1910
|
-
if (options.includes("low")) levels.push("minimal", "low");
|
|
1911
|
-
if (options.includes("medium")) levels.push("medium");
|
|
1912
|
-
if (options.includes("high")) levels.push("high", "xhigh", "max");
|
|
1913
|
-
return [...new Set(levels)];
|
|
1914
|
-
}
|
|
1915
|
-
function lmStudioReasoningEffort(level, options) {
|
|
1916
|
-
if (level === "off") return "none";
|
|
1917
|
-
if (options?.includes("on") && !options.includes("medium") && !options.includes("high")) return "low";
|
|
1918
|
-
if (level === "medium" && options?.includes("medium") !== false) return "medium";
|
|
1919
|
-
if ((level === "high" || level === "xhigh" || level === "max") && options?.includes("high") !== false) return "high";
|
|
1920
|
-
return "low";
|
|
1921
|
-
}
|
|
1922
|
-
function v0Models(data) {
|
|
1923
|
-
if (!isRecord2(data) || !Array.isArray(data.data)) return null;
|
|
1924
|
-
const models = [];
|
|
1925
|
-
for (const raw of data.data) {
|
|
1926
|
-
if (!isRecord2(raw)) continue;
|
|
1927
|
-
const key = nonEmptyString(raw.id);
|
|
1928
|
-
if (!key) continue;
|
|
1929
|
-
const state = nonEmptyString(raw.state);
|
|
1930
|
-
const loadedContext = positiveNumber(raw.loaded_context_length);
|
|
1931
|
-
const config = loadedContext === void 0 ? {} : { context_length: loadedContext };
|
|
1932
|
-
const info = {
|
|
1933
|
-
key,
|
|
1934
|
-
loadedInstances: state === "loaded" || loadedContext !== void 0 ? [{ id: key, config }] : [],
|
|
1935
|
-
metadata: { ...raw }
|
|
1936
|
-
};
|
|
1937
|
-
const type = nonEmptyString(raw.type);
|
|
1938
|
-
if (type) {
|
|
1939
|
-
info.type = type;
|
|
1940
|
-
info.vision = type === "vlm";
|
|
1941
|
-
}
|
|
1942
|
-
const maxContextLength = positiveNumber(raw.max_context_length);
|
|
1943
|
-
if (maxContextLength !== void 0) info.maxContextLength = maxContextLength;
|
|
1944
|
-
if (Array.isArray(raw.capabilities)) info.tools = raw.capabilities.includes("tool_use");
|
|
1945
|
-
models.push(info);
|
|
1946
|
-
}
|
|
1947
|
-
return models;
|
|
1948
|
-
}
|
|
1949
|
-
function openAIModels(data) {
|
|
1950
|
-
if (!isRecord2(data) || !Array.isArray(data.data)) return null;
|
|
1951
|
-
const models = [];
|
|
1952
|
-
for (const raw of data.data) {
|
|
1953
|
-
if (!isRecord2(raw)) continue;
|
|
1954
|
-
const key = nonEmptyString(raw.id);
|
|
1955
|
-
if (key) models.push({ key, loadedInstances: [], metadata: { ...raw } });
|
|
1956
|
-
}
|
|
1957
|
-
return models;
|
|
1958
|
-
}
|
|
1959
|
-
function authFailure(result) {
|
|
1960
|
-
if (result.status !== 401 && result.status !== 403) return null;
|
|
1961
|
-
return {
|
|
1962
|
-
ok: false,
|
|
1963
|
-
models: [],
|
|
1964
|
-
latencyMs: result.latencyMs,
|
|
1965
|
-
error: "LM Studio authentication required",
|
|
1966
|
-
authRequired: true
|
|
1967
|
-
};
|
|
1968
|
-
}
|
|
1969
|
-
async function listLmStudioModels(target, ctx) {
|
|
1970
|
-
if (!target.url) return { ok: false, models: [], error: "target has no url" };
|
|
1971
|
-
const root = lmStudioRootUrl(target.url);
|
|
1972
|
-
const headers = lmStudioProbeHeaders(target, ctx);
|
|
1973
|
-
const init = { headers };
|
|
1974
|
-
const v1 = await requestLmStudioJson(`${root}/api/v1/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1975
|
-
const v1Auth = authFailure(v1);
|
|
1976
|
-
if (v1Auth) return v1Auth;
|
|
1977
|
-
const parsedV1 = parseLmStudioV1Models(v1.data);
|
|
1978
|
-
if (v1.ok && parsedV1) {
|
|
1979
|
-
rememberHostCatalog(target, parsedV1);
|
|
1980
|
-
return { ok: true, models: parsedV1, tier: "0.4+", latencyMs: v1.latencyMs };
|
|
1981
|
-
}
|
|
1982
|
-
const v0 = await requestLmStudioJson(`${root}/api/v0/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1983
|
-
const v0Auth = authFailure(v0);
|
|
1984
|
-
if (v0Auth) return v0Auth;
|
|
1985
|
-
const parsedV0 = v0Models(v0.data);
|
|
1986
|
-
if (v0.ok && parsedV0) {
|
|
1987
|
-
rememberHostCatalog(target, parsedV0);
|
|
1988
|
-
return { ok: true, models: parsedV0, tier: "0.3.x", latencyMs: v0.latencyMs };
|
|
1989
|
-
}
|
|
1990
|
-
const openAI = await requestLmStudioJson(`${root}/v1/models`, init, ctx.httpTimeoutMs, ctx.signal);
|
|
1991
|
-
const openAIAuth = authFailure(openAI);
|
|
1992
|
-
if (openAIAuth) return openAIAuth;
|
|
1993
|
-
const parsedOpenAI = openAIModels(openAI.data);
|
|
1994
|
-
if (openAI.ok && parsedOpenAI) {
|
|
1995
|
-
rememberHostCatalog(target, parsedOpenAI);
|
|
1996
|
-
return { ok: true, models: parsedOpenAI, tier: "openai-compat", latencyMs: openAI.latencyMs };
|
|
1997
|
-
}
|
|
1998
|
-
return {
|
|
1999
|
-
ok: false,
|
|
2000
|
-
models: [],
|
|
2001
|
-
latencyMs: v1.latencyMs + v0.latencyMs + openAI.latencyMs,
|
|
2002
|
-
error: v1.error ?? v0.error ?? openAI.error ?? "LM Studio returned no recognized model catalog"
|
|
2003
|
-
};
|
|
2004
|
-
}
|
|
2005
|
-
async function greetLmStudio(target, ctx) {
|
|
2006
|
-
if (!target.url) return { ok: false, error: "target has no url" };
|
|
2007
|
-
const result = await requestLmStudioJson(
|
|
2008
|
-
`${lmStudioRootUrl(target.url)}/lmstudio-greeting`,
|
|
2009
|
-
{ headers: lmStudioProbeHeaders(target, ctx) },
|
|
2010
|
-
ctx.httpTimeoutMs,
|
|
2011
|
-
ctx.signal
|
|
2012
|
-
);
|
|
2013
|
-
if (result.ok && isRecord2(result.data) && result.data.lmstudio === true) {
|
|
2014
|
-
return { ok: true, latencyMs: result.latencyMs };
|
|
2015
|
-
}
|
|
2016
|
-
return { ok: false, latencyMs: result.latencyMs, error: result.error ?? "endpoint did not return {lmstudio:true}" };
|
|
2017
|
-
}
|
|
2018
|
-
function loadedContextLength(instance) {
|
|
2019
|
-
return positiveNumber(instance?.config.context_length);
|
|
2020
|
-
}
|
|
2021
|
-
|
|
2022
|
-
// src/domains/providers/runtimes/builtins.ts
|
|
2023
|
-
init_esm_shims();
|
|
2024
|
-
|
|
2025
|
-
// src/domains/providers/runtimes/antigravity/antigravity-code.ts
|
|
2026
|
-
init_esm_shims();
|
|
2027
|
-
var ANTIGRAVITY_AUTH_NOTICE = "Uses your existing Antigravity (`agy`) login. Clio stores no Antigravity credentials. agy runs its own agent harness as a subprocess; Clio maps autonomy levels onto its CLI flags but cannot mediate individual agy tool calls.";
|
|
2028
|
-
var ANTIGRAVITY_MODELS = [
|
|
2029
|
-
"Gemini 3.5 Flash (High)",
|
|
2030
|
-
"Gemini 3.5 Flash (Medium)",
|
|
2031
|
-
"Gemini 3.5 Flash (Low)",
|
|
2032
|
-
"Gemini 3.1 Pro (High)",
|
|
2033
|
-
"Gemini 3.1 Pro (Low)",
|
|
2034
|
-
"Claude Sonnet 4.6 (Thinking)",
|
|
2035
|
-
"Claude Opus 4.6 (Thinking)",
|
|
2036
|
-
"GPT-OSS 120B (Medium)"
|
|
2037
|
-
];
|
|
2038
|
-
var antigravityCapabilities = {
|
|
2039
|
-
chat: true,
|
|
2040
|
-
tools: true,
|
|
2041
|
-
toolCallFormat: "openai",
|
|
2042
|
-
reasoning: true,
|
|
2043
|
-
vision: true,
|
|
2044
|
-
audio: false,
|
|
2045
|
-
embeddings: false,
|
|
2046
|
-
rerank: false,
|
|
2047
|
-
fim: false,
|
|
2048
|
-
contextWindow: 1e6,
|
|
2049
|
-
maxTokens: 8192
|
|
2050
|
-
};
|
|
2051
|
-
var antigravityCodeRuntime = {
|
|
2052
|
-
id: "antigravity-code",
|
|
2053
|
-
displayName: "Antigravity CLI",
|
|
2054
|
-
kind: "subprocess",
|
|
2055
|
-
tier: "subscription",
|
|
2056
|
-
// Reuses the existing `google-generative-ai` api family: the worker branches
|
|
2057
|
-
// to its own runner before any pi-ai inference, so this only needs to be a
|
|
2058
|
-
// valid, google-backed family. The runner is selected by runtime id, and the
|
|
2059
|
-
// plain-text parser is named by `outputParser`.
|
|
2060
|
-
apiFamily: "google-generative-ai",
|
|
2061
|
-
auth: "none",
|
|
2062
|
-
authNotice: ANTIGRAVITY_AUTH_NOTICE,
|
|
2063
|
-
knownModels: [...ANTIGRAVITY_MODELS],
|
|
2064
|
-
binaryName: "agy",
|
|
2065
|
-
headlessCommand: "agy --print",
|
|
2066
|
-
outputParser: "antigravity-print-text",
|
|
2067
|
-
defaultCapabilities: antigravityCapabilities,
|
|
2068
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2069
|
-
return synthesizeCatalogBackedModel({
|
|
2070
|
-
target,
|
|
2071
|
-
wireModelId,
|
|
2072
|
-
kb,
|
|
2073
|
-
defaultCapabilities: antigravityCapabilities,
|
|
2074
|
-
runtimeId: "antigravity-code",
|
|
2075
|
-
api: "google-generative-ai",
|
|
2076
|
-
provider: "google",
|
|
2077
|
-
defaultBaseUrl: "antigravity://local"
|
|
2078
|
-
});
|
|
2079
|
-
}
|
|
2080
|
-
};
|
|
2081
|
-
var antigravity_code_default = antigravityCodeRuntime;
|
|
2082
|
-
|
|
2083
|
-
// src/domains/providers/runtimes/claude/claude-code.ts
|
|
2084
|
-
init_esm_shims();
|
|
2085
|
-
|
|
2086
|
-
// src/domains/providers/runtimes/claude/common.ts
|
|
2087
|
-
init_esm_shims();
|
|
2088
|
-
var CLAUDE_CODE_AUTH_NOTICE = "Uses your existing Claude Code login from the installed `claude` command. Clio stores no Claude Code credentials.";
|
|
2089
|
-
var CLAUDE_CODE_MODELS = [
|
|
2090
|
-
"sonnet",
|
|
2091
|
-
"opus",
|
|
2092
|
-
"haiku",
|
|
2093
|
-
"claude-sonnet-4-5",
|
|
2094
|
-
"claude-opus-4-5",
|
|
2095
|
-
"claude-haiku-4-5",
|
|
2096
|
-
"claude-3-7-sonnet-latest"
|
|
2097
|
-
];
|
|
2098
|
-
var claudeCodeCapabilities = {
|
|
2099
|
-
chat: true,
|
|
2100
|
-
tools: true,
|
|
2101
|
-
toolCallFormat: "anthropic",
|
|
2102
|
-
reasoning: true,
|
|
2103
|
-
thinkingFormat: "anthropic-extended",
|
|
2104
|
-
vision: true,
|
|
2105
|
-
audio: false,
|
|
2106
|
-
embeddings: false,
|
|
2107
|
-
rerank: false,
|
|
2108
|
-
fim: false,
|
|
2109
|
-
contextWindow: 2e5,
|
|
2110
|
-
maxTokens: 8192
|
|
2111
|
-
};
|
|
2112
|
-
function synthesizeClaudeDelegatedModel(input) {
|
|
2113
|
-
return synthesizeCatalogBackedModel({
|
|
2114
|
-
target: input.target,
|
|
2115
|
-
wireModelId: input.wireModelId,
|
|
2116
|
-
kb: input.kb,
|
|
2117
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2118
|
-
runtimeId: input.runtimeId,
|
|
2119
|
-
api: input.apiFamily,
|
|
2120
|
-
provider: "anthropic",
|
|
2121
|
-
defaultBaseUrl: "claude-code://local"
|
|
2122
|
-
});
|
|
2123
|
-
}
|
|
2124
|
-
|
|
2125
|
-
// src/domains/providers/runtimes/claude/claude-code.ts
|
|
2126
|
-
var claudeCodeRuntime = {
|
|
2127
|
-
id: "claude-code",
|
|
2128
|
-
displayName: "Claude Code CLI",
|
|
2129
|
-
kind: "subprocess",
|
|
2130
|
-
tier: "subscription",
|
|
2131
|
-
apiFamily: "claude-code-subprocess",
|
|
2132
|
-
auth: "claude-cli",
|
|
2133
|
-
authNotice: CLAUDE_CODE_AUTH_NOTICE,
|
|
2134
|
-
knownModels: [...CLAUDE_CODE_MODELS],
|
|
2135
|
-
binaryName: "claude",
|
|
2136
|
-
headlessCommand: "claude -p --output-format stream-json",
|
|
2137
|
-
outputParser: "claude-code-stream-json",
|
|
2138
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2139
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2140
|
-
return synthesizeClaudeDelegatedModel({
|
|
2141
|
-
target,
|
|
2142
|
-
wireModelId,
|
|
2143
|
-
kb,
|
|
2144
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2145
|
-
runtimeId: "claude-code",
|
|
2146
|
-
apiFamily: "claude-code-subprocess"
|
|
2147
|
-
});
|
|
2148
|
-
}
|
|
2149
|
-
};
|
|
2150
|
-
var claude_code_default = claudeCodeRuntime;
|
|
2151
|
-
|
|
2152
|
-
// src/domains/providers/runtimes/claude/claude-sdk.ts
|
|
2153
|
-
init_esm_shims();
|
|
2154
|
-
var claudeSdkRuntime = {
|
|
2155
|
-
id: "claude-sdk",
|
|
2156
|
-
displayName: "Claude Agent SDK",
|
|
2157
|
-
kind: "sdk",
|
|
2158
|
-
tier: "subscription",
|
|
2159
|
-
apiFamily: "claude-agent-sdk",
|
|
2160
|
-
auth: "claude-cli",
|
|
2161
|
-
authNotice: CLAUDE_CODE_AUTH_NOTICE,
|
|
2162
|
-
knownModels: [...CLAUDE_CODE_MODELS],
|
|
2163
|
-
binaryName: "claude",
|
|
2164
|
-
headlessCommand: "@anthropic-ai/claude-agent-sdk query()",
|
|
2165
|
-
outputParser: "claude-agent-sdk-messages",
|
|
2166
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2167
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2168
|
-
return synthesizeClaudeDelegatedModel({
|
|
2169
|
-
target,
|
|
2170
|
-
wireModelId,
|
|
2171
|
-
kb,
|
|
2172
|
-
defaultCapabilities: claudeCodeCapabilities,
|
|
2173
|
-
runtimeId: "claude-sdk",
|
|
2174
|
-
apiFamily: "claude-agent-sdk"
|
|
2175
|
-
});
|
|
2176
|
-
}
|
|
2177
|
-
};
|
|
2178
|
-
var claude_sdk_default = claudeSdkRuntime;
|
|
2179
|
-
|
|
2180
|
-
// src/domains/providers/runtimes/cloud/alcf.ts
|
|
2181
|
-
init_esm_shims();
|
|
2182
|
-
|
|
2183
|
-
// src/domains/providers/probe/http.ts
|
|
2184
|
-
init_esm_shims();
|
|
2185
|
-
import { performance as performance2 } from "node:perf_hooks";
|
|
2186
|
-
async function probeHttp(opts) {
|
|
2187
|
-
const { response, latencyMs, error } = await runFetch(opts);
|
|
2188
|
-
if (response === null) return { ok: false, error: error ?? "unknown transport error", latencyMs };
|
|
2189
|
-
const method = opts.method ?? "GET";
|
|
2190
|
-
if (response.ok) return { ok: true, latencyMs };
|
|
2191
|
-
if (method === "HEAD" && response.status === 405) return { ok: true, latencyMs };
|
|
2192
|
-
return {
|
|
2193
|
-
ok: false,
|
|
2194
|
-
latencyMs,
|
|
2195
|
-
error: `HTTP ${response.status}: ${response.statusText}`
|
|
2196
|
-
};
|
|
2197
|
-
}
|
|
2198
|
-
async function probeJson(opts) {
|
|
2199
|
-
const { response, latencyMs, error } = await runFetch(opts);
|
|
2200
|
-
if (response === null) return { ok: false, error: error ?? "unknown transport error", latencyMs };
|
|
2201
|
-
const method = opts.method ?? "GET";
|
|
2202
|
-
if (!response.ok && !(method === "HEAD" && response.status === 405)) {
|
|
2203
|
-
return {
|
|
2204
|
-
ok: false,
|
|
2205
|
-
latencyMs,
|
|
2206
|
-
error: `HTTP ${response.status}: ${response.statusText}`
|
|
2207
|
-
};
|
|
2208
|
-
}
|
|
2209
|
-
let data;
|
|
2210
|
-
try {
|
|
2211
|
-
data = await response.json();
|
|
2212
|
-
} catch (err) {
|
|
2213
|
-
return { ok: false, latencyMs, error: `JSON parse: ${describeError2(err)}` };
|
|
2214
|
-
}
|
|
2215
|
-
return { ok: true, latencyMs, data };
|
|
2216
|
-
}
|
|
2217
|
-
async function runFetch(opts) {
|
|
2218
|
-
const controller = new AbortController();
|
|
2219
|
-
let timedOut = false;
|
|
2220
|
-
const timer = setTimeout(() => {
|
|
2221
|
-
timedOut = true;
|
|
2222
|
-
controller.abort();
|
|
2223
|
-
}, opts.timeoutMs);
|
|
2224
|
-
const onExternalAbort = () => controller.abort();
|
|
2225
|
-
if (opts.signal) {
|
|
2226
|
-
if (opts.signal.aborted) controller.abort();
|
|
2227
|
-
else opts.signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
2228
|
-
}
|
|
2229
|
-
const init = {
|
|
2230
|
-
method: opts.method ?? "GET",
|
|
2231
|
-
signal: controller.signal
|
|
2232
|
-
};
|
|
2233
|
-
if (opts.headers) init.headers = opts.headers;
|
|
2234
|
-
if (opts.body !== void 0) init.body = opts.body;
|
|
2235
|
-
const started = performance2.now();
|
|
2236
|
-
try {
|
|
2237
|
-
const response = await fetch(opts.url, init);
|
|
2238
|
-
return { response, latencyMs: Math.round(performance2.now() - started) };
|
|
2239
|
-
} catch (err) {
|
|
2240
|
-
const latencyMs = Math.round(performance2.now() - started);
|
|
2241
|
-
if (timedOut) {
|
|
2242
|
-
return { response: null, latencyMs, error: `timeout after ${opts.timeoutMs}ms` };
|
|
2243
|
-
}
|
|
2244
|
-
if (opts.signal?.aborted) {
|
|
2245
|
-
return { response: null, latencyMs, error: "aborted by caller" };
|
|
2246
|
-
}
|
|
2247
|
-
return { response: null, latencyMs, error: describeError2(err) };
|
|
2248
|
-
} finally {
|
|
2249
|
-
clearTimeout(timer);
|
|
2250
|
-
if (opts.signal) opts.signal.removeEventListener("abort", onExternalAbort);
|
|
2251
|
-
}
|
|
2252
|
-
}
|
|
2253
|
-
function describeError2(err) {
|
|
2254
|
-
if (!(err instanceof Error)) return String(err);
|
|
2255
|
-
const cause = err.cause;
|
|
2256
|
-
if (cause instanceof Error && cause.message.length > 0) {
|
|
2257
|
-
const code = cause.code;
|
|
2258
|
-
return typeof code === "string" && !cause.message.includes(code) ? `${cause.message} (${code})` : cause.message;
|
|
2259
|
-
}
|
|
2260
|
-
return err.message;
|
|
2261
|
-
}
|
|
2262
|
-
|
|
2263
|
-
// src/domains/providers/runtimes/protocol/openai-compat.ts
|
|
2264
|
-
init_esm_shims();
|
|
2265
|
-
|
|
2266
|
-
// src/core/context-floor.ts
|
|
2267
|
-
init_esm_shims();
|
|
2268
|
-
var CLIO_MIN_CONTEXT_WINDOW = 131072;
|
|
2269
|
-
var CLIO_MIN_MAX_OUTPUT_TOKENS = 32768;
|
|
2270
|
-
var CLIO_CONTEXT_WINDOW_WARN_BELOW = 128e3;
|
|
2271
|
-
|
|
2272
|
-
// src/domains/providers/probe/reasoning.ts
|
|
2273
|
-
init_esm_shims();
|
|
2274
|
-
import { performance as performance3 } from "node:perf_hooks";
|
|
2275
|
-
var PROMPT = "What is 2+2? Think briefly, then answer.";
|
|
2276
|
-
function nonEmptyString2(value) {
|
|
2277
|
-
return typeof value === "string" && value.trim().length > 0;
|
|
2278
|
-
}
|
|
2279
|
-
function detectReasoningField(data) {
|
|
2280
|
-
const message = data.choices?.[0]?.message;
|
|
2281
|
-
if (!message) return null;
|
|
2282
|
-
if (nonEmptyString2(message.reasoning_content)) return "reasoning_content";
|
|
2283
|
-
if (nonEmptyString2(message.reasoning)) return "reasoning";
|
|
2284
|
-
if (nonEmptyString2(message.reasoning_text)) return "reasoning_text";
|
|
2285
|
-
return null;
|
|
2286
|
-
}
|
|
2287
|
-
function trimTrailingSlash(url2) {
|
|
2288
|
-
return url2.endsWith("/") ? url2.slice(0, -1) : url2;
|
|
2289
|
-
}
|
|
2290
|
-
async function probeOpenAICompatReasoning(opts) {
|
|
2291
|
-
const controller = new AbortController();
|
|
2292
|
-
let timedOut = false;
|
|
2293
|
-
const timer = setTimeout(() => {
|
|
2294
|
-
timedOut = true;
|
|
2295
|
-
controller.abort();
|
|
2296
|
-
}, opts.timeoutMs);
|
|
2297
|
-
const onExternalAbort = () => controller.abort();
|
|
2298
|
-
if (opts.signal) {
|
|
2299
|
-
if (opts.signal.aborted) controller.abort();
|
|
2300
|
-
else opts.signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
2301
|
-
}
|
|
2302
|
-
const headers = { "content-type": "application/json" };
|
|
2303
|
-
if (opts.apiKey && opts.apiKey.length > 0) headers.authorization = `Bearer ${opts.apiKey}`;
|
|
2304
|
-
const body = JSON.stringify({
|
|
2305
|
-
model: opts.modelId,
|
|
2306
|
-
messages: [{ role: "user", content: PROMPT }],
|
|
2307
|
-
max_tokens: 200,
|
|
2308
|
-
temperature: 0,
|
|
2309
|
-
stream: false,
|
|
2310
|
-
reasoning_effort: "low"
|
|
2311
|
-
});
|
|
2312
|
-
const url2 = `${trimTrailingSlash(opts.baseUrl)}/v1/chat/completions`;
|
|
2313
|
-
const started = performance3.now();
|
|
2314
|
-
try {
|
|
2315
|
-
const response = await fetch(url2, {
|
|
2316
|
-
method: "POST",
|
|
2317
|
-
headers,
|
|
2318
|
-
body,
|
|
2319
|
-
signal: controller.signal
|
|
2320
|
-
});
|
|
2321
|
-
const latencyMs = Math.round(performance3.now() - started);
|
|
2322
|
-
if (!response.ok) {
|
|
2323
|
-
return { reasoning: false, latencyMs, error: `HTTP ${response.status}: ${response.statusText}` };
|
|
2324
|
-
}
|
|
2325
|
-
const data = await response.json();
|
|
2326
|
-
const field = detectReasoningField(data);
|
|
2327
|
-
if (field) return { reasoning: true, field, latencyMs };
|
|
2328
|
-
return { reasoning: false, latencyMs };
|
|
2329
|
-
} catch (err) {
|
|
2330
|
-
const latencyMs = Math.round(performance3.now() - started);
|
|
2331
|
-
if (timedOut) return { reasoning: false, latencyMs, error: `timeout after ${opts.timeoutMs}ms` };
|
|
2332
|
-
if (opts.signal?.aborted) return { reasoning: false, latencyMs, error: "aborted by caller" };
|
|
2333
|
-
return { reasoning: false, latencyMs, error: err instanceof Error ? err.message : String(err) };
|
|
2334
|
-
} finally {
|
|
2335
|
-
clearTimeout(timer);
|
|
2336
|
-
if (opts.signal) opts.signal.removeEventListener("abort", onExternalAbort);
|
|
2337
|
-
}
|
|
2338
|
-
}
|
|
2339
|
-
|
|
2340
|
-
// src/domains/providers/runtimes/common/probe-helpers.ts
|
|
2341
|
-
init_esm_shims();
|
|
2342
|
-
|
|
2343
|
-
// src/domains/providers/types/context-window-slots.ts
|
|
2344
|
-
init_esm_shims();
|
|
2345
|
-
function formatContextWindowSlots(contextWindow, slots) {
|
|
2346
|
-
const format = (n) => Math.round(n).toLocaleString("en-US");
|
|
2347
|
-
return `${format(contextWindow)} (${format(slots.totalContextSize)} / ${slots.slots} slots)`;
|
|
2348
|
-
}
|
|
2349
|
-
|
|
2350
|
-
// src/domains/providers/runtimes/common/probe-helpers.ts
|
|
2351
|
-
async function probeUrl(url2, ctx, method = "GET") {
|
|
2352
|
-
const base = { url: url2, timeoutMs: ctx.httpTimeoutMs, method };
|
|
2353
|
-
return ctx.signal ? probeHttp({ ...base, signal: ctx.signal }) : probeHttp(base);
|
|
2354
|
-
}
|
|
2355
|
-
async function probeOpenAIModels(base, ctx, modelsPath = "/v1/models") {
|
|
2356
|
-
return (await probeOpenAIModelCatalog(base, ctx, modelsPath)).models;
|
|
2357
|
-
}
|
|
2358
|
-
var OPENAI_COMPAT_DETAIL_PATHS = ["/api/v0/models"];
|
|
2359
|
-
async function probeModelDetailRows(base, ctx, modelsPath) {
|
|
2360
|
-
const rows = /* @__PURE__ */ new Map();
|
|
2361
|
-
for (const detailPath of OPENAI_COMPAT_DETAIL_PATHS) {
|
|
2362
|
-
if (detailPath === modelsPath) continue;
|
|
2363
|
-
const opts = { url: `${base}${detailPath}`, timeoutMs: ctx.httpTimeoutMs };
|
|
2364
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2365
|
-
if (!result.ok || !Array.isArray(result.data?.data)) continue;
|
|
2366
|
-
for (const row of result.data.data) {
|
|
2367
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2368
|
-
if (!rows.has(row.id)) rows.set(row.id, row);
|
|
2369
|
-
}
|
|
2370
|
-
if (rows.size > 0) break;
|
|
2371
|
-
}
|
|
2372
|
-
return rows;
|
|
2373
|
-
}
|
|
2374
|
-
async function probeOpenAIModelCatalog(base, ctx, modelsPath = "/v1/models") {
|
|
2375
|
-
const opts = { url: `${base}${modelsPath}`, timeoutMs: ctx.httpTimeoutMs };
|
|
2376
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2377
|
-
if (!result.ok || !result.data?.data) return { models: [], modelCapabilities: {}, modelStates: {} };
|
|
2378
|
-
const detail = await probeModelDetailRows(base, ctx, modelsPath);
|
|
2379
|
-
const models = [];
|
|
2380
|
-
const modelCapabilities = {};
|
|
2381
|
-
const modelStates = {};
|
|
2382
|
-
for (const row of result.data.data) {
|
|
2383
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2384
|
-
models.push(row.id);
|
|
2385
|
-
const detailRow = detail.get(row.id);
|
|
2386
|
-
const caps = {
|
|
2387
|
-
...detailRow ? capabilitiesFromOpenAIModelEntry(detailRow) : {},
|
|
2388
|
-
...capabilitiesFromOpenAIModelEntry(row)
|
|
2389
|
-
};
|
|
2390
|
-
if (Object.keys(caps).length > 0) modelCapabilities[row.id] = caps;
|
|
2391
|
-
const loadedContext = loadedContextFromEntry(row) ?? (detailRow ? loadedContextFromEntry(detailRow) : void 0);
|
|
2392
|
-
const state = modelStateFromOpenAIModelEntry(row) ?? (detailRow ? modelStateFromOpenAIModelEntry(detailRow) : void 0) ?? // A reported loaded context is itself the residency answer: nothing
|
|
2393
|
-
// serves a window for a model it has not loaded.
|
|
2394
|
-
(loadedContext !== void 0 ? { state: "loaded" } : void 0);
|
|
2395
|
-
const contextSlots = contextSlotsFromEntry(row) ?? (detailRow ? contextSlotsFromEntry(detailRow) : void 0);
|
|
2396
|
-
const withSlots = contextSlots ? { ...state ?? { state: "unknown" }, contextSlots } : state;
|
|
2397
|
-
if (withSlots) {
|
|
2398
|
-
modelStates[row.id] = loadedContext === void 0 ? withSlots : { ...withSlots, contextLength: loadedContext };
|
|
2399
|
-
}
|
|
2400
|
-
}
|
|
2401
|
-
return { models, modelCapabilities, modelStates };
|
|
2402
|
-
}
|
|
2403
|
-
function isRecord3(value) {
|
|
2404
|
-
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
2405
|
-
}
|
|
2406
|
-
function positiveNumber2(value) {
|
|
2407
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
2408
|
-
}
|
|
2409
|
-
function firstPositiveNumber(record, keys) {
|
|
2410
|
-
for (const key of keys) {
|
|
2411
|
-
const value = positiveNumber2(record[key]);
|
|
2412
|
-
if (value !== void 0) return value;
|
|
2413
|
-
}
|
|
2414
|
-
return void 0;
|
|
2415
|
-
}
|
|
2416
|
-
function capabilityListFlag(record, name) {
|
|
2417
|
-
const list = record.capabilities;
|
|
2418
|
-
if (!Array.isArray(list)) return void 0;
|
|
2419
|
-
return list.some((entry) => entry === name) ? true : void 0;
|
|
2420
|
-
}
|
|
2421
|
-
function booleanFromAny(record, keys) {
|
|
2422
|
-
for (const key of keys) {
|
|
2423
|
-
const value = record[key];
|
|
2424
|
-
if (typeof value === "boolean") return value;
|
|
2425
|
-
}
|
|
2426
|
-
return void 0;
|
|
2427
|
-
}
|
|
2428
|
-
function nestedRecord(record, key) {
|
|
2429
|
-
const value = record[key];
|
|
2430
|
-
return isRecord3(value) ? value : null;
|
|
2431
|
-
}
|
|
2432
|
-
function firstString(record, keys) {
|
|
2433
|
-
if (!record) return void 0;
|
|
2434
|
-
for (const key of keys) {
|
|
2435
|
-
const value = record[key];
|
|
2436
|
-
if (typeof value === "string" && value.trim().length > 0) return value.trim();
|
|
2437
|
-
}
|
|
2438
|
-
return void 0;
|
|
2439
|
-
}
|
|
2440
|
-
function firstBoolean(record, keys) {
|
|
2441
|
-
if (!record) return void 0;
|
|
2442
|
-
for (const key of keys) {
|
|
2443
|
-
const value = record[key];
|
|
2444
|
-
if (typeof value === "boolean") return value;
|
|
2445
|
-
}
|
|
2446
|
-
return void 0;
|
|
2447
|
-
}
|
|
2448
|
-
function normalizeModelState(raw) {
|
|
2449
|
-
const value = raw?.trim().toLowerCase().replace(/[\s_]+/g, "-");
|
|
2450
|
-
if (!value) return void 0;
|
|
2451
|
-
if (value === "loaded" || value === "ready" || value === "running" || value === "active") return "loaded";
|
|
2452
|
-
if (value === "loading" || value === "pending" || value === "queued" || value === "starting") return "loading";
|
|
2453
|
-
if (value === "unloaded" || value === "not-loaded" || value === "idle" || value === "sleeping" || value === "stopped") {
|
|
2454
|
-
return "unloaded";
|
|
2455
|
-
}
|
|
2456
|
-
if (value === "failed" || value === "error" || value === "errored") return "failed";
|
|
2457
|
-
if (value === "unknown") return "unknown";
|
|
2458
|
-
return void 0;
|
|
2459
|
-
}
|
|
2460
|
-
function modelStateFromOpenAIModelEntry(row) {
|
|
2461
|
-
const status = nestedRecord(row, "status");
|
|
2462
|
-
const failed = firstBoolean(status, ["failed"]) ?? firstBoolean(row, ["failed"]);
|
|
2463
|
-
if (failed === true) {
|
|
2464
|
-
const detail = firstString(status, ["error", "reason", "message"]) ?? firstString(row, ["error", "reason", "message"]);
|
|
2465
|
-
return detail ? { state: "failed", detail } : { state: "failed" };
|
|
2466
|
-
}
|
|
2467
|
-
const raw = typeof row.status === "string" ? row.status : firstString(status, ["value", "state", "status"]) ?? firstString(row, ["state"]);
|
|
2468
|
-
const normalized = normalizeModelState(raw);
|
|
2469
|
-
if (normalized) {
|
|
2470
|
-
const detail = firstString(status, ["detail", "message", "reason"]);
|
|
2471
|
-
return detail ? { state: normalized, detail } : { state: normalized };
|
|
2472
|
-
}
|
|
2473
|
-
const loaded = firstBoolean(status, ["loaded"]) ?? firstBoolean(row, ["loaded"]);
|
|
2474
|
-
if (loaded === true) return { state: "loaded" };
|
|
2475
|
-
if (loaded === false) return { state: "unloaded" };
|
|
2476
|
-
return void 0;
|
|
2477
|
-
}
|
|
2478
|
-
function loadedContextFromEntry(row) {
|
|
2479
|
-
const reported = firstPositiveNumber(row, ["loaded_context_length", "loadedContextLength"]);
|
|
2480
|
-
return reported === void 0 ? void 0 : Math.floor(reported);
|
|
2481
|
-
}
|
|
2482
|
-
function statusArgsFromEntry(row) {
|
|
2483
|
-
const status = nestedRecord(row, "status");
|
|
2484
|
-
return argsFromStatus(status);
|
|
2485
|
-
}
|
|
2486
|
-
function contextSlotsFromEntry(row) {
|
|
2487
|
-
return llamaCppRequestContextWindow(parseLlamaCppServerFlags(statusArgsFromEntry(row)))?.slots;
|
|
2488
|
-
}
|
|
2489
|
-
function capabilitiesFromOpenAIModelEntry(row) {
|
|
2490
|
-
const caps = {};
|
|
2491
|
-
const meta = nestedRecord(row, "meta");
|
|
2492
|
-
const flags = parseLlamaCppServerFlags(statusArgsFromEntry(row));
|
|
2493
|
-
const contextWindow = llamaCppRequestContextWindow(flags)?.contextWindow ?? firstPositiveNumber(row, [
|
|
2494
|
-
// What is actually loaded outranks what the model could support: a
|
|
2495
|
-
// model served at 8k out of a possible 262k has an 8k window today,
|
|
2496
|
-
// and the run has to be planned against the real one.
|
|
2497
|
-
"loaded_context_length",
|
|
2498
|
-
"loadedContextLength",
|
|
2499
|
-
"context_window",
|
|
2500
|
-
"contextWindow",
|
|
2501
|
-
"context_length",
|
|
2502
|
-
"contextLength",
|
|
2503
|
-
"max_context_length",
|
|
2504
|
-
"maxContextLength",
|
|
2505
|
-
"n_ctx"
|
|
2506
|
-
]) ?? (meta ? firstPositiveNumber(meta, ["n_ctx", "n_ctx_train", "context_length", "contextWindow"]) : void 0);
|
|
2507
|
-
if (contextWindow !== void 0) caps.contextWindow = Math.floor(contextWindow);
|
|
2508
|
-
const maxTokens = positiveNumber2(flags.maxTokens) ?? firstPositiveNumber(row, [
|
|
2509
|
-
"max_output_tokens",
|
|
2510
|
-
"maxOutputTokens",
|
|
2511
|
-
"max_completion_tokens",
|
|
2512
|
-
"maxCompletionTokens",
|
|
2513
|
-
"max_tokens",
|
|
2514
|
-
"maxTokens",
|
|
2515
|
-
"n_predict"
|
|
2516
|
-
]);
|
|
2517
|
-
if (maxTokens !== void 0) caps.maxTokens = Math.floor(maxTokens);
|
|
2518
|
-
const tools = flags.jinja ?? booleanFromAny(row, ["tools", "tool_use", "toolUse", "trained_for_tool_use"]) ?? capabilityListFlag(row, "tool_use");
|
|
2519
|
-
if (tools !== void 0) caps.tools = tools;
|
|
2520
|
-
const reasoning = flags.reasoning ?? (flags.reasoningBudget !== void 0 ? true : void 0) ?? booleanFromAny(row, ["reasoning", "thinking"]);
|
|
2521
|
-
if (reasoning !== void 0) caps.reasoning = reasoning;
|
|
2522
|
-
const architecture = nestedRecord(row, "architecture");
|
|
2523
|
-
const architectureInput = architecture?.input_modalities;
|
|
2524
|
-
const modalities = Array.isArray(row.modalities) ? row.modalities : Array.isArray(row.input) ? row.input : Array.isArray(architectureInput) ? architectureInput : null;
|
|
2525
|
-
if (modalities) {
|
|
2526
|
-
caps.vision = modalities.some((entry) => entry === "image" || entry === "vision");
|
|
2527
|
-
if (modalities.some((entry) => entry === "audio")) caps.audio = true;
|
|
2528
|
-
}
|
|
2529
|
-
return caps;
|
|
2530
|
-
}
|
|
2531
|
-
function modelEntries(payload) {
|
|
2532
|
-
if (!Array.isArray(payload?.data)) return [];
|
|
2533
|
-
const out = [];
|
|
2534
|
-
for (const row of payload.data) {
|
|
2535
|
-
if (typeof row?.id !== "string" || row.id.length === 0) continue;
|
|
2536
|
-
out.push({ id: row.id, status: row.status });
|
|
2537
|
-
}
|
|
2538
|
-
return out;
|
|
2539
|
-
}
|
|
2540
|
-
async function probeOpenAIModelEntries(base, ctx) {
|
|
2541
|
-
const opts = { url: `${base}/v1/models`, timeoutMs: ctx.httpTimeoutMs };
|
|
2542
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2543
|
-
if (!result.ok) return [];
|
|
2544
|
-
return modelEntries(result.data);
|
|
2545
|
-
}
|
|
2546
|
-
function llamaCppRequestContextWindow(flags) {
|
|
2547
|
-
const total = positiveNumber2(flags.contextSize);
|
|
2548
|
-
if (total === void 0) return void 0;
|
|
2549
|
-
const parallel = positiveNumber2(flags.parallel);
|
|
2550
|
-
if (parallel === void 0 || parallel <= 1 || flags.kvUnified === true) return { contextWindow: Math.floor(total) };
|
|
2551
|
-
const slots = Math.floor(parallel);
|
|
2552
|
-
return {
|
|
2553
|
-
contextWindow: Math.floor(total / slots),
|
|
2554
|
-
slots: { totalContextSize: Math.floor(total), slots }
|
|
2555
|
-
};
|
|
2556
|
-
}
|
|
2557
|
-
function argsFromStatus(status) {
|
|
2558
|
-
if (!isRecord3(status)) return [];
|
|
2559
|
-
const args = status.args;
|
|
2560
|
-
if (Array.isArray(args)) return args.filter((entry) => typeof entry === "string");
|
|
2561
|
-
if (typeof args === "string") return args.trim().split(/\s+/).filter(Boolean);
|
|
2562
|
-
return [];
|
|
2563
|
-
}
|
|
2564
|
-
function looksLikeFlag(token) {
|
|
2565
|
-
return token.startsWith("-") && Number.isNaN(Number(token));
|
|
2566
|
-
}
|
|
2567
|
-
function valueAfter(args, ...flags) {
|
|
2568
|
-
for (const flag of flags) {
|
|
2569
|
-
const index = args.indexOf(flag);
|
|
2570
|
-
if (index < 0) continue;
|
|
2571
|
-
const value = args[index + 1];
|
|
2572
|
-
return value && !looksLikeFlag(value) ? value : void 0;
|
|
2573
|
-
}
|
|
2574
|
-
return void 0;
|
|
2575
|
-
}
|
|
2576
|
-
function numberFlag(args, ...flags) {
|
|
2577
|
-
const value = valueAfter(args, ...flags);
|
|
2578
|
-
if (value === void 0) return void 0;
|
|
2579
|
-
const parsed = Number(value);
|
|
2580
|
-
return Number.isFinite(parsed) ? parsed : void 0;
|
|
2581
|
-
}
|
|
2582
|
-
function booleanFlag(args, ...flags) {
|
|
2583
|
-
const flag = flags.find((candidate) => args.includes(candidate));
|
|
2584
|
-
if (flag === void 0) return void 0;
|
|
2585
|
-
const value = valueAfter(args, flag);
|
|
2586
|
-
if (value === void 0) return true;
|
|
2587
|
-
const normalized = value.toLowerCase();
|
|
2588
|
-
if (normalized === "true" || normalized === "on" || normalized === "1") return true;
|
|
2589
|
-
if (normalized === "false" || normalized === "off" || normalized === "0") return false;
|
|
2590
|
-
return void 0;
|
|
2591
|
-
}
|
|
2592
|
-
function parseLlamaCppServerFlags(args) {
|
|
2593
|
-
const flags = {};
|
|
2594
|
-
const ctxSize = numberFlag(args, "--ctx-size", "-c");
|
|
2595
|
-
if (ctxSize !== void 0) flags.contextSize = ctxSize;
|
|
2596
|
-
const maxTokens = numberFlag(args, "--n-predict");
|
|
2597
|
-
if (maxTokens !== void 0) flags.maxTokens = maxTokens;
|
|
2598
|
-
const flashAttention = booleanFlag(args, "--flash-attn");
|
|
2599
|
-
if (flashAttention !== void 0) flags.flashAttention = flashAttention;
|
|
2600
|
-
const jinja = booleanFlag(args, "--jinja");
|
|
2601
|
-
if (jinja !== void 0) flags.jinja = jinja;
|
|
2602
|
-
const reasoningRaw = valueAfter(args, "--reasoning");
|
|
2603
|
-
if (reasoningRaw) flags.reasoning = reasoningRaw === "on" || reasoningRaw === "true" || reasoningRaw === "1";
|
|
2604
|
-
const reasoningBudget = numberFlag(args, "--reasoning-budget");
|
|
2605
|
-
if (reasoningBudget !== void 0) flags.reasoningBudget = reasoningBudget;
|
|
2606
|
-
const temperature = numberFlag(args, "--temperature");
|
|
2607
|
-
if (temperature !== void 0) flags.temperature = temperature;
|
|
2608
|
-
const topP = numberFlag(args, "--top-p");
|
|
2609
|
-
if (topP !== void 0) flags.topP = topP;
|
|
2610
|
-
const topK = numberFlag(args, "--top-k");
|
|
2611
|
-
if (topK !== void 0) flags.topK = topK;
|
|
2612
|
-
const nGpuLayers = numberFlag(args, "--n-gpu-layers");
|
|
2613
|
-
if (nGpuLayers !== void 0) flags.nGpuLayers = nGpuLayers;
|
|
2614
|
-
const parallel = numberFlag(args, "--parallel", "-np");
|
|
2615
|
-
if (parallel !== void 0) flags.parallel = parallel;
|
|
2616
|
-
const kvUnifiedAt = Math.max(args.lastIndexOf("--kv-unified"), args.lastIndexOf("-kvu"));
|
|
2617
|
-
const noKvUnifiedAt = args.lastIndexOf("--no-kv-unified");
|
|
2618
|
-
if (noKvUnifiedAt > kvUnifiedAt) flags.kvUnified = false;
|
|
2619
|
-
else if (kvUnifiedAt >= 0) flags.kvUnified = booleanFlag(args, "--kv-unified", "-kvu") ?? true;
|
|
2620
|
-
const cacheTypeK = valueAfter(args, "--cache-type-k");
|
|
2621
|
-
if (cacheTypeK) flags.cacheTypeK = cacheTypeK;
|
|
2622
|
-
const cacheTypeV = valueAfter(args, "--cache-type-v");
|
|
2623
|
-
if (cacheTypeV) flags.cacheTypeV = cacheTypeV;
|
|
2624
|
-
const mmproj = valueAfter(args, "--mmproj");
|
|
2625
|
-
if (mmproj) flags.mmproj = mmproj;
|
|
2626
|
-
const chatTemplateKwargs = valueAfter(args, "--chat-template-kwargs");
|
|
2627
|
-
if (chatTemplateKwargs) flags.chatTemplateKwargs = chatTemplateKwargs;
|
|
2628
|
-
return flags;
|
|
2629
|
-
}
|
|
2630
|
-
function selectedModelEntry(entries, target) {
|
|
2631
|
-
const expected = target.defaultModel?.trim();
|
|
2632
|
-
if (expected) return entries.find((entry) => entry.id === expected) ?? null;
|
|
2633
|
-
return entries[0] ?? null;
|
|
2634
|
-
}
|
|
2635
|
-
function statusNotes(id, status) {
|
|
2636
|
-
if (!isRecord3(status)) return [];
|
|
2637
|
-
const notes = [];
|
|
2638
|
-
if (status.failed === true) notes.push(`llama.cpp router marks ${id} as failed`);
|
|
2639
|
-
const state = typeof status.state === "string" ? status.state : void 0;
|
|
2640
|
-
if (state === "loading") notes.push(`llama.cpp router reports ${id} is still loading`);
|
|
2641
|
-
return notes;
|
|
2642
|
-
}
|
|
2643
|
-
async function probeLlamaCppModelStatus(base, target, ctx) {
|
|
2644
|
-
const entries = await probeOpenAIModelEntries(base, ctx);
|
|
2645
|
-
const selected = selectedModelEntry(entries, target);
|
|
2646
|
-
if (!selected) return {};
|
|
2647
|
-
const args = argsFromStatus(selected.status);
|
|
2648
|
-
if (args.length === 0) return { notes: statusNotes(selected.id, selected.status) };
|
|
2649
|
-
const flags = parseLlamaCppServerFlags(args);
|
|
2650
|
-
const caps = {};
|
|
2651
|
-
const window = llamaCppRequestContextWindow(flags);
|
|
2652
|
-
if (window !== void 0) caps.contextWindow = window.contextWindow;
|
|
2653
|
-
if (flags.maxTokens !== void 0 && flags.maxTokens > 0) caps.maxTokens = flags.maxTokens;
|
|
2654
|
-
if (flags.reasoning === true || flags.reasoningBudget !== void 0) caps.reasoning = true;
|
|
2655
|
-
if (flags.mmproj) caps.vision = true;
|
|
2656
|
-
if (flags.jinja === true) caps.tools = true;
|
|
2657
|
-
const enrichment = { modelId: selected.id, serverFlags: flags };
|
|
2658
|
-
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
2659
|
-
const notes = statusNotes(selected.id, selected.status);
|
|
2660
|
-
if (window?.slots) {
|
|
2661
|
-
notes.push(
|
|
2662
|
-
`${selected.id} context window ${formatContextWindowSlots(window.contextWindow, window.slots)}: --ctx-size is split across --parallel slots without --kv-unified`
|
|
2663
|
-
);
|
|
2664
|
-
}
|
|
2665
|
-
if (notes.length > 0) enrichment.notes = notes;
|
|
2666
|
-
return enrichment;
|
|
2667
|
-
}
|
|
2668
|
-
async function detectModelMismatch(base, target, ctx) {
|
|
2669
|
-
const expected = target.defaultModel?.trim();
|
|
2670
|
-
if (!expected) return null;
|
|
2671
|
-
const ids = await probeOpenAIModels(base, ctx);
|
|
2672
|
-
if (ids.length === 0) return null;
|
|
2673
|
-
if (ids.includes(expected)) return null;
|
|
2674
|
-
const loaded = ids[0] ?? "(unknown)";
|
|
2675
|
-
return `wire model id ${expected} does not match server's loaded model ${loaded}; llama.cpp serves a single fixed model`;
|
|
2676
|
-
}
|
|
2677
|
-
async function probeLlamaCppProps(base, ctx) {
|
|
2678
|
-
const opts = { url: `${base}/props`, timeoutMs: ctx.httpTimeoutMs };
|
|
2679
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
2680
|
-
if (!result.ok || !result.data) return {};
|
|
2681
|
-
const data = result.data;
|
|
2682
|
-
const enrichment = {};
|
|
2683
|
-
const caps = {};
|
|
2684
|
-
const nCtx = data.default_generation_settings?.n_ctx;
|
|
2685
|
-
if (typeof nCtx === "number" && nCtx > 0) caps.contextWindow = nCtx;
|
|
2686
|
-
const nPredict = data.default_generation_settings?.n_predict;
|
|
2687
|
-
if (typeof nPredict === "number" && nPredict > 0) caps.maxTokens = nPredict;
|
|
2688
|
-
const vision = data.modalities?.vision;
|
|
2689
|
-
if (typeof vision === "boolean") caps.vision = vision;
|
|
2690
|
-
if (Object.keys(caps).length > 0) enrichment.discoveredCapabilities = caps;
|
|
2691
|
-
if (typeof data.build_info === "string" && data.build_info.length > 0) {
|
|
2692
|
-
enrichment.serverVersion = data.build_info;
|
|
2693
|
-
}
|
|
2694
|
-
return enrichment;
|
|
2695
|
-
}
|
|
2696
|
-
|
|
2697
|
-
// src/domains/providers/runtimes/protocol/openai-compat.ts
|
|
2698
|
-
function synthesizeOpenAICompatModel(input) {
|
|
2699
|
-
return synthLocalModel({
|
|
2700
|
-
target: input.target,
|
|
2701
|
-
wireModelId: input.wireModelId,
|
|
2702
|
-
kb: input.kb,
|
|
2703
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2704
|
-
apiFamily: input.apiFamily ?? "openai-completions",
|
|
2705
|
-
provider: input.provider,
|
|
2706
|
-
baseUrlForTarget: input.baseUrlForTarget ?? withV1
|
|
2707
|
-
});
|
|
2708
|
-
}
|
|
2709
|
-
function makeOpenAICompatRuntime(spec) {
|
|
2710
|
-
const apiFamily = spec.apiFamily ?? "openai-completions";
|
|
2711
|
-
const asIs = spec.baseUrlStyle === "asIs";
|
|
2712
|
-
const baseUrlForTarget = asIs ? withAsIs : withV1;
|
|
2713
|
-
const healthPath = spec.healthPath ?? "/v1/models";
|
|
2714
|
-
const modelsPath = spec.modelsPath ?? "/v1/models";
|
|
2715
|
-
const probeBase = asIs ? targetBaseUrl : targetRootUrl;
|
|
2716
|
-
return {
|
|
2717
|
-
id: spec.id,
|
|
2718
|
-
displayName: spec.displayName,
|
|
2719
|
-
kind: "http",
|
|
2720
|
-
tier: spec.tier,
|
|
2721
|
-
apiFamily,
|
|
2722
|
-
auth: spec.auth,
|
|
2723
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
2724
|
-
async probe(target, ctx) {
|
|
2725
|
-
const base = probeBase(target);
|
|
2726
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
2727
|
-
const health = await probeUrl(`${base}${healthPath}`, ctx);
|
|
2728
|
-
if (!health.ok) return health;
|
|
2729
|
-
const catalog = await probeOpenAIModelCatalog(base, ctx, modelsPath);
|
|
2730
|
-
const result = { ...health };
|
|
2731
|
-
if (catalog.models.length > 0) result.models = catalog.models;
|
|
2732
|
-
if (Object.keys(catalog.modelStates).length > 0) result.modelStates = catalog.modelStates;
|
|
2733
|
-
if (Object.keys(catalog.modelCapabilities).length > 0) {
|
|
2734
|
-
result.modelCapabilities = catalog.modelCapabilities;
|
|
2735
|
-
const selected = target.defaultModel?.trim();
|
|
2736
|
-
const selectedCaps = selected ? catalog.modelCapabilities[selected] : void 0;
|
|
2737
|
-
if (selected && selectedCaps) {
|
|
2738
|
-
result.discoveredCapabilities = selectedCaps;
|
|
2739
|
-
result.capabilityModelId = selected;
|
|
2740
|
-
}
|
|
2741
|
-
}
|
|
2742
|
-
return result;
|
|
2743
|
-
},
|
|
2744
|
-
async probeModels(target, ctx) {
|
|
2745
|
-
const base = probeBase(target);
|
|
2746
|
-
if (!base) return [];
|
|
2747
|
-
return probeOpenAIModels(base, ctx, modelsPath);
|
|
2748
|
-
},
|
|
2749
|
-
async probeReasoning(target, modelId, ctx) {
|
|
2750
|
-
const base = probeBase(target);
|
|
2751
|
-
if (!base) return { reasoning: false, latencyMs: 0, error: "target has no url" };
|
|
2752
|
-
const apiKeyEnv = target.auth?.apiKeyEnvVar;
|
|
2753
|
-
const apiKey = apiKeyEnv && ctx.credentialsPresent.has(apiKeyEnv) ? process.env[apiKeyEnv] : void 0;
|
|
2754
|
-
const probeOpts = {
|
|
2755
|
-
baseUrl: base,
|
|
2756
|
-
modelId,
|
|
2757
|
-
timeoutMs: Math.max(ctx.httpTimeoutMs, 8e3)
|
|
2758
|
-
};
|
|
2759
|
-
if (apiKey) probeOpts.apiKey = apiKey;
|
|
2760
|
-
if (ctx.signal) probeOpts.signal = ctx.signal;
|
|
2761
|
-
return probeOpenAICompatReasoning(probeOpts);
|
|
2762
|
-
},
|
|
2763
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2764
|
-
return synthesizeOpenAICompatModel({
|
|
2765
|
-
target,
|
|
2766
|
-
wireModelId,
|
|
2767
|
-
kb,
|
|
2768
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
2769
|
-
apiFamily,
|
|
2770
|
-
provider: spec.provider,
|
|
2771
|
-
baseUrlForTarget
|
|
2772
|
-
});
|
|
2773
|
-
}
|
|
2774
|
-
};
|
|
2775
|
-
}
|
|
2776
|
-
var defaultCapabilities = {
|
|
2777
|
-
chat: true,
|
|
2778
|
-
tools: false,
|
|
2779
|
-
reasoning: false,
|
|
2780
|
-
vision: false,
|
|
2781
|
-
audio: false,
|
|
2782
|
-
embeddings: false,
|
|
2783
|
-
rerank: false,
|
|
2784
|
-
fim: false,
|
|
2785
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
2786
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
2787
|
-
};
|
|
2788
|
-
var openai_compat_default = makeOpenAICompatRuntime({
|
|
2789
|
-
id: "openai-compat",
|
|
2790
|
-
displayName: "Generic OpenAI-compatible",
|
|
2791
|
-
provider: "openai-compat",
|
|
2792
|
-
auth: "api-key",
|
|
2793
|
-
tier: "protocol",
|
|
2794
|
-
defaultCapabilities
|
|
2795
|
-
});
|
|
2796
|
-
|
|
2797
|
-
// src/domains/providers/runtimes/cloud/alcf.ts
|
|
2798
|
-
var CATALOG_URL = "https://inference-api.alcf.anl.gov/resource_server/list-endpoints";
|
|
2799
|
-
var GATEWAY_ORIGIN = "https://inference-api.alcf.anl.gov/resource_server";
|
|
2800
|
-
var defaultCapabilities2 = {
|
|
2801
|
-
chat: true,
|
|
2802
|
-
tools: true,
|
|
2803
|
-
toolCallFormat: "openai",
|
|
2804
|
-
reasoning: true,
|
|
2805
|
-
vision: false,
|
|
2806
|
-
audio: false,
|
|
2807
|
-
embeddings: false,
|
|
2808
|
-
rerank: false,
|
|
2809
|
-
fim: false,
|
|
2810
|
-
contextWindow: 32768,
|
|
2811
|
-
maxTokens: 4096
|
|
2812
|
-
};
|
|
2813
|
-
var KNOWN_MODELS = [
|
|
2814
|
-
"openai/gpt-oss-120b",
|
|
2815
|
-
"openai/gpt-oss-20b",
|
|
2816
|
-
"gpt-oss-120b",
|
|
2817
|
-
"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
|
2818
|
-
"meta-llama/Llama-4-Scout-17B-16E-Instruct"
|
|
2819
|
-
];
|
|
2820
|
-
function clusterFromUrl(url2) {
|
|
2821
|
-
if (!url2) return null;
|
|
2822
|
-
const match = /\/resource_server\/([^/]+)\//.exec(url2);
|
|
2823
|
-
return match ? match[1] ?? null : null;
|
|
2824
|
-
}
|
|
2825
|
-
function frameworkForCluster(cluster) {
|
|
2826
|
-
return cluster === "metis" ? "api" : "vllm";
|
|
2827
|
-
}
|
|
2828
|
-
function catalogModels(payload, cluster, framework) {
|
|
2829
|
-
const models = payload.clusters?.[cluster]?.frameworks?.[framework]?.models;
|
|
2830
|
-
if (!Array.isArray(models)) return [];
|
|
2831
|
-
return models.map((model) => String(model).trim()).filter((model) => model.length > 0);
|
|
2832
|
-
}
|
|
2833
|
-
function runningModels(payload) {
|
|
2834
|
-
const out = [];
|
|
2835
|
-
for (const job of payload.running ?? []) {
|
|
2836
|
-
for (const raw of String(job.Models ?? "").split(",")) {
|
|
2837
|
-
const id = raw.trim();
|
|
2838
|
-
if (id.length > 0) out.push(id);
|
|
2839
|
-
}
|
|
2840
|
-
}
|
|
2841
|
-
return out;
|
|
2842
|
-
}
|
|
2843
|
-
function dedupe(ids) {
|
|
2844
|
-
const seen = /* @__PURE__ */ new Set();
|
|
2845
|
-
const out = [];
|
|
2846
|
-
for (const id of ids) {
|
|
2847
|
-
if (seen.has(id)) continue;
|
|
2848
|
-
seen.add(id);
|
|
2849
|
-
out.push(id);
|
|
2850
|
-
}
|
|
2851
|
-
return out;
|
|
2852
|
-
}
|
|
2853
|
-
function authHeaders(ctx) {
|
|
2854
|
-
return { Authorization: `Bearer ${ctx.authToken}`, Accept: "application/json" };
|
|
2855
|
-
}
|
|
2856
|
-
async function discover(target, ctx) {
|
|
2857
|
-
if (!target.url) return { ok: false, error: "ALCF target has no url" };
|
|
2858
|
-
const cluster = clusterFromUrl(target.url);
|
|
2859
|
-
if (!cluster) {
|
|
2860
|
-
return { ok: false, error: `cannot determine ALCF cluster from url ${target.url}` };
|
|
2861
|
-
}
|
|
2862
|
-
if (!ctx.authToken) {
|
|
2863
|
-
return { ok: false, error: "ALCF requires Globus auth; run `clio-coder auth login alcf`." };
|
|
2864
|
-
}
|
|
2865
|
-
const framework = frameworkForCluster(cluster);
|
|
2866
|
-
const headers = authHeaders(ctx);
|
|
2867
|
-
const catalog = await probeJson({
|
|
2868
|
-
url: CATALOG_URL,
|
|
2869
|
-
headers,
|
|
2870
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
2871
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
2872
|
-
});
|
|
2873
|
-
if (!catalog.ok || !catalog.data) {
|
|
2874
|
-
const reason = catalog.error ?? "unknown error";
|
|
2875
|
-
const hint = reason.includes("401") ? " (token expired or rejected; run `clio-coder auth login alcf` again)" : "";
|
|
2876
|
-
return {
|
|
2877
|
-
ok: false,
|
|
2878
|
-
error: `ALCF endpoint catalog unreachable: ${reason}${hint}`,
|
|
2879
|
-
...catalog.latencyMs !== void 0 ? { latencyMs: catalog.latencyMs } : {}
|
|
2880
|
-
};
|
|
2881
|
-
}
|
|
2882
|
-
const fromCatalog = catalogModels(catalog.data, cluster, framework);
|
|
2883
|
-
let live = [];
|
|
2884
|
-
try {
|
|
2885
|
-
const jobs = await probeJson({
|
|
2886
|
-
url: `${GATEWAY_ORIGIN}/${cluster}/jobs`,
|
|
2887
|
-
headers,
|
|
2888
|
-
timeoutMs: Math.min(ctx.httpTimeoutMs, 12e3),
|
|
2889
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
2890
|
-
});
|
|
2891
|
-
if (jobs.ok && jobs.data) live = runningModels(jobs.data);
|
|
2892
|
-
} catch {
|
|
2893
|
-
}
|
|
2894
|
-
const models = dedupe([...fromCatalog, ...live]);
|
|
2895
|
-
const result = { ok: true, models };
|
|
2896
|
-
if (catalog.latencyMs !== void 0) result.latencyMs = catalog.latencyMs;
|
|
2897
|
-
if (models.length === 0) {
|
|
2898
|
-
result.notes = [`ALCF ${cluster}/${framework} reported no models; the gateway may have no running jobs.`];
|
|
2899
|
-
}
|
|
2900
|
-
return result;
|
|
2901
|
-
}
|
|
2902
|
-
var alcfRuntime = {
|
|
2903
|
-
id: "alcf",
|
|
2904
|
-
displayName: "ALCF Inference (Globus)",
|
|
2905
|
-
kind: "http",
|
|
2906
|
-
tier: "cloud",
|
|
2907
|
-
apiFamily: "openai-completions",
|
|
2908
|
-
auth: "oauth",
|
|
2909
|
-
knownModels: KNOWN_MODELS,
|
|
2910
|
-
defaultCapabilities: defaultCapabilities2,
|
|
2911
|
-
probe(target, ctx) {
|
|
2912
|
-
return discover(target, ctx);
|
|
2913
|
-
},
|
|
2914
|
-
async probeModels(target, ctx) {
|
|
2915
|
-
const result = await discover(target, ctx);
|
|
2916
|
-
return result.models ?? [];
|
|
2917
|
-
},
|
|
2918
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2919
|
-
const model = synthesizeOpenAICompatModel({
|
|
2920
|
-
target,
|
|
2921
|
-
wireModelId,
|
|
2922
|
-
kb,
|
|
2923
|
-
defaultCapabilities: defaultCapabilities2,
|
|
2924
|
-
apiFamily: "openai-completions",
|
|
2925
|
-
provider: "alcf",
|
|
2926
|
-
baseUrlForTarget: withAsIs
|
|
2927
|
-
});
|
|
2928
|
-
const meta = model.clio;
|
|
2929
|
-
if (meta) meta.chatTemplateKwargsUnsupported = true;
|
|
2930
|
-
return model;
|
|
2931
|
-
}
|
|
2932
|
-
};
|
|
2933
|
-
var alcf_default = alcfRuntime;
|
|
2934
|
-
|
|
2935
|
-
// src/domains/providers/runtimes/cloud/anthropic.ts
|
|
2936
|
-
init_esm_shims();
|
|
2937
|
-
|
|
2938
|
-
// src/domains/providers/runtimes/protocol/anthropic-messages.ts
|
|
2939
|
-
init_esm_shims();
|
|
2940
|
-
function synthesizeAnthropicMessagesModel(input) {
|
|
2941
|
-
return synthesizeCatalogBackedModel({
|
|
2942
|
-
target: input.target,
|
|
2943
|
-
wireModelId: input.wireModelId,
|
|
2944
|
-
kb: input.kb,
|
|
2945
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
2946
|
-
runtimeId: "anthropic",
|
|
2947
|
-
api: "anthropic-messages",
|
|
2948
|
-
provider: input.provider,
|
|
2949
|
-
defaultBaseUrl: input.defaultBaseUrl
|
|
2950
|
-
});
|
|
2951
|
-
}
|
|
2952
|
-
|
|
2953
|
-
// src/domains/providers/runtimes/cloud/anthropic.ts
|
|
2954
|
-
var defaultCapabilities3 = {
|
|
2955
|
-
chat: true,
|
|
2956
|
-
tools: true,
|
|
2957
|
-
toolCallFormat: "anthropic",
|
|
2958
|
-
reasoning: true,
|
|
2959
|
-
thinkingFormat: "anthropic-extended",
|
|
2960
|
-
vision: true,
|
|
2961
|
-
audio: false,
|
|
2962
|
-
embeddings: false,
|
|
2963
|
-
rerank: false,
|
|
2964
|
-
fim: false,
|
|
2965
|
-
contextWindow: 2e5,
|
|
2966
|
-
maxTokens: 8192
|
|
2967
|
-
};
|
|
2968
|
-
var anthropicRuntime = {
|
|
2969
|
-
id: "anthropic",
|
|
2970
|
-
displayName: "Anthropic",
|
|
2971
|
-
kind: "http",
|
|
2972
|
-
tier: "cloud",
|
|
2973
|
-
apiFamily: "anthropic-messages",
|
|
2974
|
-
auth: "api-key",
|
|
2975
|
-
credentialsEnvVar: "ANTHROPIC_API_KEY",
|
|
2976
|
-
defaultCapabilities: defaultCapabilities3,
|
|
2977
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
2978
|
-
return synthesizeAnthropicMessagesModel({
|
|
2979
|
-
target,
|
|
2980
|
-
wireModelId,
|
|
2981
|
-
kb,
|
|
2982
|
-
defaultCapabilities: defaultCapabilities3,
|
|
2983
|
-
provider: "anthropic",
|
|
2984
|
-
defaultBaseUrl: "https://api.anthropic.com"
|
|
2985
|
-
});
|
|
2986
|
-
}
|
|
2987
|
-
};
|
|
2988
|
-
var anthropic_default = anthropicRuntime;
|
|
2989
|
-
|
|
2990
|
-
// src/domains/providers/runtimes/cloud/anthropic-max.ts
|
|
2991
|
-
init_esm_shims();
|
|
2992
|
-
var defaultCapabilities4 = {
|
|
2993
|
-
chat: true,
|
|
2994
|
-
tools: true,
|
|
2995
|
-
toolCallFormat: "anthropic",
|
|
2996
|
-
reasoning: true,
|
|
2997
|
-
thinkingFormat: "anthropic-extended",
|
|
2998
|
-
vision: true,
|
|
2999
|
-
audio: false,
|
|
3000
|
-
embeddings: false,
|
|
3001
|
-
rerank: false,
|
|
3002
|
-
fim: false,
|
|
3003
|
-
contextWindow: 2e5,
|
|
3004
|
-
maxTokens: 8192
|
|
3005
|
-
};
|
|
3006
|
-
var anthropicMaxRuntime = {
|
|
3007
|
-
id: "anthropic-max",
|
|
3008
|
-
displayName: "Anthropic (Claude Pro/Max)",
|
|
3009
|
-
kind: "http",
|
|
3010
|
-
tier: "cloud",
|
|
3011
|
-
apiFamily: "anthropic-messages",
|
|
3012
|
-
auth: "oauth",
|
|
3013
|
-
oauthProviderId: "anthropic",
|
|
3014
|
-
authNotice: "Connects with your Claude Pro/Max subscription via OAuth (the same path Claude Code uses). Using subscription credentials outside Anthropic's first-party apps may not align with their terms of service; enable at your own discretion.",
|
|
3015
|
-
defaultCapabilities: defaultCapabilities4,
|
|
3016
|
-
async probeModels(_target, _ctx) {
|
|
3017
|
-
return listCatalogModelsForRuntime("anthropic-max").map((model) => model.id);
|
|
3018
|
-
},
|
|
3019
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3020
|
-
return synthesizeCatalogBackedModel({
|
|
3021
|
-
target,
|
|
3022
|
-
wireModelId,
|
|
3023
|
-
kb,
|
|
3024
|
-
defaultCapabilities: defaultCapabilities4,
|
|
3025
|
-
runtimeId: "anthropic-max",
|
|
3026
|
-
api: "anthropic-messages",
|
|
3027
|
-
provider: "anthropic",
|
|
3028
|
-
defaultBaseUrl: "https://api.anthropic.com"
|
|
3029
|
-
});
|
|
3030
|
-
}
|
|
3031
|
-
};
|
|
3032
|
-
var anthropic_max_default = anthropicMaxRuntime;
|
|
3033
|
-
|
|
3034
|
-
// src/domains/providers/runtimes/cloud/bedrock.ts
|
|
3035
|
-
init_esm_shims();
|
|
3036
|
-
|
|
3037
|
-
// src/domains/providers/runtimes/protocol/bedrock.ts
|
|
3038
|
-
init_esm_shims();
|
|
3039
|
-
function synthesizeBedrockModel(input) {
|
|
3040
|
-
return synthesizeCatalogBackedModel({
|
|
3041
|
-
target: input.target,
|
|
3042
|
-
wireModelId: input.wireModelId,
|
|
3043
|
-
kb: input.kb,
|
|
3044
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3045
|
-
runtimeId: "bedrock",
|
|
3046
|
-
api: "bedrock-converse-stream",
|
|
3047
|
-
provider: "amazon-bedrock",
|
|
3048
|
-
defaultBaseUrl: ""
|
|
3049
|
-
});
|
|
3050
|
-
}
|
|
3051
|
-
|
|
3052
|
-
// src/domains/providers/runtimes/cloud/bedrock.ts
|
|
3053
|
-
var defaultCapabilities5 = {
|
|
3054
|
-
chat: true,
|
|
3055
|
-
tools: true,
|
|
3056
|
-
toolCallFormat: "anthropic",
|
|
3057
|
-
reasoning: true,
|
|
3058
|
-
thinkingFormat: "anthropic-extended",
|
|
3059
|
-
vision: false,
|
|
3060
|
-
audio: false,
|
|
3061
|
-
embeddings: false,
|
|
3062
|
-
rerank: false,
|
|
3063
|
-
fim: false,
|
|
3064
|
-
contextWindow: 2e5,
|
|
3065
|
-
maxTokens: 8192
|
|
3066
|
-
};
|
|
3067
|
-
var bedrockRuntime = {
|
|
3068
|
-
id: "bedrock",
|
|
3069
|
-
displayName: "Amazon Bedrock",
|
|
3070
|
-
kind: "http",
|
|
3071
|
-
tier: "cloud",
|
|
3072
|
-
apiFamily: "bedrock-converse-stream",
|
|
3073
|
-
auth: "aws-sdk",
|
|
3074
|
-
defaultCapabilities: defaultCapabilities5,
|
|
3075
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3076
|
-
return synthesizeBedrockModel({ target, wireModelId, kb, defaultCapabilities: defaultCapabilities5 });
|
|
3077
|
-
}
|
|
3078
|
-
};
|
|
3079
|
-
var bedrock_default = bedrockRuntime;
|
|
3080
|
-
|
|
3081
|
-
// src/domains/providers/runtimes/cloud/deepseek.ts
|
|
3082
|
-
init_esm_shims();
|
|
3083
|
-
var defaultCapabilities6 = {
|
|
3084
|
-
chat: true,
|
|
3085
|
-
tools: true,
|
|
3086
|
-
toolCallFormat: "openai",
|
|
3087
|
-
reasoning: true,
|
|
3088
|
-
thinkingFormat: "deepseek-r1",
|
|
3089
|
-
vision: false,
|
|
3090
|
-
audio: false,
|
|
3091
|
-
embeddings: false,
|
|
3092
|
-
rerank: false,
|
|
3093
|
-
fim: false,
|
|
3094
|
-
contextWindow: 128e3,
|
|
3095
|
-
maxTokens: 65536
|
|
3096
|
-
};
|
|
3097
|
-
var deepseekRuntime = {
|
|
3098
|
-
id: "deepseek",
|
|
3099
|
-
displayName: "DeepSeek",
|
|
3100
|
-
kind: "http",
|
|
3101
|
-
tier: "cloud",
|
|
3102
|
-
apiFamily: "openai-completions",
|
|
3103
|
-
auth: "api-key",
|
|
3104
|
-
credentialsEnvVar: "DEEPSEEK_API_KEY",
|
|
3105
|
-
defaultCapabilities: defaultCapabilities6,
|
|
3106
|
-
async probeModels(_target, _ctx) {
|
|
3107
|
-
return listCatalogModelsForRuntime("deepseek").map((model) => model.id);
|
|
3108
|
-
},
|
|
3109
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3110
|
-
return synthesizeCatalogBackedModel({
|
|
3111
|
-
target,
|
|
3112
|
-
wireModelId,
|
|
3113
|
-
kb,
|
|
3114
|
-
defaultCapabilities: defaultCapabilities6,
|
|
3115
|
-
runtimeId: "deepseek",
|
|
3116
|
-
api: "openai-completions",
|
|
3117
|
-
provider: "deepseek",
|
|
3118
|
-
defaultBaseUrl: "https://api.deepseek.com"
|
|
3119
|
-
});
|
|
3120
|
-
}
|
|
3121
|
-
};
|
|
3122
|
-
var deepseek_default = deepseekRuntime;
|
|
3123
|
-
|
|
3124
|
-
// src/domains/providers/runtimes/cloud/google.ts
|
|
3125
|
-
init_esm_shims();
|
|
3126
|
-
|
|
3127
|
-
// src/domains/providers/runtimes/protocol/google.ts
|
|
3128
|
-
init_esm_shims();
|
|
3129
|
-
function synthesizeGoogleModel(input) {
|
|
3130
|
-
return synthesizeCatalogBackedModel({
|
|
3131
|
-
target: input.target,
|
|
3132
|
-
wireModelId: input.wireModelId,
|
|
3133
|
-
kb: input.kb,
|
|
3134
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3135
|
-
runtimeId: "google",
|
|
3136
|
-
api: "google-generative-ai",
|
|
3137
|
-
provider: "google",
|
|
3138
|
-
defaultBaseUrl: input.defaultBaseUrl
|
|
3139
|
-
});
|
|
3140
|
-
}
|
|
3141
|
-
|
|
3142
|
-
// src/domains/providers/runtimes/cloud/google.ts
|
|
3143
|
-
var defaultCapabilities7 = {
|
|
3144
|
-
chat: true,
|
|
3145
|
-
tools: true,
|
|
3146
|
-
toolCallFormat: "openai",
|
|
3147
|
-
reasoning: true,
|
|
3148
|
-
vision: true,
|
|
3149
|
-
audio: false,
|
|
3150
|
-
embeddings: false,
|
|
3151
|
-
rerank: false,
|
|
3152
|
-
fim: false,
|
|
3153
|
-
contextWindow: 2e6,
|
|
3154
|
-
maxTokens: 8192
|
|
3155
|
-
};
|
|
3156
|
-
var googleRuntime = {
|
|
3157
|
-
id: "google",
|
|
3158
|
-
displayName: "Google Generative AI",
|
|
3159
|
-
kind: "http",
|
|
3160
|
-
tier: "cloud",
|
|
3161
|
-
apiFamily: "google-generative-ai",
|
|
3162
|
-
auth: "api-key",
|
|
3163
|
-
credentialsEnvVar: "GOOGLE_API_KEY",
|
|
3164
|
-
defaultCapabilities: defaultCapabilities7,
|
|
3165
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3166
|
-
return synthesizeGoogleModel({
|
|
3167
|
-
target,
|
|
3168
|
-
wireModelId,
|
|
3169
|
-
kb,
|
|
3170
|
-
defaultCapabilities: defaultCapabilities7,
|
|
3171
|
-
defaultBaseUrl: "https://generativelanguage.googleapis.com/v1beta"
|
|
3172
|
-
});
|
|
3173
|
-
}
|
|
3174
|
-
};
|
|
3175
|
-
var google_default = googleRuntime;
|
|
3176
|
-
|
|
3177
|
-
// src/domains/providers/runtimes/cloud/groq.ts
|
|
3178
|
-
init_esm_shims();
|
|
3179
|
-
var defaultCapabilities8 = {
|
|
3180
|
-
chat: true,
|
|
3181
|
-
tools: true,
|
|
3182
|
-
toolCallFormat: "openai",
|
|
3183
|
-
reasoning: false,
|
|
3184
|
-
vision: false,
|
|
3185
|
-
audio: false,
|
|
3186
|
-
embeddings: false,
|
|
3187
|
-
rerank: false,
|
|
3188
|
-
fim: false,
|
|
3189
|
-
contextWindow: 128e3,
|
|
3190
|
-
maxTokens: 8192
|
|
3191
|
-
};
|
|
3192
|
-
var groqRuntime = {
|
|
3193
|
-
id: "groq",
|
|
3194
|
-
displayName: "Groq",
|
|
3195
|
-
kind: "http",
|
|
3196
|
-
tier: "cloud",
|
|
3197
|
-
apiFamily: "openai-completions",
|
|
3198
|
-
auth: "api-key",
|
|
3199
|
-
credentialsEnvVar: "GROQ_API_KEY",
|
|
3200
|
-
defaultCapabilities: defaultCapabilities8,
|
|
3201
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3202
|
-
return synthesizeCatalogBackedModel({
|
|
3203
|
-
target,
|
|
3204
|
-
wireModelId,
|
|
3205
|
-
kb,
|
|
3206
|
-
defaultCapabilities: defaultCapabilities8,
|
|
3207
|
-
runtimeId: "groq",
|
|
3208
|
-
api: "openai-completions",
|
|
3209
|
-
provider: "groq",
|
|
3210
|
-
defaultBaseUrl: "https://api.groq.com/openai/v1"
|
|
3211
|
-
});
|
|
3212
|
-
}
|
|
3213
|
-
};
|
|
3214
|
-
var groq_default = groqRuntime;
|
|
3215
|
-
|
|
3216
|
-
// src/domains/providers/runtimes/cloud/mistral.ts
|
|
3217
|
-
init_esm_shims();
|
|
3218
|
-
var defaultCapabilities9 = {
|
|
3219
|
-
chat: true,
|
|
3220
|
-
tools: true,
|
|
3221
|
-
toolCallFormat: "mistral",
|
|
3222
|
-
reasoning: false,
|
|
3223
|
-
vision: false,
|
|
3224
|
-
audio: false,
|
|
3225
|
-
embeddings: false,
|
|
3226
|
-
rerank: false,
|
|
3227
|
-
fim: false,
|
|
3228
|
-
contextWindow: 128e3,
|
|
3229
|
-
maxTokens: 8192
|
|
3230
|
-
};
|
|
3231
|
-
var mistralRuntime = {
|
|
3232
|
-
id: "mistral",
|
|
3233
|
-
displayName: "Mistral AI",
|
|
3234
|
-
kind: "http",
|
|
3235
|
-
tier: "cloud",
|
|
3236
|
-
apiFamily: "mistral-conversations",
|
|
3237
|
-
auth: "api-key",
|
|
3238
|
-
credentialsEnvVar: "MISTRAL_API_KEY",
|
|
3239
|
-
defaultCapabilities: defaultCapabilities9,
|
|
3240
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3241
|
-
return synthesizeCatalogBackedModel({
|
|
3242
|
-
target,
|
|
3243
|
-
wireModelId,
|
|
3244
|
-
kb,
|
|
3245
|
-
defaultCapabilities: defaultCapabilities9,
|
|
3246
|
-
runtimeId: "mistral",
|
|
3247
|
-
api: "mistral-conversations",
|
|
3248
|
-
provider: "mistral",
|
|
3249
|
-
defaultBaseUrl: "https://api.mistral.ai"
|
|
3250
|
-
});
|
|
3251
|
-
}
|
|
3252
|
-
};
|
|
3253
|
-
var mistral_default = mistralRuntime;
|
|
3254
|
-
|
|
3255
|
-
// src/domains/providers/runtimes/cloud/openai.ts
|
|
3256
|
-
init_esm_shims();
|
|
3257
|
-
var defaultCapabilities10 = {
|
|
3258
|
-
chat: true,
|
|
3259
|
-
tools: true,
|
|
3260
|
-
toolCallFormat: "openai",
|
|
3261
|
-
reasoning: true,
|
|
3262
|
-
vision: true,
|
|
3263
|
-
audio: false,
|
|
3264
|
-
embeddings: false,
|
|
3265
|
-
rerank: false,
|
|
3266
|
-
fim: false,
|
|
3267
|
-
contextWindow: 272e3,
|
|
3268
|
-
maxTokens: 16384
|
|
3269
|
-
};
|
|
3270
|
-
var openaiRuntime = {
|
|
3271
|
-
id: "openai",
|
|
3272
|
-
displayName: "OpenAI",
|
|
3273
|
-
kind: "http",
|
|
3274
|
-
tier: "cloud",
|
|
3275
|
-
apiFamily: "openai-responses",
|
|
3276
|
-
auth: "api-key",
|
|
3277
|
-
credentialsEnvVar: "OPENAI_API_KEY",
|
|
3278
|
-
defaultCapabilities: defaultCapabilities10,
|
|
3279
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3280
|
-
return synthesizeCatalogBackedModel({
|
|
3281
|
-
target,
|
|
3282
|
-
wireModelId,
|
|
3283
|
-
kb,
|
|
3284
|
-
defaultCapabilities: defaultCapabilities10,
|
|
3285
|
-
runtimeId: "openai",
|
|
3286
|
-
api: "openai-responses",
|
|
3287
|
-
provider: "openai",
|
|
3288
|
-
defaultBaseUrl: "https://api.openai.com/v1"
|
|
3289
|
-
});
|
|
3290
|
-
}
|
|
3291
|
-
};
|
|
3292
|
-
var openai_default = openaiRuntime;
|
|
3293
|
-
|
|
3294
|
-
// src/domains/providers/runtimes/cloud/openai-codex.ts
|
|
3295
|
-
init_esm_shims();
|
|
3296
|
-
var defaultCapabilities11 = {
|
|
3297
|
-
chat: true,
|
|
3298
|
-
tools: true,
|
|
3299
|
-
toolCallFormat: "openai",
|
|
3300
|
-
reasoning: true,
|
|
3301
|
-
thinkingFormat: "openai-codex",
|
|
3302
|
-
vision: true,
|
|
3303
|
-
audio: false,
|
|
3304
|
-
embeddings: false,
|
|
3305
|
-
rerank: false,
|
|
3306
|
-
fim: false,
|
|
3307
|
-
contextWindow: 272e3,
|
|
3308
|
-
maxTokens: 16384
|
|
3309
|
-
};
|
|
3310
|
-
var openaiCodexRuntime = {
|
|
3311
|
-
id: "openai-codex",
|
|
3312
|
-
displayName: "OpenAI Codex",
|
|
3313
|
-
kind: "http",
|
|
3314
|
-
tier: "cloud",
|
|
3315
|
-
apiFamily: "openai-codex-responses",
|
|
3316
|
-
auth: "oauth",
|
|
3317
|
-
defaultCapabilities: defaultCapabilities11,
|
|
3318
|
-
async probeModels(_target, _ctx) {
|
|
3319
|
-
return listCatalogModelsForRuntime("openai-codex").map((model) => model.id);
|
|
3320
|
-
},
|
|
3321
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3322
|
-
return synthesizeCatalogBackedModel({
|
|
3323
|
-
target,
|
|
3324
|
-
wireModelId,
|
|
3325
|
-
kb,
|
|
3326
|
-
defaultCapabilities: defaultCapabilities11,
|
|
3327
|
-
runtimeId: "openai-codex",
|
|
3328
|
-
api: "openai-codex-responses",
|
|
3329
|
-
provider: "openai-codex",
|
|
3330
|
-
defaultBaseUrl: "https://chatgpt.com/backend-api"
|
|
3331
|
-
});
|
|
3332
|
-
}
|
|
3333
|
-
};
|
|
3334
|
-
var openai_codex_default = openaiCodexRuntime;
|
|
3335
|
-
|
|
3336
|
-
// src/domains/providers/runtimes/cloud/openrouter.ts
|
|
3337
|
-
init_esm_shims();
|
|
3338
|
-
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
3339
|
-
var OPENROUTER_HEADERS = {
|
|
3340
|
-
"HTTP-Referer": "https://github.com/iowarp/clio-coder",
|
|
3341
|
-
"X-OpenRouter-Title": "Clio Coder"
|
|
3342
|
-
};
|
|
3343
|
-
var defaultCapabilities12 = {
|
|
3344
|
-
chat: true,
|
|
3345
|
-
tools: true,
|
|
3346
|
-
toolCallFormat: "openai",
|
|
3347
|
-
reasoning: false,
|
|
3348
|
-
thinkingFormat: "openrouter",
|
|
3349
|
-
vision: false,
|
|
3350
|
-
audio: false,
|
|
3351
|
-
embeddings: false,
|
|
3352
|
-
rerank: false,
|
|
3353
|
-
fim: false,
|
|
3354
|
-
contextWindow: 128e3,
|
|
3355
|
-
maxTokens: 8192
|
|
3356
|
-
};
|
|
3357
|
-
function trimTrailingSlash2(value) {
|
|
3358
|
-
return value.endsWith("/") && value.length > 1 ? value.slice(0, -1) : value;
|
|
3359
|
-
}
|
|
3360
|
-
function targetBaseUrl2(target) {
|
|
3361
|
-
return trimTrailingSlash2(target.url ?? OPENROUTER_BASE_URL);
|
|
3362
|
-
}
|
|
3363
|
-
function modelsUrl(target) {
|
|
3364
|
-
return `${targetBaseUrl2(target)}/models`;
|
|
3365
|
-
}
|
|
3366
|
-
function probeHeaders(target, ctx) {
|
|
3367
|
-
const headers = { ...OPENROUTER_HEADERS, ...target.auth?.headers ?? {} };
|
|
3368
|
-
const envName = target.auth?.apiKeyEnvVar ?? "OPENROUTER_API_KEY";
|
|
3369
|
-
if (ctx.credentialsPresent.has(envName)) {
|
|
3370
|
-
const key = process.env[envName]?.trim();
|
|
3371
|
-
if (key) headers.authorization = `Bearer ${key}`;
|
|
3372
|
-
}
|
|
3373
|
-
return headers;
|
|
3374
|
-
}
|
|
3375
|
-
async function fetchModels(target, ctx) {
|
|
3376
|
-
const opts = {
|
|
3377
|
-
url: modelsUrl(target),
|
|
3378
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
3379
|
-
headers: probeHeaders(target, ctx)
|
|
3380
|
-
};
|
|
3381
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
3382
|
-
if (!result.ok) return result;
|
|
3383
|
-
const models = (result.data?.data ?? []).map((row) => typeof row?.id === "string" ? row.id : null).filter((id) => id !== null);
|
|
3384
|
-
const out = { ok: true, models };
|
|
3385
|
-
if (result.latencyMs !== void 0) out.latencyMs = result.latencyMs;
|
|
3386
|
-
const configured = target.defaultModel?.trim();
|
|
3387
|
-
if (configured && models.length > 0 && !models.includes(configured)) {
|
|
3388
|
-
out.ok = false;
|
|
3389
|
-
out.error = `configured model '${configured}' was not returned by OpenRouter`;
|
|
3390
|
-
}
|
|
3391
|
-
return out;
|
|
3392
|
-
}
|
|
3393
|
-
var openrouterRuntime = {
|
|
3394
|
-
id: "openrouter",
|
|
3395
|
-
displayName: "OpenRouter",
|
|
3396
|
-
kind: "http",
|
|
3397
|
-
tier: "cloud",
|
|
3398
|
-
apiFamily: "openai-completions",
|
|
3399
|
-
auth: "api-key",
|
|
3400
|
-
credentialsEnvVar: "OPENROUTER_API_KEY",
|
|
3401
|
-
defaultCapabilities: defaultCapabilities12,
|
|
3402
|
-
probe(target, ctx) {
|
|
3403
|
-
return fetchModels(target, ctx);
|
|
3404
|
-
},
|
|
3405
|
-
async probeModels(target, ctx) {
|
|
3406
|
-
const result = await fetchModels(target, ctx);
|
|
3407
|
-
return result.ok && result.models ? [...result.models] : [];
|
|
3408
|
-
},
|
|
3409
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3410
|
-
return synthesizeCatalogBackedModel({
|
|
3411
|
-
target,
|
|
3412
|
-
wireModelId,
|
|
3413
|
-
kb,
|
|
3414
|
-
defaultCapabilities: defaultCapabilities12,
|
|
3415
|
-
runtimeId: "openrouter",
|
|
3416
|
-
api: "openai-completions",
|
|
3417
|
-
provider: "openrouter",
|
|
3418
|
-
defaultBaseUrl: OPENROUTER_BASE_URL,
|
|
3419
|
-
defaultHeaders: OPENROUTER_HEADERS
|
|
3420
|
-
});
|
|
3421
|
-
}
|
|
3422
|
-
};
|
|
3423
|
-
var openrouter_default = openrouterRuntime;
|
|
3424
|
-
|
|
3425
|
-
// src/domains/providers/runtimes/local-native/lemonade-anthropic.ts
|
|
3426
|
-
init_esm_shims();
|
|
3427
|
-
|
|
3428
|
-
// src/domains/providers/runtimes/protocol/anthropic-compat.ts
|
|
3429
|
-
init_esm_shims();
|
|
3430
|
-
function synthesizeAnthropicCompatModel(input) {
|
|
3431
|
-
return synthLocalModel({
|
|
3432
|
-
target: input.target,
|
|
3433
|
-
wireModelId: input.wireModelId,
|
|
3434
|
-
kb: input.kb,
|
|
3435
|
-
defaultCapabilities: input.defaultCapabilities,
|
|
3436
|
-
apiFamily: "anthropic-messages",
|
|
3437
|
-
provider: input.provider,
|
|
3438
|
-
baseUrlForTarget: input.baseUrlForTarget ?? withAsIs
|
|
3439
|
-
});
|
|
3440
|
-
}
|
|
3441
|
-
function makeAnthropicCompatRuntime(spec) {
|
|
3442
|
-
const messagesPath = spec.messagesPath ?? "/v1/messages";
|
|
3443
|
-
const modelsPath = spec.modelsPath ?? "/v1/models";
|
|
3444
|
-
return {
|
|
3445
|
-
id: spec.id,
|
|
3446
|
-
displayName: spec.displayName,
|
|
3447
|
-
kind: "http",
|
|
3448
|
-
tier: spec.tier,
|
|
3449
|
-
apiFamily: "anthropic-messages",
|
|
3450
|
-
auth: spec.auth,
|
|
3451
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
3452
|
-
...spec.hidden === true ? { hidden: true } : {},
|
|
3453
|
-
async probe(target, ctx) {
|
|
3454
|
-
const base = targetBaseUrl(target);
|
|
3455
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3456
|
-
return probeUrl(`${base}${messagesPath}`, ctx, "HEAD");
|
|
3457
|
-
},
|
|
3458
|
-
async probeModels(target, ctx) {
|
|
3459
|
-
const base = targetBaseUrl(target);
|
|
3460
|
-
if (!base) return [];
|
|
3461
|
-
return probeOpenAIModels(base, ctx, modelsPath);
|
|
3462
|
-
},
|
|
3463
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3464
|
-
return synthesizeAnthropicCompatModel({
|
|
3465
|
-
target,
|
|
3466
|
-
wireModelId,
|
|
3467
|
-
kb,
|
|
3468
|
-
defaultCapabilities: spec.defaultCapabilities,
|
|
3469
|
-
provider: spec.provider
|
|
3470
|
-
});
|
|
3471
|
-
}
|
|
3472
|
-
};
|
|
3473
|
-
}
|
|
3474
|
-
var defaultCapabilities13 = {
|
|
3475
|
-
chat: true,
|
|
3476
|
-
tools: true,
|
|
3477
|
-
toolCallFormat: "anthropic",
|
|
3478
|
-
reasoning: false,
|
|
3479
|
-
vision: false,
|
|
3480
|
-
audio: false,
|
|
3481
|
-
embeddings: false,
|
|
3482
|
-
rerank: false,
|
|
3483
|
-
fim: false,
|
|
3484
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3485
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3486
|
-
};
|
|
3487
|
-
var anthropic_compat_default = makeAnthropicCompatRuntime({
|
|
3488
|
-
id: "anthropic-compat",
|
|
3489
|
-
displayName: "Generic Anthropic-compatible",
|
|
3490
|
-
provider: "anthropic-compat",
|
|
3491
|
-
auth: "api-key",
|
|
3492
|
-
tier: "protocol",
|
|
3493
|
-
defaultCapabilities: defaultCapabilities13
|
|
3494
|
-
});
|
|
3495
|
-
|
|
3496
|
-
// src/domains/providers/runtimes/local-native/lemonade-anthropic.ts
|
|
3497
|
-
var defaultCapabilities14 = {
|
|
3498
|
-
chat: true,
|
|
3499
|
-
tools: true,
|
|
3500
|
-
toolCallFormat: "anthropic",
|
|
3501
|
-
reasoning: false,
|
|
3502
|
-
vision: false,
|
|
3503
|
-
audio: false,
|
|
3504
|
-
embeddings: false,
|
|
3505
|
-
rerank: false,
|
|
3506
|
-
fim: false,
|
|
3507
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3508
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3509
|
-
};
|
|
3510
|
-
var lemonade_anthropic_default = makeAnthropicCompatRuntime({
|
|
3511
|
-
id: "lemonade-anthropic",
|
|
3512
|
-
displayName: "Lemonade (Anthropic-compat)",
|
|
3513
|
-
provider: "lemonade",
|
|
3514
|
-
auth: "api-key",
|
|
3515
|
-
tier: "local-native",
|
|
3516
|
-
defaultCapabilities: defaultCapabilities14,
|
|
3517
|
-
hidden: true
|
|
3518
|
-
});
|
|
3519
|
-
|
|
3520
|
-
// src/domains/providers/runtimes/local-native/lemonade-openai.ts
|
|
3521
|
-
init_esm_shims();
|
|
3522
|
-
var defaultCapabilities15 = {
|
|
3523
|
-
chat: true,
|
|
3524
|
-
tools: true,
|
|
3525
|
-
toolCallFormat: "openai",
|
|
3526
|
-
reasoning: false,
|
|
3527
|
-
vision: false,
|
|
3528
|
-
audio: false,
|
|
3529
|
-
embeddings: false,
|
|
3530
|
-
rerank: false,
|
|
3531
|
-
fim: false,
|
|
3532
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3533
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3534
|
-
};
|
|
3535
|
-
var lemonade_openai_default = makeOpenAICompatRuntime({
|
|
3536
|
-
id: "lemonade",
|
|
3537
|
-
displayName: "Lemonade (OpenAI-compat)",
|
|
3538
|
-
provider: "lemonade",
|
|
3539
|
-
auth: "api-key",
|
|
3540
|
-
tier: "local-native",
|
|
3541
|
-
defaultCapabilities: defaultCapabilities15
|
|
3542
|
-
});
|
|
3543
|
-
|
|
3544
|
-
// src/domains/providers/runtimes/local-native/llamacpp.ts
|
|
3545
|
-
init_esm_shims();
|
|
3546
|
-
var defaultCapabilities16 = {
|
|
3547
|
-
chat: true,
|
|
3548
|
-
tools: true,
|
|
3549
|
-
toolCallFormat: "openai",
|
|
3550
|
-
structuredOutputs: "json-schema",
|
|
3551
|
-
reasoning: false,
|
|
3552
|
-
vision: false,
|
|
3553
|
-
audio: false,
|
|
3554
|
-
embeddings: false,
|
|
3555
|
-
rerank: false,
|
|
3556
|
-
fim: false,
|
|
3557
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3558
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3559
|
-
};
|
|
3560
|
-
function targetUrl(target) {
|
|
3561
|
-
return targetRootUrl(target);
|
|
3562
|
-
}
|
|
3563
|
-
var llamacppRuntime = {
|
|
3564
|
-
id: "llamacpp",
|
|
3565
|
-
displayName: "llama.cpp",
|
|
3566
|
-
kind: "http",
|
|
3567
|
-
tier: "local-native",
|
|
3568
|
-
apiFamily: "openai-completions",
|
|
3569
|
-
auth: "api-key",
|
|
3570
|
-
defaultCapabilities: defaultCapabilities16,
|
|
3571
|
-
async probe(target, ctx) {
|
|
3572
|
-
const base = targetUrl(target);
|
|
3573
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3574
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3575
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3576
|
-
if (!health.ok) return health;
|
|
3577
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3578
|
-
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
3579
|
-
const catalog = await probeOpenAIModelCatalog(base, ctx);
|
|
3580
|
-
const result = { ok: true };
|
|
3581
|
-
if (catalog.models.length > 0) result.models = catalog.models;
|
|
3582
|
-
if (Object.keys(catalog.modelCapabilities).length > 0) result.modelCapabilities = catalog.modelCapabilities;
|
|
3583
|
-
if (Object.keys(catalog.modelStates).length > 0) result.modelStates = catalog.modelStates;
|
|
3584
|
-
if (typeof health.latencyMs === "number") result.latencyMs = health.latencyMs;
|
|
3585
|
-
const discoveredCapabilities = {
|
|
3586
|
-
...props.discoveredCapabilities ?? {},
|
|
3587
|
-
...status.discoveredCapabilities ?? {}
|
|
3588
|
-
};
|
|
3589
|
-
if (Object.keys(discoveredCapabilities).length > 0) {
|
|
3590
|
-
result.discoveredCapabilities = discoveredCapabilities;
|
|
3591
|
-
if (status.modelId) result.capabilityModelId = status.modelId;
|
|
3592
|
-
}
|
|
3593
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
3594
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3595
|
-
const notes = [...status.notes ?? [], ...note ? [note] : []];
|
|
3596
|
-
if (notes.length > 0) result.notes = notes;
|
|
3597
|
-
return result;
|
|
3598
|
-
},
|
|
3599
|
-
async probeModels(target, ctx) {
|
|
3600
|
-
const base = targetUrl(target);
|
|
3601
|
-
if (!base) return [];
|
|
3602
|
-
return probeOpenAIModels(base, ctx);
|
|
3603
|
-
},
|
|
3604
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3605
|
-
return synthLocalModel({
|
|
3606
|
-
target,
|
|
3607
|
-
wireModelId,
|
|
3608
|
-
kb,
|
|
3609
|
-
defaultCapabilities: defaultCapabilities16,
|
|
3610
|
-
apiFamily: "openai-completions",
|
|
3611
|
-
provider: "llamacpp",
|
|
3612
|
-
baseUrlForTarget: withV1
|
|
3613
|
-
});
|
|
3614
|
-
}
|
|
3615
|
-
};
|
|
3616
|
-
var llamacpp_default = llamacppRuntime;
|
|
3617
|
-
|
|
3618
|
-
// src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts
|
|
3619
|
-
init_esm_shims();
|
|
3620
|
-
var defaultCapabilities17 = {
|
|
3621
|
-
chat: true,
|
|
3622
|
-
tools: true,
|
|
3623
|
-
toolCallFormat: "anthropic",
|
|
3624
|
-
reasoning: false,
|
|
3625
|
-
vision: false,
|
|
3626
|
-
audio: false,
|
|
3627
|
-
embeddings: false,
|
|
3628
|
-
rerank: false,
|
|
3629
|
-
fim: true,
|
|
3630
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
3631
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
3632
|
-
};
|
|
3633
|
-
function url(target) {
|
|
3634
|
-
return target.url ? stripTrailingSlash(target.url) : null;
|
|
3635
|
-
}
|
|
3636
|
-
var llamacppAnthropicRuntime = {
|
|
3637
|
-
id: "llamacpp-anthropic",
|
|
3638
|
-
displayName: "llama.cpp (Anthropic-compat)",
|
|
3639
|
-
kind: "http",
|
|
3640
|
-
tier: "local-native",
|
|
3641
|
-
apiFamily: "anthropic-messages",
|
|
3642
|
-
auth: "api-key",
|
|
3643
|
-
defaultCapabilities: defaultCapabilities17,
|
|
3644
|
-
hidden: true,
|
|
3645
|
-
async probe(target, ctx) {
|
|
3646
|
-
const base = url(target);
|
|
3647
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3648
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3649
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3650
|
-
if (!health.ok) return health;
|
|
3651
|
-
const headOpts = {
|
|
3652
|
-
url: `${base}/v1/messages`,
|
|
3653
|
-
method: "HEAD",
|
|
3654
|
-
timeoutMs: ctx.httpTimeoutMs
|
|
3655
|
-
};
|
|
3656
|
-
const head = await (ctx.signal ? probeHttp({ ...headOpts, signal: ctx.signal }) : probeHttp(headOpts));
|
|
3657
|
-
if (!head.ok) return head;
|
|
3658
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3659
|
-
const enriched = { ...head };
|
|
3660
|
-
if (props.discoveredCapabilities) enriched.discoveredCapabilities = props.discoveredCapabilities;
|
|
3661
|
-
if (props.serverVersion) enriched.serverVersion = props.serverVersion;
|
|
3662
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3663
|
-
if (note) enriched.notes = [note];
|
|
3664
|
-
return enriched;
|
|
3665
|
-
},
|
|
3666
|
-
async probeModels(target, ctx) {
|
|
3667
|
-
const base = url(target);
|
|
3668
|
-
if (!base) return [];
|
|
3669
|
-
const opts = { url: `${base}/v1/models`, timeoutMs: ctx.httpTimeoutMs };
|
|
3670
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
3671
|
-
if (!result.ok || !result.data?.data) return [];
|
|
3672
|
-
return result.data.data.map((row) => typeof row?.id === "string" ? row.id : null).filter((id) => id !== null);
|
|
3673
|
-
},
|
|
3674
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3675
|
-
return synthLocalModel({
|
|
3676
|
-
target,
|
|
3677
|
-
wireModelId,
|
|
3678
|
-
kb,
|
|
3679
|
-
defaultCapabilities: defaultCapabilities17,
|
|
3680
|
-
apiFamily: "anthropic-messages",
|
|
3681
|
-
provider: "llamacpp",
|
|
3682
|
-
baseUrlForTarget: withAsIs
|
|
3683
|
-
});
|
|
3684
|
-
}
|
|
3685
|
-
};
|
|
3686
|
-
var llamacpp_anthropic_default = llamacppAnthropicRuntime;
|
|
3687
|
-
|
|
3688
|
-
// src/domains/providers/runtimes/local-native/llamacpp-completion.ts
|
|
3689
|
-
init_esm_shims();
|
|
3690
|
-
var defaultCapabilities18 = {
|
|
3691
|
-
chat: false,
|
|
3692
|
-
tools: false,
|
|
3693
|
-
reasoning: false,
|
|
3694
|
-
vision: false,
|
|
3695
|
-
audio: false,
|
|
3696
|
-
embeddings: false,
|
|
3697
|
-
rerank: false,
|
|
3698
|
-
fim: true,
|
|
3699
|
-
structuredOutputs: "gbnf",
|
|
3700
|
-
contextWindow: 8192,
|
|
3701
|
-
maxTokens: 4096
|
|
3702
|
-
};
|
|
3703
|
-
function targetUrl2(target) {
|
|
3704
|
-
return targetRootUrl(target);
|
|
3705
|
-
}
|
|
3706
|
-
function parseChunk(raw) {
|
|
3707
|
-
const chunk = {
|
|
3708
|
-
content: typeof raw.content === "string" ? raw.content : "",
|
|
3709
|
-
stop: raw.stop === true
|
|
3710
|
-
};
|
|
3711
|
-
if (raw.stop_type === "eos" || raw.stop_type === "limit" || raw.stop_type === "word" || raw.stop_type === "none") {
|
|
3712
|
-
chunk.stop_type = raw.stop_type;
|
|
3713
|
-
}
|
|
3714
|
-
if (typeof raw.tokens_predicted === "number") chunk.tokens_predicted = raw.tokens_predicted;
|
|
3715
|
-
if (typeof raw.tokens_evaluated === "number") chunk.tokens_evaluated = raw.tokens_evaluated;
|
|
3716
|
-
return chunk;
|
|
3717
|
-
}
|
|
3718
|
-
async function* streamSse(body) {
|
|
3719
|
-
const reader = body.getReader();
|
|
3720
|
-
const decoder = new TextDecoder("utf-8");
|
|
3721
|
-
let buffered = "";
|
|
3722
|
-
let droppedFrames = 0;
|
|
3723
|
-
try {
|
|
3724
|
-
while (true) {
|
|
3725
|
-
const { done, value } = await reader.read();
|
|
3726
|
-
if (done) break;
|
|
3727
|
-
buffered += decoder.decode(value, { stream: true });
|
|
3728
|
-
let nl = buffered.indexOf("\n");
|
|
3729
|
-
while (nl !== -1) {
|
|
3730
|
-
const line = buffered.slice(0, nl).trimEnd();
|
|
3731
|
-
buffered = buffered.slice(nl + 1);
|
|
3732
|
-
nl = buffered.indexOf("\n");
|
|
3733
|
-
if (line.length === 0) continue;
|
|
3734
|
-
const payload = line.startsWith("data:") ? line.slice(5).trim() : line;
|
|
3735
|
-
if (payload.length === 0 || payload === "[DONE]") continue;
|
|
3736
|
-
try {
|
|
3737
|
-
const parsed = JSON.parse(payload);
|
|
3738
|
-
const chunk = parseChunk(parsed);
|
|
3739
|
-
yield chunk;
|
|
3740
|
-
if (chunk.stop) return;
|
|
3741
|
-
} catch {
|
|
3742
|
-
droppedFrames += 1;
|
|
3743
|
-
}
|
|
3744
|
-
}
|
|
3745
|
-
}
|
|
3746
|
-
const tail = buffered.trim();
|
|
3747
|
-
if (tail.length > 0) {
|
|
3748
|
-
const payload = tail.startsWith("data:") ? tail.slice(5).trim() : tail;
|
|
3749
|
-
if (payload !== "[DONE]") {
|
|
3750
|
-
try {
|
|
3751
|
-
const parsed = JSON.parse(payload);
|
|
3752
|
-
yield parseChunk(parsed);
|
|
3753
|
-
} catch {
|
|
3754
|
-
droppedFrames += 1;
|
|
3755
|
-
}
|
|
3756
|
-
}
|
|
3757
|
-
}
|
|
3758
|
-
} finally {
|
|
3759
|
-
if (droppedFrames > 0) {
|
|
3760
|
-
process.stderr.write(`[clio:llamacpp] dropped ${droppedFrames} malformed stream frame(s)
|
|
3761
|
-
`);
|
|
3762
|
-
}
|
|
3763
|
-
reader.releaseLock();
|
|
3764
|
-
}
|
|
3765
|
-
}
|
|
3766
|
-
async function postStream(url2, body, signal) {
|
|
3767
|
-
const init = {
|
|
3768
|
-
method: "POST",
|
|
3769
|
-
headers: { "content-type": "application/json", accept: "text/event-stream" },
|
|
3770
|
-
body: JSON.stringify(body)
|
|
3771
|
-
};
|
|
3772
|
-
if (signal) init.signal = signal;
|
|
3773
|
-
const response = await fetch(url2, init);
|
|
3774
|
-
if (!response.ok || !response.body) {
|
|
3775
|
-
throw new Error(`HTTP ${response.status}: ${response.statusText}`);
|
|
3776
|
-
}
|
|
3777
|
-
return response.body;
|
|
3778
|
-
}
|
|
3779
|
-
function buildCompleteBody(opts) {
|
|
3780
|
-
const body = { prompt: opts.prompt, stream: true };
|
|
3781
|
-
if (opts.n_predict !== void 0) body.n_predict = opts.n_predict;
|
|
3782
|
-
if (opts.stop && opts.stop.length > 0) body.stop = opts.stop;
|
|
3783
|
-
if (opts.grammar) body.grammar = opts.grammar;
|
|
3784
|
-
if (opts.json_schema) body.json_schema = opts.json_schema;
|
|
3785
|
-
if (opts.cache_prompt !== void 0) body.cache_prompt = opts.cache_prompt;
|
|
3786
|
-
return body;
|
|
3787
|
-
}
|
|
3788
|
-
function buildInfillBody(opts) {
|
|
3789
|
-
const body = buildCompleteBody(opts);
|
|
3790
|
-
body.input_prefix = opts.input_prefix;
|
|
3791
|
-
body.input_suffix = opts.input_suffix;
|
|
3792
|
-
if (opts.input_extra) body.input_extra = opts.input_extra;
|
|
3793
|
-
return body;
|
|
3794
|
-
}
|
|
3795
|
-
var llamacppCompletionRuntime = {
|
|
3796
|
-
id: "llamacpp-completion",
|
|
3797
|
-
displayName: "llama.cpp (completion / infill)",
|
|
3798
|
-
kind: "http",
|
|
3799
|
-
tier: "local-native",
|
|
3800
|
-
apiFamily: "openai-completions",
|
|
3801
|
-
auth: "api-key",
|
|
3802
|
-
defaultCapabilities: defaultCapabilities18,
|
|
3803
|
-
hidden: true,
|
|
3804
|
-
async probe(target, ctx) {
|
|
3805
|
-
const base = targetUrl2(target);
|
|
3806
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3807
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3808
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3809
|
-
if (!health.ok) return health;
|
|
3810
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3811
|
-
const status = await probeLlamaCppModelStatus(base, target, ctx);
|
|
3812
|
-
const enriched = { ...health };
|
|
3813
|
-
const discoveredCapabilities = {
|
|
3814
|
-
...props.discoveredCapabilities ?? {},
|
|
3815
|
-
...status.discoveredCapabilities ?? {}
|
|
3816
|
-
};
|
|
3817
|
-
if (Object.keys(discoveredCapabilities).length > 0) {
|
|
3818
|
-
enriched.discoveredCapabilities = discoveredCapabilities;
|
|
3819
|
-
if (status.modelId) enriched.capabilityModelId = status.modelId;
|
|
3820
|
-
}
|
|
3821
|
-
if (props.serverVersion) enriched.serverVersion = props.serverVersion;
|
|
3822
|
-
const note = await detectModelMismatch(base, target, ctx);
|
|
3823
|
-
const notes = [...status.notes ?? [], ...note ? [note] : []];
|
|
3824
|
-
if (notes.length > 0) enriched.notes = notes;
|
|
3825
|
-
return enriched;
|
|
3826
|
-
},
|
|
3827
|
-
async probeModels(target, ctx) {
|
|
3828
|
-
const base = targetUrl2(target);
|
|
3829
|
-
if (!base) return [];
|
|
3830
|
-
return probeOpenAIModels(base, ctx);
|
|
3831
|
-
},
|
|
3832
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3833
|
-
return synthLocalModel({
|
|
3834
|
-
target,
|
|
3835
|
-
wireModelId,
|
|
3836
|
-
kb,
|
|
3837
|
-
defaultCapabilities: defaultCapabilities18,
|
|
3838
|
-
apiFamily: "openai-completions",
|
|
3839
|
-
provider: "llamacpp",
|
|
3840
|
-
baseUrlForTarget: withV1
|
|
3841
|
-
});
|
|
3842
|
-
},
|
|
3843
|
-
async *complete(target, opts) {
|
|
3844
|
-
const base = targetUrl2(target);
|
|
3845
|
-
if (!base) throw new Error("target has no url");
|
|
3846
|
-
const body = await postStream(`${base}/completion`, buildCompleteBody(opts), opts.signal);
|
|
3847
|
-
for await (const chunk of streamSse(body)) yield chunk;
|
|
3848
|
-
},
|
|
3849
|
-
async *infill(target, opts) {
|
|
3850
|
-
const base = targetUrl2(target);
|
|
3851
|
-
if (!base) throw new Error("target has no url");
|
|
3852
|
-
const body = await postStream(`${base}/infill`, buildInfillBody(opts), opts.signal);
|
|
3853
|
-
for await (const chunk of streamSse(body)) yield chunk;
|
|
3854
|
-
}
|
|
3855
|
-
};
|
|
3856
|
-
var llamacpp_completion_default = llamacppCompletionRuntime;
|
|
3857
|
-
|
|
3858
|
-
// src/domains/providers/runtimes/local-native/llamacpp-embed.ts
|
|
3859
|
-
init_esm_shims();
|
|
3860
|
-
var defaultCapabilities19 = {
|
|
3861
|
-
chat: false,
|
|
3862
|
-
tools: false,
|
|
3863
|
-
reasoning: false,
|
|
3864
|
-
vision: false,
|
|
3865
|
-
audio: false,
|
|
3866
|
-
embeddings: true,
|
|
3867
|
-
rerank: false,
|
|
3868
|
-
fim: false,
|
|
3869
|
-
contextWindow: 8192,
|
|
3870
|
-
maxTokens: 0
|
|
3871
|
-
};
|
|
3872
|
-
function targetUrl3(target) {
|
|
3873
|
-
return targetRootUrl(target);
|
|
3874
|
-
}
|
|
3875
|
-
function meanPool(matrix) {
|
|
3876
|
-
if (matrix.length === 0) return [];
|
|
3877
|
-
const dim = matrix[0]?.length ?? 0;
|
|
3878
|
-
const sum = new Array(dim).fill(0);
|
|
3879
|
-
for (const row of matrix) {
|
|
3880
|
-
for (let i = 0; i < dim; i++) sum[i] = (sum[i] ?? 0) + (row[i] ?? 0);
|
|
3881
|
-
}
|
|
3882
|
-
return sum.map((v) => v / matrix.length);
|
|
3883
|
-
}
|
|
3884
|
-
function flattenNativeEmbedding(entry) {
|
|
3885
|
-
const value = entry.embedding;
|
|
3886
|
-
if (!value || value.length === 0) return [];
|
|
3887
|
-
if (Array.isArray(value[0])) return meanPool(value);
|
|
3888
|
-
return value;
|
|
3889
|
-
}
|
|
3890
|
-
async function postJson(url2, body, signal) {
|
|
3891
|
-
const init = {
|
|
3892
|
-
method: "POST",
|
|
3893
|
-
headers: { "content-type": "application/json" },
|
|
3894
|
-
body: JSON.stringify(body)
|
|
3895
|
-
};
|
|
3896
|
-
if (signal) init.signal = signal;
|
|
3897
|
-
const response = await fetch(url2, init);
|
|
3898
|
-
if (!response.ok) return { status: response.status, data: null };
|
|
3899
|
-
try {
|
|
3900
|
-
const data = await response.json();
|
|
3901
|
-
return { status: response.status, data };
|
|
3902
|
-
} catch {
|
|
3903
|
-
return { status: response.status, data: null };
|
|
3904
|
-
}
|
|
3905
|
-
}
|
|
3906
|
-
var llamacppEmbedRuntime = {
|
|
3907
|
-
id: "llamacpp-embed",
|
|
3908
|
-
displayName: "llama.cpp (embeddings)",
|
|
3909
|
-
kind: "http",
|
|
3910
|
-
tier: "local-native",
|
|
3911
|
-
apiFamily: "openai-completions",
|
|
3912
|
-
auth: "api-key",
|
|
3913
|
-
defaultCapabilities: defaultCapabilities19,
|
|
3914
|
-
hidden: true,
|
|
3915
|
-
async probe(target, ctx) {
|
|
3916
|
-
const base = targetUrl3(target);
|
|
3917
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
3918
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
3919
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
3920
|
-
if (!health.ok) return health;
|
|
3921
|
-
const probeResponse = await fetch(`${base}/embedding`, {
|
|
3922
|
-
method: "POST",
|
|
3923
|
-
headers: { "content-type": "application/json" },
|
|
3924
|
-
body: JSON.stringify({ content: "probe" }),
|
|
3925
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
3926
|
-
}).catch((err) => {
|
|
3927
|
-
return new Response(null, { status: 599, statusText: String(err) });
|
|
3928
|
-
});
|
|
3929
|
-
if (!probeResponse.ok) {
|
|
3930
|
-
return { ok: false, error: `/embedding not available: HTTP ${probeResponse.status}` };
|
|
3931
|
-
}
|
|
3932
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
3933
|
-
const result = { ok: true };
|
|
3934
|
-
if (health.latencyMs !== void 0) result.latencyMs = health.latencyMs;
|
|
3935
|
-
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
3936
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
3937
|
-
return result;
|
|
3938
|
-
},
|
|
3939
|
-
async probeModels(target, ctx) {
|
|
3940
|
-
const base = targetUrl3(target);
|
|
3941
|
-
if (!base) return [];
|
|
3942
|
-
return probeOpenAIModels(base, ctx);
|
|
3943
|
-
},
|
|
3944
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
3945
|
-
return synthLocalModel({
|
|
3946
|
-
target,
|
|
3947
|
-
wireModelId,
|
|
3948
|
-
kb,
|
|
3949
|
-
defaultCapabilities: defaultCapabilities19,
|
|
3950
|
-
apiFamily: "openai-completions",
|
|
3951
|
-
provider: "llamacpp",
|
|
3952
|
-
baseUrlForTarget: withV1
|
|
3953
|
-
});
|
|
3954
|
-
},
|
|
3955
|
-
async embed(target, input, ctx) {
|
|
3956
|
-
const base = targetUrl3(target);
|
|
3957
|
-
if (!base) throw new Error("target has no url");
|
|
3958
|
-
const modelId = target.defaultModel ?? "default";
|
|
3959
|
-
const inputs = Array.isArray(input) ? input : [input];
|
|
3960
|
-
const oai = await postJson(
|
|
3961
|
-
`${base}/v1/embeddings`,
|
|
3962
|
-
{ input: inputs, model: modelId, encoding_format: "float" },
|
|
3963
|
-
ctx.signal
|
|
3964
|
-
);
|
|
3965
|
-
if (oai.data && Array.isArray(oai.data.data)) {
|
|
3966
|
-
const rows = oai.data.data;
|
|
3967
|
-
const sorted = [...rows].sort((a, b) => (a.index ?? 0) - (b.index ?? 0));
|
|
3968
|
-
const vectors2 = sorted.map((row) => row.embedding ?? []);
|
|
3969
|
-
const tokens = oai.data.usage?.total_tokens ?? oai.data.usage?.prompt_tokens ?? void 0;
|
|
3970
|
-
const dim = vectors2[0]?.length ?? 0;
|
|
3971
|
-
const result = {
|
|
3972
|
-
vectors: vectors2,
|
|
3973
|
-
model: oai.data.model ?? modelId,
|
|
3974
|
-
dimensions: dim
|
|
3975
|
-
};
|
|
3976
|
-
if (tokens !== void 0) result.tokensUsed = tokens;
|
|
3977
|
-
return result;
|
|
3978
|
-
}
|
|
3979
|
-
const probeOpts = { url: `${base}/embedding`, timeoutMs: ctx.httpTimeoutMs };
|
|
3980
|
-
const native = await (ctx.signal ? probeJson({
|
|
3981
|
-
...probeOpts,
|
|
3982
|
-
method: "POST",
|
|
3983
|
-
body: JSON.stringify({ content: inputs }),
|
|
3984
|
-
headers: { "content-type": "application/json" },
|
|
3985
|
-
signal: ctx.signal
|
|
3986
|
-
}) : probeJson({
|
|
3987
|
-
...probeOpts,
|
|
3988
|
-
method: "POST",
|
|
3989
|
-
body: JSON.stringify({ content: inputs }),
|
|
3990
|
-
headers: { "content-type": "application/json" }
|
|
3991
|
-
}));
|
|
3992
|
-
if (!native.ok || !native.data) {
|
|
3993
|
-
throw new Error(`llama.cpp embedding failed: ${native.error ?? "unknown"}`);
|
|
3994
|
-
}
|
|
3995
|
-
const items = native.data;
|
|
3996
|
-
items.sort((a, b) => (a.index ?? 0) - (b.index ?? 0));
|
|
3997
|
-
const vectors = items.map(flattenNativeEmbedding);
|
|
3998
|
-
return {
|
|
3999
|
-
vectors,
|
|
4000
|
-
model: modelId,
|
|
4001
|
-
dimensions: vectors[0]?.length ?? 0
|
|
4002
|
-
};
|
|
4003
|
-
}
|
|
4004
|
-
};
|
|
4005
|
-
var llamacpp_embed_default = llamacppEmbedRuntime;
|
|
4006
|
-
|
|
4007
|
-
// src/domains/providers/runtimes/local-native/llamacpp-rerank.ts
|
|
4008
|
-
init_esm_shims();
|
|
4009
|
-
var defaultCapabilities20 = {
|
|
4010
|
-
chat: false,
|
|
4011
|
-
tools: false,
|
|
4012
|
-
reasoning: false,
|
|
4013
|
-
vision: false,
|
|
4014
|
-
audio: false,
|
|
4015
|
-
embeddings: false,
|
|
4016
|
-
rerank: true,
|
|
4017
|
-
fim: false,
|
|
4018
|
-
contextWindow: 8192,
|
|
4019
|
-
maxTokens: 0
|
|
4020
|
-
};
|
|
4021
|
-
function targetUrl4(target) {
|
|
4022
|
-
return targetRootUrl(target);
|
|
4023
|
-
}
|
|
4024
|
-
var llamacppRerankRuntime = {
|
|
4025
|
-
id: "llamacpp-rerank",
|
|
4026
|
-
displayName: "llama.cpp (rerank)",
|
|
4027
|
-
kind: "http",
|
|
4028
|
-
tier: "local-native",
|
|
4029
|
-
apiFamily: "openai-completions",
|
|
4030
|
-
auth: "api-key",
|
|
4031
|
-
defaultCapabilities: defaultCapabilities20,
|
|
4032
|
-
hidden: true,
|
|
4033
|
-
async probe(target, ctx) {
|
|
4034
|
-
const base = targetUrl4(target);
|
|
4035
|
-
if (!base) return { ok: false, error: "target has no url" };
|
|
4036
|
-
const healthOpts = { url: `${base}/health`, timeoutMs: ctx.httpTimeoutMs };
|
|
4037
|
-
const health = await (ctx.signal ? probeHttp({ ...healthOpts, signal: ctx.signal }) : probeHttp(healthOpts));
|
|
4038
|
-
if (!health.ok) return health;
|
|
4039
|
-
const modelId = target.defaultModel ?? "default";
|
|
4040
|
-
const probeResponse = await fetch(`${base}/reranking`, {
|
|
4041
|
-
method: "POST",
|
|
4042
|
-
headers: { "content-type": "application/json" },
|
|
4043
|
-
body: JSON.stringify({ query: "probe", documents: ["a"], model: modelId }),
|
|
4044
|
-
...ctx.signal ? { signal: ctx.signal } : {}
|
|
4045
|
-
}).catch((err) => new Response(null, { status: 599, statusText: String(err) }));
|
|
4046
|
-
if (!(probeResponse.status === 200 || probeResponse.status === 202)) {
|
|
4047
|
-
return { ok: false, error: `/reranking not available: HTTP ${probeResponse.status}` };
|
|
4048
|
-
}
|
|
4049
|
-
const props = await probeLlamaCppProps(base, ctx);
|
|
4050
|
-
const result = { ok: true };
|
|
4051
|
-
if (health.latencyMs !== void 0) result.latencyMs = health.latencyMs;
|
|
4052
|
-
if (props.discoveredCapabilities) result.discoveredCapabilities = props.discoveredCapabilities;
|
|
4053
|
-
if (props.serverVersion) result.serverVersion = props.serverVersion;
|
|
4054
|
-
return result;
|
|
4055
|
-
},
|
|
4056
|
-
async probeModels(target, ctx) {
|
|
4057
|
-
const base = targetUrl4(target);
|
|
4058
|
-
if (!base) return [];
|
|
4059
|
-
return probeOpenAIModels(base, ctx);
|
|
4060
|
-
},
|
|
4061
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4062
|
-
return synthLocalModel({
|
|
4063
|
-
target,
|
|
4064
|
-
wireModelId,
|
|
4065
|
-
kb,
|
|
4066
|
-
defaultCapabilities: defaultCapabilities20,
|
|
4067
|
-
apiFamily: "openai-completions",
|
|
4068
|
-
provider: "llamacpp",
|
|
4069
|
-
baseUrlForTarget: withV1
|
|
4070
|
-
});
|
|
4071
|
-
},
|
|
4072
|
-
async rerank(target, query, documents, ctx) {
|
|
4073
|
-
const base = targetUrl4(target);
|
|
4074
|
-
if (!base) throw new Error("target has no url");
|
|
4075
|
-
const modelId = target.defaultModel ?? "default";
|
|
4076
|
-
const req = {
|
|
4077
|
-
url: `${base}/reranking`,
|
|
4078
|
-
method: "POST",
|
|
4079
|
-
timeoutMs: ctx.httpTimeoutMs,
|
|
4080
|
-
headers: { "content-type": "application/json" },
|
|
4081
|
-
body: JSON.stringify({
|
|
4082
|
-
query,
|
|
4083
|
-
documents,
|
|
4084
|
-
top_n: documents.length,
|
|
4085
|
-
model: modelId
|
|
4086
|
-
})
|
|
4087
|
-
};
|
|
4088
|
-
const result = await (ctx.signal ? probeJson({ ...req, signal: ctx.signal }) : probeJson(req));
|
|
4089
|
-
if (!result.ok || !result.data) {
|
|
4090
|
-
throw new Error(`llama.cpp rerank failed: ${result.error ?? "unknown"}`);
|
|
4091
|
-
}
|
|
4092
|
-
const rows = result.data.results ?? [];
|
|
4093
|
-
const items = rows.map((row) => {
|
|
4094
|
-
const idx = typeof row.index === "number" ? row.index : 0;
|
|
4095
|
-
const score = typeof row.relevance_score === "number" ? row.relevance_score : 0;
|
|
4096
|
-
const doc = typeof row.document === "string" ? row.document : typeof row.document?.text === "string" ? row.document.text : void 0;
|
|
4097
|
-
const item = { index: idx, score };
|
|
4098
|
-
if (doc !== void 0) item.document = doc;
|
|
4099
|
-
return item;
|
|
4100
|
-
});
|
|
4101
|
-
return { items, model: result.data.model ?? modelId };
|
|
4102
|
-
}
|
|
4103
|
-
};
|
|
4104
|
-
var llamacpp_rerank_default = llamacppRerankRuntime;
|
|
4105
|
-
|
|
4106
|
-
// src/domains/providers/runtimes/local-native/lmstudio.ts
|
|
4107
|
-
init_esm_shims();
|
|
4108
|
-
var defaultCapabilities21 = {
|
|
4109
|
-
chat: true,
|
|
4110
|
-
tools: true,
|
|
4111
|
-
toolCallFormat: "openai",
|
|
4112
|
-
structuredOutputs: "json-schema",
|
|
4113
|
-
reasoning: true,
|
|
4114
|
-
vision: true,
|
|
4115
|
-
audio: false,
|
|
4116
|
-
embeddings: false,
|
|
4117
|
-
rerank: false,
|
|
4118
|
-
fim: false,
|
|
4119
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4120
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4121
|
-
};
|
|
4122
|
-
var reasoningOptionsByTargetModel = /* @__PURE__ */ new Map();
|
|
4123
|
-
function reasoningOptionsKey(target, modelId) {
|
|
4124
|
-
const url2 = (target.url ?? "").replace(/^ws:/u, "http:").replace(/^wss:/u, "https:").replace(/\/$/u, "");
|
|
4125
|
-
return `${target.id}|${url2}|${modelId}`;
|
|
4126
|
-
}
|
|
4127
|
-
function rememberReasoningOptions(target, models) {
|
|
4128
|
-
const prefix = reasoningOptionsKey(target, "");
|
|
4129
|
-
for (const key of reasoningOptionsByTargetModel.keys()) {
|
|
4130
|
-
if (key.startsWith(prefix)) reasoningOptionsByTargetModel.delete(key);
|
|
4131
|
-
}
|
|
4132
|
-
for (const model of models) {
|
|
4133
|
-
if (!model.reasoningOptions) continue;
|
|
4134
|
-
reasoningOptionsByTargetModel.set(reasoningOptionsKey(target, model.key), model.reasoningOptions);
|
|
4135
|
-
for (const instance of model.loadedInstances) {
|
|
4136
|
-
reasoningOptionsByTargetModel.set(reasoningOptionsKey(target, instance.id), model.reasoningOptions);
|
|
1173
|
+
runtimeId: input.runtimeId,
|
|
1174
|
+
apiFamily: input.apiFamily ?? null,
|
|
1175
|
+
modelId: input.modelId,
|
|
1176
|
+
family,
|
|
1177
|
+
capabilities: input.capabilities,
|
|
1178
|
+
thinking,
|
|
1179
|
+
request: resolveRequestCapability(thinking, parser, input.runtimeId),
|
|
1180
|
+
response: {
|
|
1181
|
+
parser,
|
|
1182
|
+
stripTokenizerSentinels: true
|
|
4137
1183
|
}
|
|
4138
|
-
}
|
|
1184
|
+
};
|
|
1185
|
+
if (quirks) result.quirks = quirks;
|
|
1186
|
+
return result;
|
|
4139
1187
|
}
|
|
4140
|
-
function
|
|
4141
|
-
const
|
|
4142
|
-
|
|
1188
|
+
function resolveModelRuntimeCapabilitiesForStatus(status, wireModelId, knowledgeBase, options) {
|
|
1189
|
+
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1190
|
+
const kbHit = modelId ? knowledgeBase?.lookup(modelId) ?? null : null;
|
|
1191
|
+
const capabilities = resolveModelCapabilities(status, modelId, knowledgeBase, {
|
|
1192
|
+
detectedReasoning: options?.detectedReasoning ?? null
|
|
1193
|
+
});
|
|
1194
|
+
const runtimeId = status.runtime?.id ?? status.target.runtime;
|
|
1195
|
+
return resolveModelRuntimeCapabilities({
|
|
1196
|
+
targetId: status.target.id,
|
|
1197
|
+
runtimeId,
|
|
1198
|
+
apiFamily: status.runtime?.apiFamily ?? null,
|
|
1199
|
+
modelId,
|
|
1200
|
+
capabilities,
|
|
1201
|
+
kbHit,
|
|
1202
|
+
...thinkingHintsForCatalogModel(runtimeId, modelId),
|
|
1203
|
+
...options?.configuredThinkingLevel ? { configuredThinkingLevel: options.configuredThinkingLevel } : {}
|
|
1204
|
+
});
|
|
4143
1205
|
}
|
|
4144
|
-
function
|
|
4145
|
-
|
|
4146
|
-
|
|
4147
|
-
|
|
4148
|
-
|
|
4149
|
-
|
|
4150
|
-
|
|
4151
|
-
|
|
4152
|
-
|
|
4153
|
-
|
|
4154
|
-
|
|
1206
|
+
function resolveModelRuntimeCapabilitiesForProviders(providers, targetId, wireModelId, configuredThinkingLevel) {
|
|
1207
|
+
const id = targetId?.trim();
|
|
1208
|
+
if (!id) return null;
|
|
1209
|
+
const status = providers.list().find((entry) => entry.target.id === id);
|
|
1210
|
+
if (!status) return null;
|
|
1211
|
+
const modelId = wireModelId?.trim() || status.target.defaultModel?.trim() || "";
|
|
1212
|
+
const detectedReasoning = modelId && typeof providers.getDetectedReasoning === "function" ? providers.getDetectedReasoning(id, modelId) : null;
|
|
1213
|
+
return resolveModelRuntimeCapabilitiesForStatus(status, modelId, providers.knowledgeBase, {
|
|
1214
|
+
detectedReasoning,
|
|
1215
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1216
|
+
});
|
|
4155
1217
|
}
|
|
4156
|
-
function
|
|
1218
|
+
function thinkingHintsForModel(model) {
|
|
4157
1219
|
const out = {};
|
|
4158
|
-
if (model
|
|
4159
|
-
|
|
4160
|
-
if (
|
|
4161
|
-
|
|
4162
|
-
out.reasoning = model.reasoningOptions.some((option) => option !== "off");
|
|
4163
|
-
const contextWindow = loadedContextLength(instance) ?? model.maxContextLength;
|
|
4164
|
-
if (contextWindow !== void 0) out.contextWindow = contextWindow;
|
|
1220
|
+
if (!model) return out;
|
|
1221
|
+
const compat = model.compat;
|
|
1222
|
+
if (compat?.forceAdaptiveThinking !== void 0) out.adaptiveThinking = compat.forceAdaptiveThinking;
|
|
1223
|
+
if (model.thinkingLevelMap) out.thinkingLevelMap = model.thinkingLevelMap;
|
|
4165
1224
|
return out;
|
|
4166
1225
|
}
|
|
4167
|
-
function
|
|
4168
|
-
|
|
4169
|
-
|
|
4170
|
-
key: model.key,
|
|
4171
|
-
reasoningLevels
|
|
4172
|
-
};
|
|
4173
|
-
if (instance) {
|
|
4174
|
-
status.instanceId = instance.id;
|
|
4175
|
-
status.loadConfig = instance.config;
|
|
4176
|
-
const contextLength = loadedContextLength(instance);
|
|
4177
|
-
if (contextLength !== void 0) status.contextLength = contextLength;
|
|
4178
|
-
}
|
|
4179
|
-
return status;
|
|
1226
|
+
function thinkingHintsForCatalogModel(runtimeId, modelId) {
|
|
1227
|
+
if (!runtimeId || modelId.length === 0) return {};
|
|
1228
|
+
return thinkingHintsForModel(getCatalogModelForRuntime(runtimeId, modelId));
|
|
4180
1229
|
}
|
|
4181
|
-
function
|
|
4182
|
-
|
|
4183
|
-
|
|
4184
|
-
|
|
4185
|
-
|
|
4186
|
-
|
|
4187
|
-
|
|
4188
|
-
|
|
4189
|
-
|
|
4190
|
-
|
|
4191
|
-
|
|
4192
|
-
const modelCapabilities = {};
|
|
4193
|
-
const modelStates = {};
|
|
4194
|
-
for (const model of models) {
|
|
4195
|
-
const levels = lmStudioReasoningLevels(model.reasoningOptions);
|
|
4196
|
-
const keyInstance = model.loadedInstances[0];
|
|
4197
|
-
modelCapabilities[model.key] = capabilities(model, keyInstance);
|
|
4198
|
-
modelStates[model.key] = statusFor(model, keyInstance, levels);
|
|
4199
|
-
for (const instance of model.loadedInstances) {
|
|
4200
|
-
if (!ids.includes(instance.id)) ids.push(instance.id);
|
|
4201
|
-
modelCapabilities[instance.id] = capabilities(model, instance);
|
|
4202
|
-
const instanceStatus = statusFor(model, instance, levels);
|
|
4203
|
-
modelStates[instance.id] = instanceStatus;
|
|
4204
|
-
const resolution = resolveLmStudioInstance(target, models, instance.id, configuredModel(target));
|
|
4205
|
-
instanceStatus.detail = resolution.peerTargets.length > 0 ? `loaded on ${target.id}; also loaded on ${resolution.peerTargets.join(", ")}` : `loaded on ${target.id}`;
|
|
4206
|
-
}
|
|
4207
|
-
if (model.loadedInstances.length === 0) {
|
|
4208
|
-
ids.push(model.key);
|
|
4209
|
-
const keyStatus = modelStates[model.key];
|
|
4210
|
-
if (keyStatus) keyStatus.detail = "not loaded (LM Studio will load it on first use)";
|
|
4211
|
-
}
|
|
1230
|
+
function thinkingFormatFromModelApi(api) {
|
|
1231
|
+
switch (api) {
|
|
1232
|
+
case "anthropic-messages":
|
|
1233
|
+
case "bedrock-converse-stream":
|
|
1234
|
+
case "claude-agent-sdk":
|
|
1235
|
+
case "claude-code-subprocess":
|
|
1236
|
+
return "anthropic-extended";
|
|
1237
|
+
case "openai-codex-responses":
|
|
1238
|
+
return "openai-codex";
|
|
1239
|
+
default:
|
|
1240
|
+
return void 0;
|
|
4212
1241
|
}
|
|
4213
|
-
const selected = resolveModel(models, configuredModel(target));
|
|
4214
|
-
const result = {
|
|
4215
|
-
ok: true,
|
|
4216
|
-
models: ids,
|
|
4217
|
-
modelCapabilities,
|
|
4218
|
-
modelStates,
|
|
4219
|
-
serverVersion: catalog.tier === "0.4+" ? "LM Studio API 0.4+" : catalog.tier === "0.3.x" ? "LM Studio API 0.3.x" : "LM Studio OpenAI-compatible API",
|
|
4220
|
-
surfaces: {
|
|
4221
|
-
openaiChat: "/v1/chat/completions",
|
|
4222
|
-
...catalog.tier === "0.4+" ? { nativeV1: "/api/v1/models" } : {},
|
|
4223
|
-
...catalog.tier === "0.3.x" ? { nativeV0: "/api/v0/models" } : {}
|
|
4224
|
-
}
|
|
4225
|
-
};
|
|
4226
|
-
if (catalog.latencyMs !== void 0) result.latencyMs = catalog.latencyMs;
|
|
4227
|
-
if (selected) {
|
|
4228
|
-
result.discoveredCapabilities = capabilities(selected.model, selected.instance);
|
|
4229
|
-
result.capabilityModelId = configuredModel(target) ?? selected.model.key;
|
|
4230
|
-
}
|
|
4231
|
-
const loaded = models.flatMap(
|
|
4232
|
-
(model) => model.loadedInstances.map((instance) => {
|
|
4233
|
-
const context = loadedContextLength(instance);
|
|
4234
|
-
return `${model.key} as ${instance.id}${context ? ` at ${context} tokens` : ""}`;
|
|
4235
|
-
})
|
|
4236
|
-
);
|
|
4237
|
-
result.notes = [
|
|
4238
|
-
`LM Studio surface tier ${catalog.tier ?? "unknown"}`,
|
|
4239
|
-
`Surfaces: ${Object.values(result.surfaces ?? {}).join(", ")}`,
|
|
4240
|
-
...loaded.length > 0 ? [`Loaded instances: ${loaded.join(", ")}`] : []
|
|
4241
|
-
];
|
|
4242
|
-
return result;
|
|
4243
1242
|
}
|
|
4244
|
-
|
|
4245
|
-
|
|
4246
|
-
|
|
4247
|
-
|
|
4248
|
-
|
|
4249
|
-
|
|
4250
|
-
|
|
4251
|
-
|
|
4252
|
-
|
|
4253
|
-
|
|
4254
|
-
|
|
4255
|
-
|
|
4256
|
-
|
|
4257
|
-
|
|
4258
|
-
|
|
4259
|
-
|
|
4260
|
-
};
|
|
4261
|
-
}
|
|
4262
|
-
return probeFromCatalog(catalog, target);
|
|
4263
|
-
},
|
|
4264
|
-
async probeModels(target, ctx) {
|
|
4265
|
-
const catalog = await listLmStudioModels(target, ctx);
|
|
4266
|
-
return probeFromCatalog(catalog, target).models ?? [];
|
|
4267
|
-
},
|
|
4268
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4269
|
-
const canonicalTarget = target.runtime === "lmstudio" ? target : { ...target, runtime: "lmstudio" };
|
|
4270
|
-
const model = synthLocalModel({
|
|
4271
|
-
target: canonicalTarget,
|
|
4272
|
-
wireModelId,
|
|
4273
|
-
kb,
|
|
4274
|
-
defaultCapabilities: defaultCapabilities21,
|
|
4275
|
-
apiFamily: "openai-completions",
|
|
4276
|
-
provider: "lmstudio",
|
|
4277
|
-
baseUrlForTarget: (url2) => withV1(url2.replace(/^ws:/u, "http:").replace(/^wss:/u, "https:"))
|
|
4278
|
-
});
|
|
4279
|
-
const metadata2 = model.clio;
|
|
4280
|
-
if (metadata2) {
|
|
4281
|
-
metadata2.chatTemplateKwargsUnsupported = true;
|
|
4282
|
-
if (target.defaultModel) metadata2.lmstudioDefaultModel = target.defaultModel;
|
|
4283
|
-
const options = reasoningOptionsByTargetModel.get(reasoningOptionsKey(canonicalTarget, wireModelId));
|
|
4284
|
-
if (options) metadata2.lmstudioReasoningOptions = options;
|
|
4285
|
-
}
|
|
4286
|
-
return model;
|
|
1243
|
+
function capabilitiesFromModel(model) {
|
|
1244
|
+
const format = model.compat?.thinkingFormat ?? thinkingFormatFromModelApi(model.api);
|
|
1245
|
+
const caps = {
|
|
1246
|
+
chat: true,
|
|
1247
|
+
tools: true,
|
|
1248
|
+
reasoning: model.reasoning === true,
|
|
1249
|
+
vision: Array.isArray(model.input) && model.input.includes("image"),
|
|
1250
|
+
audio: false,
|
|
1251
|
+
embeddings: false,
|
|
1252
|
+
rerank: false,
|
|
1253
|
+
fim: false,
|
|
1254
|
+
contextWindow: model.contextWindow,
|
|
1255
|
+
maxTokens: model.maxTokens
|
|
1256
|
+
};
|
|
1257
|
+
if (format === "qwen-chat-template" || format === "openrouter" || format === "zai" || format === "anthropic-extended" || format === "deepseek-r1" || format === "openai-codex" || format === "harmony") {
|
|
1258
|
+
caps.thinkingFormat = format;
|
|
4287
1259
|
}
|
|
4288
|
-
|
|
4289
|
-
var lmstudio_default = lmstudioRuntime;
|
|
4290
|
-
|
|
4291
|
-
// src/domains/providers/runtimes/local-native/ollama-native.ts
|
|
4292
|
-
init_esm_shims();
|
|
4293
|
-
var defaultCapabilities22 = {
|
|
4294
|
-
chat: true,
|
|
4295
|
-
tools: true,
|
|
4296
|
-
toolCallFormat: "openai",
|
|
4297
|
-
reasoning: false,
|
|
4298
|
-
vision: false,
|
|
4299
|
-
audio: false,
|
|
4300
|
-
embeddings: false,
|
|
4301
|
-
rerank: false,
|
|
4302
|
-
fim: false,
|
|
4303
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4304
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4305
|
-
};
|
|
4306
|
-
function positiveNumber3(value) {
|
|
4307
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : void 0;
|
|
1260
|
+
return caps;
|
|
4308
1261
|
}
|
|
4309
|
-
|
|
4310
|
-
const
|
|
4311
|
-
const
|
|
4312
|
-
|
|
4313
|
-
|
|
4314
|
-
|
|
4315
|
-
|
|
4316
|
-
|
|
4317
|
-
|
|
4318
|
-
|
|
4319
|
-
|
|
4320
|
-
|
|
4321
|
-
|
|
4322
|
-
|
|
4323
|
-
|
|
4324
|
-
|
|
4325
|
-
}
|
|
4326
|
-
|
|
4327
|
-
|
|
4328
|
-
|
|
4329
|
-
|
|
4330
|
-
|
|
4331
|
-
|
|
4332
|
-
|
|
4333
|
-
|
|
4334
|
-
|
|
4335
|
-
|
|
4336
|
-
|
|
4337
|
-
|
|
4338
|
-
|
|
4339
|
-
|
|
4340
|
-
|
|
4341
|
-
|
|
4342
|
-
|
|
4343
|
-
return failed;
|
|
4344
|
-
}
|
|
4345
|
-
const out = { ok: true };
|
|
4346
|
-
if (result.latencyMs !== void 0) out.latencyMs = result.latencyMs;
|
|
4347
|
-
const modelStates = await probeResidentModelStates(base, ctx);
|
|
4348
|
-
if (modelStates) out.modelStates = modelStates;
|
|
4349
|
-
return out;
|
|
4350
|
-
},
|
|
4351
|
-
async probeModels(target, ctx) {
|
|
4352
|
-
const base = targetBaseUrl(target);
|
|
4353
|
-
if (!base) return [];
|
|
4354
|
-
const opts = { url: `${base}/api/tags`, timeoutMs: ctx.httpTimeoutMs };
|
|
4355
|
-
const result = await (ctx.signal ? probeJson({ ...opts, signal: ctx.signal }) : probeJson(opts));
|
|
4356
|
-
if (!result.ok || !result.data?.models) return [];
|
|
4357
|
-
return result.data.models.map((row) => typeof row?.name === "string" ? row.name : null).filter((name) => name !== null);
|
|
4358
|
-
},
|
|
4359
|
-
synthesizeModel(target, wireModelId, kb) {
|
|
4360
|
-
return synthLocalModel({
|
|
4361
|
-
target,
|
|
4362
|
-
wireModelId,
|
|
4363
|
-
kb,
|
|
4364
|
-
defaultCapabilities: defaultCapabilities22,
|
|
4365
|
-
apiFamily: "ollama-native",
|
|
4366
|
-
provider: "ollama",
|
|
4367
|
-
baseUrlForTarget: withAsIs
|
|
4368
|
-
});
|
|
4369
|
-
}
|
|
4370
|
-
};
|
|
4371
|
-
var ollama_native_default = ollamaNativeRuntime;
|
|
4372
|
-
|
|
4373
|
-
// src/domains/providers/runtimes/local-native/sglang.ts
|
|
4374
|
-
init_esm_shims();
|
|
4375
|
-
var defaultCapabilities23 = {
|
|
4376
|
-
chat: true,
|
|
4377
|
-
tools: true,
|
|
4378
|
-
toolCallFormat: "openai",
|
|
4379
|
-
reasoning: false,
|
|
4380
|
-
vision: false,
|
|
4381
|
-
audio: false,
|
|
4382
|
-
embeddings: false,
|
|
4383
|
-
rerank: false,
|
|
4384
|
-
fim: false,
|
|
4385
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4386
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4387
|
-
};
|
|
4388
|
-
var sglang_default = makeOpenAICompatRuntime({
|
|
4389
|
-
id: "sglang",
|
|
4390
|
-
displayName: "SGLang",
|
|
4391
|
-
provider: "sglang",
|
|
4392
|
-
auth: "api-key",
|
|
4393
|
-
tier: "local-native",
|
|
4394
|
-
defaultCapabilities: defaultCapabilities23
|
|
4395
|
-
});
|
|
4396
|
-
|
|
4397
|
-
// src/domains/providers/runtimes/local-native/vllm.ts
|
|
4398
|
-
init_esm_shims();
|
|
4399
|
-
var defaultCapabilities24 = {
|
|
4400
|
-
chat: true,
|
|
4401
|
-
tools: true,
|
|
4402
|
-
toolCallFormat: "openai",
|
|
4403
|
-
reasoning: false,
|
|
4404
|
-
vision: false,
|
|
4405
|
-
audio: false,
|
|
4406
|
-
embeddings: false,
|
|
4407
|
-
rerank: false,
|
|
4408
|
-
fim: false,
|
|
4409
|
-
contextWindow: CLIO_MIN_CONTEXT_WINDOW,
|
|
4410
|
-
maxTokens: CLIO_MIN_MAX_OUTPUT_TOKENS
|
|
4411
|
-
};
|
|
4412
|
-
var vllm_default = makeOpenAICompatRuntime({
|
|
4413
|
-
id: "vllm",
|
|
4414
|
-
displayName: "vLLM",
|
|
4415
|
-
provider: "vllm",
|
|
4416
|
-
auth: "api-key",
|
|
4417
|
-
tier: "local-native",
|
|
4418
|
-
defaultCapabilities: defaultCapabilities24,
|
|
4419
|
-
healthPath: "/health"
|
|
4420
|
-
});
|
|
4421
|
-
|
|
4422
|
-
// src/domains/providers/runtimes/builtins.ts
|
|
4423
|
-
var BUILTIN_RUNTIMES = [
|
|
4424
|
-
alcf_default,
|
|
4425
|
-
anthropic_default,
|
|
4426
|
-
anthropic_max_default,
|
|
4427
|
-
bedrock_default,
|
|
4428
|
-
deepseek_default,
|
|
4429
|
-
google_default,
|
|
4430
|
-
groq_default,
|
|
4431
|
-
mistral_default,
|
|
4432
|
-
openai_default,
|
|
4433
|
-
openai_codex_default,
|
|
4434
|
-
openrouter_default,
|
|
4435
|
-
lemonade_anthropic_default,
|
|
4436
|
-
lemonade_openai_default,
|
|
4437
|
-
llamacpp_default,
|
|
4438
|
-
llamacpp_anthropic_default,
|
|
4439
|
-
llamacpp_completion_default,
|
|
4440
|
-
llamacpp_embed_default,
|
|
4441
|
-
llamacpp_rerank_default,
|
|
4442
|
-
lmstudio_default,
|
|
4443
|
-
ollama_native_default,
|
|
4444
|
-
anthropic_compat_default,
|
|
4445
|
-
openai_compat_default,
|
|
4446
|
-
sglang_default,
|
|
4447
|
-
vllm_default,
|
|
4448
|
-
claude_code_default,
|
|
4449
|
-
claude_sdk_default,
|
|
4450
|
-
antigravity_code_default
|
|
4451
|
-
];
|
|
4452
|
-
function registerBuiltinRuntimes(registry) {
|
|
4453
|
-
for (const desc of BUILTIN_RUNTIMES) {
|
|
4454
|
-
if (registry.get(desc.id) !== null) continue;
|
|
4455
|
-
registry.register(desc);
|
|
4456
|
-
}
|
|
1262
|
+
function resolveModelRuntimeCapabilitiesForModel(model, configuredThinkingLevel) {
|
|
1263
|
+
const metadata2 = model.clio;
|
|
1264
|
+
const caps = capabilitiesFromModel(model);
|
|
1265
|
+
return resolveModelRuntimeCapabilities({
|
|
1266
|
+
targetId: metadata2?.targetId ?? null,
|
|
1267
|
+
runtimeId: metadata2?.runtimeId ?? model.provider,
|
|
1268
|
+
apiFamily: model.api,
|
|
1269
|
+
modelId: model.id,
|
|
1270
|
+
capabilities: caps,
|
|
1271
|
+
...thinkingHintsForModel(model),
|
|
1272
|
+
...metadata2?.quirks ? { quirks: metadata2.quirks } : {},
|
|
1273
|
+
kbHit: metadata2?.family ? {
|
|
1274
|
+
matchKind: "family",
|
|
1275
|
+
entry: {
|
|
1276
|
+
family: metadata2.family,
|
|
1277
|
+
matchPatterns: [metadata2.family],
|
|
1278
|
+
capabilities: {}
|
|
1279
|
+
}
|
|
1280
|
+
} : null,
|
|
1281
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1282
|
+
});
|
|
1283
|
+
}
|
|
1284
|
+
function resolveTargetRuntimeCapabilities(target, runtime, wireModelId, capabilities, knowledgeBase, configuredThinkingLevel) {
|
|
1285
|
+
const kbHit = knowledgeBase?.lookup(wireModelId) ?? null;
|
|
1286
|
+
return resolveModelRuntimeCapabilities({
|
|
1287
|
+
targetId: target.id,
|
|
1288
|
+
runtimeId: runtime.id,
|
|
1289
|
+
apiFamily: runtime.apiFamily,
|
|
1290
|
+
modelId: wireModelId,
|
|
1291
|
+
capabilities,
|
|
1292
|
+
kbHit,
|
|
1293
|
+
...thinkingHintsForCatalogModel(runtime.id, wireModelId),
|
|
1294
|
+
...configuredThinkingLevel ? { configuredThinkingLevel } : {}
|
|
1295
|
+
});
|
|
4457
1296
|
}
|
|
4458
1297
|
|
|
4459
1298
|
// src/domains/providers/eligibility.ts
|
|
@@ -4468,140 +1307,6 @@ function isDispatchEligibleRuntime(runtime) {
|
|
|
4468
1307
|
return isTargetEligibleRuntime(runtime);
|
|
4469
1308
|
}
|
|
4470
1309
|
|
|
4471
|
-
// src/domains/providers/support.ts
|
|
4472
|
-
init_esm_shims();
|
|
4473
|
-
var SUMMARY_BY_RUNTIME_ID = {
|
|
4474
|
-
alcf: "ALCF inference gateway (Sophia/Metis) via Globus",
|
|
4475
|
-
anthropic: "Anthropic API",
|
|
4476
|
-
"anthropic-max": "Claude Pro/Max subscription via Anthropic OAuth",
|
|
4477
|
-
bedrock: "Amazon Bedrock",
|
|
4478
|
-
"claude-code": "Claude Code subscription via installed claude CLI",
|
|
4479
|
-
"claude-sdk": "Claude Code subscription via Claude Agent SDK",
|
|
4480
|
-
deepseek: "DeepSeek API",
|
|
4481
|
-
google: "Google Gemini API",
|
|
4482
|
-
groq: "Groq API",
|
|
4483
|
-
mistral: "Mistral API",
|
|
4484
|
-
openai: "OpenAI Platform API",
|
|
4485
|
-
"openai-codex": "ChatGPT Plus/Pro via Codex OAuth",
|
|
4486
|
-
openrouter: "OpenRouter API",
|
|
4487
|
-
"ollama-native": "Ollama native API",
|
|
4488
|
-
lmstudio: "LM Studio chat over OpenAI-compatible REST with native REST model management",
|
|
4489
|
-
llamacpp: "llama.cpp server (auto-detect surface)",
|
|
4490
|
-
"anthropic-compat": "Generic Anthropic-compatible REST",
|
|
4491
|
-
"openai-compat": "Generic OpenAI-compatible REST"
|
|
4492
|
-
};
|
|
4493
|
-
function groupPriority(group) {
|
|
4494
|
-
switch (group) {
|
|
4495
|
-
case "featured":
|
|
4496
|
-
return 0;
|
|
4497
|
-
case "subscription":
|
|
4498
|
-
return 1;
|
|
4499
|
-
case "cloud-api":
|
|
4500
|
-
return 2;
|
|
4501
|
-
case "local-http":
|
|
4502
|
-
return 3;
|
|
4503
|
-
}
|
|
4504
|
-
}
|
|
4505
|
-
function supportGroupLabel(group) {
|
|
4506
|
-
switch (group) {
|
|
4507
|
-
case "featured":
|
|
4508
|
-
return "Featured";
|
|
4509
|
-
case "subscription":
|
|
4510
|
-
return "Subscriptions";
|
|
4511
|
-
case "cloud-api":
|
|
4512
|
-
return "Cloud APIs";
|
|
4513
|
-
case "local-http":
|
|
4514
|
-
return "Local HTTP";
|
|
4515
|
-
}
|
|
4516
|
-
}
|
|
4517
|
-
function classifyGroup(runtime) {
|
|
4518
|
-
if (runtime.id === "openai-codex") return "featured";
|
|
4519
|
-
if (runtime.id === "alcf") return "cloud-api";
|
|
4520
|
-
if (runtime.auth === "oauth" || runtime.auth === "claude-cli") return "subscription";
|
|
4521
|
-
if (catalogProviderForRuntime(runtime.id) || runtime.auth === "api-key" && !runtime.probe) {
|
|
4522
|
-
return "cloud-api";
|
|
4523
|
-
}
|
|
4524
|
-
return "local-http";
|
|
4525
|
-
}
|
|
4526
|
-
function knownModelsFor(runtimeId, runtime) {
|
|
4527
|
-
const catalogModels2 = listCatalogModelsForRuntime(runtimeId);
|
|
4528
|
-
if (catalogModels2.length === 0) return runtime?.knownModels ? [...runtime.knownModels] : [];
|
|
4529
|
-
return catalogModels2.map((model) => model.id);
|
|
4530
|
-
}
|
|
4531
|
-
function listKnownModelsForRuntime(runtimeId) {
|
|
4532
|
-
return knownModelsFor(runtimeId, getRuntimeIfRegistered(runtimeId));
|
|
4533
|
-
}
|
|
4534
|
-
function getRuntimeIfRegistered(runtimeId) {
|
|
4535
|
-
try {
|
|
4536
|
-
return getRuntimeRegistry().get(runtimeId);
|
|
4537
|
-
} catch {
|
|
4538
|
-
return null;
|
|
4539
|
-
}
|
|
4540
|
-
}
|
|
4541
|
-
function runtimeModelListSource(runtime) {
|
|
4542
|
-
if (runtime.knownModels && runtime.knownModels.length > 0) return "runtime";
|
|
4543
|
-
if (listCatalogModelsForRuntime(runtime.id).length > 0) return "catalog";
|
|
4544
|
-
return "none";
|
|
4545
|
-
}
|
|
4546
|
-
function describeRuntimeModels(entry, sample) {
|
|
4547
|
-
if (entry.modelSource === "catalog") return `${entry.modelHints.length} in catalog`;
|
|
4548
|
-
if (entry.modelHints.length === 0) return "-";
|
|
4549
|
-
return entry.modelHints.slice(0, sample).join(", ");
|
|
4550
|
-
}
|
|
4551
|
-
function buildProviderSupportEntry(runtime) {
|
|
4552
|
-
const modelHints = knownModelsFor(runtime.id, runtime);
|
|
4553
|
-
const modelSource = runtimeModelListSource(runtime);
|
|
4554
|
-
const defaultModel = modelSource === "catalog" ? void 0 : modelHints[0];
|
|
4555
|
-
return {
|
|
4556
|
-
runtimeId: runtime.id,
|
|
4557
|
-
label: runtime.displayName,
|
|
4558
|
-
group: classifyGroup(runtime),
|
|
4559
|
-
summary: SUMMARY_BY_RUNTIME_ID[runtime.id] ?? runtime.displayName,
|
|
4560
|
-
...defaultModel ? { defaultModel } : {},
|
|
4561
|
-
modelHints,
|
|
4562
|
-
modelSource,
|
|
4563
|
-
featured: runtime.id === "openai-codex",
|
|
4564
|
-
connectable: runtime.auth === "oauth" || runtime.auth === "api-key",
|
|
4565
|
-
supportsCustomUrl: runtime.kind === "http" && (classifyGroup(runtime) === "local-http" || runtime.id === "openai-compat" || runtime.id === "anthropic-compat" || runtime.id === "alcf")
|
|
4566
|
-
};
|
|
4567
|
-
}
|
|
4568
|
-
function compareProviderSupportEntries(a, b) {
|
|
4569
|
-
return groupPriority(a.group) - groupPriority(b.group) || (a.featured === b.featured ? 0 : a.featured ? -1 : 1) || a.label.localeCompare(b.label) || a.runtimeId.localeCompare(b.runtimeId);
|
|
4570
|
-
}
|
|
4571
|
-
function listProviderSupportEntries(runtimes, options = {}) {
|
|
4572
|
-
const filtered = options.includeHidden ? runtimes : runtimes.filter((runtime) => runtime.hidden !== true);
|
|
4573
|
-
return filtered.map((runtime) => buildProviderSupportEntry(runtime)).sort(compareProviderSupportEntries);
|
|
4574
|
-
}
|
|
4575
|
-
function configuredTargetsForRuntime(settings, runtimeId) {
|
|
4576
|
-
const canonical = getRuntimeIfRegistered(runtimeId)?.id ?? runtimeId;
|
|
4577
|
-
return settings.targets.filter(
|
|
4578
|
-
(target) => (getRuntimeIfRegistered(target.runtime)?.id ?? target.runtime) === canonical
|
|
4579
|
-
);
|
|
4580
|
-
}
|
|
4581
|
-
function resolveProviderReference(input, settings, getRuntime) {
|
|
4582
|
-
const trimmed = input.trim();
|
|
4583
|
-
if (trimmed.length === 0) return null;
|
|
4584
|
-
const target = settings.targets.find((entry) => entry.id === trimmed) ?? null;
|
|
4585
|
-
if (target) {
|
|
4586
|
-
const runtime2 = getRuntime(target.runtime);
|
|
4587
|
-
if (!runtime2) return null;
|
|
4588
|
-
return {
|
|
4589
|
-
input: trimmed,
|
|
4590
|
-
target,
|
|
4591
|
-
runtime: runtime2,
|
|
4592
|
-
authTarget: resolveAuthTarget(target, runtime2)
|
|
4593
|
-
};
|
|
4594
|
-
}
|
|
4595
|
-
const runtime = getRuntime(trimmed);
|
|
4596
|
-
if (!runtime) return null;
|
|
4597
|
-
return {
|
|
4598
|
-
input: trimmed,
|
|
4599
|
-
target: null,
|
|
4600
|
-
runtime,
|
|
4601
|
-
authTarget: resolveRuntimeAuthTarget(runtime)
|
|
4602
|
-
};
|
|
4603
|
-
}
|
|
4604
|
-
|
|
4605
1310
|
// src/domains/providers/model-discovery.ts
|
|
4606
1311
|
init_esm_shims();
|
|
4607
1312
|
function uniqueModels(ids) {
|
|
@@ -4703,24 +1408,24 @@ function diagnostic(severity, code, message) {
|
|
|
4703
1408
|
function hasError(diagnostics) {
|
|
4704
1409
|
return diagnostics.some((entry) => entry.severity === "error");
|
|
4705
1410
|
}
|
|
4706
|
-
function
|
|
1411
|
+
function statusFor(providers, target, runtime, _wireModelId) {
|
|
4707
1412
|
const existing = providers.list().find((entry) => entry.target.id === target.id);
|
|
4708
1413
|
if (existing) return existing;
|
|
4709
|
-
const
|
|
1414
|
+
const capabilities = { ...runtime.defaultCapabilities, ...target.capabilities ?? {} };
|
|
4710
1415
|
return {
|
|
4711
1416
|
target,
|
|
4712
1417
|
runtime,
|
|
4713
1418
|
available: true,
|
|
4714
1419
|
reason: "synthetic-status",
|
|
4715
1420
|
health: { status: "unknown", lastCheckAt: null, lastError: null, latencyMs: null },
|
|
4716
|
-
capabilities
|
|
1421
|
+
capabilities,
|
|
4717
1422
|
probeCapabilities: null,
|
|
4718
1423
|
probeModelId: null,
|
|
4719
1424
|
discoveredModels: runtime.knownModels ?? []
|
|
4720
1425
|
};
|
|
4721
1426
|
}
|
|
4722
|
-
function requiredCapabilitySupported(
|
|
4723
|
-
const value =
|
|
1427
|
+
function requiredCapabilitySupported(capabilities, name) {
|
|
1428
|
+
const value = capabilities[name];
|
|
4724
1429
|
return value !== void 0 && value !== false && value !== 0 && value !== "";
|
|
4725
1430
|
}
|
|
4726
1431
|
function streamingDecision(runtime) {
|
|
@@ -4730,18 +1435,18 @@ function runtimeSupportsUse(runtime, use) {
|
|
|
4730
1435
|
if (use === "dispatch") return isDispatchEligibleRuntime(runtime);
|
|
4731
1436
|
return isOrchestratorEligibleRuntime(runtime);
|
|
4732
1437
|
}
|
|
4733
|
-
function capabilityDecisions(runtime,
|
|
1438
|
+
function capabilityDecisions(runtime, capabilities) {
|
|
4734
1439
|
return {
|
|
4735
|
-
chat:
|
|
4736
|
-
tools:
|
|
4737
|
-
reasoning:
|
|
4738
|
-
vision:
|
|
1440
|
+
chat: capabilities.chat,
|
|
1441
|
+
tools: capabilities.tools,
|
|
1442
|
+
reasoning: capabilities.reasoning,
|
|
1443
|
+
vision: capabilities.vision,
|
|
4739
1444
|
streaming: streamingDecision(runtime),
|
|
4740
|
-
contextWindow:
|
|
4741
|
-
maxTokens:
|
|
1445
|
+
contextWindow: capabilities.contextWindow,
|
|
1446
|
+
maxTokens: capabilities.maxTokens
|
|
4742
1447
|
};
|
|
4743
1448
|
}
|
|
4744
|
-
function appendCapabilityDiagnostics(diagnostics, input,
|
|
1449
|
+
function appendCapabilityDiagnostics(diagnostics, input, capabilities, decisions, targetId) {
|
|
4745
1450
|
if (!decisions.chat) {
|
|
4746
1451
|
diagnostics.push(diagnostic("error", "chat-unsupported", `target '${targetId}' does not advertise chat support`));
|
|
4747
1452
|
}
|
|
@@ -4757,7 +1462,7 @@ function appendCapabilityDiagnostics(diagnostics, input, capabilities2, decision
|
|
|
4757
1462
|
);
|
|
4758
1463
|
}
|
|
4759
1464
|
for (const capability of input.requiredCapabilities ?? []) {
|
|
4760
|
-
if (!requiredCapabilitySupported(
|
|
1465
|
+
if (!requiredCapabilitySupported(capabilities, capability)) {
|
|
4761
1466
|
diagnostics.push(
|
|
4762
1467
|
diagnostic(
|
|
4763
1468
|
"error",
|
|
@@ -4874,14 +1579,14 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4874
1579
|
diagnostics: [diagnostic("error", "model-not-configured", `target '${targetId}' has no model configured`)]
|
|
4875
1580
|
};
|
|
4876
1581
|
}
|
|
4877
|
-
const status =
|
|
1582
|
+
const status = statusFor(providers, target, runtime, wireModelId);
|
|
4878
1583
|
const unknownModel = unknownModelDiagnostic(target, status, wireModelId);
|
|
4879
1584
|
if (unknownModel) diagnostics.push(unknownModel);
|
|
4880
1585
|
const requestedThinkingLevel = input.requestedThinkingLevel ?? "off";
|
|
4881
1586
|
const capabilityResolution = modelCapabilitiesFor(providers, status, wireModelId);
|
|
4882
|
-
const
|
|
1587
|
+
const capabilities = { ...capabilityResolution.capabilities };
|
|
4883
1588
|
const probedContextWindow = probeCapabilitiesForModel(status, wireModelId)?.contextWindow ?? null;
|
|
4884
|
-
const loadedContextWindow = loadedContextWindowForModel(status, wireModelId);
|
|
1589
|
+
const loadedContextWindow = loadedContextWindowForModel(status, wireModelId) ?? input.knownLoadedContextWindow ?? null;
|
|
4885
1590
|
const contextWindowDetails = resolveContextWindowDetails(
|
|
4886
1591
|
target,
|
|
4887
1592
|
runtime,
|
|
@@ -4892,7 +1597,7 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4892
1597
|
void 0,
|
|
4893
1598
|
contextSlotsForModel(status, wireModelId)
|
|
4894
1599
|
);
|
|
4895
|
-
|
|
1600
|
+
capabilities.contextWindow = contextWindowDetails.effectiveContextWindow;
|
|
4896
1601
|
if (contextWindowDetails.warning) {
|
|
4897
1602
|
diagnostics.push(diagnostic("warning", "context-window-low", contextWindowDetails.warning));
|
|
4898
1603
|
}
|
|
@@ -4903,12 +1608,12 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4903
1608
|
target,
|
|
4904
1609
|
runtime,
|
|
4905
1610
|
wireModelId,
|
|
4906
|
-
|
|
1611
|
+
capabilities,
|
|
4907
1612
|
providers.knowledgeBase,
|
|
4908
1613
|
requestedThinkingLevel
|
|
4909
1614
|
);
|
|
4910
|
-
const decisions = capabilityDecisions(runtime,
|
|
4911
|
-
appendCapabilityDiagnostics(diagnostics, input,
|
|
1615
|
+
const decisions = capabilityDecisions(runtime, capabilities);
|
|
1616
|
+
appendCapabilityDiagnostics(diagnostics, input, capabilities, decisions, targetId);
|
|
4912
1617
|
appendThinkingDiagnostics(diagnostics, modelRuntime, requestedThinkingLevel);
|
|
4913
1618
|
if (hasError(diagnostics)) return { ok: false, diagnostics };
|
|
4914
1619
|
const resolved = {
|
|
@@ -4924,7 +1629,7 @@ function resolveRuntimeTarget(providers, input) {
|
|
|
4924
1629
|
costProvenance: resolveCostProvenance(target, runtime.id, wireModelId),
|
|
4925
1630
|
requestedThinkingLevel,
|
|
4926
1631
|
effectiveThinkingLevel: modelRuntime.thinking.effectiveLevel,
|
|
4927
|
-
capabilities
|
|
1632
|
+
capabilities,
|
|
4928
1633
|
capabilityDecisions: decisions,
|
|
4929
1634
|
modelRuntime,
|
|
4930
1635
|
modelReasoningAuthoritative: capabilityResolution.reasoningAuthoritative,
|
|
@@ -4960,7 +1665,7 @@ function refineRuntimeTargetWithModelHints(target, model, knowledgeBase) {
|
|
|
4960
1665
|
const modelHintContextWindow = nonNegativeFiniteNumber(hintRecord?.contextWindow);
|
|
4961
1666
|
const windowHintDiffers = modelHintContextWindow !== void 0 && modelHintContextWindow > 0 && modelHintContextWindow !== target.contextWindowDetails.effectiveContextWindow;
|
|
4962
1667
|
if (Object.keys(patch).length === 0 && !windowHintDiffers) return target;
|
|
4963
|
-
const
|
|
1668
|
+
const capabilities = { ...target.capabilities, ...patch };
|
|
4964
1669
|
const contextWindowDetails = resolveContextWindowDetails(
|
|
4965
1670
|
target.target,
|
|
4966
1671
|
target.runtime,
|
|
@@ -4974,22 +1679,22 @@ function refineRuntimeTargetWithModelHints(target, model, knowledgeBase) {
|
|
|
4974
1679
|
modelHintContextWindow,
|
|
4975
1680
|
target.contextWindowDetails.contextWindowSlots
|
|
4976
1681
|
);
|
|
4977
|
-
|
|
1682
|
+
capabilities.contextWindow = contextWindowDetails.effectiveContextWindow;
|
|
4978
1683
|
const modelRuntime = resolveModelRuntimeCapabilities({
|
|
4979
1684
|
targetId: target.targetId,
|
|
4980
1685
|
runtimeId: target.runtimeId,
|
|
4981
1686
|
apiFamily: target.apiFamily,
|
|
4982
1687
|
modelId: target.wireModelId,
|
|
4983
|
-
capabilities
|
|
1688
|
+
capabilities,
|
|
4984
1689
|
...target.modelRuntime.quirks ? { quirks: target.modelRuntime.quirks } : {},
|
|
4985
1690
|
configuredThinkingLevel: target.requestedThinkingLevel
|
|
4986
1691
|
});
|
|
4987
|
-
const decisions = capabilityDecisions(target.runtime,
|
|
1692
|
+
const decisions = capabilityDecisions(target.runtime, capabilities);
|
|
4988
1693
|
const diagnostics = withoutStaleRuntimeDiagnostics(target.diagnostics, decisions);
|
|
4989
1694
|
appendThinkingDiagnostics(diagnostics, modelRuntime, target.requestedThinkingLevel);
|
|
4990
1695
|
return {
|
|
4991
1696
|
...target,
|
|
4992
|
-
capabilities
|
|
1697
|
+
capabilities,
|
|
4993
1698
|
capabilityDecisions: decisions,
|
|
4994
1699
|
modelRuntime,
|
|
4995
1700
|
effectiveThinkingLevel: modelRuntime.thinking.effectiveLevel,
|
|
@@ -5109,6 +1814,12 @@ function resolveContextWindowDetails(target, runtime, wireModelId, knowledgeBase
|
|
|
5109
1814
|
};
|
|
5110
1815
|
}
|
|
5111
1816
|
|
|
1817
|
+
// src/domains/providers/types/cost-provenance.ts
|
|
1818
|
+
init_esm_shims();
|
|
1819
|
+
function normalizeCostProvenance(value) {
|
|
1820
|
+
return value ?? "unknown";
|
|
1821
|
+
}
|
|
1822
|
+
|
|
5112
1823
|
// src/domains/providers/index.ts
|
|
5113
1824
|
init_esm_shims();
|
|
5114
1825
|
|
|
@@ -5126,6 +1837,18 @@ import {
|
|
|
5126
1837
|
Ollama
|
|
5127
1838
|
} from "ollama";
|
|
5128
1839
|
|
|
1840
|
+
// src/core/residency-target-key.ts
|
|
1841
|
+
init_esm_shims();
|
|
1842
|
+
var RESIDENCY_RUNTIME_IDS = /* @__PURE__ */ new Set([
|
|
1843
|
+
"llamacpp",
|
|
1844
|
+
"lmstudio",
|
|
1845
|
+
"ollama-native"
|
|
1846
|
+
]);
|
|
1847
|
+
function residencyTargetKey(runtimeId, baseUrl) {
|
|
1848
|
+
if (!RESIDENCY_RUNTIME_IDS.has(runtimeId) || typeof baseUrl !== "string") return null;
|
|
1849
|
+
return canonicalEndpointUrl(baseUrl) ?? baseUrl;
|
|
1850
|
+
}
|
|
1851
|
+
|
|
5129
1852
|
// src/engine/gemma-channel-filter.ts
|
|
5130
1853
|
init_esm_shims();
|
|
5131
1854
|
import { createAssistantMessageEventStream } from "@earendil-works/pi-ai";
|
|
@@ -5395,7 +2118,7 @@ async function reconcileOllamaResidency(model, headers) {
|
|
|
5395
2118
|
if (!baseUrl) return;
|
|
5396
2119
|
const metadata2 = model.clio;
|
|
5397
2120
|
const adapter = {
|
|
5398
|
-
targetKey:
|
|
2121
|
+
targetKey: residencyTargetKey("ollama-native", baseUrl),
|
|
5399
2122
|
targetId: ollamaTargetId(model),
|
|
5400
2123
|
runtimeId: "ollama-native",
|
|
5401
2124
|
keepModelId: model.id,
|
|
@@ -5766,7 +2489,7 @@ var ollamaNativeApiProvider = {
|
|
|
5766
2489
|
|
|
5767
2490
|
// src/engine/apis/openai-completions.ts
|
|
5768
2491
|
init_esm_shims();
|
|
5769
|
-
import { TextDecoder
|
|
2492
|
+
import { TextDecoder } from "node:util";
|
|
5770
2493
|
import {
|
|
5771
2494
|
createAssistantMessageEventStream as createAssistantMessageEventStream3
|
|
5772
2495
|
} from "@earendil-works/pi-ai";
|
|
@@ -5893,30 +2616,30 @@ function harmonyPrefixTailLength(value) {
|
|
|
5893
2616
|
|
|
5894
2617
|
// src/engine/apis/llamacpp-residency.ts
|
|
5895
2618
|
init_esm_shims();
|
|
5896
|
-
import { performance
|
|
2619
|
+
import { performance } from "node:perf_hooks";
|
|
5897
2620
|
var LOAD_TIMEOUT_MS = 12e4;
|
|
5898
2621
|
var POLL_INTERVAL_MS = 500;
|
|
5899
2622
|
function isResidentState(state) {
|
|
5900
2623
|
return state === "loaded" || state === "loading" || state === "sleeping";
|
|
5901
2624
|
}
|
|
5902
|
-
function
|
|
2625
|
+
function isRecord(value) {
|
|
5903
2626
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
5904
2627
|
}
|
|
5905
2628
|
function routerModelState(entry) {
|
|
5906
2629
|
const status = entry.status;
|
|
5907
|
-
const state =
|
|
2630
|
+
const state = isRecord(status) ? status.value : status;
|
|
5908
2631
|
if (state === "loaded" || state === "loading" || state === "sleeping" || state === "unloaded" || state === "failed") {
|
|
5909
2632
|
return state;
|
|
5910
2633
|
}
|
|
5911
2634
|
return "unknown";
|
|
5912
2635
|
}
|
|
5913
2636
|
function parseRouterModels(payload) {
|
|
5914
|
-
if (!
|
|
2637
|
+
if (!isRecord(payload)) return [];
|
|
5915
2638
|
const data = payload.data;
|
|
5916
2639
|
if (!Array.isArray(data)) return [];
|
|
5917
2640
|
const models = [];
|
|
5918
2641
|
for (const entry of data) {
|
|
5919
|
-
if (!
|
|
2642
|
+
if (!isRecord(entry) || typeof entry.id !== "string") continue;
|
|
5920
2643
|
const tags = Array.isArray(entry.tags) ? entry.tags.filter((tag) => typeof tag === "string") : [];
|
|
5921
2644
|
models.push({ id: entry.id, state: routerModelState(entry), tags });
|
|
5922
2645
|
}
|
|
@@ -5926,14 +2649,14 @@ function residentModel(model) {
|
|
|
5926
2649
|
return isResidentState(model.state);
|
|
5927
2650
|
}
|
|
5928
2651
|
function parseLlamaCppRouterProps(payload) {
|
|
5929
|
-
if (!
|
|
2652
|
+
if (!isRecord(payload)) return {};
|
|
5930
2653
|
const maxInstances = payload.max_instances;
|
|
5931
2654
|
return typeof maxInstances === "number" && Number.isFinite(maxInstances) && maxInstances > 0 ? { maxInstances: Math.floor(maxInstances) } : {};
|
|
5932
2655
|
}
|
|
5933
2656
|
function rootUrl(baseUrl) {
|
|
5934
2657
|
return baseUrl.replace(/\/+$/, "").replace(/\/v1$/, "");
|
|
5935
2658
|
}
|
|
5936
|
-
function
|
|
2659
|
+
function modelsUrl(baseUrl) {
|
|
5937
2660
|
return `${baseUrl.replace(/\/+$/, "")}/models`;
|
|
5938
2661
|
}
|
|
5939
2662
|
function propsUrl(baseUrl) {
|
|
@@ -5957,25 +2680,65 @@ async function fetchRouterProps(input, fetchImpl) {
|
|
|
5957
2680
|
}
|
|
5958
2681
|
}
|
|
5959
2682
|
async function fetchRouterModels(input, fetchImpl) {
|
|
5960
|
-
const response = await fetchImpl(
|
|
2683
|
+
const response = await fetchImpl(modelsUrl(input.baseUrl), {
|
|
5961
2684
|
signal: AbortSignal.timeout(input.timeoutMs ?? 1500)
|
|
5962
2685
|
});
|
|
5963
2686
|
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
|
5964
2687
|
return parseRouterModels(await response.json());
|
|
5965
2688
|
}
|
|
5966
|
-
|
|
5967
|
-
|
|
2689
|
+
var RouterModelPostError = class extends Error {
|
|
2690
|
+
constructor(url, status, responseBody) {
|
|
2691
|
+
super(`llama.cpp router rejected ${url}: HTTP ${status}; response: ${responseBody}`);
|
|
2692
|
+
this.url = url;
|
|
2693
|
+
this.status = status;
|
|
2694
|
+
this.responseBody = responseBody;
|
|
2695
|
+
this.name = "RouterModelPostError";
|
|
2696
|
+
}
|
|
2697
|
+
url;
|
|
2698
|
+
status;
|
|
2699
|
+
responseBody;
|
|
2700
|
+
};
|
|
2701
|
+
async function routerErrorBody(response) {
|
|
2702
|
+
try {
|
|
2703
|
+
const body = (await response.text()).trim();
|
|
2704
|
+
return body.length > 0 ? body : "<empty body>";
|
|
2705
|
+
} catch (error) {
|
|
2706
|
+
return `<unreadable body: ${error instanceof Error ? error.message : String(error)}>`;
|
|
2707
|
+
}
|
|
2708
|
+
}
|
|
2709
|
+
async function postRouterModel(fetchImpl, url, modelId) {
|
|
2710
|
+
const response = await fetchImpl(url, {
|
|
5968
2711
|
method: "POST",
|
|
5969
2712
|
headers: { "content-type": "application/json" },
|
|
5970
2713
|
body: JSON.stringify({ model: modelId }),
|
|
5971
2714
|
signal: AbortSignal.timeout(LOAD_TIMEOUT_MS)
|
|
5972
2715
|
});
|
|
5973
2716
|
if (response.ok) return;
|
|
5974
|
-
throw new
|
|
2717
|
+
throw new RouterModelPostError(url, response.status, await routerErrorBody(response));
|
|
2718
|
+
}
|
|
2719
|
+
function alreadyRunningResponse(body) {
|
|
2720
|
+
return body.toLowerCase().includes("already running");
|
|
2721
|
+
}
|
|
2722
|
+
async function postRouterLoad(input, fetchImpl, modelId) {
|
|
2723
|
+
try {
|
|
2724
|
+
await postRouterModel(fetchImpl, loadUrl(input.baseUrl), modelId);
|
|
2725
|
+
} catch (error) {
|
|
2726
|
+
if (!(error instanceof RouterModelPostError)) throw error;
|
|
2727
|
+
let model;
|
|
2728
|
+
try {
|
|
2729
|
+
model = (await fetchRouterModels(input, fetchImpl)).find((entry) => entry.id === modelId);
|
|
2730
|
+
} catch {
|
|
2731
|
+
if (alreadyRunningResponse(error.responseBody)) return;
|
|
2732
|
+
throw error;
|
|
2733
|
+
}
|
|
2734
|
+
if (alreadyRunningResponse(error.responseBody)) return;
|
|
2735
|
+
if (model !== void 0 && model.state !== "unloaded" && model.state !== "failed") return;
|
|
2736
|
+
throw error;
|
|
2737
|
+
}
|
|
5975
2738
|
}
|
|
5976
2739
|
async function waitForLoaded(input, fetchImpl, modelId) {
|
|
5977
|
-
const started =
|
|
5978
|
-
while (
|
|
2740
|
+
const started = performance.now();
|
|
2741
|
+
while (performance.now() - started < LOAD_TIMEOUT_MS) {
|
|
5979
2742
|
const models = await fetchRouterModels(input, fetchImpl);
|
|
5980
2743
|
const model = models.find((entry) => entry.id === modelId);
|
|
5981
2744
|
if (model?.state === "loaded" || model?.state === "sleeping") return;
|
|
@@ -5986,7 +2749,7 @@ async function waitForLoaded(input, fetchImpl, modelId) {
|
|
|
5986
2749
|
}
|
|
5987
2750
|
async function ensureModelLoaded(input, fetchImpl, modelId, knownState) {
|
|
5988
2751
|
if (knownState === "loaded" || knownState === "sleeping") return;
|
|
5989
|
-
if (knownState !== "loading") await
|
|
2752
|
+
if (knownState !== "loading") await postRouterLoad(input, fetchImpl, modelId);
|
|
5990
2753
|
await waitForLoaded(input, fetchImpl, modelId);
|
|
5991
2754
|
}
|
|
5992
2755
|
async function restoreDisplacedPinned(input, fetchImpl, before) {
|
|
@@ -6013,7 +2776,7 @@ async function ensureLlamaCppResidency(input) {
|
|
|
6013
2776
|
const fetchImpl = input.fetchImpl ?? fetch;
|
|
6014
2777
|
let snapshot = [];
|
|
6015
2778
|
const adapter = {
|
|
6016
|
-
targetKey:
|
|
2779
|
+
targetKey: residencyTargetKey("llamacpp", input.baseUrl),
|
|
6017
2780
|
targetId: input.targetId,
|
|
6018
2781
|
runtimeId: input.runtimeId,
|
|
6019
2782
|
keepModelId: input.keepModelId,
|
|
@@ -6112,7 +2875,7 @@ function targetForModel(model) {
|
|
|
6112
2875
|
...info.lmstudio ? { lmstudio: info.lmstudio } : {}
|
|
6113
2876
|
};
|
|
6114
2877
|
}
|
|
6115
|
-
function
|
|
2878
|
+
function resolveModel(catalog, id) {
|
|
6116
2879
|
for (const model of catalog.models) {
|
|
6117
2880
|
if (model.key === id) return { model, ...model.loadedInstances[0] ? { instance: model.loadedInstances[0] } : {} };
|
|
6118
2881
|
const instance = model.loadedInstances.find((entry) => entry.id === id);
|
|
@@ -6207,13 +2970,13 @@ async function ensureLmStudioResidency(model, options = {}) {
|
|
|
6207
2970
|
}
|
|
6208
2971
|
if (!load || Object.keys(load).length === 0 || resolution.instance) return resolution.wireModelId;
|
|
6209
2972
|
if (catalog.tier !== "0.4+") return resolution.wireModelId;
|
|
6210
|
-
const selected =
|
|
2973
|
+
const selected = resolveModel(catalog, model.id);
|
|
6211
2974
|
const modelKey = selected?.model.key ?? model.id;
|
|
6212
2975
|
let instances = catalog.models.flatMap(
|
|
6213
2976
|
(entry) => entry.loadedInstances.map((instance) => ({ modelKey: entry.key, identifier: instance.id, instance }))
|
|
6214
2977
|
);
|
|
6215
2978
|
const managed = residencyManagedFor(info.lifecycle);
|
|
6216
|
-
const targetKey =
|
|
2979
|
+
const targetKey = residencyTargetKey("lmstudio", target.url ?? model.baseUrl);
|
|
6217
2980
|
const contextLength = target.lmstudio?.load?.contextLength;
|
|
6218
2981
|
const plan = await reconcileResidency({
|
|
6219
2982
|
targetKey,
|
|
@@ -6256,17 +3019,36 @@ async function ensureLmStudioResidency(model, options = {}) {
|
|
|
6256
3019
|
});
|
|
6257
3020
|
}
|
|
6258
3021
|
}
|
|
3022
|
+
const loadAndReport = async () => {
|
|
3023
|
+
const instanceId = await loadOwnedInstance(target, targetKey, body, options.apiKey, options.signal);
|
|
3024
|
+
emitResidencyMutation({
|
|
3025
|
+
targetKey,
|
|
3026
|
+
targetId: info.targetId,
|
|
3027
|
+
runtimeId: "lmstudio",
|
|
3028
|
+
model: modelKey,
|
|
3029
|
+
operation: "load"
|
|
3030
|
+
});
|
|
3031
|
+
return instanceId ?? model.id;
|
|
3032
|
+
};
|
|
6259
3033
|
try {
|
|
6260
|
-
return await
|
|
3034
|
+
return await loadAndReport();
|
|
6261
3035
|
} catch (error) {
|
|
6262
3036
|
if (plan.decision === "observe" || plan.fallbackEvict.length === 0) throw error;
|
|
6263
3037
|
return withResidencyLock(targetKey, async () => {
|
|
6264
3038
|
for (const candidate of plan.fallbackEvict) {
|
|
6265
3039
|
for (const entry of instances.filter((resident) => resident.modelKey === candidate.modelId)) {
|
|
6266
|
-
await unloadOwnedInstance(target, targetKey, entry.identifier, options.apiKey, options.signal)
|
|
3040
|
+
if (await unloadOwnedInstance(target, targetKey, entry.identifier, options.apiKey, options.signal)) {
|
|
3041
|
+
emitResidencyMutation({
|
|
3042
|
+
targetKey,
|
|
3043
|
+
targetId: info.targetId,
|
|
3044
|
+
runtimeId: "lmstudio",
|
|
3045
|
+
model: candidate.modelId,
|
|
3046
|
+
operation: "evict"
|
|
3047
|
+
});
|
|
3048
|
+
}
|
|
6267
3049
|
}
|
|
6268
3050
|
}
|
|
6269
|
-
return
|
|
3051
|
+
return loadAndReport();
|
|
6270
3052
|
});
|
|
6271
3053
|
}
|
|
6272
3054
|
}
|
|
@@ -6288,8 +3070,36 @@ function pickSamplingProfile2(quirks, thinkingActive) {
|
|
|
6288
3070
|
function isPlainRecord(value) {
|
|
6289
3071
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
6290
3072
|
}
|
|
6291
|
-
function
|
|
6292
|
-
|
|
3073
|
+
function nonnegativeFiniteNumber(value) {
|
|
3074
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
|
|
3075
|
+
}
|
|
3076
|
+
function backendCompletionTimings(value, source) {
|
|
3077
|
+
if (!isPlainRecord(value)) return null;
|
|
3078
|
+
const promptN = nonnegativeFiniteNumber(value.prompt_n);
|
|
3079
|
+
const predictedN = nonnegativeFiniteNumber(value.predicted_n);
|
|
3080
|
+
const promptMs = nonnegativeFiniteNumber(value.prompt_ms);
|
|
3081
|
+
const predictedMs = nonnegativeFiniteNumber(value.predicted_ms);
|
|
3082
|
+
if (promptN === null || predictedN === null || promptMs === null || predictedMs === null) return null;
|
|
3083
|
+
const cacheN = nonnegativeFiniteNumber(value.cache_n);
|
|
3084
|
+
return {
|
|
3085
|
+
promptTokens: promptN + (cacheN ?? 0),
|
|
3086
|
+
cachedTokens: cacheN,
|
|
3087
|
+
predictedTokens: predictedN,
|
|
3088
|
+
promptMs,
|
|
3089
|
+
predictedMs,
|
|
3090
|
+
source
|
|
3091
|
+
};
|
|
3092
|
+
}
|
|
3093
|
+
function backendTimingsSourceForModel(model) {
|
|
3094
|
+
const metadata2 = model.clio;
|
|
3095
|
+
if (model.provider === "llamacpp" && metadata2?.runtimeId === "llamacpp") return "llamacpp-timings";
|
|
3096
|
+
if (model.provider === "lmstudio" && metadata2?.runtimeId === "lmstudio") return "lmstudio-timings";
|
|
3097
|
+
return null;
|
|
3098
|
+
}
|
|
3099
|
+
function captureCanStopEarly(capture) {
|
|
3100
|
+
return capture.modelIdDone && capture.backendTimingsSource === null;
|
|
3101
|
+
}
|
|
3102
|
+
function observeResponseMetadataLine(line, capture) {
|
|
6293
3103
|
const normalized = line.endsWith("\r") ? line.slice(0, -1) : line;
|
|
6294
3104
|
if (!normalized.startsWith("data:")) return;
|
|
6295
3105
|
const data = normalized.slice("data:".length).trimStart();
|
|
@@ -6297,23 +3107,29 @@ function observeResponseModelIdLine(line, capture) {
|
|
|
6297
3107
|
try {
|
|
6298
3108
|
const payload = JSON.parse(data);
|
|
6299
3109
|
if (!isPlainRecord(payload)) return;
|
|
6300
|
-
|
|
6301
|
-
|
|
6302
|
-
|
|
6303
|
-
|
|
3110
|
+
if (!capture.modelIdDone) {
|
|
3111
|
+
const model = payload.model;
|
|
3112
|
+
if (typeof model === "string" && model.trim().length > 0) {
|
|
3113
|
+
capture.reportedModelId = model.trim();
|
|
3114
|
+
capture.modelIdDone = true;
|
|
3115
|
+
}
|
|
3116
|
+
}
|
|
3117
|
+
if (capture.backendTimingsSource !== null) {
|
|
3118
|
+
const timings = backendCompletionTimings(payload.timings, capture.backendTimingsSource);
|
|
3119
|
+
if (timings !== null) capture.backendTimings = timings;
|
|
6304
3120
|
}
|
|
6305
3121
|
} catch {
|
|
6306
3122
|
}
|
|
6307
3123
|
}
|
|
6308
3124
|
function observeResponseModelIdBytes(chunk, capture, flush = false) {
|
|
6309
|
-
if (
|
|
3125
|
+
if (!capture.decoder) return;
|
|
6310
3126
|
capture.buffer += chunk ? capture.decoder.decode(chunk, { stream: !flush }) : capture.decoder.decode();
|
|
6311
3127
|
let newline = capture.buffer.indexOf("\n");
|
|
6312
3128
|
while (newline >= 0) {
|
|
6313
3129
|
const line = capture.buffer.slice(0, newline);
|
|
6314
3130
|
capture.buffer = capture.buffer.slice(newline + 1);
|
|
6315
|
-
|
|
6316
|
-
if (capture
|
|
3131
|
+
observeResponseMetadataLine(line, capture);
|
|
3132
|
+
if (captureCanStopEarly(capture)) {
|
|
6317
3133
|
capture.buffer = "";
|
|
6318
3134
|
capture.decoder = null;
|
|
6319
3135
|
return;
|
|
@@ -6321,10 +3137,10 @@ function observeResponseModelIdBytes(chunk, capture, flush = false) {
|
|
|
6321
3137
|
newline = capture.buffer.indexOf("\n");
|
|
6322
3138
|
}
|
|
6323
3139
|
if (flush && capture.buffer.length > 0) {
|
|
6324
|
-
|
|
3140
|
+
observeResponseMetadataLine(capture.buffer, capture);
|
|
6325
3141
|
capture.buffer = "";
|
|
6326
3142
|
}
|
|
6327
|
-
if (capture
|
|
3143
|
+
if (captureCanStopEarly(capture) || flush) capture.decoder = null;
|
|
6328
3144
|
}
|
|
6329
3145
|
function captureResponseModelId(response, capture) {
|
|
6330
3146
|
if (!response.body || !response.headers.get("content-type")?.toLowerCase().includes("text/event-stream")) {
|
|
@@ -6334,12 +3150,12 @@ function captureResponseModelId(response, capture) {
|
|
|
6334
3150
|
new TransformStream({
|
|
6335
3151
|
transform(chunk, controller) {
|
|
6336
3152
|
capture.observed = true;
|
|
6337
|
-
|
|
3153
|
+
observeResponseModelIdBytes(chunk, capture);
|
|
6338
3154
|
controller.enqueue(chunk);
|
|
6339
3155
|
},
|
|
6340
3156
|
flush() {
|
|
6341
3157
|
capture.observed = true;
|
|
6342
|
-
|
|
3158
|
+
observeResponseModelIdBytes(void 0, capture, true);
|
|
6343
3159
|
}
|
|
6344
3160
|
})
|
|
6345
3161
|
);
|
|
@@ -6349,13 +3165,15 @@ function captureResponseModelId(response, capture) {
|
|
|
6349
3165
|
headers: response.headers
|
|
6350
3166
|
});
|
|
6351
3167
|
}
|
|
6352
|
-
function withResponseModelIdCapture(options, sourceFactory) {
|
|
3168
|
+
function withResponseModelIdCapture(model, options, sourceFactory) {
|
|
6353
3169
|
const capture = {
|
|
6354
3170
|
reportedModelId: null,
|
|
6355
3171
|
observed: false,
|
|
6356
|
-
|
|
3172
|
+
modelIdDone: false,
|
|
3173
|
+
backendTimings: null,
|
|
3174
|
+
backendTimingsSource: backendTimingsSourceForModel(model),
|
|
6357
3175
|
buffer: "",
|
|
6358
|
-
decoder: new
|
|
3176
|
+
decoder: new TextDecoder()
|
|
6359
3177
|
};
|
|
6360
3178
|
const fetchImpl = options.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
6361
3179
|
const capturedOptions = {
|
|
@@ -6368,8 +3186,13 @@ function withResponseModelIdCapture(options, sourceFactory) {
|
|
|
6368
3186
|
try {
|
|
6369
3187
|
for await (const event of source) {
|
|
6370
3188
|
const observation = capture.observed ? capture.reportedModelId === null ? { state: "not-reported" } : { state: "reported", reportedModelId: capture.reportedModelId } : { state: "not-observed" };
|
|
6371
|
-
if (event.type === "done")
|
|
6372
|
-
|
|
3189
|
+
if (event.type === "done") {
|
|
3190
|
+
event.message.responseModelIdObservation = observation;
|
|
3191
|
+
if (capture.backendTimings !== null) event.message.backendTimings = capture.backendTimings;
|
|
3192
|
+
} else if (event.type === "error") {
|
|
3193
|
+
event.error.responseModelIdObservation = observation;
|
|
3194
|
+
if (capture.backendTimings !== null) event.error.backendTimings = capture.backendTimings;
|
|
3195
|
+
}
|
|
6373
3196
|
annotated.push(event);
|
|
6374
3197
|
}
|
|
6375
3198
|
annotated.end();
|
|
@@ -6652,12 +3475,12 @@ function estimateReasoningTokens(content) {
|
|
|
6652
3475
|
if (chars === 0) return 0;
|
|
6653
3476
|
return Math.max(1, Math.round(chars / REASONING_CHARS_PER_TOKEN2));
|
|
6654
3477
|
}
|
|
6655
|
-
function
|
|
3478
|
+
function positiveNumber(value) {
|
|
6656
3479
|
return typeof value === "number" && Number.isFinite(value) && value > 0;
|
|
6657
3480
|
}
|
|
6658
3481
|
function hasReportedReasoningUsage(usage) {
|
|
6659
3482
|
const aliases = usage;
|
|
6660
|
-
return
|
|
3483
|
+
return positiveNumber(aliases.reasoning) || positiveNumber(aliases.reasoningTokens) || positiveNumber(aliases.reasoning_tokens);
|
|
6661
3484
|
}
|
|
6662
3485
|
function applyOpenAICompatReasoningEstimate(message) {
|
|
6663
3486
|
if (hasReportedReasoningUsage(message.usage)) return;
|
|
@@ -6864,6 +3687,7 @@ var openAICompletionsApiProvider = {
|
|
|
6864
3687
|
stripNeverReasoningFromStream(
|
|
6865
3688
|
filterGemmaChannelStream(
|
|
6866
3689
|
withResponseModelIdCapture(
|
|
3690
|
+
model,
|
|
6867
3691
|
withRemainingContextBudget(model, effectiveContext, withSamplers),
|
|
6868
3692
|
(capturedOptions) => withLocalResidency(
|
|
6869
3693
|
model,
|
|
@@ -6895,6 +3719,7 @@ var openAICompletionsApiProvider = {
|
|
|
6895
3719
|
stripNeverReasoningFromStream(
|
|
6896
3720
|
filterGemmaChannelStream(
|
|
6897
3721
|
withResponseModelIdCapture(
|
|
3722
|
+
model,
|
|
6898
3723
|
withRemainingContextBudget(model, effectiveContext, withSamplers),
|
|
6899
3724
|
(capturedOptions) => withLocalResidency(
|
|
6900
3725
|
model,
|
|
@@ -6929,14 +3754,14 @@ function registerClioApiProviders() {
|
|
|
6929
3754
|
|
|
6930
3755
|
// src/domains/providers/knowledge-base-path.ts
|
|
6931
3756
|
init_esm_shims();
|
|
6932
|
-
import { existsSync, statSync
|
|
6933
|
-
import { delimiter, dirname, join as
|
|
3757
|
+
import { existsSync, statSync } from "node:fs";
|
|
3758
|
+
import { delimiter, dirname, join as join2 } from "node:path";
|
|
6934
3759
|
import { fileURLToPath } from "node:url";
|
|
6935
3760
|
var MODEL_CATALOG_OVERLAY_DIR = "model-catalog.d";
|
|
6936
3761
|
var MODEL_CATALOG_DIRS_ENV = "CLIO_CODER_MODEL_CATALOG_DIRS";
|
|
6937
3762
|
function isDirectory(path) {
|
|
6938
3763
|
try {
|
|
6939
|
-
return
|
|
3764
|
+
return statSync(path).isDirectory();
|
|
6940
3765
|
} catch {
|
|
6941
3766
|
return false;
|
|
6942
3767
|
}
|
|
@@ -6944,18 +3769,18 @@ function isDirectory(path) {
|
|
|
6944
3769
|
function resolveProvidersModelsDir(importMetaUrl) {
|
|
6945
3770
|
const start = dirname(fileURLToPath(importMetaUrl));
|
|
6946
3771
|
const directCandidates = [
|
|
6947
|
-
|
|
6948
|
-
|
|
6949
|
-
|
|
3772
|
+
join2(start, "models"),
|
|
3773
|
+
join2(start, "..", "domains", "providers", "models"),
|
|
3774
|
+
join2(start, "..", "providers-models")
|
|
6950
3775
|
];
|
|
6951
3776
|
for (const candidate of directCandidates) {
|
|
6952
3777
|
if (isDirectory(candidate)) return candidate;
|
|
6953
3778
|
}
|
|
6954
3779
|
let cursor = start;
|
|
6955
3780
|
for (let i = 0; i < 8; i++) {
|
|
6956
|
-
const packageJson =
|
|
6957
|
-
const sourceModels =
|
|
6958
|
-
const distModels =
|
|
3781
|
+
const packageJson = join2(cursor, "package.json");
|
|
3782
|
+
const sourceModels = join2(cursor, "src", "domains", "providers", "models");
|
|
3783
|
+
const distModels = join2(cursor, "dist", "providers-models");
|
|
6959
3784
|
if (existsSync(packageJson)) {
|
|
6960
3785
|
if (isDirectory(sourceModels)) return sourceModels;
|
|
6961
3786
|
if (isDirectory(distModels)) return distModels;
|
|
@@ -6986,8 +3811,8 @@ function resolveProviderModelCatalogDirs(importMetaUrl, options = {}) {
|
|
|
6986
3811
|
const bundled = resolveProvidersModelsDir(importMetaUrl);
|
|
6987
3812
|
const cwd = options.cwd ?? process.cwd();
|
|
6988
3813
|
const overlays = uniqueExistingDirs([
|
|
6989
|
-
|
|
6990
|
-
|
|
3814
|
+
join2(resolveClioDirs().config, MODEL_CATALOG_OVERLAY_DIR),
|
|
3815
|
+
join2(cwd, ".clio-coder", MODEL_CATALOG_OVERLAY_DIR),
|
|
6991
3816
|
...envOverlayDirs()
|
|
6992
3817
|
]);
|
|
6993
3818
|
return {
|
|
@@ -7006,7 +3831,7 @@ function resolveProviderKnowledgeBaseRoots(importMetaUrl, options = {}) {
|
|
|
7006
3831
|
|
|
7007
3832
|
// src/domains/providers/plugins.ts
|
|
7008
3833
|
init_esm_shims();
|
|
7009
|
-
import { join as
|
|
3834
|
+
import { join as join3 } from "node:path";
|
|
7010
3835
|
function extractPluginPackages(settings) {
|
|
7011
3836
|
if (!settings || typeof settings !== "object") return [];
|
|
7012
3837
|
const raw = settings.runtimePlugins;
|
|
@@ -7015,7 +3840,7 @@ function extractPluginPackages(settings) {
|
|
|
7015
3840
|
}
|
|
7016
3841
|
async function loadPluginRuntimes(registry, settings) {
|
|
7017
3842
|
const loaded = [];
|
|
7018
|
-
const pluginDir =
|
|
3843
|
+
const pluginDir = join3(clioConfigDir(), "runtimes");
|
|
7019
3844
|
const packages = extractPluginPackages(settings);
|
|
7020
3845
|
try {
|
|
7021
3846
|
const ids = await registry.loadFromDir(pluginDir, activateExternalPluginApiBridge);
|
|
@@ -7043,8 +3868,8 @@ async function loadPluginRuntimes(registry, settings) {
|
|
|
7043
3868
|
// src/domains/providers/types/knowledge-base.ts
|
|
7044
3869
|
init_esm_shims();
|
|
7045
3870
|
var import_yaml = __toESM(require_dist(), 1);
|
|
7046
|
-
import { readdirSync
|
|
7047
|
-
import { join as
|
|
3871
|
+
import { readdirSync, readFileSync, statSync as statSync2 } from "node:fs";
|
|
3872
|
+
import { join as join4 } from "node:path";
|
|
7048
3873
|
var FileKnowledgeBase = class {
|
|
7049
3874
|
roots;
|
|
7050
3875
|
loaded = [];
|
|
@@ -7098,7 +3923,7 @@ function normalizeRoots(root) {
|
|
|
7098
3923
|
if (dir.length === 0 || seen.has(dir)) continue;
|
|
7099
3924
|
let isRootDir = false;
|
|
7100
3925
|
try {
|
|
7101
|
-
isRootDir =
|
|
3926
|
+
isRootDir = statSync2(dir).isDirectory();
|
|
7102
3927
|
} catch (err) {
|
|
7103
3928
|
if (raw.optional === true) continue;
|
|
7104
3929
|
throw err;
|
|
@@ -7114,9 +3939,9 @@ function normalizeRoots(root) {
|
|
|
7114
3939
|
}
|
|
7115
3940
|
function collectYamlFiles(dir, prefix = "") {
|
|
7116
3941
|
const out = [];
|
|
7117
|
-
const entries =
|
|
3942
|
+
const entries = readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name));
|
|
7118
3943
|
for (const entry of entries) {
|
|
7119
|
-
const path =
|
|
3944
|
+
const path = join4(dir, entry.name);
|
|
7120
3945
|
const name = prefix ? `${prefix}/${entry.name}` : entry.name;
|
|
7121
3946
|
if (entry.isDirectory()) {
|
|
7122
3947
|
out.push(...collectYamlFiles(path, name));
|
|
@@ -7135,20 +3960,20 @@ function normalizeEntry(raw, file) {
|
|
|
7135
3960
|
const candidate = raw;
|
|
7136
3961
|
const family = candidate.family;
|
|
7137
3962
|
const patterns = candidate.matchPatterns;
|
|
7138
|
-
const
|
|
3963
|
+
const capabilities = candidate.capabilities;
|
|
7139
3964
|
if (typeof family !== "string" || family.length === 0) {
|
|
7140
3965
|
throw new Error(`knowledge base file ${file}: entry is missing 'family' string`);
|
|
7141
3966
|
}
|
|
7142
3967
|
if (!Array.isArray(patterns) || patterns.some((p) => typeof p !== "string")) {
|
|
7143
3968
|
throw new Error(`knowledge base file ${file}: entry '${family}' needs matchPatterns: string[]`);
|
|
7144
3969
|
}
|
|
7145
|
-
if (typeof
|
|
3970
|
+
if (typeof capabilities !== "object" || capabilities === null || Array.isArray(capabilities)) {
|
|
7146
3971
|
throw new Error(`knowledge base file ${file}: entry '${family}' needs capabilities object`);
|
|
7147
3972
|
}
|
|
7148
3973
|
const entry = {
|
|
7149
3974
|
family,
|
|
7150
3975
|
matchPatterns: patterns,
|
|
7151
|
-
capabilities
|
|
3976
|
+
capabilities
|
|
7152
3977
|
};
|
|
7153
3978
|
if (candidate.quirks !== void 0) {
|
|
7154
3979
|
if (typeof candidate.quirks !== "object" || candidate.quirks === null || Array.isArray(candidate.quirks)) {
|
|
@@ -7198,9 +4023,9 @@ function capabilitiesFor(desc, target, probe, kb) {
|
|
|
7198
4023
|
const base = capabilitiesFromCatalogModel(desc.defaultCapabilities, catalogModel);
|
|
7199
4024
|
const probeCaps = probeCapabilitiesForModel({ target, ...probe }, target.defaultModel);
|
|
7200
4025
|
const userOverride = target.capabilities ?? null;
|
|
7201
|
-
const
|
|
4026
|
+
const capabilities = mergeCapabilities(base, kbHit?.entry.capabilities ?? null, probeCaps, userOverride);
|
|
7202
4027
|
return {
|
|
7203
|
-
capabilities
|
|
4028
|
+
capabilities,
|
|
7204
4029
|
contextWindowProvenance: contextWindowProvenanceOf(kbHit, catalogModel, probeCaps, userOverride)
|
|
7205
4030
|
};
|
|
7206
4031
|
}
|
|
@@ -7255,6 +4080,15 @@ function mergeProbeResult(desc, target, probe, previous) {
|
|
|
7255
4080
|
if (probeSurfaces && Object.keys(probeSurfaces).length > 0) merge.probeSurfaces = probeSurfaces;
|
|
7256
4081
|
return merge;
|
|
7257
4082
|
}
|
|
4083
|
+
function unservedDefaultModelReason(desc, target, merge) {
|
|
4084
|
+
const model = target.defaultModel;
|
|
4085
|
+
if (!model) return null;
|
|
4086
|
+
if (merge.discoveredModelsSource !== "probe" || merge.discoveredModels.length === 0) return null;
|
|
4087
|
+
if (listKnownModelsForRuntime(desc.id).length > 0) return null;
|
|
4088
|
+
if (merge.discoveredModels.includes(model)) return null;
|
|
4089
|
+
if (merge.discoveredModelStates && model in merge.discoveredModelStates) return null;
|
|
4090
|
+
return `default model '${model}' is not advertised by the target`;
|
|
4091
|
+
}
|
|
7258
4092
|
function createProvidersBundle(context) {
|
|
7259
4093
|
const registry = getRuntimeRegistry();
|
|
7260
4094
|
const authStore = openAuthStorage();
|
|
@@ -7330,12 +4164,13 @@ function createProvidersBundle(context) {
|
|
|
7330
4164
|
}
|
|
7331
4165
|
const availability = availabilityFor(desc, target, authStatusFor);
|
|
7332
4166
|
const merge = mergeProbeResult(desc, target, probe, previous);
|
|
7333
|
-
const { capabilities
|
|
4167
|
+
const { capabilities, contextWindowProvenance } = capabilitiesFor(desc, target, merge, kb);
|
|
7334
4168
|
const healthy = probe !== null ? probe.ok : null;
|
|
4169
|
+
const unservedDefault = probe?.ok ? unservedDefaultModelReason(desc, target, merge) : null;
|
|
7335
4170
|
const health = probe === null ? previous?.health ?? emptyHealth() : {
|
|
7336
|
-
status: healthy ? "healthy" : "down",
|
|
4171
|
+
status: healthy ? unservedDefault === null ? "healthy" : "degraded" : "down",
|
|
7337
4172
|
lastCheckAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
7338
|
-
lastError: probe.error ??
|
|
4173
|
+
lastError: probe.error ?? unservedDefault,
|
|
7339
4174
|
latencyMs: probe.latencyMs ?? null
|
|
7340
4175
|
};
|
|
7341
4176
|
const available = availability.available && (probe === null || probe.ok);
|
|
@@ -7346,7 +4181,7 @@ function createProvidersBundle(context) {
|
|
|
7346
4181
|
available,
|
|
7347
4182
|
reason,
|
|
7348
4183
|
health,
|
|
7349
|
-
capabilities
|
|
4184
|
+
capabilities,
|
|
7350
4185
|
contextWindowProvenance,
|
|
7351
4186
|
probeCapabilities: merge.probeCapabilities,
|
|
7352
4187
|
probeModelCapabilities: merge.probeModelCapabilities,
|
|
@@ -7677,12 +4512,6 @@ function resolveModelReference(rawPattern, providers) {
|
|
|
7677
4512
|
return buildResult(base, pickFirst(partial), thinkingLevel);
|
|
7678
4513
|
}
|
|
7679
4514
|
|
|
7680
|
-
// src/domains/providers/types/cost-provenance.ts
|
|
7681
|
-
init_esm_shims();
|
|
7682
|
-
function normalizeCostProvenance(value) {
|
|
7683
|
-
return value ?? "unknown";
|
|
7684
|
-
}
|
|
7685
|
-
|
|
7686
4515
|
// src/domains/providers/index.ts
|
|
7687
4516
|
var ProvidersDomainModule = {
|
|
7688
4517
|
manifest: ProvidersManifest,
|
|
@@ -7690,14 +4519,15 @@ var ProvidersDomainModule = {
|
|
|
7690
4519
|
};
|
|
7691
4520
|
|
|
7692
4521
|
export {
|
|
7693
|
-
|
|
7694
|
-
AGENT_ROLE_TOOLS_REQUIRED_REASON,
|
|
7695
|
-
resolveEffectivePricing,
|
|
7696
|
-
getCatalogModelForRuntime,
|
|
7697
|
-
getRuntimeRegistry,
|
|
7698
|
-
credentialsPresent,
|
|
4522
|
+
residencyTargetKey,
|
|
7699
4523
|
setGlobalDefaultMaxOutputTokens,
|
|
7700
4524
|
resolveReservedOutputTokens,
|
|
4525
|
+
canonicalEndpointKey,
|
|
4526
|
+
endpointLabel,
|
|
4527
|
+
endpointCapacityForStatus,
|
|
4528
|
+
endpointCapacitiesForStatuses,
|
|
4529
|
+
registerForegroundStream,
|
|
4530
|
+
foregroundStreamUsage,
|
|
7701
4531
|
setResidencyNoticeSink,
|
|
7702
4532
|
setProtectedModelsProvider,
|
|
7703
4533
|
withRunOverrides,
|
|
@@ -7710,22 +4540,12 @@ export {
|
|
|
7710
4540
|
thinkingLevelFromChoiceLabel,
|
|
7711
4541
|
resolveModelRuntimeCapabilitiesForProviders,
|
|
7712
4542
|
resolveModelRuntimeCapabilitiesForModel,
|
|
7713
|
-
greetLmStudio,
|
|
7714
4543
|
registerClioApiProviders,
|
|
7715
4544
|
resolveProviderKnowledgeBaseRoots,
|
|
7716
4545
|
loadPluginRuntimes,
|
|
7717
|
-
formatContextWindowSlots,
|
|
7718
|
-
registerBuiltinRuntimes,
|
|
7719
4546
|
FileKnowledgeBase,
|
|
7720
4547
|
isOrchestratorEligibleRuntime,
|
|
7721
4548
|
isDispatchEligibleRuntime,
|
|
7722
|
-
supportGroupLabel,
|
|
7723
|
-
listKnownModelsForRuntime,
|
|
7724
|
-
describeRuntimeModels,
|
|
7725
|
-
buildProviderSupportEntry,
|
|
7726
|
-
listProviderSupportEntries,
|
|
7727
|
-
configuredTargetsForRuntime,
|
|
7728
|
-
resolveProviderReference,
|
|
7729
4549
|
modelResidencyForStatus,
|
|
7730
4550
|
modelCandidatesForStatus,
|
|
7731
4551
|
modelIdsForStatus,
|
|
@@ -7740,4 +4560,4 @@ export {
|
|
|
7740
4560
|
normalizeCostProvenance,
|
|
7741
4561
|
ProvidersDomainModule
|
|
7742
4562
|
};
|
|
7743
|
-
//# sourceMappingURL=chunk-
|
|
4563
|
+
//# sourceMappingURL=chunk-MIX5N5AC.js.map
|