@srouterhq/server 0.0.0-stage → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/client/dist/assets/am-TQ7Jmdqe.js +1 -0
- package/client/dist/assets/ar-Bf5a7yfC.js +1 -0
- package/client/dist/assets/az-CrosE1xH.js +1 -0
- package/client/dist/assets/bg-BvWnmcUz.js +1 -0
- package/client/dist/assets/bn-ams5TsJp.js +1 -0
- package/client/dist/assets/cs-Dh67CaJ_.js +1 -0
- package/client/dist/assets/da-9WlT8Iq7.js +1 -0
- package/client/dist/assets/de-lWhJJzz7.js +1 -0
- package/client/dist/assets/el-Cu7_GhNq.js +1 -0
- package/client/dist/assets/es-BZuzsBcP.js +1 -0
- package/client/dist/assets/fa-Dwdn-jKS.js +1 -0
- package/client/dist/assets/fi-BFrFyOTy.js +1 -0
- package/client/dist/assets/fr-XhbtmPpj.js +1 -0
- package/client/dist/assets/geist-cyrillic-ext-wght-normal-DjL33-gN.woff2 +0 -0
- package/client/dist/assets/geist-cyrillic-wght-normal-BEAKL7Jp.woff2 +0 -0
- package/client/dist/assets/geist-latin-ext-wght-normal-DC-KSUi6.woff2 +0 -0
- package/client/dist/assets/geist-latin-wght-normal-BgDaEnEv.woff2 +0 -0
- package/client/dist/assets/geist-mono-cyrillic-ext-wght-normal-I4S5GZfc.woff2 +0 -0
- package/client/dist/assets/geist-mono-cyrillic-wght-normal-BmXc_FBt.woff2 +0 -0
- package/client/dist/assets/geist-mono-latin-ext-wght-normal-DrnZ1wKl.woff2 +0 -0
- package/client/dist/assets/geist-mono-latin-wght-normal-B_7UjwxQ.woff2 +0 -0
- package/client/dist/assets/geist-mono-symbols2-wght-normal-GZpp1pK2.woff2 +0 -0
- package/client/dist/assets/geist-mono-vietnamese-wght-normal-D8KDMBhC.woff2 +0 -0
- package/client/dist/assets/geist-vietnamese-wght-normal-6IgcOCM7.woff2 +0 -0
- package/client/dist/assets/gu-D0uhLUB8.js +1 -0
- package/client/dist/assets/ha-CVlm9pqp.js +1 -0
- package/client/dist/assets/he-CMIFeoWr.js +1 -0
- package/client/dist/assets/hi-ENTwhBpn.js +1 -0
- package/client/dist/assets/hr-fc03vMy_.js +1 -0
- package/client/dist/assets/hu-YBo2BIYt.js +1 -0
- package/client/dist/assets/id-BEWXYX69.js +1 -0
- package/client/dist/assets/ig-BKLZFKka.js +1 -0
- package/client/dist/assets/index-D9I4KtvR.js +145 -0
- package/client/dist/assets/index-txhEmCdb.css +2 -0
- package/client/dist/assets/it-Bzmei1Y2.js +1 -0
- package/client/dist/assets/ja-BfcpuqEl.js +1 -0
- package/client/dist/assets/ka-CnWjRgLo.js +1 -0
- package/client/dist/assets/km-BAjPR3Qj.js +1 -0
- package/client/dist/assets/kn-B15WKU5m.js +1 -0
- package/client/dist/assets/ko-C61fAB-w.js +1 -0
- package/client/dist/assets/lt-CYTr4_YW.js +1 -0
- package/client/dist/assets/ml-CtqxgyH3.js +1 -0
- package/client/dist/assets/mr-X5TMY7aR.js +1 -0
- package/client/dist/assets/ms-DSg8RqPS.js +1 -0
- package/client/dist/assets/my-D7RqR8uE.js +1 -0
- package/client/dist/assets/ne-DaflxkDj.js +1 -0
- package/client/dist/assets/nl-BOqqtsC8.js +1 -0
- package/client/dist/assets/no-BSHp_StB.js +1 -0
- package/client/dist/assets/or-BkLR9Vub.js +1 -0
- package/client/dist/assets/pa-CarJ2XgI.js +1 -0
- package/client/dist/assets/pl-C2n0V-0k.js +1 -0
- package/client/dist/assets/pt-BR-CqJdNhg2.js +1 -0
- package/client/dist/assets/pt-PT-C_HpMyKz.js +1 -0
- package/client/dist/assets/ro-DVRZqSAS.js +1 -0
- package/client/dist/assets/ru-DwAJG7VC.js +1 -0
- package/client/dist/assets/si-BxUezXm0.js +1 -0
- package/client/dist/assets/sk-eI1VcIfY.js +1 -0
- package/client/dist/assets/sr-CJF3auCr.js +1 -0
- package/client/dist/assets/srouter-logo-C6ZfjGIi.svg +14 -0
- package/client/dist/assets/sv-nqvwHAq_.js +1 -0
- package/client/dist/assets/sw-DWynE2pW.js +1 -0
- package/client/dist/assets/ta-BVM12knC.js +1 -0
- package/client/dist/assets/te-oyzEkF7_.js +1 -0
- package/client/dist/assets/th-COVeE0ZQ.js +1 -0
- package/client/dist/assets/tl-DsJANnQz.js +1 -0
- package/client/dist/assets/tr-B-Fl-urO.js +1 -0
- package/client/dist/assets/uk-C5qketAO.js +1 -0
- package/client/dist/assets/ur-QU9KugfV.js +1 -0
- package/client/dist/assets/uz-DL9cG_mY.js +1 -0
- package/client/dist/assets/vi-CQiVy3Pu.js +1 -0
- package/client/dist/assets/yo-DqazMdrZ.js +1 -0
- package/client/dist/assets/zh-CN-DnYL_Kio.js +1 -0
- package/client/dist/assets/zh-TW-5bdbosKj.js +1 -0
- package/client/dist/favicon.svg +14 -0
- package/client/dist/icons.svg +24 -0
- package/client/dist/index.html +37 -0
- package/dist/app.d.ts +4 -0
- package/dist/app.d.ts.map +1 -0
- package/dist/app.js +372 -0
- package/dist/app.js.map +1 -0
- package/dist/db/index.d.ts +49 -0
- package/dist/db/index.d.ts.map +1 -0
- package/dist/db/index.js +203 -0
- package/dist/db/index.js.map +1 -0
- package/dist/db/migrate/TEMPLATE.d.ts +4 -0
- package/dist/db/migrate/TEMPLATE.d.ts.map +1 -0
- package/dist/db/migrate/TEMPLATE.js +18 -0
- package/dist/db/migrate/TEMPLATE.js.map +1 -0
- package/dist/db/migrate/cli.d.ts +2 -0
- package/dist/db/migrate/cli.d.ts.map +1 -0
- package/dist/db/migrate/cli.js +177 -0
- package/dist/db/migrate/cli.js.map +1 -0
- package/dist/db/migrate/defaults.d.ts +52 -0
- package/dist/db/migrate/defaults.d.ts.map +1 -0
- package/dist/db/migrate/defaults.js +126 -0
- package/dist/db/migrate/defaults.js.map +1 -0
- package/dist/db/migrate/runner.d.ts +16 -0
- package/dist/db/migrate/runner.d.ts.map +1 -0
- package/dist/db/migrate/runner.js +178 -0
- package/dist/db/migrate/runner.js.map +1 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts +4 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts.map +1 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.js +2240 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.js.map +1 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts +4 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts.map +1 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.js +27 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.js.map +1 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts +4 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts.map +1 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.js +30 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.js.map +1 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.d.ts +6 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.d.ts.map +1 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.js +104 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.js.map +1 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts +9 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts.map +1 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.js +22 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.js.map +1 -0
- package/dist/db/migrations/20260706_000001_request_client_info.d.ts +4 -0
- package/dist/db/migrations/20260706_000001_request_client_info.d.ts.map +1 -0
- package/dist/db/migrations/20260706_000001_request_client_info.js +30 -0
- package/dist/db/migrations/20260706_000001_request_client_info.js.map +1 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts +17 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts.map +1 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.js +20 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.js.map +1 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts +12 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts.map +1 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.js +45 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.js.map +1 -0
- package/dist/db/migrations/20260720_000001_key_health_error.d.ts +5 -0
- package/dist/db/migrations/20260720_000001_key_health_error.d.ts.map +1 -0
- package/dist/db/migrations/20260720_000001_key_health_error.js +16 -0
- package/dist/db/migrations/20260720_000001_key_health_error.js.map +1 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts +4 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js +38 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js.map +1 -0
- package/dist/db/migrations/20260726_000002_request_attempts.d.ts +4 -0
- package/dist/db/migrations/20260726_000002_request_attempts.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000002_request_attempts.js +43 -0
- package/dist/db/migrations/20260726_000002_request_attempts.js.map +1 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts +4 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.js +80 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.js.map +1 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.d.ts +4 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.js +34 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.js.map +1 -0
- package/dist/db/migrations/20260726_000005_request_served_model.d.ts +4 -0
- package/dist/db/migrations/20260726_000005_request_served_model.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000005_request_served_model.js +31 -0
- package/dist/db/migrations/20260726_000005_request_served_model.js.map +1 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts +4 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.js +29 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.js.map +1 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts +4 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts.map +1 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.js +41 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.js.map +1 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts +12 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.js +37 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.js.map +1 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts +4 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts.map +1 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js +145 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js.map +1 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts +6 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts.map +1 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js +52 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js.map +1 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.d.ts +6 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.d.ts.map +1 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.js +17 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.js.map +1 -0
- package/dist/db/migrations/20260805_000002_client_profiles.d.ts +4 -0
- package/dist/db/migrations/20260805_000002_client_profiles.d.ts.map +1 -0
- package/dist/db/migrations/20260805_000002_client_profiles.js +33 -0
- package/dist/db/migrations/20260805_000002_client_profiles.js.map +1 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts +4 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts.map +1 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.js +33 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.js.map +1 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts +4 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts.map +1 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.js +19 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.js.map +1 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.d.ts +4 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.d.ts.map +1 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.js +47 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.js.map +1 -0
- package/dist/db/migrations/20260823_000001_server_logs.d.ts +4 -0
- package/dist/db/migrations/20260823_000001_server_logs.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000001_server_logs.js +53 -0
- package/dist/db/migrations/20260823_000001_server_logs.js.map +1 -0
- package/dist/db/migrations/20260823_000002_backups_table.d.ts +8 -0
- package/dist/db/migrations/20260823_000002_backups_table.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000002_backups_table.js +22 -0
- package/dist/db/migrations/20260823_000002_backups_table.js.map +1 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts +19 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.js +28 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.js.map +1 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts +14 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.js +25 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.js.map +1 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts +4 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.js +46 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.js.map +1 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts +4 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.js +37 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.js.map +1 -0
- package/dist/db/migrations/20260901_000003_request_caller.d.ts +4 -0
- package/dist/db/migrations/20260901_000003_request_caller.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000003_request_caller.js +27 -0
- package/dist/db/migrations/20260901_000003_request_caller.js.map +1 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts +4 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts.map +1 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js +35 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js.map +1 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts +4 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts.map +1 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.js +37 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.js.map +1 -0
- package/dist/db/migrations/20260903_000002_response_cache.d.ts +4 -0
- package/dist/db/migrations/20260903_000002_response_cache.d.ts.map +1 -0
- package/dist/db/migrations/20260903_000002_response_cache.js +53 -0
- package/dist/db/migrations/20260903_000002_response_cache.js.map +1 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts +13 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts.map +1 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.js +35 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.js.map +1 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts +4 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts.map +1 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.js +68 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.js.map +1 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts +5 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts.map +1 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.js +37 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.js.map +1 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts +4 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts.map +1 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js +28 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js.map +1 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts +4 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts.map +1 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js +25 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js.map +1 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.d.ts +10 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.d.ts.map +1 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.js +78 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.js.map +1 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts +21 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts.map +1 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.js +41 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.js.map +1 -0
- package/dist/db/model-pricing.d.ts +39 -0
- package/dist/db/model-pricing.d.ts.map +1 -0
- package/dist/db/model-pricing.js +274 -0
- package/dist/db/model-pricing.js.map +1 -0
- package/dist/db/node-sqlite.d.ts +9 -0
- package/dist/db/node-sqlite.d.ts.map +1 -0
- package/dist/db/node-sqlite.js +87 -0
- package/dist/db/node-sqlite.js.map +1 -0
- package/dist/db/types.d.ts +22 -0
- package/dist/db/types.d.ts.map +1 -0
- package/dist/db/types.js +2 -0
- package/dist/db/types.js.map +1 -0
- package/dist/docs/docs-page.d.ts +2 -0
- package/dist/docs/docs-page.d.ts.map +1 -0
- package/dist/docs/docs-page.js +319 -0
- package/dist/docs/docs-page.js.map +1 -0
- package/dist/docs/openapi.d.ts +1896 -0
- package/dist/docs/openapi.d.ts.map +1 -0
- package/dist/docs/openapi.js +1096 -0
- package/dist/docs/openapi.js.map +1 -0
- package/dist/env.d.ts +2 -0
- package/dist/env.d.ts.map +1 -0
- package/dist/env.js +9 -0
- package/dist/env.js.map +1 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +144 -0
- package/dist/index.js.map +1 -0
- package/dist/lib/anthropic-documents.d.ts +41 -0
- package/dist/lib/anthropic-documents.d.ts.map +1 -0
- package/dist/lib/anthropic-documents.js +132 -0
- package/dist/lib/anthropic-documents.js.map +1 -0
- package/dist/lib/app-version.d.ts +5 -0
- package/dist/lib/app-version.d.ts.map +1 -0
- package/dist/lib/app-version.js +61 -0
- package/dist/lib/app-version.js.map +1 -0
- package/dist/lib/attempt-trace.d.ts +23 -0
- package/dist/lib/attempt-trace.d.ts.map +1 -0
- package/dist/lib/attempt-trace.js +32 -0
- package/dist/lib/attempt-trace.js.map +1 -0
- package/dist/lib/budget.d.ts +6 -0
- package/dist/lib/budget.d.ts.map +1 -0
- package/dist/lib/budget.js +40 -0
- package/dist/lib/budget.js.map +1 -0
- package/dist/lib/client-classifier.d.ts +10 -0
- package/dist/lib/client-classifier.d.ts.map +1 -0
- package/dist/lib/client-classifier.js +126 -0
- package/dist/lib/client-classifier.js.map +1 -0
- package/dist/lib/client-context.d.ts +10 -0
- package/dist/lib/client-context.d.ts.map +1 -0
- package/dist/lib/client-context.js +39 -0
- package/dist/lib/client-context.js.map +1 -0
- package/dist/lib/config.d.ts +33 -0
- package/dist/lib/config.d.ts.map +1 -0
- package/dist/lib/config.js +94 -0
- package/dist/lib/config.js.map +1 -0
- package/dist/lib/content.d.ts +33 -0
- package/dist/lib/content.d.ts.map +1 -0
- package/dist/lib/content.js +201 -0
- package/dist/lib/content.js.map +1 -0
- package/dist/lib/credential.d.ts +11 -0
- package/dist/lib/credential.d.ts.map +1 -0
- package/dist/lib/credential.js +18 -0
- package/dist/lib/credential.js.map +1 -0
- package/dist/lib/crypto.d.ts +31 -0
- package/dist/lib/crypto.d.ts.map +1 -0
- package/dist/lib/crypto.js +206 -0
- package/dist/lib/crypto.js.map +1 -0
- package/dist/lib/custom-provider-cleanup.d.ts +13 -0
- package/dist/lib/custom-provider-cleanup.d.ts.map +1 -0
- package/dist/lib/custom-provider-cleanup.js +43 -0
- package/dist/lib/custom-provider-cleanup.js.map +1 -0
- package/dist/lib/db-backup.d.ts +31 -0
- package/dist/lib/db-backup.d.ts.map +1 -0
- package/dist/lib/db-backup.js +255 -0
- package/dist/lib/db-backup.js.map +1 -0
- package/dist/lib/endpoint-scope.d.ts +42 -0
- package/dist/lib/endpoint-scope.d.ts.map +1 -0
- package/dist/lib/endpoint-scope.js +95 -0
- package/dist/lib/endpoint-scope.js.map +1 -0
- package/dist/lib/env-drift.d.ts +17 -0
- package/dist/lib/env-drift.d.ts.map +1 -0
- package/dist/lib/env-drift.js +108 -0
- package/dist/lib/env-drift.js.map +1 -0
- package/dist/lib/error-classify.d.ts +40 -0
- package/dist/lib/error-classify.d.ts.map +1 -0
- package/dist/lib/error-classify.js +687 -0
- package/dist/lib/error-classify.js.map +1 -0
- package/dist/lib/error-redaction.d.ts +8 -0
- package/dist/lib/error-redaction.d.ts.map +1 -0
- package/dist/lib/error-redaction.js +45 -0
- package/dist/lib/error-redaction.js.map +1 -0
- package/dist/lib/fallback-loop.d.ts +306 -0
- package/dist/lib/fallback-loop.d.ts.map +1 -0
- package/dist/lib/fallback-loop.js +1336 -0
- package/dist/lib/fallback-loop.js.map +1 -0
- package/dist/lib/file-permissions.d.ts +98 -0
- package/dist/lib/file-permissions.d.ts.map +1 -0
- package/dist/lib/file-permissions.js +159 -0
- package/dist/lib/file-permissions.js.map +1 -0
- package/dist/lib/gemini-wire.d.ts +76 -0
- package/dist/lib/gemini-wire.d.ts.map +1 -0
- package/dist/lib/gemini-wire.js +392 -0
- package/dist/lib/gemini-wire.js.map +1 -0
- package/dist/lib/guardrails.d.ts +37 -0
- package/dist/lib/guardrails.d.ts.map +1 -0
- package/dist/lib/guardrails.js +106 -0
- package/dist/lib/guardrails.js.map +1 -0
- package/dist/lib/header-value.d.ts +8 -0
- package/dist/lib/header-value.d.ts.map +1 -0
- package/dist/lib/header-value.js +55 -0
- package/dist/lib/header-value.js.map +1 -0
- package/dist/lib/image-normalize.d.ts +21 -0
- package/dist/lib/image-normalize.d.ts.map +1 -0
- package/dist/lib/image-normalize.js +218 -0
- package/dist/lib/image-normalize.js.map +1 -0
- package/dist/lib/inbound-chat.d.ts +48 -0
- package/dist/lib/inbound-chat.d.ts.map +1 -0
- package/dist/lib/inbound-chat.js +406 -0
- package/dist/lib/inbound-chat.js.map +1 -0
- package/dist/lib/key-parser.d.ts +79 -0
- package/dist/lib/key-parser.d.ts.map +1 -0
- package/dist/lib/key-parser.js +701 -0
- package/dist/lib/key-parser.js.map +1 -0
- package/dist/lib/key-proxy.d.ts +49 -0
- package/dist/lib/key-proxy.d.ts.map +1 -0
- package/dist/lib/key-proxy.js +92 -0
- package/dist/lib/key-proxy.js.map +1 -0
- package/dist/lib/log-redaction.d.ts +28 -0
- package/dist/lib/log-redaction.d.ts.map +1 -0
- package/dist/lib/log-redaction.js +166 -0
- package/dist/lib/log-redaction.js.map +1 -0
- package/dist/lib/model-scope.d.ts +9 -0
- package/dist/lib/model-scope.d.ts.map +1 -0
- package/dist/lib/model-scope.js +29 -0
- package/dist/lib/model-scope.js.map +1 -0
- package/dist/lib/one-time-code.d.ts +15 -0
- package/dist/lib/one-time-code.d.ts.map +1 -0
- package/dist/lib/one-time-code.js +41 -0
- package/dist/lib/one-time-code.js.map +1 -0
- package/dist/lib/output-cap.d.ts +24 -0
- package/dist/lib/output-cap.d.ts.map +1 -0
- package/dist/lib/output-cap.js +80 -0
- package/dist/lib/output-cap.js.map +1 -0
- package/dist/lib/password.d.ts +3 -0
- package/dist/lib/password.d.ts.map +1 -0
- package/dist/lib/password.js +27 -0
- package/dist/lib/password.js.map +1 -0
- package/dist/lib/process-safety-net.d.ts +32 -0
- package/dist/lib/process-safety-net.d.ts.map +1 -0
- package/dist/lib/process-safety-net.js +123 -0
- package/dist/lib/process-safety-net.js.map +1 -0
- package/dist/lib/provider-identity.d.ts +37 -0
- package/dist/lib/provider-identity.d.ts.map +1 -0
- package/dist/lib/provider-identity.js +110 -0
- package/dist/lib/provider-identity.js.map +1 -0
- package/dist/lib/provider-size-parser.d.ts +6 -0
- package/dist/lib/provider-size-parser.d.ts.map +1 -0
- package/dist/lib/provider-size-parser.js +72 -0
- package/dist/lib/provider-size-parser.js.map +1 -0
- package/dist/lib/provider-timeout.d.ts +17 -0
- package/dist/lib/provider-timeout.d.ts.map +1 -0
- package/dist/lib/provider-timeout.js +70 -0
- package/dist/lib/provider-timeout.js.map +1 -0
- package/dist/lib/proxy.d.ts +180 -0
- package/dist/lib/proxy.d.ts.map +1 -0
- package/dist/lib/proxy.js +1002 -0
- package/dist/lib/proxy.js.map +1 -0
- package/dist/lib/request-log.d.ts +4 -0
- package/dist/lib/request-log.d.ts.map +1 -0
- package/dist/lib/request-log.js +161 -0
- package/dist/lib/request-log.js.map +1 -0
- package/dist/lib/reset-code.d.ts +5 -0
- package/dist/lib/reset-code.d.ts.map +1 -0
- package/dist/lib/reset-code.js +39 -0
- package/dist/lib/reset-code.js.map +1 -0
- package/dist/lib/retry-hint.d.ts +14 -0
- package/dist/lib/retry-hint.d.ts.map +1 -0
- package/dist/lib/retry-hint.js +36 -0
- package/dist/lib/retry-hint.js.map +1 -0
- package/dist/lib/route-live.d.ts +29 -0
- package/dist/lib/route-live.d.ts.map +1 -0
- package/dist/lib/route-live.js +97 -0
- package/dist/lib/route-live.js.map +1 -0
- package/dist/lib/sampling-params.d.ts +183 -0
- package/dist/lib/sampling-params.d.ts.map +1 -0
- package/dist/lib/sampling-params.js +386 -0
- package/dist/lib/sampling-params.js.map +1 -0
- package/dist/lib/scheduler.d.ts +13 -0
- package/dist/lib/scheduler.d.ts.map +1 -0
- package/dist/lib/scheduler.js +11 -0
- package/dist/lib/scheduler.js.map +1 -0
- package/dist/lib/served-model.d.ts +16 -0
- package/dist/lib/served-model.d.ts.map +1 -0
- package/dist/lib/served-model.js +0 -0
- package/dist/lib/served-model.js.map +1 -0
- package/dist/lib/server-logs.d.ts +132 -0
- package/dist/lib/server-logs.d.ts.map +1 -0
- package/dist/lib/server-logs.js +473 -0
- package/dist/lib/server-logs.js.map +1 -0
- package/dist/lib/setup-code.d.ts +5 -0
- package/dist/lib/setup-code.d.ts.map +1 -0
- package/dist/lib/setup-code.js +35 -0
- package/dist/lib/setup-code.js.map +1 -0
- package/dist/lib/structured-output.d.ts +9 -0
- package/dist/lib/structured-output.d.ts.map +1 -0
- package/dist/lib/structured-output.js +72 -0
- package/dist/lib/structured-output.js.map +1 -0
- package/dist/lib/system-prompt.d.ts +30 -0
- package/dist/lib/system-prompt.d.ts.map +1 -0
- package/dist/lib/system-prompt.js +66 -0
- package/dist/lib/system-prompt.js.map +1 -0
- package/dist/lib/task-type.d.ts +18 -0
- package/dist/lib/task-type.d.ts.map +1 -0
- package/dist/lib/task-type.js +92 -0
- package/dist/lib/task-type.js.map +1 -0
- package/dist/lib/think-tags.d.ts +37 -0
- package/dist/lib/think-tags.d.ts.map +1 -0
- package/dist/lib/think-tags.js +203 -0
- package/dist/lib/think-tags.js.map +1 -0
- package/dist/lib/tool-args.d.ts +36 -0
- package/dist/lib/tool-args.d.ts.map +1 -0
- package/dist/lib/tool-args.js +190 -0
- package/dist/lib/tool-args.js.map +1 -0
- package/dist/lib/tool-call-rescue.d.ts +62 -0
- package/dist/lib/tool-call-rescue.d.ts.map +1 -0
- package/dist/lib/tool-call-rescue.js +258 -0
- package/dist/lib/tool-call-rescue.js.map +1 -0
- package/dist/lib/tool-capability.d.ts +13 -0
- package/dist/lib/tool-capability.d.ts.map +1 -0
- package/dist/lib/tool-capability.js +70 -0
- package/dist/lib/tool-capability.js.map +1 -0
- package/dist/lib/tool-validate.d.ts +44 -0
- package/dist/lib/tool-validate.d.ts.map +1 -0
- package/dist/lib/tool-validate.js +165 -0
- package/dist/lib/tool-validate.js.map +1 -0
- package/dist/lib/ttfb-budget.d.ts +13 -0
- package/dist/lib/ttfb-budget.d.ts.map +1 -0
- package/dist/lib/ttfb-budget.js +153 -0
- package/dist/lib/ttfb-budget.js.map +1 -0
- package/dist/lib/url-guard.d.ts +47 -0
- package/dist/lib/url-guard.d.ts.map +1 -0
- package/dist/lib/url-guard.js +229 -0
- package/dist/lib/url-guard.js.map +1 -0
- package/dist/lib/wake-detect.d.ts +12 -0
- package/dist/lib/wake-detect.d.ts.map +1 -0
- package/dist/lib/wake-detect.js +99 -0
- package/dist/lib/wake-detect.js.map +1 -0
- package/dist/middleware/errorHandler.d.ts +3 -0
- package/dist/middleware/errorHandler.d.ts.map +1 -0
- package/dist/middleware/errorHandler.js +45 -0
- package/dist/middleware/errorHandler.js.map +1 -0
- package/dist/middleware/rateLimit.d.ts +4 -0
- package/dist/middleware/rateLimit.d.ts.map +1 -0
- package/dist/middleware/rateLimit.js +125 -0
- package/dist/middleware/rateLimit.js.map +1 -0
- package/dist/middleware/requireAuth.d.ts +3 -0
- package/dist/middleware/requireAuth.d.ts.map +1 -0
- package/dist/middleware/requireAuth.js +17 -0
- package/dist/middleware/requireAuth.js.map +1 -0
- package/dist/providers/aclide.d.ts +18 -0
- package/dist/providers/aclide.d.ts.map +1 -0
- package/dist/providers/aclide.js +174 -0
- package/dist/providers/aclide.js.map +1 -0
- package/dist/providers/aihorde.d.ts +42 -0
- package/dist/providers/aihorde.d.ts.map +1 -0
- package/dist/providers/aihorde.js +192 -0
- package/dist/providers/aihorde.js.map +1 -0
- package/dist/providers/airforce.d.ts +11 -0
- package/dist/providers/airforce.d.ts.map +1 -0
- package/dist/providers/airforce.js +32 -0
- package/dist/providers/airforce.js.map +1 -0
- package/dist/providers/base.d.ts +177 -0
- package/dist/providers/base.d.ts.map +1 -0
- package/dist/providers/base.js +354 -0
- package/dist/providers/base.js.map +1 -0
- package/dist/providers/blaze.d.ts +11 -0
- package/dist/providers/blaze.d.ts.map +1 -0
- package/dist/providers/blaze.js +33 -0
- package/dist/providers/blaze.js.map +1 -0
- package/dist/providers/blockrun.d.ts +14 -0
- package/dist/providers/blockrun.d.ts.map +1 -0
- package/dist/providers/blockrun.js +36 -0
- package/dist/providers/blockrun.js.map +1 -0
- package/dist/providers/clod.d.ts +11 -0
- package/dist/providers/clod.d.ts.map +1 -0
- package/dist/providers/clod.js +39 -0
- package/dist/providers/clod.js.map +1 -0
- package/dist/providers/cloudflare.d.ts +15 -0
- package/dist/providers/cloudflare.d.ts.map +1 -0
- package/dist/providers/cloudflare.js +175 -0
- package/dist/providers/cloudflare.js.map +1 -0
- package/dist/providers/cohere.d.ts +11 -0
- package/dist/providers/cohere.d.ts.map +1 -0
- package/dist/providers/cohere.js +117 -0
- package/dist/providers/cohere.js.map +1 -0
- package/dist/providers/dreamprompting.d.ts +11 -0
- package/dist/providers/dreamprompting.d.ts.map +1 -0
- package/dist/providers/dreamprompting.js +41 -0
- package/dist/providers/dreamprompting.js.map +1 -0
- package/dist/providers/electronhub.d.ts +10 -0
- package/dist/providers/electronhub.d.ts.map +1 -0
- package/dist/providers/electronhub.js +56 -0
- package/dist/providers/electronhub.js.map +1 -0
- package/dist/providers/experiential.d.ts +10 -0
- package/dist/providers/experiential.d.ts.map +1 -0
- package/dist/providers/experiential.js +33 -0
- package/dist/providers/experiential.js.map +1 -0
- package/dist/providers/gizmo.d.ts +15 -0
- package/dist/providers/gizmo.d.ts.map +1 -0
- package/dist/providers/gizmo.js +54 -0
- package/dist/providers/gizmo.js.map +1 -0
- package/dist/providers/google.d.ts +30 -0
- package/dist/providers/google.d.ts.map +1 -0
- package/dist/providers/google.js +766 -0
- package/dist/providers/google.js.map +1 -0
- package/dist/providers/index.d.ts +13 -0
- package/dist/providers/index.d.ts.map +1 -0
- package/dist/providers/index.js +561 -0
- package/dist/providers/index.js.map +1 -0
- package/dist/providers/llmtr.d.ts +15 -0
- package/dist/providers/llmtr.d.ts.map +1 -0
- package/dist/providers/llmtr.js +51 -0
- package/dist/providers/llmtr.js.map +1 -0
- package/dist/providers/logfare.d.ts +11 -0
- package/dist/providers/logfare.d.ts.map +1 -0
- package/dist/providers/logfare.js +33 -0
- package/dist/providers/logfare.js.map +1 -0
- package/dist/providers/lucidity.d.ts +11 -0
- package/dist/providers/lucidity.d.ts.map +1 -0
- package/dist/providers/lucidity.js +37 -0
- package/dist/providers/lucidity.js.map +1 -0
- package/dist/providers/modelscope.d.ts +50 -0
- package/dist/providers/modelscope.d.ts.map +1 -0
- package/dist/providers/modelscope.js +143 -0
- package/dist/providers/modelscope.js.map +1 -0
- package/dist/providers/moondream.d.ts +17 -0
- package/dist/providers/moondream.d.ts.map +1 -0
- package/dist/providers/moondream.js +140 -0
- package/dist/providers/moondream.js.map +1 -0
- package/dist/providers/openai-compat.d.ts +140 -0
- package/dist/providers/openai-compat.d.ts.map +1 -0
- package/dist/providers/openai-compat.js +537 -0
- package/dist/providers/openai-compat.js.map +1 -0
- package/dist/providers/opencode-free.d.ts +24 -0
- package/dist/providers/opencode-free.d.ts.map +1 -0
- package/dist/providers/opencode-free.js +262 -0
- package/dist/providers/opencode-free.js.map +1 -0
- package/dist/providers/pollinations.d.ts +32 -0
- package/dist/providers/pollinations.d.ts.map +1 -0
- package/dist/providers/pollinations.js +65 -0
- package/dist/providers/pollinations.js.map +1 -0
- package/dist/providers/router9.d.ts +16 -0
- package/dist/providers/router9.d.ts.map +1 -0
- package/dist/providers/router9.js +42 -0
- package/dist/providers/router9.js.map +1 -0
- package/dist/providers/sail.d.ts +44 -0
- package/dist/providers/sail.d.ts.map +1 -0
- package/dist/providers/sail.js +302 -0
- package/dist/providers/sail.js.map +1 -0
- package/dist/providers/septor.d.ts +10 -0
- package/dist/providers/septor.d.ts.map +1 -0
- package/dist/providers/septor.js +36 -0
- package/dist/providers/septor.js.map +1 -0
- package/dist/providers/speechify.d.ts +14 -0
- package/dist/providers/speechify.d.ts.map +1 -0
- package/dist/providers/speechify.js +28 -0
- package/dist/providers/speechify.js.map +1 -0
- package/dist/providers/speka.d.ts +12 -0
- package/dist/providers/speka.d.ts.map +1 -0
- package/dist/providers/speka.js +37 -0
- package/dist/providers/speka.js.map +1 -0
- package/dist/providers/waterfall.d.ts +11 -0
- package/dist/providers/waterfall.d.ts.map +1 -0
- package/dist/providers/waterfall.js +33 -0
- package/dist/providers/waterfall.js.map +1 -0
- package/dist/providers/xkiro.d.ts +71 -0
- package/dist/providers/xkiro.d.ts.map +1 -0
- package/dist/providers/xkiro.js +119 -0
- package/dist/providers/xkiro.js.map +1 -0
- package/dist/providers/zhipu.d.ts +43 -0
- package/dist/providers/zhipu.d.ts.map +1 -0
- package/dist/providers/zhipu.js +95 -0
- package/dist/providers/zhipu.js.map +1 -0
- package/dist/routes/analytics.d.ts +2 -0
- package/dist/routes/analytics.d.ts.map +1 -0
- package/dist/routes/analytics.js +738 -0
- package/dist/routes/analytics.js.map +1 -0
- package/dist/routes/anthropic.d.ts +7 -0
- package/dist/routes/anthropic.d.ts.map +1 -0
- package/dist/routes/anthropic.js +1076 -0
- package/dist/routes/anthropic.js.map +1 -0
- package/dist/routes/auth.d.ts +2 -0
- package/dist/routes/auth.d.ts.map +1 -0
- package/dist/routes/auth.js +310 -0
- package/dist/routes/auth.js.map +1 -0
- package/dist/routes/backups.d.ts +2 -0
- package/dist/routes/backups.d.ts.map +1 -0
- package/dist/routes/backups.js +127 -0
- package/dist/routes/backups.js.map +1 -0
- package/dist/routes/cache.d.ts +2 -0
- package/dist/routes/cache.d.ts.map +1 -0
- package/dist/routes/cache.js +40 -0
- package/dist/routes/cache.js.map +1 -0
- package/dist/routes/client-profiles.d.ts +2 -0
- package/dist/routes/client-profiles.d.ts.map +1 -0
- package/dist/routes/client-profiles.js +128 -0
- package/dist/routes/client-profiles.js.map +1 -0
- package/dist/routes/compression.d.ts +2 -0
- package/dist/routes/compression.d.ts.map +1 -0
- package/dist/routes/compression.js +82 -0
- package/dist/routes/compression.js.map +1 -0
- package/dist/routes/conversations.d.ts +3 -0
- package/dist/routes/conversations.d.ts.map +1 -0
- package/dist/routes/conversations.js +215 -0
- package/dist/routes/conversations.js.map +1 -0
- package/dist/routes/docs.d.ts +2 -0
- package/dist/routes/docs.d.ts.map +1 -0
- package/dist/routes/docs.js +21 -0
- package/dist/routes/docs.js.map +1 -0
- package/dist/routes/embeddings.d.ts +2 -0
- package/dist/routes/embeddings.d.ts.map +1 -0
- package/dist/routes/embeddings.js +255 -0
- package/dist/routes/embeddings.js.map +1 -0
- package/dist/routes/fallback.d.ts +2 -0
- package/dist/routes/fallback.d.ts.map +1 -0
- package/dist/routes/fallback.js +631 -0
- package/dist/routes/fallback.js.map +1 -0
- package/dist/routes/free-tier.d.ts +22 -0
- package/dist/routes/free-tier.d.ts.map +1 -0
- package/dist/routes/free-tier.js +175 -0
- package/dist/routes/free-tier.js.map +1 -0
- package/dist/routes/gemini.d.ts +2 -0
- package/dist/routes/gemini.d.ts.map +1 -0
- package/dist/routes/gemini.js +218 -0
- package/dist/routes/gemini.js.map +1 -0
- package/dist/routes/health.d.ts +2 -0
- package/dist/routes/health.d.ts.map +1 -0
- package/dist/routes/health.js +72 -0
- package/dist/routes/health.js.map +1 -0
- package/dist/routes/keys.d.ts +17 -0
- package/dist/routes/keys.d.ts.map +1 -0
- package/dist/routes/keys.js +1787 -0
- package/dist/routes/keys.js.map +1 -0
- package/dist/routes/logs.d.ts +2 -0
- package/dist/routes/logs.d.ts.map +1 -0
- package/dist/routes/logs.js +88 -0
- package/dist/routes/logs.js.map +1 -0
- package/dist/routes/mcp.d.ts +4 -0
- package/dist/routes/mcp.d.ts.map +1 -0
- package/dist/routes/mcp.js +563 -0
- package/dist/routes/mcp.js.map +1 -0
- package/dist/routes/media.d.ts +2 -0
- package/dist/routes/media.d.ts.map +1 -0
- package/dist/routes/media.js +180 -0
- package/dist/routes/media.js.map +1 -0
- package/dist/routes/models.d.ts +2 -0
- package/dist/routes/models.d.ts.map +1 -0
- package/dist/routes/models.js +439 -0
- package/dist/routes/models.js.map +1 -0
- package/dist/routes/ollama.d.ts +7 -0
- package/dist/routes/ollama.d.ts.map +1 -0
- package/dist/routes/ollama.js +595 -0
- package/dist/routes/ollama.js.map +1 -0
- package/dist/routes/profiles.d.ts +3 -0
- package/dist/routes/profiles.d.ts.map +1 -0
- package/dist/routes/profiles.js +431 -0
- package/dist/routes/profiles.js.map +1 -0
- package/dist/routes/proxy.d.ts +32 -0
- package/dist/routes/proxy.d.ts.map +1 -0
- package/dist/routes/proxy.js +2793 -0
- package/dist/routes/proxy.js.map +1 -0
- package/dist/routes/responses.d.ts +1161 -0
- package/dist/routes/responses.d.ts.map +1 -0
- package/dist/routes/responses.js +1521 -0
- package/dist/routes/responses.js.map +1 -0
- package/dist/routes/settings.d.ts +2 -0
- package/dist/routes/settings.d.ts.map +1 -0
- package/dist/routes/settings.js +549 -0
- package/dist/routes/settings.js.map +1 -0
- package/dist/routes/status.d.ts +3 -0
- package/dist/routes/status.d.ts.map +1 -0
- package/dist/routes/status.js +214 -0
- package/dist/routes/status.js.map +1 -0
- package/dist/routes/update.d.ts +31 -0
- package/dist/routes/update.d.ts.map +1 -0
- package/dist/routes/update.js +536 -0
- package/dist/routes/update.js.map +1 -0
- package/dist/routes/url-tokens.d.ts +2 -0
- package/dist/routes/url-tokens.d.ts.map +1 -0
- package/dist/routes/url-tokens.js +31 -0
- package/dist/routes/url-tokens.js.map +1 -0
- package/dist/scripts/export-catalog.d.ts +2 -0
- package/dist/scripts/export-catalog.d.ts.map +1 -0
- package/dist/scripts/export-catalog.js +114 -0
- package/dist/scripts/export-catalog.js.map +1 -0
- package/dist/scripts/rotate-encryption-key.d.ts +81 -0
- package/dist/scripts/rotate-encryption-key.d.ts.map +1 -0
- package/dist/scripts/rotate-encryption-key.js +232 -0
- package/dist/scripts/rotate-encryption-key.js.map +1 -0
- package/dist/scripts/routing-sim.d.ts +2 -0
- package/dist/scripts/routing-sim.d.ts.map +1 -0
- package/dist/scripts/routing-sim.js +130 -0
- package/dist/scripts/routing-sim.js.map +1 -0
- package/dist/scripts/test-all-models.d.ts +2 -0
- package/dist/scripts/test-all-models.d.ts.map +1 -0
- package/dist/scripts/test-all-models.js +56 -0
- package/dist/scripts/test-all-models.js.map +1 -0
- package/dist/services/anthropic-map.d.ts +39 -0
- package/dist/services/anthropic-map.d.ts.map +1 -0
- package/dist/services/anthropic-map.js +138 -0
- package/dist/services/anthropic-map.js.map +1 -0
- package/dist/services/auth.d.ts +30 -0
- package/dist/services/auth.d.ts.map +1 -0
- package/dist/services/auth.js +124 -0
- package/dist/services/auth.js.map +1 -0
- package/dist/services/auto-discover.d.ts +77 -0
- package/dist/services/auto-discover.d.ts.map +1 -0
- package/dist/services/auto-discover.js +1047 -0
- package/dist/services/auto-discover.js.map +1 -0
- package/dist/services/backups.d.ts +67 -0
- package/dist/services/backups.d.ts.map +1 -0
- package/dist/services/backups.js +444 -0
- package/dist/services/backups.js.map +1 -0
- package/dist/services/builtin-model-discovery.d.ts +107 -0
- package/dist/services/builtin-model-discovery.d.ts.map +1 -0
- package/dist/services/builtin-model-discovery.js +296 -0
- package/dist/services/builtin-model-discovery.js.map +1 -0
- package/dist/services/cache.d.ts +190 -0
- package/dist/services/cache.d.ts.map +1 -0
- package/dist/services/cache.js +617 -0
- package/dist/services/cache.js.map +1 -0
- package/dist/services/catalog-sync.d.ts +257 -0
- package/dist/services/catalog-sync.d.ts.map +1 -0
- package/dist/services/catalog-sync.js +1041 -0
- package/dist/services/catalog-sync.js.map +1 -0
- package/dist/services/compression/config.d.ts +33 -0
- package/dist/services/compression/config.d.ts.map +1 -0
- package/dist/services/compression/config.js +158 -0
- package/dist/services/compression/config.js.map +1 -0
- package/dist/services/compression/engines/aging.d.ts +2 -0
- package/dist/services/compression/engines/aging.d.ts.map +1 -0
- package/dist/services/compression/engines/aging.js +53 -0
- package/dist/services/compression/engines/aging.js.map +1 -0
- package/dist/services/compression/engines/custom-filters.d.ts +4 -0
- package/dist/services/compression/engines/custom-filters.d.ts.map +1 -0
- package/dist/services/compression/engines/custom-filters.js +49 -0
- package/dist/services/compression/engines/custom-filters.js.map +1 -0
- package/dist/services/compression/engines/dedup.d.ts +2 -0
- package/dist/services/compression/engines/dedup.d.ts.map +1 -0
- package/dist/services/compression/engines/dedup.js +61 -0
- package/dist/services/compression/engines/dedup.js.map +1 -0
- package/dist/services/compression/engines/filter-definitions.d.ts +75 -0
- package/dist/services/compression/engines/filter-definitions.d.ts.map +1 -0
- package/dist/services/compression/engines/filter-definitions.js +45 -0
- package/dist/services/compression/engines/filter-definitions.js.map +1 -0
- package/dist/services/compression/engines/hard-budget.d.ts +2 -0
- package/dist/services/compression/engines/hard-budget.d.ts.map +1 -0
- package/dist/services/compression/engines/hard-budget.js +52 -0
- package/dist/services/compression/engines/hard-budget.js.map +1 -0
- package/dist/services/compression/engines/index.d.ts +9 -0
- package/dist/services/compression/engines/index.d.ts.map +1 -0
- package/dist/services/compression/engines/index.js +9 -0
- package/dist/services/compression/engines/index.js.map +1 -0
- package/dist/services/compression/engines/jsoncompact.d.ts +3 -0
- package/dist/services/compression/engines/jsoncompact.d.ts.map +1 -0
- package/dist/services/compression/engines/jsoncompact.js +145 -0
- package/dist/services/compression/engines/jsoncompact.js.map +1 -0
- package/dist/services/compression/engines/lite.d.ts +2 -0
- package/dist/services/compression/engines/lite.d.ts.map +1 -0
- package/dist/services/compression/engines/lite.js +34 -0
- package/dist/services/compression/engines/lite.js.map +1 -0
- package/dist/services/compression/engines/read-lifecycle.d.ts +2 -0
- package/dist/services/compression/engines/read-lifecycle.d.ts.map +1 -0
- package/dist/services/compression/engines/read-lifecycle.js +70 -0
- package/dist/services/compression/engines/read-lifecycle.js.map +1 -0
- package/dist/services/compression/engines/relevance.d.ts +2 -0
- package/dist/services/compression/engines/relevance.d.ts.map +1 -0
- package/dist/services/compression/engines/relevance.js +74 -0
- package/dist/services/compression/engines/relevance.js.map +1 -0
- package/dist/services/compression/engines/toolfilter.d.ts +2 -0
- package/dist/services/compression/engines/toolfilter.d.ts.map +1 -0
- package/dist/services/compression/engines/toolfilter.js +195 -0
- package/dist/services/compression/engines/toolfilter.js.map +1 -0
- package/dist/services/compression/fidelity-gate.d.ts +12 -0
- package/dist/services/compression/fidelity-gate.d.ts.map +1 -0
- package/dist/services/compression/fidelity-gate.js +90 -0
- package/dist/services/compression/fidelity-gate.js.map +1 -0
- package/dist/services/compression/helpers.d.ts +9 -0
- package/dist/services/compression/helpers.d.ts.map +1 -0
- package/dist/services/compression/helpers.js +52 -0
- package/dist/services/compression/helpers.js.map +1 -0
- package/dist/services/compression/pipeline.d.ts +7 -0
- package/dist/services/compression/pipeline.d.ts.map +1 -0
- package/dist/services/compression/pipeline.js +245 -0
- package/dist/services/compression/pipeline.js.map +1 -0
- package/dist/services/compression/preservation.d.ts +27 -0
- package/dist/services/compression/preservation.d.ts.map +1 -0
- package/dist/services/compression/preservation.js +109 -0
- package/dist/services/compression/preservation.js.map +1 -0
- package/dist/services/compression/registry.d.ts +6 -0
- package/dist/services/compression/registry.d.ts.map +1 -0
- package/dist/services/compression/registry.js +16 -0
- package/dist/services/compression/registry.js.map +1 -0
- package/dist/services/compression/stats.d.ts +29 -0
- package/dist/services/compression/stats.d.ts.map +1 -0
- package/dist/services/compression/stats.js +55 -0
- package/dist/services/compression/stats.js.map +1 -0
- package/dist/services/compression/types.d.ts +90 -0
- package/dist/services/compression/types.d.ts.map +1 -0
- package/dist/services/compression/types.js +2 -0
- package/dist/services/compression/types.js.map +1 -0
- package/dist/services/context-handoff.d.ts +22 -0
- package/dist/services/context-handoff.d.ts.map +1 -0
- package/dist/services/context-handoff.js +164 -0
- package/dist/services/context-handoff.js.map +1 -0
- package/dist/services/cooldown-probe.d.ts +24 -0
- package/dist/services/cooldown-probe.d.ts.map +1 -0
- package/dist/services/cooldown-probe.js +181 -0
- package/dist/services/cooldown-probe.js.map +1 -0
- package/dist/services/custom-endpoint.d.ts +58 -0
- package/dist/services/custom-endpoint.d.ts.map +1 -0
- package/dist/services/custom-endpoint.js +167 -0
- package/dist/services/custom-endpoint.js.map +1 -0
- package/dist/services/custom-media-register.d.ts +22 -0
- package/dist/services/custom-media-register.d.ts.map +1 -0
- package/dist/services/custom-media-register.js +45 -0
- package/dist/services/custom-media-register.js.map +1 -0
- package/dist/services/custom-model-register.d.ts +50 -0
- package/dist/services/custom-model-register.d.ts.map +1 -0
- package/dist/services/custom-model-register.js +103 -0
- package/dist/services/custom-model-register.js.map +1 -0
- package/dist/services/custom-model-seed.d.ts +18 -0
- package/dist/services/custom-model-seed.d.ts.map +1 -0
- package/dist/services/custom-model-seed.js +42 -0
- package/dist/services/custom-model-seed.js.map +1 -0
- package/dist/services/custom-model-sync.d.ts +37 -0
- package/dist/services/custom-model-sync.d.ts.map +1 -0
- package/dist/services/custom-model-sync.js +118 -0
- package/dist/services/custom-model-sync.js.map +1 -0
- package/dist/services/custom-model-tombstone.d.ts +5 -0
- package/dist/services/custom-model-tombstone.d.ts.map +1 -0
- package/dist/services/custom-model-tombstone.js +30 -0
- package/dist/services/custom-model-tombstone.js.map +1 -0
- package/dist/services/declarative-config.d.ts +356 -0
- package/dist/services/declarative-config.d.ts.map +1 -0
- package/dist/services/declarative-config.js +427 -0
- package/dist/services/declarative-config.js.map +1 -0
- package/dist/services/degradation.d.ts +40 -0
- package/dist/services/degradation.d.ts.map +1 -0
- package/dist/services/degradation.js +119 -0
- package/dist/services/degradation.js.map +1 -0
- package/dist/services/embeddings.d.ts +76 -0
- package/dist/services/embeddings.d.ts.map +1 -0
- package/dist/services/embeddings.js +319 -0
- package/dist/services/embeddings.js.map +1 -0
- package/dist/services/fusion.d.ts +162 -0
- package/dist/services/fusion.d.ts.map +1 -0
- package/dist/services/fusion.js +789 -0
- package/dist/services/fusion.js.map +1 -0
- package/dist/services/gemini-map.d.ts +14 -0
- package/dist/services/gemini-map.d.ts.map +1 -0
- package/dist/services/gemini-map.js +78 -0
- package/dist/services/gemini-map.js.map +1 -0
- package/dist/services/health.d.ts +52 -0
- package/dist/services/health.d.ts.map +1 -0
- package/dist/services/health.js +352 -0
- package/dist/services/health.js.map +1 -0
- package/dist/services/idempotency.d.ts +49 -0
- package/dist/services/idempotency.d.ts.map +1 -0
- package/dist/services/idempotency.js +140 -0
- package/dist/services/idempotency.js.map +1 -0
- package/dist/services/key-budget.d.ts +74 -0
- package/dist/services/key-budget.d.ts.map +1 -0
- package/dist/services/key-budget.js +200 -0
- package/dist/services/key-budget.js.map +1 -0
- package/dist/services/media.d.ts +123 -0
- package/dist/services/media.d.ts.map +1 -0
- package/dist/services/media.js +954 -0
- package/dist/services/media.js.map +1 -0
- package/dist/services/model-discovery.d.ts +126 -0
- package/dist/services/model-discovery.d.ts.map +1 -0
- package/dist/services/model-discovery.js +662 -0
- package/dist/services/model-discovery.js.map +1 -0
- package/dist/services/model-groups.d.ts +149 -0
- package/dist/services/model-groups.d.ts.map +1 -0
- package/dist/services/model-groups.js +314 -0
- package/dist/services/model-groups.js.map +1 -0
- package/dist/services/model-listing.d.ts +18 -0
- package/dist/services/model-listing.d.ts.map +1 -0
- package/dist/services/model-listing.js +94 -0
- package/dist/services/model-listing.js.map +1 -0
- package/dist/services/model-retirement.d.ts +29 -0
- package/dist/services/model-retirement.d.ts.map +1 -0
- package/dist/services/model-retirement.js +86 -0
- package/dist/services/model-retirement.js.map +1 -0
- package/dist/services/model-state.d.ts +90 -0
- package/dist/services/model-state.d.ts.map +1 -0
- package/dist/services/model-state.js +342 -0
- package/dist/services/model-state.js.map +1 -0
- package/dist/services/model-weight-overrides.d.ts +48 -0
- package/dist/services/model-weight-overrides.d.ts.map +1 -0
- package/dist/services/model-weight-overrides.js +127 -0
- package/dist/services/model-weight-overrides.js.map +1 -0
- package/dist/services/notification-emitter.d.ts +50 -0
- package/dist/services/notification-emitter.d.ts.map +1 -0
- package/dist/services/notification-emitter.js +177 -0
- package/dist/services/notification-emitter.js.map +1 -0
- package/dist/services/opencode-free-sync.d.ts +29 -0
- package/dist/services/opencode-free-sync.d.ts.map +1 -0
- package/dist/services/opencode-free-sync.js +206 -0
- package/dist/services/opencode-free-sync.js.map +1 -0
- package/dist/services/penalty-inspector.d.ts +56 -0
- package/dist/services/penalty-inspector.d.ts.map +1 -0
- package/dist/services/penalty-inspector.js +167 -0
- package/dist/services/penalty-inspector.js.map +1 -0
- package/dist/services/profile-models.d.ts +5 -0
- package/dist/services/profile-models.d.ts.map +1 -0
- package/dist/services/profile-models.js +62 -0
- package/dist/services/profile-models.js.map +1 -0
- package/dist/services/provider-credential.d.ts +19 -0
- package/dist/services/provider-credential.d.ts.map +1 -0
- package/dist/services/provider-credential.js +51 -0
- package/dist/services/provider-credential.js.map +1 -0
- package/dist/services/provider-quota.d.ts +64 -0
- package/dist/services/provider-quota.d.ts.map +1 -0
- package/dist/services/provider-quota.js +563 -0
- package/dist/services/provider-quota.js.map +1 -0
- package/dist/services/quirks.d.ts +0 -0
- package/dist/services/quirks.d.ts.map +1 -0
- package/dist/services/quirks.js +0 -0
- package/dist/services/quirks.js.map +1 -0
- package/dist/services/quota-forecast.d.ts +50 -0
- package/dist/services/quota-forecast.d.ts.map +1 -0
- package/dist/services/quota-forecast.js +150 -0
- package/dist/services/quota-forecast.js.map +1 -0
- package/dist/services/quota-outlook.d.ts +4 -0
- package/dist/services/quota-outlook.d.ts.map +1 -0
- package/dist/services/quota-outlook.js +92 -0
- package/dist/services/quota-outlook.js.map +1 -0
- package/dist/services/ratelimit.d.ts +249 -0
- package/dist/services/ratelimit.d.ts.map +1 -0
- package/dist/services/ratelimit.js +1182 -0
- package/dist/services/ratelimit.js.map +1 -0
- package/dist/services/request-retention.d.ts +60 -0
- package/dist/services/request-retention.d.ts.map +1 -0
- package/dist/services/request-retention.js +229 -0
- package/dist/services/request-retention.js.map +1 -0
- package/dist/services/router.d.ts +384 -0
- package/dist/services/router.d.ts.map +1 -0
- package/dist/services/router.js +1986 -0
- package/dist/services/router.js.map +1 -0
- package/dist/services/scoring.d.ts +156 -0
- package/dist/services/scoring.d.ts.map +1 -0
- package/dist/services/scoring.js +435 -0
- package/dist/services/scoring.js.map +1 -0
- package/dist/services/url-tokens.d.ts +15 -0
- package/dist/services/url-tokens.d.ts.map +1 -0
- package/dist/services/url-tokens.js +55 -0
- package/dist/services/url-tokens.js.map +1 -0
- package/package.json +77 -4
- package/README.md +0 -3
|
@@ -0,0 +1,1986 @@
|
|
|
1
|
+
import { getDb, getSetting, setSetting } from '../db/index.js';
|
|
2
|
+
import { getProvider, hasProvider, resolveProvider } from '../providers/index.js';
|
|
3
|
+
import { decrypt } from '../lib/crypto.js';
|
|
4
|
+
import { decryptProxyUrl } from '../lib/key-proxy.js';
|
|
5
|
+
import { canMakeRequest, canUseTokens, isOnCooldown, canUseProvider, canUseProviderMinute, canUseProviderTokens, canUseKeyConcurrency, acquireLease, releaseLease, getSoonestCooldownExpiry, modelWindowUsedFraction, } from './ratelimit.js';
|
|
6
|
+
import { BANDIT_PRESETS, DEFAULT_STRATEGY, reliabilityPosterior, expectedReliability, sampleBeta, speedScore, intelligenceScore, intelligenceComposite, headroomFactor, rateWindowHeadroomFactor, rateLimitFactor, combineScore, peakAdjustedWeights, taskAdjustedWeights, TASK_WEIGHT_SHARE, isValidPeakHour, isValidTimezone, DEFAULT_PEAK_HOURS, observedSpeedRank, TIMEOUT_LATENCY_CAP_MS, } from './scoring.js';
|
|
7
|
+
import { TIMEOUT_ERROR_MARKERS } from '../lib/error-classify.js';
|
|
8
|
+
import { checkMonthlyBudget, reserveMonthlyBudget } from './key-budget.js';
|
|
9
|
+
import { applyModelWeightOverride, getModelWeightOverrides } from './model-weight-overrides.js';
|
|
10
|
+
import { modelsWithOverriddenField } from './model-state.js';
|
|
11
|
+
import { parseBudget } from '../lib/budget.js';
|
|
12
|
+
import { platformDropsResponseFormat } from '../lib/sampling-params.js';
|
|
13
|
+
import { isUnifyEnabled, getModelGroups, resolveRequestedIdForDispatch } from './model-groups.js';
|
|
14
|
+
import { getActiveProfileId } from './profile-models.js';
|
|
15
|
+
import { customEndpointKeyIds } from './custom-endpoint.js';
|
|
16
|
+
import { isDegraded } from './degradation.js';
|
|
17
|
+
import { modelStatsKey, endpointScopeForBaseUrl } from '../lib/endpoint-scope.js';
|
|
18
|
+
import { isToolBenched } from '../lib/tool-capability.js';
|
|
19
|
+
import { parseModelScope, scopeAllows } from '../lib/model-scope.js';
|
|
20
|
+
import { getKeyQuotaHeadroom, inferQuotaPoolKey } from './provider-quota.js';
|
|
21
|
+
class RouteError extends Error {
|
|
22
|
+
status;
|
|
23
|
+
// Per-model disposition of the chain at the moment routing gave up: one line
|
|
24
|
+
// per considered model with the reason it could not serve (no key, cooldown,
|
|
25
|
+
// provider cap, rpm/rpd, tpm/tpd, context too small, …). Populated only on the
|
|
26
|
+
// synchronous "all exhausted" throw, where NO upstream was tried and nothing
|
|
27
|
+
// else logs WHY the pool was empty (issue _1: opaque routing_error 429).
|
|
28
|
+
diagnostics;
|
|
29
|
+
constructor(message, status, diagnostics) {
|
|
30
|
+
super(message);
|
|
31
|
+
this.status = status;
|
|
32
|
+
this.diagnostics = diagnostics;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
// Human-readable retry ETA from a cooldown expiry timestamp (#423). Null when
|
|
36
|
+
// nothing is cooling down or it already lapsed.
|
|
37
|
+
export function formatResetEta(soonestResetMs, now = Date.now()) {
|
|
38
|
+
if (soonestResetMs == null)
|
|
39
|
+
return null;
|
|
40
|
+
const deltaMs = soonestResetMs - now;
|
|
41
|
+
if (deltaMs <= 0)
|
|
42
|
+
return null;
|
|
43
|
+
const secs = Math.round(deltaMs / 1000);
|
|
44
|
+
if (secs < 90)
|
|
45
|
+
return `~${secs}s`;
|
|
46
|
+
const mins = Math.round(secs / 60);
|
|
47
|
+
if (mins < 90)
|
|
48
|
+
return `~${mins}m`;
|
|
49
|
+
return `~${Math.round(mins / 60)}h`;
|
|
50
|
+
}
|
|
51
|
+
const EXHAUSTION_ADVICE = 'Add more API keys or wait for rate limits to reset.';
|
|
52
|
+
// Roll the per-model diagnostics (see RouteError.diagnostics) up into a short,
|
|
53
|
+
// client-safe summary so an exhausted caller learns WHY the pool was empty
|
|
54
|
+
// instead of a bare "All models exhausted" (#423). Buckets are aggregate
|
|
55
|
+
// counts only — no key material, no per-key detail. Classifies off the whole
|
|
56
|
+
// line (model ids can contain ':' so splitting label from reason is unsafe).
|
|
57
|
+
export function summarizeExhaustion(diag, soonestResetMs, now = Date.now(), keylessSkipped = 0) {
|
|
58
|
+
const eta = formatResetEta(soonestResetMs, now);
|
|
59
|
+
const etaSuffix = eta ? ` Soonest reset ${eta}.` : '';
|
|
60
|
+
// Models dropped before the walk (#423 follow-up) are reported separately and
|
|
61
|
+
// never counted as "routes checked": they were never candidates, and folding
|
|
62
|
+
// them into the total inflates the pool the caller thinks it has.
|
|
63
|
+
const keylessSuffix = keylessSkipped > 0
|
|
64
|
+
? ` ${keylessSkipped} model${keylessSkipped === 1 ? '' : 's'} skipped: no key configured for their platform.`
|
|
65
|
+
: '';
|
|
66
|
+
if (!diag || diag.length === 0) {
|
|
67
|
+
return `All models exhausted. ${EXHAUSTION_ADVICE}${etaSuffix}${keylessSuffix}`;
|
|
68
|
+
}
|
|
69
|
+
const counts = {};
|
|
70
|
+
const bump = (bucket) => { counts[bucket] = (counts[bucket] ?? 0) + 1; };
|
|
71
|
+
for (const line of diag) {
|
|
72
|
+
const l = line.toLowerCase();
|
|
73
|
+
if (l.includes('no provider registered'))
|
|
74
|
+
bump('unsupported provider');
|
|
75
|
+
else if (/no enabled\+healthy key|no usable key|decrypt-error/.test(l))
|
|
76
|
+
bump('no usable key configured');
|
|
77
|
+
else if (l.includes('< estimated'))
|
|
78
|
+
bump('prompt too large for the model');
|
|
79
|
+
else if (l.includes('no vision support'))
|
|
80
|
+
bump('model lacks vision');
|
|
81
|
+
else if (l.includes('no tool-calling support'))
|
|
82
|
+
bump('model lacks tool-calling');
|
|
83
|
+
else if (l.includes('drops response_format'))
|
|
84
|
+
bump('platform cannot honor response_format');
|
|
85
|
+
else if (/ruled out|already-failed/.test(l))
|
|
86
|
+
bump('failed earlier this request');
|
|
87
|
+
else if (/cooldown|rpm|rpd|tpm|tpd|provider-daily-cap/.test(l))
|
|
88
|
+
bump('rate-limited or on cooldown');
|
|
89
|
+
else
|
|
90
|
+
bump('unavailable');
|
|
91
|
+
}
|
|
92
|
+
// Most actionable buckets first.
|
|
93
|
+
const order = [
|
|
94
|
+
'rate-limited or on cooldown',
|
|
95
|
+
'no usable key configured',
|
|
96
|
+
'prompt too large for the model',
|
|
97
|
+
'model lacks vision',
|
|
98
|
+
'model lacks tool-calling',
|
|
99
|
+
'platform cannot honor response_format',
|
|
100
|
+
'failed earlier this request',
|
|
101
|
+
'unsupported provider',
|
|
102
|
+
'unavailable',
|
|
103
|
+
];
|
|
104
|
+
const parts = order.filter(b => counts[b]).map(b => `${counts[b]} ${b}`);
|
|
105
|
+
const total = diag.length;
|
|
106
|
+
return `All models exhausted: ${total} route${total === 1 ? '' : 's'} checked (${parts.join(', ')}). ${EXHAUSTION_ADVICE}${etaSuffix}${keylessSuffix}`;
|
|
107
|
+
}
|
|
108
|
+
// ── Routing token estimate: cap the reserved OUTPUT, not the full max_tokens ──
|
|
109
|
+
// A client can request a huge max_tokens (e.g. 32000) it will never actually
|
|
110
|
+
// emit. Reserving that full amount against every model's context window and TPM
|
|
111
|
+
// budget falsely excludes the entire free pool (TPM 6k-30k) and returns a bogus
|
|
112
|
+
// "all models exhausted" 429 with ZERO upstream calls (#470). Providers meter
|
|
113
|
+
// ACTUAL tokens, so under-reserving only risks an upstream 429/413 the retry
|
|
114
|
+
// loop already handles, whereas over-reserving starves routing. For routing and
|
|
115
|
+
// the context-window / TPM filters we therefore reserve at most this many output
|
|
116
|
+
// tokens; the INPUT estimate is still counted in full so a genuinely large
|
|
117
|
+
// prompt still (correctly) skips a too-small model.
|
|
118
|
+
export const OUTPUT_RESERVE_CAP = 2000;
|
|
119
|
+
/**
|
|
120
|
+
* Output tokens to reserve for routing/filter purposes: the requested max_tokens
|
|
121
|
+
* clamped to OUTPUT_RESERVE_CAP (default 1000 when the client omitted it, matching
|
|
122
|
+
* the historical fallback). Callers add this to their INPUT estimate before
|
|
123
|
+
* calling routeRequest / routePinnedModel.
|
|
124
|
+
*/
|
|
125
|
+
export function routingReserveTokens(requestedMaxTokens) {
|
|
126
|
+
const requested = requestedMaxTokens != null && requestedMaxTokens > 0 ? requestedMaxTokens : 1000;
|
|
127
|
+
return Math.min(requested, OUTPUT_RESERVE_CAP);
|
|
128
|
+
}
|
|
129
|
+
// Round-robin index per platform
|
|
130
|
+
const roundRobinIndex = new Map();
|
|
131
|
+
// ── Dynamic priority: track 429s per model and demote accordingly ──
|
|
132
|
+
// Key: model_db_id → { count, lastHit, penalty }
|
|
133
|
+
const rateLimitPenalties = new Map();
|
|
134
|
+
// Penalty decays over time so models recover
|
|
135
|
+
const PENALTY_PER_429 = 3; // each 429 adds this many priority positions
|
|
136
|
+
const PENALTY_PER_FAIL = 1; // each non-limit upstream failure (5xx/timeout/empty stream)
|
|
137
|
+
const MAX_PENALTY = 10; // cap so a model doesn't sink forever
|
|
138
|
+
const DECAY_INTERVAL_MS = 2 * 60 * 1000; // penalty decays every 2 minutes
|
|
139
|
+
const DECAY_AMOUNT = 1; // remove this much penalty per decay interval
|
|
140
|
+
/**
|
|
141
|
+
* Record an upstream failure for a model — increases its penalty so it sinks in
|
|
142
|
+
* priority. Default weight is the LIGHT one for ordinary upstream failures
|
|
143
|
+
* (5xx/timeout/empty stream, +1); callers that know they saw a hard limit
|
|
144
|
+
* signal (429/402) pass the heavier weight — a quota limit is the stronger,
|
|
145
|
+
* longer-lived health cue.
|
|
146
|
+
*/
|
|
147
|
+
export function recordModelFailure(modelDbId, weight = PENALTY_PER_FAIL) {
|
|
148
|
+
const existing = rateLimitPenalties.get(modelDbId);
|
|
149
|
+
const now = Date.now();
|
|
150
|
+
if (existing) {
|
|
151
|
+
const decaySteps = Math.floor((now - existing.lastHit) / DECAY_INTERVAL_MS);
|
|
152
|
+
existing.penalty = Math.max(0, existing.penalty - decaySteps * DECAY_AMOUNT);
|
|
153
|
+
existing.count++;
|
|
154
|
+
existing.lastHit = now;
|
|
155
|
+
existing.penalty = Math.min(existing.penalty + weight, MAX_PENALTY);
|
|
156
|
+
}
|
|
157
|
+
else {
|
|
158
|
+
rateLimitPenalties.set(modelDbId, { count: 1, lastHit: now, penalty: weight });
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
/**
|
|
162
|
+
* Record a 429 for a model — heavier penalty (priority demotion) than ordinary
|
|
163
|
+
* failures, since a rate-limit signal is the strongest short-term health cue.
|
|
164
|
+
*/
|
|
165
|
+
export function recordRateLimitHit(modelDbId) {
|
|
166
|
+
recordModelFailure(modelDbId, PENALTY_PER_429);
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Record a success for a model — reduces its penalty so it rises back up.
|
|
170
|
+
*/
|
|
171
|
+
export function recordSuccess(modelDbId) {
|
|
172
|
+
const existing = rateLimitPenalties.get(modelDbId);
|
|
173
|
+
if (existing) {
|
|
174
|
+
existing.penalty = Math.max(0, existing.penalty - 1);
|
|
175
|
+
if (existing.penalty === 0) {
|
|
176
|
+
rateLimitPenalties.delete(modelDbId);
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Get the current penalty for a model (with time-based decay).
|
|
182
|
+
* Pure read — does not mutate the entry; decay is applied lazily only when
|
|
183
|
+
* recording a new hit (recordRateLimitHit) so the clock isn't reset on every
|
|
184
|
+
* routing call.
|
|
185
|
+
*/
|
|
186
|
+
function getPenalty(modelDbId) {
|
|
187
|
+
const entry = rateLimitPenalties.get(modelDbId);
|
|
188
|
+
if (!entry)
|
|
189
|
+
return 0;
|
|
190
|
+
const elapsed = Date.now() - entry.lastHit;
|
|
191
|
+
const decaySteps = Math.floor(elapsed / DECAY_INTERVAL_MS);
|
|
192
|
+
const decayed = Math.max(0, entry.penalty - decaySteps * DECAY_AMOUNT);
|
|
193
|
+
if (decayed === 0) {
|
|
194
|
+
rateLimitPenalties.delete(modelDbId);
|
|
195
|
+
return 0;
|
|
196
|
+
}
|
|
197
|
+
return decayed;
|
|
198
|
+
}
|
|
199
|
+
/**
|
|
200
|
+
* Get current penalties for all models (for the API/dashboard).
|
|
201
|
+
*/
|
|
202
|
+
export function getAllPenalties() {
|
|
203
|
+
const result = [];
|
|
204
|
+
for (const [modelDbId, entry] of rateLimitPenalties) {
|
|
205
|
+
const penalty = getPenalty(modelDbId);
|
|
206
|
+
if (penalty > 0) {
|
|
207
|
+
result.push({ modelDbId, count: entry.count, penalty });
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
return result.sort((a, b) => b.penalty - a.penalty);
|
|
211
|
+
}
|
|
212
|
+
/**
|
|
213
|
+
* Operator clear (#952): forget every model's penalty at once and report how
|
|
214
|
+
* many models were carrying one. Pairs with clearAllCooldowns — a pool stuck
|
|
215
|
+
* behind day-long benches also has its models sunk by penalties, and lifting
|
|
216
|
+
* one without the other leaves the router still avoiding them.
|
|
217
|
+
*/
|
|
218
|
+
export function clearAllPenalties() {
|
|
219
|
+
const count = getAllPenalties().length;
|
|
220
|
+
rateLimitPenalties.clear();
|
|
221
|
+
return count;
|
|
222
|
+
}
|
|
223
|
+
// ── Routing strategy (persisted) ────────────────────────────────────────────
|
|
224
|
+
const STRATEGY_KEY = 'routing_strategy';
|
|
225
|
+
const CUSTOM_WEIGHTS_KEY = 'routing_custom_weights';
|
|
226
|
+
const EXPLORE_KEY = 'routing_explore_enabled';
|
|
227
|
+
const PEAK_ADJUST_KEY = 'routing_peak_hours_adjust';
|
|
228
|
+
const PEAK_START_KEY = 'routing_peak_start_hour';
|
|
229
|
+
const PEAK_END_KEY = 'routing_peak_end_hour';
|
|
230
|
+
const PEAK_TZ_KEY = 'routing_peak_timezone';
|
|
231
|
+
const COMMUNITY_PRIOR_KEY = 'routing_community_prior';
|
|
232
|
+
const COMMUNITY_PRIOR_ENABLED_KEY = 'routing_community_prior_enabled';
|
|
233
|
+
// Headroom guardrail thresholds (#899): the remaining-budget fraction at which
|
|
234
|
+
// demotion begins and the score floor at 0 remaining. Stored as decimals
|
|
235
|
+
// (0.2 = 20%). Absent/invalid values fall back to the scoring.ts constants so
|
|
236
|
+
// existing installs are untouched.
|
|
237
|
+
export const HEADROOM_RAMP_START_KEY = 'routing_headroom_ramp_start';
|
|
238
|
+
export const HEADROOM_FLOOR_KEY = 'routing_headroom_floor';
|
|
239
|
+
export function getHeadroomThresholds() {
|
|
240
|
+
const read = (key) => {
|
|
241
|
+
const raw = getSetting(key);
|
|
242
|
+
if (raw === undefined || raw.trim() === '')
|
|
243
|
+
return undefined;
|
|
244
|
+
const n = Number(raw);
|
|
245
|
+
return Number.isFinite(n) && n >= 0 && n <= 1 ? n : undefined;
|
|
246
|
+
};
|
|
247
|
+
return { rampStart: read(HEADROOM_RAMP_START_KEY), floor: read(HEADROOM_FLOOR_KEY) };
|
|
248
|
+
}
|
|
249
|
+
// null clears a threshold back to the default; undefined leaves it untouched.
|
|
250
|
+
export function setHeadroomThresholds(rampStart, floor) {
|
|
251
|
+
const db = getDb();
|
|
252
|
+
const apply = (key, value) => {
|
|
253
|
+
if (value === undefined)
|
|
254
|
+
return;
|
|
255
|
+
if (value === null) {
|
|
256
|
+
db.prepare('DELETE FROM settings WHERE key = ?').run(key); // back to default
|
|
257
|
+
return;
|
|
258
|
+
}
|
|
259
|
+
if (!Number.isFinite(value) || value < 0 || value > 1) {
|
|
260
|
+
throw new Error(`Invalid value ${value} for ${key} (must be 0..1)`);
|
|
261
|
+
}
|
|
262
|
+
setSetting(key, String(value));
|
|
263
|
+
};
|
|
264
|
+
apply(HEADROOM_RAMP_START_KEY, rampStart);
|
|
265
|
+
apply(HEADROOM_FLOOR_KEY, floor);
|
|
266
|
+
}
|
|
267
|
+
// ── Task-type weight share (persisted) ─────────────────────────────────────
|
|
268
|
+
// #1127 follow-up: the bandit bias applied for a declared/derived task type
|
|
269
|
+
// moves `share` of one axis onto the other (code: speed → intelligence; chat:
|
|
270
|
+
// the reverse). The default matches the scoring.ts constant; operators can
|
|
271
|
+
// tune it 0..1 (0 = bias disabled) without a code change. Absent/invalid
|
|
272
|
+
// values fall back to the constant so existing installs are untouched.
|
|
273
|
+
export const TASK_WEIGHT_SHARE_KEY = 'routing_task_weight_share';
|
|
274
|
+
export function getTaskWeightShare() {
|
|
275
|
+
const raw = getSetting(TASK_WEIGHT_SHARE_KEY);
|
|
276
|
+
if (raw === undefined || raw.trim() === '')
|
|
277
|
+
return TASK_WEIGHT_SHARE;
|
|
278
|
+
const n = Number(raw);
|
|
279
|
+
return Number.isFinite(n) && n >= 0 && n <= 1 ? n : TASK_WEIGHT_SHARE;
|
|
280
|
+
}
|
|
281
|
+
// null clears back to the default; a value outside 0..1 throws.
|
|
282
|
+
export function setTaskWeightShare(value) {
|
|
283
|
+
if (value === null) {
|
|
284
|
+
getDb().prepare('DELETE FROM settings WHERE key = ?').run(TASK_WEIGHT_SHARE_KEY);
|
|
285
|
+
return;
|
|
286
|
+
}
|
|
287
|
+
if (!Number.isFinite(value) || value < 0 || value > 1) {
|
|
288
|
+
throw new Error(`Invalid value ${value} for ${TASK_WEIGHT_SHARE_KEY} (must be 0..1)`);
|
|
289
|
+
}
|
|
290
|
+
setSetting(TASK_WEIGHT_SHARE_KEY, String(value));
|
|
291
|
+
}
|
|
292
|
+
/** Chance per request that an unmeasured model gets tried first when the
|
|
293
|
+
* exploration toggle is on. The bandit's Thompson sampling already explores
|
|
294
|
+
* automatically; this guarantees a floor so models with no reliability/speed
|
|
295
|
+
* data still get sampled instead of being starved by prior-heavy rivals. */
|
|
296
|
+
export const EXPLORE_CHANCE = 0.1;
|
|
297
|
+
/** A model counts as "has data" once its decay-weighted success+failure
|
|
298
|
+
* pseudo-count reaches this many samples. */
|
|
299
|
+
export const EXPLORE_MIN_SAMPLES = 5;
|
|
300
|
+
const VALID_STRATEGIES = ['priority', 'balanced', 'smartest', 'fastest', 'reliable', 'custom'];
|
|
301
|
+
export function getRoutingStrategy() {
|
|
302
|
+
const raw = getSetting(STRATEGY_KEY);
|
|
303
|
+
return (raw && VALID_STRATEGIES.includes(raw))
|
|
304
|
+
? raw
|
|
305
|
+
: DEFAULT_STRATEGY;
|
|
306
|
+
}
|
|
307
|
+
export function setRoutingStrategy(strategy) {
|
|
308
|
+
if (!VALID_STRATEGIES.includes(strategy)) {
|
|
309
|
+
throw new Error(`Unknown routing strategy: ${strategy}`);
|
|
310
|
+
}
|
|
311
|
+
setSetting(STRATEGY_KEY, strategy);
|
|
312
|
+
}
|
|
313
|
+
// ── Exploration toggle (persisted) ─────────────────────────────────────────
|
|
314
|
+
// On by default: unmeasured models get a guaranteed chance to be tried
|
|
315
|
+
// (EXPLORE_CHANCE) so they acquire reliability/speed samples instead of
|
|
316
|
+
// losing every bandit draw. Set to '0' to opt out from the dashboard.
|
|
317
|
+
export function getExploreEnabled() {
|
|
318
|
+
return getSetting(EXPLORE_KEY) !== '0';
|
|
319
|
+
}
|
|
320
|
+
export function setExploreEnabled(enabled) {
|
|
321
|
+
setSetting(EXPLORE_KEY, enabled ? '1' : '0');
|
|
322
|
+
}
|
|
323
|
+
// ── Peak-hours adjustment (persisted, off by default) ──────────────────────
|
|
324
|
+
// Opt-in time-of-day reweighting (#760). Everything about it is operator-set:
|
|
325
|
+
// whether it runs at all, the window, and the timezone the window is read in.
|
|
326
|
+
// With the flag off, weightsFor returns the presets byte-for-byte, so an
|
|
327
|
+
// install that never touches this setting routes exactly as it did before.
|
|
328
|
+
export function getPeakHoursConfig() {
|
|
329
|
+
const startRaw = Number.parseInt(getSetting(PEAK_START_KEY) ?? '', 10);
|
|
330
|
+
const endRaw = Number.parseInt(getSetting(PEAK_END_KEY) ?? '', 10);
|
|
331
|
+
const tzRaw = getSetting(PEAK_TZ_KEY);
|
|
332
|
+
return {
|
|
333
|
+
enabled: getSetting(PEAK_ADJUST_KEY) === '1',
|
|
334
|
+
startHour: isValidPeakHour(startRaw) ? startRaw : DEFAULT_PEAK_HOURS.startHour,
|
|
335
|
+
endHour: isValidPeakHour(endRaw) ? endRaw : DEFAULT_PEAK_HOURS.endHour,
|
|
336
|
+
timezone: isValidTimezone(tzRaw) ? tzRaw : DEFAULT_PEAK_HOURS.timezone,
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
/** Persist any subset of the peak-hours settings. Throws on an out-of-range
|
|
340
|
+
* hour or an unknown IANA timezone so a bad PUT is rejected at the API rather
|
|
341
|
+
* than silently stored and then ignored on read. */
|
|
342
|
+
export function setPeakHoursConfig(patch) {
|
|
343
|
+
if (patch.startHour !== undefined && !isValidPeakHour(patch.startHour)) {
|
|
344
|
+
throw new Error('peakStartHour must be an integer between 0 and 23');
|
|
345
|
+
}
|
|
346
|
+
if (patch.endHour !== undefined && !isValidPeakHour(patch.endHour)) {
|
|
347
|
+
throw new Error('peakEndHour must be an integer between 0 and 23');
|
|
348
|
+
}
|
|
349
|
+
if (patch.timezone !== undefined && !isValidTimezone(patch.timezone)) {
|
|
350
|
+
throw new Error('peakTimezone must be a valid IANA timezone name');
|
|
351
|
+
}
|
|
352
|
+
if (patch.enabled !== undefined)
|
|
353
|
+
setSetting(PEAK_ADJUST_KEY, patch.enabled ? '1' : '0');
|
|
354
|
+
if (patch.startHour !== undefined)
|
|
355
|
+
setSetting(PEAK_START_KEY, String(patch.startHour));
|
|
356
|
+
if (patch.endHour !== undefined)
|
|
357
|
+
setSetting(PEAK_END_KEY, String(patch.endHour));
|
|
358
|
+
if (patch.timezone !== undefined)
|
|
359
|
+
setSetting(PEAK_TZ_KEY, patch.timezone);
|
|
360
|
+
}
|
|
361
|
+
// ── Key selection strategy (persisted) ─────────────────────────────────────
|
|
362
|
+
// Which of a platform's several keys to reach for, once a model has been
|
|
363
|
+
// picked. Independent of the routing strategy on purpose (#919): the strategy
|
|
364
|
+
// enum drives the MODEL bandit, so putting a key policy in it would make
|
|
365
|
+
// choosing a key policy also throw away the model ranking.
|
|
366
|
+
// 'auto' — unchanged: per-key bandit score when there is data,
|
|
367
|
+
// round-robin otherwise.
|
|
368
|
+
// 'least-remaining' — additionally rank by observed remaining quota, roomiest
|
|
369
|
+
// key first, so the key closest to its cap is held back
|
|
370
|
+
// instead of being the next one to 429.
|
|
371
|
+
const KEY_SELECTION_KEY = 'key_selection_strategy';
|
|
372
|
+
const VALID_KEY_SELECTIONS = ['auto', 'least-remaining'];
|
|
373
|
+
export const DEFAULT_KEY_SELECTION = 'auto';
|
|
374
|
+
export function getKeySelectionStrategy() {
|
|
375
|
+
const raw = getSetting(KEY_SELECTION_KEY);
|
|
376
|
+
return (raw && VALID_KEY_SELECTIONS.includes(raw))
|
|
377
|
+
? raw
|
|
378
|
+
: DEFAULT_KEY_SELECTION;
|
|
379
|
+
}
|
|
380
|
+
export function setKeySelectionStrategy(strategy) {
|
|
381
|
+
if (!VALID_KEY_SELECTIONS.includes(strategy)) {
|
|
382
|
+
throw new Error(`Unknown key selection strategy: ${strategy}`);
|
|
383
|
+
}
|
|
384
|
+
setSetting(KEY_SELECTION_KEY, strategy);
|
|
385
|
+
}
|
|
386
|
+
// ── Custom weights (persisted) ──────────────────────────────────────────────
|
|
387
|
+
// User-tuned weight vector for the 'custom' strategy. Stored normalized (sums
|
|
388
|
+
// to 1) so the dashboard percentages read cleanly; combineScore would tolerate
|
|
389
|
+
// any non-negative vector regardless. Falls back to the balanced preset until
|
|
390
|
+
// the user has saved their own.
|
|
391
|
+
export function getCustomWeights() {
|
|
392
|
+
const raw = getSetting(CUSTOM_WEIGHTS_KEY);
|
|
393
|
+
if (raw) {
|
|
394
|
+
try {
|
|
395
|
+
const w = JSON.parse(raw);
|
|
396
|
+
if ([w.reliability, w.speed, w.intelligence].every(v => Number.isFinite(v) && v >= 0) &&
|
|
397
|
+
w.reliability + w.speed + w.intelligence > 0) {
|
|
398
|
+
return { reliability: w.reliability, speed: w.speed, intelligence: w.intelligence };
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
catch { /* corrupt setting → fall through to default */ }
|
|
402
|
+
}
|
|
403
|
+
return { ...BANDIT_PRESETS.balanced };
|
|
404
|
+
}
|
|
405
|
+
export function setCustomWeights(weights) {
|
|
406
|
+
const { reliability, speed, intelligence } = weights;
|
|
407
|
+
if (![reliability, speed, intelligence].every(v => Number.isFinite(v) && v >= 0)) {
|
|
408
|
+
throw new Error('Custom weights must be non-negative numbers');
|
|
409
|
+
}
|
|
410
|
+
const sum = reliability + speed + intelligence;
|
|
411
|
+
if (sum <= 0) {
|
|
412
|
+
throw new Error('Custom weights must not all be zero');
|
|
413
|
+
}
|
|
414
|
+
setSetting(CUSTOM_WEIGHTS_KEY, JSON.stringify({
|
|
415
|
+
reliability: reliability / sum,
|
|
416
|
+
speed: speed / sum,
|
|
417
|
+
intelligence: intelligence / sum,
|
|
418
|
+
}));
|
|
419
|
+
}
|
|
420
|
+
/** Ceiling on a single prior's effective sample size. Local counts are
|
|
421
|
+
* decay-weighted (2-day half-life — a busy install still only carries on the
|
|
422
|
+
* order of a hundred effective samples), so an unbounded, undecayed community
|
|
423
|
+
* count would drown local evidence forever and collapse the Thompson-sampling
|
|
424
|
+
* variance to zero. Capping at ~50 pseudo-observations keeps a prior worth
|
|
425
|
+
* roughly half the local evidence at most: enough to seed a brand-new model,
|
|
426
|
+
* cheap for real local traffic to override. */
|
|
427
|
+
export const COMMUNITY_PRIOR_MAX_SAMPLES = 50;
|
|
428
|
+
/** Validate a raw prior map and cap each entry's effective sample size.
|
|
429
|
+
* Shared by the read path and the write path so a value is bounded no matter
|
|
430
|
+
* how it entered (fresh set, legacy stored blob, hand-edited settings row).
|
|
431
|
+
* Invalid entries (negative, all-zero, no ':') are dropped; oversized ones
|
|
432
|
+
* are rescaled preserving the success/failure ratio (980/20 → 49/1). */
|
|
433
|
+
function sanitizeCommunityPriors(priors) {
|
|
434
|
+
const clean = {};
|
|
435
|
+
if (!priors || typeof priors !== 'object')
|
|
436
|
+
return clean;
|
|
437
|
+
for (const [key, v] of Object.entries(priors)) {
|
|
438
|
+
if (key.includes(':') &&
|
|
439
|
+
v && typeof v === 'object' &&
|
|
440
|
+
Number.isFinite(v.successes) && v.successes >= 0 &&
|
|
441
|
+
Number.isFinite(v.failures) && v.failures >= 0 &&
|
|
442
|
+
v.successes + v.failures > 0) {
|
|
443
|
+
const total = v.successes + v.failures;
|
|
444
|
+
const scale = total > COMMUNITY_PRIOR_MAX_SAMPLES ? COMMUNITY_PRIOR_MAX_SAMPLES / total : 1;
|
|
445
|
+
const entry = { successes: Math.round(v.successes * scale), failures: Math.round(v.failures * scale) };
|
|
446
|
+
if (entry.successes + entry.failures > 0)
|
|
447
|
+
clean[key] = entry;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
return clean;
|
|
451
|
+
}
|
|
452
|
+
// Parsed-prior cache, same 60s shape as the stats cache: routing reads the map
|
|
453
|
+
// once per chain entry (and once per key in orderKeysByScore), so hitting
|
|
454
|
+
// sqlite + JSON.parse on every lookup is pure waste. Invalidated by the two
|
|
455
|
+
// setters and by refreshStatsCache, so tests and future ingestion see writes
|
|
456
|
+
// immediately.
|
|
457
|
+
let communityPriorCache = null;
|
|
458
|
+
let communityPriorCacheTime = 0;
|
|
459
|
+
function communityPriorState() {
|
|
460
|
+
const now = Date.now();
|
|
461
|
+
if (communityPriorCache && now - communityPriorCacheTime < CACHE_TTL_MS)
|
|
462
|
+
return communityPriorCache;
|
|
463
|
+
let map = {};
|
|
464
|
+
const raw = getSetting(COMMUNITY_PRIOR_KEY);
|
|
465
|
+
if (raw) {
|
|
466
|
+
try {
|
|
467
|
+
map = sanitizeCommunityPriors(JSON.parse(raw));
|
|
468
|
+
}
|
|
469
|
+
catch { /* corrupt setting → no priors */ }
|
|
470
|
+
}
|
|
471
|
+
communityPriorCache = { map, enabled: getSetting(COMMUNITY_PRIOR_ENABLED_KEY) === '1' };
|
|
472
|
+
communityPriorCacheTime = now;
|
|
473
|
+
return communityPriorCache;
|
|
474
|
+
}
|
|
475
|
+
function invalidateCommunityPriorCache() {
|
|
476
|
+
communityPriorCache = null;
|
|
477
|
+
}
|
|
478
|
+
/** Whether stored community priors are folded into the posterior. Default off. */
|
|
479
|
+
export function getCommunityPriorEnabled() {
|
|
480
|
+
return communityPriorState().enabled;
|
|
481
|
+
}
|
|
482
|
+
export function setCommunityPriorEnabled(enabled) {
|
|
483
|
+
setSetting(COMMUNITY_PRIOR_ENABLED_KEY, enabled ? '1' : '0');
|
|
484
|
+
invalidateCommunityPriorCache();
|
|
485
|
+
}
|
|
486
|
+
/** Community prior for one model, or undefined when none is stored.
|
|
487
|
+
* Raw read — ignores the enabled flag; routing goes through
|
|
488
|
+
* activeCommunityPrior, which honors it. */
|
|
489
|
+
export function getCommunityPrior(platform, modelId, endpointScope) {
|
|
490
|
+
return communityPriorState().map[modelStatsKey(platform, modelId, endpointScope)];
|
|
491
|
+
}
|
|
492
|
+
/** Gated read for routing: undefined unless the opt-in flag is on. */
|
|
493
|
+
function activeCommunityPrior(platform, modelId, endpointScope) {
|
|
494
|
+
const state = communityPriorState();
|
|
495
|
+
return state.enabled ? state.map[modelStatsKey(platform, modelId, endpointScope)] : undefined;
|
|
496
|
+
}
|
|
497
|
+
/** Replace the whole community-prior map (e.g. after an aggregation fetch).
|
|
498
|
+
* Invalid entries are dropped and oversized ones capped, never stored raw. */
|
|
499
|
+
export function setCommunityPriors(priors) {
|
|
500
|
+
const clean = sanitizeCommunityPriors(priors);
|
|
501
|
+
setSetting(COMMUNITY_PRIOR_KEY, JSON.stringify(clean));
|
|
502
|
+
invalidateCommunityPriorCache();
|
|
503
|
+
return Object.keys(clean).length;
|
|
504
|
+
}
|
|
505
|
+
/** Active weights plus whether the peak-hours adjustment (#760) changed them.
|
|
506
|
+
* With the setting off (the default) `weights` is the preset itself and
|
|
507
|
+
* `adjusted` is false, so nothing about routing moves with the clock.
|
|
508
|
+
* priority/custom are the operator's explicit choice and are never rewritten. */
|
|
509
|
+
function weightsWithPeak(strategy) {
|
|
510
|
+
if (strategy === 'priority')
|
|
511
|
+
return { weights: null, adjusted: false };
|
|
512
|
+
if (strategy === 'custom')
|
|
513
|
+
return { weights: getCustomWeights(), adjusted: false };
|
|
514
|
+
return peakAdjustedWeights(BANDIT_PRESETS[strategy], strategy, getPeakHoursConfig());
|
|
515
|
+
}
|
|
516
|
+
function weightsFor(strategy) {
|
|
517
|
+
return weightsWithPeak(strategy).weights;
|
|
518
|
+
}
|
|
519
|
+
/** The weight vector routing will use right now for the active strategy, and
|
|
520
|
+
* whether the peak-hours adjustment moved it. Cheap (settings reads only) —
|
|
521
|
+
* for the PUT /routing echo, which must not pay for a full score sweep. */
|
|
522
|
+
export function getActiveRoutingWeights() {
|
|
523
|
+
return weightsWithPeak(getRoutingStrategy());
|
|
524
|
+
}
|
|
525
|
+
// ── Analytics stats cache (decay-weighted) ──────────────────────────────────
|
|
526
|
+
// Instead of the fork's flat 7-day window (where a model that degrades today
|
|
527
|
+
// keeps a stale week-long average), each request is weighted by an exponential
|
|
528
|
+
// decay so recent behavior dominates while older data still stabilizes the
|
|
529
|
+
// estimate. We aggregate by (model, integer day age) in SQL — at most ~7 rows
|
|
530
|
+
// per model — then apply the per-bucket decay weight in JS.
|
|
531
|
+
const WINDOW_MS = 7 * 24 * 60 * 60 * 1000;
|
|
532
|
+
const HALF_LIFE_DAYS = 2; // a 2-day-old request counts half as much as a fresh one
|
|
533
|
+
const CACHE_TTL_MS = 60 * 1000;
|
|
534
|
+
// Keyed by modelStatsKey(): "platform:model_id" for catalog models, and
|
|
535
|
+
// "custom:model_id@base_url" for a relay model that carries an endpoint scope
|
|
536
|
+
// (#651). A single-endpoint install produces the same keys it always did.
|
|
537
|
+
let statsCache = null;
|
|
538
|
+
let keyStatsCache = null; // "platform:model_id:key_id"
|
|
539
|
+
let statsCacheTime = 0;
|
|
540
|
+
function decayWeight(ageDays) {
|
|
541
|
+
return Math.pow(0.5, Math.max(0, ageDays) / HALF_LIFE_DAYS);
|
|
542
|
+
}
|
|
543
|
+
// SQL predicate for "this row is a timed-out request" (#619). A timeout is an
|
|
544
|
+
// error row whose text carries one of the shared timeout markers
|
|
545
|
+
// (lib/error-classify.ts), which is also what the failover attempt trail
|
|
546
|
+
// classifies on. 'canceled' rows (#752 — client hung up) never reach this
|
|
547
|
+
// predicate: the stats query below filters them out entirely, because a
|
|
548
|
+
// vanished client says nothing about the model's reliability or speed. The
|
|
549
|
+
// markers are hard-coded lowercase identifiers from our own source, never
|
|
550
|
+
// user input, so interpolating them into the LIKE list is safe.
|
|
551
|
+
const IS_TIMEOUT_SQL = `(status != 'success' AND (${TIMEOUT_ERROR_MARKERS.map(m => `LOWER(COALESCE(error, '')) LIKE '%${m}%'`).join(' OR ')}))`;
|
|
552
|
+
/** api_keys.id → endpoint scope, for every custom credential on record (#651). */
|
|
553
|
+
function customEndpointScopes(db) {
|
|
554
|
+
const rows = db.prepare("SELECT id, base_url FROM api_keys WHERE platform = 'custom'")
|
|
555
|
+
.all();
|
|
556
|
+
return new Map(rows.map(r => [r.id, endpointScopeForBaseUrl(r.base_url)]));
|
|
557
|
+
}
|
|
558
|
+
export function refreshStatsCache(db, force = false) {
|
|
559
|
+
if (!force && statsCache && Date.now() - statsCacheTime < CACHE_TTL_MS)
|
|
560
|
+
return;
|
|
561
|
+
// Re-read the community priors alongside the stats they season, so a forced
|
|
562
|
+
// refresh (tests, admin actions) never routes on a stale prior snapshot.
|
|
563
|
+
invalidateCommunityPriorCache();
|
|
564
|
+
const since = new Date(Date.now() - WINDOW_MS).toISOString();
|
|
565
|
+
// Grouped by (model, key, day age): still a handful of rows per model — key
|
|
566
|
+
// count × ≤7 day buckets — so the finer grain keeps the same one-query,
|
|
567
|
+
// 60s-cached shape. Aggregated two ways below: rolled up per model (ordering)
|
|
568
|
+
// and per key (in-model key selection, #580).
|
|
569
|
+
const buckets = db.prepare(`
|
|
570
|
+
SELECT platform, model_id, key_id,
|
|
571
|
+
CAST((julianday('now') - julianday(created_at)) AS INTEGER) AS age_days,
|
|
572
|
+
COUNT(*) AS total,
|
|
573
|
+
SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) AS successes,
|
|
574
|
+
SUM(CASE WHEN status = 'success' THEN output_tokens ELSE 0 END) AS succ_out,
|
|
575
|
+
SUM(CASE WHEN status = 'success' THEN latency_ms ELSE 0 END) AS succ_lat,
|
|
576
|
+
SUM(CASE WHEN status = 'success' AND ttfb_ms IS NOT NULL THEN ttfb_ms ELSE 0 END) AS succ_ttfb_sum,
|
|
577
|
+
SUM(CASE WHEN status = 'success' AND ttfb_ms IS NOT NULL THEN 1 ELSE 0 END) AS succ_ttfb_cnt,
|
|
578
|
+
SUM(CASE WHEN ${IS_TIMEOUT_SQL} THEN 1 ELSE 0 END) AS timeouts,
|
|
579
|
+
SUM(CASE WHEN ${IS_TIMEOUT_SQL} THEN MIN(MAX(latency_ms, 0), ${TIMEOUT_LATENCY_CAP_MS}) ELSE 0 END) AS timeout_lat
|
|
580
|
+
FROM requests
|
|
581
|
+
WHERE created_at >= ? AND status <> 'canceled'
|
|
582
|
+
GROUP BY platform, model_id, key_id, age_days
|
|
583
|
+
`).all(since);
|
|
584
|
+
const emptyAcc = () => ({ wSucc: 0, wFail: 0, wOut: 0, wLat: 0, wTtfbSum: 0, wTtfbCnt: 0, wTimeouts: 0 });
|
|
585
|
+
const addBucket = (a, w, b) => {
|
|
586
|
+
a.wSucc += w * b.successes;
|
|
587
|
+
a.wFail += w * (b.total - b.successes);
|
|
588
|
+
a.wOut += w * b.succ_out;
|
|
589
|
+
a.wLat += w * (b.succ_lat + b.timeout_lat);
|
|
590
|
+
a.wTtfbSum += w * (b.succ_ttfb_sum + b.timeout_lat);
|
|
591
|
+
a.wTtfbCnt += w * (b.succ_ttfb_cnt + b.timeouts);
|
|
592
|
+
a.wTimeouts += w * b.timeouts;
|
|
593
|
+
};
|
|
594
|
+
// Which endpoint each custom credential belongs to, so a request logged
|
|
595
|
+
// against relay A's key lands in relay A's bucket and nowhere else (#651).
|
|
596
|
+
// `requests` has always recorded key_id, so pre-migration history splits
|
|
597
|
+
// correctly too; rows whose key is gone (or that never had one) fall into the
|
|
598
|
+
// un-scoped bucket, which only un-scoped rows read.
|
|
599
|
+
const scopeByKeyId = customEndpointScopes(db);
|
|
600
|
+
const scopeOf = (platform, keyId) => platform === 'custom' && keyId != null ? (scopeByKeyId.get(keyId) ?? '') : '';
|
|
601
|
+
const acc = new Map();
|
|
602
|
+
const keyAcc = new Map();
|
|
603
|
+
for (const b of buckets) {
|
|
604
|
+
const key = modelStatsKey(b.platform, b.model_id, scopeOf(b.platform, b.key_id));
|
|
605
|
+
const w = decayWeight(b.age_days);
|
|
606
|
+
let a = acc.get(key);
|
|
607
|
+
if (!a)
|
|
608
|
+
acc.set(key, a = emptyAcc());
|
|
609
|
+
addBucket(a, w, b);
|
|
610
|
+
if (b.key_id != null) {
|
|
611
|
+
const kk = `${key}:${b.key_id}`;
|
|
612
|
+
let ka = keyAcc.get(kk);
|
|
613
|
+
if (!ka)
|
|
614
|
+
keyAcc.set(kk, ka = emptyAcc());
|
|
615
|
+
addBucket(ka, w, b);
|
|
616
|
+
}
|
|
617
|
+
}
|
|
618
|
+
// Calendar-month token usage per model, for the headroom guardrail.
|
|
619
|
+
const usageRows = db.prepare(`
|
|
620
|
+
SELECT platform, model_id, key_id, COALESCE(SUM(input_tokens + output_tokens), 0) AS used
|
|
621
|
+
FROM requests
|
|
622
|
+
WHERE created_at >= datetime('now', 'start of month')
|
|
623
|
+
AND request_type = 'chat'
|
|
624
|
+
GROUP BY platform, model_id, key_id
|
|
625
|
+
`).all();
|
|
626
|
+
const usageMap = new Map();
|
|
627
|
+
for (const r of usageRows) {
|
|
628
|
+
const key = modelStatsKey(r.platform, r.model_id, scopeOf(r.platform, r.key_id));
|
|
629
|
+
usageMap.set(key, (usageMap.get(key) ?? 0) + r.used);
|
|
630
|
+
}
|
|
631
|
+
const next = new Map();
|
|
632
|
+
for (const [key, a] of acc) {
|
|
633
|
+
next.set(key, {
|
|
634
|
+
successes: a.wSucc,
|
|
635
|
+
failures: a.wFail,
|
|
636
|
+
tokPerSec: a.wLat > 0 ? (a.wOut * 1000) / a.wLat : 0,
|
|
637
|
+
avgTtfbMs: a.wTtfbCnt > 0 ? a.wTtfbSum / a.wTtfbCnt : null,
|
|
638
|
+
monthlyUsedTokens: usageMap.get(key) ?? 0,
|
|
639
|
+
speedSamples: a.wSucc + a.wTimeouts,
|
|
640
|
+
});
|
|
641
|
+
}
|
|
642
|
+
// Models with month usage but no recent window data still need a headroom number.
|
|
643
|
+
for (const [key, used] of usageMap) {
|
|
644
|
+
if (!next.has(key)) {
|
|
645
|
+
next.set(key, { successes: 0, failures: 0, tokPerSec: 0, avgTtfbMs: null, monthlyUsedTokens: used, speedSamples: 0 });
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
const nextKeys = new Map();
|
|
649
|
+
for (const [kk, a] of keyAcc) {
|
|
650
|
+
nextKeys.set(kk, {
|
|
651
|
+
successes: a.wSucc,
|
|
652
|
+
failures: a.wFail,
|
|
653
|
+
tokPerSec: a.wLat > 0 ? (a.wOut * 1000) / a.wLat : 0,
|
|
654
|
+
avgTtfbMs: a.wTtfbCnt > 0 ? a.wTtfbSum / a.wTtfbCnt : null,
|
|
655
|
+
});
|
|
656
|
+
}
|
|
657
|
+
statsCache = next;
|
|
658
|
+
keyStatsCache = nextKeys;
|
|
659
|
+
statsCacheTime = Date.now();
|
|
660
|
+
// Natural tail of a recompute: fold what we just measured back into
|
|
661
|
+
// models.speed_rank (#619). Never allowed to break routing — the caches above
|
|
662
|
+
// are already published, and a failed write just means the column keeps its
|
|
663
|
+
// previous value until the next pass.
|
|
664
|
+
if (Date.now() - speedRankWriteTime >= SPEED_RANK_WRITE_INTERVAL_MS) {
|
|
665
|
+
speedRankWriteTime = Date.now();
|
|
666
|
+
try {
|
|
667
|
+
writeObservedSpeedRanks(db);
|
|
668
|
+
}
|
|
669
|
+
catch (e) {
|
|
670
|
+
console.error('Failed to write observed speed ranks:', e);
|
|
671
|
+
}
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
// ── Observed speed_rank writeback (#619) ────────────────────────────────────
|
|
675
|
+
// models.speed_rank is the catalog's hand-assigned speed ordering and drives
|
|
676
|
+
// the dashboard's sort-by-speed preset. It was only ever WRITTEN by the seed
|
|
677
|
+
// migrations, catalog sync, and an explicit user override — never by anything
|
|
678
|
+
// that had actually watched the model run, so a relay model that hangs on half
|
|
679
|
+
// its calls kept whatever rank the catalog guessed for it forever.
|
|
680
|
+
//
|
|
681
|
+
// This folds the live speed axis back into the column, under three rules:
|
|
682
|
+
// - a model needs SPEED_RANK_MIN_SAMPLES decay-weighted speed-bearing
|
|
683
|
+
// requests (successes + timeouts) before we claim to know anything; below
|
|
684
|
+
// that it keeps its catalog value;
|
|
685
|
+
// - a user-set speed_rank override always wins — we skip those models
|
|
686
|
+
// entirely rather than fight applyModelOverrides for the column;
|
|
687
|
+
// - the UPDATE is guarded on the value actually changing, so a steady system
|
|
688
|
+
// writes nothing at all.
|
|
689
|
+
//
|
|
690
|
+
// A catalog sync re-stamps speed_rank from the catalog; the next pass simply
|
|
691
|
+
// re-derives the observed value, which is why this is a periodic write rather
|
|
692
|
+
// than a one-shot migration.
|
|
693
|
+
export const SPEED_RANK_MIN_SAMPLES = 20;
|
|
694
|
+
const SPEED_RANK_WRITE_INTERVAL_MS = 10 * 60 * 1000;
|
|
695
|
+
let speedRankWriteTime = 0;
|
|
696
|
+
/** Test hook: forget when the last writeback ran so the next refresh does one. */
|
|
697
|
+
export function resetSpeedRankWriteback() {
|
|
698
|
+
speedRankWriteTime = 0;
|
|
699
|
+
}
|
|
700
|
+
/**
|
|
701
|
+
* Write an observed speed rank for every model with enough recent samples and
|
|
702
|
+
* no user-set speed_rank override. Returns how many rows actually changed.
|
|
703
|
+
* Reads the stats cache as-is — callers refresh it first (refreshStatsCache
|
|
704
|
+
* calls this from its own tail).
|
|
705
|
+
*/
|
|
706
|
+
export function writeObservedSpeedRanks(db) {
|
|
707
|
+
if (!statsCache || statsCache.size === 0)
|
|
708
|
+
return 0;
|
|
709
|
+
const pinned = modelsWithOverriddenField(db, 'speedRank');
|
|
710
|
+
const rows = db.prepare('SELECT id, platform, model_id, speed_rank, endpoint_scope FROM models')
|
|
711
|
+
.all();
|
|
712
|
+
const update = db.prepare('UPDATE models SET speed_rank = ? WHERE id = ?');
|
|
713
|
+
let written = 0;
|
|
714
|
+
const tx = db.transaction(() => {
|
|
715
|
+
for (const row of rows) {
|
|
716
|
+
// Overrides are keyed (platform, model_id) — they only exist for
|
|
717
|
+
// catalog-managed rows, which are never endpoint-scoped — while the
|
|
718
|
+
// measured stats are per endpoint, so each relay's copy gets its own
|
|
719
|
+
// observed rank instead of one shared number (#651).
|
|
720
|
+
if (pinned.has(`${row.platform}:${row.model_id}`))
|
|
721
|
+
continue;
|
|
722
|
+
const stats = statsCache.get(modelStatsKey(row.platform, row.model_id, row.endpoint_scope));
|
|
723
|
+
if (!stats || stats.speedSamples < SPEED_RANK_MIN_SAMPLES)
|
|
724
|
+
continue;
|
|
725
|
+
const rank = observedSpeedRank(speedScore(stats.tokPerSec, stats.avgTtfbMs));
|
|
726
|
+
if (rank === row.speed_rank)
|
|
727
|
+
continue;
|
|
728
|
+
update.run(rank, row.id);
|
|
729
|
+
written++;
|
|
730
|
+
}
|
|
731
|
+
});
|
|
732
|
+
tx();
|
|
733
|
+
return written;
|
|
734
|
+
}
|
|
735
|
+
// Enabled + healthy/unknown key count per platform, for pooled-budget scaling.
|
|
736
|
+
// This is the SAME filter both /api/fallback endpoints use (issue #456): the
|
|
737
|
+
// monthly budget is a PER-KEY free-tier allowance, so N usable keys pool N× the
|
|
738
|
+
// capacity. `monthlyUsedTokens` is already summed across all keys, so budget
|
|
739
|
+
// must scale to match or the headroom guardrail damps a multi-key model to the
|
|
740
|
+
// floor after just one account's worth of tokens.
|
|
741
|
+
function usableKeyCountsByPlatform(db) {
|
|
742
|
+
const rows = db.prepare("SELECT platform, COUNT(*) AS count FROM api_keys WHERE enabled = 1 AND status IN ('healthy', 'unknown') GROUP BY platform").all();
|
|
743
|
+
return new Map(rows.map(r => [r.platform, r.count]));
|
|
744
|
+
}
|
|
745
|
+
function scoreChainEntry(entry, weights, intelMin, intelMax, sampled, keyCounts, headroomCfg) {
|
|
746
|
+
const stats = statsCache?.get(modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope));
|
|
747
|
+
const successes = stats?.successes ?? 0;
|
|
748
|
+
const failures = stats?.failures ?? 0;
|
|
749
|
+
const community = activeCommunityPrior(entry.platform, entry.model_id, entry.endpoint_scope);
|
|
750
|
+
let reliability;
|
|
751
|
+
if (sampled) {
|
|
752
|
+
const { alpha, beta } = reliabilityPosterior(successes, failures, community);
|
|
753
|
+
reliability = sampleBeta(alpha, beta);
|
|
754
|
+
}
|
|
755
|
+
else {
|
|
756
|
+
reliability = expectedReliability(successes, failures, community);
|
|
757
|
+
}
|
|
758
|
+
const speed = speedScore(stats?.tokPerSec ?? 0, stats?.avgTtfbMs ?? null);
|
|
759
|
+
const intelligence = intelligenceScore(intelligenceComposite(entry.size_label, entry.intelligence_rank), intelMin, intelMax);
|
|
760
|
+
// Scale the per-key monthly budget by the usable key count for this platform,
|
|
761
|
+
// matching the pooled `monthlyUsedTokens` aggregate (#456). Math.max(1, …) so a
|
|
762
|
+
// model whose platform currently has no usable key isn't handed a 0 budget.
|
|
763
|
+
const budget = parseBudget(entry.monthly_token_budget) * Math.max(1, keyCounts.get(entry.platform) ?? 1);
|
|
764
|
+
// Tunable headroom thresholds (#899): persisted overrides for when demotion
|
|
765
|
+
// starts and its floor; absent settings keep the scoring.ts defaults. Read
|
|
766
|
+
// ONCE per chain by the caller, not per entry — getSetting is an uncached
|
|
767
|
+
// SELECT, so reading it here cost two extra SQLite round-trips per model per
|
|
768
|
+
// request, the same reason `weights` and `keyCounts` are hoisted.
|
|
769
|
+
const monthlyHeadroom = headroomFactor(stats?.monthlyUsedTokens ?? 0, budget, headroomCfg);
|
|
770
|
+
// The same guardrail, driven by live rpm/rpd/tpm/tpd utilization instead of
|
|
771
|
+
// the monthly budget (#899). Most free tiers publish a daily request or token
|
|
772
|
+
// cap and no monthly figure at all, so without this a model sits at score #1
|
|
773
|
+
// until the request that finally 429s it. Reads a snapshot memoised inside
|
|
774
|
+
// ratelimit.ts, so this costs no query per model per request.
|
|
775
|
+
const windowHeadroom = rateWindowHeadroomFactor(modelWindowUsedFraction({ platform: entry.platform, modelId: entry.model_id, keyId: entry.key_id }, { rpm: entry.rpm_limit, rpd: entry.rpd_limit, tpm: entry.tpm_limit, tpd: entry.tpd_limit }), headroomCfg);
|
|
776
|
+
// The WORSE of the two, not their product: both express the same "this model
|
|
777
|
+
// is close to burning out" opinion on different meters, and multiplying them
|
|
778
|
+
// would push a model that is low on both to floor², below the floor the
|
|
779
|
+
// operator configured. Taking the binding constraint keeps the floor meaning
|
|
780
|
+
// what it says — the same rule getKeyQuotaHeadroom applies across metrics.
|
|
781
|
+
const headroom = Math.min(monthlyHeadroom, windowHeadroom);
|
|
782
|
+
const rl = rateLimitFactor(getPenalty(entry.model_db_id));
|
|
783
|
+
// Per-model env overrides (#738) scale the final score so a slow or
|
|
784
|
+
// poor-quality model is demoted without being disabled outright — a manual
|
|
785
|
+
// 'priority' chain can still select it.
|
|
786
|
+
const score = applyModelWeightOverride(combineScore({ reliability, speed, intelligence, headroom, rateLimit: rl }, weights), entry.model_id);
|
|
787
|
+
return { axes: { reliability, speed, intelligence }, headroom, rateLimit: rl, score };
|
|
788
|
+
}
|
|
789
|
+
/**
|
|
790
|
+
* Order the enabled fallback chain for routing.
|
|
791
|
+
* - 'priority' strategy → legacy manual order + 429 penalty (unchanged).
|
|
792
|
+
* - bandit strategy → convex score, manual priority as the deterministic
|
|
793
|
+
* tiebreaker for (near-)equal scores.
|
|
794
|
+
*
|
|
795
|
+
* `sampled` controls the bandit branch: Thompson sampling (the default) for
|
|
796
|
+
* live routing, where per-call randomness is the exploration the bandit needs;
|
|
797
|
+
* the deterministic expected score (`sampled = false`) for callers that want a
|
|
798
|
+
* STABLE ranking under the chosen strategy — the fusion panel, which should be a
|
|
799
|
+
* faithful reflection of the user's picked strategy, not a re-sampled draw each
|
|
800
|
+
* request. Priority mode is deterministic either way.
|
|
801
|
+
*/
|
|
802
|
+
function orderChain(chain, strategy, sampled = true, task) {
|
|
803
|
+
// Tier first, always: it is the one ordering input that score must not be able
|
|
804
|
+
// to override (see ChainRow.match_tier). Zero for every chain built anywhere
|
|
805
|
+
// else, so this is a no-op outside slug-fallback resolution.
|
|
806
|
+
const tier = (e) => e.match_tier ?? 0;
|
|
807
|
+
let weights = weightsFor(strategy);
|
|
808
|
+
if (!weights) {
|
|
809
|
+
// Legacy priority mode: manual chain order + the 429/failure penalty,
|
|
810
|
+
// ascending.
|
|
811
|
+
//
|
|
812
|
+
// The penalty is denominated in PRIORITY POSITIONS — PENALTY_PER_429 = 3
|
|
813
|
+
// positions per rate limit, PENALTY_PER_FAIL = 1 per upstream failure,
|
|
814
|
+
// capped at MAX_PENALTY = 10 — so adding it to the RAW priority only ever
|
|
815
|
+
// reorders a chain whose neighbours sit within 10 of each other. Nothing
|
|
816
|
+
// guarantees that, and several ordinary paths guarantee the opposite:
|
|
817
|
+
// - PUT /api/fallback validates `priority` as a bare z.number(), so any
|
|
818
|
+
// spacing the caller likes (10 / 20 / 30) is persisted verbatim;
|
|
819
|
+
// - the seed and sort-preset paths number the WHOLE catalog 1..N, while
|
|
820
|
+
// this chain is only the ENABLED subset (`JOIN models m ON
|
|
821
|
+
// m.enabled = 1`) — switching models off punches arbitrarily large
|
|
822
|
+
// holes in the surviving sequence;
|
|
823
|
+
// - resolveModelGroupCandidates hydrates scattered group members with
|
|
824
|
+
// whatever COALESCE(fc.priority, 0) they happen to carry.
|
|
825
|
+
// On any of those the penalty was silently INERT: a model 429-ing every
|
|
826
|
+
// single request stayed pinned at the head of the chain forever, and the
|
|
827
|
+
// routing panel's "effective priority" claimed a demotion that never
|
|
828
|
+
// happened.
|
|
829
|
+
//
|
|
830
|
+
// So rank first, then penalize. Sort by the manual priority, re-number the
|
|
831
|
+
// survivors densely 1..N, and add the penalty to THAT rank. Dense ranking
|
|
832
|
+
// is monotonic in priority, so an unpenalized chain comes out in exactly
|
|
833
|
+
// the order the user arranged (tier still dominates as the outer sort key,
|
|
834
|
+
// and the raw priority remains the tiebreaker); the difference is that one
|
|
835
|
+
// penalty position now means what it says — one position.
|
|
836
|
+
return chain
|
|
837
|
+
.map((e, i) => ({ e, i }))
|
|
838
|
+
.sort((a, b) => a.e.priority - b.e.priority || a.i - b.i)
|
|
839
|
+
.map(({ e, i }, rank) => ({ e, i, eff: rank + 1 + getPenalty(e.model_db_id) }))
|
|
840
|
+
.sort((a, b) => tier(a.e) - tier(b.e) || a.eff - b.eff || a.e.priority - b.e.priority || a.i - b.i)
|
|
841
|
+
.map(x => x.e);
|
|
842
|
+
}
|
|
843
|
+
// Task-type bias (#1127): a client-declared/derived task type moves part of
|
|
844
|
+
// one axis onto the other (code: speed → intelligence; chat: the reverse).
|
|
845
|
+
// Applied AFTER the peak-hours adjustment, on the same weights the rest of
|
|
846
|
+
// the chain scores with; opt-in, so absent a signal the preset stands.
|
|
847
|
+
// `fastest`, `reliable` and `custom` are exempt (see TASK_EXEMPT_STRATEGIES),
|
|
848
|
+
// and the share is operator-tunable via settings (0 disables the bias).
|
|
849
|
+
if (task) {
|
|
850
|
+
const adjusted = taskAdjustedWeights(weights, task, strategy, getTaskWeightShare());
|
|
851
|
+
weights = adjusted.adjusted ? adjusted.weights : weights;
|
|
852
|
+
}
|
|
853
|
+
const composites = chain.map(e => intelligenceComposite(e.size_label, e.intelligence_rank));
|
|
854
|
+
const intelMin = composites.length ? Math.min(...composites) : 0;
|
|
855
|
+
const intelMax = composites.length ? Math.max(...composites) : 0;
|
|
856
|
+
const keyCounts = usableKeyCountsByPlatform(getDb());
|
|
857
|
+
const headroomCfg = getHeadroomThresholds();
|
|
858
|
+
return chain
|
|
859
|
+
.map(e => ({ e, s: scoreChainEntry(e, weights, intelMin, intelMax, sampled, keyCounts, headroomCfg).score }))
|
|
860
|
+
// Higher score first WITHIN a tier; manual priority breaks ties so the chain
|
|
861
|
+
// still matters.
|
|
862
|
+
.sort((a, b) => tier(a.e) - tier(b.e) || b.s - a.s || a.e.priority - b.e.priority)
|
|
863
|
+
.map(x => x.e);
|
|
864
|
+
}
|
|
865
|
+
const GLOBAL_SORT_ALIASES = {
|
|
866
|
+
smart: 'smart', smartest: 'smart', intelligence: 'smart',
|
|
867
|
+
fast: 'fast', fastest: 'fast', speed: 'fast',
|
|
868
|
+
cheap: 'cheap', cheapest: 'cheap', price: 'cheap', budget: 'cheap',
|
|
869
|
+
reliable: 'reliable', reliability: 'reliable',
|
|
870
|
+
balanced: 'balanced',
|
|
871
|
+
};
|
|
872
|
+
/**
|
|
873
|
+
* The chain auto-routing walks.
|
|
874
|
+
*
|
|
875
|
+
* When a profile is active it IS the chain, empty or not (#1021). Falling
|
|
876
|
+
* through to `fallback_config` on an empty one meant a chain the operator had
|
|
877
|
+
* deliberately built by hand — or had not filled in yet — silently routed over
|
|
878
|
+
* the entire catalog instead, while the same chain addressed by name
|
|
879
|
+
* (`auto:<name>`) correctly refused. `fallback_config` is the chain only for an
|
|
880
|
+
* install with no profile at all.
|
|
881
|
+
*/
|
|
882
|
+
function getActiveChain(db) {
|
|
883
|
+
const profileId = getActiveProfileId(db);
|
|
884
|
+
if (profileId != null) {
|
|
885
|
+
return db.prepare(`
|
|
886
|
+
SELECT pm.model_db_id, pm.priority, pm.enabled,
|
|
887
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
888
|
+
m.size_label, m.monthly_token_budget,
|
|
889
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
890
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
891
|
+
FROM profile_models pm
|
|
892
|
+
JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
|
|
893
|
+
WHERE pm.profile_id = ?
|
|
894
|
+
ORDER BY pm.priority ASC
|
|
895
|
+
`).all(profileId);
|
|
896
|
+
}
|
|
897
|
+
return db.prepare(`
|
|
898
|
+
SELECT fc.model_db_id, fc.priority, fc.enabled,
|
|
899
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
900
|
+
m.size_label, m.monthly_token_budget,
|
|
901
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
902
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
903
|
+
FROM fallback_config fc
|
|
904
|
+
JOIN models m ON m.id = fc.model_db_id AND m.enabled = 1
|
|
905
|
+
ORDER BY fc.priority ASC
|
|
906
|
+
`).all();
|
|
907
|
+
}
|
|
908
|
+
function getChainByProfileName(db, name) {
|
|
909
|
+
const profile = db.prepare("SELECT id FROM profiles WHERE LOWER(name) = ?").get(name.toLowerCase());
|
|
910
|
+
if (!profile)
|
|
911
|
+
return null;
|
|
912
|
+
return db.prepare(`
|
|
913
|
+
SELECT pm.model_db_id, pm.priority, pm.enabled,
|
|
914
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
915
|
+
m.size_label, m.monthly_token_budget,
|
|
916
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
917
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
918
|
+
FROM profile_models pm
|
|
919
|
+
JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
|
|
920
|
+
WHERE pm.profile_id = ?
|
|
921
|
+
ORDER BY pm.priority ASC
|
|
922
|
+
`).all(profile.id);
|
|
923
|
+
}
|
|
924
|
+
function getChainByGlobalSort(db, globalAxis) {
|
|
925
|
+
// A global sort ignores the chain's ORDER, not its enable flags: a model the
|
|
926
|
+
// operator switched off — in the catalog or just for auto routing — stays off
|
|
927
|
+
// here too (#634). Models with no chain row yet (fresh catalog rows) default
|
|
928
|
+
// to in, so the sort still spans the whole catalog.
|
|
929
|
+
const profileId = getActiveProfileId(db);
|
|
930
|
+
const chainEnabled = profileId != null
|
|
931
|
+
? 'COALESCE(pm.enabled, fc.enabled, 1) = 1'
|
|
932
|
+
: 'COALESCE(fc.enabled, 1) = 1';
|
|
933
|
+
const allEnabled = db.prepare(`
|
|
934
|
+
SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
|
|
935
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
936
|
+
m.size_label, m.monthly_token_budget,
|
|
937
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
938
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
939
|
+
FROM models m
|
|
940
|
+
LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
|
|
941
|
+
${profileId != null ? 'LEFT JOIN profile_models pm ON pm.profile_id = ? AND pm.model_db_id = m.id' : ''}
|
|
942
|
+
WHERE m.enabled = 1 AND ${chainEnabled}
|
|
943
|
+
`).all(...(profileId != null ? [profileId] : []));
|
|
944
|
+
const strategyMap = {
|
|
945
|
+
'smart': 'smartest',
|
|
946
|
+
'fast': 'fastest',
|
|
947
|
+
'cheap': 'balanced',
|
|
948
|
+
'reliable': 'reliable',
|
|
949
|
+
'balanced': 'balanced'
|
|
950
|
+
};
|
|
951
|
+
const strat = strategyMap[globalAxis] || 'balanced';
|
|
952
|
+
return orderChain(allEnabled, strat);
|
|
953
|
+
}
|
|
954
|
+
/**
|
|
955
|
+
* The active chain, or a client-facing refusal when it has nothing enabled.
|
|
956
|
+
*
|
|
957
|
+
* Mirrors what `auto:<name>` already does for a named chain: say the chain is
|
|
958
|
+
* empty rather than routing the request over models the operator never put in
|
|
959
|
+
* it. Only when a profile is active — a legacy install with none keeps the
|
|
960
|
+
* ordinary "all models exhausted" exhaustion path.
|
|
961
|
+
*/
|
|
962
|
+
function activeChainOrThrow(db) {
|
|
963
|
+
const chain = getActiveChain(db);
|
|
964
|
+
if (chain.some(entry => entry.enabled))
|
|
965
|
+
return chain;
|
|
966
|
+
const profileId = getActiveProfileId(db);
|
|
967
|
+
if (profileId == null)
|
|
968
|
+
return chain;
|
|
969
|
+
const profile = db.prepare('SELECT name FROM profiles WHERE id = ?').get(profileId);
|
|
970
|
+
const err = new Error(`The active fallback chain${profile ? ` '${profile.name}'` : ''} has no enabled models. `
|
|
971
|
+
+ 'Enable models for it on the Models page, switch the active chain, or name another one with "auto:<chain>".');
|
|
972
|
+
err.status = 400;
|
|
973
|
+
throw err;
|
|
974
|
+
}
|
|
975
|
+
export function resolveRoutingChain(modelString) {
|
|
976
|
+
const db = getDb();
|
|
977
|
+
if (!modelString || modelString.toLowerCase() === 'auto') {
|
|
978
|
+
return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
|
|
979
|
+
}
|
|
980
|
+
const lower = modelString.toLowerCase();
|
|
981
|
+
if (!lower.startsWith('auto:')) {
|
|
982
|
+
return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
|
|
983
|
+
}
|
|
984
|
+
const suffix = lower.slice('auto:'.length).trim();
|
|
985
|
+
if (!suffix) {
|
|
986
|
+
return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
|
|
987
|
+
}
|
|
988
|
+
const globalAxis = GLOBAL_SORT_ALIASES[suffix];
|
|
989
|
+
if (globalAxis) {
|
|
990
|
+
const chain = getChainByGlobalSort(db, globalAxis);
|
|
991
|
+
if (chain.length === 0) {
|
|
992
|
+
const err = new Error(`No enabled models available for global sort '${suffix}'`);
|
|
993
|
+
err.status = 400;
|
|
994
|
+
throw err;
|
|
995
|
+
}
|
|
996
|
+
return { chain, strategyKey: `auto:${globalAxis}` };
|
|
997
|
+
}
|
|
998
|
+
const chain = getChainByProfileName(db, suffix);
|
|
999
|
+
if (!chain) {
|
|
1000
|
+
const err = new Error(`Profile '${suffix}' not found. Use 'auto' for the default profile, or call /v1/models for available options.`);
|
|
1001
|
+
err.status = 400;
|
|
1002
|
+
throw err;
|
|
1003
|
+
}
|
|
1004
|
+
const enabledModels = chain.filter(e => e.enabled);
|
|
1005
|
+
if (enabledModels.length === 0) {
|
|
1006
|
+
const err = new Error(`Profile '${suffix}' has no enabled models. Add models to this profile in the dashboard.`);
|
|
1007
|
+
err.status = 400;
|
|
1008
|
+
throw err;
|
|
1009
|
+
}
|
|
1010
|
+
return { chain, strategyKey: `auto:${suffix}` };
|
|
1011
|
+
}
|
|
1012
|
+
// Weights for the in-model key score (#580). Reliability dominates: the point
|
|
1013
|
+
// of per-key stats is catching a credential that FAILS (expired, drained,
|
|
1014
|
+
// region-blocked); speed differences between keys of the same platform+model
|
|
1015
|
+
// are second-order (account-tier throttling) but still worth a nudge.
|
|
1016
|
+
const KEY_SCORE_WEIGHTS = { reliability: 0.75, speed: 0.25 };
|
|
1017
|
+
/**
|
|
1018
|
+
* Order a model's candidate keys by a Thompson-sampled per-key score, mirroring
|
|
1019
|
+
* orderChain's bandit: reliability is a fresh draw from each key's Beta
|
|
1020
|
+
* posterior (so exploration is automatic and proportional to uncertainty — a
|
|
1021
|
+
* key with no data samples from the uniform prior and still gets traffic),
|
|
1022
|
+
* speed is deterministic. Returns null when NO key has recorded data, telling
|
|
1023
|
+
* the caller to keep the legacy round-robin rotation (no signal → no ranking).
|
|
1024
|
+
*/
|
|
1025
|
+
function orderKeysByScore(entry, keys) {
|
|
1026
|
+
if (keys.length < 2 || !keyStatsCache)
|
|
1027
|
+
return null;
|
|
1028
|
+
const prefix = `${modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope)}:`;
|
|
1029
|
+
if (!keys.some(k => keyStatsCache.has(prefix + k.id)))
|
|
1030
|
+
return null;
|
|
1031
|
+
// The prior is per-model, not per-key: look it up once outside the loop.
|
|
1032
|
+
const community = activeCommunityPrior(entry.platform, entry.model_id, entry.endpoint_scope);
|
|
1033
|
+
return keys
|
|
1034
|
+
.map(k => {
|
|
1035
|
+
const stats = keyStatsCache.get(prefix + k.id);
|
|
1036
|
+
const { alpha, beta } = reliabilityPosterior(stats?.successes ?? 0, stats?.failures ?? 0, community);
|
|
1037
|
+
const rel = sampleBeta(alpha, beta);
|
|
1038
|
+
const spd = speedScore(stats?.tokPerSec ?? 0, stats?.avgTtfbMs ?? null);
|
|
1039
|
+
return { k, s: KEY_SCORE_WEIGHTS.reliability * rel + KEY_SCORE_WEIGHTS.speed * spd };
|
|
1040
|
+
})
|
|
1041
|
+
.sort((a, b) => b.s - a.s || a.k.id - b.k.id)
|
|
1042
|
+
.map(x => x.k);
|
|
1043
|
+
}
|
|
1044
|
+
/** Headroom assumed for a key the quota tracker has never seen. Neutral on
|
|
1045
|
+
* purpose: an unobserved budget is no reason to prefer a key (it could be
|
|
1046
|
+
* drained) and no reason to avoid one (it could be untouched), so it sorts
|
|
1047
|
+
* between an exhausted key and a fresh one and otherwise keeps its incoming
|
|
1048
|
+
* round-robin position. */
|
|
1049
|
+
const UNKNOWN_QUOTA_HEADROOM = 0.5;
|
|
1050
|
+
/**
|
|
1051
|
+
* Whether remaining-quota weighting is meaningful for this chain entry: the
|
|
1052
|
+
* operator asked for it AND the platform meters its keys separately.
|
|
1053
|
+
*
|
|
1054
|
+
* An account-scoped pool ('<platform>::account') is ONE budget every key of the
|
|
1055
|
+
* account draws down, so "which key has more left" has no answer — every key
|
|
1056
|
+
* reports the same number, and reordering on it would only churn the rotation
|
|
1057
|
+
* for nothing (#919).
|
|
1058
|
+
*/
|
|
1059
|
+
function quotaWeightingApplies(entry) {
|
|
1060
|
+
if (getKeySelectionStrategy() !== 'least-remaining')
|
|
1061
|
+
return false;
|
|
1062
|
+
return !inferQuotaPoolKey(entry.platform, entry.model_id).endsWith('::account');
|
|
1063
|
+
}
|
|
1064
|
+
/**
|
|
1065
|
+
* Re-order an already-ordered candidate list by observed remaining quota,
|
|
1066
|
+
* roomiest first (#919 — the issue asks for higher-remaining-first, so the key
|
|
1067
|
+
* nearest its cap is tried last, not first).
|
|
1068
|
+
*
|
|
1069
|
+
* Deliberately a SORT over the caller's list rather than a second walk: the
|
|
1070
|
+
* incoming order is the round-robin rotation (or the per-key bandit ranking),
|
|
1071
|
+
* and Array#sort is stable, so keys with equal headroom — including the common
|
|
1072
|
+
* case of no observations at all — keep exactly the order they would have had.
|
|
1073
|
+
* Every gate, the custom-endpoint filter and the skip tally stay in the one
|
|
1074
|
+
* walk that follows.
|
|
1075
|
+
*/
|
|
1076
|
+
function orderKeysByRemainingQuota(entry, ordered) {
|
|
1077
|
+
const headroom = getKeyQuotaHeadroom(entry.platform);
|
|
1078
|
+
if (headroom.size === 0)
|
|
1079
|
+
return ordered;
|
|
1080
|
+
// Hoisted out of the comparator: sort calls it O(n log n) times, and the
|
|
1081
|
+
// lookup below must not re-derive anything per comparison.
|
|
1082
|
+
const room = new Map(ordered.map(k => [k.id, headroom.get(k.id) ?? UNKNOWN_QUOTA_HEADROOM]));
|
|
1083
|
+
return [...ordered].sort((a, b) => room.get(b.id) - room.get(a.id));
|
|
1084
|
+
}
|
|
1085
|
+
/**
|
|
1086
|
+
* Pick a usable key for ONE model and build its RouteResult, or return null if
|
|
1087
|
+
* the model has no key that can serve the request right now (all cooled down,
|
|
1088
|
+
* over quota, undecryptable, or no provider). Factored out of routeRequest so
|
|
1089
|
+
* the fusion panel can HARD-PIN a model: walk that model's keys without ever
|
|
1090
|
+
* falling through to a different model (issue #326 — soft preference collapses
|
|
1091
|
+
* panel diversity under rate limits). Keys are tried in per-key bandit-score
|
|
1092
|
+
* order when any of them has recorded data (#580), else round-robin.
|
|
1093
|
+
* Request-level filters (vision/tools/context window) stay in the caller; this
|
|
1094
|
+
* only does key selection + accounting pre-checks.
|
|
1095
|
+
*/
|
|
1096
|
+
function selectKeyForModel(entry, estimatedTokens, skipKeys, diag) {
|
|
1097
|
+
const db = getDb();
|
|
1098
|
+
const label = `${entry.platform}/${entry.model_id}`;
|
|
1099
|
+
if (!hasProvider(entry.platform)) {
|
|
1100
|
+
diag?.push(`${label}: no provider registered`);
|
|
1101
|
+
return null;
|
|
1102
|
+
}
|
|
1103
|
+
const provider = getProvider(entry.platform);
|
|
1104
|
+
const allKeys = db.prepare("SELECT * FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(entry.platform);
|
|
1105
|
+
if (allKeys.length === 0) {
|
|
1106
|
+
diag?.push(`${label}: no enabled+healthy key for platform`);
|
|
1107
|
+
return null;
|
|
1108
|
+
}
|
|
1109
|
+
// Scoped keys (#657) are dropped before the walk: a key whose model scope
|
|
1110
|
+
// excludes this model is not a candidate at all — it neither takes a
|
|
1111
|
+
// round-robin slot nor burns an attempt on a guaranteed 403. Parsed once per
|
|
1112
|
+
// key row.
|
|
1113
|
+
const keys = allKeys.filter(k => scopeAllows(parseModelScope(k.model_scope_json), entry.model_id));
|
|
1114
|
+
if (keys.length === 0) {
|
|
1115
|
+
diag?.push(`${label}: no usable key — ${allKeys.length} key(s) scoped to other models`);
|
|
1116
|
+
return null;
|
|
1117
|
+
}
|
|
1118
|
+
// Tally the gate that rejected each key, so the exhaustion diagnostic can say
|
|
1119
|
+
// *why* a model with keys still couldn't serve (all on cooldown vs over quota).
|
|
1120
|
+
const skipTally = {};
|
|
1121
|
+
const note = (reason) => { skipTally[reason] = (skipTally[reason] ?? 0) + 1; };
|
|
1122
|
+
const limits = {
|
|
1123
|
+
rpm: entry.rpm_limit,
|
|
1124
|
+
rpd: entry.rpd_limit,
|
|
1125
|
+
tpm: entry.tpm_limit,
|
|
1126
|
+
tpd: entry.tpd_limit,
|
|
1127
|
+
};
|
|
1128
|
+
// Score-ordered walk over this model's keys (#580): when any key has recorded
|
|
1129
|
+
// reliability/speed data, try them best-sampled-score first so a chronically
|
|
1130
|
+
// failing key stops soaking up every Nth request. The stats cache is the same
|
|
1131
|
+
// 60s-TTL aggregate the model-level bandit uses (refresh is a no-op when
|
|
1132
|
+
// fresh, and cheap when not). With no data at all, keep the legacy rotation.
|
|
1133
|
+
refreshStatsCache(db);
|
|
1134
|
+
// Scoped so two relays offering the same model id don't share one rotation
|
|
1135
|
+
// cursor over the platform's key list (#651).
|
|
1136
|
+
const rrKey = modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope);
|
|
1137
|
+
let idx = roundRobinIndex.get(rrKey) ?? 0;
|
|
1138
|
+
let ranked = orderKeysByScore(entry, keys);
|
|
1139
|
+
// Remaining-quota weighting (#919) layers on top: it re-sorts whatever order
|
|
1140
|
+
// we were going to walk anyway — the bandit ranking when there is per-key
|
|
1141
|
+
// data, otherwise the round-robin rotation starting at the live cursor — so
|
|
1142
|
+
// ties fall back to that order instead of to rowid.
|
|
1143
|
+
if (keys.length > 1 && quotaWeightingApplies(entry)) {
|
|
1144
|
+
const base = ranked ?? Array.from({ length: keys.length }, (_, i) => keys[(idx + i) % keys.length]);
|
|
1145
|
+
ranked = orderKeysByRemainingQuota(entry, base);
|
|
1146
|
+
}
|
|
1147
|
+
// A custom model belongs to exactly one endpoint (#212), but an endpoint can
|
|
1148
|
+
// hold several credentials — so the pool is every key on the same base_url,
|
|
1149
|
+
// rotated like any other platform's keys (#619). Legacy rows (key_id NULL)
|
|
1150
|
+
// keep the old any-key match.
|
|
1151
|
+
const endpointKeyIds = entry.platform === 'custom' && entry.key_id != null
|
|
1152
|
+
? customEndpointKeyIds(db, entry.key_id)
|
|
1153
|
+
: null;
|
|
1154
|
+
for (let attempt = 0; attempt < keys.length; attempt++) {
|
|
1155
|
+
const key = ranked ? ranked[attempt] : keys[idx % keys.length];
|
|
1156
|
+
idx++;
|
|
1157
|
+
if (endpointKeyIds && !endpointKeyIds.has(key.id)) {
|
|
1158
|
+
note('custom-key-mismatch');
|
|
1159
|
+
continue;
|
|
1160
|
+
}
|
|
1161
|
+
const skipId = `${entry.platform}:${entry.model_id}:${key.id}`;
|
|
1162
|
+
if (skipKeys?.has(skipId)) {
|
|
1163
|
+
note('already-failed-this-request');
|
|
1164
|
+
continue;
|
|
1165
|
+
}
|
|
1166
|
+
if (isOnCooldown(entry.platform, entry.model_id, key.id)) {
|
|
1167
|
+
note('cooldown');
|
|
1168
|
+
continue;
|
|
1169
|
+
}
|
|
1170
|
+
if (!canUseProvider(entry.platform, key.id)) {
|
|
1171
|
+
note('provider-daily-cap');
|
|
1172
|
+
continue;
|
|
1173
|
+
}
|
|
1174
|
+
// Account-wide per-minute budget, checked before the per-model gates: a model
|
|
1175
|
+
// with a NULL rpm_limit would otherwise sail past them and spend a budget its
|
|
1176
|
+
// siblings share.
|
|
1177
|
+
if (!canUseProviderMinute(entry.platform, key.id)) {
|
|
1178
|
+
note('provider-minute-cap');
|
|
1179
|
+
continue;
|
|
1180
|
+
}
|
|
1181
|
+
// Skip a key that already has its allowed requests in the air. Without this,
|
|
1182
|
+
// parallel streams all pick the same key and 429 each other on providers that
|
|
1183
|
+
// meter concurrency per credential.
|
|
1184
|
+
if (!canUseKeyConcurrency(entry.platform, key.id)) {
|
|
1185
|
+
note('key-concurrency');
|
|
1186
|
+
continue;
|
|
1187
|
+
}
|
|
1188
|
+
if (!canMakeRequest(entry.platform, entry.model_id, key.id, limits)) {
|
|
1189
|
+
note('rpm/rpd-limit');
|
|
1190
|
+
continue;
|
|
1191
|
+
}
|
|
1192
|
+
if (!canUseTokens(entry.platform, entry.model_id, key.id, estimatedTokens, limits)) {
|
|
1193
|
+
note('tpm/tpd-limit');
|
|
1194
|
+
continue;
|
|
1195
|
+
}
|
|
1196
|
+
if (!canUseProviderTokens(entry.platform, key.id, entry.model_id, estimatedTokens)) {
|
|
1197
|
+
note('provider-daily-token-cap');
|
|
1198
|
+
continue;
|
|
1199
|
+
}
|
|
1200
|
+
// Monthly budget (#1158): a key whose request/token caps are spent for the
|
|
1201
|
+
// current UTC month is not a candidate — same skip semantics as the daily
|
|
1202
|
+
// gates above. The Retry-After (next-month boundary) surfaces through the
|
|
1203
|
+
// fallback exhaustion path rather than blocking here.
|
|
1204
|
+
if (!checkMonthlyBudget(key.id, estimatedTokens).allowed) {
|
|
1205
|
+
note('monthly-budget-cap');
|
|
1206
|
+
continue;
|
|
1207
|
+
}
|
|
1208
|
+
let decryptedKey;
|
|
1209
|
+
try {
|
|
1210
|
+
decryptedKey = decrypt(key.encrypted_key, key.iv, key.auth_tag);
|
|
1211
|
+
}
|
|
1212
|
+
catch {
|
|
1213
|
+
db.prepare("UPDATE api_keys SET status = 'error', last_checked_at = datetime('now') WHERE id = ?")
|
|
1214
|
+
.run(key.id);
|
|
1215
|
+
note('decrypt-error');
|
|
1216
|
+
continue;
|
|
1217
|
+
}
|
|
1218
|
+
const resolvedProvider = entry.platform === 'custom'
|
|
1219
|
+
? resolveProvider('custom', key.base_url)
|
|
1220
|
+
: provider;
|
|
1221
|
+
if (!resolvedProvider) {
|
|
1222
|
+
note('no-resolved-provider');
|
|
1223
|
+
continue;
|
|
1224
|
+
}
|
|
1225
|
+
roundRobinIndex.set(rrKey, idx);
|
|
1226
|
+
// Taken only once the key has cleared every gate and is definitely being
|
|
1227
|
+
// returned, so a rejected candidate never consumes concurrency budget.
|
|
1228
|
+
const proxyUrl = decryptProxyUrl(key);
|
|
1229
|
+
const budget = reserveMonthlyBudget(key.id, estimatedTokens);
|
|
1230
|
+
if (!budget.allowed) {
|
|
1231
|
+
note('monthly-budget-cap');
|
|
1232
|
+
continue;
|
|
1233
|
+
}
|
|
1234
|
+
const leaseId = acquireLease(entry.platform, entry.model_id, key.id, estimatedTokens);
|
|
1235
|
+
return {
|
|
1236
|
+
provider: resolvedProvider,
|
|
1237
|
+
modelId: entry.model_id,
|
|
1238
|
+
modelDbId: entry.model_db_id,
|
|
1239
|
+
apiKey: decryptedKey,
|
|
1240
|
+
keyId: key.id,
|
|
1241
|
+
keyLabel: key.label || null,
|
|
1242
|
+
// Decrypted once here, at the point the row is already in hand (#590).
|
|
1243
|
+
proxyUrl,
|
|
1244
|
+
platform: entry.platform,
|
|
1245
|
+
displayName: entry.display_name,
|
|
1246
|
+
contextWindow: entry.context_window,
|
|
1247
|
+
endpointScope: entry.endpoint_scope ?? '',
|
|
1248
|
+
rpdLimit: limits.rpd,
|
|
1249
|
+
tpdLimit: limits.tpd,
|
|
1250
|
+
release: () => { releaseLease(leaseId); budget.release(); },
|
|
1251
|
+
};
|
|
1252
|
+
}
|
|
1253
|
+
// No usable key for this model. Advance the round-robin index anyway so we
|
|
1254
|
+
// don't get stuck re-trying the same exhausted key first next time.
|
|
1255
|
+
roundRobinIndex.set(rrKey, idx);
|
|
1256
|
+
const summary = Object.entries(skipTally).map(([r, n]) => `${r}:${n}`).join(', ') || 'no usable key';
|
|
1257
|
+
diag?.push(`${label}: ${keys.length} key(s) — ${summary}`);
|
|
1258
|
+
return null;
|
|
1259
|
+
}
|
|
1260
|
+
/**
|
|
1261
|
+
* Whether the model still has ANOTHER key that could serve it right now, given
|
|
1262
|
+
* the key that just failed (excludingKeyId) and any keys already ruled out this
|
|
1263
|
+
* request (skipKeys, in the "platform:modelId:keyId" form). Applies the same
|
|
1264
|
+
* gates selectKeyForModel uses — enabled + healthy status, not on cooldown,
|
|
1265
|
+
* under the provider daily cap, and under rpm/rpd/tpm/tpd — so the answer means
|
|
1266
|
+
* "a real, dispatchable alternative exists".
|
|
1267
|
+
*
|
|
1268
|
+
* Used by the retry loops to decide whether a single key's 429 should demote the
|
|
1269
|
+
* WHOLE model (the model-level 429 penalty). It should not: the per-key cooldown
|
|
1270
|
+
* already isolates the failing key, so demoting the model while a sibling key can
|
|
1271
|
+
* still serve it wrongly sinks a healthy model in the scorer (#454). We only
|
|
1272
|
+
* record the model-level hit when this returns false — i.e. the 429 exhausted the
|
|
1273
|
+
* model, not just one of its keys.
|
|
1274
|
+
*/
|
|
1275
|
+
export function hasOtherUsableKey(modelDbId, excludingKeyId, skipKeys) {
|
|
1276
|
+
const db = getDb();
|
|
1277
|
+
const m = db.prepare(`
|
|
1278
|
+
SELECT platform, model_id, rpm_limit, rpd_limit, tpm_limit, tpd_limit, key_id
|
|
1279
|
+
FROM models WHERE id = ?
|
|
1280
|
+
`).get(modelDbId);
|
|
1281
|
+
if (!m)
|
|
1282
|
+
return false;
|
|
1283
|
+
const limits = { rpm: m.rpm_limit, rpd: m.rpd_limit, tpm: m.tpm_limit, tpd: m.tpd_limit };
|
|
1284
|
+
const keys = db.prepare("SELECT id, model_scope_json FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(m.platform);
|
|
1285
|
+
// Keys of the model's own custom endpoint (#212, #619); a key belonging to a
|
|
1286
|
+
// DIFFERENT endpoint cannot serve it, so it doesn't count as an alternative.
|
|
1287
|
+
const endpointKeyIds = m.platform === 'custom' && m.key_id != null
|
|
1288
|
+
? customEndpointKeyIds(db, m.key_id)
|
|
1289
|
+
: null;
|
|
1290
|
+
for (const k of keys) {
|
|
1291
|
+
if (k.id === excludingKeyId)
|
|
1292
|
+
continue;
|
|
1293
|
+
if (endpointKeyIds && !endpointKeyIds.has(k.id))
|
|
1294
|
+
continue;
|
|
1295
|
+
// A sibling scoped away from this model can never serve it (#657) — counting
|
|
1296
|
+
// it would wrongly suppress the model-level penalty this gate exists for.
|
|
1297
|
+
if (!scopeAllows(parseModelScope(k.model_scope_json), m.model_id))
|
|
1298
|
+
continue;
|
|
1299
|
+
if (skipKeys?.has(`${m.platform}:${m.model_id}:${k.id}`))
|
|
1300
|
+
continue;
|
|
1301
|
+
if (isOnCooldown(m.platform, m.model_id, k.id))
|
|
1302
|
+
continue;
|
|
1303
|
+
if (!canUseProvider(m.platform, k.id))
|
|
1304
|
+
continue;
|
|
1305
|
+
if (!canUseProviderMinute(m.platform, k.id))
|
|
1306
|
+
continue;
|
|
1307
|
+
if (!canMakeRequest(m.platform, m.model_id, k.id, limits))
|
|
1308
|
+
continue;
|
|
1309
|
+
// A per-minute token spike on the failed key doesn't mean a fresh key lacks
|
|
1310
|
+
// headroom; a nominal 1-token probe only rules out a key already at its
|
|
1311
|
+
// TPM/TPD ceiling.
|
|
1312
|
+
if (!canUseTokens(m.platform, m.model_id, k.id, 1, limits))
|
|
1313
|
+
continue;
|
|
1314
|
+
if (!canUseProviderTokens(m.platform, k.id, m.model_id, 1))
|
|
1315
|
+
continue;
|
|
1316
|
+
return true;
|
|
1317
|
+
}
|
|
1318
|
+
return false;
|
|
1319
|
+
}
|
|
1320
|
+
/**
|
|
1321
|
+
* Can ANY key serve this model right now? The same gates hasOtherUsableKey
|
|
1322
|
+
* applies — scope (#657), per-key cooldown, and the provider/model rate and
|
|
1323
|
+
* token windows — with no key excluded. /v1/models uses it to tell a `ready`
|
|
1324
|
+
* model from an `exhausted` one (#1100), so the listing cannot claim a model
|
|
1325
|
+
* the router would immediately skip.
|
|
1326
|
+
*/
|
|
1327
|
+
export function hasUsableKeyForModel(modelDbId) {
|
|
1328
|
+
// Key ids are AUTOINCREMENT and start at 1, so -1 excludes nothing.
|
|
1329
|
+
return hasOtherUsableKey(modelDbId, -1);
|
|
1330
|
+
}
|
|
1331
|
+
/**
|
|
1332
|
+
* Every key that can be ROUTED to this model: enabled + healthy/unknown, not
|
|
1333
|
+
* scoped away from the model (#657), and — for a custom model — belonging to
|
|
1334
|
+
* the model's own endpoint (#212, #619). Deliberately ignores the transient
|
|
1335
|
+
* gates hasOtherUsableKey applies (cooldown, quotas): the caller here is the
|
|
1336
|
+
* model-level bench, which needs the full key set to take a sick model out of
|
|
1337
|
+
* rotation, not "who could serve the next request".
|
|
1338
|
+
*/
|
|
1339
|
+
export function routableKeyIdsForModel(modelDbId) {
|
|
1340
|
+
const db = getDb();
|
|
1341
|
+
const m = db.prepare('SELECT platform, model_id, key_id FROM models WHERE id = ?')
|
|
1342
|
+
.get(modelDbId);
|
|
1343
|
+
if (!m)
|
|
1344
|
+
return [];
|
|
1345
|
+
const keys = db.prepare("SELECT id, model_scope_json FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(m.platform);
|
|
1346
|
+
const endpointKeyIds = m.platform === 'custom' && m.key_id != null
|
|
1347
|
+
? customEndpointKeyIds(db, m.key_id)
|
|
1348
|
+
: null;
|
|
1349
|
+
return keys
|
|
1350
|
+
.filter(k => !endpointKeyIds || endpointKeyIds.has(k.id))
|
|
1351
|
+
.filter(k => scopeAllows(parseModelScope(k.model_scope_json), m.model_id))
|
|
1352
|
+
.map(k => k.id);
|
|
1353
|
+
}
|
|
1354
|
+
/**
|
|
1355
|
+
* Fetch a single enabled model's chain row by its db id.
|
|
1356
|
+
*/
|
|
1357
|
+
function getModelChainRow(db, modelDbId) {
|
|
1358
|
+
return db.prepare(`
|
|
1359
|
+
SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
|
|
1360
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
1361
|
+
m.size_label, m.monthly_token_budget,
|
|
1362
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
1363
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
1364
|
+
FROM models m
|
|
1365
|
+
WHERE m.id = ? AND m.enabled = 1
|
|
1366
|
+
`).get(modelDbId);
|
|
1367
|
+
}
|
|
1368
|
+
/**
|
|
1369
|
+
* Safety margin applied when ranking a model against an estimated request size.
|
|
1370
|
+
* The estimate is a chars/4 heuristic that under-counts dense payloads (JSON,
|
|
1371
|
+
* code, CJK) by up to ~2x; without any accounting for that gap such requests
|
|
1372
|
+
* were routed to models whose real tokenizer count exceeded the window and the
|
|
1373
|
+
* provider rejected them with a 400 mid-chain (kilo: "maximum context length is
|
|
1374
|
+
* 262144 tokens" on requests estimated <=256000).
|
|
1375
|
+
*
|
|
1376
|
+
* The margin is a SOFT preference, not a hard filter (#956 review): /v1/models
|
|
1377
|
+
* advertises the RAW window, so clients legitimately pack requests right up to
|
|
1378
|
+
* it. Excluding margin-violating models outright would turn an upstream 400
|
|
1379
|
+
* that the retry loop already classifies and handles (`context_too_large`)
|
|
1380
|
+
* into a regression: "all models exhausted" with zero attempts. Callers
|
|
1381
|
+
* therefore try margin-fitting candidates first and only fall back to raw
|
|
1382
|
+
* advertised-window fits (see fitsContextWindowStrict) when nothing else can
|
|
1383
|
+
* serve the request.
|
|
1384
|
+
*/
|
|
1385
|
+
export const CONTEXT_WINDOW_SAFETY_FACTOR = 1.25;
|
|
1386
|
+
// Platforms whose pre-dispatch trim guard already caps the dispatched input
|
|
1387
|
+
// below the live context ceiling (lib/content.ts truncateMessagesForGithub):
|
|
1388
|
+
// the guard — not the routing estimate — is what guarantees the fit there, so
|
|
1389
|
+
// applying the factor too would only make that guard unreachable. The margin
|
|
1390
|
+
// checks below treat these platforms as strict comparisons.
|
|
1391
|
+
const TRIM_GUARDED_PLATFORMS = new Set(['github']);
|
|
1392
|
+
/** True when `estimatedTokens` fits the RAW advertised window (null window =
|
|
1393
|
+
* unknown, never filtered — same convention as the auto-router). This is the
|
|
1394
|
+
* comparison /v1/models publishes and the soft-preference fallback tier. */
|
|
1395
|
+
export function fitsContextWindowStrict(contextWindow, estimatedTokens) {
|
|
1396
|
+
return contextWindow == null || estimatedTokens <= contextWindow;
|
|
1397
|
+
}
|
|
1398
|
+
/** True when `estimatedTokens` plausibly fits `contextWindow` WITH the safety
|
|
1399
|
+
* margin. The chars/4 heuristic portion is scaled by the factor; an explicit
|
|
1400
|
+
* output reserve derived from the client's max_tokens (`routingReserveTokens`)
|
|
1401
|
+
* is already an exact count and is added UNSCALED (#956 review). Trim-guarded
|
|
1402
|
+
* platforms compare strictly — their guard guarantees the fit. */
|
|
1403
|
+
export function fitsContextWindow(platform, contextWindow, estimatedTokens, exactOutputReserve = 0) {
|
|
1404
|
+
if (contextWindow == null)
|
|
1405
|
+
return true;
|
|
1406
|
+
// Raw advertised comparison first — the margin can only shrink eligibility.
|
|
1407
|
+
if (estimatedTokens > contextWindow)
|
|
1408
|
+
return false;
|
|
1409
|
+
if (TRIM_GUARDED_PLATFORMS.has(platform))
|
|
1410
|
+
return true;
|
|
1411
|
+
const reserve = Math.max(0, exactOutputReserve);
|
|
1412
|
+
const heuristic = Math.max(0, estimatedTokens - reserve);
|
|
1413
|
+
return heuristic * CONTEXT_WINDOW_SAFETY_FACTOR + reserve <= contextWindow;
|
|
1414
|
+
}
|
|
1415
|
+
/**
|
|
1416
|
+
* Route to ONE specific model, hard-pinned. Rotates across that model's keys
|
|
1417
|
+
* (cooldowns, quotas, decryption all honored) but NEVER substitutes a different
|
|
1418
|
+
* model — returns null if the pinned model can't serve right now. This is what
|
|
1419
|
+
* makes a fusion panel genuinely diverse: a rate-limited slot is dropped, not
|
|
1420
|
+
* silently collapsed onto whatever else is available. `skipKeys` lets a slot
|
|
1421
|
+
* exclude keys it already failed on this request.
|
|
1422
|
+
*/
|
|
1423
|
+
export function routePinnedModel(modelDbId, estimatedTokens = 1000, skipKeys) {
|
|
1424
|
+
const db = getDb();
|
|
1425
|
+
const entry = getModelChainRow(db, modelDbId);
|
|
1426
|
+
if (!entry)
|
|
1427
|
+
return null;
|
|
1428
|
+
// Strict comparison only (#956 review): a pinned slot has no substitute, so
|
|
1429
|
+
// refusing on a margin violation would drop the slot outright where the
|
|
1430
|
+
// pre-margin behavior was one dispatch attempt (a mid-chain context_too_large
|
|
1431
|
+
// 400 is classified and retried downstream). Nothing is multiplied here —
|
|
1432
|
+
// estimatedTokens already carries the exact capped output reserve (#470).
|
|
1433
|
+
if (!fitsContextWindowStrict(entry.context_window, estimatedTokens))
|
|
1434
|
+
return null;
|
|
1435
|
+
if (entry.tpm_limit != null && estimatedTokens > entry.tpm_limit)
|
|
1436
|
+
return null;
|
|
1437
|
+
return selectKeyForModel(entry, estimatedTokens, skipKeys);
|
|
1438
|
+
}
|
|
1439
|
+
/**
|
|
1440
|
+
* Resolve a logical model group's member db ids to an ordered ChainRow[] for
|
|
1441
|
+
* strict group-pin routing (the "unify" feature). Each catalog-enabled member
|
|
1442
|
+
* is hydrated as a ChainRow carrying its active-profile/manual priority, then
|
|
1443
|
+
* ordered by the active strategy via orderChain. Auto-chain enabled/disabled is
|
|
1444
|
+
* intentionally ignored here because an explicit model request should still be
|
|
1445
|
+
* able to use a direct model that the user removed from auto routing.
|
|
1446
|
+
*
|
|
1447
|
+
* Pass the result to routeRequest() as `prefetchedChain` and DO NOT pass a
|
|
1448
|
+
* `preferredModelDbId` that isn't already one of these rows — otherwise the
|
|
1449
|
+
* preferred-model injection in routeRequest would unshift an off-group model and
|
|
1450
|
+
* the pin would no longer be strict (it could answer with a different model).
|
|
1451
|
+
*/
|
|
1452
|
+
export function resolveModelGroupCandidates(memberDbIds,
|
|
1453
|
+
/**
|
|
1454
|
+
* Members that were reached only through a group's auto-derived slug, not the
|
|
1455
|
+
* id the client wrote (#651). They stay in the chain — resolution must never
|
|
1456
|
+
* shrink — but as a strictly lower tier, so they can serve only once every
|
|
1457
|
+
* literal match is exhausted. Omit it and every row is an equal candidate,
|
|
1458
|
+
* which is what every other caller wants.
|
|
1459
|
+
*/
|
|
1460
|
+
demotedDbIds) {
|
|
1461
|
+
const db = getDb();
|
|
1462
|
+
const strategy = getRoutingStrategy();
|
|
1463
|
+
if (strategy !== 'priority')
|
|
1464
|
+
refreshStatsCache(db);
|
|
1465
|
+
const activeProfileId = getActiveProfileId(db);
|
|
1466
|
+
const selectMember = activeProfileId == null
|
|
1467
|
+
? db.prepare(`
|
|
1468
|
+
SELECT m.id as model_db_id, COALESCE(fc.priority, 0) as priority,
|
|
1469
|
+
1 as enabled,
|
|
1470
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
1471
|
+
m.size_label, m.monthly_token_budget,
|
|
1472
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
1473
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
1474
|
+
FROM models m
|
|
1475
|
+
LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
|
|
1476
|
+
WHERE m.id = ? AND m.enabled = 1
|
|
1477
|
+
`)
|
|
1478
|
+
: db.prepare(`
|
|
1479
|
+
SELECT m.id as model_db_id, COALESCE(pm.priority, fc.priority, 0) as priority,
|
|
1480
|
+
1 as enabled,
|
|
1481
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
1482
|
+
m.size_label, m.monthly_token_budget,
|
|
1483
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
1484
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
1485
|
+
FROM models m
|
|
1486
|
+
LEFT JOIN profile_models pm ON pm.profile_id = ? AND pm.model_db_id = m.id
|
|
1487
|
+
LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
|
|
1488
|
+
WHERE m.id = ? AND m.enabled = 1
|
|
1489
|
+
`);
|
|
1490
|
+
const rows = [];
|
|
1491
|
+
for (const id of memberDbIds) {
|
|
1492
|
+
const row = (activeProfileId == null ? selectMember.get(id) : selectMember.get(activeProfileId, id));
|
|
1493
|
+
if (!row)
|
|
1494
|
+
continue;
|
|
1495
|
+
row.match_tier = demotedDbIds?.has(id) ? 1 : 0;
|
|
1496
|
+
rows.push(row);
|
|
1497
|
+
}
|
|
1498
|
+
return orderChain(rows, strategy);
|
|
1499
|
+
}
|
|
1500
|
+
/**
|
|
1501
|
+
* The active fallback chain ordered by the current routing strategy, surfaced
|
|
1502
|
+
* for fusion panel selection. Same ordering the normal auto-router would walk,
|
|
1503
|
+
* so the panel's auto-pick draws from the highest-scored models first and the
|
|
1504
|
+
* fusion layer just needs to apply provider-diversity on top.
|
|
1505
|
+
*/
|
|
1506
|
+
export function getOrderedFusionChain(estimatedTokens, exactOutputReserve = 0) {
|
|
1507
|
+
const db = getDb();
|
|
1508
|
+
const strategy = getRoutingStrategy();
|
|
1509
|
+
if (strategy !== 'priority')
|
|
1510
|
+
refreshStatsCache(db);
|
|
1511
|
+
const chain = getActiveChain(db).filter(e => e.enabled);
|
|
1512
|
+
// Only consider models that can ACTUALLY be served RIGHT NOW — applying the
|
|
1513
|
+
// same gate selectKeyForModel uses when the router walks the chain: the model
|
|
1514
|
+
// must have a key that is enabled + healthy, NOT on cooldown (e.g. a
|
|
1515
|
+
// HuggingFace key benched for a day after a 402 "Payment Required"), within
|
|
1516
|
+
// the provider's daily request cap, and under its per-minute/day request
|
|
1517
|
+
// limits. Without this, a high-strategy-ranked model whose only key is
|
|
1518
|
+
// currently cooled down (huggingface/Kimi-K2.6) would claim a panel slot it
|
|
1519
|
+
// can't fill — surfacing as "no available key" and pushing out a usable model,
|
|
1520
|
+
// which also makes the panel look like it's ignoring the routing strategy.
|
|
1521
|
+
//
|
|
1522
|
+
// The SIZE gates matter as much as the key gates: a model whose context window
|
|
1523
|
+
// cannot hold the prompt can NEVER fill its slot, yet diversifyChain keeps
|
|
1524
|
+
// handing it one on every request when it is its platform's only representative.
|
|
1525
|
+
// That leaves one panel slot dead on arrival and reports the failure as the
|
|
1526
|
+
// misleading "no available key for model". Passing a placeholder token count
|
|
1527
|
+
// here made both size gates no-ops.
|
|
1528
|
+
const usableKeys = db.prepare("SELECT id, platform, model_scope_json FROM api_keys WHERE enabled = 1 AND status IN ('healthy', 'unknown')").all();
|
|
1529
|
+
// Scope parsed once per key row (#657); the servable filter below re-checks
|
|
1530
|
+
// membership per model.
|
|
1531
|
+
const keysByPlatform = new Map();
|
|
1532
|
+
for (const k of usableKeys) {
|
|
1533
|
+
const entry = { id: k.id, scope: parseModelScope(k.model_scope_json) };
|
|
1534
|
+
const arr = keysByPlatform.get(k.platform);
|
|
1535
|
+
if (arr)
|
|
1536
|
+
arr.push(entry);
|
|
1537
|
+
else
|
|
1538
|
+
keysByPlatform.set(k.platform, [entry]);
|
|
1539
|
+
}
|
|
1540
|
+
// Soft preference (#956 review): prefer models whose window holds the estimate
|
|
1541
|
+
// WITH the safety margin; /v1/models still advertises the raw window, so if
|
|
1542
|
+
// NOTHING survives that pass, re-run allowing raw advertised-window fits
|
|
1543
|
+
// rather than handing back an empty chain — a request packed to the
|
|
1544
|
+
// advertised window keeps its one attempt (the mid-chain context_too_large
|
|
1545
|
+
// 400 is classified and retried downstream).
|
|
1546
|
+
const passesContextGate = (e, allowMarginViolators) => {
|
|
1547
|
+
// A null context_window means "unknown", not "zero": same convention the
|
|
1548
|
+
// auto-router uses, so an unspecified window is never itself a reason to skip.
|
|
1549
|
+
if (fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve))
|
|
1550
|
+
return true;
|
|
1551
|
+
return allowMarginViolators && fitsContextWindowStrict(e.context_window, estimatedTokens);
|
|
1552
|
+
};
|
|
1553
|
+
const servableFilter = (allowMarginViolators) => chain.filter(e => {
|
|
1554
|
+
if (!passesContextGate(e, allowMarginViolators))
|
|
1555
|
+
return false;
|
|
1556
|
+
const keyIds = keysByPlatform.get(e.platform);
|
|
1557
|
+
if (!keyIds)
|
|
1558
|
+
return false;
|
|
1559
|
+
// Same endpoint-pool rule the router applies (#619).
|
|
1560
|
+
const endpointKeyIds = e.platform === 'custom' && e.key_id != null
|
|
1561
|
+
? customEndpointKeyIds(db, e.key_id)
|
|
1562
|
+
: null;
|
|
1563
|
+
const limits = { rpm: e.rpm_limit, rpd: e.rpd_limit, tpm: e.tpm_limit, tpd: e.tpd_limit };
|
|
1564
|
+
return keyIds.some(({ id: kid, scope }) => scopeAllows(scope, e.model_id) &&
|
|
1565
|
+
(endpointKeyIds == null || endpointKeyIds.has(kid)) &&
|
|
1566
|
+
!isOnCooldown(e.platform, e.model_id, kid) &&
|
|
1567
|
+
canUseProvider(e.platform, kid) &&
|
|
1568
|
+
canUseProviderMinute(e.platform, kid) &&
|
|
1569
|
+
canMakeRequest(e.platform, e.model_id, kid, limits) &&
|
|
1570
|
+
canUseProviderTokens(e.platform, kid, e.model_id, estimatedTokens));
|
|
1571
|
+
});
|
|
1572
|
+
let servable = servableFilter(false);
|
|
1573
|
+
if (servable.length === 0)
|
|
1574
|
+
servable = servableFilter(true);
|
|
1575
|
+
// Deterministic (expected-score) ordering so the panel faithfully follows the
|
|
1576
|
+
// user's picked routing strategy instead of re-sampling a fresh draw each call.
|
|
1577
|
+
const ordered = orderChain(servable, strategy, false);
|
|
1578
|
+
return ordered.map(e => ({
|
|
1579
|
+
modelDbId: e.model_db_id,
|
|
1580
|
+
platform: e.platform,
|
|
1581
|
+
modelId: e.model_id,
|
|
1582
|
+
displayName: e.display_name,
|
|
1583
|
+
sizeLabel: e.size_label,
|
|
1584
|
+
supportsVision: e.supports_vision,
|
|
1585
|
+
supportsTools: e.supports_tools,
|
|
1586
|
+
}));
|
|
1587
|
+
}
|
|
1588
|
+
/**
|
|
1589
|
+
* Resolve an explicit model id (as a client would type it) to a fusion
|
|
1590
|
+
* candidate, or null when it isn't a known enabled model. Prefers an enabled
|
|
1591
|
+
* row; dedupes a model id that exists on multiple platforms by intelligence
|
|
1592
|
+
* rank, matching how /v1/models picks a representative row.
|
|
1593
|
+
*/
|
|
1594
|
+
export function resolveFusionCandidate(modelId) {
|
|
1595
|
+
const db = getDb();
|
|
1596
|
+
const rows = db.prepare(`
|
|
1597
|
+
SELECT m.id as model_db_id, m.platform, m.model_id, m.display_name,
|
|
1598
|
+
m.size_label, m.supports_vision, m.supports_tools
|
|
1599
|
+
FROM models m
|
|
1600
|
+
WHERE m.model_id = ? AND m.enabled = 1
|
|
1601
|
+
ORDER BY m.intelligence_rank ASC, m.id ASC
|
|
1602
|
+
`).all(modelId);
|
|
1603
|
+
if (rows.length > 0) {
|
|
1604
|
+
// A logical model can have several enabled provider rows. Prefer one with
|
|
1605
|
+
// an enabled, healthy/unknown key before falling back to the deterministic
|
|
1606
|
+
// ranking. Without this check a duplicate alias can pin fusion to a
|
|
1607
|
+
// keyless provider (for example a free relay) while a configured provider
|
|
1608
|
+
// for the same model is available.
|
|
1609
|
+
const row = rows.find(candidate => routableKeyIdsForModel(candidate.model_db_id).length > 0) ?? rows[0];
|
|
1610
|
+
return {
|
|
1611
|
+
modelDbId: row.model_db_id,
|
|
1612
|
+
platform: row.platform,
|
|
1613
|
+
modelId: row.model_id,
|
|
1614
|
+
displayName: row.display_name,
|
|
1615
|
+
sizeLabel: row.size_label,
|
|
1616
|
+
supportsVision: row.supports_vision,
|
|
1617
|
+
supportsTools: row.supports_tools,
|
|
1618
|
+
};
|
|
1619
|
+
}
|
|
1620
|
+
// Unify ON: a fusion picker value may be a canonical GROUP id rather than a
|
|
1621
|
+
// raw model_id. Resolve it to the group's best-ordered enabled member so
|
|
1622
|
+
// saved fusion configs that use canonical ids keep working. Exact model_id
|
|
1623
|
+
// match above always wins first, so OFF mode and legacy configs are untouched.
|
|
1624
|
+
if (isUnifyEnabled()) {
|
|
1625
|
+
const resolved = resolveRequestedIdForDispatch(modelId, getModelGroups());
|
|
1626
|
+
if (resolved && resolved.memberDbIds.length > 0) {
|
|
1627
|
+
const candidates = resolveModelGroupCandidates(resolved.memberDbIds, resolved.demotedDbIds);
|
|
1628
|
+
// Bare unified aliases may resolve to a provider row that is enabled in
|
|
1629
|
+
// the catalog but has no usable key. Select the first routable member so
|
|
1630
|
+
// an explicit fusion panel does not waste a slot on that dead end.
|
|
1631
|
+
const top = candidates.find(candidate => routableKeyIdsForModel(candidate.model_db_id).length > 0) ?? candidates[0];
|
|
1632
|
+
if (top) {
|
|
1633
|
+
return {
|
|
1634
|
+
modelDbId: top.model_db_id,
|
|
1635
|
+
platform: top.platform,
|
|
1636
|
+
modelId: top.model_id,
|
|
1637
|
+
displayName: top.display_name,
|
|
1638
|
+
sizeLabel: top.size_label,
|
|
1639
|
+
supportsVision: top.supports_vision,
|
|
1640
|
+
supportsTools: top.supports_tools,
|
|
1641
|
+
};
|
|
1642
|
+
}
|
|
1643
|
+
}
|
|
1644
|
+
}
|
|
1645
|
+
return null;
|
|
1646
|
+
}
|
|
1647
|
+
export function routeRequest(estimatedTokens = 1000, skipKeys, preferredModelDbId, requireVision = false, requireTools = false, skipModels, prefetchedChain, requireStructured = false, skipPlatforms, exactOutputReserve = 0, task) {
|
|
1648
|
+
const db = getDb();
|
|
1649
|
+
const strategy = getRoutingStrategy();
|
|
1650
|
+
if (strategy !== 'priority')
|
|
1651
|
+
refreshStatsCache(db);
|
|
1652
|
+
const chain = (prefetchedChain ?? getActiveChain(db)).filter(e => e.enabled);
|
|
1653
|
+
const sortedChain = orderChain(chain, strategy, true, task);
|
|
1654
|
+
const keyCounts = usableKeyCountsByPlatform(db);
|
|
1655
|
+
// Exploration toggle (#685/#707 follow-up): when enabled, give a model with
|
|
1656
|
+
// no reliability/speed samples a guaranteed chance to be tried, so it stops
|
|
1657
|
+
// losing every bandit draw to prior-heavy rivals. With EXPLORE_CHANCE
|
|
1658
|
+
// probability, pick one unmeasured model uniformly and try it first; if it
|
|
1659
|
+
// fails, the loop falls through to the scored order as usual. Only for
|
|
1660
|
+
// bandit strategies — Manual is the operator's explicit order.
|
|
1661
|
+
// Candidates the main loop would immediately reject for THIS request are
|
|
1662
|
+
// excluded up front (ruled-out models plus the request-level capability
|
|
1663
|
+
// gates: vision/tools/structured output/context window): promoting a model
|
|
1664
|
+
// that can't serve the request ahead of capable ones would just get it
|
|
1665
|
+
// skipped a moment later (e.g. an image request must never randomly probe a
|
|
1666
|
+
// text-only model).
|
|
1667
|
+
// While the gateway is in degraded mode (#904) exploration is skipped
|
|
1668
|
+
// entirely: probing unmeasured models during a fleet-wide outage just burns
|
|
1669
|
+
// retry budget on the same dead providers, and the scored order of known
|
|
1670
|
+
// survivors is the only thing worth trying.
|
|
1671
|
+
if (strategy !== 'priority' && getExploreEnabled() && !isDegraded() && Math.random() < EXPLORE_CHANCE) {
|
|
1672
|
+
// A model the operator zeroed out via MODEL_ROUTING_OVERRIDES never wins a
|
|
1673
|
+
// bandit draw, so it would stay under EXPLORE_MIN_SAMPLES forever and become
|
|
1674
|
+
// a perpetual probe target — the explicit ban outranks exploration.
|
|
1675
|
+
const overrides = getModelWeightOverrides();
|
|
1676
|
+
const unmeasured = sortedChain.filter(e => {
|
|
1677
|
+
if (overrides.get(e.model_id) === 0)
|
|
1678
|
+
return false;
|
|
1679
|
+
if ((keyCounts.get(e.platform) ?? 0) <= 0)
|
|
1680
|
+
return false;
|
|
1681
|
+
const stats = statsCache?.get(modelStatsKey(e.platform, e.model_id, e.endpoint_scope));
|
|
1682
|
+
if ((stats?.successes ?? 0) + (stats?.failures ?? 0) >= EXPLORE_MIN_SAMPLES)
|
|
1683
|
+
return false;
|
|
1684
|
+
// Mirror the main loop's gates below so exploration only samples
|
|
1685
|
+
// candidates that can actually serve this request.
|
|
1686
|
+
if (skipModels?.has(e.model_db_id))
|
|
1687
|
+
return false;
|
|
1688
|
+
if (skipPlatforms?.has(e.platform))
|
|
1689
|
+
return false;
|
|
1690
|
+
if (requireVision && !e.supports_vision)
|
|
1691
|
+
return false;
|
|
1692
|
+
if (requireTools && !e.supports_tools)
|
|
1693
|
+
return false;
|
|
1694
|
+
// Never spend the exploration slot on a model that keeps rejecting tool
|
|
1695
|
+
// requests (#1230); it stays reachable at the back of the main walk.
|
|
1696
|
+
if (requireTools && isToolBenched(e.platform, e.model_id, e.endpoint_scope))
|
|
1697
|
+
return false;
|
|
1698
|
+
if (requireStructured && platformDropsResponseFormat(e.platform))
|
|
1699
|
+
return false;
|
|
1700
|
+
if (!fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve))
|
|
1701
|
+
return false;
|
|
1702
|
+
if (e.tpm_limit != null && estimatedTokens > e.tpm_limit)
|
|
1703
|
+
return false;
|
|
1704
|
+
return true;
|
|
1705
|
+
});
|
|
1706
|
+
if (unmeasured.length > 0) {
|
|
1707
|
+
const probe = unmeasured[Math.floor(Math.random() * unmeasured.length)];
|
|
1708
|
+
const idx = sortedChain.findIndex(e => e.model_db_id === probe.model_db_id);
|
|
1709
|
+
if (idx > 0) {
|
|
1710
|
+
const [probeRow] = sortedChain.splice(idx, 1);
|
|
1711
|
+
sortedChain.unshift(probeRow);
|
|
1712
|
+
}
|
|
1713
|
+
}
|
|
1714
|
+
}
|
|
1715
|
+
// Sticky session / Explicit pinning: move preferred model to front of chain
|
|
1716
|
+
if (preferredModelDbId) {
|
|
1717
|
+
const idx = sortedChain.findIndex(e => e.model_db_id === preferredModelDbId);
|
|
1718
|
+
if (idx >= 0) {
|
|
1719
|
+
if (idx > 0) {
|
|
1720
|
+
const [preferred] = sortedChain.splice(idx, 1);
|
|
1721
|
+
sortedChain.unshift(preferred);
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1724
|
+
else {
|
|
1725
|
+
// The requested model is not in the current routing chain (e.g. it's a
|
|
1726
|
+
// custom model or not added to the active profile). We must fulfill the
|
|
1727
|
+
// explicit request by injecting it at the front.
|
|
1728
|
+
const pinnedRow = db.prepare(`
|
|
1729
|
+
SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
|
|
1730
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
1731
|
+
m.size_label, m.monthly_token_budget,
|
|
1732
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
1733
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
1734
|
+
FROM models m
|
|
1735
|
+
WHERE m.id = ? AND m.enabled = 1
|
|
1736
|
+
`).get(preferredModelDbId);
|
|
1737
|
+
if (pinnedRow) {
|
|
1738
|
+
sortedChain.unshift(pinnedRow);
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
1741
|
+
}
|
|
1742
|
+
// Drop models whose platform has NO enabled+healthy key before the walk. Such
|
|
1743
|
+
// a row can never produce a route (selectKeyForModel's first query returns
|
|
1744
|
+
// empty for it), so walking it is pure overhead on every request, and its diag
|
|
1745
|
+
// line pads the exhaustion summary with a constant that has nothing to do with
|
|
1746
|
+
// why THIS request failed. On the clean tier that was 14 of 36 rows, reported
|
|
1747
|
+
// as "37 routes checked" when only 22 were ever candidates.
|
|
1748
|
+
//
|
|
1749
|
+
// An explicit pin is exempt: the client named that model, so it still gets
|
|
1750
|
+
// walked and still reports "no enabled+healthy key for platform" against its
|
|
1751
|
+
// own label rather than vanishing into an aggregate.
|
|
1752
|
+
const isRoutable = (e) => e.model_db_id === preferredModelDbId || (keyCounts.get(e.platform) ?? 0) > 0;
|
|
1753
|
+
const routableChain = sortedChain.filter(isRoutable);
|
|
1754
|
+
const keylessSkipped = sortedChain.length - routableChain.length;
|
|
1755
|
+
// One aggregate line, not one per model: the platforms stay visible to anyone
|
|
1756
|
+
// reading RouteError.diagnostics (and keep routingExhaustionBody classifying a
|
|
1757
|
+
// fully-unconfigured pool as 503 config, not a 429 rate limit), without N
|
|
1758
|
+
// near-identical rows drowning the request's real reasons.
|
|
1759
|
+
const keylessLine = keylessSkipped > 0
|
|
1760
|
+
? `${keylessSkipped} model(s) skipped: no enabled+healthy key for platform (${[...new Set(sortedChain.filter(e => !isRoutable(e)).map(e => e.platform))].sort().join(', ')})`
|
|
1761
|
+
: null;
|
|
1762
|
+
// Per-model disposition, attached to the exhaustion error when the loop falls
|
|
1763
|
+
// through with no route — the only record of WHY the pool was empty on the
|
|
1764
|
+
// synchronous "all exhausted" path (nothing downstream logs it). See issue _1.
|
|
1765
|
+
const diag = [];
|
|
1766
|
+
// Margin as a SOFT preference (#956 review): /v1/models advertises the raw
|
|
1767
|
+
// window, so clients legitimately pack requests right up to it — excluding
|
|
1768
|
+
// those models outright turned an already-handled upstream 400 into "all
|
|
1769
|
+
// models exhausted" with zero attempts. Keep the operator's order intact but
|
|
1770
|
+
// sweep margin-fitting models first; ones that only fit the advertised window
|
|
1771
|
+
// stay eligible behind them. Worst case is one classified context_too_large
|
|
1772
|
+
// hop instead of no route at all.
|
|
1773
|
+
//
|
|
1774
|
+
// Same soft treatment for models that keep answering tool requests with a 400
|
|
1775
|
+
// (#1230, lib/tool-capability.ts): on a tool request they go to the very back
|
|
1776
|
+
// instead of being excluded, so the worst case is the old order, never an
|
|
1777
|
+
// empty pool. An explicit pin keeps its place: the client named that model.
|
|
1778
|
+
const servingChain = [];
|
|
1779
|
+
const marginDeferred = [];
|
|
1780
|
+
const toolDeferred = [];
|
|
1781
|
+
for (const e of routableChain) {
|
|
1782
|
+
if (requireTools && e.model_db_id !== preferredModelDbId && isToolBenched(e.platform, e.model_id, e.endpoint_scope)) {
|
|
1783
|
+
toolDeferred.push(e);
|
|
1784
|
+
continue;
|
|
1785
|
+
}
|
|
1786
|
+
(fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve) ? servingChain : marginDeferred).push(e);
|
|
1787
|
+
}
|
|
1788
|
+
servingChain.push(...marginDeferred, ...toolDeferred);
|
|
1789
|
+
for (const entry of servingChain) {
|
|
1790
|
+
const label = `${entry.platform}/${entry.model_id}`;
|
|
1791
|
+
// Models the caller has ruled out for this request — e.g. a 404
|
|
1792
|
+
// "model removed upstream" already seen this request: trying the same
|
|
1793
|
+
// model again on a different key would just burn another attempt on the
|
|
1794
|
+
// same dead route (PR #111, credits @barbotkonv).
|
|
1795
|
+
if (skipModels?.has(entry.model_db_id)) {
|
|
1796
|
+
diag.push(`${label}: ruled out earlier this request`);
|
|
1797
|
+
continue;
|
|
1798
|
+
}
|
|
1799
|
+
// Platforms the caller has ruled out wholesale (#788): a provider-level
|
|
1800
|
+
// failure this request — a 5xx, a timeout, a dead socket — is about the
|
|
1801
|
+
// PROVIDER, so its other keys and its other models would fail the same way.
|
|
1802
|
+
// Skipping the platform moves failover to the next provider instead of
|
|
1803
|
+
// burning one hop per key. Request-scoped; nothing is benched by this.
|
|
1804
|
+
if (skipPlatforms?.has(entry.platform)) {
|
|
1805
|
+
diag.push(`${label}: provider ruled out earlier this request`);
|
|
1806
|
+
continue;
|
|
1807
|
+
}
|
|
1808
|
+
// Vision requests skip text-only models — including a sticky/preferred one,
|
|
1809
|
+
// which is correct: don't pin an image turn to a model that can't see it.
|
|
1810
|
+
if (requireVision && !entry.supports_vision) {
|
|
1811
|
+
diag.push(`${label}: no vision support`);
|
|
1812
|
+
continue;
|
|
1813
|
+
}
|
|
1814
|
+
// Tool-bearing requests skip models that can't emit structured tool_calls.
|
|
1815
|
+
// A model that "answers" a tool request with the call serialized as text
|
|
1816
|
+
// looks successful at the transport level while the client's harness sees
|
|
1817
|
+
// nothing — worse than a failover. Applies to sticky models too, same
|
|
1818
|
+
// reasoning as vision above.
|
|
1819
|
+
if (requireTools && !entry.supports_tools) {
|
|
1820
|
+
diag.push(`${label}: no tool-calling support`);
|
|
1821
|
+
continue;
|
|
1822
|
+
}
|
|
1823
|
+
// Structured-output routing (#514 follow-up): when the request carries a
|
|
1824
|
+
// response_format, skip platforms whose param policy can't even receive it
|
|
1825
|
+
// (the param would be dropped before send, so the model would answer in
|
|
1826
|
+
// prose and burn a failover hop). Platform-level fast path — model-level
|
|
1827
|
+
// capability isn't in the catalog; models that accept the param but ignore
|
|
1828
|
+
// it are caught by the non-stream JSON enforcement downstream.
|
|
1829
|
+
if (requireStructured && platformDropsResponseFormat(entry.platform)) {
|
|
1830
|
+
diag.push(`${label}: platform drops response_format`);
|
|
1831
|
+
continue;
|
|
1832
|
+
}
|
|
1833
|
+
// Context-aware routing fast path (#167): skip a model whose RAW advertised
|
|
1834
|
+
// window cannot hold the request — a dispatch there is a guaranteed 413.
|
|
1835
|
+
// Margin-violating-but-raw-fitting models are NOT skipped (soft preference,
|
|
1836
|
+
// #956 review): they were merely deferred to the back of the sweep above.
|
|
1837
|
+
// estimatedTokens is the INPUT estimate plus a CAPPED output reserve
|
|
1838
|
+
// (routingReserveTokens, #470), so a huge client-set max_tokens no longer
|
|
1839
|
+
// excludes the model — the input must fit, not input+full max_tokens. A 413
|
|
1840
|
+
// that slips through is still retryable downstream, and the failed model is
|
|
1841
|
+
// put on cooldown — so this is a fast-path, not the only guard. If every
|
|
1842
|
+
// model is too small, the loop falls through and the caller gets the normal
|
|
1843
|
+
// "all models exhausted" error rather than a wasted sweep.
|
|
1844
|
+
if (!fitsContextWindowStrict(entry.context_window, estimatedTokens)) {
|
|
1845
|
+
// Keep the `< estimated` substring — summarizeExhaustion buckets prompt
|
|
1846
|
+
// overflow off it. Emit the EFFECTIVE number so the line doesn't read as
|
|
1847
|
+
// false (#956 review): e.g. `context 131072 < estimated 106000 x1.25 = 132500`.
|
|
1848
|
+
const reserve = Math.max(0, exactOutputReserve);
|
|
1849
|
+
const requiredWithMargin = Math.ceil(Math.max(0, estimatedTokens - reserve) * CONTEXT_WINDOW_SAFETY_FACTOR + reserve);
|
|
1850
|
+
diag.push(TRIM_GUARDED_PLATFORMS.has(entry.platform)
|
|
1851
|
+
? `${label}: context ${entry.context_window} < estimated ${estimatedTokens}`
|
|
1852
|
+
: `${label}: context ${entry.context_window} < estimated ${estimatedTokens} x${CONTEXT_WINDOW_SAFETY_FACTOR} = ${requiredWithMargin}`);
|
|
1853
|
+
continue;
|
|
1854
|
+
}
|
|
1855
|
+
// Same guard for a model with a small per-minute token budget: a request
|
|
1856
|
+
// whose input alone exceeds tpm_limit can never fit one minute of quota and
|
|
1857
|
+
// returns a guaranteed 413 (e.g. Groq gpt-oss-120b: 131k context but 8k TPM).
|
|
1858
|
+
// estimatedTokens carries the same capped output reserve, mirroring the
|
|
1859
|
+
// check above (#470).
|
|
1860
|
+
if (entry.tpm_limit != null && estimatedTokens > entry.tpm_limit) {
|
|
1861
|
+
diag.push(`${label}: tpm_limit ${entry.tpm_limit} < estimated ${estimatedTokens}`);
|
|
1862
|
+
continue;
|
|
1863
|
+
}
|
|
1864
|
+
// Key selection + accounting pre-checks for this one model. Returns the
|
|
1865
|
+
// first usable key's RouteResult, or null when the model has no key that
|
|
1866
|
+
// can serve right now — in which case we fall through to the next model in
|
|
1867
|
+
// the sorted chain for THIS request (no explicit penalty needed).
|
|
1868
|
+
const route = selectKeyForModel(entry, estimatedTokens, skipKeys, diag);
|
|
1869
|
+
if (route)
|
|
1870
|
+
return route;
|
|
1871
|
+
}
|
|
1872
|
+
// The aggregate keyless line rides in diagnostics but NOT in the summary's
|
|
1873
|
+
// route count: those models were never candidates for this request.
|
|
1874
|
+
throw new RouteError(summarizeExhaustion(diag, getSoonestCooldownExpiry(), Date.now(), keylessSkipped), 429, keylessLine ? [...diag, keylessLine] : diag);
|
|
1875
|
+
}
|
|
1876
|
+
export function getRoutingScores() {
|
|
1877
|
+
const db = getDb();
|
|
1878
|
+
const strategy = getRoutingStrategy();
|
|
1879
|
+
refreshStatsCache(db);
|
|
1880
|
+
// Score the whole enabled catalog, not just the active chain (#1047). Since
|
|
1881
|
+
// #1023 the dashboard table lists every catalog row through the chain, and
|
|
1882
|
+
// merging it against chain-only scores left every not-yet-opted-in row with
|
|
1883
|
+
// blank reliability/speed — which read as "each new chain starts from
|
|
1884
|
+
// scratch" even though the underlying per-model stats are global. Rows the
|
|
1885
|
+
// chain names keep its enabled flag; the rest score as disabled. Display
|
|
1886
|
+
// only: routeRequest still walks getActiveChain.
|
|
1887
|
+
const profileId = getActiveProfileId(db);
|
|
1888
|
+
const chain = db.prepare(`
|
|
1889
|
+
SELECT m.id AS model_db_id, COALESCE(c.priority, 0) AS priority, COALESCE(c.enabled, 0) AS enabled,
|
|
1890
|
+
m.platform, m.model_id, m.display_name, m.intelligence_rank,
|
|
1891
|
+
m.size_label, m.monthly_token_budget,
|
|
1892
|
+
m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
|
|
1893
|
+
m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
|
|
1894
|
+
FROM models m
|
|
1895
|
+
LEFT JOIN ${profileId != null
|
|
1896
|
+
? '(SELECT model_db_id, priority, enabled FROM profile_models WHERE profile_id = ?) c'
|
|
1897
|
+
: '(SELECT model_db_id, priority, enabled FROM fallback_config) c'}
|
|
1898
|
+
ON c.model_db_id = m.id
|
|
1899
|
+
WHERE m.enabled = 1
|
|
1900
|
+
ORDER BY c.priority ASC
|
|
1901
|
+
`).all(...(profileId != null ? [profileId] : []));
|
|
1902
|
+
// For display we score under 'balanced' weights when in priority mode, so the
|
|
1903
|
+
// table still shows a meaningful ranking even with the bandit turned off.
|
|
1904
|
+
const weights = weightsFor(strategy) ?? BANDIT_PRESETS.balanced;
|
|
1905
|
+
const composites = chain.map(e => intelligenceComposite(e.size_label, e.intelligence_rank));
|
|
1906
|
+
const intelMin = composites.length ? Math.min(...composites) : 0;
|
|
1907
|
+
const intelMax = composites.length ? Math.max(...composites) : 0;
|
|
1908
|
+
const keyCounts = usableKeyCountsByPlatform(db);
|
|
1909
|
+
const headroomCfg = getHeadroomThresholds();
|
|
1910
|
+
const scores = chain.map(entry => {
|
|
1911
|
+
const scored = scoreChainEntry(entry, weights, intelMin, intelMax, false, keyCounts, headroomCfg);
|
|
1912
|
+
const stats = statsCache?.get(modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope));
|
|
1913
|
+
return {
|
|
1914
|
+
modelDbId: entry.model_db_id,
|
|
1915
|
+
platform: entry.platform,
|
|
1916
|
+
modelId: entry.model_id,
|
|
1917
|
+
displayName: entry.display_name,
|
|
1918
|
+
enabled: entry.enabled === 1,
|
|
1919
|
+
reliability: scored.axes.reliability,
|
|
1920
|
+
speed: scored.axes.speed,
|
|
1921
|
+
intelligence: scored.axes.intelligence,
|
|
1922
|
+
headroom: scored.headroom,
|
|
1923
|
+
rateLimit: scored.rateLimit,
|
|
1924
|
+
score: scored.score,
|
|
1925
|
+
totalRequests: Math.round((stats?.successes ?? 0) + (stats?.failures ?? 0)),
|
|
1926
|
+
};
|
|
1927
|
+
}).sort((a, b) => b.score - a.score);
|
|
1928
|
+
// customWeights is always present (the saved vector, or the balanced default)
|
|
1929
|
+
// so the dashboard's custom-weight sliders can render even before the user
|
|
1930
|
+
// has saved their own — distinct from `weights`, which is null in priority
|
|
1931
|
+
// mode and the active preset otherwise.
|
|
1932
|
+
// exploreEnabled must ride along here too: the dashboard checkbox renders
|
|
1933
|
+
// from GET /routing, so omitting it would make the toggle look permanently
|
|
1934
|
+
// off (and impossible to turn off) after a refetch. Same for the key
|
|
1935
|
+
// selection picker (#919).
|
|
1936
|
+
// peakAdjusted tells the dashboard whether the weight summary it is about to
|
|
1937
|
+
// render is the raw preset or a peak-hours variant of it (#760) — without it
|
|
1938
|
+
// the numbers would change under the operator with nothing to explain why.
|
|
1939
|
+
const active = weightsWithPeak(strategy);
|
|
1940
|
+
return {
|
|
1941
|
+
strategy,
|
|
1942
|
+
keySelectionStrategy: getKeySelectionStrategy(),
|
|
1943
|
+
weights: active.weights,
|
|
1944
|
+
customWeights: getCustomWeights(),
|
|
1945
|
+
exploreEnabled: getExploreEnabled(),
|
|
1946
|
+
peakAdjusted: active.adjusted,
|
|
1947
|
+
peakHours: getPeakHoursConfig(),
|
|
1948
|
+
scores,
|
|
1949
|
+
};
|
|
1950
|
+
}
|
|
1951
|
+
/**
|
|
1952
|
+
* Filter a sticky-session pin down to something still routable (#634).
|
|
1953
|
+
*
|
|
1954
|
+
* A sticky entry holds a model db id for up to 30 minutes, so it goes stale the
|
|
1955
|
+
* moment the operator disables that model — in the catalog, or just for auto
|
|
1956
|
+
* routing. It must NOT be handed to routeRequest as-is: an off-chain preferred
|
|
1957
|
+
* id is treated as an explicit pin and injected ahead of the chain, which is
|
|
1958
|
+
* right for a client that named the model and wrong for a pin the client never
|
|
1959
|
+
* asked for. Dropping it here falls the request through to normal auto routing.
|
|
1960
|
+
*
|
|
1961
|
+
* Pass the same chain the request will route over (the prefetched auto chain);
|
|
1962
|
+
* omit it to check the active chain, which is what routeRequest would use.
|
|
1963
|
+
*/
|
|
1964
|
+
export function resolveStickyPreference(stickyModelDbId, chain) {
|
|
1965
|
+
if (stickyModelDbId == null)
|
|
1966
|
+
return undefined;
|
|
1967
|
+
const rows = chain ?? getActiveChain(getDb());
|
|
1968
|
+
return rows.some(entry => entry.model_db_id === stickyModelDbId && entry.enabled)
|
|
1969
|
+
? stickyModelDbId
|
|
1970
|
+
: undefined;
|
|
1971
|
+
}
|
|
1972
|
+
// Whether at least one vision-capable model is enabled in the fallback chain.
|
|
1973
|
+
// Used to give image requests a clear "enable a vision model" error instead of
|
|
1974
|
+
// the generic exhaustion message when none is configured (#118, #125).
|
|
1975
|
+
export function hasEnabledVisionModel() {
|
|
1976
|
+
const db = getDb();
|
|
1977
|
+
return getActiveChain(db).some(entry => entry.enabled === 1 && entry.supports_vision === 1);
|
|
1978
|
+
}
|
|
1979
|
+
// Whether at least one tool-capable model is enabled in the fallback chain.
|
|
1980
|
+
// Same role as hasEnabledVisionModel: a clear up-front error for tool-bearing
|
|
1981
|
+
// requests beats routing them to a model that mangles the tool call.
|
|
1982
|
+
export function hasEnabledToolsModel() {
|
|
1983
|
+
const db = getDb();
|
|
1984
|
+
return getActiveChain(db).some(entry => entry.enabled === 1 && entry.supports_tools === 1);
|
|
1985
|
+
}
|
|
1986
|
+
//# sourceMappingURL=router.js.map
|