@srouterhq/server 0.0.0-stage → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/client/dist/assets/am-TQ7Jmdqe.js +1 -0
- package/client/dist/assets/ar-Bf5a7yfC.js +1 -0
- package/client/dist/assets/az-CrosE1xH.js +1 -0
- package/client/dist/assets/bg-BvWnmcUz.js +1 -0
- package/client/dist/assets/bn-ams5TsJp.js +1 -0
- package/client/dist/assets/cs-Dh67CaJ_.js +1 -0
- package/client/dist/assets/da-9WlT8Iq7.js +1 -0
- package/client/dist/assets/de-lWhJJzz7.js +1 -0
- package/client/dist/assets/el-Cu7_GhNq.js +1 -0
- package/client/dist/assets/es-BZuzsBcP.js +1 -0
- package/client/dist/assets/fa-Dwdn-jKS.js +1 -0
- package/client/dist/assets/fi-BFrFyOTy.js +1 -0
- package/client/dist/assets/fr-XhbtmPpj.js +1 -0
- package/client/dist/assets/geist-cyrillic-ext-wght-normal-DjL33-gN.woff2 +0 -0
- package/client/dist/assets/geist-cyrillic-wght-normal-BEAKL7Jp.woff2 +0 -0
- package/client/dist/assets/geist-latin-ext-wght-normal-DC-KSUi6.woff2 +0 -0
- package/client/dist/assets/geist-latin-wght-normal-BgDaEnEv.woff2 +0 -0
- package/client/dist/assets/geist-mono-cyrillic-ext-wght-normal-I4S5GZfc.woff2 +0 -0
- package/client/dist/assets/geist-mono-cyrillic-wght-normal-BmXc_FBt.woff2 +0 -0
- package/client/dist/assets/geist-mono-latin-ext-wght-normal-DrnZ1wKl.woff2 +0 -0
- package/client/dist/assets/geist-mono-latin-wght-normal-B_7UjwxQ.woff2 +0 -0
- package/client/dist/assets/geist-mono-symbols2-wght-normal-GZpp1pK2.woff2 +0 -0
- package/client/dist/assets/geist-mono-vietnamese-wght-normal-D8KDMBhC.woff2 +0 -0
- package/client/dist/assets/geist-vietnamese-wght-normal-6IgcOCM7.woff2 +0 -0
- package/client/dist/assets/gu-D0uhLUB8.js +1 -0
- package/client/dist/assets/ha-CVlm9pqp.js +1 -0
- package/client/dist/assets/he-CMIFeoWr.js +1 -0
- package/client/dist/assets/hi-ENTwhBpn.js +1 -0
- package/client/dist/assets/hr-fc03vMy_.js +1 -0
- package/client/dist/assets/hu-YBo2BIYt.js +1 -0
- package/client/dist/assets/id-BEWXYX69.js +1 -0
- package/client/dist/assets/ig-BKLZFKka.js +1 -0
- package/client/dist/assets/index-D9I4KtvR.js +145 -0
- package/client/dist/assets/index-txhEmCdb.css +2 -0
- package/client/dist/assets/it-Bzmei1Y2.js +1 -0
- package/client/dist/assets/ja-BfcpuqEl.js +1 -0
- package/client/dist/assets/ka-CnWjRgLo.js +1 -0
- package/client/dist/assets/km-BAjPR3Qj.js +1 -0
- package/client/dist/assets/kn-B15WKU5m.js +1 -0
- package/client/dist/assets/ko-C61fAB-w.js +1 -0
- package/client/dist/assets/lt-CYTr4_YW.js +1 -0
- package/client/dist/assets/ml-CtqxgyH3.js +1 -0
- package/client/dist/assets/mr-X5TMY7aR.js +1 -0
- package/client/dist/assets/ms-DSg8RqPS.js +1 -0
- package/client/dist/assets/my-D7RqR8uE.js +1 -0
- package/client/dist/assets/ne-DaflxkDj.js +1 -0
- package/client/dist/assets/nl-BOqqtsC8.js +1 -0
- package/client/dist/assets/no-BSHp_StB.js +1 -0
- package/client/dist/assets/or-BkLR9Vub.js +1 -0
- package/client/dist/assets/pa-CarJ2XgI.js +1 -0
- package/client/dist/assets/pl-C2n0V-0k.js +1 -0
- package/client/dist/assets/pt-BR-CqJdNhg2.js +1 -0
- package/client/dist/assets/pt-PT-C_HpMyKz.js +1 -0
- package/client/dist/assets/ro-DVRZqSAS.js +1 -0
- package/client/dist/assets/ru-DwAJG7VC.js +1 -0
- package/client/dist/assets/si-BxUezXm0.js +1 -0
- package/client/dist/assets/sk-eI1VcIfY.js +1 -0
- package/client/dist/assets/sr-CJF3auCr.js +1 -0
- package/client/dist/assets/srouter-logo-C6ZfjGIi.svg +14 -0
- package/client/dist/assets/sv-nqvwHAq_.js +1 -0
- package/client/dist/assets/sw-DWynE2pW.js +1 -0
- package/client/dist/assets/ta-BVM12knC.js +1 -0
- package/client/dist/assets/te-oyzEkF7_.js +1 -0
- package/client/dist/assets/th-COVeE0ZQ.js +1 -0
- package/client/dist/assets/tl-DsJANnQz.js +1 -0
- package/client/dist/assets/tr-B-Fl-urO.js +1 -0
- package/client/dist/assets/uk-C5qketAO.js +1 -0
- package/client/dist/assets/ur-QU9KugfV.js +1 -0
- package/client/dist/assets/uz-DL9cG_mY.js +1 -0
- package/client/dist/assets/vi-CQiVy3Pu.js +1 -0
- package/client/dist/assets/yo-DqazMdrZ.js +1 -0
- package/client/dist/assets/zh-CN-DnYL_Kio.js +1 -0
- package/client/dist/assets/zh-TW-5bdbosKj.js +1 -0
- package/client/dist/favicon.svg +14 -0
- package/client/dist/icons.svg +24 -0
- package/client/dist/index.html +37 -0
- package/dist/app.d.ts +4 -0
- package/dist/app.d.ts.map +1 -0
- package/dist/app.js +372 -0
- package/dist/app.js.map +1 -0
- package/dist/db/index.d.ts +49 -0
- package/dist/db/index.d.ts.map +1 -0
- package/dist/db/index.js +203 -0
- package/dist/db/index.js.map +1 -0
- package/dist/db/migrate/TEMPLATE.d.ts +4 -0
- package/dist/db/migrate/TEMPLATE.d.ts.map +1 -0
- package/dist/db/migrate/TEMPLATE.js +18 -0
- package/dist/db/migrate/TEMPLATE.js.map +1 -0
- package/dist/db/migrate/cli.d.ts +2 -0
- package/dist/db/migrate/cli.d.ts.map +1 -0
- package/dist/db/migrate/cli.js +177 -0
- package/dist/db/migrate/cli.js.map +1 -0
- package/dist/db/migrate/defaults.d.ts +52 -0
- package/dist/db/migrate/defaults.d.ts.map +1 -0
- package/dist/db/migrate/defaults.js +126 -0
- package/dist/db/migrate/defaults.js.map +1 -0
- package/dist/db/migrate/runner.d.ts +16 -0
- package/dist/db/migrate/runner.d.ts.map +1 -0
- package/dist/db/migrate/runner.js +178 -0
- package/dist/db/migrate/runner.js.map +1 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts +4 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts.map +1 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.js +2240 -0
- package/dist/db/migrations/20260101_000000_legacy_baseline.js.map +1 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts +4 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts.map +1 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.js +27 -0
- package/dist/db/migrations/20260627_000001_custom_provider_modalities.js.map +1 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts +4 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts.map +1 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.js +30 -0
- package/dist/db/migrations/20260627_000002_catalog_model_state.js.map +1 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.d.ts +6 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.d.ts.map +1 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.js +104 -0
- package/dist/db/migrations/20260628_120000_request_aggregates.js.map +1 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts +9 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts.map +1 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.js +22 -0
- package/dist/db/migrations/20260630_000001_github_gpt41_context.js.map +1 -0
- package/dist/db/migrations/20260706_000001_request_client_info.d.ts +4 -0
- package/dist/db/migrations/20260706_000001_request_client_info.d.ts.map +1 -0
- package/dist/db/migrations/20260706_000001_request_client_info.js +30 -0
- package/dist/db/migrations/20260706_000001_request_client_info.js.map +1 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts +17 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts.map +1 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.js +20 -0
- package/dist/db/migrations/20260706_000002_custom_model_tool_support.js.map +1 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts +12 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts.map +1 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.js +45 -0
- package/dist/db/migrations/20260714_000001_profile_chain_backfill.js.map +1 -0
- package/dist/db/migrations/20260720_000001_key_health_error.d.ts +5 -0
- package/dist/db/migrations/20260720_000001_key_health_error.d.ts.map +1 -0
- package/dist/db/migrations/20260720_000001_key_health_error.js +16 -0
- package/dist/db/migrations/20260720_000001_key_health_error.js.map +1 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts +4 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js +38 -0
- package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js.map +1 -0
- package/dist/db/migrations/20260726_000002_request_attempts.d.ts +4 -0
- package/dist/db/migrations/20260726_000002_request_attempts.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000002_request_attempts.js +43 -0
- package/dist/db/migrations/20260726_000002_request_attempts.js.map +1 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts +4 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.js +80 -0
- package/dist/db/migrations/20260726_000003_model_source_provenance.js.map +1 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.d.ts +4 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.js +34 -0
- package/dist/db/migrations/20260726_000004_media_model_meta.js.map +1 -0
- package/dist/db/migrations/20260726_000005_request_served_model.d.ts +4 -0
- package/dist/db/migrations/20260726_000005_request_served_model.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000005_request_served_model.js +31 -0
- package/dist/db/migrations/20260726_000005_request_served_model.js.map +1 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts +4 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts.map +1 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.js +29 -0
- package/dist/db/migrations/20260726_000006_attempt_error_summary.js.map +1 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts +4 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts.map +1 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.js +41 -0
- package/dist/db/migrations/20260727_000001_agent_compatibility.js.map +1 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts +12 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts.map +1 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.js +37 -0
- package/dist/db/migrations/20260728_000001_tombstone_provenance.js.map +1 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts +4 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts.map +1 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js +145 -0
- package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js.map +1 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts +6 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts.map +1 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js +52 -0
- package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js.map +1 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.d.ts +6 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.d.ts.map +1 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.js +17 -0
- package/dist/db/migrations/20260805_000001_key_model_scope.js.map +1 -0
- package/dist/db/migrations/20260805_000002_client_profiles.d.ts +4 -0
- package/dist/db/migrations/20260805_000002_client_profiles.d.ts.map +1 -0
- package/dist/db/migrations/20260805_000002_client_profiles.js +33 -0
- package/dist/db/migrations/20260805_000002_client_profiles.js.map +1 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts +4 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts.map +1 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.js +33 -0
- package/dist/db/migrations/20260810_000001_api_key_proxy.js.map +1 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts +4 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts.map +1 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.js +19 -0
- package/dist/db/migrations/20260819_000001_custom_model_tombstones.js.map +1 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.d.ts +4 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.d.ts.map +1 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.js +47 -0
- package/dist/db/migrations/20260820_000001_playground_conversations.js.map +1 -0
- package/dist/db/migrations/20260823_000001_server_logs.d.ts +4 -0
- package/dist/db/migrations/20260823_000001_server_logs.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000001_server_logs.js +53 -0
- package/dist/db/migrations/20260823_000001_server_logs.js.map +1 -0
- package/dist/db/migrations/20260823_000002_backups_table.d.ts +8 -0
- package/dist/db/migrations/20260823_000002_backups_table.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000002_backups_table.js +22 -0
- package/dist/db/migrations/20260823_000002_backups_table.js.map +1 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts +19 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.js +28 -0
- package/dist/db/migrations/20260823_000003_attempt_key_label.js.map +1 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts +14 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts.map +1 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.js +25 -0
- package/dist/db/migrations/20260823_000004_profile_auto_include.js.map +1 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts +4 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.js +46 -0
- package/dist/db/migrations/20260901_000001_idempotency_claims.js.map +1 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts +4 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.js +37 -0
- package/dist/db/migrations/20260901_000002_quota_observation_lookup.js.map +1 -0
- package/dist/db/migrations/20260901_000003_request_caller.d.ts +4 -0
- package/dist/db/migrations/20260901_000003_request_caller.d.ts.map +1 -0
- package/dist/db/migrations/20260901_000003_request_caller.js +27 -0
- package/dist/db/migrations/20260901_000003_request_caller.js.map +1 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts +4 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts.map +1 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js +35 -0
- package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js.map +1 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts +4 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts.map +1 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.js +37 -0
- package/dist/db/migrations/20260903_000001_mcp_enabled_default.js.map +1 -0
- package/dist/db/migrations/20260903_000002_response_cache.d.ts +4 -0
- package/dist/db/migrations/20260903_000002_response_cache.d.ts.map +1 -0
- package/dist/db/migrations/20260903_000002_response_cache.js +53 -0
- package/dist/db/migrations/20260903_000002_response_cache.js.map +1 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts +13 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts.map +1 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.js +35 -0
- package/dist/db/migrations/20260904_000001_key_monthly_budget.js.map +1 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts +4 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts.map +1 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.js +68 -0
- package/dist/db/migrations/20260913_000001_request_model_attribution.js.map +1 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts +5 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts.map +1 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.js +37 -0
- package/dist/db/migrations/20260914_000001_key_monthly_usage.js.map +1 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts +4 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts.map +1 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js +28 -0
- package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js.map +1 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts +4 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts.map +1 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js +25 -0
- package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js.map +1 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.d.ts +10 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.d.ts.map +1 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.js +78 -0
- package/dist/db/migrations/20261005_000001_key_budget_period.js.map +1 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts +21 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts.map +1 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.js +41 -0
- package/dist/db/migrations/20261005_000002_key_budget_period_auto.js.map +1 -0
- package/dist/db/model-pricing.d.ts +39 -0
- package/dist/db/model-pricing.d.ts.map +1 -0
- package/dist/db/model-pricing.js +274 -0
- package/dist/db/model-pricing.js.map +1 -0
- package/dist/db/node-sqlite.d.ts +9 -0
- package/dist/db/node-sqlite.d.ts.map +1 -0
- package/dist/db/node-sqlite.js +87 -0
- package/dist/db/node-sqlite.js.map +1 -0
- package/dist/db/types.d.ts +22 -0
- package/dist/db/types.d.ts.map +1 -0
- package/dist/db/types.js +2 -0
- package/dist/db/types.js.map +1 -0
- package/dist/docs/docs-page.d.ts +2 -0
- package/dist/docs/docs-page.d.ts.map +1 -0
- package/dist/docs/docs-page.js +319 -0
- package/dist/docs/docs-page.js.map +1 -0
- package/dist/docs/openapi.d.ts +1896 -0
- package/dist/docs/openapi.d.ts.map +1 -0
- package/dist/docs/openapi.js +1096 -0
- package/dist/docs/openapi.js.map +1 -0
- package/dist/env.d.ts +2 -0
- package/dist/env.d.ts.map +1 -0
- package/dist/env.js +9 -0
- package/dist/env.js.map +1 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +144 -0
- package/dist/index.js.map +1 -0
- package/dist/lib/anthropic-documents.d.ts +41 -0
- package/dist/lib/anthropic-documents.d.ts.map +1 -0
- package/dist/lib/anthropic-documents.js +132 -0
- package/dist/lib/anthropic-documents.js.map +1 -0
- package/dist/lib/app-version.d.ts +5 -0
- package/dist/lib/app-version.d.ts.map +1 -0
- package/dist/lib/app-version.js +61 -0
- package/dist/lib/app-version.js.map +1 -0
- package/dist/lib/attempt-trace.d.ts +23 -0
- package/dist/lib/attempt-trace.d.ts.map +1 -0
- package/dist/lib/attempt-trace.js +32 -0
- package/dist/lib/attempt-trace.js.map +1 -0
- package/dist/lib/budget.d.ts +6 -0
- package/dist/lib/budget.d.ts.map +1 -0
- package/dist/lib/budget.js +40 -0
- package/dist/lib/budget.js.map +1 -0
- package/dist/lib/client-classifier.d.ts +10 -0
- package/dist/lib/client-classifier.d.ts.map +1 -0
- package/dist/lib/client-classifier.js +126 -0
- package/dist/lib/client-classifier.js.map +1 -0
- package/dist/lib/client-context.d.ts +10 -0
- package/dist/lib/client-context.d.ts.map +1 -0
- package/dist/lib/client-context.js +39 -0
- package/dist/lib/client-context.js.map +1 -0
- package/dist/lib/config.d.ts +33 -0
- package/dist/lib/config.d.ts.map +1 -0
- package/dist/lib/config.js +94 -0
- package/dist/lib/config.js.map +1 -0
- package/dist/lib/content.d.ts +33 -0
- package/dist/lib/content.d.ts.map +1 -0
- package/dist/lib/content.js +201 -0
- package/dist/lib/content.js.map +1 -0
- package/dist/lib/credential.d.ts +11 -0
- package/dist/lib/credential.d.ts.map +1 -0
- package/dist/lib/credential.js +18 -0
- package/dist/lib/credential.js.map +1 -0
- package/dist/lib/crypto.d.ts +31 -0
- package/dist/lib/crypto.d.ts.map +1 -0
- package/dist/lib/crypto.js +206 -0
- package/dist/lib/crypto.js.map +1 -0
- package/dist/lib/custom-provider-cleanup.d.ts +13 -0
- package/dist/lib/custom-provider-cleanup.d.ts.map +1 -0
- package/dist/lib/custom-provider-cleanup.js +43 -0
- package/dist/lib/custom-provider-cleanup.js.map +1 -0
- package/dist/lib/db-backup.d.ts +31 -0
- package/dist/lib/db-backup.d.ts.map +1 -0
- package/dist/lib/db-backup.js +255 -0
- package/dist/lib/db-backup.js.map +1 -0
- package/dist/lib/endpoint-scope.d.ts +42 -0
- package/dist/lib/endpoint-scope.d.ts.map +1 -0
- package/dist/lib/endpoint-scope.js +95 -0
- package/dist/lib/endpoint-scope.js.map +1 -0
- package/dist/lib/env-drift.d.ts +17 -0
- package/dist/lib/env-drift.d.ts.map +1 -0
- package/dist/lib/env-drift.js +108 -0
- package/dist/lib/env-drift.js.map +1 -0
- package/dist/lib/error-classify.d.ts +40 -0
- package/dist/lib/error-classify.d.ts.map +1 -0
- package/dist/lib/error-classify.js +687 -0
- package/dist/lib/error-classify.js.map +1 -0
- package/dist/lib/error-redaction.d.ts +8 -0
- package/dist/lib/error-redaction.d.ts.map +1 -0
- package/dist/lib/error-redaction.js +45 -0
- package/dist/lib/error-redaction.js.map +1 -0
- package/dist/lib/fallback-loop.d.ts +306 -0
- package/dist/lib/fallback-loop.d.ts.map +1 -0
- package/dist/lib/fallback-loop.js +1336 -0
- package/dist/lib/fallback-loop.js.map +1 -0
- package/dist/lib/file-permissions.d.ts +98 -0
- package/dist/lib/file-permissions.d.ts.map +1 -0
- package/dist/lib/file-permissions.js +159 -0
- package/dist/lib/file-permissions.js.map +1 -0
- package/dist/lib/gemini-wire.d.ts +76 -0
- package/dist/lib/gemini-wire.d.ts.map +1 -0
- package/dist/lib/gemini-wire.js +392 -0
- package/dist/lib/gemini-wire.js.map +1 -0
- package/dist/lib/guardrails.d.ts +37 -0
- package/dist/lib/guardrails.d.ts.map +1 -0
- package/dist/lib/guardrails.js +106 -0
- package/dist/lib/guardrails.js.map +1 -0
- package/dist/lib/header-value.d.ts +8 -0
- package/dist/lib/header-value.d.ts.map +1 -0
- package/dist/lib/header-value.js +55 -0
- package/dist/lib/header-value.js.map +1 -0
- package/dist/lib/image-normalize.d.ts +21 -0
- package/dist/lib/image-normalize.d.ts.map +1 -0
- package/dist/lib/image-normalize.js +218 -0
- package/dist/lib/image-normalize.js.map +1 -0
- package/dist/lib/inbound-chat.d.ts +48 -0
- package/dist/lib/inbound-chat.d.ts.map +1 -0
- package/dist/lib/inbound-chat.js +406 -0
- package/dist/lib/inbound-chat.js.map +1 -0
- package/dist/lib/key-parser.d.ts +79 -0
- package/dist/lib/key-parser.d.ts.map +1 -0
- package/dist/lib/key-parser.js +701 -0
- package/dist/lib/key-parser.js.map +1 -0
- package/dist/lib/key-proxy.d.ts +49 -0
- package/dist/lib/key-proxy.d.ts.map +1 -0
- package/dist/lib/key-proxy.js +92 -0
- package/dist/lib/key-proxy.js.map +1 -0
- package/dist/lib/log-redaction.d.ts +28 -0
- package/dist/lib/log-redaction.d.ts.map +1 -0
- package/dist/lib/log-redaction.js +166 -0
- package/dist/lib/log-redaction.js.map +1 -0
- package/dist/lib/model-scope.d.ts +9 -0
- package/dist/lib/model-scope.d.ts.map +1 -0
- package/dist/lib/model-scope.js +29 -0
- package/dist/lib/model-scope.js.map +1 -0
- package/dist/lib/one-time-code.d.ts +15 -0
- package/dist/lib/one-time-code.d.ts.map +1 -0
- package/dist/lib/one-time-code.js +41 -0
- package/dist/lib/one-time-code.js.map +1 -0
- package/dist/lib/output-cap.d.ts +24 -0
- package/dist/lib/output-cap.d.ts.map +1 -0
- package/dist/lib/output-cap.js +80 -0
- package/dist/lib/output-cap.js.map +1 -0
- package/dist/lib/password.d.ts +3 -0
- package/dist/lib/password.d.ts.map +1 -0
- package/dist/lib/password.js +27 -0
- package/dist/lib/password.js.map +1 -0
- package/dist/lib/process-safety-net.d.ts +32 -0
- package/dist/lib/process-safety-net.d.ts.map +1 -0
- package/dist/lib/process-safety-net.js +123 -0
- package/dist/lib/process-safety-net.js.map +1 -0
- package/dist/lib/provider-identity.d.ts +37 -0
- package/dist/lib/provider-identity.d.ts.map +1 -0
- package/dist/lib/provider-identity.js +110 -0
- package/dist/lib/provider-identity.js.map +1 -0
- package/dist/lib/provider-size-parser.d.ts +6 -0
- package/dist/lib/provider-size-parser.d.ts.map +1 -0
- package/dist/lib/provider-size-parser.js +72 -0
- package/dist/lib/provider-size-parser.js.map +1 -0
- package/dist/lib/provider-timeout.d.ts +17 -0
- package/dist/lib/provider-timeout.d.ts.map +1 -0
- package/dist/lib/provider-timeout.js +70 -0
- package/dist/lib/provider-timeout.js.map +1 -0
- package/dist/lib/proxy.d.ts +180 -0
- package/dist/lib/proxy.d.ts.map +1 -0
- package/dist/lib/proxy.js +1002 -0
- package/dist/lib/proxy.js.map +1 -0
- package/dist/lib/request-log.d.ts +4 -0
- package/dist/lib/request-log.d.ts.map +1 -0
- package/dist/lib/request-log.js +161 -0
- package/dist/lib/request-log.js.map +1 -0
- package/dist/lib/reset-code.d.ts +5 -0
- package/dist/lib/reset-code.d.ts.map +1 -0
- package/dist/lib/reset-code.js +39 -0
- package/dist/lib/reset-code.js.map +1 -0
- package/dist/lib/retry-hint.d.ts +14 -0
- package/dist/lib/retry-hint.d.ts.map +1 -0
- package/dist/lib/retry-hint.js +36 -0
- package/dist/lib/retry-hint.js.map +1 -0
- package/dist/lib/route-live.d.ts +29 -0
- package/dist/lib/route-live.d.ts.map +1 -0
- package/dist/lib/route-live.js +97 -0
- package/dist/lib/route-live.js.map +1 -0
- package/dist/lib/sampling-params.d.ts +183 -0
- package/dist/lib/sampling-params.d.ts.map +1 -0
- package/dist/lib/sampling-params.js +386 -0
- package/dist/lib/sampling-params.js.map +1 -0
- package/dist/lib/scheduler.d.ts +13 -0
- package/dist/lib/scheduler.d.ts.map +1 -0
- package/dist/lib/scheduler.js +11 -0
- package/dist/lib/scheduler.js.map +1 -0
- package/dist/lib/served-model.d.ts +16 -0
- package/dist/lib/served-model.d.ts.map +1 -0
- package/dist/lib/served-model.js +0 -0
- package/dist/lib/served-model.js.map +1 -0
- package/dist/lib/server-logs.d.ts +132 -0
- package/dist/lib/server-logs.d.ts.map +1 -0
- package/dist/lib/server-logs.js +473 -0
- package/dist/lib/server-logs.js.map +1 -0
- package/dist/lib/setup-code.d.ts +5 -0
- package/dist/lib/setup-code.d.ts.map +1 -0
- package/dist/lib/setup-code.js +35 -0
- package/dist/lib/setup-code.js.map +1 -0
- package/dist/lib/structured-output.d.ts +9 -0
- package/dist/lib/structured-output.d.ts.map +1 -0
- package/dist/lib/structured-output.js +72 -0
- package/dist/lib/structured-output.js.map +1 -0
- package/dist/lib/system-prompt.d.ts +30 -0
- package/dist/lib/system-prompt.d.ts.map +1 -0
- package/dist/lib/system-prompt.js +66 -0
- package/dist/lib/system-prompt.js.map +1 -0
- package/dist/lib/task-type.d.ts +18 -0
- package/dist/lib/task-type.d.ts.map +1 -0
- package/dist/lib/task-type.js +92 -0
- package/dist/lib/task-type.js.map +1 -0
- package/dist/lib/think-tags.d.ts +37 -0
- package/dist/lib/think-tags.d.ts.map +1 -0
- package/dist/lib/think-tags.js +203 -0
- package/dist/lib/think-tags.js.map +1 -0
- package/dist/lib/tool-args.d.ts +36 -0
- package/dist/lib/tool-args.d.ts.map +1 -0
- package/dist/lib/tool-args.js +190 -0
- package/dist/lib/tool-args.js.map +1 -0
- package/dist/lib/tool-call-rescue.d.ts +62 -0
- package/dist/lib/tool-call-rescue.d.ts.map +1 -0
- package/dist/lib/tool-call-rescue.js +258 -0
- package/dist/lib/tool-call-rescue.js.map +1 -0
- package/dist/lib/tool-capability.d.ts +13 -0
- package/dist/lib/tool-capability.d.ts.map +1 -0
- package/dist/lib/tool-capability.js +70 -0
- package/dist/lib/tool-capability.js.map +1 -0
- package/dist/lib/tool-validate.d.ts +44 -0
- package/dist/lib/tool-validate.d.ts.map +1 -0
- package/dist/lib/tool-validate.js +165 -0
- package/dist/lib/tool-validate.js.map +1 -0
- package/dist/lib/ttfb-budget.d.ts +13 -0
- package/dist/lib/ttfb-budget.d.ts.map +1 -0
- package/dist/lib/ttfb-budget.js +153 -0
- package/dist/lib/ttfb-budget.js.map +1 -0
- package/dist/lib/url-guard.d.ts +47 -0
- package/dist/lib/url-guard.d.ts.map +1 -0
- package/dist/lib/url-guard.js +229 -0
- package/dist/lib/url-guard.js.map +1 -0
- package/dist/lib/wake-detect.d.ts +12 -0
- package/dist/lib/wake-detect.d.ts.map +1 -0
- package/dist/lib/wake-detect.js +99 -0
- package/dist/lib/wake-detect.js.map +1 -0
- package/dist/middleware/errorHandler.d.ts +3 -0
- package/dist/middleware/errorHandler.d.ts.map +1 -0
- package/dist/middleware/errorHandler.js +45 -0
- package/dist/middleware/errorHandler.js.map +1 -0
- package/dist/middleware/rateLimit.d.ts +4 -0
- package/dist/middleware/rateLimit.d.ts.map +1 -0
- package/dist/middleware/rateLimit.js +125 -0
- package/dist/middleware/rateLimit.js.map +1 -0
- package/dist/middleware/requireAuth.d.ts +3 -0
- package/dist/middleware/requireAuth.d.ts.map +1 -0
- package/dist/middleware/requireAuth.js +17 -0
- package/dist/middleware/requireAuth.js.map +1 -0
- package/dist/providers/aclide.d.ts +18 -0
- package/dist/providers/aclide.d.ts.map +1 -0
- package/dist/providers/aclide.js +174 -0
- package/dist/providers/aclide.js.map +1 -0
- package/dist/providers/aihorde.d.ts +42 -0
- package/dist/providers/aihorde.d.ts.map +1 -0
- package/dist/providers/aihorde.js +192 -0
- package/dist/providers/aihorde.js.map +1 -0
- package/dist/providers/airforce.d.ts +11 -0
- package/dist/providers/airforce.d.ts.map +1 -0
- package/dist/providers/airforce.js +32 -0
- package/dist/providers/airforce.js.map +1 -0
- package/dist/providers/base.d.ts +177 -0
- package/dist/providers/base.d.ts.map +1 -0
- package/dist/providers/base.js +354 -0
- package/dist/providers/base.js.map +1 -0
- package/dist/providers/blaze.d.ts +11 -0
- package/dist/providers/blaze.d.ts.map +1 -0
- package/dist/providers/blaze.js +33 -0
- package/dist/providers/blaze.js.map +1 -0
- package/dist/providers/blockrun.d.ts +14 -0
- package/dist/providers/blockrun.d.ts.map +1 -0
- package/dist/providers/blockrun.js +36 -0
- package/dist/providers/blockrun.js.map +1 -0
- package/dist/providers/clod.d.ts +11 -0
- package/dist/providers/clod.d.ts.map +1 -0
- package/dist/providers/clod.js +39 -0
- package/dist/providers/clod.js.map +1 -0
- package/dist/providers/cloudflare.d.ts +15 -0
- package/dist/providers/cloudflare.d.ts.map +1 -0
- package/dist/providers/cloudflare.js +175 -0
- package/dist/providers/cloudflare.js.map +1 -0
- package/dist/providers/cohere.d.ts +11 -0
- package/dist/providers/cohere.d.ts.map +1 -0
- package/dist/providers/cohere.js +117 -0
- package/dist/providers/cohere.js.map +1 -0
- package/dist/providers/dreamprompting.d.ts +11 -0
- package/dist/providers/dreamprompting.d.ts.map +1 -0
- package/dist/providers/dreamprompting.js +41 -0
- package/dist/providers/dreamprompting.js.map +1 -0
- package/dist/providers/electronhub.d.ts +10 -0
- package/dist/providers/electronhub.d.ts.map +1 -0
- package/dist/providers/electronhub.js +56 -0
- package/dist/providers/electronhub.js.map +1 -0
- package/dist/providers/experiential.d.ts +10 -0
- package/dist/providers/experiential.d.ts.map +1 -0
- package/dist/providers/experiential.js +33 -0
- package/dist/providers/experiential.js.map +1 -0
- package/dist/providers/gizmo.d.ts +15 -0
- package/dist/providers/gizmo.d.ts.map +1 -0
- package/dist/providers/gizmo.js +54 -0
- package/dist/providers/gizmo.js.map +1 -0
- package/dist/providers/google.d.ts +30 -0
- package/dist/providers/google.d.ts.map +1 -0
- package/dist/providers/google.js +766 -0
- package/dist/providers/google.js.map +1 -0
- package/dist/providers/index.d.ts +13 -0
- package/dist/providers/index.d.ts.map +1 -0
- package/dist/providers/index.js +561 -0
- package/dist/providers/index.js.map +1 -0
- package/dist/providers/llmtr.d.ts +15 -0
- package/dist/providers/llmtr.d.ts.map +1 -0
- package/dist/providers/llmtr.js +51 -0
- package/dist/providers/llmtr.js.map +1 -0
- package/dist/providers/logfare.d.ts +11 -0
- package/dist/providers/logfare.d.ts.map +1 -0
- package/dist/providers/logfare.js +33 -0
- package/dist/providers/logfare.js.map +1 -0
- package/dist/providers/lucidity.d.ts +11 -0
- package/dist/providers/lucidity.d.ts.map +1 -0
- package/dist/providers/lucidity.js +37 -0
- package/dist/providers/lucidity.js.map +1 -0
- package/dist/providers/modelscope.d.ts +50 -0
- package/dist/providers/modelscope.d.ts.map +1 -0
- package/dist/providers/modelscope.js +143 -0
- package/dist/providers/modelscope.js.map +1 -0
- package/dist/providers/moondream.d.ts +17 -0
- package/dist/providers/moondream.d.ts.map +1 -0
- package/dist/providers/moondream.js +140 -0
- package/dist/providers/moondream.js.map +1 -0
- package/dist/providers/openai-compat.d.ts +140 -0
- package/dist/providers/openai-compat.d.ts.map +1 -0
- package/dist/providers/openai-compat.js +537 -0
- package/dist/providers/openai-compat.js.map +1 -0
- package/dist/providers/opencode-free.d.ts +24 -0
- package/dist/providers/opencode-free.d.ts.map +1 -0
- package/dist/providers/opencode-free.js +262 -0
- package/dist/providers/opencode-free.js.map +1 -0
- package/dist/providers/pollinations.d.ts +32 -0
- package/dist/providers/pollinations.d.ts.map +1 -0
- package/dist/providers/pollinations.js +65 -0
- package/dist/providers/pollinations.js.map +1 -0
- package/dist/providers/router9.d.ts +16 -0
- package/dist/providers/router9.d.ts.map +1 -0
- package/dist/providers/router9.js +42 -0
- package/dist/providers/router9.js.map +1 -0
- package/dist/providers/sail.d.ts +44 -0
- package/dist/providers/sail.d.ts.map +1 -0
- package/dist/providers/sail.js +302 -0
- package/dist/providers/sail.js.map +1 -0
- package/dist/providers/septor.d.ts +10 -0
- package/dist/providers/septor.d.ts.map +1 -0
- package/dist/providers/septor.js +36 -0
- package/dist/providers/septor.js.map +1 -0
- package/dist/providers/speechify.d.ts +14 -0
- package/dist/providers/speechify.d.ts.map +1 -0
- package/dist/providers/speechify.js +28 -0
- package/dist/providers/speechify.js.map +1 -0
- package/dist/providers/speka.d.ts +12 -0
- package/dist/providers/speka.d.ts.map +1 -0
- package/dist/providers/speka.js +37 -0
- package/dist/providers/speka.js.map +1 -0
- package/dist/providers/waterfall.d.ts +11 -0
- package/dist/providers/waterfall.d.ts.map +1 -0
- package/dist/providers/waterfall.js +33 -0
- package/dist/providers/waterfall.js.map +1 -0
- package/dist/providers/xkiro.d.ts +71 -0
- package/dist/providers/xkiro.d.ts.map +1 -0
- package/dist/providers/xkiro.js +119 -0
- package/dist/providers/xkiro.js.map +1 -0
- package/dist/providers/zhipu.d.ts +43 -0
- package/dist/providers/zhipu.d.ts.map +1 -0
- package/dist/providers/zhipu.js +95 -0
- package/dist/providers/zhipu.js.map +1 -0
- package/dist/routes/analytics.d.ts +2 -0
- package/dist/routes/analytics.d.ts.map +1 -0
- package/dist/routes/analytics.js +738 -0
- package/dist/routes/analytics.js.map +1 -0
- package/dist/routes/anthropic.d.ts +7 -0
- package/dist/routes/anthropic.d.ts.map +1 -0
- package/dist/routes/anthropic.js +1076 -0
- package/dist/routes/anthropic.js.map +1 -0
- package/dist/routes/auth.d.ts +2 -0
- package/dist/routes/auth.d.ts.map +1 -0
- package/dist/routes/auth.js +310 -0
- package/dist/routes/auth.js.map +1 -0
- package/dist/routes/backups.d.ts +2 -0
- package/dist/routes/backups.d.ts.map +1 -0
- package/dist/routes/backups.js +127 -0
- package/dist/routes/backups.js.map +1 -0
- package/dist/routes/cache.d.ts +2 -0
- package/dist/routes/cache.d.ts.map +1 -0
- package/dist/routes/cache.js +40 -0
- package/dist/routes/cache.js.map +1 -0
- package/dist/routes/client-profiles.d.ts +2 -0
- package/dist/routes/client-profiles.d.ts.map +1 -0
- package/dist/routes/client-profiles.js +128 -0
- package/dist/routes/client-profiles.js.map +1 -0
- package/dist/routes/compression.d.ts +2 -0
- package/dist/routes/compression.d.ts.map +1 -0
- package/dist/routes/compression.js +82 -0
- package/dist/routes/compression.js.map +1 -0
- package/dist/routes/conversations.d.ts +3 -0
- package/dist/routes/conversations.d.ts.map +1 -0
- package/dist/routes/conversations.js +215 -0
- package/dist/routes/conversations.js.map +1 -0
- package/dist/routes/docs.d.ts +2 -0
- package/dist/routes/docs.d.ts.map +1 -0
- package/dist/routes/docs.js +21 -0
- package/dist/routes/docs.js.map +1 -0
- package/dist/routes/embeddings.d.ts +2 -0
- package/dist/routes/embeddings.d.ts.map +1 -0
- package/dist/routes/embeddings.js +255 -0
- package/dist/routes/embeddings.js.map +1 -0
- package/dist/routes/fallback.d.ts +2 -0
- package/dist/routes/fallback.d.ts.map +1 -0
- package/dist/routes/fallback.js +631 -0
- package/dist/routes/fallback.js.map +1 -0
- package/dist/routes/free-tier.d.ts +22 -0
- package/dist/routes/free-tier.d.ts.map +1 -0
- package/dist/routes/free-tier.js +175 -0
- package/dist/routes/free-tier.js.map +1 -0
- package/dist/routes/gemini.d.ts +2 -0
- package/dist/routes/gemini.d.ts.map +1 -0
- package/dist/routes/gemini.js +218 -0
- package/dist/routes/gemini.js.map +1 -0
- package/dist/routes/health.d.ts +2 -0
- package/dist/routes/health.d.ts.map +1 -0
- package/dist/routes/health.js +72 -0
- package/dist/routes/health.js.map +1 -0
- package/dist/routes/keys.d.ts +17 -0
- package/dist/routes/keys.d.ts.map +1 -0
- package/dist/routes/keys.js +1787 -0
- package/dist/routes/keys.js.map +1 -0
- package/dist/routes/logs.d.ts +2 -0
- package/dist/routes/logs.d.ts.map +1 -0
- package/dist/routes/logs.js +88 -0
- package/dist/routes/logs.js.map +1 -0
- package/dist/routes/mcp.d.ts +4 -0
- package/dist/routes/mcp.d.ts.map +1 -0
- package/dist/routes/mcp.js +563 -0
- package/dist/routes/mcp.js.map +1 -0
- package/dist/routes/media.d.ts +2 -0
- package/dist/routes/media.d.ts.map +1 -0
- package/dist/routes/media.js +180 -0
- package/dist/routes/media.js.map +1 -0
- package/dist/routes/models.d.ts +2 -0
- package/dist/routes/models.d.ts.map +1 -0
- package/dist/routes/models.js +439 -0
- package/dist/routes/models.js.map +1 -0
- package/dist/routes/ollama.d.ts +7 -0
- package/dist/routes/ollama.d.ts.map +1 -0
- package/dist/routes/ollama.js +595 -0
- package/dist/routes/ollama.js.map +1 -0
- package/dist/routes/profiles.d.ts +3 -0
- package/dist/routes/profiles.d.ts.map +1 -0
- package/dist/routes/profiles.js +431 -0
- package/dist/routes/profiles.js.map +1 -0
- package/dist/routes/proxy.d.ts +32 -0
- package/dist/routes/proxy.d.ts.map +1 -0
- package/dist/routes/proxy.js +2793 -0
- package/dist/routes/proxy.js.map +1 -0
- package/dist/routes/responses.d.ts +1161 -0
- package/dist/routes/responses.d.ts.map +1 -0
- package/dist/routes/responses.js +1521 -0
- package/dist/routes/responses.js.map +1 -0
- package/dist/routes/settings.d.ts +2 -0
- package/dist/routes/settings.d.ts.map +1 -0
- package/dist/routes/settings.js +549 -0
- package/dist/routes/settings.js.map +1 -0
- package/dist/routes/status.d.ts +3 -0
- package/dist/routes/status.d.ts.map +1 -0
- package/dist/routes/status.js +214 -0
- package/dist/routes/status.js.map +1 -0
- package/dist/routes/update.d.ts +31 -0
- package/dist/routes/update.d.ts.map +1 -0
- package/dist/routes/update.js +536 -0
- package/dist/routes/update.js.map +1 -0
- package/dist/routes/url-tokens.d.ts +2 -0
- package/dist/routes/url-tokens.d.ts.map +1 -0
- package/dist/routes/url-tokens.js +31 -0
- package/dist/routes/url-tokens.js.map +1 -0
- package/dist/scripts/export-catalog.d.ts +2 -0
- package/dist/scripts/export-catalog.d.ts.map +1 -0
- package/dist/scripts/export-catalog.js +114 -0
- package/dist/scripts/export-catalog.js.map +1 -0
- package/dist/scripts/rotate-encryption-key.d.ts +81 -0
- package/dist/scripts/rotate-encryption-key.d.ts.map +1 -0
- package/dist/scripts/rotate-encryption-key.js +232 -0
- package/dist/scripts/rotate-encryption-key.js.map +1 -0
- package/dist/scripts/routing-sim.d.ts +2 -0
- package/dist/scripts/routing-sim.d.ts.map +1 -0
- package/dist/scripts/routing-sim.js +130 -0
- package/dist/scripts/routing-sim.js.map +1 -0
- package/dist/scripts/test-all-models.d.ts +2 -0
- package/dist/scripts/test-all-models.d.ts.map +1 -0
- package/dist/scripts/test-all-models.js +56 -0
- package/dist/scripts/test-all-models.js.map +1 -0
- package/dist/services/anthropic-map.d.ts +39 -0
- package/dist/services/anthropic-map.d.ts.map +1 -0
- package/dist/services/anthropic-map.js +138 -0
- package/dist/services/anthropic-map.js.map +1 -0
- package/dist/services/auth.d.ts +30 -0
- package/dist/services/auth.d.ts.map +1 -0
- package/dist/services/auth.js +124 -0
- package/dist/services/auth.js.map +1 -0
- package/dist/services/auto-discover.d.ts +77 -0
- package/dist/services/auto-discover.d.ts.map +1 -0
- package/dist/services/auto-discover.js +1047 -0
- package/dist/services/auto-discover.js.map +1 -0
- package/dist/services/backups.d.ts +67 -0
- package/dist/services/backups.d.ts.map +1 -0
- package/dist/services/backups.js +444 -0
- package/dist/services/backups.js.map +1 -0
- package/dist/services/builtin-model-discovery.d.ts +107 -0
- package/dist/services/builtin-model-discovery.d.ts.map +1 -0
- package/dist/services/builtin-model-discovery.js +296 -0
- package/dist/services/builtin-model-discovery.js.map +1 -0
- package/dist/services/cache.d.ts +190 -0
- package/dist/services/cache.d.ts.map +1 -0
- package/dist/services/cache.js +617 -0
- package/dist/services/cache.js.map +1 -0
- package/dist/services/catalog-sync.d.ts +257 -0
- package/dist/services/catalog-sync.d.ts.map +1 -0
- package/dist/services/catalog-sync.js +1041 -0
- package/dist/services/catalog-sync.js.map +1 -0
- package/dist/services/compression/config.d.ts +33 -0
- package/dist/services/compression/config.d.ts.map +1 -0
- package/dist/services/compression/config.js +158 -0
- package/dist/services/compression/config.js.map +1 -0
- package/dist/services/compression/engines/aging.d.ts +2 -0
- package/dist/services/compression/engines/aging.d.ts.map +1 -0
- package/dist/services/compression/engines/aging.js +53 -0
- package/dist/services/compression/engines/aging.js.map +1 -0
- package/dist/services/compression/engines/custom-filters.d.ts +4 -0
- package/dist/services/compression/engines/custom-filters.d.ts.map +1 -0
- package/dist/services/compression/engines/custom-filters.js +49 -0
- package/dist/services/compression/engines/custom-filters.js.map +1 -0
- package/dist/services/compression/engines/dedup.d.ts +2 -0
- package/dist/services/compression/engines/dedup.d.ts.map +1 -0
- package/dist/services/compression/engines/dedup.js +61 -0
- package/dist/services/compression/engines/dedup.js.map +1 -0
- package/dist/services/compression/engines/filter-definitions.d.ts +75 -0
- package/dist/services/compression/engines/filter-definitions.d.ts.map +1 -0
- package/dist/services/compression/engines/filter-definitions.js +45 -0
- package/dist/services/compression/engines/filter-definitions.js.map +1 -0
- package/dist/services/compression/engines/hard-budget.d.ts +2 -0
- package/dist/services/compression/engines/hard-budget.d.ts.map +1 -0
- package/dist/services/compression/engines/hard-budget.js +52 -0
- package/dist/services/compression/engines/hard-budget.js.map +1 -0
- package/dist/services/compression/engines/index.d.ts +9 -0
- package/dist/services/compression/engines/index.d.ts.map +1 -0
- package/dist/services/compression/engines/index.js +9 -0
- package/dist/services/compression/engines/index.js.map +1 -0
- package/dist/services/compression/engines/jsoncompact.d.ts +3 -0
- package/dist/services/compression/engines/jsoncompact.d.ts.map +1 -0
- package/dist/services/compression/engines/jsoncompact.js +145 -0
- package/dist/services/compression/engines/jsoncompact.js.map +1 -0
- package/dist/services/compression/engines/lite.d.ts +2 -0
- package/dist/services/compression/engines/lite.d.ts.map +1 -0
- package/dist/services/compression/engines/lite.js +34 -0
- package/dist/services/compression/engines/lite.js.map +1 -0
- package/dist/services/compression/engines/read-lifecycle.d.ts +2 -0
- package/dist/services/compression/engines/read-lifecycle.d.ts.map +1 -0
- package/dist/services/compression/engines/read-lifecycle.js +70 -0
- package/dist/services/compression/engines/read-lifecycle.js.map +1 -0
- package/dist/services/compression/engines/relevance.d.ts +2 -0
- package/dist/services/compression/engines/relevance.d.ts.map +1 -0
- package/dist/services/compression/engines/relevance.js +74 -0
- package/dist/services/compression/engines/relevance.js.map +1 -0
- package/dist/services/compression/engines/toolfilter.d.ts +2 -0
- package/dist/services/compression/engines/toolfilter.d.ts.map +1 -0
- package/dist/services/compression/engines/toolfilter.js +195 -0
- package/dist/services/compression/engines/toolfilter.js.map +1 -0
- package/dist/services/compression/fidelity-gate.d.ts +12 -0
- package/dist/services/compression/fidelity-gate.d.ts.map +1 -0
- package/dist/services/compression/fidelity-gate.js +90 -0
- package/dist/services/compression/fidelity-gate.js.map +1 -0
- package/dist/services/compression/helpers.d.ts +9 -0
- package/dist/services/compression/helpers.d.ts.map +1 -0
- package/dist/services/compression/helpers.js +52 -0
- package/dist/services/compression/helpers.js.map +1 -0
- package/dist/services/compression/pipeline.d.ts +7 -0
- package/dist/services/compression/pipeline.d.ts.map +1 -0
- package/dist/services/compression/pipeline.js +245 -0
- package/dist/services/compression/pipeline.js.map +1 -0
- package/dist/services/compression/preservation.d.ts +27 -0
- package/dist/services/compression/preservation.d.ts.map +1 -0
- package/dist/services/compression/preservation.js +109 -0
- package/dist/services/compression/preservation.js.map +1 -0
- package/dist/services/compression/registry.d.ts +6 -0
- package/dist/services/compression/registry.d.ts.map +1 -0
- package/dist/services/compression/registry.js +16 -0
- package/dist/services/compression/registry.js.map +1 -0
- package/dist/services/compression/stats.d.ts +29 -0
- package/dist/services/compression/stats.d.ts.map +1 -0
- package/dist/services/compression/stats.js +55 -0
- package/dist/services/compression/stats.js.map +1 -0
- package/dist/services/compression/types.d.ts +90 -0
- package/dist/services/compression/types.d.ts.map +1 -0
- package/dist/services/compression/types.js +2 -0
- package/dist/services/compression/types.js.map +1 -0
- package/dist/services/context-handoff.d.ts +22 -0
- package/dist/services/context-handoff.d.ts.map +1 -0
- package/dist/services/context-handoff.js +164 -0
- package/dist/services/context-handoff.js.map +1 -0
- package/dist/services/cooldown-probe.d.ts +24 -0
- package/dist/services/cooldown-probe.d.ts.map +1 -0
- package/dist/services/cooldown-probe.js +181 -0
- package/dist/services/cooldown-probe.js.map +1 -0
- package/dist/services/custom-endpoint.d.ts +58 -0
- package/dist/services/custom-endpoint.d.ts.map +1 -0
- package/dist/services/custom-endpoint.js +167 -0
- package/dist/services/custom-endpoint.js.map +1 -0
- package/dist/services/custom-media-register.d.ts +22 -0
- package/dist/services/custom-media-register.d.ts.map +1 -0
- package/dist/services/custom-media-register.js +45 -0
- package/dist/services/custom-media-register.js.map +1 -0
- package/dist/services/custom-model-register.d.ts +50 -0
- package/dist/services/custom-model-register.d.ts.map +1 -0
- package/dist/services/custom-model-register.js +103 -0
- package/dist/services/custom-model-register.js.map +1 -0
- package/dist/services/custom-model-seed.d.ts +18 -0
- package/dist/services/custom-model-seed.d.ts.map +1 -0
- package/dist/services/custom-model-seed.js +42 -0
- package/dist/services/custom-model-seed.js.map +1 -0
- package/dist/services/custom-model-sync.d.ts +37 -0
- package/dist/services/custom-model-sync.d.ts.map +1 -0
- package/dist/services/custom-model-sync.js +118 -0
- package/dist/services/custom-model-sync.js.map +1 -0
- package/dist/services/custom-model-tombstone.d.ts +5 -0
- package/dist/services/custom-model-tombstone.d.ts.map +1 -0
- package/dist/services/custom-model-tombstone.js +30 -0
- package/dist/services/custom-model-tombstone.js.map +1 -0
- package/dist/services/declarative-config.d.ts +356 -0
- package/dist/services/declarative-config.d.ts.map +1 -0
- package/dist/services/declarative-config.js +427 -0
- package/dist/services/declarative-config.js.map +1 -0
- package/dist/services/degradation.d.ts +40 -0
- package/dist/services/degradation.d.ts.map +1 -0
- package/dist/services/degradation.js +119 -0
- package/dist/services/degradation.js.map +1 -0
- package/dist/services/embeddings.d.ts +76 -0
- package/dist/services/embeddings.d.ts.map +1 -0
- package/dist/services/embeddings.js +319 -0
- package/dist/services/embeddings.js.map +1 -0
- package/dist/services/fusion.d.ts +162 -0
- package/dist/services/fusion.d.ts.map +1 -0
- package/dist/services/fusion.js +789 -0
- package/dist/services/fusion.js.map +1 -0
- package/dist/services/gemini-map.d.ts +14 -0
- package/dist/services/gemini-map.d.ts.map +1 -0
- package/dist/services/gemini-map.js +78 -0
- package/dist/services/gemini-map.js.map +1 -0
- package/dist/services/health.d.ts +52 -0
- package/dist/services/health.d.ts.map +1 -0
- package/dist/services/health.js +352 -0
- package/dist/services/health.js.map +1 -0
- package/dist/services/idempotency.d.ts +49 -0
- package/dist/services/idempotency.d.ts.map +1 -0
- package/dist/services/idempotency.js +140 -0
- package/dist/services/idempotency.js.map +1 -0
- package/dist/services/key-budget.d.ts +74 -0
- package/dist/services/key-budget.d.ts.map +1 -0
- package/dist/services/key-budget.js +200 -0
- package/dist/services/key-budget.js.map +1 -0
- package/dist/services/media.d.ts +123 -0
- package/dist/services/media.d.ts.map +1 -0
- package/dist/services/media.js +954 -0
- package/dist/services/media.js.map +1 -0
- package/dist/services/model-discovery.d.ts +126 -0
- package/dist/services/model-discovery.d.ts.map +1 -0
- package/dist/services/model-discovery.js +662 -0
- package/dist/services/model-discovery.js.map +1 -0
- package/dist/services/model-groups.d.ts +149 -0
- package/dist/services/model-groups.d.ts.map +1 -0
- package/dist/services/model-groups.js +314 -0
- package/dist/services/model-groups.js.map +1 -0
- package/dist/services/model-listing.d.ts +18 -0
- package/dist/services/model-listing.d.ts.map +1 -0
- package/dist/services/model-listing.js +94 -0
- package/dist/services/model-listing.js.map +1 -0
- package/dist/services/model-retirement.d.ts +29 -0
- package/dist/services/model-retirement.d.ts.map +1 -0
- package/dist/services/model-retirement.js +86 -0
- package/dist/services/model-retirement.js.map +1 -0
- package/dist/services/model-state.d.ts +90 -0
- package/dist/services/model-state.d.ts.map +1 -0
- package/dist/services/model-state.js +342 -0
- package/dist/services/model-state.js.map +1 -0
- package/dist/services/model-weight-overrides.d.ts +48 -0
- package/dist/services/model-weight-overrides.d.ts.map +1 -0
- package/dist/services/model-weight-overrides.js +127 -0
- package/dist/services/model-weight-overrides.js.map +1 -0
- package/dist/services/notification-emitter.d.ts +50 -0
- package/dist/services/notification-emitter.d.ts.map +1 -0
- package/dist/services/notification-emitter.js +177 -0
- package/dist/services/notification-emitter.js.map +1 -0
- package/dist/services/opencode-free-sync.d.ts +29 -0
- package/dist/services/opencode-free-sync.d.ts.map +1 -0
- package/dist/services/opencode-free-sync.js +206 -0
- package/dist/services/opencode-free-sync.js.map +1 -0
- package/dist/services/penalty-inspector.d.ts +56 -0
- package/dist/services/penalty-inspector.d.ts.map +1 -0
- package/dist/services/penalty-inspector.js +167 -0
- package/dist/services/penalty-inspector.js.map +1 -0
- package/dist/services/profile-models.d.ts +5 -0
- package/dist/services/profile-models.d.ts.map +1 -0
- package/dist/services/profile-models.js +62 -0
- package/dist/services/profile-models.js.map +1 -0
- package/dist/services/provider-credential.d.ts +19 -0
- package/dist/services/provider-credential.d.ts.map +1 -0
- package/dist/services/provider-credential.js +51 -0
- package/dist/services/provider-credential.js.map +1 -0
- package/dist/services/provider-quota.d.ts +64 -0
- package/dist/services/provider-quota.d.ts.map +1 -0
- package/dist/services/provider-quota.js +563 -0
- package/dist/services/provider-quota.js.map +1 -0
- package/dist/services/quirks.d.ts +0 -0
- package/dist/services/quirks.d.ts.map +1 -0
- package/dist/services/quirks.js +0 -0
- package/dist/services/quirks.js.map +1 -0
- package/dist/services/quota-forecast.d.ts +50 -0
- package/dist/services/quota-forecast.d.ts.map +1 -0
- package/dist/services/quota-forecast.js +150 -0
- package/dist/services/quota-forecast.js.map +1 -0
- package/dist/services/quota-outlook.d.ts +4 -0
- package/dist/services/quota-outlook.d.ts.map +1 -0
- package/dist/services/quota-outlook.js +92 -0
- package/dist/services/quota-outlook.js.map +1 -0
- package/dist/services/ratelimit.d.ts +249 -0
- package/dist/services/ratelimit.d.ts.map +1 -0
- package/dist/services/ratelimit.js +1182 -0
- package/dist/services/ratelimit.js.map +1 -0
- package/dist/services/request-retention.d.ts +60 -0
- package/dist/services/request-retention.d.ts.map +1 -0
- package/dist/services/request-retention.js +229 -0
- package/dist/services/request-retention.js.map +1 -0
- package/dist/services/router.d.ts +384 -0
- package/dist/services/router.d.ts.map +1 -0
- package/dist/services/router.js +1986 -0
- package/dist/services/router.js.map +1 -0
- package/dist/services/scoring.d.ts +156 -0
- package/dist/services/scoring.d.ts.map +1 -0
- package/dist/services/scoring.js +435 -0
- package/dist/services/scoring.js.map +1 -0
- package/dist/services/url-tokens.d.ts +15 -0
- package/dist/services/url-tokens.d.ts.map +1 -0
- package/dist/services/url-tokens.js +55 -0
- package/dist/services/url-tokens.js.map +1 -0
- package/package.json +77 -4
- package/README.md +0 -3
|
@@ -0,0 +1,2793 @@
|
|
|
1
|
+
import crypto from 'crypto';
|
|
2
|
+
import { Router } from 'express';
|
|
3
|
+
import { z } from 'zod';
|
|
4
|
+
import { routeRequest, resolveRoutingChain, resolveModelGroupCandidates, resolveStickyPreference, hasEnabledVisionModel, hasEnabledToolsModel, routingReserveTokens } from '../services/router.js';
|
|
5
|
+
import { secondsUntilNextMonth } from '../services/key-budget.js';
|
|
6
|
+
import { runEmbeddings, EmbeddingsError } from '../services/embeddings.js';
|
|
7
|
+
import { retryAfterSeconds } from '../lib/retry-hint.js';
|
|
8
|
+
import { runImageGeneration, runVideoGeneration, runSpeech, runTranscription, MediaError, MAX_TRANSCRIPTION_BYTES } from '../services/media.js';
|
|
9
|
+
import multer from 'multer';
|
|
10
|
+
import { getDb } from '../db/index.js';
|
|
11
|
+
import { resolveAuth, prependSystemPrompt } from '../lib/system-prompt.js';
|
|
12
|
+
import { contentToString, estimateInputTokens, messageHasImage, normalizeOutboundContent, sanitizeResponse, truncateMessagesForGithub } from '../lib/content.js';
|
|
13
|
+
import { routeOutputBudget } from '../lib/output-cap.js';
|
|
14
|
+
import { resolveTaskType } from '../lib/task-type.js';
|
|
15
|
+
import { normalizeMessageImages } from '../lib/image-normalize.js';
|
|
16
|
+
import { repairToolArguments, toolSchemaMap } from '../lib/tool-args.js';
|
|
17
|
+
import { invalidToolArgumentsError, invalidToolCallReasons, isToolArgumentValidationEnabled } from '../lib/tool-validate.js';
|
|
18
|
+
import { sanitizeProviderErrorMessage } from '../lib/error-redaction.js';
|
|
19
|
+
import { rescueInlineToolCalls, startsWithDialectMarker, couldBecomeDialectMarker, containsDialectMarker } from '../lib/tool-call-rescue.js';
|
|
20
|
+
import { getContextHandoffMode, recordIncomingMessages, maybeInjectContextHandoff, recordSuccessfulModel, hasPriorModel, HANDOFF_MAX_TOKENS } from '../services/context-handoff.js';
|
|
21
|
+
import { isFusionModel, runFusion, fusionConfigSchema, FusionError, FUSION_MODEL_ID } from '../services/fusion.js';
|
|
22
|
+
import { isRetryableError, isPaymentRequiredError, isModelNotFoundError, isModelAccessForbiddenError, isClientAbortError, newClientAbortError, newHedgeAbortError, isUpstreamClassificationOutput } from '../lib/error-classify.js';
|
|
23
|
+
import { logRequest } from '../lib/request-log.js';
|
|
24
|
+
import { observeServedModel } from '../lib/served-model.js';
|
|
25
|
+
import { parseCacheDirective, cacheActive, isCacheableTemperature, computeCacheKey, getCachedResponse, storeCachedResponse, getCachedStreamResponse, storeCachedStreamResponse, STREAM_CACHE_MAX_BYTES } from '../services/cache.js';
|
|
26
|
+
import { normalizeIdempotencyKey, hashIdempotencyKey, computeIdempotencyFingerprint, lookupIdempotencyReplay, storeIdempotencyResult } from '../services/idempotency.js';
|
|
27
|
+
import { runFallbackLoop, newFallbackState, fallbackRoutingTokens, recordUpstreamSuccess, exhaustedRetryError, setFallbackHeaders, exhaustionErrorPayload, setExhaustionHeaders } from '../lib/fallback-loop.js';
|
|
28
|
+
import { routedViaValue, safeHeaderValue } from '../lib/header-value.js';
|
|
29
|
+
import { applyTokenBudget, tokenBudgetMessage } from '../lib/guardrails.js';
|
|
30
|
+
import { samplingParamSchemaFields, pickSamplingParams, supportedParametersForPlatforms } from '../lib/sampling-params.js';
|
|
31
|
+
import { enforceJsonContent } from '../lib/structured-output.js';
|
|
32
|
+
import { inferQuotaPoolKey } from '../services/provider-quota.js';
|
|
33
|
+
import { isUnifyEnabled, getModelGroups, resolveRequestedIdForDispatch } from '../services/model-groups.js';
|
|
34
|
+
import { buildModelListing } from '../services/model-listing.js';
|
|
35
|
+
import { claudeFamilyDiscoveryEntries } from '../services/anthropic-map.js';
|
|
36
|
+
import { compressRequest, formatCompressionHeader } from '../services/compression/pipeline.js';
|
|
37
|
+
import { recordRouteTrace } from '../lib/route-live.js';
|
|
38
|
+
export const proxyRouter = Router();
|
|
39
|
+
// Virtual "auto" model. Clients like Hermes require a non-empty `model` field
|
|
40
|
+
// on every request, but srouter's whole point is to pick the model itself.
|
|
41
|
+
// Requesting this id means "let the router decide" — identical to omitting
|
|
42
|
+
// `model` entirely.
|
|
43
|
+
const AUTO_MODEL_ID = 'auto';
|
|
44
|
+
function isAutoModel(modelId) {
|
|
45
|
+
if (!modelId)
|
|
46
|
+
return true;
|
|
47
|
+
const lower = modelId.toLowerCase();
|
|
48
|
+
return lower === AUTO_MODEL_ID || lower.startsWith(`${AUTO_MODEL_ID}:`);
|
|
49
|
+
}
|
|
50
|
+
// timingSafeStringEqual moved to lib/system-prompt.ts (resolveAuth needs it
|
|
51
|
+
// and importing it back from this route would be a cycle). Re-exported here
|
|
52
|
+
// for existing importers (anthropic, gemini, mcp, ollama, status, url-tokens).
|
|
53
|
+
export { timingSafeStringEqual } from '../lib/system-prompt.js';
|
|
54
|
+
// Shared auth gate for the /v1 inference endpoints (#411): accepts the unified
|
|
55
|
+
// key (default behavior, no enforced prompt) or an enabled client-profile key
|
|
56
|
+
// (which may carry a server-enforced system prompt). Profile keys are ONLY
|
|
57
|
+
// valid here — never on the /api dashboard surface. Writes the 401 itself so
|
|
58
|
+
// call sites can simply bail on null.
|
|
59
|
+
function requireInferenceAuth(req, res) {
|
|
60
|
+
const auth = resolveAuth(extractApiToken(req));
|
|
61
|
+
if (!auth) {
|
|
62
|
+
res.status(401).json({ error: { message: 'Invalid API key', type: 'authentication_error' } });
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
return auth;
|
|
66
|
+
}
|
|
67
|
+
// Extract the unified API key from an incoming request. Accepts both the
|
|
68
|
+
// OpenAI Bearer, Anthropic x-api-key, and Gemini x-goog-api-key headers.
|
|
69
|
+
// Query credentials remain scoped to the Gemini router.
|
|
70
|
+
export function extractApiToken(req) {
|
|
71
|
+
const bearer = req.headers.authorization?.replace(/^Bearer\s+/i, '').trim();
|
|
72
|
+
if (bearer)
|
|
73
|
+
return bearer;
|
|
74
|
+
const apiKeyHeader = req.headers['x-api-key'];
|
|
75
|
+
const xApiKey = Array.isArray(apiKeyHeader) ? apiKeyHeader[0] : apiKeyHeader;
|
|
76
|
+
const trimmed = xApiKey?.trim();
|
|
77
|
+
if (trimmed)
|
|
78
|
+
return trimmed;
|
|
79
|
+
const googleHeader = req.headers['x-goog-api-key'];
|
|
80
|
+
const googleKey = Array.isArray(googleHeader) ? googleHeader[0] : googleHeader;
|
|
81
|
+
return googleKey?.trim() || undefined;
|
|
82
|
+
}
|
|
83
|
+
function quotaContextForRoute(route, endpoint) {
|
|
84
|
+
return {
|
|
85
|
+
platform: route.platform,
|
|
86
|
+
keyId: route.keyId,
|
|
87
|
+
modelId: route.modelId,
|
|
88
|
+
quotaPoolKey: inferQuotaPoolKey(route.platform, route.modelId),
|
|
89
|
+
endpoint,
|
|
90
|
+
origin: 'proxy',
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
export function getRequestGroupId(req) {
|
|
94
|
+
const raw = req.headers['x-request-id'];
|
|
95
|
+
const value = Array.isArray(raw) ? raw[0] : raw;
|
|
96
|
+
const trimmed = value?.trim();
|
|
97
|
+
return trimmed || crypto.randomUUID();
|
|
98
|
+
}
|
|
99
|
+
function shortRequestId(requestId) {
|
|
100
|
+
return requestId.replace(/-/g, '').slice(0, 6);
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* Stamp the execution id onto a body that was stored by an EARLIER request
|
|
104
|
+
* (response cache hit, idempotency replay). The id goes on the outbound copy
|
|
105
|
+
* only — never on the stored entry — so a replay reports the id of THIS
|
|
106
|
+
* request instead of the one that first filled the store, which is what makes
|
|
107
|
+
* the field safe to trust on every response. Stored bodies are typed
|
|
108
|
+
* `unknown`; a non-object entry (corrupt) is passed through untouched rather
|
|
109
|
+
* than spread into indexed characters.
|
|
110
|
+
*/
|
|
111
|
+
function withExecutionId(body, executionId) {
|
|
112
|
+
if (!body || typeof body !== 'object' || Array.isArray(body))
|
|
113
|
+
return body;
|
|
114
|
+
return { ...body, execution_id: executionId };
|
|
115
|
+
}
|
|
116
|
+
export function traceRouteEvent(scope, opts) {
|
|
117
|
+
const parts = [
|
|
118
|
+
`[${scope}]`,
|
|
119
|
+
new Date().toISOString().slice(11, 19),
|
|
120
|
+
opts.event,
|
|
121
|
+
shortRequestId(opts.requestId),
|
|
122
|
+
`a${opts.attempt}`,
|
|
123
|
+
opts.platform,
|
|
124
|
+
'-',
|
|
125
|
+
opts.model,
|
|
126
|
+
];
|
|
127
|
+
if (opts.requestedModel)
|
|
128
|
+
parts.push(`req=${opts.requestedModel}`);
|
|
129
|
+
if (opts.latencyMs != null)
|
|
130
|
+
parts.push(`lat=${opts.latencyMs}ms`);
|
|
131
|
+
if (opts.inputTokens != null)
|
|
132
|
+
parts.push(`in=${opts.inputTokens}`);
|
|
133
|
+
if (opts.outputTokens != null)
|
|
134
|
+
parts.push(`out=${opts.outputTokens}`);
|
|
135
|
+
if (opts.error)
|
|
136
|
+
parts.push(`err=${JSON.stringify(opts.error)}`);
|
|
137
|
+
// Mirror the trace into the in-memory live tracker the dashboard's routing
|
|
138
|
+
// strip reads (lib/route-live). One call here covers every Proxy/Responses
|
|
139
|
+
// start/next/ok/fail/canceled without touching the request path.
|
|
140
|
+
recordRouteTrace(opts);
|
|
141
|
+
console.log(parts.join(' '));
|
|
142
|
+
}
|
|
143
|
+
// exhaustedRetryError moved to lib/fallback-loop.ts (the shared retry loop needs
|
|
144
|
+
// it and importing it back from a route would be a cycle). Re-exported here for
|
|
145
|
+
// existing importers (routes/responses.ts, proxy-retry.test.ts historically).
|
|
146
|
+
export { exhaustedRetryError };
|
|
147
|
+
// Sticky sessions: track which model served each "session"
|
|
148
|
+
// Key: hash of first user message → model_db_id
|
|
149
|
+
// This prevents model switching mid-conversation which causes hallucination
|
|
150
|
+
const stickySessionMap = new Map();
|
|
151
|
+
const STICKY_TTL_MS = 30 * 60 * 1000; // 30 min session TTL
|
|
152
|
+
// #797: per-session memory of the last assistant turn's thinking trace.
|
|
153
|
+
// DeepSeek thinking models on OpenCode Zen 400 on a follow-up turn unless the
|
|
154
|
+
// prior `reasoning_content` is replayed; opencode (and other AI-SDK clients)
|
|
155
|
+
// strip the field when re-serializing history, so the proxy restores what it
|
|
156
|
+
// itself returned last turn. Non-thinking sessions never record an entry.
|
|
157
|
+
// The trace is stored WITH the model key that produced it: a remembered trace
|
|
158
|
+
// is only ever replayed to that same platform+model, so a session that fails
|
|
159
|
+
// over (or auto-routes elsewhere on the next turn) never carries one model's
|
|
160
|
+
// thinking into another provider's payload.
|
|
161
|
+
const reasoningMemory = new Map();
|
|
162
|
+
const REASONING_TTL_MS = 30 * 60 * 1000; // 30 min, matching sticky sessions
|
|
163
|
+
// Platforms that reject an assistant turn WITHOUT `reasoning_content` once the
|
|
164
|
+
// conversation is in thinking mode, i.e. where the field has to be present on
|
|
165
|
+
// every assistant message and older turns need an empty-string filler. Only
|
|
166
|
+
// OpenCode Zen is on record for this (the DeepSeek thinking semantics behind
|
|
167
|
+
// #255/#797); everywhere else only the turn we actually have a trace for is
|
|
168
|
+
// touched, so no other provider's bytes change.
|
|
169
|
+
const PLATFORMS_REQUIRING_REASONING_ECHO = new Set(['opencode']);
|
|
170
|
+
function rememberReasoning(sessionKey, modelKey, reasoning) {
|
|
171
|
+
if (!sessionKey || !reasoning)
|
|
172
|
+
return;
|
|
173
|
+
reasoningMemory.set(sessionKey, { reasoning, modelKey, lastUsed: Date.now() });
|
|
174
|
+
if (reasoningMemory.size > 500) {
|
|
175
|
+
const now = Date.now();
|
|
176
|
+
for (const [key, entry] of reasoningMemory) {
|
|
177
|
+
if (now - entry.lastUsed > REASONING_TTL_MS)
|
|
178
|
+
reasoningMemory.delete(key);
|
|
179
|
+
}
|
|
180
|
+
// Hard cap: traces are far bigger than a sticky entry, so an all-fresh map
|
|
181
|
+
// must not grow without bound. Evict oldest by lastUsed, as setStickyModel does.
|
|
182
|
+
if (reasoningMemory.size > 1000) {
|
|
183
|
+
const entries = [...reasoningMemory.entries()].sort((a, b) => a[1].lastUsed - b[1].lastUsed);
|
|
184
|
+
const toEvict = reasoningMemory.size - 1000;
|
|
185
|
+
for (let i = 0; i < toEvict; i++)
|
|
186
|
+
reasoningMemory.delete(entries[i][0]);
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
// The remembered trace for this session, or undefined when there is none, it
|
|
191
|
+
// expired, or it came from a different model than the one about to be called.
|
|
192
|
+
// An expired entry is dropped on read rather than left for the size sweep.
|
|
193
|
+
export function clearReasoningMemory() {
|
|
194
|
+
reasoningMemory.clear();
|
|
195
|
+
stickySessionMap.clear();
|
|
196
|
+
}
|
|
197
|
+
function rememberedReasoningFor(sessionKey, modelKey) {
|
|
198
|
+
if (!sessionKey)
|
|
199
|
+
return undefined;
|
|
200
|
+
const entry = reasoningMemory.get(sessionKey);
|
|
201
|
+
if (!entry)
|
|
202
|
+
return undefined;
|
|
203
|
+
if (Date.now() - entry.lastUsed > REASONING_TTL_MS) {
|
|
204
|
+
reasoningMemory.delete(sessionKey);
|
|
205
|
+
return undefined;
|
|
206
|
+
}
|
|
207
|
+
return entry.modelKey === modelKey ? entry.reasoning : undefined;
|
|
208
|
+
}
|
|
209
|
+
// Put the remembered trace back on the newest assistant turn that lost it.
|
|
210
|
+
// Returns a NEW array (with new objects for the messages it changes) whenever
|
|
211
|
+
// it changes anything — the caller's `messages` are what handoff recording,
|
|
212
|
+
// logging, compression and the response cache already saw, and must not move
|
|
213
|
+
// under them. Returns the input untouched when there is nothing to restore.
|
|
214
|
+
function restoreSessionReasoning(messages, reasoning, platform) {
|
|
215
|
+
// Older assistant turns get "" — DeepSeek requires the field on every
|
|
216
|
+
// assistant message, and an empty string satisfies it (see opencode issue
|
|
217
|
+
// #24104). Only for platforms that actually enforce that; elsewhere just the
|
|
218
|
+
// one turn we have a real trace for is touched.
|
|
219
|
+
const fillOlderTurns = PLATFORMS_REQUIRING_REASONING_ECHO.has(platform);
|
|
220
|
+
let restored;
|
|
221
|
+
let restoredLatest = false;
|
|
222
|
+
for (let i = messages.length - 1; i >= 0; i -= 1) {
|
|
223
|
+
const m = messages[i];
|
|
224
|
+
if (m.role !== 'assistant')
|
|
225
|
+
continue;
|
|
226
|
+
// The client kept the field — nothing was dropped, leave it alone.
|
|
227
|
+
if (typeof m.reasoning_content === 'string' && m.reasoning_content.length > 0)
|
|
228
|
+
continue;
|
|
229
|
+
restored ??= [...messages];
|
|
230
|
+
restored[i] = { ...m, reasoning_content: restoredLatest ? '' : reasoning };
|
|
231
|
+
if (!restoredLatest && !fillOlderTurns)
|
|
232
|
+
break;
|
|
233
|
+
restoredLatest = true;
|
|
234
|
+
}
|
|
235
|
+
return restored ?? messages;
|
|
236
|
+
}
|
|
237
|
+
function getSessionKey(messages, sessionIdHeader, strategyKey) {
|
|
238
|
+
if (sessionIdHeader) {
|
|
239
|
+
return strategyKey ? `hdr:${sessionIdHeader}::${strategyKey}` : `hdr:${sessionIdHeader}`;
|
|
240
|
+
}
|
|
241
|
+
const firstUser = messages.find(m => m.role === 'user');
|
|
242
|
+
if (!firstUser)
|
|
243
|
+
return '';
|
|
244
|
+
const text = contentToString(firstUser.content ?? '');
|
|
245
|
+
if (!text)
|
|
246
|
+
return '';
|
|
247
|
+
const payload = strategyKey ? `${text}::${strategyKey}` : text;
|
|
248
|
+
return crypto.createHash('sha1').update(payload).digest('hex');
|
|
249
|
+
}
|
|
250
|
+
export function getStickyModel(messages, sessionIdHeader, strategyKey) {
|
|
251
|
+
const hasAssistant = messages.some(m => m.role === 'assistant');
|
|
252
|
+
if (!hasAssistant)
|
|
253
|
+
return undefined;
|
|
254
|
+
const key = getSessionKey(messages, sessionIdHeader, strategyKey);
|
|
255
|
+
if (!key)
|
|
256
|
+
return undefined;
|
|
257
|
+
const entry = stickySessionMap.get(key);
|
|
258
|
+
if (!entry)
|
|
259
|
+
return undefined;
|
|
260
|
+
if (Date.now() - entry.lastUsed > STICKY_TTL_MS) {
|
|
261
|
+
stickySessionMap.delete(key);
|
|
262
|
+
return undefined;
|
|
263
|
+
}
|
|
264
|
+
return entry.modelDbId;
|
|
265
|
+
}
|
|
266
|
+
export function setStickyModel(messages, modelDbId, sessionIdHeader, strategyKey) {
|
|
267
|
+
const key = getSessionKey(messages, sessionIdHeader, strategyKey);
|
|
268
|
+
if (!key)
|
|
269
|
+
return;
|
|
270
|
+
stickySessionMap.set(key, { modelDbId, lastUsed: Date.now() });
|
|
271
|
+
// Cleanup old entries
|
|
272
|
+
if (stickySessionMap.size > 500) {
|
|
273
|
+
const now = Date.now();
|
|
274
|
+
for (const [k, v] of stickySessionMap) {
|
|
275
|
+
if (now - v.lastUsed > STICKY_TTL_MS)
|
|
276
|
+
stickySessionMap.delete(k);
|
|
277
|
+
}
|
|
278
|
+
// Hard cap: if still over 1000 after pruning expired entries, evict oldest by lastUsed
|
|
279
|
+
if (stickySessionMap.size > 1000) {
|
|
280
|
+
const entries = [...stickySessionMap.entries()].sort((a, b) => a[1].lastUsed - b[1].lastUsed);
|
|
281
|
+
const toEvict = stickySessionMap.size - 1000;
|
|
282
|
+
for (let i = 0; i < toEvict; i++) {
|
|
283
|
+
stickySessionMap.delete(entries[i][0]);
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
// OpenAI-compatible /models endpoint (used by Hermes for metadata)
|
|
289
|
+
// shows API models which is linked by the user
|
|
290
|
+
proxyRouter.get('/models', (req, res) => {
|
|
291
|
+
if (!requireInferenceAuth(req, res))
|
|
292
|
+
return;
|
|
293
|
+
// By default we return the WHOLE catalog (one row per model id), each tagged
|
|
294
|
+
// with whether it is currently usable, so a client can see everything and know
|
|
295
|
+
// what's connected vs. disabled/keyless (#242). `?available=true` (aliases
|
|
296
|
+
// `?connected=true`, `?ready=true`) narrows the list to only models that can
|
|
297
|
+
// serve a request right now — the previous default behavior. The `ready`
|
|
298
|
+
// alias is the machine-readable filter a meta-gateway uses (#433) to ask
|
|
299
|
+
// "which models can this instance actually serve now". `available` is computed as
|
|
300
|
+
// "enabled AND an enabled key can serve it"; dedup prefers an available
|
|
301
|
+
// instance of a model id over a disabled/keyless one.
|
|
302
|
+
// Shared catalog listing (one source of truth for the OpenAI and Anthropic
|
|
303
|
+
// /v1/models endpoints — see services/model-listing.ts). `autoContextWindow`
|
|
304
|
+
// is the honest ceiling for the virtual "auto" model: the largest context
|
|
305
|
+
// window among models that can serve a request right now. Advertising null
|
|
306
|
+
// makes OpenAI-compatible clients (opencode, Continue) fall back to their own
|
|
307
|
+
// conservative default and truncate long inputs before they reach us (#282).
|
|
308
|
+
const { models: allListed, autoContextWindow } = buildModelListing();
|
|
309
|
+
const q = String(req.query.available ?? req.query.connected ?? req.query.ready ?? '').toLowerCase();
|
|
310
|
+
const onlyAvailable = q === '1' || q === 'true' || q === 'yes';
|
|
311
|
+
const listed = onlyAvailable ? allListed.filter(m => m.available === 1) : allListed;
|
|
312
|
+
// Named fallback chains (#960/#895): every user-defined profile is exposed
|
|
313
|
+
// as an `auto:<name>` model so a client can pick a specific fallback chain
|
|
314
|
+
// per request (auto:my-group) instead of only the active one. Available iff
|
|
315
|
+
// at least one model in that profile's chain can serve a request right now.
|
|
316
|
+
const profileRows = getDb().prepare(`
|
|
317
|
+
SELECT p.id, p.name,
|
|
318
|
+
EXISTS (
|
|
319
|
+
SELECT 1
|
|
320
|
+
FROM profile_models pm
|
|
321
|
+
JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
|
|
322
|
+
WHERE pm.profile_id = p.id AND pm.enabled = 1
|
|
323
|
+
AND EXISTS (
|
|
324
|
+
SELECT 1 FROM api_keys k
|
|
325
|
+
WHERE k.platform = m.platform AND k.enabled = 1
|
|
326
|
+
AND (m.key_id IS NULL OR k.id = m.key_id)
|
|
327
|
+
)
|
|
328
|
+
) AS usable,
|
|
329
|
+
(SELECT MAX(m2.context_window)
|
|
330
|
+
FROM profile_models pm2
|
|
331
|
+
JOIN models m2 ON m2.id = pm2.model_db_id AND m2.enabled = 1
|
|
332
|
+
WHERE pm2.profile_id = p.id AND pm2.enabled = 1) AS max_ctx
|
|
333
|
+
FROM profiles p
|
|
334
|
+
WHERE p.type = 'custom'
|
|
335
|
+
ORDER BY p.sort_order, p.id
|
|
336
|
+
`).all();
|
|
337
|
+
// Claude-family discovery entries (#880). The Anthropic-shaped GET /v1/models
|
|
338
|
+
// in routes/anthropic.ts already lists one id per Claude family so clients
|
|
339
|
+
// that only accept Claude-looking ids can discover anything at all — but that
|
|
340
|
+
// handler only answers when the caller sends an `anthropic-version` header.
|
|
341
|
+
// Claude Desktop's gateway picker fetches this path WITHOUT that header, so
|
|
342
|
+
// it fell through to the OpenAI-shaped listing below and still saw zero
|
|
343
|
+
// Claude-shaped ids. Emit the same entries here, from the same builder, so
|
|
344
|
+
// both shapes agree on what the gateway will serve. Listed only when
|
|
345
|
+
// something can actually serve them, and never when the id would collide
|
|
346
|
+
// with a real catalog row.
|
|
347
|
+
const listedIds = new Set(listed.map(m => m.id));
|
|
348
|
+
const claudeFamilyEntries = allListed.some(m => m.available === 1)
|
|
349
|
+
? claudeFamilyDiscoveryEntries()
|
|
350
|
+
.filter(a => !listedIds.has(a.id))
|
|
351
|
+
.map(a => ({
|
|
352
|
+
id: a.id,
|
|
353
|
+
object: 'model',
|
|
354
|
+
created: 0,
|
|
355
|
+
owned_by: 'srouter',
|
|
356
|
+
name: a.displayName,
|
|
357
|
+
context_window: autoContextWindow,
|
|
358
|
+
context_length: autoContextWindow,
|
|
359
|
+
available: true,
|
|
360
|
+
unavailable_reason: null,
|
|
361
|
+
}))
|
|
362
|
+
: [];
|
|
363
|
+
// Machine-readable execution filter: `?execution_status=ready` narrows to
|
|
364
|
+
// models a request can actually serve RIGHT NOW (a key that is not scoped
|
|
365
|
+
// away, cooling down or out of window). `ready` implies available;
|
|
366
|
+
// `needsKey`/`exhausted` select the rest. Matched case-insensitively because
|
|
367
|
+
// the value itself is camelCase; an unrecognised value filters nothing, the
|
|
368
|
+
// same way an unrecognised `?available=` does. Like `?available=`, this
|
|
369
|
+
// narrows only the catalog rows: `auto`, `fusion` and the named chains are
|
|
370
|
+
// router entries, not models with keys of their own.
|
|
371
|
+
const esValues = {
|
|
372
|
+
ready: 'ready', needskey: 'needsKey', exhausted: 'exhausted',
|
|
373
|
+
};
|
|
374
|
+
const es = esValues[String(req.query.execution_status ?? '').toLowerCase()];
|
|
375
|
+
const esFiltered = es ? listed.filter(m => m.executionStatus === es) : listed;
|
|
376
|
+
res.json({
|
|
377
|
+
object: 'list',
|
|
378
|
+
data: [
|
|
379
|
+
{
|
|
380
|
+
id: AUTO_MODEL_ID,
|
|
381
|
+
object: 'model',
|
|
382
|
+
created: 0,
|
|
383
|
+
owned_by: 'srouter',
|
|
384
|
+
name: 'Auto (router picks the best available model)',
|
|
385
|
+
context_window: autoContextWindow,
|
|
386
|
+
// `context_length` is OpenRouter's field name and the one most
|
|
387
|
+
// OpenAI-compatible clients read; emit both so whichever a client
|
|
388
|
+
// looks for is populated. Additive — clients ignore unknown fields.
|
|
389
|
+
context_length: autoContextWindow,
|
|
390
|
+
available: true,
|
|
391
|
+
unavailable_reason: null,
|
|
392
|
+
},
|
|
393
|
+
{
|
|
394
|
+
id: FUSION_MODEL_ID,
|
|
395
|
+
object: 'model',
|
|
396
|
+
created: 0,
|
|
397
|
+
owned_by: 'srouter',
|
|
398
|
+
name: 'Fusion (panel of models answer in parallel, a judge synthesizes one answer)',
|
|
399
|
+
context_window: autoContextWindow,
|
|
400
|
+
context_length: autoContextWindow,
|
|
401
|
+
// Available whenever auto is — fusion needs at least one routable model.
|
|
402
|
+
available: autoContextWindow != null,
|
|
403
|
+
unavailable_reason: autoContextWindow != null ? null : 'no_models',
|
|
404
|
+
},
|
|
405
|
+
...claudeFamilyEntries,
|
|
406
|
+
...profileRows.map(p => ({
|
|
407
|
+
id: `auto:${p.name.toLowerCase()}`,
|
|
408
|
+
object: 'model',
|
|
409
|
+
created: 0,
|
|
410
|
+
owned_by: 'srouter',
|
|
411
|
+
name: `Auto: ${p.name} (named fallback chain)`,
|
|
412
|
+
context_window: p.max_ctx,
|
|
413
|
+
context_length: p.max_ctx,
|
|
414
|
+
available: p.usable === 1,
|
|
415
|
+
unavailable_reason: p.usable === 1 ? null : 'no_models',
|
|
416
|
+
})),
|
|
417
|
+
...esFiltered.map(m => ({
|
|
418
|
+
id: m.id,
|
|
419
|
+
object: 'model',
|
|
420
|
+
created: 0,
|
|
421
|
+
owned_by: m.ownedBy,
|
|
422
|
+
name: m.name,
|
|
423
|
+
context_window: m.contextWindow,
|
|
424
|
+
context_length: m.contextWindow,
|
|
425
|
+
// Non-standard but additive: OpenAI clients ignore unknown fields.
|
|
426
|
+
available: m.available === 1,
|
|
427
|
+
unavailable_reason: m.available === 1 ? null : (m.enabled === 1 ? 'no_key' : 'disabled'),
|
|
428
|
+
// Machine-readable dynamic status: 'ready' | 'needsKey' | 'exhausted'
|
|
429
|
+
// (see services/model-listing.ts). Agents can filter with
|
|
430
|
+
// ?execution_status=ready and route around exhausted models.
|
|
431
|
+
execution_status: m.executionStatus,
|
|
432
|
+
// OpenRouter's field name; agents use it to pick knobs per model. For
|
|
433
|
+
// a unify group this is the intersection over member platforms — a
|
|
434
|
+
// param is only advertised when every platform the router might pick
|
|
435
|
+
// honors it.
|
|
436
|
+
supported_parameters: supportedParametersForPlatforms(m.platforms, { tools: m.supportsTools }),
|
|
437
|
+
})),
|
|
438
|
+
],
|
|
439
|
+
});
|
|
440
|
+
});
|
|
441
|
+
const MAX_RETRIES = 20;
|
|
442
|
+
// Echo-tolerant tool calls: agents replay OUR responses back as history, and
|
|
443
|
+
// not all of them preserve the strict OpenAI shape. `type` may be dropped
|
|
444
|
+
// (re-added on forward), Gemini-lineage agents (Qwen Code, AionUI) often
|
|
445
|
+
// send `arguments` as a parsed object instead of a JSON string, and `id` may
|
|
446
|
+
// be missing or empty (ids aren't a Gemini concept) — all get normalized
|
|
447
|
+
// below rather than 400-ing the whole session. Missing ids are synthesized
|
|
448
|
+
// and paired with their tool-result messages by order. (#200)
|
|
449
|
+
const toolCallSchema = z.object({
|
|
450
|
+
id: z.string().optional(),
|
|
451
|
+
type: z.literal('function').optional(),
|
|
452
|
+
function: z.object({
|
|
453
|
+
name: z.string().min(1),
|
|
454
|
+
arguments: z.union([z.string(), z.record(z.string(), z.unknown())]),
|
|
455
|
+
}),
|
|
456
|
+
thought_signature: z.string().optional(),
|
|
457
|
+
});
|
|
458
|
+
const toolCallArgsToString = (args) => typeof args === 'string' ? args : JSON.stringify(args);
|
|
459
|
+
// OpenAI multimodal envelope. Clients like opencode / continue.dev send
|
|
460
|
+
// content as an array of typed blocks even when only text is present, and
|
|
461
|
+
// Gemini-lineage agents send part-style blocks like `{ "text": "..." }` with
|
|
462
|
+
// no `type` at all. Accept any object (or bare string) as a block; flatten to
|
|
463
|
+
// string for providers that don't support arrays (Cohere, Cloudflare).
|
|
464
|
+
// Non-text blocks pass z validation but get dropped by contentToString —
|
|
465
|
+
// vision/audio still isn't supported. (#200)
|
|
466
|
+
const contentBlockSchema = z.union([z.string(), z.record(z.string(), z.unknown())]);
|
|
467
|
+
const contentSchema = z.union([z.string(), z.array(contentBlockSchema)]);
|
|
468
|
+
const systemMessageSchema = z.object({
|
|
469
|
+
role: z.literal('system'),
|
|
470
|
+
content: contentSchema,
|
|
471
|
+
name: z.string().optional(),
|
|
472
|
+
});
|
|
473
|
+
// OpenAI's newer SDKs send the system prompt as role:"developer"; accept it
|
|
474
|
+
// and forward as "system" — none of the routed providers know the developer
|
|
475
|
+
// role. (#200)
|
|
476
|
+
const developerMessageSchema = z.object({
|
|
477
|
+
role: z.literal('developer'),
|
|
478
|
+
content: contentSchema,
|
|
479
|
+
name: z.string().optional(),
|
|
480
|
+
});
|
|
481
|
+
const userMessageSchema = z.object({
|
|
482
|
+
role: z.literal('user'),
|
|
483
|
+
content: contentSchema,
|
|
484
|
+
name: z.string().optional(),
|
|
485
|
+
});
|
|
486
|
+
// Assistant turns may carry empty/null content and no tool_calls — OpenAI
|
|
487
|
+
// accepts these in conversation history (a turn that produced no visible text,
|
|
488
|
+
// a placeholder, a tool turn whose content was emptied), and clients replay
|
|
489
|
+
// them verbatim. We accept them too and coerce empty/null content to "" before
|
|
490
|
+
// forwarding (see message build below) rather than 400-ing a payload OpenAI
|
|
491
|
+
// would take. (#165)
|
|
492
|
+
const assistantMessageSchema = z.object({
|
|
493
|
+
role: z.literal('assistant'),
|
|
494
|
+
content: z.union([contentSchema, z.null()]).optional(),
|
|
495
|
+
name: z.string().optional(),
|
|
496
|
+
// tool_calls: null (not just missing) is what several agents replay for
|
|
497
|
+
// no-tool assistant turns — aionrs (AionUI's engine) writes it into every
|
|
498
|
+
// session-resumed assistant echo. Treated as absent. (#200)
|
|
499
|
+
tool_calls: z.array(toolCallSchema).nullable().optional(),
|
|
500
|
+
// Thinking trace echoed back by a client. DeepSeek thinking models on
|
|
501
|
+
// OpenCode Zen 400 ("reasoning_content in thinking mode must be passed back")
|
|
502
|
+
// unless the prior turn's reasoning_content is replayed, so keep it through
|
|
503
|
+
// validation instead of stripping it. See issue #255.
|
|
504
|
+
reasoning_content: z.string().nullable().optional(),
|
|
505
|
+
// Moonshot's "partial" prefill flag. A plain z.object (no .passthrough())
|
|
506
|
+
// would silently strip it; keep it through validation so it can be forwarded
|
|
507
|
+
// to Moonshot/Kimi models, which document it. See issue #1038.
|
|
508
|
+
partial: z.boolean().optional(),
|
|
509
|
+
});
|
|
510
|
+
// Tool results may arrive with null/missing content (a tool that returned
|
|
511
|
+
// nothing) and a missing/empty tool_call_id (Gemini-lineage agents) — coerced
|
|
512
|
+
// to "" and paired by order with the preceding tool_calls respectively. (#200)
|
|
513
|
+
const toolMessageSchema = z.object({
|
|
514
|
+
role: z.literal('tool'),
|
|
515
|
+
content: z.union([contentSchema, z.null()]).optional(),
|
|
516
|
+
tool_call_id: z.string().optional(),
|
|
517
|
+
name: z.string().optional(),
|
|
518
|
+
});
|
|
519
|
+
// Legacy function-calling shape (pre-tools OpenAI API). Old clients still
|
|
520
|
+
// replay these in history; forwarded as a tool message. (#200)
|
|
521
|
+
const functionMessageSchema = z.object({
|
|
522
|
+
role: z.literal('function'),
|
|
523
|
+
name: z.string().min(1),
|
|
524
|
+
content: z.union([contentSchema, z.null()]).optional(),
|
|
525
|
+
});
|
|
526
|
+
const toolDefinitionSchema = z.object({
|
|
527
|
+
// Some agents omit `type` on tool definitions; re-defaulted to 'function'
|
|
528
|
+
// on forward. (#200)
|
|
529
|
+
type: z.literal('function').optional(),
|
|
530
|
+
function: z.object({
|
|
531
|
+
name: z.string().min(1),
|
|
532
|
+
description: z.string().optional(),
|
|
533
|
+
parameters: z.record(z.string(), z.unknown()).optional(),
|
|
534
|
+
strict: z.boolean().optional(),
|
|
535
|
+
}),
|
|
536
|
+
});
|
|
537
|
+
const toolChoiceSchema = z.union([
|
|
538
|
+
// 'any' is the Mistral/Gemini wording for OpenAI's 'required'; mapped on
|
|
539
|
+
// forward. (#200)
|
|
540
|
+
z.enum(['none', 'auto', 'required', 'any']),
|
|
541
|
+
z.object({
|
|
542
|
+
type: z.literal('function'),
|
|
543
|
+
function: z.object({
|
|
544
|
+
name: z.string().min(1),
|
|
545
|
+
}),
|
|
546
|
+
}),
|
|
547
|
+
]);
|
|
548
|
+
const stopSchema = z.union([z.string(), z.array(z.string()).min(1).max(64)]);
|
|
549
|
+
function providerSafeStop(stop) {
|
|
550
|
+
if (!Array.isArray(stop))
|
|
551
|
+
return stop;
|
|
552
|
+
return stop.slice(0, 4);
|
|
553
|
+
}
|
|
554
|
+
const chatCompletionSchema = z.object({
|
|
555
|
+
messages: z.array(z.union([
|
|
556
|
+
systemMessageSchema,
|
|
557
|
+
developerMessageSchema,
|
|
558
|
+
userMessageSchema,
|
|
559
|
+
assistantMessageSchema,
|
|
560
|
+
toolMessageSchema,
|
|
561
|
+
functionMessageSchema,
|
|
562
|
+
])).min(1),
|
|
563
|
+
model: z.string().optional(),
|
|
564
|
+
temperature: z.number().min(0).max(2).optional(),
|
|
565
|
+
// Some clients send max_tokens <= 0 (or -1) to mean "no limit"; accepted and
|
|
566
|
+
// treated as unset on forward. (#200)
|
|
567
|
+
max_tokens: z.number().int().optional(),
|
|
568
|
+
top_p: z.number().min(0).max(1).optional(),
|
|
569
|
+
stop: stopSchema.optional(),
|
|
570
|
+
stream: z.boolean().optional(),
|
|
571
|
+
stream_options: z.object({
|
|
572
|
+
include_usage: z.boolean().optional(),
|
|
573
|
+
}).optional(),
|
|
574
|
+
// Top-level tool knobs may arrive as explicit nulls from clients that
|
|
575
|
+
// serialize every field of their request struct; all treated as absent
|
|
576
|
+
// and never forwarded as null. (#200)
|
|
577
|
+
tools: z.array(toolDefinitionSchema).nullable().optional(),
|
|
578
|
+
tool_choice: toolChoiceSchema.nullable().optional(),
|
|
579
|
+
parallel_tool_calls: z.boolean().nullable().optional(),
|
|
580
|
+
// Fusion config — only meaningful when `model` is the virtual "fusion" id.
|
|
581
|
+
// Ignored for every other model. See services/fusion.ts.
|
|
582
|
+
fusion: fusionConfigSchema.optional(),
|
|
583
|
+
// Extended sampling + structured-output params (top_k, seed, penalties,
|
|
584
|
+
// logit_bias, logprobs, response_format, max_completion_tokens…), forwarded
|
|
585
|
+
// per the platform policy in lib/sampling-params.ts.
|
|
586
|
+
...samplingParamSchemaFields,
|
|
587
|
+
});
|
|
588
|
+
// Upstream-error classifiers live in lib/error-classify.ts so the fusion
|
|
589
|
+
// service can share them without an import cycle; imported above for internal
|
|
590
|
+
// use and re-exported here for existing importers (routes/responses.ts,
|
|
591
|
+
// proxy-retry.test.ts) that pull them from this module.
|
|
592
|
+
export { isRetryableError, isPaymentRequiredError, isModelNotFoundError, isModelAccessForbiddenError };
|
|
593
|
+
// Pull the incremental text out of a streaming chunk for token counting.
|
|
594
|
+
// Must tolerate chunks that carry no `choices` array at all: some providers
|
|
595
|
+
// (e.g. Groq) emit usage/keepalive frames shaped like `{usage:{...}}` with no
|
|
596
|
+
// `choices`. Indexing `chunk.choices[0]` on those throws "Cannot read
|
|
597
|
+
// properties of undefined (reading '0')", which — once the SSE stream has
|
|
598
|
+
// started — aborts the response mid-flight with no chance to fall back.
|
|
599
|
+
export function streamChunkText(chunk) {
|
|
600
|
+
return chunk?.choices?.[0]?.delta?.content ?? '';
|
|
601
|
+
}
|
|
602
|
+
// Pull the incremental reasoning text out of a streaming chunk. Reasoning
|
|
603
|
+
// models stream thinking via `reasoning_content` (Z.ai, DeepSeek-style — the
|
|
604
|
+
// <think> extractor in base.ts normalizes inline tags into the same field) or
|
|
605
|
+
// `reasoning` (Ollama-style) before the first visible answer token; both
|
|
606
|
+
// spellings must count for ttfb and output-token estimates. Same shape
|
|
607
|
+
// tolerance as streamChunkText. (#764)
|
|
608
|
+
export function streamReasoningText(chunk) {
|
|
609
|
+
const delta = chunk?.choices?.[0]?.delta;
|
|
610
|
+
const r = delta?.reasoning_content ?? delta?.reasoning;
|
|
611
|
+
return typeof r === 'string' ? r : '';
|
|
612
|
+
}
|
|
613
|
+
// OpenAI-compatible embeddings endpoint, routed through the embeddings family
|
|
614
|
+
// catalog: `model: "auto"` (or omitted) → the configured default family; a
|
|
615
|
+
// family name or provider model id → that family's provider chain. Failover
|
|
616
|
+
// only happens WITHIN a family (same model on another provider) — never across
|
|
617
|
+
// models, since vectors from different models are incompatible.
|
|
618
|
+
const EmbeddingsBody = z.object({
|
|
619
|
+
model: z.string().optional(),
|
|
620
|
+
input: z.union([z.string(), z.array(z.string())]),
|
|
621
|
+
// Optional output-dimension override forwarded to providers that support MRL
|
|
622
|
+
// truncation (NVIDIA NeMo NIM, Google Gemini Embedding, OpenAI v3). Validation
|
|
623
|
+
// only — bounds checking happens upstream (the provider rejects out-of-range
|
|
624
|
+
// values with a clear 400).
|
|
625
|
+
dimensions: z.number().int().positive().optional(),
|
|
626
|
+
});
|
|
627
|
+
proxyRouter.post('/embeddings', async (req, res) => {
|
|
628
|
+
if (!requireInferenceAuth(req, res))
|
|
629
|
+
return;
|
|
630
|
+
const parsed = EmbeddingsBody.safeParse(req.body);
|
|
631
|
+
if (!parsed.success) {
|
|
632
|
+
res.status(400).json({ error: { message: 'Invalid request: `input` is required', type: 'invalid_request_error' } });
|
|
633
|
+
return;
|
|
634
|
+
}
|
|
635
|
+
const inputs = Array.isArray(parsed.data.input) ? parsed.data.input : [parsed.data.input];
|
|
636
|
+
try {
|
|
637
|
+
const result = await runEmbeddings(parsed.data.model, inputs, parsed.data.dimensions);
|
|
638
|
+
res.json({
|
|
639
|
+
object: 'list',
|
|
640
|
+
data: result.vectors.map((values, i) => ({ object: 'embedding', index: i, embedding: values })),
|
|
641
|
+
model: result.family,
|
|
642
|
+
provider: result.platform,
|
|
643
|
+
usage: { prompt_tokens: result.inputTokens, total_tokens: result.inputTokens },
|
|
644
|
+
});
|
|
645
|
+
}
|
|
646
|
+
catch (err) {
|
|
647
|
+
const status = err instanceof EmbeddingsError ? err.status : 502;
|
|
648
|
+
const code = err instanceof EmbeddingsError ? inferenceBudgetCode(err, res) : {};
|
|
649
|
+
const type = status === 400 ? 'invalid_request_error' : status === 429 ? 'rate_limit_error' : 'server_error';
|
|
650
|
+
res.status(status).json({ error: { message: `embedding error: ${err?.message ?? 'unknown'}`, type, ...code } });
|
|
651
|
+
}
|
|
652
|
+
});
|
|
653
|
+
// OpenAI-compatible image generation. Routed through the media catalog (its own
|
|
654
|
+
// table, never the chat router): `model: "auto"` (or omitted) tries every enabled
|
|
655
|
+
// image provider in order; a provider model id pins to that one. Failover is
|
|
656
|
+
// across providers, never across modalities. See services/media.ts.
|
|
657
|
+
const ImageBody = z.object({
|
|
658
|
+
model: z.string().optional(),
|
|
659
|
+
prompt: z.string().min(1),
|
|
660
|
+
n: z.number().int().positive().max(4).optional(),
|
|
661
|
+
size: z.string().optional(),
|
|
662
|
+
response_format: z.enum(['url', 'b64_json']).optional(),
|
|
663
|
+
});
|
|
664
|
+
function inferenceBudgetCode(error, res) {
|
|
665
|
+
// retryAfterMs is set by the embeddings/media services only when the whole
|
|
666
|
+
// chain was rate limited (soonest stated back-off, budget resets included),
|
|
667
|
+
// so it wins over the month-long budget reset when a sibling returns sooner.
|
|
668
|
+
if (error.retryAfterMs !== undefined)
|
|
669
|
+
res.setHeader('Retry-After', retryAfterSeconds(error.retryAfterMs));
|
|
670
|
+
else if (error.code === 'quota_exceeded')
|
|
671
|
+
res.setHeader('Retry-After', secondsUntilNextMonth());
|
|
672
|
+
return error.code ? { code: error.code } : {};
|
|
673
|
+
}
|
|
674
|
+
function mediaErrorType(status) {
|
|
675
|
+
if (status === 400 || status === 413)
|
|
676
|
+
return 'invalid_request_error';
|
|
677
|
+
if (status === 401)
|
|
678
|
+
return 'authentication_error';
|
|
679
|
+
if (status === 429)
|
|
680
|
+
return 'rate_limit_error';
|
|
681
|
+
return 'server_error';
|
|
682
|
+
}
|
|
683
|
+
proxyRouter.post('/images/generations', async (req, res) => {
|
|
684
|
+
if (!requireInferenceAuth(req, res))
|
|
685
|
+
return;
|
|
686
|
+
const parsed = ImageBody.safeParse(req.body);
|
|
687
|
+
if (!parsed.success) {
|
|
688
|
+
res.status(400).json({ error: { message: 'Invalid request: `prompt` is required', type: 'invalid_request_error' } });
|
|
689
|
+
return;
|
|
690
|
+
}
|
|
691
|
+
try {
|
|
692
|
+
const result = await runImageGeneration(parsed.data.model, {
|
|
693
|
+
prompt: parsed.data.prompt, n: parsed.data.n, size: parsed.data.size,
|
|
694
|
+
});
|
|
695
|
+
res.json({
|
|
696
|
+
created: Math.floor(Date.now() / 1000),
|
|
697
|
+
data: result.images,
|
|
698
|
+
model: result.modelId,
|
|
699
|
+
provider: result.platform,
|
|
700
|
+
});
|
|
701
|
+
}
|
|
702
|
+
catch (err) {
|
|
703
|
+
const status = err instanceof MediaError ? err.status : 502;
|
|
704
|
+
const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
|
|
705
|
+
const httpStatus = status >= 400 && status < 600 ? status : 502;
|
|
706
|
+
res.status(httpStatus).json({ error: { message: `image generation error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
|
|
707
|
+
}
|
|
708
|
+
});
|
|
709
|
+
// Text-to-video generation. Providers may use a synchronous binary response
|
|
710
|
+
// (Pollinations) or an asynchronous queue internally (Hugging Face/fal.ai), but
|
|
711
|
+
// this gateway presents one bounded request and returns the completed MP4.
|
|
712
|
+
const VideoBody = z.object({
|
|
713
|
+
model: z.string().optional(),
|
|
714
|
+
prompt: z.string().min(1),
|
|
715
|
+
duration: z.number().int().min(1).max(120).optional(),
|
|
716
|
+
aspect_ratio: z.enum(['16:9', '9:16']).optional(),
|
|
717
|
+
image: z.string().url().optional(),
|
|
718
|
+
seed: z.number().int().min(-1).max(2_147_483_647).optional(),
|
|
719
|
+
audio: z.boolean().optional(),
|
|
720
|
+
});
|
|
721
|
+
proxyRouter.post('/videos/generations', async (req, res) => {
|
|
722
|
+
if (!requireInferenceAuth(req, res))
|
|
723
|
+
return;
|
|
724
|
+
const parsed = VideoBody.safeParse(req.body);
|
|
725
|
+
if (!parsed.success) {
|
|
726
|
+
res.status(400).json({
|
|
727
|
+
error: {
|
|
728
|
+
message: 'Invalid request: `prompt` is required and video options must use supported values',
|
|
729
|
+
type: 'invalid_request_error',
|
|
730
|
+
},
|
|
731
|
+
});
|
|
732
|
+
return;
|
|
733
|
+
}
|
|
734
|
+
// A video job runs for minutes, so a caller that hangs up must actually stop
|
|
735
|
+
// the work: without this the gateway would keep polling the provider and then
|
|
736
|
+
// fail over to a second one, both charged to the operator, for a response
|
|
737
|
+
// nobody is waiting for. 'close' also fires on normal completion, which
|
|
738
|
+
// writableEnded distinguishes.
|
|
739
|
+
const clientAbort = new AbortController();
|
|
740
|
+
res.on('close', () => {
|
|
741
|
+
if (!res.writableEnded)
|
|
742
|
+
clientAbort.abort();
|
|
743
|
+
});
|
|
744
|
+
try {
|
|
745
|
+
const result = await runVideoGeneration(parsed.data.model, {
|
|
746
|
+
prompt: parsed.data.prompt,
|
|
747
|
+
duration: parsed.data.duration,
|
|
748
|
+
aspectRatio: parsed.data.aspect_ratio,
|
|
749
|
+
image: parsed.data.image,
|
|
750
|
+
seed: parsed.data.seed,
|
|
751
|
+
audio: parsed.data.audio,
|
|
752
|
+
}, clientAbort.signal);
|
|
753
|
+
res.setHeader('Content-Type', result.contentType);
|
|
754
|
+
res.setHeader('Cache-Control', 'no-store');
|
|
755
|
+
res.setHeader('X-Provider', safeHeaderValue(result.platform));
|
|
756
|
+
res.setHeader('X-Model', safeHeaderValue(result.modelId));
|
|
757
|
+
res.send(result.video);
|
|
758
|
+
}
|
|
759
|
+
catch (err) {
|
|
760
|
+
// Nothing to report to a socket that is already gone.
|
|
761
|
+
if (clientAbort.signal.aborted || res.writableEnded)
|
|
762
|
+
return;
|
|
763
|
+
const status = err instanceof MediaError ? err.status : 502;
|
|
764
|
+
const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
|
|
765
|
+
const httpStatus = status >= 400 && status < 600 ? status : 502;
|
|
766
|
+
res.status(httpStatus).json({
|
|
767
|
+
error: { message: `video generation error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code },
|
|
768
|
+
});
|
|
769
|
+
}
|
|
770
|
+
});
|
|
771
|
+
// OpenAI-compatible text-to-speech. Returns raw audio bytes (OpenAI's /audio/speech
|
|
772
|
+
// shape). Same media-catalog routing as images.
|
|
773
|
+
const SpeechBody = z.object({
|
|
774
|
+
model: z.string().optional(),
|
|
775
|
+
input: z.string().min(1),
|
|
776
|
+
voice: z.string().optional(),
|
|
777
|
+
response_format: z.string().optional(),
|
|
778
|
+
});
|
|
779
|
+
proxyRouter.post('/audio/speech', async (req, res) => {
|
|
780
|
+
if (!requireInferenceAuth(req, res))
|
|
781
|
+
return;
|
|
782
|
+
const parsed = SpeechBody.safeParse(req.body);
|
|
783
|
+
if (!parsed.success) {
|
|
784
|
+
res.status(400).json({ error: { message: 'Invalid request: `input` is required', type: 'invalid_request_error' } });
|
|
785
|
+
return;
|
|
786
|
+
}
|
|
787
|
+
try {
|
|
788
|
+
const result = await runSpeech(parsed.data.model, {
|
|
789
|
+
input: parsed.data.input, voice: parsed.data.voice, format: parsed.data.response_format,
|
|
790
|
+
});
|
|
791
|
+
res.setHeader('Content-Type', result.contentType);
|
|
792
|
+
res.setHeader('X-Provider', safeHeaderValue(result.platform));
|
|
793
|
+
res.send(result.audio);
|
|
794
|
+
}
|
|
795
|
+
catch (err) {
|
|
796
|
+
const status = err instanceof MediaError ? err.status : 502;
|
|
797
|
+
const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
|
|
798
|
+
const httpStatus = status >= 400 && status < 600 ? status : 502;
|
|
799
|
+
res.status(httpStatus).json({ error: { message: `speech error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
|
|
800
|
+
}
|
|
801
|
+
});
|
|
802
|
+
// OpenAI-compatible speech-to-text (/v1/audio/transcriptions). Multipart form
|
|
803
|
+
// upload, held in memory only (multer memoryStorage — audio bytes never touch
|
|
804
|
+
// disk), routed through the STT provider chain in services/media.ts with the
|
|
805
|
+
// same key/failover/cooldown machinery as the other media endpoints. The STT
|
|
806
|
+
// registry (media_models, modality='transcription') is maintained by the
|
|
807
|
+
// published catalog's `transcriptionModels` array via catalog-sync, plus any
|
|
808
|
+
// OpenAI-compatible endpoint the operator registered themselves through
|
|
809
|
+
// POST /api/media/custom; on an install that has never synced one and has no
|
|
810
|
+
// custom row, the endpoint answers 503 with code 'no_transcription_models'.
|
|
811
|
+
//
|
|
812
|
+
// response_format: 'json' (default, {"text": ...}), 'text' (plain string),
|
|
813
|
+
// 'verbose_json' (OpenAI verbose shape when the provider returns segments,
|
|
814
|
+
// graceful fallback to the plain json shape otherwise), 'vtt' (only from
|
|
815
|
+
// providers that produce it natively — Cloudflare whisper). 'srt' is not
|
|
816
|
+
// produced natively by any configured provider and is refused with 400
|
|
817
|
+
// unsupported_format rather than synthesized.
|
|
818
|
+
const transcriptionUpload = multer({
|
|
819
|
+
storage: multer.memoryStorage(),
|
|
820
|
+
limits: { fileSize: MAX_TRANSCRIPTION_BYTES, files: 1 },
|
|
821
|
+
});
|
|
822
|
+
const TRANSCRIPTION_FORMATS = new Set(['json', 'text', 'verbose_json', 'srt', 'vtt']);
|
|
823
|
+
function transcriptionBadRequest(res, message, code) {
|
|
824
|
+
res.status(400).json({ error: { message, type: 'invalid_request_error', ...(code ? { code } : {}) } });
|
|
825
|
+
}
|
|
826
|
+
proxyRouter.post('/audio/transcriptions', (req, res, next) => {
|
|
827
|
+
// Auth before the multipart body is parsed: an unauthenticated caller's
|
|
828
|
+
// upload is never buffered.
|
|
829
|
+
if (!requireInferenceAuth(req, res))
|
|
830
|
+
return;
|
|
831
|
+
transcriptionUpload.single('file')(req, res, (err) => {
|
|
832
|
+
if (err) {
|
|
833
|
+
if (err instanceof multer.MulterError && err.code === 'LIMIT_FILE_SIZE') {
|
|
834
|
+
res.status(413).json({
|
|
835
|
+
error: {
|
|
836
|
+
message: `Audio file too large: the maximum upload size is ${MAX_TRANSCRIPTION_BYTES / (1024 * 1024)} MB.`,
|
|
837
|
+
type: 'invalid_request_error',
|
|
838
|
+
code: 'file_too_large',
|
|
839
|
+
},
|
|
840
|
+
});
|
|
841
|
+
return;
|
|
842
|
+
}
|
|
843
|
+
transcriptionBadRequest(res, 'Malformed multipart/form-data upload.');
|
|
844
|
+
return;
|
|
845
|
+
}
|
|
846
|
+
next();
|
|
847
|
+
});
|
|
848
|
+
}, async (req, res) => {
|
|
849
|
+
const file = req.file;
|
|
850
|
+
if (!file || !file.buffer?.length) {
|
|
851
|
+
transcriptionBadRequest(res, 'Invalid request: `file` is required (multipart/form-data audio upload).');
|
|
852
|
+
return;
|
|
853
|
+
}
|
|
854
|
+
const model = typeof req.body?.model === 'string' ? req.body.model.trim() : '';
|
|
855
|
+
if (!model) {
|
|
856
|
+
transcriptionBadRequest(res, "Invalid request: `model` is required (use 'whisper-1' or 'auto' to let the router decide).");
|
|
857
|
+
return;
|
|
858
|
+
}
|
|
859
|
+
const rawFormat = typeof req.body?.response_format === 'string' ? req.body.response_format.trim() : '';
|
|
860
|
+
const responseFormat = rawFormat || 'json';
|
|
861
|
+
if (!TRANSCRIPTION_FORMATS.has(responseFormat)) {
|
|
862
|
+
transcriptionBadRequest(res, `Invalid response_format '${responseFormat}'. Supported: json, text, verbose_json, vtt.`);
|
|
863
|
+
return;
|
|
864
|
+
}
|
|
865
|
+
if (responseFormat === 'srt') {
|
|
866
|
+
transcriptionBadRequest(res, "response_format 'srt' is not supported: no configured provider produces srt natively. Use json, text, verbose_json, or vtt.", 'unsupported_format');
|
|
867
|
+
return;
|
|
868
|
+
}
|
|
869
|
+
let temperature;
|
|
870
|
+
if (req.body?.temperature !== undefined && req.body.temperature !== '') {
|
|
871
|
+
temperature = Number(req.body.temperature);
|
|
872
|
+
if (!Number.isFinite(temperature) || temperature < 0 || temperature > 1) {
|
|
873
|
+
transcriptionBadRequest(res, 'Invalid temperature: must be a number between 0 and 1.');
|
|
874
|
+
return;
|
|
875
|
+
}
|
|
876
|
+
}
|
|
877
|
+
const language = typeof req.body?.language === 'string' && req.body.language.trim() ? req.body.language.trim() : undefined;
|
|
878
|
+
const prompt = typeof req.body?.prompt === 'string' && req.body.prompt ? req.body.prompt : undefined;
|
|
879
|
+
try {
|
|
880
|
+
const result = await runTranscription(model, {
|
|
881
|
+
file: file.buffer,
|
|
882
|
+
filename: file.originalname || 'audio',
|
|
883
|
+
mimeType: file.mimetype,
|
|
884
|
+
language,
|
|
885
|
+
prompt,
|
|
886
|
+
temperature,
|
|
887
|
+
responseFormat,
|
|
888
|
+
});
|
|
889
|
+
res.setHeader('X-Provider', safeHeaderValue(result.platform));
|
|
890
|
+
res.setHeader('X-Model', safeHeaderValue(result.modelId));
|
|
891
|
+
if (responseFormat === 'text') {
|
|
892
|
+
res.type('text/plain').send(result.text);
|
|
893
|
+
return;
|
|
894
|
+
}
|
|
895
|
+
if (responseFormat === 'vtt') {
|
|
896
|
+
res.type('text/vtt').send(result.vtt ?? '');
|
|
897
|
+
return;
|
|
898
|
+
}
|
|
899
|
+
if (responseFormat === 'verbose_json' && Array.isArray(result.segments) && result.segments.length > 0) {
|
|
900
|
+
res.json({
|
|
901
|
+
task: 'transcribe',
|
|
902
|
+
language: result.language ?? null,
|
|
903
|
+
duration: result.duration ?? null,
|
|
904
|
+
text: result.text,
|
|
905
|
+
segments: result.segments,
|
|
906
|
+
});
|
|
907
|
+
return;
|
|
908
|
+
}
|
|
909
|
+
res.json({ text: result.text });
|
|
910
|
+
}
|
|
911
|
+
catch (err) {
|
|
912
|
+
const status = err instanceof MediaError ? err.status : 502;
|
|
913
|
+
const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
|
|
914
|
+
const httpStatus = status >= 400 && status < 600 ? status : 502;
|
|
915
|
+
res.status(httpStatus).json({ error: { message: `transcription error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
|
|
916
|
+
}
|
|
917
|
+
});
|
|
918
|
+
const CompletionBody = z.object({
|
|
919
|
+
model: z.string().optional(),
|
|
920
|
+
prompt: z.string(),
|
|
921
|
+
suffix: z.string().optional(),
|
|
922
|
+
temperature: z.number().min(0).max(2).optional(),
|
|
923
|
+
max_tokens: z.number().int().optional(),
|
|
924
|
+
top_p: z.number().min(0).max(1).optional(),
|
|
925
|
+
stop: stopSchema.optional(),
|
|
926
|
+
stream: z.boolean().optional(),
|
|
927
|
+
});
|
|
928
|
+
function completionPromptToMessages(prompt, suffix) {
|
|
929
|
+
const hasSuffix = suffix !== undefined && suffix.length > 0;
|
|
930
|
+
return [
|
|
931
|
+
{
|
|
932
|
+
role: 'system',
|
|
933
|
+
content: [
|
|
934
|
+
'You are a code autocomplete engine.',
|
|
935
|
+
'Complete at the cursor and return only the text to insert.',
|
|
936
|
+
'Do not include markdown fences, explanations, or repeat surrounding code.',
|
|
937
|
+
].join(' '),
|
|
938
|
+
},
|
|
939
|
+
{
|
|
940
|
+
role: 'user',
|
|
941
|
+
content: hasSuffix
|
|
942
|
+
? `Prefix before cursor:\n${prompt}\n\nSuffix after cursor:\n${suffix}\n\nCompletion to insert:`
|
|
943
|
+
: `Prefix before cursor:\n${prompt}\n\nCompletion to insert:`,
|
|
944
|
+
},
|
|
945
|
+
];
|
|
946
|
+
}
|
|
947
|
+
function completionTextFromChat(result) {
|
|
948
|
+
return contentToString(result?.choices?.[0]?.message?.content ?? '');
|
|
949
|
+
}
|
|
950
|
+
// Non-streaming counterpart of streamReasoningText: reasoning models attach
|
|
951
|
+
// thinking to the completed message as `reasoning_content` or `reasoning`.
|
|
952
|
+
// Included in the chars/4 output estimate so analytics and the rate-limit
|
|
953
|
+
// ledger aren't undercounted for thinking models. (#764)
|
|
954
|
+
export function completionReasoningText(result) {
|
|
955
|
+
const msg = result?.choices?.[0]?.message;
|
|
956
|
+
const r = msg?.reasoning_content ?? msg?.reasoning;
|
|
957
|
+
return typeof r === 'string' ? r : '';
|
|
958
|
+
}
|
|
959
|
+
function completionIdFromChat(id) {
|
|
960
|
+
if (!id)
|
|
961
|
+
return `cmpl-${Date.now()}`;
|
|
962
|
+
return id.startsWith('cmpl-') ? id : `cmpl-${id}`;
|
|
963
|
+
}
|
|
964
|
+
function legacyCompletionChunk(route, chunk, text) {
|
|
965
|
+
return {
|
|
966
|
+
id: completionIdFromChat(chunk?.id),
|
|
967
|
+
object: 'text_completion',
|
|
968
|
+
created: chunk?.created ?? Math.floor(Date.now() / 1000),
|
|
969
|
+
model: route.modelId,
|
|
970
|
+
choices: [{
|
|
971
|
+
text,
|
|
972
|
+
index: chunk?.choices?.[0]?.index ?? 0,
|
|
973
|
+
logprobs: null,
|
|
974
|
+
finish_reason: chunk?.choices?.[0]?.finish_reason ?? null,
|
|
975
|
+
}],
|
|
976
|
+
};
|
|
977
|
+
}
|
|
978
|
+
// OpenAI-compatible legacy completions endpoint. Editor ghost-text clients
|
|
979
|
+
// (notably Continue autocomplete) still send prompt/suffix requests here; route
|
|
980
|
+
// those through chat models while preserving the legacy text_completion shape.
|
|
981
|
+
proxyRouter.post('/completions', async (req, res) => {
|
|
982
|
+
const start = Date.now();
|
|
983
|
+
const requestGroupId = getRequestGroupId(req);
|
|
984
|
+
res.setHeader('X-Request-ID', requestGroupId);
|
|
985
|
+
const auth = requireInferenceAuth(req, res);
|
|
986
|
+
if (!auth)
|
|
987
|
+
return;
|
|
988
|
+
const parsed = CompletionBody.safeParse(req.body);
|
|
989
|
+
if (!parsed.success) {
|
|
990
|
+
const detail = parsed.error.errors
|
|
991
|
+
.map(e => (e.path.length ? `${e.path.join('.')}: ${e.message}` : e.message))
|
|
992
|
+
.slice(0, 5)
|
|
993
|
+
.join(', ');
|
|
994
|
+
res.status(400).json({
|
|
995
|
+
error: { message: `Invalid request: ${detail}`, type: 'invalid_request_error' },
|
|
996
|
+
execution_id: requestGroupId,
|
|
997
|
+
});
|
|
998
|
+
return;
|
|
999
|
+
}
|
|
1000
|
+
const { model: requestedModel, prompt, suffix, temperature, top_p, stream } = parsed.data;
|
|
1001
|
+
const requestedModelLabel = requestedModel ?? 'auto';
|
|
1002
|
+
const max_tokens = parsed.data.max_tokens != null && parsed.data.max_tokens > 0
|
|
1003
|
+
? parsed.data.max_tokens : 128;
|
|
1004
|
+
const stop = providerSafeStop(parsed.data.stop);
|
|
1005
|
+
// A profile's enforced prompt goes ahead of the autocomplete system message.
|
|
1006
|
+
const messages = prependSystemPrompt(completionPromptToMessages(prompt, suffix), auth.systemPrompt);
|
|
1007
|
+
const estimatedInputTokens = messages.reduce((sum, m) => sum + Math.ceil(contentToString(m.content).length / 4), 0);
|
|
1008
|
+
// Cap the reserved output so a huge client-set max_tokens doesn't falsely
|
|
1009
|
+
// exclude the whole model pool (#470); input is still counted in full. The
|
|
1010
|
+
// reserve is passed to the router separately: it is an exact count and must
|
|
1011
|
+
// not be inflated by the context-window safety margin (#956 review).
|
|
1012
|
+
const outputReserve = routingReserveTokens(max_tokens);
|
|
1013
|
+
const estimatedTotal = estimatedInputTokens + outputReserve;
|
|
1014
|
+
// Guardrail: per-request token budget (request_max_tokens_budget, default
|
|
1015
|
+
// off). max_tokens always has a value on this surface (default 128), so a
|
|
1016
|
+
// violation can only reject — no capping branch.
|
|
1017
|
+
const budgetCheck = applyTokenBudget(estimatedInputTokens, max_tokens);
|
|
1018
|
+
if (budgetCheck.rejection) {
|
|
1019
|
+
res.status(413).json({
|
|
1020
|
+
error: { message: tokenBudgetMessage(budgetCheck.rejection), type: 'invalid_request_error', code: 'request_token_budget' },
|
|
1021
|
+
execution_id: requestGroupId,
|
|
1022
|
+
});
|
|
1023
|
+
return;
|
|
1024
|
+
}
|
|
1025
|
+
let resolvedChain;
|
|
1026
|
+
if (isAutoModel(requestedModel)) {
|
|
1027
|
+
resolvedChain = resolveRoutingChain(requestedModel);
|
|
1028
|
+
}
|
|
1029
|
+
let preferredModel;
|
|
1030
|
+
let groupChain;
|
|
1031
|
+
if (!isAutoModel(requestedModel) && requestedModel) {
|
|
1032
|
+
const db = getDb();
|
|
1033
|
+
const resolved = isUnifyEnabled() ? resolveRequestedIdForDispatch(requestedModel, getModelGroups()) : null;
|
|
1034
|
+
const members = resolved?.memberDbIds ?? null;
|
|
1035
|
+
if (members && members.length > 0) {
|
|
1036
|
+
groupChain = resolveModelGroupCandidates(members, resolved.demotedDbIds);
|
|
1037
|
+
if (groupChain.length === 0) {
|
|
1038
|
+
const placeholders = members.map(() => '?').join(',');
|
|
1039
|
+
const anyEnabled = db.prepare(`SELECT 1 FROM models WHERE id IN (${placeholders}) AND enabled = 1 LIMIT 1`).get(...members);
|
|
1040
|
+
// Honest statuses: a model whose providers exist but have no usable key
|
|
1041
|
+
// is a server-side configuration gap (503), not a client mistake; a
|
|
1042
|
+
// disabled/unknown model is a 404 model_not_found (OpenAI semantics).
|
|
1043
|
+
if (anyEnabled) {
|
|
1044
|
+
res.status(503).json({
|
|
1045
|
+
error: {
|
|
1046
|
+
message: `Model '${requestedModel}' has no providers with an enabled key. Add a provider API key for it, use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1047
|
+
type: 'service_unavailable',
|
|
1048
|
+
code: 'no_providers_configured',
|
|
1049
|
+
},
|
|
1050
|
+
execution_id: requestGroupId,
|
|
1051
|
+
});
|
|
1052
|
+
}
|
|
1053
|
+
else {
|
|
1054
|
+
res.status(404).json({
|
|
1055
|
+
error: {
|
|
1056
|
+
message: `Model '${requestedModel}' is disabled. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1057
|
+
type: 'invalid_request_error',
|
|
1058
|
+
code: 'model_not_found',
|
|
1059
|
+
},
|
|
1060
|
+
execution_id: requestGroupId,
|
|
1061
|
+
});
|
|
1062
|
+
}
|
|
1063
|
+
return;
|
|
1064
|
+
}
|
|
1065
|
+
}
|
|
1066
|
+
else {
|
|
1067
|
+
const enabled = db.prepare('SELECT id FROM models WHERE model_id = ? AND enabled = 1').get(requestedModel);
|
|
1068
|
+
if (enabled) {
|
|
1069
|
+
preferredModel = enabled.id;
|
|
1070
|
+
}
|
|
1071
|
+
else {
|
|
1072
|
+
const disabled = db.prepare('SELECT id FROM models WHERE model_id = ?').get(requestedModel);
|
|
1073
|
+
const reason = disabled ? 'is disabled' : 'is not in the catalog';
|
|
1074
|
+
res.status(404).json({
|
|
1075
|
+
error: {
|
|
1076
|
+
message: `Model '${requestedModel}' ${reason}. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1077
|
+
type: 'invalid_request_error',
|
|
1078
|
+
code: 'model_not_found',
|
|
1079
|
+
},
|
|
1080
|
+
execution_id: requestGroupId,
|
|
1081
|
+
});
|
|
1082
|
+
return;
|
|
1083
|
+
}
|
|
1084
|
+
}
|
|
1085
|
+
}
|
|
1086
|
+
const pinnedModelId = requestedModel && !isAutoModel(requestedModel) ? requestedModel : null;
|
|
1087
|
+
const state = newFallbackState();
|
|
1088
|
+
const attemptLog = [];
|
|
1089
|
+
// Client-disconnect fan-out: the flag stops the loop before the NEXT
|
|
1090
|
+
// attempt; the AbortController (threaded to the provider as
|
|
1091
|
+
// CompletionOptions.signal) additionally cancels the IN-FLIGHT upstream
|
|
1092
|
+
// fetch and any body/stream read, so tokens stop burning and the in-flight
|
|
1093
|
+
// lease frees immediately. 'close' also fires on normal completion —
|
|
1094
|
+
// writableEnded distinguishes a real disconnect.
|
|
1095
|
+
let clientGone = false;
|
|
1096
|
+
const clientAbort = new AbortController();
|
|
1097
|
+
// Fallback-v2 hedging: the loop aborts this controller (via abortInFlight)
|
|
1098
|
+
// when the wall-clock retry budget expires mid-attempt, canceling the
|
|
1099
|
+
// in-flight upstream instead of waiting for a stalled attempt to time out.
|
|
1100
|
+
const hedgeAbort = new AbortController();
|
|
1101
|
+
res.on('close', () => {
|
|
1102
|
+
if (!res.writableEnded) {
|
|
1103
|
+
clientGone = true;
|
|
1104
|
+
clientAbort.abort(newClientAbortError());
|
|
1105
|
+
}
|
|
1106
|
+
});
|
|
1107
|
+
// Legacy /completions is a thin adapter over the shared fallback loop
|
|
1108
|
+
// (lib/fallback-loop.ts): the cooldown/skip/penalty/exhaustion machinery is
|
|
1109
|
+
// shared; only the text_completion request/stream translation lives here.
|
|
1110
|
+
await runFallbackLoop({
|
|
1111
|
+
maxRetries: MAX_RETRIES,
|
|
1112
|
+
state,
|
|
1113
|
+
attemptLog,
|
|
1114
|
+
logIdentity: { surface: 'legacy completions', requestId: requestGroupId, requestedModel: requestedModelLabel },
|
|
1115
|
+
clientGone: () => clientGone,
|
|
1116
|
+
abortInFlight: () => hedgeAbort.abort(newHedgeAbortError()),
|
|
1117
|
+
route: () => {
|
|
1118
|
+
// #507: see inbound-chat.ts — inflate the routing estimate from any
|
|
1119
|
+
// provider-reported REQUESTED size latched onto state so the existing
|
|
1120
|
+
// size gates in router.ts skip low-TPM / small-context models on retry.
|
|
1121
|
+
const routingTotal = fallbackRoutingTokens(state, estimatedTotal, outputReserve);
|
|
1122
|
+
return routeRequest(routingTotal, state.skipKeys.size > 0 ? state.skipKeys : undefined, preferredModel, false, false, state.skipModels.size > 0 ? state.skipModels : undefined, groupChain ?? resolvedChain?.chain, false, state.skipPlatforms.size > 0 ? state.skipPlatforms : undefined, outputReserve);
|
|
1123
|
+
},
|
|
1124
|
+
dispatch: async (route, attempt, ctx) => {
|
|
1125
|
+
const contextBudget = routeOutputBudget(route, estimatedInputTokens);
|
|
1126
|
+
traceRouteEvent('Proxy', {
|
|
1127
|
+
event: attempt === 0 ? 'start' : 'next',
|
|
1128
|
+
requestId: requestGroupId,
|
|
1129
|
+
attempt,
|
|
1130
|
+
platform: route.platform,
|
|
1131
|
+
model: route.modelId,
|
|
1132
|
+
requestedModel: attempt === 0 ? requestedModelLabel : undefined,
|
|
1133
|
+
});
|
|
1134
|
+
// Same GitHub input ceiling as /chat/completions below: trim the
|
|
1135
|
+
// dispatched copy so a long legacy prompt doesn't 413 the github hop.
|
|
1136
|
+
const dispatchMessages = route.platform === 'github'
|
|
1137
|
+
? truncateMessagesForGithub(messages)
|
|
1138
|
+
: messages;
|
|
1139
|
+
if (stream) {
|
|
1140
|
+
let totalOutputTokens = 0;
|
|
1141
|
+
let headerSent = false;
|
|
1142
|
+
let ttfbMs = null;
|
|
1143
|
+
let sawText = false;
|
|
1144
|
+
let upstreamFinish = null;
|
|
1145
|
+
const buffered = [];
|
|
1146
|
+
const flushHeaders = () => {
|
|
1147
|
+
if (headerSent)
|
|
1148
|
+
return;
|
|
1149
|
+
// #764: ttfb is recorded on the first token of ANY kind (content or
|
|
1150
|
+
// reasoning) in the pump loop below; this call only backfills streams
|
|
1151
|
+
// that reached the commit point without one.
|
|
1152
|
+
if (ttfbMs === null)
|
|
1153
|
+
ttfbMs = Date.now() - start;
|
|
1154
|
+
res.setHeader('Content-Type', 'text/event-stream');
|
|
1155
|
+
res.setHeader('Cache-Control', 'no-cache');
|
|
1156
|
+
res.setHeader('Connection', 'keep-alive');
|
|
1157
|
+
res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
|
|
1158
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
1159
|
+
headerSent = true;
|
|
1160
|
+
// Committed: the answer is on its way, so the retry budget must no
|
|
1161
|
+
// longer cancel this attempt (it could not fail over now anyway).
|
|
1162
|
+
ctx.disarmHedge();
|
|
1163
|
+
for (const frame of buffered)
|
|
1164
|
+
res.write(`data: ${JSON.stringify(frame)}\n\n`);
|
|
1165
|
+
buffered.length = 0;
|
|
1166
|
+
};
|
|
1167
|
+
try {
|
|
1168
|
+
const gen = route.provider.streamChatCompletion(route.apiKey, dispatchMessages, route.modelId, { temperature, max_tokens, top_p, stop, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
|
|
1169
|
+
for await (const chunk of gen) {
|
|
1170
|
+
if (clientGone)
|
|
1171
|
+
break; // client hung up: stop pulling; reader.cancel() aborts upstream
|
|
1172
|
+
const text = streamChunkText(chunk);
|
|
1173
|
+
if (text.length > 0)
|
|
1174
|
+
sawText = true;
|
|
1175
|
+
// #764: reasoning models stream thinking before any visible text.
|
|
1176
|
+
// ttfb must count the first token of ANY kind — otherwise the speed
|
|
1177
|
+
// shown is the thinking tail, or NULL when headers never flush.
|
|
1178
|
+
const reasoning = streamReasoningText(chunk);
|
|
1179
|
+
if (ttfbMs === null && (text.length > 0 || reasoning.length > 0)) {
|
|
1180
|
+
ttfbMs = Date.now() - start;
|
|
1181
|
+
}
|
|
1182
|
+
const finish = chunk?.choices?.[0]?.finish_reason;
|
|
1183
|
+
if (finish)
|
|
1184
|
+
upstreamFinish = finish;
|
|
1185
|
+
// #764: reasoning tokens are real output consumption — count them
|
|
1186
|
+
// so analytics and the rate-limit ledger aren't undercounted.
|
|
1187
|
+
totalOutputTokens += Math.ceil((text.length + reasoning.length) / 4);
|
|
1188
|
+
const frame = legacyCompletionChunk(route, chunk, text);
|
|
1189
|
+
// Commit point: hold headers until the first real text, so a stream
|
|
1190
|
+
// that dies before producing any fails over invisibly.
|
|
1191
|
+
if (!headerSent && !sawText) {
|
|
1192
|
+
buffered.push(frame);
|
|
1193
|
+
continue;
|
|
1194
|
+
}
|
|
1195
|
+
flushHeaders();
|
|
1196
|
+
res.write(`data: ${JSON.stringify(frame)}\n\n`);
|
|
1197
|
+
}
|
|
1198
|
+
// Disconnect before the commit point: the break above fired with no
|
|
1199
|
+
// text seen, which is indistinguishable from an empty completion
|
|
1200
|
+
// below — but it is CLIENT behavior, not a provider failure. Without
|
|
1201
|
+
// this check every Ctrl-C during a reasoning model's TTFB window
|
|
1202
|
+
// benched the healthy model+key for 90s and logged a provider error.
|
|
1203
|
+
if (clientGone && !headerSent && !sawText) {
|
|
1204
|
+
console.log(`[Proxy] client disconnected before first token from ${route.displayName} — dropping attempt without benching`);
|
|
1205
|
+
traceRouteEvent('Proxy', {
|
|
1206
|
+
event: 'canceled',
|
|
1207
|
+
requestId: requestGroupId,
|
|
1208
|
+
attempt,
|
|
1209
|
+
platform: route.platform,
|
|
1210
|
+
model: route.modelId,
|
|
1211
|
+
});
|
|
1212
|
+
return 'committed';
|
|
1213
|
+
}
|
|
1214
|
+
if (!sawText) {
|
|
1215
|
+
// finish_reason 'length' means the model spent the whole output
|
|
1216
|
+
// budget before any visible text (hidden reasoning) — fail over,
|
|
1217
|
+
// but skip the cooldown/penalty: not a provider-health signal.
|
|
1218
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName} (legacy stream produced no text)`), upstreamFinish === 'length' ? { skipBench: true } : {});
|
|
1219
|
+
}
|
|
1220
|
+
flushHeaders();
|
|
1221
|
+
res.write('data: [DONE]\n\n');
|
|
1222
|
+
res.end();
|
|
1223
|
+
recordUpstreamSuccess(route, estimatedInputTokens + totalOutputTokens);
|
|
1224
|
+
traceRouteEvent('Proxy', {
|
|
1225
|
+
event: 'ok',
|
|
1226
|
+
requestId: requestGroupId,
|
|
1227
|
+
attempt,
|
|
1228
|
+
platform: route.platform,
|
|
1229
|
+
model: route.modelId,
|
|
1230
|
+
latencyMs: Date.now() - start,
|
|
1231
|
+
inputTokens: estimatedInputTokens,
|
|
1232
|
+
outputTokens: totalOutputTokens,
|
|
1233
|
+
});
|
|
1234
|
+
logRequest(route.platform, route.modelId, route.keyId, 'success', estimatedInputTokens, totalOutputTokens, Date.now() - start, null, ttfbMs, pinnedModelId, null, 'http');
|
|
1235
|
+
return 'done';
|
|
1236
|
+
}
|
|
1237
|
+
catch (streamErr) {
|
|
1238
|
+
// Client abort mid-stream: the pump's own `if (clientGone) break`
|
|
1239
|
+
// can lose the race against the fetch-signal rejection, so the
|
|
1240
|
+
// abort may surface here instead. Rethrow — the shared loop's
|
|
1241
|
+
// client-abort branch stops the ladder without benching or an
|
|
1242
|
+
// error log row (the socket is gone; nothing to render).
|
|
1243
|
+
if (isClientAbortError(streamErr))
|
|
1244
|
+
throw streamErr;
|
|
1245
|
+
if (headerSent) {
|
|
1246
|
+
console.error(`[Proxy] Mid-stream legacy completion error from ${route.displayName}:`, streamErr.message);
|
|
1247
|
+
const payload = { error: { message: `Provider error (${route.displayName}): stream interrupted`, type: 'stream_error' } };
|
|
1248
|
+
try {
|
|
1249
|
+
res.write(`data: ${JSON.stringify(payload)}\n\n`);
|
|
1250
|
+
}
|
|
1251
|
+
catch { /* socket gone */ }
|
|
1252
|
+
try {
|
|
1253
|
+
res.write('data: [DONE]\n\n');
|
|
1254
|
+
res.end();
|
|
1255
|
+
}
|
|
1256
|
+
catch { /* socket gone */ }
|
|
1257
|
+
traceRouteEvent('Proxy', {
|
|
1258
|
+
event: 'fail',
|
|
1259
|
+
requestId: requestGroupId,
|
|
1260
|
+
attempt,
|
|
1261
|
+
platform: route.platform,
|
|
1262
|
+
model: route.modelId,
|
|
1263
|
+
latencyMs: Date.now() - start,
|
|
1264
|
+
error: sanitizeProviderErrorMessage(streamErr.message),
|
|
1265
|
+
});
|
|
1266
|
+
logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, sanitizeProviderErrorMessage(streamErr.message), ttfbMs, pinnedModelId, null, 'http');
|
|
1267
|
+
return 'committed';
|
|
1268
|
+
}
|
|
1269
|
+
throw streamErr;
|
|
1270
|
+
}
|
|
1271
|
+
}
|
|
1272
|
+
const result = await route.provider.chatCompletion(route.apiKey, dispatchMessages, route.modelId, { temperature, max_tokens, top_p, stop, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
|
|
1273
|
+
const text = completionTextFromChat(result);
|
|
1274
|
+
if (!text) {
|
|
1275
|
+
// finish_reason 'length' = output budget consumed by hidden reasoning
|
|
1276
|
+
// before any visible text: fail over without a cooldown/penalty.
|
|
1277
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName}`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
|
|
1278
|
+
}
|
|
1279
|
+
// #809: a bare "safe"/"unsafe" classification word from a relay is an
|
|
1280
|
+
// upstream filter, not the requested model — fail over like an empty
|
|
1281
|
+
// completion.
|
|
1282
|
+
if (isUpstreamClassificationOutput(text, route.platform)) {
|
|
1283
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
|
|
1284
|
+
}
|
|
1285
|
+
// Usage fallback: providers that omit `usage` used to be logged as 0
|
|
1286
|
+
// tokens, silently undercounting analytics and the rate-limit ledger.
|
|
1287
|
+
// Fall back to the same chars/4 estimate the streaming path uses,
|
|
1288
|
+
// including reasoning tokens (thinking models). (#764)
|
|
1289
|
+
const promptTokens = result.usage?.prompt_tokens ?? estimatedInputTokens;
|
|
1290
|
+
const completionTokens = result.usage?.completion_tokens
|
|
1291
|
+
?? Math.ceil((text.length + completionReasoningText(result).length) / 4);
|
|
1292
|
+
const totalTokens = result.usage?.total_tokens ?? (promptTokens + completionTokens);
|
|
1293
|
+
recordUpstreamSuccess(route, totalTokens);
|
|
1294
|
+
res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
|
|
1295
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
1296
|
+
res.json({
|
|
1297
|
+
id: completionIdFromChat(result.id),
|
|
1298
|
+
object: 'text_completion',
|
|
1299
|
+
created: result.created ?? Math.floor(Date.now() / 1000),
|
|
1300
|
+
model: route.modelId,
|
|
1301
|
+
choices: [{
|
|
1302
|
+
text,
|
|
1303
|
+
index: result.choices?.[0]?.index ?? 0,
|
|
1304
|
+
logprobs: null,
|
|
1305
|
+
finish_reason: result.choices?.[0]?.finish_reason ?? 'stop',
|
|
1306
|
+
}],
|
|
1307
|
+
// `usage` is required by the OpenAI completions spec. The fallback
|
|
1308
|
+
// counts computed just above (chars/4 when the provider omits usage,
|
|
1309
|
+
// #764) existed but were dropped here — clients like editor
|
|
1310
|
+
// ghost-text plugins read usage to throttle and saw `undefined`.
|
|
1311
|
+
usage: result.usage ?? {
|
|
1312
|
+
prompt_tokens: promptTokens,
|
|
1313
|
+
completion_tokens: completionTokens,
|
|
1314
|
+
total_tokens: totalTokens,
|
|
1315
|
+
estimated: true,
|
|
1316
|
+
},
|
|
1317
|
+
execution_id: requestGroupId,
|
|
1318
|
+
});
|
|
1319
|
+
traceRouteEvent('Proxy', {
|
|
1320
|
+
event: 'ok',
|
|
1321
|
+
requestId: requestGroupId,
|
|
1322
|
+
attempt,
|
|
1323
|
+
platform: route.platform,
|
|
1324
|
+
model: route.modelId,
|
|
1325
|
+
latencyMs: Date.now() - start,
|
|
1326
|
+
inputTokens: promptTokens,
|
|
1327
|
+
outputTokens: completionTokens,
|
|
1328
|
+
});
|
|
1329
|
+
logRequest(route.platform, route.modelId, route.keyId, 'success', promptTokens, completionTokens, Date.now() - start, null, null, pinnedModelId, null, 'http');
|
|
1330
|
+
return 'done';
|
|
1331
|
+
},
|
|
1332
|
+
logFailure: (route, err, attempt) => {
|
|
1333
|
+
const latency = Date.now() - start;
|
|
1334
|
+
const safeError = sanitizeProviderErrorMessage(err.message);
|
|
1335
|
+
traceRouteEvent('Proxy', {
|
|
1336
|
+
event: 'fail',
|
|
1337
|
+
requestId: requestGroupId,
|
|
1338
|
+
attempt,
|
|
1339
|
+
platform: route.platform,
|
|
1340
|
+
model: route.modelId,
|
|
1341
|
+
latencyMs: latency,
|
|
1342
|
+
error: safeError,
|
|
1343
|
+
});
|
|
1344
|
+
logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, 0, latency, safeError, null, pinnedModelId, null, 'http');
|
|
1345
|
+
},
|
|
1346
|
+
onFatal: (route, err, attempt) => {
|
|
1347
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
1348
|
+
res.status(502).json({
|
|
1349
|
+
error: {
|
|
1350
|
+
message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(err.message)}`,
|
|
1351
|
+
type: 'provider_error',
|
|
1352
|
+
},
|
|
1353
|
+
execution_id: requestGroupId,
|
|
1354
|
+
});
|
|
1355
|
+
},
|
|
1356
|
+
onRoutingExhausted: (lastError, routeErr, exhaustion, info) => {
|
|
1357
|
+
setFallbackHeaders(res, info.attempts.length, info.attempts);
|
|
1358
|
+
setExhaustionHeaders(res, exhaustion);
|
|
1359
|
+
res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
|
|
1360
|
+
},
|
|
1361
|
+
onExhausted: (exhaustion, info) => {
|
|
1362
|
+
setFallbackHeaders(res, info.attempts.length, info.attempts);
|
|
1363
|
+
setExhaustionHeaders(res, exhaustion);
|
|
1364
|
+
res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
|
|
1365
|
+
},
|
|
1366
|
+
});
|
|
1367
|
+
});
|
|
1368
|
+
proxyRouter.post('/chat/completions', async (req, res) => {
|
|
1369
|
+
const start = Date.now();
|
|
1370
|
+
const requestGroupId = getRequestGroupId(req);
|
|
1371
|
+
res.setHeader('X-Request-ID', requestGroupId);
|
|
1372
|
+
// Authenticate every proxy request, including loopback callers. Browser
|
|
1373
|
+
// pages can reach localhost, so socket locality is not a reliable
|
|
1374
|
+
// authorization boundary. Client-profile keys resolve here too, carrying
|
|
1375
|
+
// their server-enforced system prompt (#411).
|
|
1376
|
+
const auth = requireInferenceAuth(req, res);
|
|
1377
|
+
if (!auth)
|
|
1378
|
+
return;
|
|
1379
|
+
// Validate request
|
|
1380
|
+
const parsed = chatCompletionSchema.safeParse(req.body);
|
|
1381
|
+
if (!parsed.success) {
|
|
1382
|
+
// Path-qualified issues ("messages.1.content: Invalid input" beats a bare
|
|
1383
|
+
// "Invalid input") and a server-side breadcrumb — these rejections never
|
|
1384
|
+
// reach the request log, which made #200 nearly undebuggable.
|
|
1385
|
+
const detail = parsed.error.errors
|
|
1386
|
+
.map(e => (e.path.length ? `${e.path.join('.')}: ${e.message}` : e.message))
|
|
1387
|
+
.slice(0, 5)
|
|
1388
|
+
.join(', ');
|
|
1389
|
+
console.warn(`[proxy] 400 invalid /chat/completions request: ${detail}`);
|
|
1390
|
+
res.status(400).json({
|
|
1391
|
+
error: {
|
|
1392
|
+
message: `Invalid request: ${detail}`,
|
|
1393
|
+
type: 'invalid_request_error',
|
|
1394
|
+
},
|
|
1395
|
+
execution_id: requestGroupId,
|
|
1396
|
+
});
|
|
1397
|
+
return;
|
|
1398
|
+
}
|
|
1399
|
+
const { model: requestedModel, temperature, top_p, stream } = parsed.data;
|
|
1400
|
+
const requestedModelLabel = requestedModel ?? 'auto';
|
|
1401
|
+
// Agent-tolerant knob normalization (#200): max_tokens <= 0 means "no
|
|
1402
|
+
// limit" in several clients → unset; tool_choice 'any' is OpenAI's
|
|
1403
|
+
// 'required'; tool definitions get their 'function' type re-defaulted.
|
|
1404
|
+
// `max_completion_tokens` is OpenAI's newer alias — honored when max_tokens
|
|
1405
|
+
// itself is absent. `let`: the token-budget guardrail below may cap an
|
|
1406
|
+
// absent max_tokens to the budget remainder before the options objects are
|
|
1407
|
+
// built from it.
|
|
1408
|
+
const requestedMaxTokens = parsed.data.max_tokens ?? parsed.data.max_completion_tokens;
|
|
1409
|
+
let max_tokens = requestedMaxTokens != null && requestedMaxTokens > 0
|
|
1410
|
+
? requestedMaxTokens : undefined;
|
|
1411
|
+
// Extended sampling/output params (seed, penalties, response_format…),
|
|
1412
|
+
// spread into every options object below — including fusion fan-out.
|
|
1413
|
+
const samplingParams = pickSamplingParams(parsed.data);
|
|
1414
|
+
const stop = providerSafeStop(parsed.data.stop);
|
|
1415
|
+
const tool_choice = parsed.data.tool_choice === 'any' ? 'required' : parsed.data.tool_choice ?? undefined;
|
|
1416
|
+
const tools = parsed.data.tools?.map(t => ({ ...t, type: 'function' }));
|
|
1417
|
+
const parallel_tool_calls = parsed.data.parallel_tool_calls ?? undefined;
|
|
1418
|
+
// Pairing state for id-less tool calls (#200): every tool_call id (given or
|
|
1419
|
+
// synthesized) queues up here; a tool message without a tool_call_id takes
|
|
1420
|
+
// the oldest unanswered one, which matches the single-call-per-turn flow
|
|
1421
|
+
// Gemini-lineage agents produce.
|
|
1422
|
+
const pendingToolCallIds = [];
|
|
1423
|
+
let syntheticIdCounter = 0;
|
|
1424
|
+
const takeToolCallId = (given) => {
|
|
1425
|
+
if (given && given.length > 0) {
|
|
1426
|
+
const qi = pendingToolCallIds.indexOf(given);
|
|
1427
|
+
if (qi !== -1)
|
|
1428
|
+
pendingToolCallIds.splice(qi, 1);
|
|
1429
|
+
return given;
|
|
1430
|
+
}
|
|
1431
|
+
return pendingToolCallIds.shift() ?? `call_auto_${++syntheticIdCounter}`;
|
|
1432
|
+
};
|
|
1433
|
+
let messages = parsed.data.messages.map((m) => {
|
|
1434
|
+
if (m.role === 'assistant') {
|
|
1435
|
+
const hasToolCalls = (m.tool_calls?.length ?? 0) > 0;
|
|
1436
|
+
// With tool_calls, content: null is the correct OpenAI shape — keep it.
|
|
1437
|
+
// Without tool_calls, coerce empty/null content to "" so strict upstreams
|
|
1438
|
+
// don't choke on a null-content assistant turn we just accepted. (#165)
|
|
1439
|
+
const isEmptyContent = m.content == null
|
|
1440
|
+
|| (typeof m.content === 'string' && m.content.length === 0)
|
|
1441
|
+
|| (Array.isArray(m.content) && m.content.length === 0);
|
|
1442
|
+
const assistantContent = hasToolCalls
|
|
1443
|
+
? (m.content ?? null)
|
|
1444
|
+
: (isEmptyContent ? '' : m.content);
|
|
1445
|
+
return {
|
|
1446
|
+
role: 'assistant',
|
|
1447
|
+
content: assistantContent,
|
|
1448
|
+
...(m.name ? { name: m.name } : {}),
|
|
1449
|
+
// Replay the thinking trace verbatim. DeepSeek thinking models on
|
|
1450
|
+
// OpenCode Zen reject a follow-up turn that drops it; other providers
|
|
1451
|
+
// ignore the unknown field. Same round-trip rationale as
|
|
1452
|
+
// thought_signature below. (#255)
|
|
1453
|
+
...(typeof m.reasoning_content === 'string' && m.reasoning_content.length > 0
|
|
1454
|
+
? { reasoning_content: m.reasoning_content }
|
|
1455
|
+
: {}),
|
|
1456
|
+
// Moonshot's "partial" prefill flag: keep it through the message build
|
|
1457
|
+
// (the schema already preserves it); the provider layer decides whether
|
|
1458
|
+
// the routed model understands it and strips it otherwise. (#1038)
|
|
1459
|
+
...(m.partial === true ? { partial: true } : {}),
|
|
1460
|
+
// hasToolCalls (not a bare truthiness check) so null AND empty-array
|
|
1461
|
+
// tool_calls are dropped rather than forwarded — strict upstreams
|
|
1462
|
+
// reject both shapes. (#200)
|
|
1463
|
+
...(hasToolCalls ? { tool_calls: m.tool_calls.map(tc => {
|
|
1464
|
+
// Normalize echo-tolerant inputs back to the strict OpenAI shape
|
|
1465
|
+
// before forwarding (see toolCallSchema); synthesize missing ids
|
|
1466
|
+
// and queue every id for order-based tool-result pairing. (#200)
|
|
1467
|
+
const id = tc.id && tc.id.length > 0 ? tc.id : `call_auto_${++syntheticIdCounter}`;
|
|
1468
|
+
pendingToolCallIds.push(id);
|
|
1469
|
+
return {
|
|
1470
|
+
id,
|
|
1471
|
+
type: 'function',
|
|
1472
|
+
function: { name: tc.function.name, arguments: toolCallArgsToString(tc.function.arguments) },
|
|
1473
|
+
thought_signature: tc.thought_signature,
|
|
1474
|
+
};
|
|
1475
|
+
}) } : {}),
|
|
1476
|
+
};
|
|
1477
|
+
}
|
|
1478
|
+
if (m.role === 'tool') {
|
|
1479
|
+
return {
|
|
1480
|
+
role: 'tool',
|
|
1481
|
+
// Null/missing content (a tool that returned nothing) → "". (#200)
|
|
1482
|
+
content: m.content ?? '',
|
|
1483
|
+
tool_call_id: takeToolCallId(m.tool_call_id),
|
|
1484
|
+
...(m.name ? { name: m.name } : {}),
|
|
1485
|
+
};
|
|
1486
|
+
}
|
|
1487
|
+
// Legacy function-calling result → forward as a tool message, paired by
|
|
1488
|
+
// order like an id-less tool message. (#200)
|
|
1489
|
+
if (m.role === 'function') {
|
|
1490
|
+
return {
|
|
1491
|
+
role: 'tool',
|
|
1492
|
+
content: m.content ?? '',
|
|
1493
|
+
tool_call_id: takeToolCallId(undefined),
|
|
1494
|
+
name: m.name,
|
|
1495
|
+
};
|
|
1496
|
+
}
|
|
1497
|
+
return {
|
|
1498
|
+
// 'developer' is OpenAI's newer name for the system role — providers
|
|
1499
|
+
// downstream only know 'system'. (#200)
|
|
1500
|
+
role: m.role === 'developer' ? 'system' : m.role,
|
|
1501
|
+
content: m.content,
|
|
1502
|
+
...(m.name ? { name: m.name } : {}),
|
|
1503
|
+
};
|
|
1504
|
+
});
|
|
1505
|
+
let cacheControlPrefixLength = 0;
|
|
1506
|
+
parsed.data.messages.forEach((message, index) => {
|
|
1507
|
+
const content = message.content;
|
|
1508
|
+
if (Array.isArray(content)
|
|
1509
|
+
&& content.some(block => block && typeof block === 'object' && 'cache_control' in block)) {
|
|
1510
|
+
cacheControlPrefixLength = index + 1;
|
|
1511
|
+
}
|
|
1512
|
+
});
|
|
1513
|
+
const compressionResult = compressRequest(messages, {
|
|
1514
|
+
header: req.headers['x-freellm-compress'],
|
|
1515
|
+
tools,
|
|
1516
|
+
cacheControlPrefixLength,
|
|
1517
|
+
});
|
|
1518
|
+
messages = compressionResult.messages;
|
|
1519
|
+
res.setHeader('X-FreeLLM-Compress', formatCompressionHeader(compressionResult));
|
|
1520
|
+
// Server-enforced system prompt (#411): injected AFTER compression so it is
|
|
1521
|
+
// never compressed away, and FIRST in the list so a caller-supplied system
|
|
1522
|
+
// message follows it and cannot override it. Constant per profile, so the
|
|
1523
|
+
// provider-side cache prefix stays stable across requests. Neutral no-op for
|
|
1524
|
+
// the unified key and for profiles without a prompt.
|
|
1525
|
+
messages = prependSystemPrompt(messages, auth.systemPrompt);
|
|
1526
|
+
// Downscale over-threshold inline images before estimation/routing so the
|
|
1527
|
+
// token budget, payload limits, and upstream transfer all see the shrunk
|
|
1528
|
+
// bytes (see lib/image-normalize.ts). Mutates the image blocks in place.
|
|
1529
|
+
await normalizeMessageImages(messages);
|
|
1530
|
+
// Token estimation is intentionally a heuristic (~4 chars per token). Used
|
|
1531
|
+
// for routing decisions (skip a model whose budget is too small) and for
|
|
1532
|
+
// streaming bookkeeping where the provider doesn't echo a final usage count.
|
|
1533
|
+
// Non-streaming requests reconcile against the provider's real `usage` block;
|
|
1534
|
+
// streaming does the same when stream_options.include_usage produces a final
|
|
1535
|
+
// usage frame, and otherwise falls back to this estimate.
|
|
1536
|
+
const estimatedInputTokens = estimateInputTokens(messages, tools);
|
|
1537
|
+
// Image requests must route to a vision-capable model. Reject up front with a
|
|
1538
|
+
// clear message when none is enabled, rather than silently dropping the image
|
|
1539
|
+
// or surfacing the generic "all models exhausted" error (#118, #125). Add a
|
|
1540
|
+
// rough per-image token cost so budget routing isn't skewed by content the
|
|
1541
|
+
// heuristic above (text-only) can't see.
|
|
1542
|
+
const hasImage = messageHasImage(messages);
|
|
1543
|
+
if (hasImage && !hasEnabledVisionModel()) {
|
|
1544
|
+
res.status(422).json({
|
|
1545
|
+
error: {
|
|
1546
|
+
message: 'This request includes an image, but no vision-capable model is enabled. Enable a vision model (e.g. Gemini 2.5 Flash, Llama 4 Scout) in the Fallback Chain.',
|
|
1547
|
+
type: 'invalid_request_error',
|
|
1548
|
+
code: 'no_vision_model',
|
|
1549
|
+
},
|
|
1550
|
+
execution_id: requestGroupId,
|
|
1551
|
+
});
|
|
1552
|
+
return;
|
|
1553
|
+
}
|
|
1554
|
+
const IMAGE_TOKEN_ESTIMATE = 1000;
|
|
1555
|
+
const imageCount = messages.reduce((n, m) => n + (Array.isArray(m.content) ? m.content.filter(b => b?.type === 'image_url' || b?.type === 'image').length : 0), 0);
|
|
1556
|
+
// The reserved output is capped (routingReserveTokens, #470) so an oversized
|
|
1557
|
+
// client max_tokens can't starve routing; input + images count in full. The
|
|
1558
|
+
// reserve is threaded to the router separately: it is exact and must not be
|
|
1559
|
+
// inflated by the context-window safety margin (#956 review).
|
|
1560
|
+
const outputReserve = routingReserveTokens(max_tokens);
|
|
1561
|
+
const estimatedTotal = estimatedInputTokens + imageCount * IMAGE_TOKEN_ESTIMATE + outputReserve;
|
|
1562
|
+
// Tool-bearing requests must route to a model that emits STRUCTURED
|
|
1563
|
+
// tool_calls. A model without real function-calling support serializes the
|
|
1564
|
+
// call into its text answer — the request "succeeds" but the client's tool
|
|
1565
|
+
// loop sees nothing, which is strictly worse than an error. Same up-front
|
|
1566
|
+
// gate pattern as vision above.
|
|
1567
|
+
const wantsTools = (tools?.length ?? 0) > 0;
|
|
1568
|
+
if (wantsTools && !hasEnabledToolsModel()) {
|
|
1569
|
+
res.status(422).json({
|
|
1570
|
+
error: {
|
|
1571
|
+
message: 'This request includes tools, but no tool-capable model is enabled. Enable a tool-calling model (e.g. GPT-OSS 120B, Gemini 3.5 Flash, GLM-4.7) in the Fallback Chain.',
|
|
1572
|
+
type: 'invalid_request_error',
|
|
1573
|
+
code: 'no_tools_model',
|
|
1574
|
+
},
|
|
1575
|
+
execution_id: requestGroupId,
|
|
1576
|
+
});
|
|
1577
|
+
return;
|
|
1578
|
+
}
|
|
1579
|
+
// Guardrail: per-request token budget (request_max_tokens_budget, default
|
|
1580
|
+
// off). Estimated input (incl. images) + requested output must fit the
|
|
1581
|
+
// ceiling; a request with no max_tokens gets its output capped to the
|
|
1582
|
+
// remainder instead. Sits before the Fusion branch so fan-out inherits the
|
|
1583
|
+
// capped max_tokens too.
|
|
1584
|
+
const budgetCheck = applyTokenBudget(estimatedInputTokens + imageCount * IMAGE_TOKEN_ESTIMATE, max_tokens);
|
|
1585
|
+
if (budgetCheck.rejection) {
|
|
1586
|
+
res.status(413).json({
|
|
1587
|
+
error: { message: tokenBudgetMessage(budgetCheck.rejection), type: 'invalid_request_error', code: 'request_token_budget' },
|
|
1588
|
+
execution_id: requestGroupId,
|
|
1589
|
+
});
|
|
1590
|
+
return;
|
|
1591
|
+
}
|
|
1592
|
+
max_tokens = budgetCheck.maxTokens;
|
|
1593
|
+
// Client-disconnect fan-out: the flag stops the loop before the NEXT
|
|
1594
|
+
// attempt; the AbortController (threaded to the provider as
|
|
1595
|
+
// CompletionOptions.signal) additionally cancels the IN-FLIGHT upstream
|
|
1596
|
+
// fetch and any body/stream read, so tokens stop burning and the in-flight
|
|
1597
|
+
// lease frees immediately. 'close' also fires on normal completion —
|
|
1598
|
+
// writableEnded distinguishes a real disconnect.
|
|
1599
|
+
//
|
|
1600
|
+
// This controller is declared before the Fusion branch because Fusion uses
|
|
1601
|
+
// the same cancellation signal as the ordinary fallback loop below.
|
|
1602
|
+
let clientGone = false;
|
|
1603
|
+
const clientAbort = new AbortController();
|
|
1604
|
+
res.on('close', () => {
|
|
1605
|
+
if (!res.writableEnded) {
|
|
1606
|
+
clientGone = true;
|
|
1607
|
+
clientAbort.abort(newClientAbortError());
|
|
1608
|
+
}
|
|
1609
|
+
});
|
|
1610
|
+
// ── Fusion: multi-model synthesis ──────────────────────────────────────────
|
|
1611
|
+
// The virtual "fusion" model fans the prompt out to a panel of diverse models
|
|
1612
|
+
// in parallel, then a judge synthesizes one answer. It routes each panel/judge
|
|
1613
|
+
// sub-call through the normal path (cooldowns, quotas, analytics), so it
|
|
1614
|
+
// behaves like a normal model from the client's side — just K+1x the tokens.
|
|
1615
|
+
// Image requests run on vision-capable panel members; tool requests run on
|
|
1616
|
+
// tool-capable members and return the first structured tool call directly.
|
|
1617
|
+
if (isFusionModel(requestedModel)) {
|
|
1618
|
+
// Keep Fusion's sequential tool fallback tied to the caller's socket. The
|
|
1619
|
+
// Responses route already threads this signal; the legacy Chat surface
|
|
1620
|
+
// needs the same cancellation so a disconnected client does not leave a
|
|
1621
|
+
// bounded-but-expensive provider call running in the background.
|
|
1622
|
+
const fusionOptions = { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, ...samplingParams, signal: clientAbort.signal };
|
|
1623
|
+
const fusionConfig = parsed.data.fusion ?? {};
|
|
1624
|
+
if (stream) {
|
|
1625
|
+
// Streaming fusion: open the SSE response immediately and emit additive
|
|
1626
|
+
// `_fusion` frames (no `choices`, so standard OpenAI clients skip them) as
|
|
1627
|
+
// each panel model settles and when the judge runs — the Playground shows
|
|
1628
|
+
// these arriving in a collapsible trace. The final synthesized answer is
|
|
1629
|
+
// then streamed as normal content deltas, so plain clients still get it.
|
|
1630
|
+
res.setHeader('Content-Type', 'text/event-stream');
|
|
1631
|
+
res.setHeader('Cache-Control', 'no-cache');
|
|
1632
|
+
res.setHeader('Connection', 'keep-alive');
|
|
1633
|
+
const writeFrame = (o) => { try {
|
|
1634
|
+
res.write(`data: ${JSON.stringify(o)}\n\n`);
|
|
1635
|
+
}
|
|
1636
|
+
catch { /* socket gone */ } };
|
|
1637
|
+
const streamId = `fusion-${Date.now()}-${crypto.randomBytes(3).toString('hex')}`;
|
|
1638
|
+
const base = { id: streamId, object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: FUSION_MODEL_ID };
|
|
1639
|
+
// Track whether the judge already streamed content so we don't re-emit it.
|
|
1640
|
+
let answerStarted = false;
|
|
1641
|
+
try {
|
|
1642
|
+
const { response } = await runFusion({
|
|
1643
|
+
messages,
|
|
1644
|
+
config: fusionConfig,
|
|
1645
|
+
options: fusionOptions,
|
|
1646
|
+
estimatedTokens: estimatedTotal,
|
|
1647
|
+
vision: hasImage,
|
|
1648
|
+
hooks: {
|
|
1649
|
+
// `a` already carries a sanitized error for failed slots; content is
|
|
1650
|
+
// the model's own answer and is forwarded as-is.
|
|
1651
|
+
onPanel: (a) => writeFrame({
|
|
1652
|
+
...base,
|
|
1653
|
+
choices: [{ index: 0, delta: {}, finish_reason: null }],
|
|
1654
|
+
_fusion: { event: 'panel', ...a },
|
|
1655
|
+
}),
|
|
1656
|
+
onJudge: (j) => writeFrame({
|
|
1657
|
+
...base,
|
|
1658
|
+
choices: [{ index: 0, delta: {}, finish_reason: null }],
|
|
1659
|
+
_fusion: { event: 'judge', ...j },
|
|
1660
|
+
}),
|
|
1661
|
+
// Stream the judge's synthesis live as standard content deltas, so
|
|
1662
|
+
// the final answer appears as it's written instead of after the wait.
|
|
1663
|
+
onJudgeDelta: (delta) => {
|
|
1664
|
+
if (!answerStarted) {
|
|
1665
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
|
|
1666
|
+
answerStarted = true;
|
|
1667
|
+
}
|
|
1668
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { content: delta }, finish_reason: null }] });
|
|
1669
|
+
},
|
|
1670
|
+
},
|
|
1671
|
+
});
|
|
1672
|
+
// best_of / single-survivor / judge-fell-back-to-best-of never streamed
|
|
1673
|
+
// a delta — emit the final answer as one chunk in that case.
|
|
1674
|
+
const finalMsg = response.choices[0]?.message;
|
|
1675
|
+
const finalToolCalls = finalMsg?.tool_calls;
|
|
1676
|
+
const hasFinalToolCalls = Array.isArray(finalToolCalls) && finalToolCalls.length > 0;
|
|
1677
|
+
if (hasFinalToolCalls) {
|
|
1678
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
|
|
1679
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { tool_calls: finalToolCalls }, finish_reason: null }] });
|
|
1680
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }], usage: response.usage });
|
|
1681
|
+
}
|
|
1682
|
+
else {
|
|
1683
|
+
if (!answerStarted) {
|
|
1684
|
+
const finalText = contentToString(finalMsg?.content ?? '');
|
|
1685
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
|
|
1686
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: { content: finalText }, finish_reason: null }] });
|
|
1687
|
+
}
|
|
1688
|
+
writeFrame({ ...base, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], usage: response.usage });
|
|
1689
|
+
}
|
|
1690
|
+
}
|
|
1691
|
+
catch (err) {
|
|
1692
|
+
const message = err instanceof FusionError ? err.message : `fusion error: ${sanitizeProviderErrorMessage(err?.message)}`;
|
|
1693
|
+
const type = err instanceof FusionError
|
|
1694
|
+
? (err.status === 429 ? 'rate_limit_error' : err.status >= 500 ? 'server_error' : 'invalid_request_error')
|
|
1695
|
+
: 'server_error';
|
|
1696
|
+
writeFrame({ error: { message, type } });
|
|
1697
|
+
}
|
|
1698
|
+
try {
|
|
1699
|
+
res.write('data: [DONE]\n\n');
|
|
1700
|
+
res.end();
|
|
1701
|
+
}
|
|
1702
|
+
catch { /* socket gone */ }
|
|
1703
|
+
return;
|
|
1704
|
+
}
|
|
1705
|
+
try {
|
|
1706
|
+
const { response, routedVia } = await runFusion({
|
|
1707
|
+
messages,
|
|
1708
|
+
config: fusionConfig,
|
|
1709
|
+
options: fusionOptions,
|
|
1710
|
+
estimatedTokens: estimatedTotal,
|
|
1711
|
+
vision: hasImage,
|
|
1712
|
+
});
|
|
1713
|
+
// Structured-output enforcement for fusion (#516 scope gap): the panel/
|
|
1714
|
+
// judge output got no format check, so model:"fusion" could hand back
|
|
1715
|
+
// prose as a "success" for a json_schema request. Fusion has no failover
|
|
1716
|
+
// machinery to hand this to — heal what's healable, otherwise answer
|
|
1717
|
+
// honestly instead of pretending. (Streaming fusion stays unenforced,
|
|
1718
|
+
// same boundary as every other streamed response.)
|
|
1719
|
+
const fusionMsg = response?.choices?.[0]?.message;
|
|
1720
|
+
if (samplingParams.response_format && fusionMsg && !fusionMsg.tool_calls?.length) {
|
|
1721
|
+
const fusionText = contentToString(fusionMsg.content ?? '');
|
|
1722
|
+
if (fusionText) {
|
|
1723
|
+
const enforced = enforceJsonContent(fusionText);
|
|
1724
|
+
if (!enforced.ok) {
|
|
1725
|
+
res.status(502).json({ error: { message: `fusion produced non-JSON output despite response_format=${samplingParams.response_format.type} — retry, or pin a structured-output-capable model instead of "fusion"`, type: 'server_error' }, execution_id: requestGroupId });
|
|
1726
|
+
return;
|
|
1727
|
+
}
|
|
1728
|
+
if (enforced.healed)
|
|
1729
|
+
fusionMsg.content = enforced.content;
|
|
1730
|
+
}
|
|
1731
|
+
}
|
|
1732
|
+
res.setHeader('X-Routed-Via', safeHeaderValue(routedVia));
|
|
1733
|
+
res.json({ ...response, execution_id: requestGroupId });
|
|
1734
|
+
}
|
|
1735
|
+
catch (err) {
|
|
1736
|
+
if (err instanceof FusionError) {
|
|
1737
|
+
res.status(err.status).json({
|
|
1738
|
+
error: {
|
|
1739
|
+
message: err.message,
|
|
1740
|
+
type: err.status === 429 ? 'rate_limit_error' : err.status >= 500 ? 'server_error' : 'invalid_request_error',
|
|
1741
|
+
},
|
|
1742
|
+
execution_id: requestGroupId,
|
|
1743
|
+
});
|
|
1744
|
+
}
|
|
1745
|
+
else {
|
|
1746
|
+
res.status(502).json({ error: { message: `fusion error: ${sanitizeProviderErrorMessage(err?.message)}`, type: 'server_error' }, execution_id: requestGroupId });
|
|
1747
|
+
}
|
|
1748
|
+
}
|
|
1749
|
+
return;
|
|
1750
|
+
}
|
|
1751
|
+
// ── Response cache (services/cache.ts) ──
|
|
1752
|
+
// Opt-in exact-match cache. An identical earlier request is replayed from an
|
|
1753
|
+
// in-memory LRU without spending any provider quota. Computed here, after
|
|
1754
|
+
// message + sampling-param normalization but before any routing/session work,
|
|
1755
|
+
// so a hit short-circuits the whole pipeline. Only NON-streaming requests at a
|
|
1756
|
+
// cacheable temperature are eligible (v1 scope: streaming always bypasses); a
|
|
1757
|
+
// per-request `X-FreeLLM-Cache` header can force or bypass. Off unless enabled
|
|
1758
|
+
// via the RESPONSE_CACHE env var or the response_cache_enabled setting.
|
|
1759
|
+
const cacheDirective = parseCacheDirective(req.headers['x-freellm-cache'], req.headers['cache-control']);
|
|
1760
|
+
// Streaming requests participate in the cache too: same canonical key (the
|
|
1761
|
+
// request content is identical), but hits are looked up in the streaming
|
|
1762
|
+
// store and replayed as SSE rather than JSON (see below).
|
|
1763
|
+
const cacheKey = (cacheActive(cacheDirective) && isCacheableTemperature(temperature))
|
|
1764
|
+
? computeCacheKey({
|
|
1765
|
+
model: requestedModel, messages, temperature, top_p, max_tokens, tools, tool_choice,
|
|
1766
|
+
// Normalized stop (providerSafeStop), i.e. what is actually forwarded.
|
|
1767
|
+
stop,
|
|
1768
|
+
// The knobs below are NOT in chatCompletionSchema, so zod strips them
|
|
1769
|
+
// from parsed.data; read them from the raw body. They still change what
|
|
1770
|
+
// answer the client is asking for, so requests differing only in one of
|
|
1771
|
+
// them must never collide on a cached entry. Explicit null is coerced
|
|
1772
|
+
// to undefined (dropped from the key) to match how the proxy treats
|
|
1773
|
+
// null-valued optional knobs as absent.
|
|
1774
|
+
response_format: req.body?.response_format ?? undefined,
|
|
1775
|
+
n: req.body?.n ?? undefined,
|
|
1776
|
+
seed: req.body?.seed ?? undefined,
|
|
1777
|
+
presence_penalty: req.body?.presence_penalty ?? undefined,
|
|
1778
|
+
frequency_penalty: req.body?.frequency_penalty ?? undefined,
|
|
1779
|
+
logit_bias: req.body?.logit_bias ?? undefined,
|
|
1780
|
+
logprobs: req.body?.logprobs ?? undefined,
|
|
1781
|
+
top_logprobs: req.body?.top_logprobs ?? undefined,
|
|
1782
|
+
// Normalized reasoning knob (flat field or object form) — a different
|
|
1783
|
+
// effort asks for a different answer, so it must never collide.
|
|
1784
|
+
reasoning_effort: samplingParams.reasoning_effort ?? undefined,
|
|
1785
|
+
compression: compressionResult.cacheKey,
|
|
1786
|
+
})
|
|
1787
|
+
: null;
|
|
1788
|
+
if (cacheKey) {
|
|
1789
|
+
if (stream) {
|
|
1790
|
+
// Streaming hit: replay the captured SSE frame sequence verbatim —
|
|
1791
|
+
// same zero-quota rationale as a JSON hit, with first-byte semantics
|
|
1792
|
+
// preserved (the frames include the leading content chunks, so the
|
|
1793
|
+
// first token arrives immediately).
|
|
1794
|
+
const streamHit = getCachedStreamResponse(cacheKey);
|
|
1795
|
+
if (streamHit) {
|
|
1796
|
+
res.setHeader('Content-Type', 'text/event-stream');
|
|
1797
|
+
res.setHeader('Cache-Control', 'no-cache');
|
|
1798
|
+
res.setHeader('Connection', 'keep-alive');
|
|
1799
|
+
res.setHeader('X-Routed-Via', 'cache');
|
|
1800
|
+
res.setHeader('X-FreeLLM-Cache', 'HIT');
|
|
1801
|
+
res.write(streamHit.sse);
|
|
1802
|
+
res.end();
|
|
1803
|
+
return;
|
|
1804
|
+
}
|
|
1805
|
+
}
|
|
1806
|
+
else {
|
|
1807
|
+
const hit = getCachedResponse(cacheKey);
|
|
1808
|
+
if (hit) {
|
|
1809
|
+
// A hit consumes NO provider quota, so recordRequest/recordTokens are
|
|
1810
|
+
// deliberately skipped and the reply is not re-logged as provider usage.
|
|
1811
|
+
// The savings are reported separately by GET /api/cache/stats.
|
|
1812
|
+
res.setHeader('X-Routed-Via', 'cache');
|
|
1813
|
+
res.setHeader('X-FreeLLM-Cache', 'HIT');
|
|
1814
|
+
res.json(withExecutionId(hit.body, requestGroupId));
|
|
1815
|
+
return;
|
|
1816
|
+
}
|
|
1817
|
+
}
|
|
1818
|
+
}
|
|
1819
|
+
// ── Idempotency-Key (services/idempotency.ts) ──
|
|
1820
|
+
// Optional caller-scoped dedup for NON-streaming requests: a client that
|
|
1821
|
+
// times out and retries with the same Idempotency-Key gets the ORIGINAL
|
|
1822
|
+
// response replayed (zero provider cost) instead of burning a second
|
|
1823
|
+
// free-tier slot. Only a SHA-256 hash of the key is stored. Reusing a key
|
|
1824
|
+
// with different request content is a 409 conflict. Streaming always
|
|
1825
|
+
// bypasses (like the response cache) — a stream cannot be replayed as a
|
|
1826
|
+
// unit, and the open connection is itself the retry signal.
|
|
1827
|
+
const idemKeyRaw = req.headers['idempotency-key'] ?? req.headers['Idempotency-Key'];
|
|
1828
|
+
const idemKey = !stream ? normalizeIdempotencyKey(idemKeyRaw) : null;
|
|
1829
|
+
const idemFingerprint = idemKey
|
|
1830
|
+
? computeIdempotencyFingerprint({
|
|
1831
|
+
model: requestedModel,
|
|
1832
|
+
messages,
|
|
1833
|
+
temperature,
|
|
1834
|
+
top_p,
|
|
1835
|
+
max_tokens,
|
|
1836
|
+
tools,
|
|
1837
|
+
tool_choice,
|
|
1838
|
+
})
|
|
1839
|
+
: null;
|
|
1840
|
+
if (idemKey && idemFingerprint) {
|
|
1841
|
+
const keyHash = hashIdempotencyKey(idemKey);
|
|
1842
|
+
const claim = lookupIdempotencyReplay(keyHash, idemFingerprint);
|
|
1843
|
+
if (claim.kind === 'replay') {
|
|
1844
|
+
// Replay consumes NO provider quota — same zero-cost rationale as a
|
|
1845
|
+
// cache hit, so request/usage bookkeeping is skipped here too.
|
|
1846
|
+
res.setHeader('X-Routed-Via', 'idempotency');
|
|
1847
|
+
res.status(claim.status).json(withExecutionId(claim.body, requestGroupId));
|
|
1848
|
+
return;
|
|
1849
|
+
}
|
|
1850
|
+
if (claim.kind === 'conflict') {
|
|
1851
|
+
res.status(409).json({
|
|
1852
|
+
error: {
|
|
1853
|
+
message: 'idempotency_key_conflict',
|
|
1854
|
+
type: 'invalid_request_error',
|
|
1855
|
+
},
|
|
1856
|
+
execution_id: requestGroupId,
|
|
1857
|
+
});
|
|
1858
|
+
return;
|
|
1859
|
+
}
|
|
1860
|
+
// kind === 'miss': no prior claim (or it expired) — proceed normally
|
|
1861
|
+
// and persist the result on success below.
|
|
1862
|
+
}
|
|
1863
|
+
// Optional client-managed session affinity (see getSessionKey). Express
|
|
1864
|
+
// lower-cases header names; a repeated header arrives as an array — take
|
|
1865
|
+
// the first value.
|
|
1866
|
+
const rawSessionId = req.headers['x-session-id'];
|
|
1867
|
+
const sessionIdHeader = Array.isArray(rawSessionId) ? rawSessionId[0] : rawSessionId;
|
|
1868
|
+
let resolvedChain;
|
|
1869
|
+
let strategyKey;
|
|
1870
|
+
if (isAutoModel(requestedModel)) {
|
|
1871
|
+
resolvedChain = resolveRoutingChain(requestedModel);
|
|
1872
|
+
strategyKey = resolvedChain.strategyKey;
|
|
1873
|
+
}
|
|
1874
|
+
// Context handoff only applies to auto-routed requests. Pinned-model requests
|
|
1875
|
+
// are deliberate client choices; injecting "you are taking over" there would
|
|
1876
|
+
// be semantically wrong.
|
|
1877
|
+
const isAutoRouted = !requestedModel || isAutoModel(requestedModel);
|
|
1878
|
+
const handoffMode = isAutoRouted ? getContextHandoffMode() : 'off';
|
|
1879
|
+
const sessionKey = handoffMode !== 'off' ? getSessionKey(messages, sessionIdHeader, strategyKey) : '';
|
|
1880
|
+
if (handoffMode !== 'off' && sessionKey) {
|
|
1881
|
+
recordIncomingMessages(sessionKey, messages);
|
|
1882
|
+
}
|
|
1883
|
+
// #797: key for the per-session thinking-trace memory. Read and written
|
|
1884
|
+
// inside the dispatch loop, where the routed platform/model is known — the
|
|
1885
|
+
// restore only ever touches the OUTBOUND copy, never `messages`, so handoff
|
|
1886
|
+
// recording above, request logging, compression and the response cache all
|
|
1887
|
+
// keep seeing exactly what the client sent.
|
|
1888
|
+
//
|
|
1889
|
+
// Header-less clients are covered: without x-session-id, getSessionKey hashes
|
|
1890
|
+
// the FIRST user message, which does not change as the conversation grows —
|
|
1891
|
+
// the same stability sticky sessions already rely on. Its known weakness is
|
|
1892
|
+
// shared here: two conversations opening with identical text share a key, so
|
|
1893
|
+
// the same-model + field-actually-missing gates below are what keep a
|
|
1894
|
+
// mis-keyed restore from reaching a payload it does not belong in.
|
|
1895
|
+
const reasoningSessionKey = getSessionKey(messages, sessionIdHeader, strategyKey);
|
|
1896
|
+
// A handoff can only fire when a prior model is on record for this session.
|
|
1897
|
+
// Check after recordIncomingMessages, which clears the prior model on a
|
|
1898
|
+
// fresh conversation. Stable across the retry loop (the prior model only
|
|
1899
|
+
// changes on a success, which returns), so compute it once here.
|
|
1900
|
+
const handoffPossible = handoffMode !== 'off' && !!sessionKey && hasPriorModel(sessionKey);
|
|
1901
|
+
// Explicit `model` field pins routing. If the catalog has no enabled row
|
|
1902
|
+
// matching the requested id, return 400 — silently auto-routing to a
|
|
1903
|
+
// different model would be surprising to OpenAI-compatible clients.
|
|
1904
|
+
// Sticky-session is the fallback when no `model` field was sent at all.
|
|
1905
|
+
let preferredModel;
|
|
1906
|
+
// When the pinned model is a unified group, this holds the group's ordered
|
|
1907
|
+
// members and is passed to routeRequest as the STRICT chain (no other model
|
|
1908
|
+
// is ever reached). Undefined for auto and legacy single-row pins.
|
|
1909
|
+
let groupChain;
|
|
1910
|
+
// Sticky scope: auto requests bucket by routing strategy; a unified group pin
|
|
1911
|
+
// buckets by the canonical id the client sent, so the group prefers its last
|
|
1912
|
+
// successful provider without leaking stickiness across groups.
|
|
1913
|
+
let stickyStrategyKey = strategyKey;
|
|
1914
|
+
if (isAutoModel(requestedModel)) {
|
|
1915
|
+
preferredModel = resolveStickyPreference(getStickyModel(messages, sessionIdHeader, strategyKey), resolvedChain?.chain);
|
|
1916
|
+
}
|
|
1917
|
+
else if (requestedModel) {
|
|
1918
|
+
const db = getDb();
|
|
1919
|
+
// Unify ON: a requested id (canonical slug OR any provider's model_id) maps
|
|
1920
|
+
// to the whole logical-model group, and we route STRICTLY across only its
|
|
1921
|
+
// providers — failing over between them, never to a different model (#335).
|
|
1922
|
+
const resolved = isUnifyEnabled() ? resolveRequestedIdForDispatch(requestedModel, getModelGroups()) : null;
|
|
1923
|
+
const members = resolved?.memberDbIds ?? null;
|
|
1924
|
+
if (members && members.length > 0) {
|
|
1925
|
+
groupChain = resolveModelGroupCandidates(members, resolved.demotedDbIds);
|
|
1926
|
+
if (groupChain.length === 0) {
|
|
1927
|
+
// Distinguish a catalog-disabled model (404 model_not_found, OpenAI
|
|
1928
|
+
// semantics) from one whose providers are present but unusable
|
|
1929
|
+
// (chain-disabled / no key) — the latter is a server-side
|
|
1930
|
+
// configuration gap, so it renders an honest 503.
|
|
1931
|
+
const placeholders = members.map(() => '?').join(',');
|
|
1932
|
+
const anyEnabled = db.prepare(`SELECT 1 FROM models WHERE id IN (${placeholders}) AND enabled = 1 LIMIT 1`).get(...members);
|
|
1933
|
+
if (anyEnabled) {
|
|
1934
|
+
res.status(503).json({
|
|
1935
|
+
error: {
|
|
1936
|
+
message: `Model '${requestedModel}' has no providers with an enabled key. Add a provider API key for it, use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1937
|
+
type: 'service_unavailable',
|
|
1938
|
+
code: 'no_providers_configured',
|
|
1939
|
+
},
|
|
1940
|
+
execution_id: requestGroupId,
|
|
1941
|
+
});
|
|
1942
|
+
}
|
|
1943
|
+
else {
|
|
1944
|
+
res.status(404).json({
|
|
1945
|
+
error: {
|
|
1946
|
+
message: `Model '${requestedModel}' is disabled. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1947
|
+
type: 'invalid_request_error',
|
|
1948
|
+
code: 'model_not_found',
|
|
1949
|
+
},
|
|
1950
|
+
execution_id: requestGroupId,
|
|
1951
|
+
});
|
|
1952
|
+
}
|
|
1953
|
+
return;
|
|
1954
|
+
}
|
|
1955
|
+
stickyStrategyKey = requestedModel;
|
|
1956
|
+
const sticky = getStickyModel(messages, sessionIdHeader, stickyStrategyKey);
|
|
1957
|
+
// Only prefer the sticky member if it's actually IN this group — passing a
|
|
1958
|
+
// non-member as preferredModelDbId would make routeRequest inject an
|
|
1959
|
+
// off-group model and break strict pinning.
|
|
1960
|
+
preferredModel = (sticky != null && groupChain.some(r => r.model_db_id === sticky)) ? sticky : undefined;
|
|
1961
|
+
}
|
|
1962
|
+
else {
|
|
1963
|
+
// Unify OFF, or an id that isn't in the catalog: legacy single-row pin.
|
|
1964
|
+
const enabled = db.prepare('SELECT id FROM models WHERE model_id = ? AND enabled = 1').get(requestedModel);
|
|
1965
|
+
if (enabled) {
|
|
1966
|
+
preferredModel = enabled.id;
|
|
1967
|
+
}
|
|
1968
|
+
else {
|
|
1969
|
+
const disabled = db.prepare('SELECT id FROM models WHERE model_id = ?').get(requestedModel);
|
|
1970
|
+
const reason = disabled ? 'is disabled' : 'is not in the catalog';
|
|
1971
|
+
res.status(404).json({
|
|
1972
|
+
error: {
|
|
1973
|
+
message: `Model '${requestedModel}' ${reason}. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
|
|
1974
|
+
type: 'invalid_request_error',
|
|
1975
|
+
code: 'model_not_found',
|
|
1976
|
+
},
|
|
1977
|
+
execution_id: requestGroupId,
|
|
1978
|
+
});
|
|
1979
|
+
return;
|
|
1980
|
+
}
|
|
1981
|
+
}
|
|
1982
|
+
}
|
|
1983
|
+
else {
|
|
1984
|
+
preferredModel = resolveStickyPreference(getStickyModel(messages, sessionIdHeader, strategyKey), resolvedChain?.chain);
|
|
1985
|
+
}
|
|
1986
|
+
// For analytics: the model id the client pinned, null when auto-routed
|
|
1987
|
+
// ('auto' or omitted). Logged with every request row so pinned vs auto
|
|
1988
|
+
// traffic and failover overrides are visible.
|
|
1989
|
+
const pinnedModelId = requestedModel && !isAutoModel(requestedModel) ? requestedModel : null;
|
|
1990
|
+
// Retry loop: on 429/rate limit, skip that model+key and try the next one.
|
|
1991
|
+
// The attempt iteration, cooldown/skip/penalty bookkeeping, and exhaustion
|
|
1992
|
+
// rendering are the shared fallback loop (lib/fallback-loop.ts). What stays
|
|
1993
|
+
// here is /chat/completions-specific: the response-cache MISS store, the
|
|
1994
|
+
// context-handoff injection, group/unified-chain routing, and the OpenAI
|
|
1995
|
+
// stream turn-integrity framing.
|
|
1996
|
+
const state = newFallbackState();
|
|
1997
|
+
// Lets the failover loop learn which models reject tool calls (#1230).
|
|
1998
|
+
state.wantsTools = wantsTools;
|
|
1999
|
+
const attemptLog = [];
|
|
2000
|
+
// Fallback-v2 hedging: the loop aborts this controller (via abortInFlight)
|
|
2001
|
+
// when the wall-clock retry budget expires mid-attempt, canceling the
|
|
2002
|
+
// in-flight upstream instead of waiting for a stalled attempt to time out.
|
|
2003
|
+
const hedgeAbort = new AbortController();
|
|
2004
|
+
await runFallbackLoop({
|
|
2005
|
+
maxRetries: MAX_RETRIES,
|
|
2006
|
+
state,
|
|
2007
|
+
attemptLog,
|
|
2008
|
+
logIdentity: { surface: 'chat completions', requestId: requestGroupId, requestedModel: requestedModelLabel },
|
|
2009
|
+
clientGone: () => clientGone,
|
|
2010
|
+
abortInFlight: () => hedgeAbort.abort(newHedgeAbortError()),
|
|
2011
|
+
route: () => {
|
|
2012
|
+
// When a handoff could fire this turn, pad the token estimate so the router's
|
|
2013
|
+
// context-window and TPM checks account for the extra system message overhead.
|
|
2014
|
+
// We don't know the selected model key until after routeRequest() returns, so
|
|
2015
|
+
// the padding is conservative on turns where injection is *possible* (a prior
|
|
2016
|
+
// model is on record). Turns where injection can't happen — every turn 1, and
|
|
2017
|
+
// sessions that never switched — pay no headroom tax.
|
|
2018
|
+
const routingEstimate = fallbackRoutingTokens(state, estimatedTotal, outputReserve) + (handoffPossible ? HANDOFF_MAX_TOKENS : 0);
|
|
2019
|
+
// Task-type routing (#1127): the client can declare code/chat intent via
|
|
2020
|
+
// header; otherwise a bounded rule derives it (tools present / code
|
|
2021
|
+
// markers). undefined keeps the preset weights untouched.
|
|
2022
|
+
const taskType = resolveTaskType(req, tools, messages);
|
|
2023
|
+
return routeRequest(routingEstimate, state.skipKeys.size > 0 ? state.skipKeys : undefined, preferredModel, hasImage, wantsTools, state.skipModels.size > 0 ? state.skipModels : undefined, groupChain ?? resolvedChain?.chain, samplingParams.response_format !== undefined, state.skipPlatforms.size > 0 ? state.skipPlatforms : undefined, outputReserve, taskType);
|
|
2024
|
+
},
|
|
2025
|
+
dispatch: async (route, attempt, ctx) => {
|
|
2026
|
+
const contextBudget = routeOutputBudget(route, estimatedInputTokens);
|
|
2027
|
+
const modelKey = `${route.platform}:${route.modelId}`;
|
|
2028
|
+
traceRouteEvent('Proxy', {
|
|
2029
|
+
event: attempt === 0 ? 'start' : 'next',
|
|
2030
|
+
requestId: requestGroupId,
|
|
2031
|
+
attempt,
|
|
2032
|
+
platform: route.platform,
|
|
2033
|
+
model: route.modelId,
|
|
2034
|
+
requestedModel: attempt === 0 ? requestedModelLabel : undefined,
|
|
2035
|
+
});
|
|
2036
|
+
let outboundMessages = messages;
|
|
2037
|
+
// #797: thinking trace accumulated from this turn's streamed deltas, then
|
|
2038
|
+
// remembered per-session so a follow-up whose client stripped the field
|
|
2039
|
+
// can have it restored (see restore block above).
|
|
2040
|
+
let streamReasoning = '';
|
|
2041
|
+
// Extra input tokens the injected handoff adds on this turn (0 when not
|
|
2042
|
+
// injected). Folded into the streaming success accounting, where token
|
|
2043
|
+
// counts are estimated; the non-stream path uses the provider's usage,
|
|
2044
|
+
// which already counts the injected message.
|
|
2045
|
+
let injectedHandoffTokens = 0;
|
|
2046
|
+
if (handoffMode !== 'off' && sessionKey) {
|
|
2047
|
+
const handoff = maybeInjectContextHandoff({ mode: handoffMode, sessionKey, messages, selectedModelKey: modelKey });
|
|
2048
|
+
if (handoff.injected)
|
|
2049
|
+
console.log(`[Proxy] Context handoff injected (session ${sessionKey.slice(0, 8)}…, model switch detected)`);
|
|
2050
|
+
outboundMessages = handoff.messages;
|
|
2051
|
+
injectedHandoffTokens = handoff.injectedTokens;
|
|
2052
|
+
}
|
|
2053
|
+
// #797: restore the thinking trace this proxy emitted last turn, for THIS
|
|
2054
|
+
// model, when the client dropped it on replay. Scoped to the outbound copy
|
|
2055
|
+
// and to the model that produced the trace, so a failover hop and every
|
|
2056
|
+
// provider that never needed the field send the client's bytes unchanged.
|
|
2057
|
+
const rememberedReasoning = rememberedReasoningFor(reasoningSessionKey, modelKey);
|
|
2058
|
+
if (rememberedReasoning) {
|
|
2059
|
+
outboundMessages = restoreSessionReasoning(outboundMessages, rememberedReasoning, route.platform);
|
|
2060
|
+
}
|
|
2061
|
+
// GitHub Models 413s a history above its input ceiling instead of
|
|
2062
|
+
// truncating it, so a long conversation burns the github hop of every
|
|
2063
|
+
// chain it appears in. Trim the outbound copy to what the platform will
|
|
2064
|
+
// accept — scoped to this attempt, so the next candidate still sees the
|
|
2065
|
+
// client's full history. A no-op returning the same array when the
|
|
2066
|
+
// request already fits.
|
|
2067
|
+
if (route.platform === 'github') {
|
|
2068
|
+
outboundMessages = truncateMessagesForGithub(outboundMessages);
|
|
2069
|
+
}
|
|
2070
|
+
if (stream) {
|
|
2071
|
+
// — Stream turn-integrity (#231 audit) —
|
|
2072
|
+
// The old loop forwarded upstream chunks verbatim and called any
|
|
2073
|
+
// stream that produced bytes a success. Live failure modes that
|
|
2074
|
+
// slipped through: in-band `{"error":...}` frames delivered as dead
|
|
2075
|
+
// turns, tool calls with no terminal finish_reason, inline tool-call
|
|
2076
|
+
// dialect emitted as text, truncations logged as success. This loop
|
|
2077
|
+
// validates the TURN, not the transport:
|
|
2078
|
+
// - headers are held until the first real payload, so anything that
|
|
2079
|
+
// dies before producing one fails over invisibly;
|
|
2080
|
+
// - text that starts with an inline tool-call dialect marker is held
|
|
2081
|
+
// and rescued into structured tool_calls (or failed over);
|
|
2082
|
+
// - tool_call deltas are buffered, argument-repaired, and emitted as
|
|
2083
|
+
// one complete chunk, always followed by finish_reason
|
|
2084
|
+
// "tool_calls" — agents never see calls without a terminal reason;
|
|
2085
|
+
// - a stream that ends with neither content nor calls is an empty
|
|
2086
|
+
// completion and fails over like the non-stream path.
|
|
2087
|
+
let totalOutputTokens = 0;
|
|
2088
|
+
let headerSent = false;
|
|
2089
|
+
let ttfbMs = null;
|
|
2090
|
+
// Hold-window state: 'undecided' until the first text either matches
|
|
2091
|
+
// a dialect marker (→ 'dialect': buffer everything, rescue at end),
|
|
2092
|
+
// carries a structured-output request (→ 'json': buffer everything,
|
|
2093
|
+
// enforce JSON at end) or provably cannot (→ 'passthrough': flush and
|
|
2094
|
+
// stream normally).
|
|
2095
|
+
let mode = 'undecided';
|
|
2096
|
+
let heldText = '';
|
|
2097
|
+
const preamble = []; // role-only chunks held until flush
|
|
2098
|
+
const toolCallAcc = new Map();
|
|
2099
|
+
let upstreamFinish = null;
|
|
2100
|
+
let usageChunk = null;
|
|
2101
|
+
let lastMeta = {};
|
|
2102
|
+
// Raw upstream-reported model, captured off the first frame that
|
|
2103
|
+
// carries one — BEFORE the per-frame overwrite below destroys it.
|
|
2104
|
+
// Only evidence when a provider serves a different model than routed
|
|
2105
|
+
// (#534); compared/persisted on success via observeServedModel.
|
|
2106
|
+
let upstreamModel = null;
|
|
2107
|
+
// Every `data: ...` frame the client sees, captured for a possible
|
|
2108
|
+
// streaming cache store on success (exact SSE replay on a later hit).
|
|
2109
|
+
// Collected ONLY when this request is actually cacheable: with the
|
|
2110
|
+
// cache off — the default — a stream must retain nothing, so the
|
|
2111
|
+
// buffer stays null and every frame is written straight through.
|
|
2112
|
+
const streamFrames = cacheKey ? [] : null;
|
|
2113
|
+
let streamFrameBytes = 0;
|
|
2114
|
+
// Flipped off once the answer outgrows what is worth holding; the
|
|
2115
|
+
// buffer is dropped and this response is not stored.
|
|
2116
|
+
let streamCacheable = cacheKey !== null;
|
|
2117
|
+
const collectFrame = (frame) => {
|
|
2118
|
+
if (!streamCacheable || !streamFrames)
|
|
2119
|
+
return;
|
|
2120
|
+
streamFrameBytes += Buffer.byteLength(frame);
|
|
2121
|
+
if (streamFrameBytes > STREAM_CACHE_MAX_BYTES) {
|
|
2122
|
+
streamCacheable = false;
|
|
2123
|
+
streamFrames.length = 0;
|
|
2124
|
+
return;
|
|
2125
|
+
}
|
|
2126
|
+
streamFrames.push(frame);
|
|
2127
|
+
};
|
|
2128
|
+
const flushHeaders = () => {
|
|
2129
|
+
if (headerSent)
|
|
2130
|
+
return;
|
|
2131
|
+
// #764: backfill only — the pump loop already records ttfb on the
|
|
2132
|
+
// first token (content or reasoning) it sees.
|
|
2133
|
+
if (ttfbMs === null)
|
|
2134
|
+
ttfbMs = Date.now() - start;
|
|
2135
|
+
res.setHeader('Content-Type', 'text/event-stream');
|
|
2136
|
+
res.setHeader('Cache-Control', 'no-cache');
|
|
2137
|
+
res.setHeader('Connection', 'keep-alive');
|
|
2138
|
+
res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
|
|
2139
|
+
res.setHeader('X-FreeLLM-Cache', cacheKey ? 'MISS' : 'OFF');
|
|
2140
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
2141
|
+
headerSent = true;
|
|
2142
|
+
// Committed: the answer is on its way, so the retry budget must no
|
|
2143
|
+
// longer cancel this attempt (it could not fail over now anyway).
|
|
2144
|
+
ctx.disarmHedge();
|
|
2145
|
+
for (const p of preamble) {
|
|
2146
|
+
const frame = `data: ${JSON.stringify(p)}\n\n`;
|
|
2147
|
+
collectFrame(frame);
|
|
2148
|
+
res.write(frame);
|
|
2149
|
+
}
|
|
2150
|
+
preamble.length = 0;
|
|
2151
|
+
};
|
|
2152
|
+
const mkChunk = (delta, finish) => ({
|
|
2153
|
+
id: lastMeta.id ?? `chatcmpl-${Date.now()}`,
|
|
2154
|
+
object: 'chat.completion.chunk',
|
|
2155
|
+
created: lastMeta.created ?? Math.floor(Date.now() / 1000),
|
|
2156
|
+
model: lastMeta.model ?? route.modelId,
|
|
2157
|
+
choices: [{ index: 0, delta, finish_reason: finish }],
|
|
2158
|
+
});
|
|
2159
|
+
const writeChunk = (c) => {
|
|
2160
|
+
const frame = `data: ${JSON.stringify(c)}\n\n`;
|
|
2161
|
+
collectFrame(frame);
|
|
2162
|
+
res.write(frame);
|
|
2163
|
+
};
|
|
2164
|
+
try {
|
|
2165
|
+
const gen = route.provider.streamChatCompletion(route.apiKey, outboundMessages, route.modelId, { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, stream_options: parsed.data.stream_options, ...samplingParams, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
|
|
2166
|
+
for await (const chunk of gen) {
|
|
2167
|
+
if (clientGone)
|
|
2168
|
+
break; // client hung up: stop pulling; reader.cancel() aborts upstream
|
|
2169
|
+
// Provider metadata is not authoritative for the public gateway
|
|
2170
|
+
// response. Some OpenAI-compatible providers (notably Reka) return
|
|
2171
|
+
// the literal model name "default" even when a concrete model was
|
|
2172
|
+
// requested. Normalize every streamed frame at the proxy boundary
|
|
2173
|
+
// so clients consistently see the model that was actually routed.
|
|
2174
|
+
const rawChunkModel = chunk.model;
|
|
2175
|
+
if (upstreamModel == null && typeof rawChunkModel === 'string' && rawChunkModel.length > 0) {
|
|
2176
|
+
upstreamModel = rawChunkModel;
|
|
2177
|
+
}
|
|
2178
|
+
const anyChunk = { ...chunk, model: route.modelId };
|
|
2179
|
+
// In-band upstream error frame (observed live: Groq emits
|
|
2180
|
+
// {"error":{...,"code":"tool_use_failed"}} inside a 200 SSE
|
|
2181
|
+
// stream). Before headers: retryable, the next model gets the
|
|
2182
|
+
// request. After: surface an error frame instead of pretending
|
|
2183
|
+
// the turn succeeded.
|
|
2184
|
+
if (anyChunk.error && !anyChunk.choices) {
|
|
2185
|
+
const msg = anyChunk.error.message ?? JSON.stringify(anyChunk.error).slice(0, 200);
|
|
2186
|
+
if (!headerSent)
|
|
2187
|
+
throw new Error(`in-band provider error from ${route.displayName}: ${msg}`);
|
|
2188
|
+
console.error(`[Proxy] In-band error frame from ${route.displayName} mid-stream:`, msg);
|
|
2189
|
+
writeChunk({ error: { message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(String(msg))}`, type: 'stream_error' } });
|
|
2190
|
+
try {
|
|
2191
|
+
res.write('data: [DONE]\n\n');
|
|
2192
|
+
res.end();
|
|
2193
|
+
}
|
|
2194
|
+
catch { /* socket gone */ }
|
|
2195
|
+
traceRouteEvent('Proxy', {
|
|
2196
|
+
event: 'fail',
|
|
2197
|
+
requestId: requestGroupId,
|
|
2198
|
+
attempt,
|
|
2199
|
+
platform: route.platform,
|
|
2200
|
+
model: route.modelId,
|
|
2201
|
+
latencyMs: Date.now() - start,
|
|
2202
|
+
error: sanitizeProviderErrorMessage(String(msg)),
|
|
2203
|
+
});
|
|
2204
|
+
logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, `in-band error frame: ${sanitizeProviderErrorMessage(String(msg))}`, ttfbMs, pinnedModelId, null, 'http');
|
|
2205
|
+
return 'committed';
|
|
2206
|
+
}
|
|
2207
|
+
if (anyChunk.id)
|
|
2208
|
+
lastMeta = { id: anyChunk.id, model: anyChunk.model, created: anyChunk.created };
|
|
2209
|
+
// Usage arrives either on its own frame (OpenAI's
|
|
2210
|
+
// stream_options.include_usage shape) or bundled onto the last
|
|
2211
|
+
// choice-bearing frame — several providers do the latter, and
|
|
2212
|
+
// reading it only off choice-less frames threw their real token
|
|
2213
|
+
// counts away and left accounting on the chars/4 estimate. Capture
|
|
2214
|
+
// it wherever it lands, reduced to a usage-only frame: the frame's
|
|
2215
|
+
// deltas are re-emitted through our own framing below, so holding
|
|
2216
|
+
// the original verbatim would duplicate content (or a finish_reason
|
|
2217
|
+
// the client already saw) when it is written back after our finish
|
|
2218
|
+
// chunk to preserve OpenAI ordering.
|
|
2219
|
+
if (anyChunk.usage)
|
|
2220
|
+
usageChunk = { ...anyChunk, choices: [], usage: anyChunk.usage };
|
|
2221
|
+
const choice = anyChunk.choices?.[0];
|
|
2222
|
+
if (!choice)
|
|
2223
|
+
continue;
|
|
2224
|
+
if (choice.finish_reason)
|
|
2225
|
+
upstreamFinish = choice.finish_reason;
|
|
2226
|
+
// #797: accumulate this turn's thinking trace (native reasoning
|
|
2227
|
+
// deltas; the <think> extractor in base.ts has already normalized
|
|
2228
|
+
// inline tags into reasoning_content) so a follow-up request whose
|
|
2229
|
+
// client stripped it can have it restored from session memory.
|
|
2230
|
+
// Shared with the #764 ttfb/token accounting below.
|
|
2231
|
+
const reasoning = streamReasoningText(anyChunk);
|
|
2232
|
+
if (reasoning.length > 0)
|
|
2233
|
+
streamReasoning += reasoning;
|
|
2234
|
+
// Buffer tool_call deltas — emitted complete + repaired at end.
|
|
2235
|
+
for (const tc of choice.delta?.tool_calls ?? []) {
|
|
2236
|
+
const idx = tc.index ?? 0;
|
|
2237
|
+
if (!toolCallAcc.has(idx))
|
|
2238
|
+
toolCallAcc.set(idx, { id: undefined, name: '', args: '' });
|
|
2239
|
+
const acc = toolCallAcc.get(idx);
|
|
2240
|
+
if (tc.id && !acc.id)
|
|
2241
|
+
acc.id = tc.id;
|
|
2242
|
+
if (tc.function?.name)
|
|
2243
|
+
acc.name += tc.function.name;
|
|
2244
|
+
if (tc.function?.arguments)
|
|
2245
|
+
acc.args += tc.function.arguments;
|
|
2246
|
+
}
|
|
2247
|
+
normalizeOutboundContent(anyChunk);
|
|
2248
|
+
sanitizeResponse(anyChunk);
|
|
2249
|
+
const text = typeof choice.delta?.content === 'string' ? choice.delta.content : '';
|
|
2250
|
+
// #764: ttfb = first token of ANY kind, not just visible content —
|
|
2251
|
+
// reasoning models stream thinking long before the first answer
|
|
2252
|
+
// token, and the old code deferred ttfb until header flush (or left
|
|
2253
|
+
// it NULL on long-thinking turns that never flushed).
|
|
2254
|
+
if (ttfbMs === null && (text.length > 0 || reasoning.length > 0)) {
|
|
2255
|
+
ttfbMs = Date.now() - start;
|
|
2256
|
+
}
|
|
2257
|
+
if (text.length === 0) {
|
|
2258
|
+
// Role preamble / keep-alive: hold until first payload decides
|
|
2259
|
+
// the mode, forward afterwards. tool_calls and finish_reason are
|
|
2260
|
+
// stripped — both are re-emitted complete at the end (OpenRouter
|
|
2261
|
+
// attaches tool_call deltas to chunks that also carry role/
|
|
2262
|
+
// reasoning keys; forwarding them raw would duplicate the call).
|
|
2263
|
+
// #764: thinking-only chunks still consumed tokens — count them.
|
|
2264
|
+
if (reasoning.length > 0)
|
|
2265
|
+
totalOutputTokens += Math.ceil(reasoning.length / 4);
|
|
2266
|
+
if (choice.delta && Object.keys(choice.delta).some(k => k !== 'content' && k !== 'tool_calls' && choice.delta[k] != null)) {
|
|
2267
|
+
// `usage: undefined` (dropped by JSON.stringify): it was held
|
|
2268
|
+
// above and is re-emitted once, after our finish chunk.
|
|
2269
|
+
const cleaned = { ...anyChunk, usage: undefined, choices: [{ ...choice, delta: { ...choice.delta, tool_calls: undefined }, finish_reason: null }] };
|
|
2270
|
+
if (headerSent)
|
|
2271
|
+
writeChunk(cleaned);
|
|
2272
|
+
else
|
|
2273
|
+
preamble.push(cleaned);
|
|
2274
|
+
}
|
|
2275
|
+
continue;
|
|
2276
|
+
}
|
|
2277
|
+
// #764: count reasoning tokens with the same chars/4 estimate so
|
|
2278
|
+
// analytics and rate-limit reflect real consumption of thinking
|
|
2279
|
+
// models (a chunk can carry both reasoning and text).
|
|
2280
|
+
totalOutputTokens += Math.ceil((text.length + reasoning.length) / 4);
|
|
2281
|
+
if (mode === 'passthrough') {
|
|
2282
|
+
// Same rule as the preamble path: usage rides the held frame,
|
|
2283
|
+
// not this one, so the client sees it exactly once and last.
|
|
2284
|
+
writeChunk({ ...anyChunk, usage: undefined, choices: [{ ...choice, delta: { ...choice.delta, tool_calls: undefined }, finish_reason: null }] });
|
|
2285
|
+
continue;
|
|
2286
|
+
}
|
|
2287
|
+
heldText += text;
|
|
2288
|
+
if (mode === 'dialect' || mode === 'json')
|
|
2289
|
+
continue;
|
|
2290
|
+
const probe = heldText.trimStart();
|
|
2291
|
+
if (wantsTools && startsWithDialectMarker(probe)) {
|
|
2292
|
+
mode = 'dialect';
|
|
2293
|
+
}
|
|
2294
|
+
else if (samplingParams.response_format) {
|
|
2295
|
+
// Structured-output request (#933): hold ALL text until the
|
|
2296
|
+
// stream ends, then enforce JSON (mirrors the non-stream check
|
|
2297
|
+
// below). Streaming bytes are already committed once headers
|
|
2298
|
+
// flush, so a model that answers in prose despite the forwarded
|
|
2299
|
+
// response_format must be caught here, before any byte leaves —
|
|
2300
|
+
// the client asked for machine-readable output, not an essay.
|
|
2301
|
+
mode = 'json';
|
|
2302
|
+
}
|
|
2303
|
+
else if (!wantsTools || !couldBecomeDialectMarker(probe) || probe.length > 256) {
|
|
2304
|
+
mode = 'passthrough';
|
|
2305
|
+
flushHeaders();
|
|
2306
|
+
writeChunk(mkChunk({ content: heldText }, null));
|
|
2307
|
+
heldText = '';
|
|
2308
|
+
}
|
|
2309
|
+
// else: still a strict prefix of a marker — keep holding.
|
|
2310
|
+
}
|
|
2311
|
+
// — Stream ended cleanly (provider saw [DONE] or a finish_reason) —
|
|
2312
|
+
// Assemble buffered tool calls: synthesize missing ids, repair
|
|
2313
|
+
// double-encoded arguments against the request's schemas, drop
|
|
2314
|
+
// calls whose args still aren't valid JSON.
|
|
2315
|
+
const schemas = toolSchemaMap(tools);
|
|
2316
|
+
let syntheticStreamIds = 0;
|
|
2317
|
+
const completedCalls = [...toolCallAcc.entries()]
|
|
2318
|
+
.sort((a, b) => a[0] - b[0])
|
|
2319
|
+
.map(([, acc]) => ({
|
|
2320
|
+
id: acc.id && acc.id.length > 0 ? acc.id : `call_stream_${++syntheticStreamIds}`,
|
|
2321
|
+
type: 'function',
|
|
2322
|
+
function: { name: acc.name, arguments: repairToolArguments(acc.args || '{}', schemas.get(acc.name)) },
|
|
2323
|
+
}))
|
|
2324
|
+
.filter(c => { try {
|
|
2325
|
+
JSON.parse(c.function.arguments);
|
|
2326
|
+
return c.function.name.length > 0;
|
|
2327
|
+
}
|
|
2328
|
+
catch {
|
|
2329
|
+
return false;
|
|
2330
|
+
} });
|
|
2331
|
+
// Dialect rescue: the held text is an inline tool call in some
|
|
2332
|
+
// model's private syntax. Parse it into structured calls or treat
|
|
2333
|
+
// the turn as dead (headers were never sent in dialect mode, so
|
|
2334
|
+
// failing over is free).
|
|
2335
|
+
if (wantsTools && (mode === 'dialect' || (mode === 'undecided' && heldText.length > 0 && containsDialectMarker(heldText)))) {
|
|
2336
|
+
const rescue = rescueInlineToolCalls(heldText, new Set((tools ?? []).map(t => t.function.name)));
|
|
2337
|
+
if (rescue.detected) {
|
|
2338
|
+
if (!rescue.calls)
|
|
2339
|
+
throw new Error(`unparseable inline tool-call dialect from ${route.displayName}: ${heldText.slice(0, 120)}`);
|
|
2340
|
+
let rescuedIds = 0;
|
|
2341
|
+
for (const c of rescue.calls) {
|
|
2342
|
+
completedCalls.push({ id: `call_rescued_${++rescuedIds}`, type: 'function', function: { name: c.name, arguments: repairToolArguments(c.arguments, schemas.get(c.name)) } });
|
|
2343
|
+
}
|
|
2344
|
+
heldText = rescue.cleanText;
|
|
2345
|
+
console.log(`[Proxy] Rescued ${rescuedIds} inline tool call(s) from ${route.displayName} into structured tool_calls`);
|
|
2346
|
+
}
|
|
2347
|
+
}
|
|
2348
|
+
// Opt-in schema verdict, taken AFTER the rescue so it covers the
|
|
2349
|
+
// calls the rescue reconstructed from prose — those are the ones
|
|
2350
|
+
// most likely to be malformed, and running first exempted exactly
|
|
2351
|
+
// them. `!headerSent` is the whole licence to throw here: the commit
|
|
2352
|
+
// point is held until the first meaningful content, so the common
|
|
2353
|
+
// tool-call turn (no prose before the call) has sent no bytes yet and
|
|
2354
|
+
// can still fail over invisibly. A turn that already flushed prose is
|
|
2355
|
+
// past the point of no return — the catch below would have to tear
|
|
2356
|
+
// the SSE stream down with a `stream_error`, which is strictly worse
|
|
2357
|
+
// for the client than forwarding a tool call the schema dislikes.
|
|
2358
|
+
// Off-by-default or not, this check must never turn a served answer
|
|
2359
|
+
// into a broken one.
|
|
2360
|
+
if (isToolArgumentValidationEnabled() && !headerSent && completedCalls.length > 0) {
|
|
2361
|
+
const invalid = invalidToolCallReasons(completedCalls, schemas);
|
|
2362
|
+
if (invalid.length > 0)
|
|
2363
|
+
throw invalidToolArgumentsError(route.displayName, invalid);
|
|
2364
|
+
}
|
|
2365
|
+
// Disconnect before the commit point: nothing usable was (or will
|
|
2366
|
+
// be) delivered, and that is CLIENT behavior, not a provider
|
|
2367
|
+
// failure — do not let it fall through to the empty-completion
|
|
2368
|
+
// throw below, which would bench a healthy model+key for 90s and
|
|
2369
|
+
// log a provider error for every Ctrl-C during a reasoning model's
|
|
2370
|
+
// TTFB window.
|
|
2371
|
+
if (clientGone && !headerSent && heldText.trim().length === 0 && completedCalls.length === 0) {
|
|
2372
|
+
console.log(`[Proxy] client disconnected before first token from ${route.displayName} — dropping attempt without benching`);
|
|
2373
|
+
traceRouteEvent('Proxy', {
|
|
2374
|
+
event: 'canceled',
|
|
2375
|
+
requestId: requestGroupId,
|
|
2376
|
+
attempt,
|
|
2377
|
+
platform: route.platform,
|
|
2378
|
+
model: route.modelId,
|
|
2379
|
+
});
|
|
2380
|
+
return 'committed';
|
|
2381
|
+
}
|
|
2382
|
+
const hasText = headerSent || heldText.trim().length > 0;
|
|
2383
|
+
if (!hasText && completedCalls.length === 0) {
|
|
2384
|
+
// Nothing usable came out — same failover semantics as the
|
|
2385
|
+
// non-stream empty-completion path. Headers can't have been sent
|
|
2386
|
+
// (header flush requires payload), so the client never notices.
|
|
2387
|
+
// finish_reason 'length' = the model spent the whole output budget
|
|
2388
|
+
// on hidden reasoning before any visible text: fail over, but skip
|
|
2389
|
+
// the cooldown/penalty (not a provider-health signal).
|
|
2390
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName} (stream produced no content and no tool calls)`), upstreamFinish === 'length' ? { skipBench: true } : {});
|
|
2391
|
+
}
|
|
2392
|
+
// #809: a bare "safe"/"unsafe" classification word streamed by a
|
|
2393
|
+
// relay is an upstream filter, not the requested model — fail over
|
|
2394
|
+
// like an empty completion.
|
|
2395
|
+
if (isUpstreamClassificationOutput(heldText, route.platform) && completedCalls.length === 0) {
|
|
2396
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), upstreamFinish === 'length' ? { skipBench: true } : {});
|
|
2397
|
+
}
|
|
2398
|
+
// Structured-output enforcement for streams (#933): the non-stream
|
|
2399
|
+
// path checks JSON before returning; the stream path must too, or a
|
|
2400
|
+
// model that answers in prose despite the forwarded response_format
|
|
2401
|
+
// ships the essay as a "success" — the worst case for a
|
|
2402
|
+
// machine-readable request. json mode held every byte (headers never
|
|
2403
|
+
// flushed), so failing over here is free: skipBench (provider
|
|
2404
|
+
// healthy, the MODEL misbehaved) + skipModelForRequest (a sibling
|
|
2405
|
+
// key would misbehave identically). Mirrors proxy.ts non-stream.
|
|
2406
|
+
if (mode === 'json' && samplingParams.response_format && completedCalls.length === 0) {
|
|
2407
|
+
const enforced = enforceJsonContent(heldText);
|
|
2408
|
+
if (!enforced.ok) {
|
|
2409
|
+
const truncated = upstreamFinish === 'length';
|
|
2410
|
+
throw Object.assign(new Error(truncated
|
|
2411
|
+
? `truncated JSON from ${route.displayName} (finish_reason=length — raise max_tokens for this ${samplingParams.response_format.type} request)`
|
|
2412
|
+
: `${route.displayName} ignored response_format (returned non-JSON despite ${samplingParams.response_format.type})`), { skipBench: true, skipModelForRequest: true });
|
|
2413
|
+
}
|
|
2414
|
+
if (enforced.healed)
|
|
2415
|
+
heldText = enforced.content;
|
|
2416
|
+
}
|
|
2417
|
+
flushHeaders();
|
|
2418
|
+
if (heldText.length > 0) {
|
|
2419
|
+
writeChunk(mkChunk({ content: heldText }, null));
|
|
2420
|
+
}
|
|
2421
|
+
if (completedCalls.length > 0) {
|
|
2422
|
+
writeChunk(mkChunk({ tool_calls: completedCalls.map((c, i) => ({ index: i, ...c })) }, null));
|
|
2423
|
+
totalOutputTokens += Math.ceil(completedCalls.reduce((n, c) => n + c.function.arguments.length, 0) / 4);
|
|
2424
|
+
}
|
|
2425
|
+
// Terminal finish_reason, ALWAYS present: calls win over a sloppy
|
|
2426
|
+
// upstream 'stop'; 'length'/'content_filter' survive for pure-text
|
|
2427
|
+
// turns; missing upstream reason is synthesized.
|
|
2428
|
+
const finish = completedCalls.length > 0
|
|
2429
|
+
? 'tool_calls'
|
|
2430
|
+
: (upstreamFinish && upstreamFinish !== 'tool_calls' ? upstreamFinish : 'stop');
|
|
2431
|
+
writeChunk(mkChunk({}, finish));
|
|
2432
|
+
// One prompt-token estimate for both the injected usage frame below
|
|
2433
|
+
// and the accounting fallback after it, so a client that reads the
|
|
2434
|
+
// frame and the row this request writes can never disagree. Images
|
|
2435
|
+
// are billed at the same flat per-image estimate the routing budget
|
|
2436
|
+
// uses (the chars/4 pass sees text only).
|
|
2437
|
+
const estimatedPromptTokens = estimatedInputTokens + injectedHandoffTokens + imageCount * IMAGE_TOKEN_ESTIMATE;
|
|
2438
|
+
if (usageChunk) {
|
|
2439
|
+
writeChunk(usageChunk);
|
|
2440
|
+
}
|
|
2441
|
+
else {
|
|
2442
|
+
// Some OpenAI-compatible upstreams never echo a final usage
|
|
2443
|
+
// frame — neither when stream_options.include_usage is requested
|
|
2444
|
+
// nor otherwise. Strict clients (Hermes, Cline, Continue) treat a
|
|
2445
|
+
// missing usage block as "no accounting happened" and skip
|
|
2446
|
+
// per-call token/cost/billing_provider writes entirely; agents
|
|
2447
|
+
// that read usage for context-window display (e.g. #1084) show 0.
|
|
2448
|
+
//
|
|
2449
|
+
// So inject the estimate whenever the upstream never sent one —
|
|
2450
|
+
// regardless of whether the client asked for include_usage. The
|
|
2451
|
+
// numbers are this gateway's own chars/4 estimate (the same total
|
|
2452
|
+
// the accounting below records), never the upstream's accounting,
|
|
2453
|
+
// so the block is flagged `estimated: true` rather than passed
|
|
2454
|
+
// off as real counts.
|
|
2455
|
+
const completionTokens = totalOutputTokens;
|
|
2456
|
+
writeChunk({
|
|
2457
|
+
id: lastMeta.id ?? `chatcmpl-${Date.now()}`,
|
|
2458
|
+
object: 'chat.completion.chunk',
|
|
2459
|
+
created: lastMeta.created ?? Math.floor(Date.now() / 1000),
|
|
2460
|
+
model: lastMeta.model ?? route.modelId,
|
|
2461
|
+
choices: [],
|
|
2462
|
+
usage: {
|
|
2463
|
+
prompt_tokens: estimatedPromptTokens,
|
|
2464
|
+
completion_tokens: completionTokens,
|
|
2465
|
+
total_tokens: estimatedPromptTokens + completionTokens,
|
|
2466
|
+
estimated: true,
|
|
2467
|
+
},
|
|
2468
|
+
});
|
|
2469
|
+
}
|
|
2470
|
+
const doneFrame = 'data: [DONE]\n\n';
|
|
2471
|
+
collectFrame(doneFrame);
|
|
2472
|
+
res.write(doneFrame);
|
|
2473
|
+
res.end();
|
|
2474
|
+
const upstreamUsage = usageChunk?.usage;
|
|
2475
|
+
const inputTokens = upstreamUsage?.prompt_tokens ?? estimatedPromptTokens;
|
|
2476
|
+
const outputTokens = upstreamUsage?.completion_tokens ?? totalOutputTokens;
|
|
2477
|
+
const totalTokens = upstreamUsage?.total_tokens ?? (inputTokens + outputTokens);
|
|
2478
|
+
recordUpstreamSuccess(route, totalTokens, state);
|
|
2479
|
+
// Cache the freshly-generated SSE sequence so an identical later
|
|
2480
|
+
// stream request is replayed without spending another free-tier
|
|
2481
|
+
// slot. A truncated turn (finish 'length') is NOT cached, matching
|
|
2482
|
+
// the JSON cache policy — replaying a cut-off answer would be worse
|
|
2483
|
+
// than regenerating — and neither is one that outgrew the buffer
|
|
2484
|
+
// ceiling. A stream that errored or was aborted mid-flight never
|
|
2485
|
+
// reaches here at all (the catch below owns that path).
|
|
2486
|
+
if (cacheKey && streamFrames && streamCacheable && finish !== 'length') {
|
|
2487
|
+
storeCachedStreamResponse(cacheKey, {
|
|
2488
|
+
frames: streamFrames,
|
|
2489
|
+
platform: route.platform,
|
|
2490
|
+
modelId: route.modelId,
|
|
2491
|
+
keyId: route.keyId,
|
|
2492
|
+
promptTokens: inputTokens,
|
|
2493
|
+
completionTokens: outputTokens,
|
|
2494
|
+
});
|
|
2495
|
+
}
|
|
2496
|
+
setStickyModel(messages, route.modelDbId, sessionIdHeader, stickyStrategyKey);
|
|
2497
|
+
if (handoffMode !== 'off' && sessionKey)
|
|
2498
|
+
recordSuccessfulModel({ sessionKey, modelKey });
|
|
2499
|
+
// #797: remember this turn's thinking trace so the next request from
|
|
2500
|
+
// the same session can restore it (clients strip it on replay).
|
|
2501
|
+
if (streamReasoning.length > 0)
|
|
2502
|
+
rememberReasoning(reasoningSessionKey, modelKey, streamReasoning);
|
|
2503
|
+
traceRouteEvent('Proxy', {
|
|
2504
|
+
event: 'ok',
|
|
2505
|
+
requestId: requestGroupId,
|
|
2506
|
+
attempt,
|
|
2507
|
+
platform: route.platform,
|
|
2508
|
+
model: route.modelId,
|
|
2509
|
+
latencyMs: Date.now() - start,
|
|
2510
|
+
inputTokens,
|
|
2511
|
+
outputTokens,
|
|
2512
|
+
});
|
|
2513
|
+
logRequest(route.platform, route.modelId, route.keyId, 'success', inputTokens, outputTokens, Date.now() - start, null, ttfbMs, pinnedModelId, observeServedModel({ platform: route.platform, requestedModel: route.modelId, servedModel: upstreamModel }), 'http');
|
|
2514
|
+
return 'done';
|
|
2515
|
+
}
|
|
2516
|
+
catch (streamErr) {
|
|
2517
|
+
// Client abort mid-stream: the pump's own `if (clientGone) break`
|
|
2518
|
+
// can lose the race against the fetch-signal rejection, so the
|
|
2519
|
+
// abort may surface here instead. Rethrow — the shared loop's
|
|
2520
|
+
// client-abort branch stops the ladder without benching or an
|
|
2521
|
+
// error log row (the socket is gone; nothing to render).
|
|
2522
|
+
if (isClientAbortError(streamErr))
|
|
2523
|
+
throw streamErr;
|
|
2524
|
+
if (headerSent) {
|
|
2525
|
+
// Mid-stream error after real payload reached the client — finish
|
|
2526
|
+
// the SSE response honestly instead of leaving the client hanging.
|
|
2527
|
+
console.error(`[Proxy] Mid-stream error from ${route.displayName}:`, streamErr.message);
|
|
2528
|
+
const payload = { error: { message: `Provider error (${route.displayName}): stream interrupted`, type: 'stream_error' } };
|
|
2529
|
+
try {
|
|
2530
|
+
res.write(`data: ${JSON.stringify(payload)}\n\n`);
|
|
2531
|
+
}
|
|
2532
|
+
catch { /* socket gone */ }
|
|
2533
|
+
try {
|
|
2534
|
+
res.write('data: [DONE]\n\n');
|
|
2535
|
+
res.end();
|
|
2536
|
+
}
|
|
2537
|
+
catch { /* socket gone */ }
|
|
2538
|
+
traceRouteEvent('Proxy', {
|
|
2539
|
+
event: 'fail',
|
|
2540
|
+
requestId: requestGroupId,
|
|
2541
|
+
attempt,
|
|
2542
|
+
platform: route.platform,
|
|
2543
|
+
model: route.modelId,
|
|
2544
|
+
latencyMs: Date.now() - start,
|
|
2545
|
+
error: sanitizeProviderErrorMessage(streamErr.message),
|
|
2546
|
+
});
|
|
2547
|
+
logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, sanitizeProviderErrorMessage(streamErr.message), ttfbMs, pinnedModelId, null, 'http');
|
|
2548
|
+
return 'committed';
|
|
2549
|
+
}
|
|
2550
|
+
// Headers never sent — bubble to the shared loop, which cooldowns this
|
|
2551
|
+
// model+key and tries the next one. Covers upstream HTTP errors, in-band
|
|
2552
|
+
// error frames, abrupt EOF, stalls, empty completions, and unparseable
|
|
2553
|
+
// dialect turns alike.
|
|
2554
|
+
throw streamErr;
|
|
2555
|
+
}
|
|
2556
|
+
}
|
|
2557
|
+
else {
|
|
2558
|
+
const result = await route.provider.chatCompletion(route.apiKey, outboundMessages, route.modelId, { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, ...samplingParams, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
|
|
2559
|
+
// Raw upstream-reported model, captured BEFORE the contract overwrite
|
|
2560
|
+
// below destroys it — the only evidence when a provider silently
|
|
2561
|
+
// serves a different model than requested (#534). The OpenAI-compat,
|
|
2562
|
+
// cohere, and cloudflare adapters pass the upstream body through, so
|
|
2563
|
+
// result.model here is still the provider's own claim; google/aihorde
|
|
2564
|
+
// synthesize their responses with the routed id (no upstream signal).
|
|
2565
|
+
const upstreamModel = typeof result.model === 'string' ? result.model : null;
|
|
2566
|
+
// Upstream `model` fields are provider-controlled and can be a generic
|
|
2567
|
+
// placeholder such as Reka's "default". The gateway contract exposes
|
|
2568
|
+
// the concrete routed model, consistently across every provider.
|
|
2569
|
+
result.model = route.modelId;
|
|
2570
|
+
// Empty completion (no text, no tool calls) → fail over rather than
|
|
2571
|
+
// return a transport-level "success" the caller can't act on. Mirrors
|
|
2572
|
+
// the zero-chunk streaming case above. Throwing hands it to the shared
|
|
2573
|
+
// loop, which classifies "empty completion" as retryable and applies the
|
|
2574
|
+
// same cooldown/skip/penalty bookkeeping as every other failure.
|
|
2575
|
+
const respMsg = result.choices?.[0]?.message;
|
|
2576
|
+
const respText = contentToString(respMsg?.content ?? '');
|
|
2577
|
+
if (!respText && (respMsg?.tool_calls?.length ?? 0) === 0) {
|
|
2578
|
+
// finish_reason 'length' = the model spent the whole output budget on
|
|
2579
|
+
// hidden reasoning before any visible text (observed live: 5 of 11
|
|
2580
|
+
// hops in one chain). Still fail over, but skipBench tells the shared
|
|
2581
|
+
// loop not to cooldown/penalize a healthy model for a truncated turn.
|
|
2582
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName}`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
|
|
2583
|
+
}
|
|
2584
|
+
// #809: a bare "safe"/"unsafe" classification word from a relay is an
|
|
2585
|
+
// upstream filter, not the requested model — fail over like an empty
|
|
2586
|
+
// completion.
|
|
2587
|
+
if (isUpstreamClassificationOutput(respText, route.platform) && (respMsg?.tool_calls?.length ?? 0) === 0) {
|
|
2588
|
+
throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
|
|
2589
|
+
}
|
|
2590
|
+
// Inline tool-call dialect rescue (#231 audit): a tool-bearing
|
|
2591
|
+
// request answered with the call serialized as TEXT (a mid-
|
|
2592
|
+
// conversation model switch makes the new model imitate the previous
|
|
2593
|
+
// model's private syntax). Re-parse it into structured tool_calls so
|
|
2594
|
+
// the client's agent loop keeps working; a detected-but-unparseable
|
|
2595
|
+
// dialect is a dead turn and fails over like an empty completion.
|
|
2596
|
+
if (wantsTools && respMsg && (respMsg.tool_calls?.length ?? 0) === 0 && respText) {
|
|
2597
|
+
const rescue = rescueInlineToolCalls(respText, new Set((tools ?? []).map(t => t.function.name)));
|
|
2598
|
+
if (rescue.detected) {
|
|
2599
|
+
if (!rescue.calls) {
|
|
2600
|
+
throw new Error(`unparseable inline tool-call dialect from ${route.displayName}: ${respText.slice(0, 120)}`);
|
|
2601
|
+
}
|
|
2602
|
+
const schemas = toolSchemaMap(tools);
|
|
2603
|
+
respMsg.tool_calls = rescue.calls.map((c, i) => ({
|
|
2604
|
+
id: `call_rescued_${i + 1}`,
|
|
2605
|
+
type: 'function',
|
|
2606
|
+
function: { name: c.name, arguments: repairToolArguments(c.arguments, schemas.get(c.name)) },
|
|
2607
|
+
}));
|
|
2608
|
+
respMsg.content = rescue.cleanText.length > 0 ? rescue.cleanText : null;
|
|
2609
|
+
if (result.choices?.[0])
|
|
2610
|
+
result.choices[0].finish_reason = 'tool_calls';
|
|
2611
|
+
console.log(`[Proxy] Rescued ${rescue.calls.length} inline tool call(s) from ${route.displayName} into structured tool_calls`);
|
|
2612
|
+
}
|
|
2613
|
+
}
|
|
2614
|
+
// Structured-output enforcement (#514 follow-up): the client asked for
|
|
2615
|
+
// JSON; a model that answered in prose despite the forwarded
|
|
2616
|
+
// response_format must not be returned as a "success". Heal the common
|
|
2617
|
+
// almost-right shapes (fenced block, prose-wrapped JSON) in place;
|
|
2618
|
+
// otherwise fail over. Deliberately AFTER the dialect rescue (matching
|
|
2619
|
+
// responses.ts): an inline tool-call turn isn't JSON either, and
|
|
2620
|
+
// gating it first burned a failover hop on turns the rescue converts.
|
|
2621
|
+
// skipBench: the provider is healthy — the MODEL misbehaved — so no
|
|
2622
|
+
// cooldown/penalty; skipModelForRequest: a sibling key would misbehave
|
|
2623
|
+
// identically, so rule out the whole model for this request.
|
|
2624
|
+
if (samplingParams.response_format && respText && (respMsg?.tool_calls?.length ?? 0) === 0) {
|
|
2625
|
+
const enforced = enforceJsonContent(respText);
|
|
2626
|
+
if (!enforced.ok) {
|
|
2627
|
+
// finish_reason 'length' = the JSON was CUT OFF by max_tokens, not
|
|
2628
|
+
// ignored — same failover (a terser model may fit the budget), but
|
|
2629
|
+
// an honest error class/trail instead of "ignored response_format".
|
|
2630
|
+
const truncated = result.choices?.[0]?.finish_reason === 'length';
|
|
2631
|
+
throw Object.assign(new Error(truncated
|
|
2632
|
+
? `truncated JSON from ${route.displayName} (finish_reason=length — raise max_tokens for this ${samplingParams.response_format.type} request)`
|
|
2633
|
+
: `${route.displayName} ignored response_format (returned non-JSON despite ${samplingParams.response_format.type})`), { skipBench: true, skipModelForRequest: true });
|
|
2634
|
+
}
|
|
2635
|
+
if (enforced.healed && respMsg) {
|
|
2636
|
+
respMsg.content = enforced.content;
|
|
2637
|
+
}
|
|
2638
|
+
}
|
|
2639
|
+
// Repair double-encoded tool arguments against the request's tool
|
|
2640
|
+
// schemas (e.g. GLM emitting an array parameter as a JSON string),
|
|
2641
|
+
// so strict clients don't reject the call. Schema-gated — a true
|
|
2642
|
+
// string parameter is never touched. See lib/tool-args.ts.
|
|
2643
|
+
//
|
|
2644
|
+
// Deliberately BEFORE the success bookkeeping below: the opt-in schema
|
|
2645
|
+
// verdict that follows can still fail this attempt over, and crediting
|
|
2646
|
+
// recordUpstreamSuccess / rememberReasoning / setStickyModel to an
|
|
2647
|
+
// attempt we are about to discard would bill a model that never served
|
|
2648
|
+
// the turn and pin the session to it for the next one.
|
|
2649
|
+
if (respMsg?.tool_calls?.length) {
|
|
2650
|
+
const schemas = toolSchemaMap(tools);
|
|
2651
|
+
for (const tc of respMsg.tool_calls) {
|
|
2652
|
+
if (tc?.function?.arguments != null) {
|
|
2653
|
+
tc.function.arguments = repairToolArguments(tc.function.arguments, schemas.get(tc.function.name));
|
|
2654
|
+
}
|
|
2655
|
+
}
|
|
2656
|
+
// Whatever the repair could not fix is still broken. Opt-in, and
|
|
2657
|
+
// thrown before anything is written, so failover is invisible.
|
|
2658
|
+
if (isToolArgumentValidationEnabled()) {
|
|
2659
|
+
const invalid = invalidToolCallReasons(respMsg.tool_calls, schemas);
|
|
2660
|
+
if (invalid.length > 0)
|
|
2661
|
+
throw invalidToolArgumentsError(route.displayName, invalid);
|
|
2662
|
+
}
|
|
2663
|
+
}
|
|
2664
|
+
// Usage fallback: providers that omit `usage` used to be logged as 0
|
|
2665
|
+
// tokens, silently undercounting analytics and the rate-limit ledger.
|
|
2666
|
+
// Fall back to the same chars/4 estimate the streaming path uses (tool
|
|
2667
|
+
// arguments included, mirroring the stream accounting; counted after
|
|
2668
|
+
// the repair above, so it measures the bytes actually sent, and
|
|
2669
|
+
// reasoning tokens included too, so thinking models aren't
|
|
2670
|
+
// undercounted — #764).
|
|
2671
|
+
const respToolArgChars = (respMsg?.tool_calls ?? []).reduce((n, tc) => n + (tc?.function?.arguments?.length ?? 0), 0);
|
|
2672
|
+
const promptTokens = result.usage?.prompt_tokens ?? estimatedInputTokens;
|
|
2673
|
+
const completionTokens = result.usage?.completion_tokens
|
|
2674
|
+
?? Math.ceil((contentToString(respMsg?.content ?? '').length + completionReasoningText(result).length + respToolArgChars) / 4);
|
|
2675
|
+
const totalTokens = result.usage?.total_tokens ?? (promptTokens + completionTokens);
|
|
2676
|
+
recordUpstreamSuccess(route, totalTokens, state);
|
|
2677
|
+
// #797: remember this turn's thinking trace so the next request from
|
|
2678
|
+
// the same session can restore it (clients strip it on replay).
|
|
2679
|
+
// normalizeChoices keeps reasoning_content on the message even when it
|
|
2680
|
+
// folds the trace into an otherwise-empty content field.
|
|
2681
|
+
if (typeof respMsg?.reasoning_content === 'string' && respMsg.reasoning_content.length > 0) {
|
|
2682
|
+
rememberReasoning(reasoningSessionKey, modelKey, respMsg.reasoning_content);
|
|
2683
|
+
}
|
|
2684
|
+
// Use stickyStrategyKey (not the global strategyKey) so a group-pinned
|
|
2685
|
+
// request writes its sticky entry under the SAME key the next turn reads
|
|
2686
|
+
// from (set to the requested model id at the top of the loop). Matches the
|
|
2687
|
+
// streaming success path; without it, "prefer last successful provider"
|
|
2688
|
+
// is lost for non-streaming group-pinned sessions. (#341 review)
|
|
2689
|
+
setStickyModel(messages, route.modelDbId, sessionIdHeader, stickyStrategyKey);
|
|
2690
|
+
if (handoffMode !== 'off' && sessionKey)
|
|
2691
|
+
recordSuccessfulModel({ sessionKey, modelKey });
|
|
2692
|
+
res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
|
|
2693
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
2694
|
+
// Normalize array-shaped message.content to a string on the way out (#166).
|
|
2695
|
+
const outboundBody = sanitizeResponse(normalizeOutboundContent(result));
|
|
2696
|
+
res.setHeader('X-FreeLLM-Cache', cacheKey ? 'MISS' : 'OFF');
|
|
2697
|
+
// #1084: agents show zero context usage when the upstream omits
|
|
2698
|
+
// `usage` (many free-tier providers do). Fall back to the same
|
|
2699
|
+
// chars/4 estimate used for accounting above, flagged `estimated:
|
|
2700
|
+
// true` so a cost-accounting client can tell it apart from the
|
|
2701
|
+
// upstream's real counts.
|
|
2702
|
+
if (!outboundBody.usage) {
|
|
2703
|
+
outboundBody.usage = {
|
|
2704
|
+
prompt_tokens: promptTokens,
|
|
2705
|
+
completion_tokens: completionTokens,
|
|
2706
|
+
total_tokens: totalTokens,
|
|
2707
|
+
estimated: true,
|
|
2708
|
+
};
|
|
2709
|
+
}
|
|
2710
|
+
// Persist the completed response for Idempotency-Key replays. Only
|
|
2711
|
+
// non-streaming requests with a valid key reach here; a truncated turn
|
|
2712
|
+
// (finish_reason 'length') is NOT stored — replaying a cut-off answer
|
|
2713
|
+
// would be worse than regenerating, matching the cache policy below.
|
|
2714
|
+
if (idemKey
|
|
2715
|
+
&& idemFingerprint
|
|
2716
|
+
&& result.choices?.[0]?.finish_reason !== 'length') {
|
|
2717
|
+
storeIdempotencyResult(hashIdempotencyKey(idemKey), idemFingerprint, 200, outboundBody, requestGroupId);
|
|
2718
|
+
}
|
|
2719
|
+
// #1102: expose the execution id in the body so clients (incl. AI
|
|
2720
|
+
// agents that ignore headers) can trace this response to its analytics
|
|
2721
|
+
// attempt trail. Additive — OpenAI clients ignore unknown fields.
|
|
2722
|
+
res.json({ execution_id: requestGroupId, ...outboundBody });
|
|
2723
|
+
// Cache the freshly-generated answer so an identical later request is
|
|
2724
|
+
// served from memory without spending another free-tier slot. A
|
|
2725
|
+
// truncated turn (finish_reason 'length') is NOT cached: replaying a
|
|
2726
|
+
// cut-off answer forever would be worse than regenerating.
|
|
2727
|
+
if (cacheKey && result.choices?.[0]?.finish_reason !== 'length') {
|
|
2728
|
+
storeCachedResponse(cacheKey, {
|
|
2729
|
+
body: outboundBody,
|
|
2730
|
+
platform: route.platform,
|
|
2731
|
+
modelId: route.modelId,
|
|
2732
|
+
keyId: route.keyId,
|
|
2733
|
+
promptTokens,
|
|
2734
|
+
completionTokens,
|
|
2735
|
+
});
|
|
2736
|
+
}
|
|
2737
|
+
traceRouteEvent('Proxy', {
|
|
2738
|
+
event: 'ok',
|
|
2739
|
+
requestId: requestGroupId,
|
|
2740
|
+
attempt,
|
|
2741
|
+
platform: route.platform,
|
|
2742
|
+
model: route.modelId,
|
|
2743
|
+
latencyMs: Date.now() - start,
|
|
2744
|
+
inputTokens: promptTokens,
|
|
2745
|
+
outputTokens: completionTokens,
|
|
2746
|
+
});
|
|
2747
|
+
logRequest(route.platform, route.modelId, route.keyId, 'success', promptTokens, completionTokens, Date.now() - start, null, null, pinnedModelId, observeServedModel({ platform: route.platform, requestedModel: route.modelId, servedModel: upstreamModel }), 'http');
|
|
2748
|
+
return 'done';
|
|
2749
|
+
}
|
|
2750
|
+
},
|
|
2751
|
+
logFailure: (route, err, attempt) => {
|
|
2752
|
+
const latency = Date.now() - start;
|
|
2753
|
+
const safeError = sanitizeProviderErrorMessage(err.message);
|
|
2754
|
+
traceRouteEvent('Proxy', {
|
|
2755
|
+
event: 'fail',
|
|
2756
|
+
requestId: requestGroupId,
|
|
2757
|
+
attempt,
|
|
2758
|
+
platform: route.platform,
|
|
2759
|
+
model: route.modelId,
|
|
2760
|
+
latencyMs: latency,
|
|
2761
|
+
error: safeError,
|
|
2762
|
+
});
|
|
2763
|
+
logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, 0, latency, safeError, null, pinnedModelId, null, 'http');
|
|
2764
|
+
},
|
|
2765
|
+
onFatal: (route, err, attempt) => {
|
|
2766
|
+
// Non-retryable error (bare 4xx, etc.): don't retry.
|
|
2767
|
+
setFallbackHeaders(res, attempt, attemptLog);
|
|
2768
|
+
res.status(502).json({
|
|
2769
|
+
error: {
|
|
2770
|
+
message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(err.message)}`,
|
|
2771
|
+
type: 'provider_error',
|
|
2772
|
+
},
|
|
2773
|
+
execution_id: requestGroupId,
|
|
2774
|
+
});
|
|
2775
|
+
},
|
|
2776
|
+
onRoutingExhausted: (lastError, routeErr, exhaustion, info) => {
|
|
2777
|
+
// No more models available.
|
|
2778
|
+
setFallbackHeaders(res, info.attempts.length, info.attempts);
|
|
2779
|
+
setExhaustionHeaders(res, exhaustion);
|
|
2780
|
+
res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
|
|
2781
|
+
},
|
|
2782
|
+
onExhausted: (exhaustion, info) => {
|
|
2783
|
+
setFallbackHeaders(res, info.attempts.length, info.attempts);
|
|
2784
|
+
setExhaustionHeaders(res, exhaustion);
|
|
2785
|
+
res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
|
|
2786
|
+
},
|
|
2787
|
+
});
|
|
2788
|
+
});
|
|
2789
|
+
// logRequest moved to lib/request-log.ts (shared with the fusion service to
|
|
2790
|
+
// avoid an import cycle); imported above for internal use and re-exported here
|
|
2791
|
+
// for routes/responses.ts which imports it from this module.
|
|
2792
|
+
export { logRequest };
|
|
2793
|
+
//# sourceMappingURL=proxy.js.map
|