@srouterhq/server 0.0.0-stage → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (1027) hide show
  1. package/LICENSE +21 -0
  2. package/client/dist/assets/am-TQ7Jmdqe.js +1 -0
  3. package/client/dist/assets/ar-Bf5a7yfC.js +1 -0
  4. package/client/dist/assets/az-CrosE1xH.js +1 -0
  5. package/client/dist/assets/bg-BvWnmcUz.js +1 -0
  6. package/client/dist/assets/bn-ams5TsJp.js +1 -0
  7. package/client/dist/assets/cs-Dh67CaJ_.js +1 -0
  8. package/client/dist/assets/da-9WlT8Iq7.js +1 -0
  9. package/client/dist/assets/de-lWhJJzz7.js +1 -0
  10. package/client/dist/assets/el-Cu7_GhNq.js +1 -0
  11. package/client/dist/assets/es-BZuzsBcP.js +1 -0
  12. package/client/dist/assets/fa-Dwdn-jKS.js +1 -0
  13. package/client/dist/assets/fi-BFrFyOTy.js +1 -0
  14. package/client/dist/assets/fr-XhbtmPpj.js +1 -0
  15. package/client/dist/assets/geist-cyrillic-ext-wght-normal-DjL33-gN.woff2 +0 -0
  16. package/client/dist/assets/geist-cyrillic-wght-normal-BEAKL7Jp.woff2 +0 -0
  17. package/client/dist/assets/geist-latin-ext-wght-normal-DC-KSUi6.woff2 +0 -0
  18. package/client/dist/assets/geist-latin-wght-normal-BgDaEnEv.woff2 +0 -0
  19. package/client/dist/assets/geist-mono-cyrillic-ext-wght-normal-I4S5GZfc.woff2 +0 -0
  20. package/client/dist/assets/geist-mono-cyrillic-wght-normal-BmXc_FBt.woff2 +0 -0
  21. package/client/dist/assets/geist-mono-latin-ext-wght-normal-DrnZ1wKl.woff2 +0 -0
  22. package/client/dist/assets/geist-mono-latin-wght-normal-B_7UjwxQ.woff2 +0 -0
  23. package/client/dist/assets/geist-mono-symbols2-wght-normal-GZpp1pK2.woff2 +0 -0
  24. package/client/dist/assets/geist-mono-vietnamese-wght-normal-D8KDMBhC.woff2 +0 -0
  25. package/client/dist/assets/geist-vietnamese-wght-normal-6IgcOCM7.woff2 +0 -0
  26. package/client/dist/assets/gu-D0uhLUB8.js +1 -0
  27. package/client/dist/assets/ha-CVlm9pqp.js +1 -0
  28. package/client/dist/assets/he-CMIFeoWr.js +1 -0
  29. package/client/dist/assets/hi-ENTwhBpn.js +1 -0
  30. package/client/dist/assets/hr-fc03vMy_.js +1 -0
  31. package/client/dist/assets/hu-YBo2BIYt.js +1 -0
  32. package/client/dist/assets/id-BEWXYX69.js +1 -0
  33. package/client/dist/assets/ig-BKLZFKka.js +1 -0
  34. package/client/dist/assets/index-D9I4KtvR.js +145 -0
  35. package/client/dist/assets/index-txhEmCdb.css +2 -0
  36. package/client/dist/assets/it-Bzmei1Y2.js +1 -0
  37. package/client/dist/assets/ja-BfcpuqEl.js +1 -0
  38. package/client/dist/assets/ka-CnWjRgLo.js +1 -0
  39. package/client/dist/assets/km-BAjPR3Qj.js +1 -0
  40. package/client/dist/assets/kn-B15WKU5m.js +1 -0
  41. package/client/dist/assets/ko-C61fAB-w.js +1 -0
  42. package/client/dist/assets/lt-CYTr4_YW.js +1 -0
  43. package/client/dist/assets/ml-CtqxgyH3.js +1 -0
  44. package/client/dist/assets/mr-X5TMY7aR.js +1 -0
  45. package/client/dist/assets/ms-DSg8RqPS.js +1 -0
  46. package/client/dist/assets/my-D7RqR8uE.js +1 -0
  47. package/client/dist/assets/ne-DaflxkDj.js +1 -0
  48. package/client/dist/assets/nl-BOqqtsC8.js +1 -0
  49. package/client/dist/assets/no-BSHp_StB.js +1 -0
  50. package/client/dist/assets/or-BkLR9Vub.js +1 -0
  51. package/client/dist/assets/pa-CarJ2XgI.js +1 -0
  52. package/client/dist/assets/pl-C2n0V-0k.js +1 -0
  53. package/client/dist/assets/pt-BR-CqJdNhg2.js +1 -0
  54. package/client/dist/assets/pt-PT-C_HpMyKz.js +1 -0
  55. package/client/dist/assets/ro-DVRZqSAS.js +1 -0
  56. package/client/dist/assets/ru-DwAJG7VC.js +1 -0
  57. package/client/dist/assets/si-BxUezXm0.js +1 -0
  58. package/client/dist/assets/sk-eI1VcIfY.js +1 -0
  59. package/client/dist/assets/sr-CJF3auCr.js +1 -0
  60. package/client/dist/assets/srouter-logo-C6ZfjGIi.svg +14 -0
  61. package/client/dist/assets/sv-nqvwHAq_.js +1 -0
  62. package/client/dist/assets/sw-DWynE2pW.js +1 -0
  63. package/client/dist/assets/ta-BVM12knC.js +1 -0
  64. package/client/dist/assets/te-oyzEkF7_.js +1 -0
  65. package/client/dist/assets/th-COVeE0ZQ.js +1 -0
  66. package/client/dist/assets/tl-DsJANnQz.js +1 -0
  67. package/client/dist/assets/tr-B-Fl-urO.js +1 -0
  68. package/client/dist/assets/uk-C5qketAO.js +1 -0
  69. package/client/dist/assets/ur-QU9KugfV.js +1 -0
  70. package/client/dist/assets/uz-DL9cG_mY.js +1 -0
  71. package/client/dist/assets/vi-CQiVy3Pu.js +1 -0
  72. package/client/dist/assets/yo-DqazMdrZ.js +1 -0
  73. package/client/dist/assets/zh-CN-DnYL_Kio.js +1 -0
  74. package/client/dist/assets/zh-TW-5bdbosKj.js +1 -0
  75. package/client/dist/favicon.svg +14 -0
  76. package/client/dist/icons.svg +24 -0
  77. package/client/dist/index.html +37 -0
  78. package/dist/app.d.ts +4 -0
  79. package/dist/app.d.ts.map +1 -0
  80. package/dist/app.js +372 -0
  81. package/dist/app.js.map +1 -0
  82. package/dist/db/index.d.ts +49 -0
  83. package/dist/db/index.d.ts.map +1 -0
  84. package/dist/db/index.js +203 -0
  85. package/dist/db/index.js.map +1 -0
  86. package/dist/db/migrate/TEMPLATE.d.ts +4 -0
  87. package/dist/db/migrate/TEMPLATE.d.ts.map +1 -0
  88. package/dist/db/migrate/TEMPLATE.js +18 -0
  89. package/dist/db/migrate/TEMPLATE.js.map +1 -0
  90. package/dist/db/migrate/cli.d.ts +2 -0
  91. package/dist/db/migrate/cli.d.ts.map +1 -0
  92. package/dist/db/migrate/cli.js +177 -0
  93. package/dist/db/migrate/cli.js.map +1 -0
  94. package/dist/db/migrate/defaults.d.ts +52 -0
  95. package/dist/db/migrate/defaults.d.ts.map +1 -0
  96. package/dist/db/migrate/defaults.js +126 -0
  97. package/dist/db/migrate/defaults.js.map +1 -0
  98. package/dist/db/migrate/runner.d.ts +16 -0
  99. package/dist/db/migrate/runner.d.ts.map +1 -0
  100. package/dist/db/migrate/runner.js +178 -0
  101. package/dist/db/migrate/runner.js.map +1 -0
  102. package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts +4 -0
  103. package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts.map +1 -0
  104. package/dist/db/migrations/20260101_000000_legacy_baseline.js +2240 -0
  105. package/dist/db/migrations/20260101_000000_legacy_baseline.js.map +1 -0
  106. package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts +4 -0
  107. package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts.map +1 -0
  108. package/dist/db/migrations/20260627_000001_custom_provider_modalities.js +27 -0
  109. package/dist/db/migrations/20260627_000001_custom_provider_modalities.js.map +1 -0
  110. package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts +4 -0
  111. package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts.map +1 -0
  112. package/dist/db/migrations/20260627_000002_catalog_model_state.js +30 -0
  113. package/dist/db/migrations/20260627_000002_catalog_model_state.js.map +1 -0
  114. package/dist/db/migrations/20260628_120000_request_aggregates.d.ts +6 -0
  115. package/dist/db/migrations/20260628_120000_request_aggregates.d.ts.map +1 -0
  116. package/dist/db/migrations/20260628_120000_request_aggregates.js +104 -0
  117. package/dist/db/migrations/20260628_120000_request_aggregates.js.map +1 -0
  118. package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts +9 -0
  119. package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts.map +1 -0
  120. package/dist/db/migrations/20260630_000001_github_gpt41_context.js +22 -0
  121. package/dist/db/migrations/20260630_000001_github_gpt41_context.js.map +1 -0
  122. package/dist/db/migrations/20260706_000001_request_client_info.d.ts +4 -0
  123. package/dist/db/migrations/20260706_000001_request_client_info.d.ts.map +1 -0
  124. package/dist/db/migrations/20260706_000001_request_client_info.js +30 -0
  125. package/dist/db/migrations/20260706_000001_request_client_info.js.map +1 -0
  126. package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts +17 -0
  127. package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts.map +1 -0
  128. package/dist/db/migrations/20260706_000002_custom_model_tool_support.js +20 -0
  129. package/dist/db/migrations/20260706_000002_custom_model_tool_support.js.map +1 -0
  130. package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts +12 -0
  131. package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts.map +1 -0
  132. package/dist/db/migrations/20260714_000001_profile_chain_backfill.js +45 -0
  133. package/dist/db/migrations/20260714_000001_profile_chain_backfill.js.map +1 -0
  134. package/dist/db/migrations/20260720_000001_key_health_error.d.ts +5 -0
  135. package/dist/db/migrations/20260720_000001_key_health_error.d.ts.map +1 -0
  136. package/dist/db/migrations/20260720_000001_key_health_error.js +16 -0
  137. package/dist/db/migrations/20260720_000001_key_health_error.js.map +1 -0
  138. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts +4 -0
  139. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts.map +1 -0
  140. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js +38 -0
  141. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js.map +1 -0
  142. package/dist/db/migrations/20260726_000002_request_attempts.d.ts +4 -0
  143. package/dist/db/migrations/20260726_000002_request_attempts.d.ts.map +1 -0
  144. package/dist/db/migrations/20260726_000002_request_attempts.js +43 -0
  145. package/dist/db/migrations/20260726_000002_request_attempts.js.map +1 -0
  146. package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts +4 -0
  147. package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts.map +1 -0
  148. package/dist/db/migrations/20260726_000003_model_source_provenance.js +80 -0
  149. package/dist/db/migrations/20260726_000003_model_source_provenance.js.map +1 -0
  150. package/dist/db/migrations/20260726_000004_media_model_meta.d.ts +4 -0
  151. package/dist/db/migrations/20260726_000004_media_model_meta.d.ts.map +1 -0
  152. package/dist/db/migrations/20260726_000004_media_model_meta.js +34 -0
  153. package/dist/db/migrations/20260726_000004_media_model_meta.js.map +1 -0
  154. package/dist/db/migrations/20260726_000005_request_served_model.d.ts +4 -0
  155. package/dist/db/migrations/20260726_000005_request_served_model.d.ts.map +1 -0
  156. package/dist/db/migrations/20260726_000005_request_served_model.js +31 -0
  157. package/dist/db/migrations/20260726_000005_request_served_model.js.map +1 -0
  158. package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts +4 -0
  159. package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts.map +1 -0
  160. package/dist/db/migrations/20260726_000006_attempt_error_summary.js +29 -0
  161. package/dist/db/migrations/20260726_000006_attempt_error_summary.js.map +1 -0
  162. package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts +4 -0
  163. package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts.map +1 -0
  164. package/dist/db/migrations/20260727_000001_agent_compatibility.js +41 -0
  165. package/dist/db/migrations/20260727_000001_agent_compatibility.js.map +1 -0
  166. package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts +12 -0
  167. package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts.map +1 -0
  168. package/dist/db/migrations/20260728_000001_tombstone_provenance.js +37 -0
  169. package/dist/db/migrations/20260728_000001_tombstone_provenance.js.map +1 -0
  170. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts +4 -0
  171. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts.map +1 -0
  172. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js +145 -0
  173. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js.map +1 -0
  174. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts +6 -0
  175. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts.map +1 -0
  176. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js +52 -0
  177. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js.map +1 -0
  178. package/dist/db/migrations/20260805_000001_key_model_scope.d.ts +6 -0
  179. package/dist/db/migrations/20260805_000001_key_model_scope.d.ts.map +1 -0
  180. package/dist/db/migrations/20260805_000001_key_model_scope.js +17 -0
  181. package/dist/db/migrations/20260805_000001_key_model_scope.js.map +1 -0
  182. package/dist/db/migrations/20260805_000002_client_profiles.d.ts +4 -0
  183. package/dist/db/migrations/20260805_000002_client_profiles.d.ts.map +1 -0
  184. package/dist/db/migrations/20260805_000002_client_profiles.js +33 -0
  185. package/dist/db/migrations/20260805_000002_client_profiles.js.map +1 -0
  186. package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts +4 -0
  187. package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts.map +1 -0
  188. package/dist/db/migrations/20260810_000001_api_key_proxy.js +33 -0
  189. package/dist/db/migrations/20260810_000001_api_key_proxy.js.map +1 -0
  190. package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts +4 -0
  191. package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts.map +1 -0
  192. package/dist/db/migrations/20260819_000001_custom_model_tombstones.js +19 -0
  193. package/dist/db/migrations/20260819_000001_custom_model_tombstones.js.map +1 -0
  194. package/dist/db/migrations/20260820_000001_playground_conversations.d.ts +4 -0
  195. package/dist/db/migrations/20260820_000001_playground_conversations.d.ts.map +1 -0
  196. package/dist/db/migrations/20260820_000001_playground_conversations.js +47 -0
  197. package/dist/db/migrations/20260820_000001_playground_conversations.js.map +1 -0
  198. package/dist/db/migrations/20260823_000001_server_logs.d.ts +4 -0
  199. package/dist/db/migrations/20260823_000001_server_logs.d.ts.map +1 -0
  200. package/dist/db/migrations/20260823_000001_server_logs.js +53 -0
  201. package/dist/db/migrations/20260823_000001_server_logs.js.map +1 -0
  202. package/dist/db/migrations/20260823_000002_backups_table.d.ts +8 -0
  203. package/dist/db/migrations/20260823_000002_backups_table.d.ts.map +1 -0
  204. package/dist/db/migrations/20260823_000002_backups_table.js +22 -0
  205. package/dist/db/migrations/20260823_000002_backups_table.js.map +1 -0
  206. package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts +19 -0
  207. package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts.map +1 -0
  208. package/dist/db/migrations/20260823_000003_attempt_key_label.js +28 -0
  209. package/dist/db/migrations/20260823_000003_attempt_key_label.js.map +1 -0
  210. package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts +14 -0
  211. package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts.map +1 -0
  212. package/dist/db/migrations/20260823_000004_profile_auto_include.js +25 -0
  213. package/dist/db/migrations/20260823_000004_profile_auto_include.js.map +1 -0
  214. package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts +4 -0
  215. package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts.map +1 -0
  216. package/dist/db/migrations/20260901_000001_idempotency_claims.js +46 -0
  217. package/dist/db/migrations/20260901_000001_idempotency_claims.js.map +1 -0
  218. package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts +4 -0
  219. package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts.map +1 -0
  220. package/dist/db/migrations/20260901_000002_quota_observation_lookup.js +37 -0
  221. package/dist/db/migrations/20260901_000002_quota_observation_lookup.js.map +1 -0
  222. package/dist/db/migrations/20260901_000003_request_caller.d.ts +4 -0
  223. package/dist/db/migrations/20260901_000003_request_caller.d.ts.map +1 -0
  224. package/dist/db/migrations/20260901_000003_request_caller.js +27 -0
  225. package/dist/db/migrations/20260901_000003_request_caller.js.map +1 -0
  226. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts +4 -0
  227. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts.map +1 -0
  228. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js +35 -0
  229. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js.map +1 -0
  230. package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts +4 -0
  231. package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts.map +1 -0
  232. package/dist/db/migrations/20260903_000001_mcp_enabled_default.js +37 -0
  233. package/dist/db/migrations/20260903_000001_mcp_enabled_default.js.map +1 -0
  234. package/dist/db/migrations/20260903_000002_response_cache.d.ts +4 -0
  235. package/dist/db/migrations/20260903_000002_response_cache.d.ts.map +1 -0
  236. package/dist/db/migrations/20260903_000002_response_cache.js +53 -0
  237. package/dist/db/migrations/20260903_000002_response_cache.js.map +1 -0
  238. package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts +13 -0
  239. package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts.map +1 -0
  240. package/dist/db/migrations/20260904_000001_key_monthly_budget.js +35 -0
  241. package/dist/db/migrations/20260904_000001_key_monthly_budget.js.map +1 -0
  242. package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts +4 -0
  243. package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts.map +1 -0
  244. package/dist/db/migrations/20260913_000001_request_model_attribution.js +68 -0
  245. package/dist/db/migrations/20260913_000001_request_model_attribution.js.map +1 -0
  246. package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts +5 -0
  247. package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts.map +1 -0
  248. package/dist/db/migrations/20260914_000001_key_monthly_usage.js +37 -0
  249. package/dist/db/migrations/20260914_000001_key_monthly_usage.js.map +1 -0
  250. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts +4 -0
  251. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts.map +1 -0
  252. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js +28 -0
  253. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js.map +1 -0
  254. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts +4 -0
  255. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts.map +1 -0
  256. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js +25 -0
  257. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js.map +1 -0
  258. package/dist/db/migrations/20261005_000001_key_budget_period.d.ts +10 -0
  259. package/dist/db/migrations/20261005_000001_key_budget_period.d.ts.map +1 -0
  260. package/dist/db/migrations/20261005_000001_key_budget_period.js +78 -0
  261. package/dist/db/migrations/20261005_000001_key_budget_period.js.map +1 -0
  262. package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts +21 -0
  263. package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts.map +1 -0
  264. package/dist/db/migrations/20261005_000002_key_budget_period_auto.js +41 -0
  265. package/dist/db/migrations/20261005_000002_key_budget_period_auto.js.map +1 -0
  266. package/dist/db/model-pricing.d.ts +39 -0
  267. package/dist/db/model-pricing.d.ts.map +1 -0
  268. package/dist/db/model-pricing.js +274 -0
  269. package/dist/db/model-pricing.js.map +1 -0
  270. package/dist/db/node-sqlite.d.ts +9 -0
  271. package/dist/db/node-sqlite.d.ts.map +1 -0
  272. package/dist/db/node-sqlite.js +87 -0
  273. package/dist/db/node-sqlite.js.map +1 -0
  274. package/dist/db/types.d.ts +22 -0
  275. package/dist/db/types.d.ts.map +1 -0
  276. package/dist/db/types.js +2 -0
  277. package/dist/db/types.js.map +1 -0
  278. package/dist/docs/docs-page.d.ts +2 -0
  279. package/dist/docs/docs-page.d.ts.map +1 -0
  280. package/dist/docs/docs-page.js +319 -0
  281. package/dist/docs/docs-page.js.map +1 -0
  282. package/dist/docs/openapi.d.ts +1896 -0
  283. package/dist/docs/openapi.d.ts.map +1 -0
  284. package/dist/docs/openapi.js +1096 -0
  285. package/dist/docs/openapi.js.map +1 -0
  286. package/dist/env.d.ts +2 -0
  287. package/dist/env.d.ts.map +1 -0
  288. package/dist/env.js +9 -0
  289. package/dist/env.js.map +1 -0
  290. package/dist/index.d.ts +2 -0
  291. package/dist/index.d.ts.map +1 -0
  292. package/dist/index.js +144 -0
  293. package/dist/index.js.map +1 -0
  294. package/dist/lib/anthropic-documents.d.ts +41 -0
  295. package/dist/lib/anthropic-documents.d.ts.map +1 -0
  296. package/dist/lib/anthropic-documents.js +132 -0
  297. package/dist/lib/anthropic-documents.js.map +1 -0
  298. package/dist/lib/app-version.d.ts +5 -0
  299. package/dist/lib/app-version.d.ts.map +1 -0
  300. package/dist/lib/app-version.js +61 -0
  301. package/dist/lib/app-version.js.map +1 -0
  302. package/dist/lib/attempt-trace.d.ts +23 -0
  303. package/dist/lib/attempt-trace.d.ts.map +1 -0
  304. package/dist/lib/attempt-trace.js +32 -0
  305. package/dist/lib/attempt-trace.js.map +1 -0
  306. package/dist/lib/budget.d.ts +6 -0
  307. package/dist/lib/budget.d.ts.map +1 -0
  308. package/dist/lib/budget.js +40 -0
  309. package/dist/lib/budget.js.map +1 -0
  310. package/dist/lib/client-classifier.d.ts +10 -0
  311. package/dist/lib/client-classifier.d.ts.map +1 -0
  312. package/dist/lib/client-classifier.js +126 -0
  313. package/dist/lib/client-classifier.js.map +1 -0
  314. package/dist/lib/client-context.d.ts +10 -0
  315. package/dist/lib/client-context.d.ts.map +1 -0
  316. package/dist/lib/client-context.js +39 -0
  317. package/dist/lib/client-context.js.map +1 -0
  318. package/dist/lib/config.d.ts +33 -0
  319. package/dist/lib/config.d.ts.map +1 -0
  320. package/dist/lib/config.js +94 -0
  321. package/dist/lib/config.js.map +1 -0
  322. package/dist/lib/content.d.ts +33 -0
  323. package/dist/lib/content.d.ts.map +1 -0
  324. package/dist/lib/content.js +201 -0
  325. package/dist/lib/content.js.map +1 -0
  326. package/dist/lib/credential.d.ts +11 -0
  327. package/dist/lib/credential.d.ts.map +1 -0
  328. package/dist/lib/credential.js +18 -0
  329. package/dist/lib/credential.js.map +1 -0
  330. package/dist/lib/crypto.d.ts +31 -0
  331. package/dist/lib/crypto.d.ts.map +1 -0
  332. package/dist/lib/crypto.js +206 -0
  333. package/dist/lib/crypto.js.map +1 -0
  334. package/dist/lib/custom-provider-cleanup.d.ts +13 -0
  335. package/dist/lib/custom-provider-cleanup.d.ts.map +1 -0
  336. package/dist/lib/custom-provider-cleanup.js +43 -0
  337. package/dist/lib/custom-provider-cleanup.js.map +1 -0
  338. package/dist/lib/db-backup.d.ts +31 -0
  339. package/dist/lib/db-backup.d.ts.map +1 -0
  340. package/dist/lib/db-backup.js +255 -0
  341. package/dist/lib/db-backup.js.map +1 -0
  342. package/dist/lib/endpoint-scope.d.ts +42 -0
  343. package/dist/lib/endpoint-scope.d.ts.map +1 -0
  344. package/dist/lib/endpoint-scope.js +95 -0
  345. package/dist/lib/endpoint-scope.js.map +1 -0
  346. package/dist/lib/env-drift.d.ts +17 -0
  347. package/dist/lib/env-drift.d.ts.map +1 -0
  348. package/dist/lib/env-drift.js +108 -0
  349. package/dist/lib/env-drift.js.map +1 -0
  350. package/dist/lib/error-classify.d.ts +40 -0
  351. package/dist/lib/error-classify.d.ts.map +1 -0
  352. package/dist/lib/error-classify.js +687 -0
  353. package/dist/lib/error-classify.js.map +1 -0
  354. package/dist/lib/error-redaction.d.ts +8 -0
  355. package/dist/lib/error-redaction.d.ts.map +1 -0
  356. package/dist/lib/error-redaction.js +45 -0
  357. package/dist/lib/error-redaction.js.map +1 -0
  358. package/dist/lib/fallback-loop.d.ts +306 -0
  359. package/dist/lib/fallback-loop.d.ts.map +1 -0
  360. package/dist/lib/fallback-loop.js +1336 -0
  361. package/dist/lib/fallback-loop.js.map +1 -0
  362. package/dist/lib/file-permissions.d.ts +98 -0
  363. package/dist/lib/file-permissions.d.ts.map +1 -0
  364. package/dist/lib/file-permissions.js +159 -0
  365. package/dist/lib/file-permissions.js.map +1 -0
  366. package/dist/lib/gemini-wire.d.ts +76 -0
  367. package/dist/lib/gemini-wire.d.ts.map +1 -0
  368. package/dist/lib/gemini-wire.js +392 -0
  369. package/dist/lib/gemini-wire.js.map +1 -0
  370. package/dist/lib/guardrails.d.ts +37 -0
  371. package/dist/lib/guardrails.d.ts.map +1 -0
  372. package/dist/lib/guardrails.js +106 -0
  373. package/dist/lib/guardrails.js.map +1 -0
  374. package/dist/lib/header-value.d.ts +8 -0
  375. package/dist/lib/header-value.d.ts.map +1 -0
  376. package/dist/lib/header-value.js +55 -0
  377. package/dist/lib/header-value.js.map +1 -0
  378. package/dist/lib/image-normalize.d.ts +21 -0
  379. package/dist/lib/image-normalize.d.ts.map +1 -0
  380. package/dist/lib/image-normalize.js +218 -0
  381. package/dist/lib/image-normalize.js.map +1 -0
  382. package/dist/lib/inbound-chat.d.ts +48 -0
  383. package/dist/lib/inbound-chat.d.ts.map +1 -0
  384. package/dist/lib/inbound-chat.js +406 -0
  385. package/dist/lib/inbound-chat.js.map +1 -0
  386. package/dist/lib/key-parser.d.ts +79 -0
  387. package/dist/lib/key-parser.d.ts.map +1 -0
  388. package/dist/lib/key-parser.js +701 -0
  389. package/dist/lib/key-parser.js.map +1 -0
  390. package/dist/lib/key-proxy.d.ts +49 -0
  391. package/dist/lib/key-proxy.d.ts.map +1 -0
  392. package/dist/lib/key-proxy.js +92 -0
  393. package/dist/lib/key-proxy.js.map +1 -0
  394. package/dist/lib/log-redaction.d.ts +28 -0
  395. package/dist/lib/log-redaction.d.ts.map +1 -0
  396. package/dist/lib/log-redaction.js +166 -0
  397. package/dist/lib/log-redaction.js.map +1 -0
  398. package/dist/lib/model-scope.d.ts +9 -0
  399. package/dist/lib/model-scope.d.ts.map +1 -0
  400. package/dist/lib/model-scope.js +29 -0
  401. package/dist/lib/model-scope.js.map +1 -0
  402. package/dist/lib/one-time-code.d.ts +15 -0
  403. package/dist/lib/one-time-code.d.ts.map +1 -0
  404. package/dist/lib/one-time-code.js +41 -0
  405. package/dist/lib/one-time-code.js.map +1 -0
  406. package/dist/lib/output-cap.d.ts +24 -0
  407. package/dist/lib/output-cap.d.ts.map +1 -0
  408. package/dist/lib/output-cap.js +80 -0
  409. package/dist/lib/output-cap.js.map +1 -0
  410. package/dist/lib/password.d.ts +3 -0
  411. package/dist/lib/password.d.ts.map +1 -0
  412. package/dist/lib/password.js +27 -0
  413. package/dist/lib/password.js.map +1 -0
  414. package/dist/lib/process-safety-net.d.ts +32 -0
  415. package/dist/lib/process-safety-net.d.ts.map +1 -0
  416. package/dist/lib/process-safety-net.js +123 -0
  417. package/dist/lib/process-safety-net.js.map +1 -0
  418. package/dist/lib/provider-identity.d.ts +37 -0
  419. package/dist/lib/provider-identity.d.ts.map +1 -0
  420. package/dist/lib/provider-identity.js +110 -0
  421. package/dist/lib/provider-identity.js.map +1 -0
  422. package/dist/lib/provider-size-parser.d.ts +6 -0
  423. package/dist/lib/provider-size-parser.d.ts.map +1 -0
  424. package/dist/lib/provider-size-parser.js +72 -0
  425. package/dist/lib/provider-size-parser.js.map +1 -0
  426. package/dist/lib/provider-timeout.d.ts +17 -0
  427. package/dist/lib/provider-timeout.d.ts.map +1 -0
  428. package/dist/lib/provider-timeout.js +70 -0
  429. package/dist/lib/provider-timeout.js.map +1 -0
  430. package/dist/lib/proxy.d.ts +180 -0
  431. package/dist/lib/proxy.d.ts.map +1 -0
  432. package/dist/lib/proxy.js +1002 -0
  433. package/dist/lib/proxy.js.map +1 -0
  434. package/dist/lib/request-log.d.ts +4 -0
  435. package/dist/lib/request-log.d.ts.map +1 -0
  436. package/dist/lib/request-log.js +161 -0
  437. package/dist/lib/request-log.js.map +1 -0
  438. package/dist/lib/reset-code.d.ts +5 -0
  439. package/dist/lib/reset-code.d.ts.map +1 -0
  440. package/dist/lib/reset-code.js +39 -0
  441. package/dist/lib/reset-code.js.map +1 -0
  442. package/dist/lib/retry-hint.d.ts +14 -0
  443. package/dist/lib/retry-hint.d.ts.map +1 -0
  444. package/dist/lib/retry-hint.js +36 -0
  445. package/dist/lib/retry-hint.js.map +1 -0
  446. package/dist/lib/route-live.d.ts +29 -0
  447. package/dist/lib/route-live.d.ts.map +1 -0
  448. package/dist/lib/route-live.js +97 -0
  449. package/dist/lib/route-live.js.map +1 -0
  450. package/dist/lib/sampling-params.d.ts +183 -0
  451. package/dist/lib/sampling-params.d.ts.map +1 -0
  452. package/dist/lib/sampling-params.js +386 -0
  453. package/dist/lib/sampling-params.js.map +1 -0
  454. package/dist/lib/scheduler.d.ts +13 -0
  455. package/dist/lib/scheduler.d.ts.map +1 -0
  456. package/dist/lib/scheduler.js +11 -0
  457. package/dist/lib/scheduler.js.map +1 -0
  458. package/dist/lib/served-model.d.ts +16 -0
  459. package/dist/lib/served-model.d.ts.map +1 -0
  460. package/dist/lib/served-model.js +0 -0
  461. package/dist/lib/served-model.js.map +1 -0
  462. package/dist/lib/server-logs.d.ts +132 -0
  463. package/dist/lib/server-logs.d.ts.map +1 -0
  464. package/dist/lib/server-logs.js +473 -0
  465. package/dist/lib/server-logs.js.map +1 -0
  466. package/dist/lib/setup-code.d.ts +5 -0
  467. package/dist/lib/setup-code.d.ts.map +1 -0
  468. package/dist/lib/setup-code.js +35 -0
  469. package/dist/lib/setup-code.js.map +1 -0
  470. package/dist/lib/structured-output.d.ts +9 -0
  471. package/dist/lib/structured-output.d.ts.map +1 -0
  472. package/dist/lib/structured-output.js +72 -0
  473. package/dist/lib/structured-output.js.map +1 -0
  474. package/dist/lib/system-prompt.d.ts +30 -0
  475. package/dist/lib/system-prompt.d.ts.map +1 -0
  476. package/dist/lib/system-prompt.js +66 -0
  477. package/dist/lib/system-prompt.js.map +1 -0
  478. package/dist/lib/task-type.d.ts +18 -0
  479. package/dist/lib/task-type.d.ts.map +1 -0
  480. package/dist/lib/task-type.js +92 -0
  481. package/dist/lib/task-type.js.map +1 -0
  482. package/dist/lib/think-tags.d.ts +37 -0
  483. package/dist/lib/think-tags.d.ts.map +1 -0
  484. package/dist/lib/think-tags.js +203 -0
  485. package/dist/lib/think-tags.js.map +1 -0
  486. package/dist/lib/tool-args.d.ts +36 -0
  487. package/dist/lib/tool-args.d.ts.map +1 -0
  488. package/dist/lib/tool-args.js +190 -0
  489. package/dist/lib/tool-args.js.map +1 -0
  490. package/dist/lib/tool-call-rescue.d.ts +62 -0
  491. package/dist/lib/tool-call-rescue.d.ts.map +1 -0
  492. package/dist/lib/tool-call-rescue.js +258 -0
  493. package/dist/lib/tool-call-rescue.js.map +1 -0
  494. package/dist/lib/tool-capability.d.ts +13 -0
  495. package/dist/lib/tool-capability.d.ts.map +1 -0
  496. package/dist/lib/tool-capability.js +70 -0
  497. package/dist/lib/tool-capability.js.map +1 -0
  498. package/dist/lib/tool-validate.d.ts +44 -0
  499. package/dist/lib/tool-validate.d.ts.map +1 -0
  500. package/dist/lib/tool-validate.js +165 -0
  501. package/dist/lib/tool-validate.js.map +1 -0
  502. package/dist/lib/ttfb-budget.d.ts +13 -0
  503. package/dist/lib/ttfb-budget.d.ts.map +1 -0
  504. package/dist/lib/ttfb-budget.js +153 -0
  505. package/dist/lib/ttfb-budget.js.map +1 -0
  506. package/dist/lib/url-guard.d.ts +47 -0
  507. package/dist/lib/url-guard.d.ts.map +1 -0
  508. package/dist/lib/url-guard.js +229 -0
  509. package/dist/lib/url-guard.js.map +1 -0
  510. package/dist/lib/wake-detect.d.ts +12 -0
  511. package/dist/lib/wake-detect.d.ts.map +1 -0
  512. package/dist/lib/wake-detect.js +99 -0
  513. package/dist/lib/wake-detect.js.map +1 -0
  514. package/dist/middleware/errorHandler.d.ts +3 -0
  515. package/dist/middleware/errorHandler.d.ts.map +1 -0
  516. package/dist/middleware/errorHandler.js +45 -0
  517. package/dist/middleware/errorHandler.js.map +1 -0
  518. package/dist/middleware/rateLimit.d.ts +4 -0
  519. package/dist/middleware/rateLimit.d.ts.map +1 -0
  520. package/dist/middleware/rateLimit.js +125 -0
  521. package/dist/middleware/rateLimit.js.map +1 -0
  522. package/dist/middleware/requireAuth.d.ts +3 -0
  523. package/dist/middleware/requireAuth.d.ts.map +1 -0
  524. package/dist/middleware/requireAuth.js +17 -0
  525. package/dist/middleware/requireAuth.js.map +1 -0
  526. package/dist/providers/aclide.d.ts +18 -0
  527. package/dist/providers/aclide.d.ts.map +1 -0
  528. package/dist/providers/aclide.js +174 -0
  529. package/dist/providers/aclide.js.map +1 -0
  530. package/dist/providers/aihorde.d.ts +42 -0
  531. package/dist/providers/aihorde.d.ts.map +1 -0
  532. package/dist/providers/aihorde.js +192 -0
  533. package/dist/providers/aihorde.js.map +1 -0
  534. package/dist/providers/airforce.d.ts +11 -0
  535. package/dist/providers/airforce.d.ts.map +1 -0
  536. package/dist/providers/airforce.js +32 -0
  537. package/dist/providers/airforce.js.map +1 -0
  538. package/dist/providers/base.d.ts +177 -0
  539. package/dist/providers/base.d.ts.map +1 -0
  540. package/dist/providers/base.js +354 -0
  541. package/dist/providers/base.js.map +1 -0
  542. package/dist/providers/blaze.d.ts +11 -0
  543. package/dist/providers/blaze.d.ts.map +1 -0
  544. package/dist/providers/blaze.js +33 -0
  545. package/dist/providers/blaze.js.map +1 -0
  546. package/dist/providers/blockrun.d.ts +14 -0
  547. package/dist/providers/blockrun.d.ts.map +1 -0
  548. package/dist/providers/blockrun.js +36 -0
  549. package/dist/providers/blockrun.js.map +1 -0
  550. package/dist/providers/clod.d.ts +11 -0
  551. package/dist/providers/clod.d.ts.map +1 -0
  552. package/dist/providers/clod.js +39 -0
  553. package/dist/providers/clod.js.map +1 -0
  554. package/dist/providers/cloudflare.d.ts +15 -0
  555. package/dist/providers/cloudflare.d.ts.map +1 -0
  556. package/dist/providers/cloudflare.js +175 -0
  557. package/dist/providers/cloudflare.js.map +1 -0
  558. package/dist/providers/cohere.d.ts +11 -0
  559. package/dist/providers/cohere.d.ts.map +1 -0
  560. package/dist/providers/cohere.js +117 -0
  561. package/dist/providers/cohere.js.map +1 -0
  562. package/dist/providers/dreamprompting.d.ts +11 -0
  563. package/dist/providers/dreamprompting.d.ts.map +1 -0
  564. package/dist/providers/dreamprompting.js +41 -0
  565. package/dist/providers/dreamprompting.js.map +1 -0
  566. package/dist/providers/electronhub.d.ts +10 -0
  567. package/dist/providers/electronhub.d.ts.map +1 -0
  568. package/dist/providers/electronhub.js +56 -0
  569. package/dist/providers/electronhub.js.map +1 -0
  570. package/dist/providers/experiential.d.ts +10 -0
  571. package/dist/providers/experiential.d.ts.map +1 -0
  572. package/dist/providers/experiential.js +33 -0
  573. package/dist/providers/experiential.js.map +1 -0
  574. package/dist/providers/gizmo.d.ts +15 -0
  575. package/dist/providers/gizmo.d.ts.map +1 -0
  576. package/dist/providers/gizmo.js +54 -0
  577. package/dist/providers/gizmo.js.map +1 -0
  578. package/dist/providers/google.d.ts +30 -0
  579. package/dist/providers/google.d.ts.map +1 -0
  580. package/dist/providers/google.js +766 -0
  581. package/dist/providers/google.js.map +1 -0
  582. package/dist/providers/index.d.ts +13 -0
  583. package/dist/providers/index.d.ts.map +1 -0
  584. package/dist/providers/index.js +561 -0
  585. package/dist/providers/index.js.map +1 -0
  586. package/dist/providers/llmtr.d.ts +15 -0
  587. package/dist/providers/llmtr.d.ts.map +1 -0
  588. package/dist/providers/llmtr.js +51 -0
  589. package/dist/providers/llmtr.js.map +1 -0
  590. package/dist/providers/logfare.d.ts +11 -0
  591. package/dist/providers/logfare.d.ts.map +1 -0
  592. package/dist/providers/logfare.js +33 -0
  593. package/dist/providers/logfare.js.map +1 -0
  594. package/dist/providers/lucidity.d.ts +11 -0
  595. package/dist/providers/lucidity.d.ts.map +1 -0
  596. package/dist/providers/lucidity.js +37 -0
  597. package/dist/providers/lucidity.js.map +1 -0
  598. package/dist/providers/modelscope.d.ts +50 -0
  599. package/dist/providers/modelscope.d.ts.map +1 -0
  600. package/dist/providers/modelscope.js +143 -0
  601. package/dist/providers/modelscope.js.map +1 -0
  602. package/dist/providers/moondream.d.ts +17 -0
  603. package/dist/providers/moondream.d.ts.map +1 -0
  604. package/dist/providers/moondream.js +140 -0
  605. package/dist/providers/moondream.js.map +1 -0
  606. package/dist/providers/openai-compat.d.ts +140 -0
  607. package/dist/providers/openai-compat.d.ts.map +1 -0
  608. package/dist/providers/openai-compat.js +537 -0
  609. package/dist/providers/openai-compat.js.map +1 -0
  610. package/dist/providers/opencode-free.d.ts +24 -0
  611. package/dist/providers/opencode-free.d.ts.map +1 -0
  612. package/dist/providers/opencode-free.js +262 -0
  613. package/dist/providers/opencode-free.js.map +1 -0
  614. package/dist/providers/pollinations.d.ts +32 -0
  615. package/dist/providers/pollinations.d.ts.map +1 -0
  616. package/dist/providers/pollinations.js +65 -0
  617. package/dist/providers/pollinations.js.map +1 -0
  618. package/dist/providers/router9.d.ts +16 -0
  619. package/dist/providers/router9.d.ts.map +1 -0
  620. package/dist/providers/router9.js +42 -0
  621. package/dist/providers/router9.js.map +1 -0
  622. package/dist/providers/sail.d.ts +44 -0
  623. package/dist/providers/sail.d.ts.map +1 -0
  624. package/dist/providers/sail.js +302 -0
  625. package/dist/providers/sail.js.map +1 -0
  626. package/dist/providers/septor.d.ts +10 -0
  627. package/dist/providers/septor.d.ts.map +1 -0
  628. package/dist/providers/septor.js +36 -0
  629. package/dist/providers/septor.js.map +1 -0
  630. package/dist/providers/speechify.d.ts +14 -0
  631. package/dist/providers/speechify.d.ts.map +1 -0
  632. package/dist/providers/speechify.js +28 -0
  633. package/dist/providers/speechify.js.map +1 -0
  634. package/dist/providers/speka.d.ts +12 -0
  635. package/dist/providers/speka.d.ts.map +1 -0
  636. package/dist/providers/speka.js +37 -0
  637. package/dist/providers/speka.js.map +1 -0
  638. package/dist/providers/waterfall.d.ts +11 -0
  639. package/dist/providers/waterfall.d.ts.map +1 -0
  640. package/dist/providers/waterfall.js +33 -0
  641. package/dist/providers/waterfall.js.map +1 -0
  642. package/dist/providers/xkiro.d.ts +71 -0
  643. package/dist/providers/xkiro.d.ts.map +1 -0
  644. package/dist/providers/xkiro.js +119 -0
  645. package/dist/providers/xkiro.js.map +1 -0
  646. package/dist/providers/zhipu.d.ts +43 -0
  647. package/dist/providers/zhipu.d.ts.map +1 -0
  648. package/dist/providers/zhipu.js +95 -0
  649. package/dist/providers/zhipu.js.map +1 -0
  650. package/dist/routes/analytics.d.ts +2 -0
  651. package/dist/routes/analytics.d.ts.map +1 -0
  652. package/dist/routes/analytics.js +738 -0
  653. package/dist/routes/analytics.js.map +1 -0
  654. package/dist/routes/anthropic.d.ts +7 -0
  655. package/dist/routes/anthropic.d.ts.map +1 -0
  656. package/dist/routes/anthropic.js +1076 -0
  657. package/dist/routes/anthropic.js.map +1 -0
  658. package/dist/routes/auth.d.ts +2 -0
  659. package/dist/routes/auth.d.ts.map +1 -0
  660. package/dist/routes/auth.js +310 -0
  661. package/dist/routes/auth.js.map +1 -0
  662. package/dist/routes/backups.d.ts +2 -0
  663. package/dist/routes/backups.d.ts.map +1 -0
  664. package/dist/routes/backups.js +127 -0
  665. package/dist/routes/backups.js.map +1 -0
  666. package/dist/routes/cache.d.ts +2 -0
  667. package/dist/routes/cache.d.ts.map +1 -0
  668. package/dist/routes/cache.js +40 -0
  669. package/dist/routes/cache.js.map +1 -0
  670. package/dist/routes/client-profiles.d.ts +2 -0
  671. package/dist/routes/client-profiles.d.ts.map +1 -0
  672. package/dist/routes/client-profiles.js +128 -0
  673. package/dist/routes/client-profiles.js.map +1 -0
  674. package/dist/routes/compression.d.ts +2 -0
  675. package/dist/routes/compression.d.ts.map +1 -0
  676. package/dist/routes/compression.js +82 -0
  677. package/dist/routes/compression.js.map +1 -0
  678. package/dist/routes/conversations.d.ts +3 -0
  679. package/dist/routes/conversations.d.ts.map +1 -0
  680. package/dist/routes/conversations.js +215 -0
  681. package/dist/routes/conversations.js.map +1 -0
  682. package/dist/routes/docs.d.ts +2 -0
  683. package/dist/routes/docs.d.ts.map +1 -0
  684. package/dist/routes/docs.js +21 -0
  685. package/dist/routes/docs.js.map +1 -0
  686. package/dist/routes/embeddings.d.ts +2 -0
  687. package/dist/routes/embeddings.d.ts.map +1 -0
  688. package/dist/routes/embeddings.js +255 -0
  689. package/dist/routes/embeddings.js.map +1 -0
  690. package/dist/routes/fallback.d.ts +2 -0
  691. package/dist/routes/fallback.d.ts.map +1 -0
  692. package/dist/routes/fallback.js +631 -0
  693. package/dist/routes/fallback.js.map +1 -0
  694. package/dist/routes/free-tier.d.ts +22 -0
  695. package/dist/routes/free-tier.d.ts.map +1 -0
  696. package/dist/routes/free-tier.js +175 -0
  697. package/dist/routes/free-tier.js.map +1 -0
  698. package/dist/routes/gemini.d.ts +2 -0
  699. package/dist/routes/gemini.d.ts.map +1 -0
  700. package/dist/routes/gemini.js +218 -0
  701. package/dist/routes/gemini.js.map +1 -0
  702. package/dist/routes/health.d.ts +2 -0
  703. package/dist/routes/health.d.ts.map +1 -0
  704. package/dist/routes/health.js +72 -0
  705. package/dist/routes/health.js.map +1 -0
  706. package/dist/routes/keys.d.ts +17 -0
  707. package/dist/routes/keys.d.ts.map +1 -0
  708. package/dist/routes/keys.js +1787 -0
  709. package/dist/routes/keys.js.map +1 -0
  710. package/dist/routes/logs.d.ts +2 -0
  711. package/dist/routes/logs.d.ts.map +1 -0
  712. package/dist/routes/logs.js +88 -0
  713. package/dist/routes/logs.js.map +1 -0
  714. package/dist/routes/mcp.d.ts +4 -0
  715. package/dist/routes/mcp.d.ts.map +1 -0
  716. package/dist/routes/mcp.js +563 -0
  717. package/dist/routes/mcp.js.map +1 -0
  718. package/dist/routes/media.d.ts +2 -0
  719. package/dist/routes/media.d.ts.map +1 -0
  720. package/dist/routes/media.js +180 -0
  721. package/dist/routes/media.js.map +1 -0
  722. package/dist/routes/models.d.ts +2 -0
  723. package/dist/routes/models.d.ts.map +1 -0
  724. package/dist/routes/models.js +439 -0
  725. package/dist/routes/models.js.map +1 -0
  726. package/dist/routes/ollama.d.ts +7 -0
  727. package/dist/routes/ollama.d.ts.map +1 -0
  728. package/dist/routes/ollama.js +595 -0
  729. package/dist/routes/ollama.js.map +1 -0
  730. package/dist/routes/profiles.d.ts +3 -0
  731. package/dist/routes/profiles.d.ts.map +1 -0
  732. package/dist/routes/profiles.js +431 -0
  733. package/dist/routes/profiles.js.map +1 -0
  734. package/dist/routes/proxy.d.ts +32 -0
  735. package/dist/routes/proxy.d.ts.map +1 -0
  736. package/dist/routes/proxy.js +2793 -0
  737. package/dist/routes/proxy.js.map +1 -0
  738. package/dist/routes/responses.d.ts +1161 -0
  739. package/dist/routes/responses.d.ts.map +1 -0
  740. package/dist/routes/responses.js +1521 -0
  741. package/dist/routes/responses.js.map +1 -0
  742. package/dist/routes/settings.d.ts +2 -0
  743. package/dist/routes/settings.d.ts.map +1 -0
  744. package/dist/routes/settings.js +549 -0
  745. package/dist/routes/settings.js.map +1 -0
  746. package/dist/routes/status.d.ts +3 -0
  747. package/dist/routes/status.d.ts.map +1 -0
  748. package/dist/routes/status.js +214 -0
  749. package/dist/routes/status.js.map +1 -0
  750. package/dist/routes/update.d.ts +31 -0
  751. package/dist/routes/update.d.ts.map +1 -0
  752. package/dist/routes/update.js +536 -0
  753. package/dist/routes/update.js.map +1 -0
  754. package/dist/routes/url-tokens.d.ts +2 -0
  755. package/dist/routes/url-tokens.d.ts.map +1 -0
  756. package/dist/routes/url-tokens.js +31 -0
  757. package/dist/routes/url-tokens.js.map +1 -0
  758. package/dist/scripts/export-catalog.d.ts +2 -0
  759. package/dist/scripts/export-catalog.d.ts.map +1 -0
  760. package/dist/scripts/export-catalog.js +114 -0
  761. package/dist/scripts/export-catalog.js.map +1 -0
  762. package/dist/scripts/rotate-encryption-key.d.ts +81 -0
  763. package/dist/scripts/rotate-encryption-key.d.ts.map +1 -0
  764. package/dist/scripts/rotate-encryption-key.js +232 -0
  765. package/dist/scripts/rotate-encryption-key.js.map +1 -0
  766. package/dist/scripts/routing-sim.d.ts +2 -0
  767. package/dist/scripts/routing-sim.d.ts.map +1 -0
  768. package/dist/scripts/routing-sim.js +130 -0
  769. package/dist/scripts/routing-sim.js.map +1 -0
  770. package/dist/scripts/test-all-models.d.ts +2 -0
  771. package/dist/scripts/test-all-models.d.ts.map +1 -0
  772. package/dist/scripts/test-all-models.js +56 -0
  773. package/dist/scripts/test-all-models.js.map +1 -0
  774. package/dist/services/anthropic-map.d.ts +39 -0
  775. package/dist/services/anthropic-map.d.ts.map +1 -0
  776. package/dist/services/anthropic-map.js +138 -0
  777. package/dist/services/anthropic-map.js.map +1 -0
  778. package/dist/services/auth.d.ts +30 -0
  779. package/dist/services/auth.d.ts.map +1 -0
  780. package/dist/services/auth.js +124 -0
  781. package/dist/services/auth.js.map +1 -0
  782. package/dist/services/auto-discover.d.ts +77 -0
  783. package/dist/services/auto-discover.d.ts.map +1 -0
  784. package/dist/services/auto-discover.js +1047 -0
  785. package/dist/services/auto-discover.js.map +1 -0
  786. package/dist/services/backups.d.ts +67 -0
  787. package/dist/services/backups.d.ts.map +1 -0
  788. package/dist/services/backups.js +444 -0
  789. package/dist/services/backups.js.map +1 -0
  790. package/dist/services/builtin-model-discovery.d.ts +107 -0
  791. package/dist/services/builtin-model-discovery.d.ts.map +1 -0
  792. package/dist/services/builtin-model-discovery.js +296 -0
  793. package/dist/services/builtin-model-discovery.js.map +1 -0
  794. package/dist/services/cache.d.ts +190 -0
  795. package/dist/services/cache.d.ts.map +1 -0
  796. package/dist/services/cache.js +617 -0
  797. package/dist/services/cache.js.map +1 -0
  798. package/dist/services/catalog-sync.d.ts +257 -0
  799. package/dist/services/catalog-sync.d.ts.map +1 -0
  800. package/dist/services/catalog-sync.js +1041 -0
  801. package/dist/services/catalog-sync.js.map +1 -0
  802. package/dist/services/compression/config.d.ts +33 -0
  803. package/dist/services/compression/config.d.ts.map +1 -0
  804. package/dist/services/compression/config.js +158 -0
  805. package/dist/services/compression/config.js.map +1 -0
  806. package/dist/services/compression/engines/aging.d.ts +2 -0
  807. package/dist/services/compression/engines/aging.d.ts.map +1 -0
  808. package/dist/services/compression/engines/aging.js +53 -0
  809. package/dist/services/compression/engines/aging.js.map +1 -0
  810. package/dist/services/compression/engines/custom-filters.d.ts +4 -0
  811. package/dist/services/compression/engines/custom-filters.d.ts.map +1 -0
  812. package/dist/services/compression/engines/custom-filters.js +49 -0
  813. package/dist/services/compression/engines/custom-filters.js.map +1 -0
  814. package/dist/services/compression/engines/dedup.d.ts +2 -0
  815. package/dist/services/compression/engines/dedup.d.ts.map +1 -0
  816. package/dist/services/compression/engines/dedup.js +61 -0
  817. package/dist/services/compression/engines/dedup.js.map +1 -0
  818. package/dist/services/compression/engines/filter-definitions.d.ts +75 -0
  819. package/dist/services/compression/engines/filter-definitions.d.ts.map +1 -0
  820. package/dist/services/compression/engines/filter-definitions.js +45 -0
  821. package/dist/services/compression/engines/filter-definitions.js.map +1 -0
  822. package/dist/services/compression/engines/hard-budget.d.ts +2 -0
  823. package/dist/services/compression/engines/hard-budget.d.ts.map +1 -0
  824. package/dist/services/compression/engines/hard-budget.js +52 -0
  825. package/dist/services/compression/engines/hard-budget.js.map +1 -0
  826. package/dist/services/compression/engines/index.d.ts +9 -0
  827. package/dist/services/compression/engines/index.d.ts.map +1 -0
  828. package/dist/services/compression/engines/index.js +9 -0
  829. package/dist/services/compression/engines/index.js.map +1 -0
  830. package/dist/services/compression/engines/jsoncompact.d.ts +3 -0
  831. package/dist/services/compression/engines/jsoncompact.d.ts.map +1 -0
  832. package/dist/services/compression/engines/jsoncompact.js +145 -0
  833. package/dist/services/compression/engines/jsoncompact.js.map +1 -0
  834. package/dist/services/compression/engines/lite.d.ts +2 -0
  835. package/dist/services/compression/engines/lite.d.ts.map +1 -0
  836. package/dist/services/compression/engines/lite.js +34 -0
  837. package/dist/services/compression/engines/lite.js.map +1 -0
  838. package/dist/services/compression/engines/read-lifecycle.d.ts +2 -0
  839. package/dist/services/compression/engines/read-lifecycle.d.ts.map +1 -0
  840. package/dist/services/compression/engines/read-lifecycle.js +70 -0
  841. package/dist/services/compression/engines/read-lifecycle.js.map +1 -0
  842. package/dist/services/compression/engines/relevance.d.ts +2 -0
  843. package/dist/services/compression/engines/relevance.d.ts.map +1 -0
  844. package/dist/services/compression/engines/relevance.js +74 -0
  845. package/dist/services/compression/engines/relevance.js.map +1 -0
  846. package/dist/services/compression/engines/toolfilter.d.ts +2 -0
  847. package/dist/services/compression/engines/toolfilter.d.ts.map +1 -0
  848. package/dist/services/compression/engines/toolfilter.js +195 -0
  849. package/dist/services/compression/engines/toolfilter.js.map +1 -0
  850. package/dist/services/compression/fidelity-gate.d.ts +12 -0
  851. package/dist/services/compression/fidelity-gate.d.ts.map +1 -0
  852. package/dist/services/compression/fidelity-gate.js +90 -0
  853. package/dist/services/compression/fidelity-gate.js.map +1 -0
  854. package/dist/services/compression/helpers.d.ts +9 -0
  855. package/dist/services/compression/helpers.d.ts.map +1 -0
  856. package/dist/services/compression/helpers.js +52 -0
  857. package/dist/services/compression/helpers.js.map +1 -0
  858. package/dist/services/compression/pipeline.d.ts +7 -0
  859. package/dist/services/compression/pipeline.d.ts.map +1 -0
  860. package/dist/services/compression/pipeline.js +245 -0
  861. package/dist/services/compression/pipeline.js.map +1 -0
  862. package/dist/services/compression/preservation.d.ts +27 -0
  863. package/dist/services/compression/preservation.d.ts.map +1 -0
  864. package/dist/services/compression/preservation.js +109 -0
  865. package/dist/services/compression/preservation.js.map +1 -0
  866. package/dist/services/compression/registry.d.ts +6 -0
  867. package/dist/services/compression/registry.d.ts.map +1 -0
  868. package/dist/services/compression/registry.js +16 -0
  869. package/dist/services/compression/registry.js.map +1 -0
  870. package/dist/services/compression/stats.d.ts +29 -0
  871. package/dist/services/compression/stats.d.ts.map +1 -0
  872. package/dist/services/compression/stats.js +55 -0
  873. package/dist/services/compression/stats.js.map +1 -0
  874. package/dist/services/compression/types.d.ts +90 -0
  875. package/dist/services/compression/types.d.ts.map +1 -0
  876. package/dist/services/compression/types.js +2 -0
  877. package/dist/services/compression/types.js.map +1 -0
  878. package/dist/services/context-handoff.d.ts +22 -0
  879. package/dist/services/context-handoff.d.ts.map +1 -0
  880. package/dist/services/context-handoff.js +164 -0
  881. package/dist/services/context-handoff.js.map +1 -0
  882. package/dist/services/cooldown-probe.d.ts +24 -0
  883. package/dist/services/cooldown-probe.d.ts.map +1 -0
  884. package/dist/services/cooldown-probe.js +181 -0
  885. package/dist/services/cooldown-probe.js.map +1 -0
  886. package/dist/services/custom-endpoint.d.ts +58 -0
  887. package/dist/services/custom-endpoint.d.ts.map +1 -0
  888. package/dist/services/custom-endpoint.js +167 -0
  889. package/dist/services/custom-endpoint.js.map +1 -0
  890. package/dist/services/custom-media-register.d.ts +22 -0
  891. package/dist/services/custom-media-register.d.ts.map +1 -0
  892. package/dist/services/custom-media-register.js +45 -0
  893. package/dist/services/custom-media-register.js.map +1 -0
  894. package/dist/services/custom-model-register.d.ts +50 -0
  895. package/dist/services/custom-model-register.d.ts.map +1 -0
  896. package/dist/services/custom-model-register.js +103 -0
  897. package/dist/services/custom-model-register.js.map +1 -0
  898. package/dist/services/custom-model-seed.d.ts +18 -0
  899. package/dist/services/custom-model-seed.d.ts.map +1 -0
  900. package/dist/services/custom-model-seed.js +42 -0
  901. package/dist/services/custom-model-seed.js.map +1 -0
  902. package/dist/services/custom-model-sync.d.ts +37 -0
  903. package/dist/services/custom-model-sync.d.ts.map +1 -0
  904. package/dist/services/custom-model-sync.js +118 -0
  905. package/dist/services/custom-model-sync.js.map +1 -0
  906. package/dist/services/custom-model-tombstone.d.ts +5 -0
  907. package/dist/services/custom-model-tombstone.d.ts.map +1 -0
  908. package/dist/services/custom-model-tombstone.js +30 -0
  909. package/dist/services/custom-model-tombstone.js.map +1 -0
  910. package/dist/services/declarative-config.d.ts +356 -0
  911. package/dist/services/declarative-config.d.ts.map +1 -0
  912. package/dist/services/declarative-config.js +427 -0
  913. package/dist/services/declarative-config.js.map +1 -0
  914. package/dist/services/degradation.d.ts +40 -0
  915. package/dist/services/degradation.d.ts.map +1 -0
  916. package/dist/services/degradation.js +119 -0
  917. package/dist/services/degradation.js.map +1 -0
  918. package/dist/services/embeddings.d.ts +76 -0
  919. package/dist/services/embeddings.d.ts.map +1 -0
  920. package/dist/services/embeddings.js +319 -0
  921. package/dist/services/embeddings.js.map +1 -0
  922. package/dist/services/fusion.d.ts +162 -0
  923. package/dist/services/fusion.d.ts.map +1 -0
  924. package/dist/services/fusion.js +789 -0
  925. package/dist/services/fusion.js.map +1 -0
  926. package/dist/services/gemini-map.d.ts +14 -0
  927. package/dist/services/gemini-map.d.ts.map +1 -0
  928. package/dist/services/gemini-map.js +78 -0
  929. package/dist/services/gemini-map.js.map +1 -0
  930. package/dist/services/health.d.ts +52 -0
  931. package/dist/services/health.d.ts.map +1 -0
  932. package/dist/services/health.js +352 -0
  933. package/dist/services/health.js.map +1 -0
  934. package/dist/services/idempotency.d.ts +49 -0
  935. package/dist/services/idempotency.d.ts.map +1 -0
  936. package/dist/services/idempotency.js +140 -0
  937. package/dist/services/idempotency.js.map +1 -0
  938. package/dist/services/key-budget.d.ts +74 -0
  939. package/dist/services/key-budget.d.ts.map +1 -0
  940. package/dist/services/key-budget.js +200 -0
  941. package/dist/services/key-budget.js.map +1 -0
  942. package/dist/services/media.d.ts +123 -0
  943. package/dist/services/media.d.ts.map +1 -0
  944. package/dist/services/media.js +954 -0
  945. package/dist/services/media.js.map +1 -0
  946. package/dist/services/model-discovery.d.ts +126 -0
  947. package/dist/services/model-discovery.d.ts.map +1 -0
  948. package/dist/services/model-discovery.js +662 -0
  949. package/dist/services/model-discovery.js.map +1 -0
  950. package/dist/services/model-groups.d.ts +149 -0
  951. package/dist/services/model-groups.d.ts.map +1 -0
  952. package/dist/services/model-groups.js +314 -0
  953. package/dist/services/model-groups.js.map +1 -0
  954. package/dist/services/model-listing.d.ts +18 -0
  955. package/dist/services/model-listing.d.ts.map +1 -0
  956. package/dist/services/model-listing.js +94 -0
  957. package/dist/services/model-listing.js.map +1 -0
  958. package/dist/services/model-retirement.d.ts +29 -0
  959. package/dist/services/model-retirement.d.ts.map +1 -0
  960. package/dist/services/model-retirement.js +86 -0
  961. package/dist/services/model-retirement.js.map +1 -0
  962. package/dist/services/model-state.d.ts +90 -0
  963. package/dist/services/model-state.d.ts.map +1 -0
  964. package/dist/services/model-state.js +342 -0
  965. package/dist/services/model-state.js.map +1 -0
  966. package/dist/services/model-weight-overrides.d.ts +48 -0
  967. package/dist/services/model-weight-overrides.d.ts.map +1 -0
  968. package/dist/services/model-weight-overrides.js +127 -0
  969. package/dist/services/model-weight-overrides.js.map +1 -0
  970. package/dist/services/notification-emitter.d.ts +50 -0
  971. package/dist/services/notification-emitter.d.ts.map +1 -0
  972. package/dist/services/notification-emitter.js +177 -0
  973. package/dist/services/notification-emitter.js.map +1 -0
  974. package/dist/services/opencode-free-sync.d.ts +29 -0
  975. package/dist/services/opencode-free-sync.d.ts.map +1 -0
  976. package/dist/services/opencode-free-sync.js +206 -0
  977. package/dist/services/opencode-free-sync.js.map +1 -0
  978. package/dist/services/penalty-inspector.d.ts +56 -0
  979. package/dist/services/penalty-inspector.d.ts.map +1 -0
  980. package/dist/services/penalty-inspector.js +167 -0
  981. package/dist/services/penalty-inspector.js.map +1 -0
  982. package/dist/services/profile-models.d.ts +5 -0
  983. package/dist/services/profile-models.d.ts.map +1 -0
  984. package/dist/services/profile-models.js +62 -0
  985. package/dist/services/profile-models.js.map +1 -0
  986. package/dist/services/provider-credential.d.ts +19 -0
  987. package/dist/services/provider-credential.d.ts.map +1 -0
  988. package/dist/services/provider-credential.js +51 -0
  989. package/dist/services/provider-credential.js.map +1 -0
  990. package/dist/services/provider-quota.d.ts +64 -0
  991. package/dist/services/provider-quota.d.ts.map +1 -0
  992. package/dist/services/provider-quota.js +563 -0
  993. package/dist/services/provider-quota.js.map +1 -0
  994. package/dist/services/quirks.d.ts +0 -0
  995. package/dist/services/quirks.d.ts.map +1 -0
  996. package/dist/services/quirks.js +0 -0
  997. package/dist/services/quirks.js.map +1 -0
  998. package/dist/services/quota-forecast.d.ts +50 -0
  999. package/dist/services/quota-forecast.d.ts.map +1 -0
  1000. package/dist/services/quota-forecast.js +150 -0
  1001. package/dist/services/quota-forecast.js.map +1 -0
  1002. package/dist/services/quota-outlook.d.ts +4 -0
  1003. package/dist/services/quota-outlook.d.ts.map +1 -0
  1004. package/dist/services/quota-outlook.js +92 -0
  1005. package/dist/services/quota-outlook.js.map +1 -0
  1006. package/dist/services/ratelimit.d.ts +249 -0
  1007. package/dist/services/ratelimit.d.ts.map +1 -0
  1008. package/dist/services/ratelimit.js +1182 -0
  1009. package/dist/services/ratelimit.js.map +1 -0
  1010. package/dist/services/request-retention.d.ts +60 -0
  1011. package/dist/services/request-retention.d.ts.map +1 -0
  1012. package/dist/services/request-retention.js +229 -0
  1013. package/dist/services/request-retention.js.map +1 -0
  1014. package/dist/services/router.d.ts +384 -0
  1015. package/dist/services/router.d.ts.map +1 -0
  1016. package/dist/services/router.js +1986 -0
  1017. package/dist/services/router.js.map +1 -0
  1018. package/dist/services/scoring.d.ts +156 -0
  1019. package/dist/services/scoring.d.ts.map +1 -0
  1020. package/dist/services/scoring.js +435 -0
  1021. package/dist/services/scoring.js.map +1 -0
  1022. package/dist/services/url-tokens.d.ts +15 -0
  1023. package/dist/services/url-tokens.d.ts.map +1 -0
  1024. package/dist/services/url-tokens.js +55 -0
  1025. package/dist/services/url-tokens.js.map +1 -0
  1026. package/package.json +77 -4
  1027. package/README.md +0 -3
@@ -0,0 +1,1986 @@
1
+ import { getDb, getSetting, setSetting } from '../db/index.js';
2
+ import { getProvider, hasProvider, resolveProvider } from '../providers/index.js';
3
+ import { decrypt } from '../lib/crypto.js';
4
+ import { decryptProxyUrl } from '../lib/key-proxy.js';
5
+ import { canMakeRequest, canUseTokens, isOnCooldown, canUseProvider, canUseProviderMinute, canUseProviderTokens, canUseKeyConcurrency, acquireLease, releaseLease, getSoonestCooldownExpiry, modelWindowUsedFraction, } from './ratelimit.js';
6
+ import { BANDIT_PRESETS, DEFAULT_STRATEGY, reliabilityPosterior, expectedReliability, sampleBeta, speedScore, intelligenceScore, intelligenceComposite, headroomFactor, rateWindowHeadroomFactor, rateLimitFactor, combineScore, peakAdjustedWeights, taskAdjustedWeights, TASK_WEIGHT_SHARE, isValidPeakHour, isValidTimezone, DEFAULT_PEAK_HOURS, observedSpeedRank, TIMEOUT_LATENCY_CAP_MS, } from './scoring.js';
7
+ import { TIMEOUT_ERROR_MARKERS } from '../lib/error-classify.js';
8
+ import { checkMonthlyBudget, reserveMonthlyBudget } from './key-budget.js';
9
+ import { applyModelWeightOverride, getModelWeightOverrides } from './model-weight-overrides.js';
10
+ import { modelsWithOverriddenField } from './model-state.js';
11
+ import { parseBudget } from '../lib/budget.js';
12
+ import { platformDropsResponseFormat } from '../lib/sampling-params.js';
13
+ import { isUnifyEnabled, getModelGroups, resolveRequestedIdForDispatch } from './model-groups.js';
14
+ import { getActiveProfileId } from './profile-models.js';
15
+ import { customEndpointKeyIds } from './custom-endpoint.js';
16
+ import { isDegraded } from './degradation.js';
17
+ import { modelStatsKey, endpointScopeForBaseUrl } from '../lib/endpoint-scope.js';
18
+ import { isToolBenched } from '../lib/tool-capability.js';
19
+ import { parseModelScope, scopeAllows } from '../lib/model-scope.js';
20
+ import { getKeyQuotaHeadroom, inferQuotaPoolKey } from './provider-quota.js';
21
+ class RouteError extends Error {
22
+ status;
23
+ // Per-model disposition of the chain at the moment routing gave up: one line
24
+ // per considered model with the reason it could not serve (no key, cooldown,
25
+ // provider cap, rpm/rpd, tpm/tpd, context too small, …). Populated only on the
26
+ // synchronous "all exhausted" throw, where NO upstream was tried and nothing
27
+ // else logs WHY the pool was empty (issue _1: opaque routing_error 429).
28
+ diagnostics;
29
+ constructor(message, status, diagnostics) {
30
+ super(message);
31
+ this.status = status;
32
+ this.diagnostics = diagnostics;
33
+ }
34
+ }
35
+ // Human-readable retry ETA from a cooldown expiry timestamp (#423). Null when
36
+ // nothing is cooling down or it already lapsed.
37
+ export function formatResetEta(soonestResetMs, now = Date.now()) {
38
+ if (soonestResetMs == null)
39
+ return null;
40
+ const deltaMs = soonestResetMs - now;
41
+ if (deltaMs <= 0)
42
+ return null;
43
+ const secs = Math.round(deltaMs / 1000);
44
+ if (secs < 90)
45
+ return `~${secs}s`;
46
+ const mins = Math.round(secs / 60);
47
+ if (mins < 90)
48
+ return `~${mins}m`;
49
+ return `~${Math.round(mins / 60)}h`;
50
+ }
51
+ const EXHAUSTION_ADVICE = 'Add more API keys or wait for rate limits to reset.';
52
+ // Roll the per-model diagnostics (see RouteError.diagnostics) up into a short,
53
+ // client-safe summary so an exhausted caller learns WHY the pool was empty
54
+ // instead of a bare "All models exhausted" (#423). Buckets are aggregate
55
+ // counts only — no key material, no per-key detail. Classifies off the whole
56
+ // line (model ids can contain ':' so splitting label from reason is unsafe).
57
+ export function summarizeExhaustion(diag, soonestResetMs, now = Date.now(), keylessSkipped = 0) {
58
+ const eta = formatResetEta(soonestResetMs, now);
59
+ const etaSuffix = eta ? ` Soonest reset ${eta}.` : '';
60
+ // Models dropped before the walk (#423 follow-up) are reported separately and
61
+ // never counted as "routes checked": they were never candidates, and folding
62
+ // them into the total inflates the pool the caller thinks it has.
63
+ const keylessSuffix = keylessSkipped > 0
64
+ ? ` ${keylessSkipped} model${keylessSkipped === 1 ? '' : 's'} skipped: no key configured for their platform.`
65
+ : '';
66
+ if (!diag || diag.length === 0) {
67
+ return `All models exhausted. ${EXHAUSTION_ADVICE}${etaSuffix}${keylessSuffix}`;
68
+ }
69
+ const counts = {};
70
+ const bump = (bucket) => { counts[bucket] = (counts[bucket] ?? 0) + 1; };
71
+ for (const line of diag) {
72
+ const l = line.toLowerCase();
73
+ if (l.includes('no provider registered'))
74
+ bump('unsupported provider');
75
+ else if (/no enabled\+healthy key|no usable key|decrypt-error/.test(l))
76
+ bump('no usable key configured');
77
+ else if (l.includes('< estimated'))
78
+ bump('prompt too large for the model');
79
+ else if (l.includes('no vision support'))
80
+ bump('model lacks vision');
81
+ else if (l.includes('no tool-calling support'))
82
+ bump('model lacks tool-calling');
83
+ else if (l.includes('drops response_format'))
84
+ bump('platform cannot honor response_format');
85
+ else if (/ruled out|already-failed/.test(l))
86
+ bump('failed earlier this request');
87
+ else if (/cooldown|rpm|rpd|tpm|tpd|provider-daily-cap/.test(l))
88
+ bump('rate-limited or on cooldown');
89
+ else
90
+ bump('unavailable');
91
+ }
92
+ // Most actionable buckets first.
93
+ const order = [
94
+ 'rate-limited or on cooldown',
95
+ 'no usable key configured',
96
+ 'prompt too large for the model',
97
+ 'model lacks vision',
98
+ 'model lacks tool-calling',
99
+ 'platform cannot honor response_format',
100
+ 'failed earlier this request',
101
+ 'unsupported provider',
102
+ 'unavailable',
103
+ ];
104
+ const parts = order.filter(b => counts[b]).map(b => `${counts[b]} ${b}`);
105
+ const total = diag.length;
106
+ return `All models exhausted: ${total} route${total === 1 ? '' : 's'} checked (${parts.join(', ')}). ${EXHAUSTION_ADVICE}${etaSuffix}${keylessSuffix}`;
107
+ }
108
+ // ── Routing token estimate: cap the reserved OUTPUT, not the full max_tokens ──
109
+ // A client can request a huge max_tokens (e.g. 32000) it will never actually
110
+ // emit. Reserving that full amount against every model's context window and TPM
111
+ // budget falsely excludes the entire free pool (TPM 6k-30k) and returns a bogus
112
+ // "all models exhausted" 429 with ZERO upstream calls (#470). Providers meter
113
+ // ACTUAL tokens, so under-reserving only risks an upstream 429/413 the retry
114
+ // loop already handles, whereas over-reserving starves routing. For routing and
115
+ // the context-window / TPM filters we therefore reserve at most this many output
116
+ // tokens; the INPUT estimate is still counted in full so a genuinely large
117
+ // prompt still (correctly) skips a too-small model.
118
+ export const OUTPUT_RESERVE_CAP = 2000;
119
+ /**
120
+ * Output tokens to reserve for routing/filter purposes: the requested max_tokens
121
+ * clamped to OUTPUT_RESERVE_CAP (default 1000 when the client omitted it, matching
122
+ * the historical fallback). Callers add this to their INPUT estimate before
123
+ * calling routeRequest / routePinnedModel.
124
+ */
125
+ export function routingReserveTokens(requestedMaxTokens) {
126
+ const requested = requestedMaxTokens != null && requestedMaxTokens > 0 ? requestedMaxTokens : 1000;
127
+ return Math.min(requested, OUTPUT_RESERVE_CAP);
128
+ }
129
+ // Round-robin index per platform
130
+ const roundRobinIndex = new Map();
131
+ // ── Dynamic priority: track 429s per model and demote accordingly ──
132
+ // Key: model_db_id → { count, lastHit, penalty }
133
+ const rateLimitPenalties = new Map();
134
+ // Penalty decays over time so models recover
135
+ const PENALTY_PER_429 = 3; // each 429 adds this many priority positions
136
+ const PENALTY_PER_FAIL = 1; // each non-limit upstream failure (5xx/timeout/empty stream)
137
+ const MAX_PENALTY = 10; // cap so a model doesn't sink forever
138
+ const DECAY_INTERVAL_MS = 2 * 60 * 1000; // penalty decays every 2 minutes
139
+ const DECAY_AMOUNT = 1; // remove this much penalty per decay interval
140
+ /**
141
+ * Record an upstream failure for a model — increases its penalty so it sinks in
142
+ * priority. Default weight is the LIGHT one for ordinary upstream failures
143
+ * (5xx/timeout/empty stream, +1); callers that know they saw a hard limit
144
+ * signal (429/402) pass the heavier weight — a quota limit is the stronger,
145
+ * longer-lived health cue.
146
+ */
147
+ export function recordModelFailure(modelDbId, weight = PENALTY_PER_FAIL) {
148
+ const existing = rateLimitPenalties.get(modelDbId);
149
+ const now = Date.now();
150
+ if (existing) {
151
+ const decaySteps = Math.floor((now - existing.lastHit) / DECAY_INTERVAL_MS);
152
+ existing.penalty = Math.max(0, existing.penalty - decaySteps * DECAY_AMOUNT);
153
+ existing.count++;
154
+ existing.lastHit = now;
155
+ existing.penalty = Math.min(existing.penalty + weight, MAX_PENALTY);
156
+ }
157
+ else {
158
+ rateLimitPenalties.set(modelDbId, { count: 1, lastHit: now, penalty: weight });
159
+ }
160
+ }
161
+ /**
162
+ * Record a 429 for a model — heavier penalty (priority demotion) than ordinary
163
+ * failures, since a rate-limit signal is the strongest short-term health cue.
164
+ */
165
+ export function recordRateLimitHit(modelDbId) {
166
+ recordModelFailure(modelDbId, PENALTY_PER_429);
167
+ }
168
+ /**
169
+ * Record a success for a model — reduces its penalty so it rises back up.
170
+ */
171
+ export function recordSuccess(modelDbId) {
172
+ const existing = rateLimitPenalties.get(modelDbId);
173
+ if (existing) {
174
+ existing.penalty = Math.max(0, existing.penalty - 1);
175
+ if (existing.penalty === 0) {
176
+ rateLimitPenalties.delete(modelDbId);
177
+ }
178
+ }
179
+ }
180
+ /**
181
+ * Get the current penalty for a model (with time-based decay).
182
+ * Pure read — does not mutate the entry; decay is applied lazily only when
183
+ * recording a new hit (recordRateLimitHit) so the clock isn't reset on every
184
+ * routing call.
185
+ */
186
+ function getPenalty(modelDbId) {
187
+ const entry = rateLimitPenalties.get(modelDbId);
188
+ if (!entry)
189
+ return 0;
190
+ const elapsed = Date.now() - entry.lastHit;
191
+ const decaySteps = Math.floor(elapsed / DECAY_INTERVAL_MS);
192
+ const decayed = Math.max(0, entry.penalty - decaySteps * DECAY_AMOUNT);
193
+ if (decayed === 0) {
194
+ rateLimitPenalties.delete(modelDbId);
195
+ return 0;
196
+ }
197
+ return decayed;
198
+ }
199
+ /**
200
+ * Get current penalties for all models (for the API/dashboard).
201
+ */
202
+ export function getAllPenalties() {
203
+ const result = [];
204
+ for (const [modelDbId, entry] of rateLimitPenalties) {
205
+ const penalty = getPenalty(modelDbId);
206
+ if (penalty > 0) {
207
+ result.push({ modelDbId, count: entry.count, penalty });
208
+ }
209
+ }
210
+ return result.sort((a, b) => b.penalty - a.penalty);
211
+ }
212
+ /**
213
+ * Operator clear (#952): forget every model's penalty at once and report how
214
+ * many models were carrying one. Pairs with clearAllCooldowns — a pool stuck
215
+ * behind day-long benches also has its models sunk by penalties, and lifting
216
+ * one without the other leaves the router still avoiding them.
217
+ */
218
+ export function clearAllPenalties() {
219
+ const count = getAllPenalties().length;
220
+ rateLimitPenalties.clear();
221
+ return count;
222
+ }
223
+ // ── Routing strategy (persisted) ────────────────────────────────────────────
224
+ const STRATEGY_KEY = 'routing_strategy';
225
+ const CUSTOM_WEIGHTS_KEY = 'routing_custom_weights';
226
+ const EXPLORE_KEY = 'routing_explore_enabled';
227
+ const PEAK_ADJUST_KEY = 'routing_peak_hours_adjust';
228
+ const PEAK_START_KEY = 'routing_peak_start_hour';
229
+ const PEAK_END_KEY = 'routing_peak_end_hour';
230
+ const PEAK_TZ_KEY = 'routing_peak_timezone';
231
+ const COMMUNITY_PRIOR_KEY = 'routing_community_prior';
232
+ const COMMUNITY_PRIOR_ENABLED_KEY = 'routing_community_prior_enabled';
233
+ // Headroom guardrail thresholds (#899): the remaining-budget fraction at which
234
+ // demotion begins and the score floor at 0 remaining. Stored as decimals
235
+ // (0.2 = 20%). Absent/invalid values fall back to the scoring.ts constants so
236
+ // existing installs are untouched.
237
+ export const HEADROOM_RAMP_START_KEY = 'routing_headroom_ramp_start';
238
+ export const HEADROOM_FLOOR_KEY = 'routing_headroom_floor';
239
+ export function getHeadroomThresholds() {
240
+ const read = (key) => {
241
+ const raw = getSetting(key);
242
+ if (raw === undefined || raw.trim() === '')
243
+ return undefined;
244
+ const n = Number(raw);
245
+ return Number.isFinite(n) && n >= 0 && n <= 1 ? n : undefined;
246
+ };
247
+ return { rampStart: read(HEADROOM_RAMP_START_KEY), floor: read(HEADROOM_FLOOR_KEY) };
248
+ }
249
+ // null clears a threshold back to the default; undefined leaves it untouched.
250
+ export function setHeadroomThresholds(rampStart, floor) {
251
+ const db = getDb();
252
+ const apply = (key, value) => {
253
+ if (value === undefined)
254
+ return;
255
+ if (value === null) {
256
+ db.prepare('DELETE FROM settings WHERE key = ?').run(key); // back to default
257
+ return;
258
+ }
259
+ if (!Number.isFinite(value) || value < 0 || value > 1) {
260
+ throw new Error(`Invalid value ${value} for ${key} (must be 0..1)`);
261
+ }
262
+ setSetting(key, String(value));
263
+ };
264
+ apply(HEADROOM_RAMP_START_KEY, rampStart);
265
+ apply(HEADROOM_FLOOR_KEY, floor);
266
+ }
267
+ // ── Task-type weight share (persisted) ─────────────────────────────────────
268
+ // #1127 follow-up: the bandit bias applied for a declared/derived task type
269
+ // moves `share` of one axis onto the other (code: speed → intelligence; chat:
270
+ // the reverse). The default matches the scoring.ts constant; operators can
271
+ // tune it 0..1 (0 = bias disabled) without a code change. Absent/invalid
272
+ // values fall back to the constant so existing installs are untouched.
273
+ export const TASK_WEIGHT_SHARE_KEY = 'routing_task_weight_share';
274
+ export function getTaskWeightShare() {
275
+ const raw = getSetting(TASK_WEIGHT_SHARE_KEY);
276
+ if (raw === undefined || raw.trim() === '')
277
+ return TASK_WEIGHT_SHARE;
278
+ const n = Number(raw);
279
+ return Number.isFinite(n) && n >= 0 && n <= 1 ? n : TASK_WEIGHT_SHARE;
280
+ }
281
+ // null clears back to the default; a value outside 0..1 throws.
282
+ export function setTaskWeightShare(value) {
283
+ if (value === null) {
284
+ getDb().prepare('DELETE FROM settings WHERE key = ?').run(TASK_WEIGHT_SHARE_KEY);
285
+ return;
286
+ }
287
+ if (!Number.isFinite(value) || value < 0 || value > 1) {
288
+ throw new Error(`Invalid value ${value} for ${TASK_WEIGHT_SHARE_KEY} (must be 0..1)`);
289
+ }
290
+ setSetting(TASK_WEIGHT_SHARE_KEY, String(value));
291
+ }
292
+ /** Chance per request that an unmeasured model gets tried first when the
293
+ * exploration toggle is on. The bandit's Thompson sampling already explores
294
+ * automatically; this guarantees a floor so models with no reliability/speed
295
+ * data still get sampled instead of being starved by prior-heavy rivals. */
296
+ export const EXPLORE_CHANCE = 0.1;
297
+ /** A model counts as "has data" once its decay-weighted success+failure
298
+ * pseudo-count reaches this many samples. */
299
+ export const EXPLORE_MIN_SAMPLES = 5;
300
+ const VALID_STRATEGIES = ['priority', 'balanced', 'smartest', 'fastest', 'reliable', 'custom'];
301
+ export function getRoutingStrategy() {
302
+ const raw = getSetting(STRATEGY_KEY);
303
+ return (raw && VALID_STRATEGIES.includes(raw))
304
+ ? raw
305
+ : DEFAULT_STRATEGY;
306
+ }
307
+ export function setRoutingStrategy(strategy) {
308
+ if (!VALID_STRATEGIES.includes(strategy)) {
309
+ throw new Error(`Unknown routing strategy: ${strategy}`);
310
+ }
311
+ setSetting(STRATEGY_KEY, strategy);
312
+ }
313
+ // ── Exploration toggle (persisted) ─────────────────────────────────────────
314
+ // On by default: unmeasured models get a guaranteed chance to be tried
315
+ // (EXPLORE_CHANCE) so they acquire reliability/speed samples instead of
316
+ // losing every bandit draw. Set to '0' to opt out from the dashboard.
317
+ export function getExploreEnabled() {
318
+ return getSetting(EXPLORE_KEY) !== '0';
319
+ }
320
+ export function setExploreEnabled(enabled) {
321
+ setSetting(EXPLORE_KEY, enabled ? '1' : '0');
322
+ }
323
+ // ── Peak-hours adjustment (persisted, off by default) ──────────────────────
324
+ // Opt-in time-of-day reweighting (#760). Everything about it is operator-set:
325
+ // whether it runs at all, the window, and the timezone the window is read in.
326
+ // With the flag off, weightsFor returns the presets byte-for-byte, so an
327
+ // install that never touches this setting routes exactly as it did before.
328
+ export function getPeakHoursConfig() {
329
+ const startRaw = Number.parseInt(getSetting(PEAK_START_KEY) ?? '', 10);
330
+ const endRaw = Number.parseInt(getSetting(PEAK_END_KEY) ?? '', 10);
331
+ const tzRaw = getSetting(PEAK_TZ_KEY);
332
+ return {
333
+ enabled: getSetting(PEAK_ADJUST_KEY) === '1',
334
+ startHour: isValidPeakHour(startRaw) ? startRaw : DEFAULT_PEAK_HOURS.startHour,
335
+ endHour: isValidPeakHour(endRaw) ? endRaw : DEFAULT_PEAK_HOURS.endHour,
336
+ timezone: isValidTimezone(tzRaw) ? tzRaw : DEFAULT_PEAK_HOURS.timezone,
337
+ };
338
+ }
339
+ /** Persist any subset of the peak-hours settings. Throws on an out-of-range
340
+ * hour or an unknown IANA timezone so a bad PUT is rejected at the API rather
341
+ * than silently stored and then ignored on read. */
342
+ export function setPeakHoursConfig(patch) {
343
+ if (patch.startHour !== undefined && !isValidPeakHour(patch.startHour)) {
344
+ throw new Error('peakStartHour must be an integer between 0 and 23');
345
+ }
346
+ if (patch.endHour !== undefined && !isValidPeakHour(patch.endHour)) {
347
+ throw new Error('peakEndHour must be an integer between 0 and 23');
348
+ }
349
+ if (patch.timezone !== undefined && !isValidTimezone(patch.timezone)) {
350
+ throw new Error('peakTimezone must be a valid IANA timezone name');
351
+ }
352
+ if (patch.enabled !== undefined)
353
+ setSetting(PEAK_ADJUST_KEY, patch.enabled ? '1' : '0');
354
+ if (patch.startHour !== undefined)
355
+ setSetting(PEAK_START_KEY, String(patch.startHour));
356
+ if (patch.endHour !== undefined)
357
+ setSetting(PEAK_END_KEY, String(patch.endHour));
358
+ if (patch.timezone !== undefined)
359
+ setSetting(PEAK_TZ_KEY, patch.timezone);
360
+ }
361
+ // ── Key selection strategy (persisted) ─────────────────────────────────────
362
+ // Which of a platform's several keys to reach for, once a model has been
363
+ // picked. Independent of the routing strategy on purpose (#919): the strategy
364
+ // enum drives the MODEL bandit, so putting a key policy in it would make
365
+ // choosing a key policy also throw away the model ranking.
366
+ // 'auto' — unchanged: per-key bandit score when there is data,
367
+ // round-robin otherwise.
368
+ // 'least-remaining' — additionally rank by observed remaining quota, roomiest
369
+ // key first, so the key closest to its cap is held back
370
+ // instead of being the next one to 429.
371
+ const KEY_SELECTION_KEY = 'key_selection_strategy';
372
+ const VALID_KEY_SELECTIONS = ['auto', 'least-remaining'];
373
+ export const DEFAULT_KEY_SELECTION = 'auto';
374
+ export function getKeySelectionStrategy() {
375
+ const raw = getSetting(KEY_SELECTION_KEY);
376
+ return (raw && VALID_KEY_SELECTIONS.includes(raw))
377
+ ? raw
378
+ : DEFAULT_KEY_SELECTION;
379
+ }
380
+ export function setKeySelectionStrategy(strategy) {
381
+ if (!VALID_KEY_SELECTIONS.includes(strategy)) {
382
+ throw new Error(`Unknown key selection strategy: ${strategy}`);
383
+ }
384
+ setSetting(KEY_SELECTION_KEY, strategy);
385
+ }
386
+ // ── Custom weights (persisted) ──────────────────────────────────────────────
387
+ // User-tuned weight vector for the 'custom' strategy. Stored normalized (sums
388
+ // to 1) so the dashboard percentages read cleanly; combineScore would tolerate
389
+ // any non-negative vector regardless. Falls back to the balanced preset until
390
+ // the user has saved their own.
391
+ export function getCustomWeights() {
392
+ const raw = getSetting(CUSTOM_WEIGHTS_KEY);
393
+ if (raw) {
394
+ try {
395
+ const w = JSON.parse(raw);
396
+ if ([w.reliability, w.speed, w.intelligence].every(v => Number.isFinite(v) && v >= 0) &&
397
+ w.reliability + w.speed + w.intelligence > 0) {
398
+ return { reliability: w.reliability, speed: w.speed, intelligence: w.intelligence };
399
+ }
400
+ }
401
+ catch { /* corrupt setting → fall through to default */ }
402
+ }
403
+ return { ...BANDIT_PRESETS.balanced };
404
+ }
405
+ export function setCustomWeights(weights) {
406
+ const { reliability, speed, intelligence } = weights;
407
+ if (![reliability, speed, intelligence].every(v => Number.isFinite(v) && v >= 0)) {
408
+ throw new Error('Custom weights must be non-negative numbers');
409
+ }
410
+ const sum = reliability + speed + intelligence;
411
+ if (sum <= 0) {
412
+ throw new Error('Custom weights must not all be zero');
413
+ }
414
+ setSetting(CUSTOM_WEIGHTS_KEY, JSON.stringify({
415
+ reliability: reliability / sum,
416
+ speed: speed / sum,
417
+ intelligence: intelligence / sum,
418
+ }));
419
+ }
420
+ /** Ceiling on a single prior's effective sample size. Local counts are
421
+ * decay-weighted (2-day half-life — a busy install still only carries on the
422
+ * order of a hundred effective samples), so an unbounded, undecayed community
423
+ * count would drown local evidence forever and collapse the Thompson-sampling
424
+ * variance to zero. Capping at ~50 pseudo-observations keeps a prior worth
425
+ * roughly half the local evidence at most: enough to seed a brand-new model,
426
+ * cheap for real local traffic to override. */
427
+ export const COMMUNITY_PRIOR_MAX_SAMPLES = 50;
428
+ /** Validate a raw prior map and cap each entry's effective sample size.
429
+ * Shared by the read path and the write path so a value is bounded no matter
430
+ * how it entered (fresh set, legacy stored blob, hand-edited settings row).
431
+ * Invalid entries (negative, all-zero, no ':') are dropped; oversized ones
432
+ * are rescaled preserving the success/failure ratio (980/20 → 49/1). */
433
+ function sanitizeCommunityPriors(priors) {
434
+ const clean = {};
435
+ if (!priors || typeof priors !== 'object')
436
+ return clean;
437
+ for (const [key, v] of Object.entries(priors)) {
438
+ if (key.includes(':') &&
439
+ v && typeof v === 'object' &&
440
+ Number.isFinite(v.successes) && v.successes >= 0 &&
441
+ Number.isFinite(v.failures) && v.failures >= 0 &&
442
+ v.successes + v.failures > 0) {
443
+ const total = v.successes + v.failures;
444
+ const scale = total > COMMUNITY_PRIOR_MAX_SAMPLES ? COMMUNITY_PRIOR_MAX_SAMPLES / total : 1;
445
+ const entry = { successes: Math.round(v.successes * scale), failures: Math.round(v.failures * scale) };
446
+ if (entry.successes + entry.failures > 0)
447
+ clean[key] = entry;
448
+ }
449
+ }
450
+ return clean;
451
+ }
452
+ // Parsed-prior cache, same 60s shape as the stats cache: routing reads the map
453
+ // once per chain entry (and once per key in orderKeysByScore), so hitting
454
+ // sqlite + JSON.parse on every lookup is pure waste. Invalidated by the two
455
+ // setters and by refreshStatsCache, so tests and future ingestion see writes
456
+ // immediately.
457
+ let communityPriorCache = null;
458
+ let communityPriorCacheTime = 0;
459
+ function communityPriorState() {
460
+ const now = Date.now();
461
+ if (communityPriorCache && now - communityPriorCacheTime < CACHE_TTL_MS)
462
+ return communityPriorCache;
463
+ let map = {};
464
+ const raw = getSetting(COMMUNITY_PRIOR_KEY);
465
+ if (raw) {
466
+ try {
467
+ map = sanitizeCommunityPriors(JSON.parse(raw));
468
+ }
469
+ catch { /* corrupt setting → no priors */ }
470
+ }
471
+ communityPriorCache = { map, enabled: getSetting(COMMUNITY_PRIOR_ENABLED_KEY) === '1' };
472
+ communityPriorCacheTime = now;
473
+ return communityPriorCache;
474
+ }
475
+ function invalidateCommunityPriorCache() {
476
+ communityPriorCache = null;
477
+ }
478
+ /** Whether stored community priors are folded into the posterior. Default off. */
479
+ export function getCommunityPriorEnabled() {
480
+ return communityPriorState().enabled;
481
+ }
482
+ export function setCommunityPriorEnabled(enabled) {
483
+ setSetting(COMMUNITY_PRIOR_ENABLED_KEY, enabled ? '1' : '0');
484
+ invalidateCommunityPriorCache();
485
+ }
486
+ /** Community prior for one model, or undefined when none is stored.
487
+ * Raw read — ignores the enabled flag; routing goes through
488
+ * activeCommunityPrior, which honors it. */
489
+ export function getCommunityPrior(platform, modelId, endpointScope) {
490
+ return communityPriorState().map[modelStatsKey(platform, modelId, endpointScope)];
491
+ }
492
+ /** Gated read for routing: undefined unless the opt-in flag is on. */
493
+ function activeCommunityPrior(platform, modelId, endpointScope) {
494
+ const state = communityPriorState();
495
+ return state.enabled ? state.map[modelStatsKey(platform, modelId, endpointScope)] : undefined;
496
+ }
497
+ /** Replace the whole community-prior map (e.g. after an aggregation fetch).
498
+ * Invalid entries are dropped and oversized ones capped, never stored raw. */
499
+ export function setCommunityPriors(priors) {
500
+ const clean = sanitizeCommunityPriors(priors);
501
+ setSetting(COMMUNITY_PRIOR_KEY, JSON.stringify(clean));
502
+ invalidateCommunityPriorCache();
503
+ return Object.keys(clean).length;
504
+ }
505
+ /** Active weights plus whether the peak-hours adjustment (#760) changed them.
506
+ * With the setting off (the default) `weights` is the preset itself and
507
+ * `adjusted` is false, so nothing about routing moves with the clock.
508
+ * priority/custom are the operator's explicit choice and are never rewritten. */
509
+ function weightsWithPeak(strategy) {
510
+ if (strategy === 'priority')
511
+ return { weights: null, adjusted: false };
512
+ if (strategy === 'custom')
513
+ return { weights: getCustomWeights(), adjusted: false };
514
+ return peakAdjustedWeights(BANDIT_PRESETS[strategy], strategy, getPeakHoursConfig());
515
+ }
516
+ function weightsFor(strategy) {
517
+ return weightsWithPeak(strategy).weights;
518
+ }
519
+ /** The weight vector routing will use right now for the active strategy, and
520
+ * whether the peak-hours adjustment moved it. Cheap (settings reads only) —
521
+ * for the PUT /routing echo, which must not pay for a full score sweep. */
522
+ export function getActiveRoutingWeights() {
523
+ return weightsWithPeak(getRoutingStrategy());
524
+ }
525
+ // ── Analytics stats cache (decay-weighted) ──────────────────────────────────
526
+ // Instead of the fork's flat 7-day window (where a model that degrades today
527
+ // keeps a stale week-long average), each request is weighted by an exponential
528
+ // decay so recent behavior dominates while older data still stabilizes the
529
+ // estimate. We aggregate by (model, integer day age) in SQL — at most ~7 rows
530
+ // per model — then apply the per-bucket decay weight in JS.
531
+ const WINDOW_MS = 7 * 24 * 60 * 60 * 1000;
532
+ const HALF_LIFE_DAYS = 2; // a 2-day-old request counts half as much as a fresh one
533
+ const CACHE_TTL_MS = 60 * 1000;
534
+ // Keyed by modelStatsKey(): "platform:model_id" for catalog models, and
535
+ // "custom:model_id@base_url" for a relay model that carries an endpoint scope
536
+ // (#651). A single-endpoint install produces the same keys it always did.
537
+ let statsCache = null;
538
+ let keyStatsCache = null; // "platform:model_id:key_id"
539
+ let statsCacheTime = 0;
540
+ function decayWeight(ageDays) {
541
+ return Math.pow(0.5, Math.max(0, ageDays) / HALF_LIFE_DAYS);
542
+ }
543
+ // SQL predicate for "this row is a timed-out request" (#619). A timeout is an
544
+ // error row whose text carries one of the shared timeout markers
545
+ // (lib/error-classify.ts), which is also what the failover attempt trail
546
+ // classifies on. 'canceled' rows (#752 — client hung up) never reach this
547
+ // predicate: the stats query below filters them out entirely, because a
548
+ // vanished client says nothing about the model's reliability or speed. The
549
+ // markers are hard-coded lowercase identifiers from our own source, never
550
+ // user input, so interpolating them into the LIKE list is safe.
551
+ const IS_TIMEOUT_SQL = `(status != 'success' AND (${TIMEOUT_ERROR_MARKERS.map(m => `LOWER(COALESCE(error, '')) LIKE '%${m}%'`).join(' OR ')}))`;
552
+ /** api_keys.id → endpoint scope, for every custom credential on record (#651). */
553
+ function customEndpointScopes(db) {
554
+ const rows = db.prepare("SELECT id, base_url FROM api_keys WHERE platform = 'custom'")
555
+ .all();
556
+ return new Map(rows.map(r => [r.id, endpointScopeForBaseUrl(r.base_url)]));
557
+ }
558
+ export function refreshStatsCache(db, force = false) {
559
+ if (!force && statsCache && Date.now() - statsCacheTime < CACHE_TTL_MS)
560
+ return;
561
+ // Re-read the community priors alongside the stats they season, so a forced
562
+ // refresh (tests, admin actions) never routes on a stale prior snapshot.
563
+ invalidateCommunityPriorCache();
564
+ const since = new Date(Date.now() - WINDOW_MS).toISOString();
565
+ // Grouped by (model, key, day age): still a handful of rows per model — key
566
+ // count × ≤7 day buckets — so the finer grain keeps the same one-query,
567
+ // 60s-cached shape. Aggregated two ways below: rolled up per model (ordering)
568
+ // and per key (in-model key selection, #580).
569
+ const buckets = db.prepare(`
570
+ SELECT platform, model_id, key_id,
571
+ CAST((julianday('now') - julianday(created_at)) AS INTEGER) AS age_days,
572
+ COUNT(*) AS total,
573
+ SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) AS successes,
574
+ SUM(CASE WHEN status = 'success' THEN output_tokens ELSE 0 END) AS succ_out,
575
+ SUM(CASE WHEN status = 'success' THEN latency_ms ELSE 0 END) AS succ_lat,
576
+ SUM(CASE WHEN status = 'success' AND ttfb_ms IS NOT NULL THEN ttfb_ms ELSE 0 END) AS succ_ttfb_sum,
577
+ SUM(CASE WHEN status = 'success' AND ttfb_ms IS NOT NULL THEN 1 ELSE 0 END) AS succ_ttfb_cnt,
578
+ SUM(CASE WHEN ${IS_TIMEOUT_SQL} THEN 1 ELSE 0 END) AS timeouts,
579
+ SUM(CASE WHEN ${IS_TIMEOUT_SQL} THEN MIN(MAX(latency_ms, 0), ${TIMEOUT_LATENCY_CAP_MS}) ELSE 0 END) AS timeout_lat
580
+ FROM requests
581
+ WHERE created_at >= ? AND status <> 'canceled'
582
+ GROUP BY platform, model_id, key_id, age_days
583
+ `).all(since);
584
+ const emptyAcc = () => ({ wSucc: 0, wFail: 0, wOut: 0, wLat: 0, wTtfbSum: 0, wTtfbCnt: 0, wTimeouts: 0 });
585
+ const addBucket = (a, w, b) => {
586
+ a.wSucc += w * b.successes;
587
+ a.wFail += w * (b.total - b.successes);
588
+ a.wOut += w * b.succ_out;
589
+ a.wLat += w * (b.succ_lat + b.timeout_lat);
590
+ a.wTtfbSum += w * (b.succ_ttfb_sum + b.timeout_lat);
591
+ a.wTtfbCnt += w * (b.succ_ttfb_cnt + b.timeouts);
592
+ a.wTimeouts += w * b.timeouts;
593
+ };
594
+ // Which endpoint each custom credential belongs to, so a request logged
595
+ // against relay A's key lands in relay A's bucket and nowhere else (#651).
596
+ // `requests` has always recorded key_id, so pre-migration history splits
597
+ // correctly too; rows whose key is gone (or that never had one) fall into the
598
+ // un-scoped bucket, which only un-scoped rows read.
599
+ const scopeByKeyId = customEndpointScopes(db);
600
+ const scopeOf = (platform, keyId) => platform === 'custom' && keyId != null ? (scopeByKeyId.get(keyId) ?? '') : '';
601
+ const acc = new Map();
602
+ const keyAcc = new Map();
603
+ for (const b of buckets) {
604
+ const key = modelStatsKey(b.platform, b.model_id, scopeOf(b.platform, b.key_id));
605
+ const w = decayWeight(b.age_days);
606
+ let a = acc.get(key);
607
+ if (!a)
608
+ acc.set(key, a = emptyAcc());
609
+ addBucket(a, w, b);
610
+ if (b.key_id != null) {
611
+ const kk = `${key}:${b.key_id}`;
612
+ let ka = keyAcc.get(kk);
613
+ if (!ka)
614
+ keyAcc.set(kk, ka = emptyAcc());
615
+ addBucket(ka, w, b);
616
+ }
617
+ }
618
+ // Calendar-month token usage per model, for the headroom guardrail.
619
+ const usageRows = db.prepare(`
620
+ SELECT platform, model_id, key_id, COALESCE(SUM(input_tokens + output_tokens), 0) AS used
621
+ FROM requests
622
+ WHERE created_at >= datetime('now', 'start of month')
623
+ AND request_type = 'chat'
624
+ GROUP BY platform, model_id, key_id
625
+ `).all();
626
+ const usageMap = new Map();
627
+ for (const r of usageRows) {
628
+ const key = modelStatsKey(r.platform, r.model_id, scopeOf(r.platform, r.key_id));
629
+ usageMap.set(key, (usageMap.get(key) ?? 0) + r.used);
630
+ }
631
+ const next = new Map();
632
+ for (const [key, a] of acc) {
633
+ next.set(key, {
634
+ successes: a.wSucc,
635
+ failures: a.wFail,
636
+ tokPerSec: a.wLat > 0 ? (a.wOut * 1000) / a.wLat : 0,
637
+ avgTtfbMs: a.wTtfbCnt > 0 ? a.wTtfbSum / a.wTtfbCnt : null,
638
+ monthlyUsedTokens: usageMap.get(key) ?? 0,
639
+ speedSamples: a.wSucc + a.wTimeouts,
640
+ });
641
+ }
642
+ // Models with month usage but no recent window data still need a headroom number.
643
+ for (const [key, used] of usageMap) {
644
+ if (!next.has(key)) {
645
+ next.set(key, { successes: 0, failures: 0, tokPerSec: 0, avgTtfbMs: null, monthlyUsedTokens: used, speedSamples: 0 });
646
+ }
647
+ }
648
+ const nextKeys = new Map();
649
+ for (const [kk, a] of keyAcc) {
650
+ nextKeys.set(kk, {
651
+ successes: a.wSucc,
652
+ failures: a.wFail,
653
+ tokPerSec: a.wLat > 0 ? (a.wOut * 1000) / a.wLat : 0,
654
+ avgTtfbMs: a.wTtfbCnt > 0 ? a.wTtfbSum / a.wTtfbCnt : null,
655
+ });
656
+ }
657
+ statsCache = next;
658
+ keyStatsCache = nextKeys;
659
+ statsCacheTime = Date.now();
660
+ // Natural tail of a recompute: fold what we just measured back into
661
+ // models.speed_rank (#619). Never allowed to break routing — the caches above
662
+ // are already published, and a failed write just means the column keeps its
663
+ // previous value until the next pass.
664
+ if (Date.now() - speedRankWriteTime >= SPEED_RANK_WRITE_INTERVAL_MS) {
665
+ speedRankWriteTime = Date.now();
666
+ try {
667
+ writeObservedSpeedRanks(db);
668
+ }
669
+ catch (e) {
670
+ console.error('Failed to write observed speed ranks:', e);
671
+ }
672
+ }
673
+ }
674
+ // ── Observed speed_rank writeback (#619) ────────────────────────────────────
675
+ // models.speed_rank is the catalog's hand-assigned speed ordering and drives
676
+ // the dashboard's sort-by-speed preset. It was only ever WRITTEN by the seed
677
+ // migrations, catalog sync, and an explicit user override — never by anything
678
+ // that had actually watched the model run, so a relay model that hangs on half
679
+ // its calls kept whatever rank the catalog guessed for it forever.
680
+ //
681
+ // This folds the live speed axis back into the column, under three rules:
682
+ // - a model needs SPEED_RANK_MIN_SAMPLES decay-weighted speed-bearing
683
+ // requests (successes + timeouts) before we claim to know anything; below
684
+ // that it keeps its catalog value;
685
+ // - a user-set speed_rank override always wins — we skip those models
686
+ // entirely rather than fight applyModelOverrides for the column;
687
+ // - the UPDATE is guarded on the value actually changing, so a steady system
688
+ // writes nothing at all.
689
+ //
690
+ // A catalog sync re-stamps speed_rank from the catalog; the next pass simply
691
+ // re-derives the observed value, which is why this is a periodic write rather
692
+ // than a one-shot migration.
693
+ export const SPEED_RANK_MIN_SAMPLES = 20;
694
+ const SPEED_RANK_WRITE_INTERVAL_MS = 10 * 60 * 1000;
695
+ let speedRankWriteTime = 0;
696
+ /** Test hook: forget when the last writeback ran so the next refresh does one. */
697
+ export function resetSpeedRankWriteback() {
698
+ speedRankWriteTime = 0;
699
+ }
700
+ /**
701
+ * Write an observed speed rank for every model with enough recent samples and
702
+ * no user-set speed_rank override. Returns how many rows actually changed.
703
+ * Reads the stats cache as-is — callers refresh it first (refreshStatsCache
704
+ * calls this from its own tail).
705
+ */
706
+ export function writeObservedSpeedRanks(db) {
707
+ if (!statsCache || statsCache.size === 0)
708
+ return 0;
709
+ const pinned = modelsWithOverriddenField(db, 'speedRank');
710
+ const rows = db.prepare('SELECT id, platform, model_id, speed_rank, endpoint_scope FROM models')
711
+ .all();
712
+ const update = db.prepare('UPDATE models SET speed_rank = ? WHERE id = ?');
713
+ let written = 0;
714
+ const tx = db.transaction(() => {
715
+ for (const row of rows) {
716
+ // Overrides are keyed (platform, model_id) — they only exist for
717
+ // catalog-managed rows, which are never endpoint-scoped — while the
718
+ // measured stats are per endpoint, so each relay's copy gets its own
719
+ // observed rank instead of one shared number (#651).
720
+ if (pinned.has(`${row.platform}:${row.model_id}`))
721
+ continue;
722
+ const stats = statsCache.get(modelStatsKey(row.platform, row.model_id, row.endpoint_scope));
723
+ if (!stats || stats.speedSamples < SPEED_RANK_MIN_SAMPLES)
724
+ continue;
725
+ const rank = observedSpeedRank(speedScore(stats.tokPerSec, stats.avgTtfbMs));
726
+ if (rank === row.speed_rank)
727
+ continue;
728
+ update.run(rank, row.id);
729
+ written++;
730
+ }
731
+ });
732
+ tx();
733
+ return written;
734
+ }
735
+ // Enabled + healthy/unknown key count per platform, for pooled-budget scaling.
736
+ // This is the SAME filter both /api/fallback endpoints use (issue #456): the
737
+ // monthly budget is a PER-KEY free-tier allowance, so N usable keys pool N× the
738
+ // capacity. `monthlyUsedTokens` is already summed across all keys, so budget
739
+ // must scale to match or the headroom guardrail damps a multi-key model to the
740
+ // floor after just one account's worth of tokens.
741
+ function usableKeyCountsByPlatform(db) {
742
+ const rows = db.prepare("SELECT platform, COUNT(*) AS count FROM api_keys WHERE enabled = 1 AND status IN ('healthy', 'unknown') GROUP BY platform").all();
743
+ return new Map(rows.map(r => [r.platform, r.count]));
744
+ }
745
+ function scoreChainEntry(entry, weights, intelMin, intelMax, sampled, keyCounts, headroomCfg) {
746
+ const stats = statsCache?.get(modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope));
747
+ const successes = stats?.successes ?? 0;
748
+ const failures = stats?.failures ?? 0;
749
+ const community = activeCommunityPrior(entry.platform, entry.model_id, entry.endpoint_scope);
750
+ let reliability;
751
+ if (sampled) {
752
+ const { alpha, beta } = reliabilityPosterior(successes, failures, community);
753
+ reliability = sampleBeta(alpha, beta);
754
+ }
755
+ else {
756
+ reliability = expectedReliability(successes, failures, community);
757
+ }
758
+ const speed = speedScore(stats?.tokPerSec ?? 0, stats?.avgTtfbMs ?? null);
759
+ const intelligence = intelligenceScore(intelligenceComposite(entry.size_label, entry.intelligence_rank), intelMin, intelMax);
760
+ // Scale the per-key monthly budget by the usable key count for this platform,
761
+ // matching the pooled `monthlyUsedTokens` aggregate (#456). Math.max(1, …) so a
762
+ // model whose platform currently has no usable key isn't handed a 0 budget.
763
+ const budget = parseBudget(entry.monthly_token_budget) * Math.max(1, keyCounts.get(entry.platform) ?? 1);
764
+ // Tunable headroom thresholds (#899): persisted overrides for when demotion
765
+ // starts and its floor; absent settings keep the scoring.ts defaults. Read
766
+ // ONCE per chain by the caller, not per entry — getSetting is an uncached
767
+ // SELECT, so reading it here cost two extra SQLite round-trips per model per
768
+ // request, the same reason `weights` and `keyCounts` are hoisted.
769
+ const monthlyHeadroom = headroomFactor(stats?.monthlyUsedTokens ?? 0, budget, headroomCfg);
770
+ // The same guardrail, driven by live rpm/rpd/tpm/tpd utilization instead of
771
+ // the monthly budget (#899). Most free tiers publish a daily request or token
772
+ // cap and no monthly figure at all, so without this a model sits at score #1
773
+ // until the request that finally 429s it. Reads a snapshot memoised inside
774
+ // ratelimit.ts, so this costs no query per model per request.
775
+ const windowHeadroom = rateWindowHeadroomFactor(modelWindowUsedFraction({ platform: entry.platform, modelId: entry.model_id, keyId: entry.key_id }, { rpm: entry.rpm_limit, rpd: entry.rpd_limit, tpm: entry.tpm_limit, tpd: entry.tpd_limit }), headroomCfg);
776
+ // The WORSE of the two, not their product: both express the same "this model
777
+ // is close to burning out" opinion on different meters, and multiplying them
778
+ // would push a model that is low on both to floor², below the floor the
779
+ // operator configured. Taking the binding constraint keeps the floor meaning
780
+ // what it says — the same rule getKeyQuotaHeadroom applies across metrics.
781
+ const headroom = Math.min(monthlyHeadroom, windowHeadroom);
782
+ const rl = rateLimitFactor(getPenalty(entry.model_db_id));
783
+ // Per-model env overrides (#738) scale the final score so a slow or
784
+ // poor-quality model is demoted without being disabled outright — a manual
785
+ // 'priority' chain can still select it.
786
+ const score = applyModelWeightOverride(combineScore({ reliability, speed, intelligence, headroom, rateLimit: rl }, weights), entry.model_id);
787
+ return { axes: { reliability, speed, intelligence }, headroom, rateLimit: rl, score };
788
+ }
789
+ /**
790
+ * Order the enabled fallback chain for routing.
791
+ * - 'priority' strategy → legacy manual order + 429 penalty (unchanged).
792
+ * - bandit strategy → convex score, manual priority as the deterministic
793
+ * tiebreaker for (near-)equal scores.
794
+ *
795
+ * `sampled` controls the bandit branch: Thompson sampling (the default) for
796
+ * live routing, where per-call randomness is the exploration the bandit needs;
797
+ * the deterministic expected score (`sampled = false`) for callers that want a
798
+ * STABLE ranking under the chosen strategy — the fusion panel, which should be a
799
+ * faithful reflection of the user's picked strategy, not a re-sampled draw each
800
+ * request. Priority mode is deterministic either way.
801
+ */
802
+ function orderChain(chain, strategy, sampled = true, task) {
803
+ // Tier first, always: it is the one ordering input that score must not be able
804
+ // to override (see ChainRow.match_tier). Zero for every chain built anywhere
805
+ // else, so this is a no-op outside slug-fallback resolution.
806
+ const tier = (e) => e.match_tier ?? 0;
807
+ let weights = weightsFor(strategy);
808
+ if (!weights) {
809
+ // Legacy priority mode: manual chain order + the 429/failure penalty,
810
+ // ascending.
811
+ //
812
+ // The penalty is denominated in PRIORITY POSITIONS — PENALTY_PER_429 = 3
813
+ // positions per rate limit, PENALTY_PER_FAIL = 1 per upstream failure,
814
+ // capped at MAX_PENALTY = 10 — so adding it to the RAW priority only ever
815
+ // reorders a chain whose neighbours sit within 10 of each other. Nothing
816
+ // guarantees that, and several ordinary paths guarantee the opposite:
817
+ // - PUT /api/fallback validates `priority` as a bare z.number(), so any
818
+ // spacing the caller likes (10 / 20 / 30) is persisted verbatim;
819
+ // - the seed and sort-preset paths number the WHOLE catalog 1..N, while
820
+ // this chain is only the ENABLED subset (`JOIN models m ON
821
+ // m.enabled = 1`) — switching models off punches arbitrarily large
822
+ // holes in the surviving sequence;
823
+ // - resolveModelGroupCandidates hydrates scattered group members with
824
+ // whatever COALESCE(fc.priority, 0) they happen to carry.
825
+ // On any of those the penalty was silently INERT: a model 429-ing every
826
+ // single request stayed pinned at the head of the chain forever, and the
827
+ // routing panel's "effective priority" claimed a demotion that never
828
+ // happened.
829
+ //
830
+ // So rank first, then penalize. Sort by the manual priority, re-number the
831
+ // survivors densely 1..N, and add the penalty to THAT rank. Dense ranking
832
+ // is monotonic in priority, so an unpenalized chain comes out in exactly
833
+ // the order the user arranged (tier still dominates as the outer sort key,
834
+ // and the raw priority remains the tiebreaker); the difference is that one
835
+ // penalty position now means what it says — one position.
836
+ return chain
837
+ .map((e, i) => ({ e, i }))
838
+ .sort((a, b) => a.e.priority - b.e.priority || a.i - b.i)
839
+ .map(({ e, i }, rank) => ({ e, i, eff: rank + 1 + getPenalty(e.model_db_id) }))
840
+ .sort((a, b) => tier(a.e) - tier(b.e) || a.eff - b.eff || a.e.priority - b.e.priority || a.i - b.i)
841
+ .map(x => x.e);
842
+ }
843
+ // Task-type bias (#1127): a client-declared/derived task type moves part of
844
+ // one axis onto the other (code: speed → intelligence; chat: the reverse).
845
+ // Applied AFTER the peak-hours adjustment, on the same weights the rest of
846
+ // the chain scores with; opt-in, so absent a signal the preset stands.
847
+ // `fastest`, `reliable` and `custom` are exempt (see TASK_EXEMPT_STRATEGIES),
848
+ // and the share is operator-tunable via settings (0 disables the bias).
849
+ if (task) {
850
+ const adjusted = taskAdjustedWeights(weights, task, strategy, getTaskWeightShare());
851
+ weights = adjusted.adjusted ? adjusted.weights : weights;
852
+ }
853
+ const composites = chain.map(e => intelligenceComposite(e.size_label, e.intelligence_rank));
854
+ const intelMin = composites.length ? Math.min(...composites) : 0;
855
+ const intelMax = composites.length ? Math.max(...composites) : 0;
856
+ const keyCounts = usableKeyCountsByPlatform(getDb());
857
+ const headroomCfg = getHeadroomThresholds();
858
+ return chain
859
+ .map(e => ({ e, s: scoreChainEntry(e, weights, intelMin, intelMax, sampled, keyCounts, headroomCfg).score }))
860
+ // Higher score first WITHIN a tier; manual priority breaks ties so the chain
861
+ // still matters.
862
+ .sort((a, b) => tier(a.e) - tier(b.e) || b.s - a.s || a.e.priority - b.e.priority)
863
+ .map(x => x.e);
864
+ }
865
+ const GLOBAL_SORT_ALIASES = {
866
+ smart: 'smart', smartest: 'smart', intelligence: 'smart',
867
+ fast: 'fast', fastest: 'fast', speed: 'fast',
868
+ cheap: 'cheap', cheapest: 'cheap', price: 'cheap', budget: 'cheap',
869
+ reliable: 'reliable', reliability: 'reliable',
870
+ balanced: 'balanced',
871
+ };
872
+ /**
873
+ * The chain auto-routing walks.
874
+ *
875
+ * When a profile is active it IS the chain, empty or not (#1021). Falling
876
+ * through to `fallback_config` on an empty one meant a chain the operator had
877
+ * deliberately built by hand — or had not filled in yet — silently routed over
878
+ * the entire catalog instead, while the same chain addressed by name
879
+ * (`auto:<name>`) correctly refused. `fallback_config` is the chain only for an
880
+ * install with no profile at all.
881
+ */
882
+ function getActiveChain(db) {
883
+ const profileId = getActiveProfileId(db);
884
+ if (profileId != null) {
885
+ return db.prepare(`
886
+ SELECT pm.model_db_id, pm.priority, pm.enabled,
887
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
888
+ m.size_label, m.monthly_token_budget,
889
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
890
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
891
+ FROM profile_models pm
892
+ JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
893
+ WHERE pm.profile_id = ?
894
+ ORDER BY pm.priority ASC
895
+ `).all(profileId);
896
+ }
897
+ return db.prepare(`
898
+ SELECT fc.model_db_id, fc.priority, fc.enabled,
899
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
900
+ m.size_label, m.monthly_token_budget,
901
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
902
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
903
+ FROM fallback_config fc
904
+ JOIN models m ON m.id = fc.model_db_id AND m.enabled = 1
905
+ ORDER BY fc.priority ASC
906
+ `).all();
907
+ }
908
+ function getChainByProfileName(db, name) {
909
+ const profile = db.prepare("SELECT id FROM profiles WHERE LOWER(name) = ?").get(name.toLowerCase());
910
+ if (!profile)
911
+ return null;
912
+ return db.prepare(`
913
+ SELECT pm.model_db_id, pm.priority, pm.enabled,
914
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
915
+ m.size_label, m.monthly_token_budget,
916
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
917
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
918
+ FROM profile_models pm
919
+ JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
920
+ WHERE pm.profile_id = ?
921
+ ORDER BY pm.priority ASC
922
+ `).all(profile.id);
923
+ }
924
+ function getChainByGlobalSort(db, globalAxis) {
925
+ // A global sort ignores the chain's ORDER, not its enable flags: a model the
926
+ // operator switched off — in the catalog or just for auto routing — stays off
927
+ // here too (#634). Models with no chain row yet (fresh catalog rows) default
928
+ // to in, so the sort still spans the whole catalog.
929
+ const profileId = getActiveProfileId(db);
930
+ const chainEnabled = profileId != null
931
+ ? 'COALESCE(pm.enabled, fc.enabled, 1) = 1'
932
+ : 'COALESCE(fc.enabled, 1) = 1';
933
+ const allEnabled = db.prepare(`
934
+ SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
935
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
936
+ m.size_label, m.monthly_token_budget,
937
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
938
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
939
+ FROM models m
940
+ LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
941
+ ${profileId != null ? 'LEFT JOIN profile_models pm ON pm.profile_id = ? AND pm.model_db_id = m.id' : ''}
942
+ WHERE m.enabled = 1 AND ${chainEnabled}
943
+ `).all(...(profileId != null ? [profileId] : []));
944
+ const strategyMap = {
945
+ 'smart': 'smartest',
946
+ 'fast': 'fastest',
947
+ 'cheap': 'balanced',
948
+ 'reliable': 'reliable',
949
+ 'balanced': 'balanced'
950
+ };
951
+ const strat = strategyMap[globalAxis] || 'balanced';
952
+ return orderChain(allEnabled, strat);
953
+ }
954
+ /**
955
+ * The active chain, or a client-facing refusal when it has nothing enabled.
956
+ *
957
+ * Mirrors what `auto:<name>` already does for a named chain: say the chain is
958
+ * empty rather than routing the request over models the operator never put in
959
+ * it. Only when a profile is active — a legacy install with none keeps the
960
+ * ordinary "all models exhausted" exhaustion path.
961
+ */
962
+ function activeChainOrThrow(db) {
963
+ const chain = getActiveChain(db);
964
+ if (chain.some(entry => entry.enabled))
965
+ return chain;
966
+ const profileId = getActiveProfileId(db);
967
+ if (profileId == null)
968
+ return chain;
969
+ const profile = db.prepare('SELECT name FROM profiles WHERE id = ?').get(profileId);
970
+ const err = new Error(`The active fallback chain${profile ? ` '${profile.name}'` : ''} has no enabled models. `
971
+ + 'Enable models for it on the Models page, switch the active chain, or name another one with "auto:<chain>".');
972
+ err.status = 400;
973
+ throw err;
974
+ }
975
+ export function resolveRoutingChain(modelString) {
976
+ const db = getDb();
977
+ if (!modelString || modelString.toLowerCase() === 'auto') {
978
+ return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
979
+ }
980
+ const lower = modelString.toLowerCase();
981
+ if (!lower.startsWith('auto:')) {
982
+ return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
983
+ }
984
+ const suffix = lower.slice('auto:'.length).trim();
985
+ if (!suffix) {
986
+ return { chain: activeChainOrThrow(db), strategyKey: 'auto' };
987
+ }
988
+ const globalAxis = GLOBAL_SORT_ALIASES[suffix];
989
+ if (globalAxis) {
990
+ const chain = getChainByGlobalSort(db, globalAxis);
991
+ if (chain.length === 0) {
992
+ const err = new Error(`No enabled models available for global sort '${suffix}'`);
993
+ err.status = 400;
994
+ throw err;
995
+ }
996
+ return { chain, strategyKey: `auto:${globalAxis}` };
997
+ }
998
+ const chain = getChainByProfileName(db, suffix);
999
+ if (!chain) {
1000
+ const err = new Error(`Profile '${suffix}' not found. Use 'auto' for the default profile, or call /v1/models for available options.`);
1001
+ err.status = 400;
1002
+ throw err;
1003
+ }
1004
+ const enabledModels = chain.filter(e => e.enabled);
1005
+ if (enabledModels.length === 0) {
1006
+ const err = new Error(`Profile '${suffix}' has no enabled models. Add models to this profile in the dashboard.`);
1007
+ err.status = 400;
1008
+ throw err;
1009
+ }
1010
+ return { chain, strategyKey: `auto:${suffix}` };
1011
+ }
1012
+ // Weights for the in-model key score (#580). Reliability dominates: the point
1013
+ // of per-key stats is catching a credential that FAILS (expired, drained,
1014
+ // region-blocked); speed differences between keys of the same platform+model
1015
+ // are second-order (account-tier throttling) but still worth a nudge.
1016
+ const KEY_SCORE_WEIGHTS = { reliability: 0.75, speed: 0.25 };
1017
+ /**
1018
+ * Order a model's candidate keys by a Thompson-sampled per-key score, mirroring
1019
+ * orderChain's bandit: reliability is a fresh draw from each key's Beta
1020
+ * posterior (so exploration is automatic and proportional to uncertainty — a
1021
+ * key with no data samples from the uniform prior and still gets traffic),
1022
+ * speed is deterministic. Returns null when NO key has recorded data, telling
1023
+ * the caller to keep the legacy round-robin rotation (no signal → no ranking).
1024
+ */
1025
+ function orderKeysByScore(entry, keys) {
1026
+ if (keys.length < 2 || !keyStatsCache)
1027
+ return null;
1028
+ const prefix = `${modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope)}:`;
1029
+ if (!keys.some(k => keyStatsCache.has(prefix + k.id)))
1030
+ return null;
1031
+ // The prior is per-model, not per-key: look it up once outside the loop.
1032
+ const community = activeCommunityPrior(entry.platform, entry.model_id, entry.endpoint_scope);
1033
+ return keys
1034
+ .map(k => {
1035
+ const stats = keyStatsCache.get(prefix + k.id);
1036
+ const { alpha, beta } = reliabilityPosterior(stats?.successes ?? 0, stats?.failures ?? 0, community);
1037
+ const rel = sampleBeta(alpha, beta);
1038
+ const spd = speedScore(stats?.tokPerSec ?? 0, stats?.avgTtfbMs ?? null);
1039
+ return { k, s: KEY_SCORE_WEIGHTS.reliability * rel + KEY_SCORE_WEIGHTS.speed * spd };
1040
+ })
1041
+ .sort((a, b) => b.s - a.s || a.k.id - b.k.id)
1042
+ .map(x => x.k);
1043
+ }
1044
+ /** Headroom assumed for a key the quota tracker has never seen. Neutral on
1045
+ * purpose: an unobserved budget is no reason to prefer a key (it could be
1046
+ * drained) and no reason to avoid one (it could be untouched), so it sorts
1047
+ * between an exhausted key and a fresh one and otherwise keeps its incoming
1048
+ * round-robin position. */
1049
+ const UNKNOWN_QUOTA_HEADROOM = 0.5;
1050
+ /**
1051
+ * Whether remaining-quota weighting is meaningful for this chain entry: the
1052
+ * operator asked for it AND the platform meters its keys separately.
1053
+ *
1054
+ * An account-scoped pool ('<platform>::account') is ONE budget every key of the
1055
+ * account draws down, so "which key has more left" has no answer — every key
1056
+ * reports the same number, and reordering on it would only churn the rotation
1057
+ * for nothing (#919).
1058
+ */
1059
+ function quotaWeightingApplies(entry) {
1060
+ if (getKeySelectionStrategy() !== 'least-remaining')
1061
+ return false;
1062
+ return !inferQuotaPoolKey(entry.platform, entry.model_id).endsWith('::account');
1063
+ }
1064
+ /**
1065
+ * Re-order an already-ordered candidate list by observed remaining quota,
1066
+ * roomiest first (#919 — the issue asks for higher-remaining-first, so the key
1067
+ * nearest its cap is tried last, not first).
1068
+ *
1069
+ * Deliberately a SORT over the caller's list rather than a second walk: the
1070
+ * incoming order is the round-robin rotation (or the per-key bandit ranking),
1071
+ * and Array#sort is stable, so keys with equal headroom — including the common
1072
+ * case of no observations at all — keep exactly the order they would have had.
1073
+ * Every gate, the custom-endpoint filter and the skip tally stay in the one
1074
+ * walk that follows.
1075
+ */
1076
+ function orderKeysByRemainingQuota(entry, ordered) {
1077
+ const headroom = getKeyQuotaHeadroom(entry.platform);
1078
+ if (headroom.size === 0)
1079
+ return ordered;
1080
+ // Hoisted out of the comparator: sort calls it O(n log n) times, and the
1081
+ // lookup below must not re-derive anything per comparison.
1082
+ const room = new Map(ordered.map(k => [k.id, headroom.get(k.id) ?? UNKNOWN_QUOTA_HEADROOM]));
1083
+ return [...ordered].sort((a, b) => room.get(b.id) - room.get(a.id));
1084
+ }
1085
+ /**
1086
+ * Pick a usable key for ONE model and build its RouteResult, or return null if
1087
+ * the model has no key that can serve the request right now (all cooled down,
1088
+ * over quota, undecryptable, or no provider). Factored out of routeRequest so
1089
+ * the fusion panel can HARD-PIN a model: walk that model's keys without ever
1090
+ * falling through to a different model (issue #326 — soft preference collapses
1091
+ * panel diversity under rate limits). Keys are tried in per-key bandit-score
1092
+ * order when any of them has recorded data (#580), else round-robin.
1093
+ * Request-level filters (vision/tools/context window) stay in the caller; this
1094
+ * only does key selection + accounting pre-checks.
1095
+ */
1096
+ function selectKeyForModel(entry, estimatedTokens, skipKeys, diag) {
1097
+ const db = getDb();
1098
+ const label = `${entry.platform}/${entry.model_id}`;
1099
+ if (!hasProvider(entry.platform)) {
1100
+ diag?.push(`${label}: no provider registered`);
1101
+ return null;
1102
+ }
1103
+ const provider = getProvider(entry.platform);
1104
+ const allKeys = db.prepare("SELECT * FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(entry.platform);
1105
+ if (allKeys.length === 0) {
1106
+ diag?.push(`${label}: no enabled+healthy key for platform`);
1107
+ return null;
1108
+ }
1109
+ // Scoped keys (#657) are dropped before the walk: a key whose model scope
1110
+ // excludes this model is not a candidate at all — it neither takes a
1111
+ // round-robin slot nor burns an attempt on a guaranteed 403. Parsed once per
1112
+ // key row.
1113
+ const keys = allKeys.filter(k => scopeAllows(parseModelScope(k.model_scope_json), entry.model_id));
1114
+ if (keys.length === 0) {
1115
+ diag?.push(`${label}: no usable key — ${allKeys.length} key(s) scoped to other models`);
1116
+ return null;
1117
+ }
1118
+ // Tally the gate that rejected each key, so the exhaustion diagnostic can say
1119
+ // *why* a model with keys still couldn't serve (all on cooldown vs over quota).
1120
+ const skipTally = {};
1121
+ const note = (reason) => { skipTally[reason] = (skipTally[reason] ?? 0) + 1; };
1122
+ const limits = {
1123
+ rpm: entry.rpm_limit,
1124
+ rpd: entry.rpd_limit,
1125
+ tpm: entry.tpm_limit,
1126
+ tpd: entry.tpd_limit,
1127
+ };
1128
+ // Score-ordered walk over this model's keys (#580): when any key has recorded
1129
+ // reliability/speed data, try them best-sampled-score first so a chronically
1130
+ // failing key stops soaking up every Nth request. The stats cache is the same
1131
+ // 60s-TTL aggregate the model-level bandit uses (refresh is a no-op when
1132
+ // fresh, and cheap when not). With no data at all, keep the legacy rotation.
1133
+ refreshStatsCache(db);
1134
+ // Scoped so two relays offering the same model id don't share one rotation
1135
+ // cursor over the platform's key list (#651).
1136
+ const rrKey = modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope);
1137
+ let idx = roundRobinIndex.get(rrKey) ?? 0;
1138
+ let ranked = orderKeysByScore(entry, keys);
1139
+ // Remaining-quota weighting (#919) layers on top: it re-sorts whatever order
1140
+ // we were going to walk anyway — the bandit ranking when there is per-key
1141
+ // data, otherwise the round-robin rotation starting at the live cursor — so
1142
+ // ties fall back to that order instead of to rowid.
1143
+ if (keys.length > 1 && quotaWeightingApplies(entry)) {
1144
+ const base = ranked ?? Array.from({ length: keys.length }, (_, i) => keys[(idx + i) % keys.length]);
1145
+ ranked = orderKeysByRemainingQuota(entry, base);
1146
+ }
1147
+ // A custom model belongs to exactly one endpoint (#212), but an endpoint can
1148
+ // hold several credentials — so the pool is every key on the same base_url,
1149
+ // rotated like any other platform's keys (#619). Legacy rows (key_id NULL)
1150
+ // keep the old any-key match.
1151
+ const endpointKeyIds = entry.platform === 'custom' && entry.key_id != null
1152
+ ? customEndpointKeyIds(db, entry.key_id)
1153
+ : null;
1154
+ for (let attempt = 0; attempt < keys.length; attempt++) {
1155
+ const key = ranked ? ranked[attempt] : keys[idx % keys.length];
1156
+ idx++;
1157
+ if (endpointKeyIds && !endpointKeyIds.has(key.id)) {
1158
+ note('custom-key-mismatch');
1159
+ continue;
1160
+ }
1161
+ const skipId = `${entry.platform}:${entry.model_id}:${key.id}`;
1162
+ if (skipKeys?.has(skipId)) {
1163
+ note('already-failed-this-request');
1164
+ continue;
1165
+ }
1166
+ if (isOnCooldown(entry.platform, entry.model_id, key.id)) {
1167
+ note('cooldown');
1168
+ continue;
1169
+ }
1170
+ if (!canUseProvider(entry.platform, key.id)) {
1171
+ note('provider-daily-cap');
1172
+ continue;
1173
+ }
1174
+ // Account-wide per-minute budget, checked before the per-model gates: a model
1175
+ // with a NULL rpm_limit would otherwise sail past them and spend a budget its
1176
+ // siblings share.
1177
+ if (!canUseProviderMinute(entry.platform, key.id)) {
1178
+ note('provider-minute-cap');
1179
+ continue;
1180
+ }
1181
+ // Skip a key that already has its allowed requests in the air. Without this,
1182
+ // parallel streams all pick the same key and 429 each other on providers that
1183
+ // meter concurrency per credential.
1184
+ if (!canUseKeyConcurrency(entry.platform, key.id)) {
1185
+ note('key-concurrency');
1186
+ continue;
1187
+ }
1188
+ if (!canMakeRequest(entry.platform, entry.model_id, key.id, limits)) {
1189
+ note('rpm/rpd-limit');
1190
+ continue;
1191
+ }
1192
+ if (!canUseTokens(entry.platform, entry.model_id, key.id, estimatedTokens, limits)) {
1193
+ note('tpm/tpd-limit');
1194
+ continue;
1195
+ }
1196
+ if (!canUseProviderTokens(entry.platform, key.id, entry.model_id, estimatedTokens)) {
1197
+ note('provider-daily-token-cap');
1198
+ continue;
1199
+ }
1200
+ // Monthly budget (#1158): a key whose request/token caps are spent for the
1201
+ // current UTC month is not a candidate — same skip semantics as the daily
1202
+ // gates above. The Retry-After (next-month boundary) surfaces through the
1203
+ // fallback exhaustion path rather than blocking here.
1204
+ if (!checkMonthlyBudget(key.id, estimatedTokens).allowed) {
1205
+ note('monthly-budget-cap');
1206
+ continue;
1207
+ }
1208
+ let decryptedKey;
1209
+ try {
1210
+ decryptedKey = decrypt(key.encrypted_key, key.iv, key.auth_tag);
1211
+ }
1212
+ catch {
1213
+ db.prepare("UPDATE api_keys SET status = 'error', last_checked_at = datetime('now') WHERE id = ?")
1214
+ .run(key.id);
1215
+ note('decrypt-error');
1216
+ continue;
1217
+ }
1218
+ const resolvedProvider = entry.platform === 'custom'
1219
+ ? resolveProvider('custom', key.base_url)
1220
+ : provider;
1221
+ if (!resolvedProvider) {
1222
+ note('no-resolved-provider');
1223
+ continue;
1224
+ }
1225
+ roundRobinIndex.set(rrKey, idx);
1226
+ // Taken only once the key has cleared every gate and is definitely being
1227
+ // returned, so a rejected candidate never consumes concurrency budget.
1228
+ const proxyUrl = decryptProxyUrl(key);
1229
+ const budget = reserveMonthlyBudget(key.id, estimatedTokens);
1230
+ if (!budget.allowed) {
1231
+ note('monthly-budget-cap');
1232
+ continue;
1233
+ }
1234
+ const leaseId = acquireLease(entry.platform, entry.model_id, key.id, estimatedTokens);
1235
+ return {
1236
+ provider: resolvedProvider,
1237
+ modelId: entry.model_id,
1238
+ modelDbId: entry.model_db_id,
1239
+ apiKey: decryptedKey,
1240
+ keyId: key.id,
1241
+ keyLabel: key.label || null,
1242
+ // Decrypted once here, at the point the row is already in hand (#590).
1243
+ proxyUrl,
1244
+ platform: entry.platform,
1245
+ displayName: entry.display_name,
1246
+ contextWindow: entry.context_window,
1247
+ endpointScope: entry.endpoint_scope ?? '',
1248
+ rpdLimit: limits.rpd,
1249
+ tpdLimit: limits.tpd,
1250
+ release: () => { releaseLease(leaseId); budget.release(); },
1251
+ };
1252
+ }
1253
+ // No usable key for this model. Advance the round-robin index anyway so we
1254
+ // don't get stuck re-trying the same exhausted key first next time.
1255
+ roundRobinIndex.set(rrKey, idx);
1256
+ const summary = Object.entries(skipTally).map(([r, n]) => `${r}:${n}`).join(', ') || 'no usable key';
1257
+ diag?.push(`${label}: ${keys.length} key(s) — ${summary}`);
1258
+ return null;
1259
+ }
1260
+ /**
1261
+ * Whether the model still has ANOTHER key that could serve it right now, given
1262
+ * the key that just failed (excludingKeyId) and any keys already ruled out this
1263
+ * request (skipKeys, in the "platform:modelId:keyId" form). Applies the same
1264
+ * gates selectKeyForModel uses — enabled + healthy status, not on cooldown,
1265
+ * under the provider daily cap, and under rpm/rpd/tpm/tpd — so the answer means
1266
+ * "a real, dispatchable alternative exists".
1267
+ *
1268
+ * Used by the retry loops to decide whether a single key's 429 should demote the
1269
+ * WHOLE model (the model-level 429 penalty). It should not: the per-key cooldown
1270
+ * already isolates the failing key, so demoting the model while a sibling key can
1271
+ * still serve it wrongly sinks a healthy model in the scorer (#454). We only
1272
+ * record the model-level hit when this returns false — i.e. the 429 exhausted the
1273
+ * model, not just one of its keys.
1274
+ */
1275
+ export function hasOtherUsableKey(modelDbId, excludingKeyId, skipKeys) {
1276
+ const db = getDb();
1277
+ const m = db.prepare(`
1278
+ SELECT platform, model_id, rpm_limit, rpd_limit, tpm_limit, tpd_limit, key_id
1279
+ FROM models WHERE id = ?
1280
+ `).get(modelDbId);
1281
+ if (!m)
1282
+ return false;
1283
+ const limits = { rpm: m.rpm_limit, rpd: m.rpd_limit, tpm: m.tpm_limit, tpd: m.tpd_limit };
1284
+ const keys = db.prepare("SELECT id, model_scope_json FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(m.platform);
1285
+ // Keys of the model's own custom endpoint (#212, #619); a key belonging to a
1286
+ // DIFFERENT endpoint cannot serve it, so it doesn't count as an alternative.
1287
+ const endpointKeyIds = m.platform === 'custom' && m.key_id != null
1288
+ ? customEndpointKeyIds(db, m.key_id)
1289
+ : null;
1290
+ for (const k of keys) {
1291
+ if (k.id === excludingKeyId)
1292
+ continue;
1293
+ if (endpointKeyIds && !endpointKeyIds.has(k.id))
1294
+ continue;
1295
+ // A sibling scoped away from this model can never serve it (#657) — counting
1296
+ // it would wrongly suppress the model-level penalty this gate exists for.
1297
+ if (!scopeAllows(parseModelScope(k.model_scope_json), m.model_id))
1298
+ continue;
1299
+ if (skipKeys?.has(`${m.platform}:${m.model_id}:${k.id}`))
1300
+ continue;
1301
+ if (isOnCooldown(m.platform, m.model_id, k.id))
1302
+ continue;
1303
+ if (!canUseProvider(m.platform, k.id))
1304
+ continue;
1305
+ if (!canUseProviderMinute(m.platform, k.id))
1306
+ continue;
1307
+ if (!canMakeRequest(m.platform, m.model_id, k.id, limits))
1308
+ continue;
1309
+ // A per-minute token spike on the failed key doesn't mean a fresh key lacks
1310
+ // headroom; a nominal 1-token probe only rules out a key already at its
1311
+ // TPM/TPD ceiling.
1312
+ if (!canUseTokens(m.platform, m.model_id, k.id, 1, limits))
1313
+ continue;
1314
+ if (!canUseProviderTokens(m.platform, k.id, m.model_id, 1))
1315
+ continue;
1316
+ return true;
1317
+ }
1318
+ return false;
1319
+ }
1320
+ /**
1321
+ * Can ANY key serve this model right now? The same gates hasOtherUsableKey
1322
+ * applies — scope (#657), per-key cooldown, and the provider/model rate and
1323
+ * token windows — with no key excluded. /v1/models uses it to tell a `ready`
1324
+ * model from an `exhausted` one (#1100), so the listing cannot claim a model
1325
+ * the router would immediately skip.
1326
+ */
1327
+ export function hasUsableKeyForModel(modelDbId) {
1328
+ // Key ids are AUTOINCREMENT and start at 1, so -1 excludes nothing.
1329
+ return hasOtherUsableKey(modelDbId, -1);
1330
+ }
1331
+ /**
1332
+ * Every key that can be ROUTED to this model: enabled + healthy/unknown, not
1333
+ * scoped away from the model (#657), and — for a custom model — belonging to
1334
+ * the model's own endpoint (#212, #619). Deliberately ignores the transient
1335
+ * gates hasOtherUsableKey applies (cooldown, quotas): the caller here is the
1336
+ * model-level bench, which needs the full key set to take a sick model out of
1337
+ * rotation, not "who could serve the next request".
1338
+ */
1339
+ export function routableKeyIdsForModel(modelDbId) {
1340
+ const db = getDb();
1341
+ const m = db.prepare('SELECT platform, model_id, key_id FROM models WHERE id = ?')
1342
+ .get(modelDbId);
1343
+ if (!m)
1344
+ return [];
1345
+ const keys = db.prepare("SELECT id, model_scope_json FROM api_keys WHERE platform = ? AND enabled = 1 AND status IN ('healthy', 'unknown')").all(m.platform);
1346
+ const endpointKeyIds = m.platform === 'custom' && m.key_id != null
1347
+ ? customEndpointKeyIds(db, m.key_id)
1348
+ : null;
1349
+ return keys
1350
+ .filter(k => !endpointKeyIds || endpointKeyIds.has(k.id))
1351
+ .filter(k => scopeAllows(parseModelScope(k.model_scope_json), m.model_id))
1352
+ .map(k => k.id);
1353
+ }
1354
+ /**
1355
+ * Fetch a single enabled model's chain row by its db id.
1356
+ */
1357
+ function getModelChainRow(db, modelDbId) {
1358
+ return db.prepare(`
1359
+ SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
1360
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
1361
+ m.size_label, m.monthly_token_budget,
1362
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
1363
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
1364
+ FROM models m
1365
+ WHERE m.id = ? AND m.enabled = 1
1366
+ `).get(modelDbId);
1367
+ }
1368
+ /**
1369
+ * Safety margin applied when ranking a model against an estimated request size.
1370
+ * The estimate is a chars/4 heuristic that under-counts dense payloads (JSON,
1371
+ * code, CJK) by up to ~2x; without any accounting for that gap such requests
1372
+ * were routed to models whose real tokenizer count exceeded the window and the
1373
+ * provider rejected them with a 400 mid-chain (kilo: "maximum context length is
1374
+ * 262144 tokens" on requests estimated <=256000).
1375
+ *
1376
+ * The margin is a SOFT preference, not a hard filter (#956 review): /v1/models
1377
+ * advertises the RAW window, so clients legitimately pack requests right up to
1378
+ * it. Excluding margin-violating models outright would turn an upstream 400
1379
+ * that the retry loop already classifies and handles (`context_too_large`)
1380
+ * into a regression: "all models exhausted" with zero attempts. Callers
1381
+ * therefore try margin-fitting candidates first and only fall back to raw
1382
+ * advertised-window fits (see fitsContextWindowStrict) when nothing else can
1383
+ * serve the request.
1384
+ */
1385
+ export const CONTEXT_WINDOW_SAFETY_FACTOR = 1.25;
1386
+ // Platforms whose pre-dispatch trim guard already caps the dispatched input
1387
+ // below the live context ceiling (lib/content.ts truncateMessagesForGithub):
1388
+ // the guard — not the routing estimate — is what guarantees the fit there, so
1389
+ // applying the factor too would only make that guard unreachable. The margin
1390
+ // checks below treat these platforms as strict comparisons.
1391
+ const TRIM_GUARDED_PLATFORMS = new Set(['github']);
1392
+ /** True when `estimatedTokens` fits the RAW advertised window (null window =
1393
+ * unknown, never filtered — same convention as the auto-router). This is the
1394
+ * comparison /v1/models publishes and the soft-preference fallback tier. */
1395
+ export function fitsContextWindowStrict(contextWindow, estimatedTokens) {
1396
+ return contextWindow == null || estimatedTokens <= contextWindow;
1397
+ }
1398
+ /** True when `estimatedTokens` plausibly fits `contextWindow` WITH the safety
1399
+ * margin. The chars/4 heuristic portion is scaled by the factor; an explicit
1400
+ * output reserve derived from the client's max_tokens (`routingReserveTokens`)
1401
+ * is already an exact count and is added UNSCALED (#956 review). Trim-guarded
1402
+ * platforms compare strictly — their guard guarantees the fit. */
1403
+ export function fitsContextWindow(platform, contextWindow, estimatedTokens, exactOutputReserve = 0) {
1404
+ if (contextWindow == null)
1405
+ return true;
1406
+ // Raw advertised comparison first — the margin can only shrink eligibility.
1407
+ if (estimatedTokens > contextWindow)
1408
+ return false;
1409
+ if (TRIM_GUARDED_PLATFORMS.has(platform))
1410
+ return true;
1411
+ const reserve = Math.max(0, exactOutputReserve);
1412
+ const heuristic = Math.max(0, estimatedTokens - reserve);
1413
+ return heuristic * CONTEXT_WINDOW_SAFETY_FACTOR + reserve <= contextWindow;
1414
+ }
1415
+ /**
1416
+ * Route to ONE specific model, hard-pinned. Rotates across that model's keys
1417
+ * (cooldowns, quotas, decryption all honored) but NEVER substitutes a different
1418
+ * model — returns null if the pinned model can't serve right now. This is what
1419
+ * makes a fusion panel genuinely diverse: a rate-limited slot is dropped, not
1420
+ * silently collapsed onto whatever else is available. `skipKeys` lets a slot
1421
+ * exclude keys it already failed on this request.
1422
+ */
1423
+ export function routePinnedModel(modelDbId, estimatedTokens = 1000, skipKeys) {
1424
+ const db = getDb();
1425
+ const entry = getModelChainRow(db, modelDbId);
1426
+ if (!entry)
1427
+ return null;
1428
+ // Strict comparison only (#956 review): a pinned slot has no substitute, so
1429
+ // refusing on a margin violation would drop the slot outright where the
1430
+ // pre-margin behavior was one dispatch attempt (a mid-chain context_too_large
1431
+ // 400 is classified and retried downstream). Nothing is multiplied here —
1432
+ // estimatedTokens already carries the exact capped output reserve (#470).
1433
+ if (!fitsContextWindowStrict(entry.context_window, estimatedTokens))
1434
+ return null;
1435
+ if (entry.tpm_limit != null && estimatedTokens > entry.tpm_limit)
1436
+ return null;
1437
+ return selectKeyForModel(entry, estimatedTokens, skipKeys);
1438
+ }
1439
+ /**
1440
+ * Resolve a logical model group's member db ids to an ordered ChainRow[] for
1441
+ * strict group-pin routing (the "unify" feature). Each catalog-enabled member
1442
+ * is hydrated as a ChainRow carrying its active-profile/manual priority, then
1443
+ * ordered by the active strategy via orderChain. Auto-chain enabled/disabled is
1444
+ * intentionally ignored here because an explicit model request should still be
1445
+ * able to use a direct model that the user removed from auto routing.
1446
+ *
1447
+ * Pass the result to routeRequest() as `prefetchedChain` and DO NOT pass a
1448
+ * `preferredModelDbId` that isn't already one of these rows — otherwise the
1449
+ * preferred-model injection in routeRequest would unshift an off-group model and
1450
+ * the pin would no longer be strict (it could answer with a different model).
1451
+ */
1452
+ export function resolveModelGroupCandidates(memberDbIds,
1453
+ /**
1454
+ * Members that were reached only through a group's auto-derived slug, not the
1455
+ * id the client wrote (#651). They stay in the chain — resolution must never
1456
+ * shrink — but as a strictly lower tier, so they can serve only once every
1457
+ * literal match is exhausted. Omit it and every row is an equal candidate,
1458
+ * which is what every other caller wants.
1459
+ */
1460
+ demotedDbIds) {
1461
+ const db = getDb();
1462
+ const strategy = getRoutingStrategy();
1463
+ if (strategy !== 'priority')
1464
+ refreshStatsCache(db);
1465
+ const activeProfileId = getActiveProfileId(db);
1466
+ const selectMember = activeProfileId == null
1467
+ ? db.prepare(`
1468
+ SELECT m.id as model_db_id, COALESCE(fc.priority, 0) as priority,
1469
+ 1 as enabled,
1470
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
1471
+ m.size_label, m.monthly_token_budget,
1472
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
1473
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
1474
+ FROM models m
1475
+ LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
1476
+ WHERE m.id = ? AND m.enabled = 1
1477
+ `)
1478
+ : db.prepare(`
1479
+ SELECT m.id as model_db_id, COALESCE(pm.priority, fc.priority, 0) as priority,
1480
+ 1 as enabled,
1481
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
1482
+ m.size_label, m.monthly_token_budget,
1483
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
1484
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
1485
+ FROM models m
1486
+ LEFT JOIN profile_models pm ON pm.profile_id = ? AND pm.model_db_id = m.id
1487
+ LEFT JOIN fallback_config fc ON fc.model_db_id = m.id
1488
+ WHERE m.id = ? AND m.enabled = 1
1489
+ `);
1490
+ const rows = [];
1491
+ for (const id of memberDbIds) {
1492
+ const row = (activeProfileId == null ? selectMember.get(id) : selectMember.get(activeProfileId, id));
1493
+ if (!row)
1494
+ continue;
1495
+ row.match_tier = demotedDbIds?.has(id) ? 1 : 0;
1496
+ rows.push(row);
1497
+ }
1498
+ return orderChain(rows, strategy);
1499
+ }
1500
+ /**
1501
+ * The active fallback chain ordered by the current routing strategy, surfaced
1502
+ * for fusion panel selection. Same ordering the normal auto-router would walk,
1503
+ * so the panel's auto-pick draws from the highest-scored models first and the
1504
+ * fusion layer just needs to apply provider-diversity on top.
1505
+ */
1506
+ export function getOrderedFusionChain(estimatedTokens, exactOutputReserve = 0) {
1507
+ const db = getDb();
1508
+ const strategy = getRoutingStrategy();
1509
+ if (strategy !== 'priority')
1510
+ refreshStatsCache(db);
1511
+ const chain = getActiveChain(db).filter(e => e.enabled);
1512
+ // Only consider models that can ACTUALLY be served RIGHT NOW — applying the
1513
+ // same gate selectKeyForModel uses when the router walks the chain: the model
1514
+ // must have a key that is enabled + healthy, NOT on cooldown (e.g. a
1515
+ // HuggingFace key benched for a day after a 402 "Payment Required"), within
1516
+ // the provider's daily request cap, and under its per-minute/day request
1517
+ // limits. Without this, a high-strategy-ranked model whose only key is
1518
+ // currently cooled down (huggingface/Kimi-K2.6) would claim a panel slot it
1519
+ // can't fill — surfacing as "no available key" and pushing out a usable model,
1520
+ // which also makes the panel look like it's ignoring the routing strategy.
1521
+ //
1522
+ // The SIZE gates matter as much as the key gates: a model whose context window
1523
+ // cannot hold the prompt can NEVER fill its slot, yet diversifyChain keeps
1524
+ // handing it one on every request when it is its platform's only representative.
1525
+ // That leaves one panel slot dead on arrival and reports the failure as the
1526
+ // misleading "no available key for model". Passing a placeholder token count
1527
+ // here made both size gates no-ops.
1528
+ const usableKeys = db.prepare("SELECT id, platform, model_scope_json FROM api_keys WHERE enabled = 1 AND status IN ('healthy', 'unknown')").all();
1529
+ // Scope parsed once per key row (#657); the servable filter below re-checks
1530
+ // membership per model.
1531
+ const keysByPlatform = new Map();
1532
+ for (const k of usableKeys) {
1533
+ const entry = { id: k.id, scope: parseModelScope(k.model_scope_json) };
1534
+ const arr = keysByPlatform.get(k.platform);
1535
+ if (arr)
1536
+ arr.push(entry);
1537
+ else
1538
+ keysByPlatform.set(k.platform, [entry]);
1539
+ }
1540
+ // Soft preference (#956 review): prefer models whose window holds the estimate
1541
+ // WITH the safety margin; /v1/models still advertises the raw window, so if
1542
+ // NOTHING survives that pass, re-run allowing raw advertised-window fits
1543
+ // rather than handing back an empty chain — a request packed to the
1544
+ // advertised window keeps its one attempt (the mid-chain context_too_large
1545
+ // 400 is classified and retried downstream).
1546
+ const passesContextGate = (e, allowMarginViolators) => {
1547
+ // A null context_window means "unknown", not "zero": same convention the
1548
+ // auto-router uses, so an unspecified window is never itself a reason to skip.
1549
+ if (fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve))
1550
+ return true;
1551
+ return allowMarginViolators && fitsContextWindowStrict(e.context_window, estimatedTokens);
1552
+ };
1553
+ const servableFilter = (allowMarginViolators) => chain.filter(e => {
1554
+ if (!passesContextGate(e, allowMarginViolators))
1555
+ return false;
1556
+ const keyIds = keysByPlatform.get(e.platform);
1557
+ if (!keyIds)
1558
+ return false;
1559
+ // Same endpoint-pool rule the router applies (#619).
1560
+ const endpointKeyIds = e.platform === 'custom' && e.key_id != null
1561
+ ? customEndpointKeyIds(db, e.key_id)
1562
+ : null;
1563
+ const limits = { rpm: e.rpm_limit, rpd: e.rpd_limit, tpm: e.tpm_limit, tpd: e.tpd_limit };
1564
+ return keyIds.some(({ id: kid, scope }) => scopeAllows(scope, e.model_id) &&
1565
+ (endpointKeyIds == null || endpointKeyIds.has(kid)) &&
1566
+ !isOnCooldown(e.platform, e.model_id, kid) &&
1567
+ canUseProvider(e.platform, kid) &&
1568
+ canUseProviderMinute(e.platform, kid) &&
1569
+ canMakeRequest(e.platform, e.model_id, kid, limits) &&
1570
+ canUseProviderTokens(e.platform, kid, e.model_id, estimatedTokens));
1571
+ });
1572
+ let servable = servableFilter(false);
1573
+ if (servable.length === 0)
1574
+ servable = servableFilter(true);
1575
+ // Deterministic (expected-score) ordering so the panel faithfully follows the
1576
+ // user's picked routing strategy instead of re-sampling a fresh draw each call.
1577
+ const ordered = orderChain(servable, strategy, false);
1578
+ return ordered.map(e => ({
1579
+ modelDbId: e.model_db_id,
1580
+ platform: e.platform,
1581
+ modelId: e.model_id,
1582
+ displayName: e.display_name,
1583
+ sizeLabel: e.size_label,
1584
+ supportsVision: e.supports_vision,
1585
+ supportsTools: e.supports_tools,
1586
+ }));
1587
+ }
1588
+ /**
1589
+ * Resolve an explicit model id (as a client would type it) to a fusion
1590
+ * candidate, or null when it isn't a known enabled model. Prefers an enabled
1591
+ * row; dedupes a model id that exists on multiple platforms by intelligence
1592
+ * rank, matching how /v1/models picks a representative row.
1593
+ */
1594
+ export function resolveFusionCandidate(modelId) {
1595
+ const db = getDb();
1596
+ const rows = db.prepare(`
1597
+ SELECT m.id as model_db_id, m.platform, m.model_id, m.display_name,
1598
+ m.size_label, m.supports_vision, m.supports_tools
1599
+ FROM models m
1600
+ WHERE m.model_id = ? AND m.enabled = 1
1601
+ ORDER BY m.intelligence_rank ASC, m.id ASC
1602
+ `).all(modelId);
1603
+ if (rows.length > 0) {
1604
+ // A logical model can have several enabled provider rows. Prefer one with
1605
+ // an enabled, healthy/unknown key before falling back to the deterministic
1606
+ // ranking. Without this check a duplicate alias can pin fusion to a
1607
+ // keyless provider (for example a free relay) while a configured provider
1608
+ // for the same model is available.
1609
+ const row = rows.find(candidate => routableKeyIdsForModel(candidate.model_db_id).length > 0) ?? rows[0];
1610
+ return {
1611
+ modelDbId: row.model_db_id,
1612
+ platform: row.platform,
1613
+ modelId: row.model_id,
1614
+ displayName: row.display_name,
1615
+ sizeLabel: row.size_label,
1616
+ supportsVision: row.supports_vision,
1617
+ supportsTools: row.supports_tools,
1618
+ };
1619
+ }
1620
+ // Unify ON: a fusion picker value may be a canonical GROUP id rather than a
1621
+ // raw model_id. Resolve it to the group's best-ordered enabled member so
1622
+ // saved fusion configs that use canonical ids keep working. Exact model_id
1623
+ // match above always wins first, so OFF mode and legacy configs are untouched.
1624
+ if (isUnifyEnabled()) {
1625
+ const resolved = resolveRequestedIdForDispatch(modelId, getModelGroups());
1626
+ if (resolved && resolved.memberDbIds.length > 0) {
1627
+ const candidates = resolveModelGroupCandidates(resolved.memberDbIds, resolved.demotedDbIds);
1628
+ // Bare unified aliases may resolve to a provider row that is enabled in
1629
+ // the catalog but has no usable key. Select the first routable member so
1630
+ // an explicit fusion panel does not waste a slot on that dead end.
1631
+ const top = candidates.find(candidate => routableKeyIdsForModel(candidate.model_db_id).length > 0) ?? candidates[0];
1632
+ if (top) {
1633
+ return {
1634
+ modelDbId: top.model_db_id,
1635
+ platform: top.platform,
1636
+ modelId: top.model_id,
1637
+ displayName: top.display_name,
1638
+ sizeLabel: top.size_label,
1639
+ supportsVision: top.supports_vision,
1640
+ supportsTools: top.supports_tools,
1641
+ };
1642
+ }
1643
+ }
1644
+ }
1645
+ return null;
1646
+ }
1647
+ export function routeRequest(estimatedTokens = 1000, skipKeys, preferredModelDbId, requireVision = false, requireTools = false, skipModels, prefetchedChain, requireStructured = false, skipPlatforms, exactOutputReserve = 0, task) {
1648
+ const db = getDb();
1649
+ const strategy = getRoutingStrategy();
1650
+ if (strategy !== 'priority')
1651
+ refreshStatsCache(db);
1652
+ const chain = (prefetchedChain ?? getActiveChain(db)).filter(e => e.enabled);
1653
+ const sortedChain = orderChain(chain, strategy, true, task);
1654
+ const keyCounts = usableKeyCountsByPlatform(db);
1655
+ // Exploration toggle (#685/#707 follow-up): when enabled, give a model with
1656
+ // no reliability/speed samples a guaranteed chance to be tried, so it stops
1657
+ // losing every bandit draw to prior-heavy rivals. With EXPLORE_CHANCE
1658
+ // probability, pick one unmeasured model uniformly and try it first; if it
1659
+ // fails, the loop falls through to the scored order as usual. Only for
1660
+ // bandit strategies — Manual is the operator's explicit order.
1661
+ // Candidates the main loop would immediately reject for THIS request are
1662
+ // excluded up front (ruled-out models plus the request-level capability
1663
+ // gates: vision/tools/structured output/context window): promoting a model
1664
+ // that can't serve the request ahead of capable ones would just get it
1665
+ // skipped a moment later (e.g. an image request must never randomly probe a
1666
+ // text-only model).
1667
+ // While the gateway is in degraded mode (#904) exploration is skipped
1668
+ // entirely: probing unmeasured models during a fleet-wide outage just burns
1669
+ // retry budget on the same dead providers, and the scored order of known
1670
+ // survivors is the only thing worth trying.
1671
+ if (strategy !== 'priority' && getExploreEnabled() && !isDegraded() && Math.random() < EXPLORE_CHANCE) {
1672
+ // A model the operator zeroed out via MODEL_ROUTING_OVERRIDES never wins a
1673
+ // bandit draw, so it would stay under EXPLORE_MIN_SAMPLES forever and become
1674
+ // a perpetual probe target — the explicit ban outranks exploration.
1675
+ const overrides = getModelWeightOverrides();
1676
+ const unmeasured = sortedChain.filter(e => {
1677
+ if (overrides.get(e.model_id) === 0)
1678
+ return false;
1679
+ if ((keyCounts.get(e.platform) ?? 0) <= 0)
1680
+ return false;
1681
+ const stats = statsCache?.get(modelStatsKey(e.platform, e.model_id, e.endpoint_scope));
1682
+ if ((stats?.successes ?? 0) + (stats?.failures ?? 0) >= EXPLORE_MIN_SAMPLES)
1683
+ return false;
1684
+ // Mirror the main loop's gates below so exploration only samples
1685
+ // candidates that can actually serve this request.
1686
+ if (skipModels?.has(e.model_db_id))
1687
+ return false;
1688
+ if (skipPlatforms?.has(e.platform))
1689
+ return false;
1690
+ if (requireVision && !e.supports_vision)
1691
+ return false;
1692
+ if (requireTools && !e.supports_tools)
1693
+ return false;
1694
+ // Never spend the exploration slot on a model that keeps rejecting tool
1695
+ // requests (#1230); it stays reachable at the back of the main walk.
1696
+ if (requireTools && isToolBenched(e.platform, e.model_id, e.endpoint_scope))
1697
+ return false;
1698
+ if (requireStructured && platformDropsResponseFormat(e.platform))
1699
+ return false;
1700
+ if (!fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve))
1701
+ return false;
1702
+ if (e.tpm_limit != null && estimatedTokens > e.tpm_limit)
1703
+ return false;
1704
+ return true;
1705
+ });
1706
+ if (unmeasured.length > 0) {
1707
+ const probe = unmeasured[Math.floor(Math.random() * unmeasured.length)];
1708
+ const idx = sortedChain.findIndex(e => e.model_db_id === probe.model_db_id);
1709
+ if (idx > 0) {
1710
+ const [probeRow] = sortedChain.splice(idx, 1);
1711
+ sortedChain.unshift(probeRow);
1712
+ }
1713
+ }
1714
+ }
1715
+ // Sticky session / Explicit pinning: move preferred model to front of chain
1716
+ if (preferredModelDbId) {
1717
+ const idx = sortedChain.findIndex(e => e.model_db_id === preferredModelDbId);
1718
+ if (idx >= 0) {
1719
+ if (idx > 0) {
1720
+ const [preferred] = sortedChain.splice(idx, 1);
1721
+ sortedChain.unshift(preferred);
1722
+ }
1723
+ }
1724
+ else {
1725
+ // The requested model is not in the current routing chain (e.g. it's a
1726
+ // custom model or not added to the active profile). We must fulfill the
1727
+ // explicit request by injecting it at the front.
1728
+ const pinnedRow = db.prepare(`
1729
+ SELECT m.id as model_db_id, 0 as priority, 1 as enabled,
1730
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
1731
+ m.size_label, m.monthly_token_budget,
1732
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
1733
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
1734
+ FROM models m
1735
+ WHERE m.id = ? AND m.enabled = 1
1736
+ `).get(preferredModelDbId);
1737
+ if (pinnedRow) {
1738
+ sortedChain.unshift(pinnedRow);
1739
+ }
1740
+ }
1741
+ }
1742
+ // Drop models whose platform has NO enabled+healthy key before the walk. Such
1743
+ // a row can never produce a route (selectKeyForModel's first query returns
1744
+ // empty for it), so walking it is pure overhead on every request, and its diag
1745
+ // line pads the exhaustion summary with a constant that has nothing to do with
1746
+ // why THIS request failed. On the clean tier that was 14 of 36 rows, reported
1747
+ // as "37 routes checked" when only 22 were ever candidates.
1748
+ //
1749
+ // An explicit pin is exempt: the client named that model, so it still gets
1750
+ // walked and still reports "no enabled+healthy key for platform" against its
1751
+ // own label rather than vanishing into an aggregate.
1752
+ const isRoutable = (e) => e.model_db_id === preferredModelDbId || (keyCounts.get(e.platform) ?? 0) > 0;
1753
+ const routableChain = sortedChain.filter(isRoutable);
1754
+ const keylessSkipped = sortedChain.length - routableChain.length;
1755
+ // One aggregate line, not one per model: the platforms stay visible to anyone
1756
+ // reading RouteError.diagnostics (and keep routingExhaustionBody classifying a
1757
+ // fully-unconfigured pool as 503 config, not a 429 rate limit), without N
1758
+ // near-identical rows drowning the request's real reasons.
1759
+ const keylessLine = keylessSkipped > 0
1760
+ ? `${keylessSkipped} model(s) skipped: no enabled+healthy key for platform (${[...new Set(sortedChain.filter(e => !isRoutable(e)).map(e => e.platform))].sort().join(', ')})`
1761
+ : null;
1762
+ // Per-model disposition, attached to the exhaustion error when the loop falls
1763
+ // through with no route — the only record of WHY the pool was empty on the
1764
+ // synchronous "all exhausted" path (nothing downstream logs it). See issue _1.
1765
+ const diag = [];
1766
+ // Margin as a SOFT preference (#956 review): /v1/models advertises the raw
1767
+ // window, so clients legitimately pack requests right up to it — excluding
1768
+ // those models outright turned an already-handled upstream 400 into "all
1769
+ // models exhausted" with zero attempts. Keep the operator's order intact but
1770
+ // sweep margin-fitting models first; ones that only fit the advertised window
1771
+ // stay eligible behind them. Worst case is one classified context_too_large
1772
+ // hop instead of no route at all.
1773
+ //
1774
+ // Same soft treatment for models that keep answering tool requests with a 400
1775
+ // (#1230, lib/tool-capability.ts): on a tool request they go to the very back
1776
+ // instead of being excluded, so the worst case is the old order, never an
1777
+ // empty pool. An explicit pin keeps its place: the client named that model.
1778
+ const servingChain = [];
1779
+ const marginDeferred = [];
1780
+ const toolDeferred = [];
1781
+ for (const e of routableChain) {
1782
+ if (requireTools && e.model_db_id !== preferredModelDbId && isToolBenched(e.platform, e.model_id, e.endpoint_scope)) {
1783
+ toolDeferred.push(e);
1784
+ continue;
1785
+ }
1786
+ (fitsContextWindow(e.platform, e.context_window, estimatedTokens, exactOutputReserve) ? servingChain : marginDeferred).push(e);
1787
+ }
1788
+ servingChain.push(...marginDeferred, ...toolDeferred);
1789
+ for (const entry of servingChain) {
1790
+ const label = `${entry.platform}/${entry.model_id}`;
1791
+ // Models the caller has ruled out for this request — e.g. a 404
1792
+ // "model removed upstream" already seen this request: trying the same
1793
+ // model again on a different key would just burn another attempt on the
1794
+ // same dead route (PR #111, credits @barbotkonv).
1795
+ if (skipModels?.has(entry.model_db_id)) {
1796
+ diag.push(`${label}: ruled out earlier this request`);
1797
+ continue;
1798
+ }
1799
+ // Platforms the caller has ruled out wholesale (#788): a provider-level
1800
+ // failure this request — a 5xx, a timeout, a dead socket — is about the
1801
+ // PROVIDER, so its other keys and its other models would fail the same way.
1802
+ // Skipping the platform moves failover to the next provider instead of
1803
+ // burning one hop per key. Request-scoped; nothing is benched by this.
1804
+ if (skipPlatforms?.has(entry.platform)) {
1805
+ diag.push(`${label}: provider ruled out earlier this request`);
1806
+ continue;
1807
+ }
1808
+ // Vision requests skip text-only models — including a sticky/preferred one,
1809
+ // which is correct: don't pin an image turn to a model that can't see it.
1810
+ if (requireVision && !entry.supports_vision) {
1811
+ diag.push(`${label}: no vision support`);
1812
+ continue;
1813
+ }
1814
+ // Tool-bearing requests skip models that can't emit structured tool_calls.
1815
+ // A model that "answers" a tool request with the call serialized as text
1816
+ // looks successful at the transport level while the client's harness sees
1817
+ // nothing — worse than a failover. Applies to sticky models too, same
1818
+ // reasoning as vision above.
1819
+ if (requireTools && !entry.supports_tools) {
1820
+ diag.push(`${label}: no tool-calling support`);
1821
+ continue;
1822
+ }
1823
+ // Structured-output routing (#514 follow-up): when the request carries a
1824
+ // response_format, skip platforms whose param policy can't even receive it
1825
+ // (the param would be dropped before send, so the model would answer in
1826
+ // prose and burn a failover hop). Platform-level fast path — model-level
1827
+ // capability isn't in the catalog; models that accept the param but ignore
1828
+ // it are caught by the non-stream JSON enforcement downstream.
1829
+ if (requireStructured && platformDropsResponseFormat(entry.platform)) {
1830
+ diag.push(`${label}: platform drops response_format`);
1831
+ continue;
1832
+ }
1833
+ // Context-aware routing fast path (#167): skip a model whose RAW advertised
1834
+ // window cannot hold the request — a dispatch there is a guaranteed 413.
1835
+ // Margin-violating-but-raw-fitting models are NOT skipped (soft preference,
1836
+ // #956 review): they were merely deferred to the back of the sweep above.
1837
+ // estimatedTokens is the INPUT estimate plus a CAPPED output reserve
1838
+ // (routingReserveTokens, #470), so a huge client-set max_tokens no longer
1839
+ // excludes the model — the input must fit, not input+full max_tokens. A 413
1840
+ // that slips through is still retryable downstream, and the failed model is
1841
+ // put on cooldown — so this is a fast-path, not the only guard. If every
1842
+ // model is too small, the loop falls through and the caller gets the normal
1843
+ // "all models exhausted" error rather than a wasted sweep.
1844
+ if (!fitsContextWindowStrict(entry.context_window, estimatedTokens)) {
1845
+ // Keep the `< estimated` substring — summarizeExhaustion buckets prompt
1846
+ // overflow off it. Emit the EFFECTIVE number so the line doesn't read as
1847
+ // false (#956 review): e.g. `context 131072 < estimated 106000 x1.25 = 132500`.
1848
+ const reserve = Math.max(0, exactOutputReserve);
1849
+ const requiredWithMargin = Math.ceil(Math.max(0, estimatedTokens - reserve) * CONTEXT_WINDOW_SAFETY_FACTOR + reserve);
1850
+ diag.push(TRIM_GUARDED_PLATFORMS.has(entry.platform)
1851
+ ? `${label}: context ${entry.context_window} < estimated ${estimatedTokens}`
1852
+ : `${label}: context ${entry.context_window} < estimated ${estimatedTokens} x${CONTEXT_WINDOW_SAFETY_FACTOR} = ${requiredWithMargin}`);
1853
+ continue;
1854
+ }
1855
+ // Same guard for a model with a small per-minute token budget: a request
1856
+ // whose input alone exceeds tpm_limit can never fit one minute of quota and
1857
+ // returns a guaranteed 413 (e.g. Groq gpt-oss-120b: 131k context but 8k TPM).
1858
+ // estimatedTokens carries the same capped output reserve, mirroring the
1859
+ // check above (#470).
1860
+ if (entry.tpm_limit != null && estimatedTokens > entry.tpm_limit) {
1861
+ diag.push(`${label}: tpm_limit ${entry.tpm_limit} < estimated ${estimatedTokens}`);
1862
+ continue;
1863
+ }
1864
+ // Key selection + accounting pre-checks for this one model. Returns the
1865
+ // first usable key's RouteResult, or null when the model has no key that
1866
+ // can serve right now — in which case we fall through to the next model in
1867
+ // the sorted chain for THIS request (no explicit penalty needed).
1868
+ const route = selectKeyForModel(entry, estimatedTokens, skipKeys, diag);
1869
+ if (route)
1870
+ return route;
1871
+ }
1872
+ // The aggregate keyless line rides in diagnostics but NOT in the summary's
1873
+ // route count: those models were never candidates for this request.
1874
+ throw new RouteError(summarizeExhaustion(diag, getSoonestCooldownExpiry(), Date.now(), keylessSkipped), 429, keylessLine ? [...diag, keylessLine] : diag);
1875
+ }
1876
+ export function getRoutingScores() {
1877
+ const db = getDb();
1878
+ const strategy = getRoutingStrategy();
1879
+ refreshStatsCache(db);
1880
+ // Score the whole enabled catalog, not just the active chain (#1047). Since
1881
+ // #1023 the dashboard table lists every catalog row through the chain, and
1882
+ // merging it against chain-only scores left every not-yet-opted-in row with
1883
+ // blank reliability/speed — which read as "each new chain starts from
1884
+ // scratch" even though the underlying per-model stats are global. Rows the
1885
+ // chain names keep its enabled flag; the rest score as disabled. Display
1886
+ // only: routeRequest still walks getActiveChain.
1887
+ const profileId = getActiveProfileId(db);
1888
+ const chain = db.prepare(`
1889
+ SELECT m.id AS model_db_id, COALESCE(c.priority, 0) AS priority, COALESCE(c.enabled, 0) AS enabled,
1890
+ m.platform, m.model_id, m.display_name, m.intelligence_rank,
1891
+ m.size_label, m.monthly_token_budget,
1892
+ m.rpm_limit, m.rpd_limit, m.tpm_limit, m.tpd_limit, m.supports_vision,
1893
+ m.supports_tools, m.context_window, m.key_id, m.endpoint_scope
1894
+ FROM models m
1895
+ LEFT JOIN ${profileId != null
1896
+ ? '(SELECT model_db_id, priority, enabled FROM profile_models WHERE profile_id = ?) c'
1897
+ : '(SELECT model_db_id, priority, enabled FROM fallback_config) c'}
1898
+ ON c.model_db_id = m.id
1899
+ WHERE m.enabled = 1
1900
+ ORDER BY c.priority ASC
1901
+ `).all(...(profileId != null ? [profileId] : []));
1902
+ // For display we score under 'balanced' weights when in priority mode, so the
1903
+ // table still shows a meaningful ranking even with the bandit turned off.
1904
+ const weights = weightsFor(strategy) ?? BANDIT_PRESETS.balanced;
1905
+ const composites = chain.map(e => intelligenceComposite(e.size_label, e.intelligence_rank));
1906
+ const intelMin = composites.length ? Math.min(...composites) : 0;
1907
+ const intelMax = composites.length ? Math.max(...composites) : 0;
1908
+ const keyCounts = usableKeyCountsByPlatform(db);
1909
+ const headroomCfg = getHeadroomThresholds();
1910
+ const scores = chain.map(entry => {
1911
+ const scored = scoreChainEntry(entry, weights, intelMin, intelMax, false, keyCounts, headroomCfg);
1912
+ const stats = statsCache?.get(modelStatsKey(entry.platform, entry.model_id, entry.endpoint_scope));
1913
+ return {
1914
+ modelDbId: entry.model_db_id,
1915
+ platform: entry.platform,
1916
+ modelId: entry.model_id,
1917
+ displayName: entry.display_name,
1918
+ enabled: entry.enabled === 1,
1919
+ reliability: scored.axes.reliability,
1920
+ speed: scored.axes.speed,
1921
+ intelligence: scored.axes.intelligence,
1922
+ headroom: scored.headroom,
1923
+ rateLimit: scored.rateLimit,
1924
+ score: scored.score,
1925
+ totalRequests: Math.round((stats?.successes ?? 0) + (stats?.failures ?? 0)),
1926
+ };
1927
+ }).sort((a, b) => b.score - a.score);
1928
+ // customWeights is always present (the saved vector, or the balanced default)
1929
+ // so the dashboard's custom-weight sliders can render even before the user
1930
+ // has saved their own — distinct from `weights`, which is null in priority
1931
+ // mode and the active preset otherwise.
1932
+ // exploreEnabled must ride along here too: the dashboard checkbox renders
1933
+ // from GET /routing, so omitting it would make the toggle look permanently
1934
+ // off (and impossible to turn off) after a refetch. Same for the key
1935
+ // selection picker (#919).
1936
+ // peakAdjusted tells the dashboard whether the weight summary it is about to
1937
+ // render is the raw preset or a peak-hours variant of it (#760) — without it
1938
+ // the numbers would change under the operator with nothing to explain why.
1939
+ const active = weightsWithPeak(strategy);
1940
+ return {
1941
+ strategy,
1942
+ keySelectionStrategy: getKeySelectionStrategy(),
1943
+ weights: active.weights,
1944
+ customWeights: getCustomWeights(),
1945
+ exploreEnabled: getExploreEnabled(),
1946
+ peakAdjusted: active.adjusted,
1947
+ peakHours: getPeakHoursConfig(),
1948
+ scores,
1949
+ };
1950
+ }
1951
+ /**
1952
+ * Filter a sticky-session pin down to something still routable (#634).
1953
+ *
1954
+ * A sticky entry holds a model db id for up to 30 minutes, so it goes stale the
1955
+ * moment the operator disables that model — in the catalog, or just for auto
1956
+ * routing. It must NOT be handed to routeRequest as-is: an off-chain preferred
1957
+ * id is treated as an explicit pin and injected ahead of the chain, which is
1958
+ * right for a client that named the model and wrong for a pin the client never
1959
+ * asked for. Dropping it here falls the request through to normal auto routing.
1960
+ *
1961
+ * Pass the same chain the request will route over (the prefetched auto chain);
1962
+ * omit it to check the active chain, which is what routeRequest would use.
1963
+ */
1964
+ export function resolveStickyPreference(stickyModelDbId, chain) {
1965
+ if (stickyModelDbId == null)
1966
+ return undefined;
1967
+ const rows = chain ?? getActiveChain(getDb());
1968
+ return rows.some(entry => entry.model_db_id === stickyModelDbId && entry.enabled)
1969
+ ? stickyModelDbId
1970
+ : undefined;
1971
+ }
1972
+ // Whether at least one vision-capable model is enabled in the fallback chain.
1973
+ // Used to give image requests a clear "enable a vision model" error instead of
1974
+ // the generic exhaustion message when none is configured (#118, #125).
1975
+ export function hasEnabledVisionModel() {
1976
+ const db = getDb();
1977
+ return getActiveChain(db).some(entry => entry.enabled === 1 && entry.supports_vision === 1);
1978
+ }
1979
+ // Whether at least one tool-capable model is enabled in the fallback chain.
1980
+ // Same role as hasEnabledVisionModel: a clear up-front error for tool-bearing
1981
+ // requests beats routing them to a model that mangles the tool call.
1982
+ export function hasEnabledToolsModel() {
1983
+ const db = getDb();
1984
+ return getActiveChain(db).some(entry => entry.enabled === 1 && entry.supports_tools === 1);
1985
+ }
1986
+ //# sourceMappingURL=router.js.map