@srouterhq/server 0.0.0-stage → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (1027) hide show
  1. package/LICENSE +21 -0
  2. package/client/dist/assets/am-TQ7Jmdqe.js +1 -0
  3. package/client/dist/assets/ar-Bf5a7yfC.js +1 -0
  4. package/client/dist/assets/az-CrosE1xH.js +1 -0
  5. package/client/dist/assets/bg-BvWnmcUz.js +1 -0
  6. package/client/dist/assets/bn-ams5TsJp.js +1 -0
  7. package/client/dist/assets/cs-Dh67CaJ_.js +1 -0
  8. package/client/dist/assets/da-9WlT8Iq7.js +1 -0
  9. package/client/dist/assets/de-lWhJJzz7.js +1 -0
  10. package/client/dist/assets/el-Cu7_GhNq.js +1 -0
  11. package/client/dist/assets/es-BZuzsBcP.js +1 -0
  12. package/client/dist/assets/fa-Dwdn-jKS.js +1 -0
  13. package/client/dist/assets/fi-BFrFyOTy.js +1 -0
  14. package/client/dist/assets/fr-XhbtmPpj.js +1 -0
  15. package/client/dist/assets/geist-cyrillic-ext-wght-normal-DjL33-gN.woff2 +0 -0
  16. package/client/dist/assets/geist-cyrillic-wght-normal-BEAKL7Jp.woff2 +0 -0
  17. package/client/dist/assets/geist-latin-ext-wght-normal-DC-KSUi6.woff2 +0 -0
  18. package/client/dist/assets/geist-latin-wght-normal-BgDaEnEv.woff2 +0 -0
  19. package/client/dist/assets/geist-mono-cyrillic-ext-wght-normal-I4S5GZfc.woff2 +0 -0
  20. package/client/dist/assets/geist-mono-cyrillic-wght-normal-BmXc_FBt.woff2 +0 -0
  21. package/client/dist/assets/geist-mono-latin-ext-wght-normal-DrnZ1wKl.woff2 +0 -0
  22. package/client/dist/assets/geist-mono-latin-wght-normal-B_7UjwxQ.woff2 +0 -0
  23. package/client/dist/assets/geist-mono-symbols2-wght-normal-GZpp1pK2.woff2 +0 -0
  24. package/client/dist/assets/geist-mono-vietnamese-wght-normal-D8KDMBhC.woff2 +0 -0
  25. package/client/dist/assets/geist-vietnamese-wght-normal-6IgcOCM7.woff2 +0 -0
  26. package/client/dist/assets/gu-D0uhLUB8.js +1 -0
  27. package/client/dist/assets/ha-CVlm9pqp.js +1 -0
  28. package/client/dist/assets/he-CMIFeoWr.js +1 -0
  29. package/client/dist/assets/hi-ENTwhBpn.js +1 -0
  30. package/client/dist/assets/hr-fc03vMy_.js +1 -0
  31. package/client/dist/assets/hu-YBo2BIYt.js +1 -0
  32. package/client/dist/assets/id-BEWXYX69.js +1 -0
  33. package/client/dist/assets/ig-BKLZFKka.js +1 -0
  34. package/client/dist/assets/index-D9I4KtvR.js +145 -0
  35. package/client/dist/assets/index-txhEmCdb.css +2 -0
  36. package/client/dist/assets/it-Bzmei1Y2.js +1 -0
  37. package/client/dist/assets/ja-BfcpuqEl.js +1 -0
  38. package/client/dist/assets/ka-CnWjRgLo.js +1 -0
  39. package/client/dist/assets/km-BAjPR3Qj.js +1 -0
  40. package/client/dist/assets/kn-B15WKU5m.js +1 -0
  41. package/client/dist/assets/ko-C61fAB-w.js +1 -0
  42. package/client/dist/assets/lt-CYTr4_YW.js +1 -0
  43. package/client/dist/assets/ml-CtqxgyH3.js +1 -0
  44. package/client/dist/assets/mr-X5TMY7aR.js +1 -0
  45. package/client/dist/assets/ms-DSg8RqPS.js +1 -0
  46. package/client/dist/assets/my-D7RqR8uE.js +1 -0
  47. package/client/dist/assets/ne-DaflxkDj.js +1 -0
  48. package/client/dist/assets/nl-BOqqtsC8.js +1 -0
  49. package/client/dist/assets/no-BSHp_StB.js +1 -0
  50. package/client/dist/assets/or-BkLR9Vub.js +1 -0
  51. package/client/dist/assets/pa-CarJ2XgI.js +1 -0
  52. package/client/dist/assets/pl-C2n0V-0k.js +1 -0
  53. package/client/dist/assets/pt-BR-CqJdNhg2.js +1 -0
  54. package/client/dist/assets/pt-PT-C_HpMyKz.js +1 -0
  55. package/client/dist/assets/ro-DVRZqSAS.js +1 -0
  56. package/client/dist/assets/ru-DwAJG7VC.js +1 -0
  57. package/client/dist/assets/si-BxUezXm0.js +1 -0
  58. package/client/dist/assets/sk-eI1VcIfY.js +1 -0
  59. package/client/dist/assets/sr-CJF3auCr.js +1 -0
  60. package/client/dist/assets/srouter-logo-C6ZfjGIi.svg +14 -0
  61. package/client/dist/assets/sv-nqvwHAq_.js +1 -0
  62. package/client/dist/assets/sw-DWynE2pW.js +1 -0
  63. package/client/dist/assets/ta-BVM12knC.js +1 -0
  64. package/client/dist/assets/te-oyzEkF7_.js +1 -0
  65. package/client/dist/assets/th-COVeE0ZQ.js +1 -0
  66. package/client/dist/assets/tl-DsJANnQz.js +1 -0
  67. package/client/dist/assets/tr-B-Fl-urO.js +1 -0
  68. package/client/dist/assets/uk-C5qketAO.js +1 -0
  69. package/client/dist/assets/ur-QU9KugfV.js +1 -0
  70. package/client/dist/assets/uz-DL9cG_mY.js +1 -0
  71. package/client/dist/assets/vi-CQiVy3Pu.js +1 -0
  72. package/client/dist/assets/yo-DqazMdrZ.js +1 -0
  73. package/client/dist/assets/zh-CN-DnYL_Kio.js +1 -0
  74. package/client/dist/assets/zh-TW-5bdbosKj.js +1 -0
  75. package/client/dist/favicon.svg +14 -0
  76. package/client/dist/icons.svg +24 -0
  77. package/client/dist/index.html +37 -0
  78. package/dist/app.d.ts +4 -0
  79. package/dist/app.d.ts.map +1 -0
  80. package/dist/app.js +372 -0
  81. package/dist/app.js.map +1 -0
  82. package/dist/db/index.d.ts +49 -0
  83. package/dist/db/index.d.ts.map +1 -0
  84. package/dist/db/index.js +203 -0
  85. package/dist/db/index.js.map +1 -0
  86. package/dist/db/migrate/TEMPLATE.d.ts +4 -0
  87. package/dist/db/migrate/TEMPLATE.d.ts.map +1 -0
  88. package/dist/db/migrate/TEMPLATE.js +18 -0
  89. package/dist/db/migrate/TEMPLATE.js.map +1 -0
  90. package/dist/db/migrate/cli.d.ts +2 -0
  91. package/dist/db/migrate/cli.d.ts.map +1 -0
  92. package/dist/db/migrate/cli.js +177 -0
  93. package/dist/db/migrate/cli.js.map +1 -0
  94. package/dist/db/migrate/defaults.d.ts +52 -0
  95. package/dist/db/migrate/defaults.d.ts.map +1 -0
  96. package/dist/db/migrate/defaults.js +126 -0
  97. package/dist/db/migrate/defaults.js.map +1 -0
  98. package/dist/db/migrate/runner.d.ts +16 -0
  99. package/dist/db/migrate/runner.d.ts.map +1 -0
  100. package/dist/db/migrate/runner.js +178 -0
  101. package/dist/db/migrate/runner.js.map +1 -0
  102. package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts +4 -0
  103. package/dist/db/migrations/20260101_000000_legacy_baseline.d.ts.map +1 -0
  104. package/dist/db/migrations/20260101_000000_legacy_baseline.js +2240 -0
  105. package/dist/db/migrations/20260101_000000_legacy_baseline.js.map +1 -0
  106. package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts +4 -0
  107. package/dist/db/migrations/20260627_000001_custom_provider_modalities.d.ts.map +1 -0
  108. package/dist/db/migrations/20260627_000001_custom_provider_modalities.js +27 -0
  109. package/dist/db/migrations/20260627_000001_custom_provider_modalities.js.map +1 -0
  110. package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts +4 -0
  111. package/dist/db/migrations/20260627_000002_catalog_model_state.d.ts.map +1 -0
  112. package/dist/db/migrations/20260627_000002_catalog_model_state.js +30 -0
  113. package/dist/db/migrations/20260627_000002_catalog_model_state.js.map +1 -0
  114. package/dist/db/migrations/20260628_120000_request_aggregates.d.ts +6 -0
  115. package/dist/db/migrations/20260628_120000_request_aggregates.d.ts.map +1 -0
  116. package/dist/db/migrations/20260628_120000_request_aggregates.js +104 -0
  117. package/dist/db/migrations/20260628_120000_request_aggregates.js.map +1 -0
  118. package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts +9 -0
  119. package/dist/db/migrations/20260630_000001_github_gpt41_context.d.ts.map +1 -0
  120. package/dist/db/migrations/20260630_000001_github_gpt41_context.js +22 -0
  121. package/dist/db/migrations/20260630_000001_github_gpt41_context.js.map +1 -0
  122. package/dist/db/migrations/20260706_000001_request_client_info.d.ts +4 -0
  123. package/dist/db/migrations/20260706_000001_request_client_info.d.ts.map +1 -0
  124. package/dist/db/migrations/20260706_000001_request_client_info.js +30 -0
  125. package/dist/db/migrations/20260706_000001_request_client_info.js.map +1 -0
  126. package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts +17 -0
  127. package/dist/db/migrations/20260706_000002_custom_model_tool_support.d.ts.map +1 -0
  128. package/dist/db/migrations/20260706_000002_custom_model_tool_support.js +20 -0
  129. package/dist/db/migrations/20260706_000002_custom_model_tool_support.js.map +1 -0
  130. package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts +12 -0
  131. package/dist/db/migrations/20260714_000001_profile_chain_backfill.d.ts.map +1 -0
  132. package/dist/db/migrations/20260714_000001_profile_chain_backfill.js +45 -0
  133. package/dist/db/migrations/20260714_000001_profile_chain_backfill.js.map +1 -0
  134. package/dist/db/migrations/20260720_000001_key_health_error.d.ts +5 -0
  135. package/dist/db/migrations/20260720_000001_key_health_error.d.ts.map +1 -0
  136. package/dist/db/migrations/20260720_000001_key_health_error.js +16 -0
  137. package/dist/db/migrations/20260720_000001_key_health_error.js.map +1 -0
  138. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts +4 -0
  139. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.d.ts.map +1 -0
  140. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js +38 -0
  141. package/dist/db/migrations/20260726_000001_cooldown_probe_provenance.js.map +1 -0
  142. package/dist/db/migrations/20260726_000002_request_attempts.d.ts +4 -0
  143. package/dist/db/migrations/20260726_000002_request_attempts.d.ts.map +1 -0
  144. package/dist/db/migrations/20260726_000002_request_attempts.js +43 -0
  145. package/dist/db/migrations/20260726_000002_request_attempts.js.map +1 -0
  146. package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts +4 -0
  147. package/dist/db/migrations/20260726_000003_model_source_provenance.d.ts.map +1 -0
  148. package/dist/db/migrations/20260726_000003_model_source_provenance.js +80 -0
  149. package/dist/db/migrations/20260726_000003_model_source_provenance.js.map +1 -0
  150. package/dist/db/migrations/20260726_000004_media_model_meta.d.ts +4 -0
  151. package/dist/db/migrations/20260726_000004_media_model_meta.d.ts.map +1 -0
  152. package/dist/db/migrations/20260726_000004_media_model_meta.js +34 -0
  153. package/dist/db/migrations/20260726_000004_media_model_meta.js.map +1 -0
  154. package/dist/db/migrations/20260726_000005_request_served_model.d.ts +4 -0
  155. package/dist/db/migrations/20260726_000005_request_served_model.d.ts.map +1 -0
  156. package/dist/db/migrations/20260726_000005_request_served_model.js +31 -0
  157. package/dist/db/migrations/20260726_000005_request_served_model.js.map +1 -0
  158. package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts +4 -0
  159. package/dist/db/migrations/20260726_000006_attempt_error_summary.d.ts.map +1 -0
  160. package/dist/db/migrations/20260726_000006_attempt_error_summary.js +29 -0
  161. package/dist/db/migrations/20260726_000006_attempt_error_summary.js.map +1 -0
  162. package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts +4 -0
  163. package/dist/db/migrations/20260727_000001_agent_compatibility.d.ts.map +1 -0
  164. package/dist/db/migrations/20260727_000001_agent_compatibility.js +41 -0
  165. package/dist/db/migrations/20260727_000001_agent_compatibility.js.map +1 -0
  166. package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts +12 -0
  167. package/dist/db/migrations/20260728_000001_tombstone_provenance.d.ts.map +1 -0
  168. package/dist/db/migrations/20260728_000001_tombstone_provenance.js +37 -0
  169. package/dist/db/migrations/20260728_000001_tombstone_provenance.js.map +1 -0
  170. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts +4 -0
  171. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.d.ts.map +1 -0
  172. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js +145 -0
  173. package/dist/db/migrations/20260729_000001_custom_model_endpoint_identity.js.map +1 -0
  174. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts +6 -0
  175. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.d.ts.map +1 -0
  176. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js +52 -0
  177. package/dist/db/migrations/20260802_000001_custom_endpoint_host_labels.js.map +1 -0
  178. package/dist/db/migrations/20260805_000001_key_model_scope.d.ts +6 -0
  179. package/dist/db/migrations/20260805_000001_key_model_scope.d.ts.map +1 -0
  180. package/dist/db/migrations/20260805_000001_key_model_scope.js +17 -0
  181. package/dist/db/migrations/20260805_000001_key_model_scope.js.map +1 -0
  182. package/dist/db/migrations/20260805_000002_client_profiles.d.ts +4 -0
  183. package/dist/db/migrations/20260805_000002_client_profiles.d.ts.map +1 -0
  184. package/dist/db/migrations/20260805_000002_client_profiles.js +33 -0
  185. package/dist/db/migrations/20260805_000002_client_profiles.js.map +1 -0
  186. package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts +4 -0
  187. package/dist/db/migrations/20260810_000001_api_key_proxy.d.ts.map +1 -0
  188. package/dist/db/migrations/20260810_000001_api_key_proxy.js +33 -0
  189. package/dist/db/migrations/20260810_000001_api_key_proxy.js.map +1 -0
  190. package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts +4 -0
  191. package/dist/db/migrations/20260819_000001_custom_model_tombstones.d.ts.map +1 -0
  192. package/dist/db/migrations/20260819_000001_custom_model_tombstones.js +19 -0
  193. package/dist/db/migrations/20260819_000001_custom_model_tombstones.js.map +1 -0
  194. package/dist/db/migrations/20260820_000001_playground_conversations.d.ts +4 -0
  195. package/dist/db/migrations/20260820_000001_playground_conversations.d.ts.map +1 -0
  196. package/dist/db/migrations/20260820_000001_playground_conversations.js +47 -0
  197. package/dist/db/migrations/20260820_000001_playground_conversations.js.map +1 -0
  198. package/dist/db/migrations/20260823_000001_server_logs.d.ts +4 -0
  199. package/dist/db/migrations/20260823_000001_server_logs.d.ts.map +1 -0
  200. package/dist/db/migrations/20260823_000001_server_logs.js +53 -0
  201. package/dist/db/migrations/20260823_000001_server_logs.js.map +1 -0
  202. package/dist/db/migrations/20260823_000002_backups_table.d.ts +8 -0
  203. package/dist/db/migrations/20260823_000002_backups_table.d.ts.map +1 -0
  204. package/dist/db/migrations/20260823_000002_backups_table.js +22 -0
  205. package/dist/db/migrations/20260823_000002_backups_table.js.map +1 -0
  206. package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts +19 -0
  207. package/dist/db/migrations/20260823_000003_attempt_key_label.d.ts.map +1 -0
  208. package/dist/db/migrations/20260823_000003_attempt_key_label.js +28 -0
  209. package/dist/db/migrations/20260823_000003_attempt_key_label.js.map +1 -0
  210. package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts +14 -0
  211. package/dist/db/migrations/20260823_000004_profile_auto_include.d.ts.map +1 -0
  212. package/dist/db/migrations/20260823_000004_profile_auto_include.js +25 -0
  213. package/dist/db/migrations/20260823_000004_profile_auto_include.js.map +1 -0
  214. package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts +4 -0
  215. package/dist/db/migrations/20260901_000001_idempotency_claims.d.ts.map +1 -0
  216. package/dist/db/migrations/20260901_000001_idempotency_claims.js +46 -0
  217. package/dist/db/migrations/20260901_000001_idempotency_claims.js.map +1 -0
  218. package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts +4 -0
  219. package/dist/db/migrations/20260901_000002_quota_observation_lookup.d.ts.map +1 -0
  220. package/dist/db/migrations/20260901_000002_quota_observation_lookup.js +37 -0
  221. package/dist/db/migrations/20260901_000002_quota_observation_lookup.js.map +1 -0
  222. package/dist/db/migrations/20260901_000003_request_caller.d.ts +4 -0
  223. package/dist/db/migrations/20260901_000003_request_caller.d.ts.map +1 -0
  224. package/dist/db/migrations/20260901_000003_request_caller.js +27 -0
  225. package/dist/db/migrations/20260901_000003_request_caller.js.map +1 -0
  226. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts +4 -0
  227. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.d.ts.map +1 -0
  228. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js +35 -0
  229. package/dist/db/migrations/20260902_000001_analytics_latency_percentile_index.js.map +1 -0
  230. package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts +4 -0
  231. package/dist/db/migrations/20260903_000001_mcp_enabled_default.d.ts.map +1 -0
  232. package/dist/db/migrations/20260903_000001_mcp_enabled_default.js +37 -0
  233. package/dist/db/migrations/20260903_000001_mcp_enabled_default.js.map +1 -0
  234. package/dist/db/migrations/20260903_000002_response_cache.d.ts +4 -0
  235. package/dist/db/migrations/20260903_000002_response_cache.d.ts.map +1 -0
  236. package/dist/db/migrations/20260903_000002_response_cache.js +53 -0
  237. package/dist/db/migrations/20260903_000002_response_cache.js.map +1 -0
  238. package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts +13 -0
  239. package/dist/db/migrations/20260904_000001_key_monthly_budget.d.ts.map +1 -0
  240. package/dist/db/migrations/20260904_000001_key_monthly_budget.js +35 -0
  241. package/dist/db/migrations/20260904_000001_key_monthly_budget.js.map +1 -0
  242. package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts +4 -0
  243. package/dist/db/migrations/20260913_000001_request_model_attribution.d.ts.map +1 -0
  244. package/dist/db/migrations/20260913_000001_request_model_attribution.js +68 -0
  245. package/dist/db/migrations/20260913_000001_request_model_attribution.js.map +1 -0
  246. package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts +5 -0
  247. package/dist/db/migrations/20260914_000001_key_monthly_usage.d.ts.map +1 -0
  248. package/dist/db/migrations/20260914_000001_key_monthly_usage.js +37 -0
  249. package/dist/db/migrations/20260914_000001_key_monthly_usage.js.map +1 -0
  250. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts +4 -0
  251. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.d.ts.map +1 -0
  252. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js +28 -0
  253. package/dist/db/migrations/20260915_000001_quota_snapshot_freshness.js.map +1 -0
  254. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts +4 -0
  255. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.d.ts.map +1 -0
  256. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js +25 -0
  257. package/dist/db/migrations/20261004_000001_unified_key_sr_prefix.js.map +1 -0
  258. package/dist/db/migrations/20261005_000001_key_budget_period.d.ts +10 -0
  259. package/dist/db/migrations/20261005_000001_key_budget_period.d.ts.map +1 -0
  260. package/dist/db/migrations/20261005_000001_key_budget_period.js +78 -0
  261. package/dist/db/migrations/20261005_000001_key_budget_period.js.map +1 -0
  262. package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts +21 -0
  263. package/dist/db/migrations/20261005_000002_key_budget_period_auto.d.ts.map +1 -0
  264. package/dist/db/migrations/20261005_000002_key_budget_period_auto.js +41 -0
  265. package/dist/db/migrations/20261005_000002_key_budget_period_auto.js.map +1 -0
  266. package/dist/db/model-pricing.d.ts +39 -0
  267. package/dist/db/model-pricing.d.ts.map +1 -0
  268. package/dist/db/model-pricing.js +274 -0
  269. package/dist/db/model-pricing.js.map +1 -0
  270. package/dist/db/node-sqlite.d.ts +9 -0
  271. package/dist/db/node-sqlite.d.ts.map +1 -0
  272. package/dist/db/node-sqlite.js +87 -0
  273. package/dist/db/node-sqlite.js.map +1 -0
  274. package/dist/db/types.d.ts +22 -0
  275. package/dist/db/types.d.ts.map +1 -0
  276. package/dist/db/types.js +2 -0
  277. package/dist/db/types.js.map +1 -0
  278. package/dist/docs/docs-page.d.ts +2 -0
  279. package/dist/docs/docs-page.d.ts.map +1 -0
  280. package/dist/docs/docs-page.js +319 -0
  281. package/dist/docs/docs-page.js.map +1 -0
  282. package/dist/docs/openapi.d.ts +1896 -0
  283. package/dist/docs/openapi.d.ts.map +1 -0
  284. package/dist/docs/openapi.js +1096 -0
  285. package/dist/docs/openapi.js.map +1 -0
  286. package/dist/env.d.ts +2 -0
  287. package/dist/env.d.ts.map +1 -0
  288. package/dist/env.js +9 -0
  289. package/dist/env.js.map +1 -0
  290. package/dist/index.d.ts +2 -0
  291. package/dist/index.d.ts.map +1 -0
  292. package/dist/index.js +144 -0
  293. package/dist/index.js.map +1 -0
  294. package/dist/lib/anthropic-documents.d.ts +41 -0
  295. package/dist/lib/anthropic-documents.d.ts.map +1 -0
  296. package/dist/lib/anthropic-documents.js +132 -0
  297. package/dist/lib/anthropic-documents.js.map +1 -0
  298. package/dist/lib/app-version.d.ts +5 -0
  299. package/dist/lib/app-version.d.ts.map +1 -0
  300. package/dist/lib/app-version.js +61 -0
  301. package/dist/lib/app-version.js.map +1 -0
  302. package/dist/lib/attempt-trace.d.ts +23 -0
  303. package/dist/lib/attempt-trace.d.ts.map +1 -0
  304. package/dist/lib/attempt-trace.js +32 -0
  305. package/dist/lib/attempt-trace.js.map +1 -0
  306. package/dist/lib/budget.d.ts +6 -0
  307. package/dist/lib/budget.d.ts.map +1 -0
  308. package/dist/lib/budget.js +40 -0
  309. package/dist/lib/budget.js.map +1 -0
  310. package/dist/lib/client-classifier.d.ts +10 -0
  311. package/dist/lib/client-classifier.d.ts.map +1 -0
  312. package/dist/lib/client-classifier.js +126 -0
  313. package/dist/lib/client-classifier.js.map +1 -0
  314. package/dist/lib/client-context.d.ts +10 -0
  315. package/dist/lib/client-context.d.ts.map +1 -0
  316. package/dist/lib/client-context.js +39 -0
  317. package/dist/lib/client-context.js.map +1 -0
  318. package/dist/lib/config.d.ts +33 -0
  319. package/dist/lib/config.d.ts.map +1 -0
  320. package/dist/lib/config.js +94 -0
  321. package/dist/lib/config.js.map +1 -0
  322. package/dist/lib/content.d.ts +33 -0
  323. package/dist/lib/content.d.ts.map +1 -0
  324. package/dist/lib/content.js +201 -0
  325. package/dist/lib/content.js.map +1 -0
  326. package/dist/lib/credential.d.ts +11 -0
  327. package/dist/lib/credential.d.ts.map +1 -0
  328. package/dist/lib/credential.js +18 -0
  329. package/dist/lib/credential.js.map +1 -0
  330. package/dist/lib/crypto.d.ts +31 -0
  331. package/dist/lib/crypto.d.ts.map +1 -0
  332. package/dist/lib/crypto.js +206 -0
  333. package/dist/lib/crypto.js.map +1 -0
  334. package/dist/lib/custom-provider-cleanup.d.ts +13 -0
  335. package/dist/lib/custom-provider-cleanup.d.ts.map +1 -0
  336. package/dist/lib/custom-provider-cleanup.js +43 -0
  337. package/dist/lib/custom-provider-cleanup.js.map +1 -0
  338. package/dist/lib/db-backup.d.ts +31 -0
  339. package/dist/lib/db-backup.d.ts.map +1 -0
  340. package/dist/lib/db-backup.js +255 -0
  341. package/dist/lib/db-backup.js.map +1 -0
  342. package/dist/lib/endpoint-scope.d.ts +42 -0
  343. package/dist/lib/endpoint-scope.d.ts.map +1 -0
  344. package/dist/lib/endpoint-scope.js +95 -0
  345. package/dist/lib/endpoint-scope.js.map +1 -0
  346. package/dist/lib/env-drift.d.ts +17 -0
  347. package/dist/lib/env-drift.d.ts.map +1 -0
  348. package/dist/lib/env-drift.js +108 -0
  349. package/dist/lib/env-drift.js.map +1 -0
  350. package/dist/lib/error-classify.d.ts +40 -0
  351. package/dist/lib/error-classify.d.ts.map +1 -0
  352. package/dist/lib/error-classify.js +687 -0
  353. package/dist/lib/error-classify.js.map +1 -0
  354. package/dist/lib/error-redaction.d.ts +8 -0
  355. package/dist/lib/error-redaction.d.ts.map +1 -0
  356. package/dist/lib/error-redaction.js +45 -0
  357. package/dist/lib/error-redaction.js.map +1 -0
  358. package/dist/lib/fallback-loop.d.ts +306 -0
  359. package/dist/lib/fallback-loop.d.ts.map +1 -0
  360. package/dist/lib/fallback-loop.js +1336 -0
  361. package/dist/lib/fallback-loop.js.map +1 -0
  362. package/dist/lib/file-permissions.d.ts +98 -0
  363. package/dist/lib/file-permissions.d.ts.map +1 -0
  364. package/dist/lib/file-permissions.js +159 -0
  365. package/dist/lib/file-permissions.js.map +1 -0
  366. package/dist/lib/gemini-wire.d.ts +76 -0
  367. package/dist/lib/gemini-wire.d.ts.map +1 -0
  368. package/dist/lib/gemini-wire.js +392 -0
  369. package/dist/lib/gemini-wire.js.map +1 -0
  370. package/dist/lib/guardrails.d.ts +37 -0
  371. package/dist/lib/guardrails.d.ts.map +1 -0
  372. package/dist/lib/guardrails.js +106 -0
  373. package/dist/lib/guardrails.js.map +1 -0
  374. package/dist/lib/header-value.d.ts +8 -0
  375. package/dist/lib/header-value.d.ts.map +1 -0
  376. package/dist/lib/header-value.js +55 -0
  377. package/dist/lib/header-value.js.map +1 -0
  378. package/dist/lib/image-normalize.d.ts +21 -0
  379. package/dist/lib/image-normalize.d.ts.map +1 -0
  380. package/dist/lib/image-normalize.js +218 -0
  381. package/dist/lib/image-normalize.js.map +1 -0
  382. package/dist/lib/inbound-chat.d.ts +48 -0
  383. package/dist/lib/inbound-chat.d.ts.map +1 -0
  384. package/dist/lib/inbound-chat.js +406 -0
  385. package/dist/lib/inbound-chat.js.map +1 -0
  386. package/dist/lib/key-parser.d.ts +79 -0
  387. package/dist/lib/key-parser.d.ts.map +1 -0
  388. package/dist/lib/key-parser.js +701 -0
  389. package/dist/lib/key-parser.js.map +1 -0
  390. package/dist/lib/key-proxy.d.ts +49 -0
  391. package/dist/lib/key-proxy.d.ts.map +1 -0
  392. package/dist/lib/key-proxy.js +92 -0
  393. package/dist/lib/key-proxy.js.map +1 -0
  394. package/dist/lib/log-redaction.d.ts +28 -0
  395. package/dist/lib/log-redaction.d.ts.map +1 -0
  396. package/dist/lib/log-redaction.js +166 -0
  397. package/dist/lib/log-redaction.js.map +1 -0
  398. package/dist/lib/model-scope.d.ts +9 -0
  399. package/dist/lib/model-scope.d.ts.map +1 -0
  400. package/dist/lib/model-scope.js +29 -0
  401. package/dist/lib/model-scope.js.map +1 -0
  402. package/dist/lib/one-time-code.d.ts +15 -0
  403. package/dist/lib/one-time-code.d.ts.map +1 -0
  404. package/dist/lib/one-time-code.js +41 -0
  405. package/dist/lib/one-time-code.js.map +1 -0
  406. package/dist/lib/output-cap.d.ts +24 -0
  407. package/dist/lib/output-cap.d.ts.map +1 -0
  408. package/dist/lib/output-cap.js +80 -0
  409. package/dist/lib/output-cap.js.map +1 -0
  410. package/dist/lib/password.d.ts +3 -0
  411. package/dist/lib/password.d.ts.map +1 -0
  412. package/dist/lib/password.js +27 -0
  413. package/dist/lib/password.js.map +1 -0
  414. package/dist/lib/process-safety-net.d.ts +32 -0
  415. package/dist/lib/process-safety-net.d.ts.map +1 -0
  416. package/dist/lib/process-safety-net.js +123 -0
  417. package/dist/lib/process-safety-net.js.map +1 -0
  418. package/dist/lib/provider-identity.d.ts +37 -0
  419. package/dist/lib/provider-identity.d.ts.map +1 -0
  420. package/dist/lib/provider-identity.js +110 -0
  421. package/dist/lib/provider-identity.js.map +1 -0
  422. package/dist/lib/provider-size-parser.d.ts +6 -0
  423. package/dist/lib/provider-size-parser.d.ts.map +1 -0
  424. package/dist/lib/provider-size-parser.js +72 -0
  425. package/dist/lib/provider-size-parser.js.map +1 -0
  426. package/dist/lib/provider-timeout.d.ts +17 -0
  427. package/dist/lib/provider-timeout.d.ts.map +1 -0
  428. package/dist/lib/provider-timeout.js +70 -0
  429. package/dist/lib/provider-timeout.js.map +1 -0
  430. package/dist/lib/proxy.d.ts +180 -0
  431. package/dist/lib/proxy.d.ts.map +1 -0
  432. package/dist/lib/proxy.js +1002 -0
  433. package/dist/lib/proxy.js.map +1 -0
  434. package/dist/lib/request-log.d.ts +4 -0
  435. package/dist/lib/request-log.d.ts.map +1 -0
  436. package/dist/lib/request-log.js +161 -0
  437. package/dist/lib/request-log.js.map +1 -0
  438. package/dist/lib/reset-code.d.ts +5 -0
  439. package/dist/lib/reset-code.d.ts.map +1 -0
  440. package/dist/lib/reset-code.js +39 -0
  441. package/dist/lib/reset-code.js.map +1 -0
  442. package/dist/lib/retry-hint.d.ts +14 -0
  443. package/dist/lib/retry-hint.d.ts.map +1 -0
  444. package/dist/lib/retry-hint.js +36 -0
  445. package/dist/lib/retry-hint.js.map +1 -0
  446. package/dist/lib/route-live.d.ts +29 -0
  447. package/dist/lib/route-live.d.ts.map +1 -0
  448. package/dist/lib/route-live.js +97 -0
  449. package/dist/lib/route-live.js.map +1 -0
  450. package/dist/lib/sampling-params.d.ts +183 -0
  451. package/dist/lib/sampling-params.d.ts.map +1 -0
  452. package/dist/lib/sampling-params.js +386 -0
  453. package/dist/lib/sampling-params.js.map +1 -0
  454. package/dist/lib/scheduler.d.ts +13 -0
  455. package/dist/lib/scheduler.d.ts.map +1 -0
  456. package/dist/lib/scheduler.js +11 -0
  457. package/dist/lib/scheduler.js.map +1 -0
  458. package/dist/lib/served-model.d.ts +16 -0
  459. package/dist/lib/served-model.d.ts.map +1 -0
  460. package/dist/lib/served-model.js +0 -0
  461. package/dist/lib/served-model.js.map +1 -0
  462. package/dist/lib/server-logs.d.ts +132 -0
  463. package/dist/lib/server-logs.d.ts.map +1 -0
  464. package/dist/lib/server-logs.js +473 -0
  465. package/dist/lib/server-logs.js.map +1 -0
  466. package/dist/lib/setup-code.d.ts +5 -0
  467. package/dist/lib/setup-code.d.ts.map +1 -0
  468. package/dist/lib/setup-code.js +35 -0
  469. package/dist/lib/setup-code.js.map +1 -0
  470. package/dist/lib/structured-output.d.ts +9 -0
  471. package/dist/lib/structured-output.d.ts.map +1 -0
  472. package/dist/lib/structured-output.js +72 -0
  473. package/dist/lib/structured-output.js.map +1 -0
  474. package/dist/lib/system-prompt.d.ts +30 -0
  475. package/dist/lib/system-prompt.d.ts.map +1 -0
  476. package/dist/lib/system-prompt.js +66 -0
  477. package/dist/lib/system-prompt.js.map +1 -0
  478. package/dist/lib/task-type.d.ts +18 -0
  479. package/dist/lib/task-type.d.ts.map +1 -0
  480. package/dist/lib/task-type.js +92 -0
  481. package/dist/lib/task-type.js.map +1 -0
  482. package/dist/lib/think-tags.d.ts +37 -0
  483. package/dist/lib/think-tags.d.ts.map +1 -0
  484. package/dist/lib/think-tags.js +203 -0
  485. package/dist/lib/think-tags.js.map +1 -0
  486. package/dist/lib/tool-args.d.ts +36 -0
  487. package/dist/lib/tool-args.d.ts.map +1 -0
  488. package/dist/lib/tool-args.js +190 -0
  489. package/dist/lib/tool-args.js.map +1 -0
  490. package/dist/lib/tool-call-rescue.d.ts +62 -0
  491. package/dist/lib/tool-call-rescue.d.ts.map +1 -0
  492. package/dist/lib/tool-call-rescue.js +258 -0
  493. package/dist/lib/tool-call-rescue.js.map +1 -0
  494. package/dist/lib/tool-capability.d.ts +13 -0
  495. package/dist/lib/tool-capability.d.ts.map +1 -0
  496. package/dist/lib/tool-capability.js +70 -0
  497. package/dist/lib/tool-capability.js.map +1 -0
  498. package/dist/lib/tool-validate.d.ts +44 -0
  499. package/dist/lib/tool-validate.d.ts.map +1 -0
  500. package/dist/lib/tool-validate.js +165 -0
  501. package/dist/lib/tool-validate.js.map +1 -0
  502. package/dist/lib/ttfb-budget.d.ts +13 -0
  503. package/dist/lib/ttfb-budget.d.ts.map +1 -0
  504. package/dist/lib/ttfb-budget.js +153 -0
  505. package/dist/lib/ttfb-budget.js.map +1 -0
  506. package/dist/lib/url-guard.d.ts +47 -0
  507. package/dist/lib/url-guard.d.ts.map +1 -0
  508. package/dist/lib/url-guard.js +229 -0
  509. package/dist/lib/url-guard.js.map +1 -0
  510. package/dist/lib/wake-detect.d.ts +12 -0
  511. package/dist/lib/wake-detect.d.ts.map +1 -0
  512. package/dist/lib/wake-detect.js +99 -0
  513. package/dist/lib/wake-detect.js.map +1 -0
  514. package/dist/middleware/errorHandler.d.ts +3 -0
  515. package/dist/middleware/errorHandler.d.ts.map +1 -0
  516. package/dist/middleware/errorHandler.js +45 -0
  517. package/dist/middleware/errorHandler.js.map +1 -0
  518. package/dist/middleware/rateLimit.d.ts +4 -0
  519. package/dist/middleware/rateLimit.d.ts.map +1 -0
  520. package/dist/middleware/rateLimit.js +125 -0
  521. package/dist/middleware/rateLimit.js.map +1 -0
  522. package/dist/middleware/requireAuth.d.ts +3 -0
  523. package/dist/middleware/requireAuth.d.ts.map +1 -0
  524. package/dist/middleware/requireAuth.js +17 -0
  525. package/dist/middleware/requireAuth.js.map +1 -0
  526. package/dist/providers/aclide.d.ts +18 -0
  527. package/dist/providers/aclide.d.ts.map +1 -0
  528. package/dist/providers/aclide.js +174 -0
  529. package/dist/providers/aclide.js.map +1 -0
  530. package/dist/providers/aihorde.d.ts +42 -0
  531. package/dist/providers/aihorde.d.ts.map +1 -0
  532. package/dist/providers/aihorde.js +192 -0
  533. package/dist/providers/aihorde.js.map +1 -0
  534. package/dist/providers/airforce.d.ts +11 -0
  535. package/dist/providers/airforce.d.ts.map +1 -0
  536. package/dist/providers/airforce.js +32 -0
  537. package/dist/providers/airforce.js.map +1 -0
  538. package/dist/providers/base.d.ts +177 -0
  539. package/dist/providers/base.d.ts.map +1 -0
  540. package/dist/providers/base.js +354 -0
  541. package/dist/providers/base.js.map +1 -0
  542. package/dist/providers/blaze.d.ts +11 -0
  543. package/dist/providers/blaze.d.ts.map +1 -0
  544. package/dist/providers/blaze.js +33 -0
  545. package/dist/providers/blaze.js.map +1 -0
  546. package/dist/providers/blockrun.d.ts +14 -0
  547. package/dist/providers/blockrun.d.ts.map +1 -0
  548. package/dist/providers/blockrun.js +36 -0
  549. package/dist/providers/blockrun.js.map +1 -0
  550. package/dist/providers/clod.d.ts +11 -0
  551. package/dist/providers/clod.d.ts.map +1 -0
  552. package/dist/providers/clod.js +39 -0
  553. package/dist/providers/clod.js.map +1 -0
  554. package/dist/providers/cloudflare.d.ts +15 -0
  555. package/dist/providers/cloudflare.d.ts.map +1 -0
  556. package/dist/providers/cloudflare.js +175 -0
  557. package/dist/providers/cloudflare.js.map +1 -0
  558. package/dist/providers/cohere.d.ts +11 -0
  559. package/dist/providers/cohere.d.ts.map +1 -0
  560. package/dist/providers/cohere.js +117 -0
  561. package/dist/providers/cohere.js.map +1 -0
  562. package/dist/providers/dreamprompting.d.ts +11 -0
  563. package/dist/providers/dreamprompting.d.ts.map +1 -0
  564. package/dist/providers/dreamprompting.js +41 -0
  565. package/dist/providers/dreamprompting.js.map +1 -0
  566. package/dist/providers/electronhub.d.ts +10 -0
  567. package/dist/providers/electronhub.d.ts.map +1 -0
  568. package/dist/providers/electronhub.js +56 -0
  569. package/dist/providers/electronhub.js.map +1 -0
  570. package/dist/providers/experiential.d.ts +10 -0
  571. package/dist/providers/experiential.d.ts.map +1 -0
  572. package/dist/providers/experiential.js +33 -0
  573. package/dist/providers/experiential.js.map +1 -0
  574. package/dist/providers/gizmo.d.ts +15 -0
  575. package/dist/providers/gizmo.d.ts.map +1 -0
  576. package/dist/providers/gizmo.js +54 -0
  577. package/dist/providers/gizmo.js.map +1 -0
  578. package/dist/providers/google.d.ts +30 -0
  579. package/dist/providers/google.d.ts.map +1 -0
  580. package/dist/providers/google.js +766 -0
  581. package/dist/providers/google.js.map +1 -0
  582. package/dist/providers/index.d.ts +13 -0
  583. package/dist/providers/index.d.ts.map +1 -0
  584. package/dist/providers/index.js +561 -0
  585. package/dist/providers/index.js.map +1 -0
  586. package/dist/providers/llmtr.d.ts +15 -0
  587. package/dist/providers/llmtr.d.ts.map +1 -0
  588. package/dist/providers/llmtr.js +51 -0
  589. package/dist/providers/llmtr.js.map +1 -0
  590. package/dist/providers/logfare.d.ts +11 -0
  591. package/dist/providers/logfare.d.ts.map +1 -0
  592. package/dist/providers/logfare.js +33 -0
  593. package/dist/providers/logfare.js.map +1 -0
  594. package/dist/providers/lucidity.d.ts +11 -0
  595. package/dist/providers/lucidity.d.ts.map +1 -0
  596. package/dist/providers/lucidity.js +37 -0
  597. package/dist/providers/lucidity.js.map +1 -0
  598. package/dist/providers/modelscope.d.ts +50 -0
  599. package/dist/providers/modelscope.d.ts.map +1 -0
  600. package/dist/providers/modelscope.js +143 -0
  601. package/dist/providers/modelscope.js.map +1 -0
  602. package/dist/providers/moondream.d.ts +17 -0
  603. package/dist/providers/moondream.d.ts.map +1 -0
  604. package/dist/providers/moondream.js +140 -0
  605. package/dist/providers/moondream.js.map +1 -0
  606. package/dist/providers/openai-compat.d.ts +140 -0
  607. package/dist/providers/openai-compat.d.ts.map +1 -0
  608. package/dist/providers/openai-compat.js +537 -0
  609. package/dist/providers/openai-compat.js.map +1 -0
  610. package/dist/providers/opencode-free.d.ts +24 -0
  611. package/dist/providers/opencode-free.d.ts.map +1 -0
  612. package/dist/providers/opencode-free.js +262 -0
  613. package/dist/providers/opencode-free.js.map +1 -0
  614. package/dist/providers/pollinations.d.ts +32 -0
  615. package/dist/providers/pollinations.d.ts.map +1 -0
  616. package/dist/providers/pollinations.js +65 -0
  617. package/dist/providers/pollinations.js.map +1 -0
  618. package/dist/providers/router9.d.ts +16 -0
  619. package/dist/providers/router9.d.ts.map +1 -0
  620. package/dist/providers/router9.js +42 -0
  621. package/dist/providers/router9.js.map +1 -0
  622. package/dist/providers/sail.d.ts +44 -0
  623. package/dist/providers/sail.d.ts.map +1 -0
  624. package/dist/providers/sail.js +302 -0
  625. package/dist/providers/sail.js.map +1 -0
  626. package/dist/providers/septor.d.ts +10 -0
  627. package/dist/providers/septor.d.ts.map +1 -0
  628. package/dist/providers/septor.js +36 -0
  629. package/dist/providers/septor.js.map +1 -0
  630. package/dist/providers/speechify.d.ts +14 -0
  631. package/dist/providers/speechify.d.ts.map +1 -0
  632. package/dist/providers/speechify.js +28 -0
  633. package/dist/providers/speechify.js.map +1 -0
  634. package/dist/providers/speka.d.ts +12 -0
  635. package/dist/providers/speka.d.ts.map +1 -0
  636. package/dist/providers/speka.js +37 -0
  637. package/dist/providers/speka.js.map +1 -0
  638. package/dist/providers/waterfall.d.ts +11 -0
  639. package/dist/providers/waterfall.d.ts.map +1 -0
  640. package/dist/providers/waterfall.js +33 -0
  641. package/dist/providers/waterfall.js.map +1 -0
  642. package/dist/providers/xkiro.d.ts +71 -0
  643. package/dist/providers/xkiro.d.ts.map +1 -0
  644. package/dist/providers/xkiro.js +119 -0
  645. package/dist/providers/xkiro.js.map +1 -0
  646. package/dist/providers/zhipu.d.ts +43 -0
  647. package/dist/providers/zhipu.d.ts.map +1 -0
  648. package/dist/providers/zhipu.js +95 -0
  649. package/dist/providers/zhipu.js.map +1 -0
  650. package/dist/routes/analytics.d.ts +2 -0
  651. package/dist/routes/analytics.d.ts.map +1 -0
  652. package/dist/routes/analytics.js +738 -0
  653. package/dist/routes/analytics.js.map +1 -0
  654. package/dist/routes/anthropic.d.ts +7 -0
  655. package/dist/routes/anthropic.d.ts.map +1 -0
  656. package/dist/routes/anthropic.js +1076 -0
  657. package/dist/routes/anthropic.js.map +1 -0
  658. package/dist/routes/auth.d.ts +2 -0
  659. package/dist/routes/auth.d.ts.map +1 -0
  660. package/dist/routes/auth.js +310 -0
  661. package/dist/routes/auth.js.map +1 -0
  662. package/dist/routes/backups.d.ts +2 -0
  663. package/dist/routes/backups.d.ts.map +1 -0
  664. package/dist/routes/backups.js +127 -0
  665. package/dist/routes/backups.js.map +1 -0
  666. package/dist/routes/cache.d.ts +2 -0
  667. package/dist/routes/cache.d.ts.map +1 -0
  668. package/dist/routes/cache.js +40 -0
  669. package/dist/routes/cache.js.map +1 -0
  670. package/dist/routes/client-profiles.d.ts +2 -0
  671. package/dist/routes/client-profiles.d.ts.map +1 -0
  672. package/dist/routes/client-profiles.js +128 -0
  673. package/dist/routes/client-profiles.js.map +1 -0
  674. package/dist/routes/compression.d.ts +2 -0
  675. package/dist/routes/compression.d.ts.map +1 -0
  676. package/dist/routes/compression.js +82 -0
  677. package/dist/routes/compression.js.map +1 -0
  678. package/dist/routes/conversations.d.ts +3 -0
  679. package/dist/routes/conversations.d.ts.map +1 -0
  680. package/dist/routes/conversations.js +215 -0
  681. package/dist/routes/conversations.js.map +1 -0
  682. package/dist/routes/docs.d.ts +2 -0
  683. package/dist/routes/docs.d.ts.map +1 -0
  684. package/dist/routes/docs.js +21 -0
  685. package/dist/routes/docs.js.map +1 -0
  686. package/dist/routes/embeddings.d.ts +2 -0
  687. package/dist/routes/embeddings.d.ts.map +1 -0
  688. package/dist/routes/embeddings.js +255 -0
  689. package/dist/routes/embeddings.js.map +1 -0
  690. package/dist/routes/fallback.d.ts +2 -0
  691. package/dist/routes/fallback.d.ts.map +1 -0
  692. package/dist/routes/fallback.js +631 -0
  693. package/dist/routes/fallback.js.map +1 -0
  694. package/dist/routes/free-tier.d.ts +22 -0
  695. package/dist/routes/free-tier.d.ts.map +1 -0
  696. package/dist/routes/free-tier.js +175 -0
  697. package/dist/routes/free-tier.js.map +1 -0
  698. package/dist/routes/gemini.d.ts +2 -0
  699. package/dist/routes/gemini.d.ts.map +1 -0
  700. package/dist/routes/gemini.js +218 -0
  701. package/dist/routes/gemini.js.map +1 -0
  702. package/dist/routes/health.d.ts +2 -0
  703. package/dist/routes/health.d.ts.map +1 -0
  704. package/dist/routes/health.js +72 -0
  705. package/dist/routes/health.js.map +1 -0
  706. package/dist/routes/keys.d.ts +17 -0
  707. package/dist/routes/keys.d.ts.map +1 -0
  708. package/dist/routes/keys.js +1787 -0
  709. package/dist/routes/keys.js.map +1 -0
  710. package/dist/routes/logs.d.ts +2 -0
  711. package/dist/routes/logs.d.ts.map +1 -0
  712. package/dist/routes/logs.js +88 -0
  713. package/dist/routes/logs.js.map +1 -0
  714. package/dist/routes/mcp.d.ts +4 -0
  715. package/dist/routes/mcp.d.ts.map +1 -0
  716. package/dist/routes/mcp.js +563 -0
  717. package/dist/routes/mcp.js.map +1 -0
  718. package/dist/routes/media.d.ts +2 -0
  719. package/dist/routes/media.d.ts.map +1 -0
  720. package/dist/routes/media.js +180 -0
  721. package/dist/routes/media.js.map +1 -0
  722. package/dist/routes/models.d.ts +2 -0
  723. package/dist/routes/models.d.ts.map +1 -0
  724. package/dist/routes/models.js +439 -0
  725. package/dist/routes/models.js.map +1 -0
  726. package/dist/routes/ollama.d.ts +7 -0
  727. package/dist/routes/ollama.d.ts.map +1 -0
  728. package/dist/routes/ollama.js +595 -0
  729. package/dist/routes/ollama.js.map +1 -0
  730. package/dist/routes/profiles.d.ts +3 -0
  731. package/dist/routes/profiles.d.ts.map +1 -0
  732. package/dist/routes/profiles.js +431 -0
  733. package/dist/routes/profiles.js.map +1 -0
  734. package/dist/routes/proxy.d.ts +32 -0
  735. package/dist/routes/proxy.d.ts.map +1 -0
  736. package/dist/routes/proxy.js +2793 -0
  737. package/dist/routes/proxy.js.map +1 -0
  738. package/dist/routes/responses.d.ts +1161 -0
  739. package/dist/routes/responses.d.ts.map +1 -0
  740. package/dist/routes/responses.js +1521 -0
  741. package/dist/routes/responses.js.map +1 -0
  742. package/dist/routes/settings.d.ts +2 -0
  743. package/dist/routes/settings.d.ts.map +1 -0
  744. package/dist/routes/settings.js +549 -0
  745. package/dist/routes/settings.js.map +1 -0
  746. package/dist/routes/status.d.ts +3 -0
  747. package/dist/routes/status.d.ts.map +1 -0
  748. package/dist/routes/status.js +214 -0
  749. package/dist/routes/status.js.map +1 -0
  750. package/dist/routes/update.d.ts +31 -0
  751. package/dist/routes/update.d.ts.map +1 -0
  752. package/dist/routes/update.js +536 -0
  753. package/dist/routes/update.js.map +1 -0
  754. package/dist/routes/url-tokens.d.ts +2 -0
  755. package/dist/routes/url-tokens.d.ts.map +1 -0
  756. package/dist/routes/url-tokens.js +31 -0
  757. package/dist/routes/url-tokens.js.map +1 -0
  758. package/dist/scripts/export-catalog.d.ts +2 -0
  759. package/dist/scripts/export-catalog.d.ts.map +1 -0
  760. package/dist/scripts/export-catalog.js +114 -0
  761. package/dist/scripts/export-catalog.js.map +1 -0
  762. package/dist/scripts/rotate-encryption-key.d.ts +81 -0
  763. package/dist/scripts/rotate-encryption-key.d.ts.map +1 -0
  764. package/dist/scripts/rotate-encryption-key.js +232 -0
  765. package/dist/scripts/rotate-encryption-key.js.map +1 -0
  766. package/dist/scripts/routing-sim.d.ts +2 -0
  767. package/dist/scripts/routing-sim.d.ts.map +1 -0
  768. package/dist/scripts/routing-sim.js +130 -0
  769. package/dist/scripts/routing-sim.js.map +1 -0
  770. package/dist/scripts/test-all-models.d.ts +2 -0
  771. package/dist/scripts/test-all-models.d.ts.map +1 -0
  772. package/dist/scripts/test-all-models.js +56 -0
  773. package/dist/scripts/test-all-models.js.map +1 -0
  774. package/dist/services/anthropic-map.d.ts +39 -0
  775. package/dist/services/anthropic-map.d.ts.map +1 -0
  776. package/dist/services/anthropic-map.js +138 -0
  777. package/dist/services/anthropic-map.js.map +1 -0
  778. package/dist/services/auth.d.ts +30 -0
  779. package/dist/services/auth.d.ts.map +1 -0
  780. package/dist/services/auth.js +124 -0
  781. package/dist/services/auth.js.map +1 -0
  782. package/dist/services/auto-discover.d.ts +77 -0
  783. package/dist/services/auto-discover.d.ts.map +1 -0
  784. package/dist/services/auto-discover.js +1047 -0
  785. package/dist/services/auto-discover.js.map +1 -0
  786. package/dist/services/backups.d.ts +67 -0
  787. package/dist/services/backups.d.ts.map +1 -0
  788. package/dist/services/backups.js +444 -0
  789. package/dist/services/backups.js.map +1 -0
  790. package/dist/services/builtin-model-discovery.d.ts +107 -0
  791. package/dist/services/builtin-model-discovery.d.ts.map +1 -0
  792. package/dist/services/builtin-model-discovery.js +296 -0
  793. package/dist/services/builtin-model-discovery.js.map +1 -0
  794. package/dist/services/cache.d.ts +190 -0
  795. package/dist/services/cache.d.ts.map +1 -0
  796. package/dist/services/cache.js +617 -0
  797. package/dist/services/cache.js.map +1 -0
  798. package/dist/services/catalog-sync.d.ts +257 -0
  799. package/dist/services/catalog-sync.d.ts.map +1 -0
  800. package/dist/services/catalog-sync.js +1041 -0
  801. package/dist/services/catalog-sync.js.map +1 -0
  802. package/dist/services/compression/config.d.ts +33 -0
  803. package/dist/services/compression/config.d.ts.map +1 -0
  804. package/dist/services/compression/config.js +158 -0
  805. package/dist/services/compression/config.js.map +1 -0
  806. package/dist/services/compression/engines/aging.d.ts +2 -0
  807. package/dist/services/compression/engines/aging.d.ts.map +1 -0
  808. package/dist/services/compression/engines/aging.js +53 -0
  809. package/dist/services/compression/engines/aging.js.map +1 -0
  810. package/dist/services/compression/engines/custom-filters.d.ts +4 -0
  811. package/dist/services/compression/engines/custom-filters.d.ts.map +1 -0
  812. package/dist/services/compression/engines/custom-filters.js +49 -0
  813. package/dist/services/compression/engines/custom-filters.js.map +1 -0
  814. package/dist/services/compression/engines/dedup.d.ts +2 -0
  815. package/dist/services/compression/engines/dedup.d.ts.map +1 -0
  816. package/dist/services/compression/engines/dedup.js +61 -0
  817. package/dist/services/compression/engines/dedup.js.map +1 -0
  818. package/dist/services/compression/engines/filter-definitions.d.ts +75 -0
  819. package/dist/services/compression/engines/filter-definitions.d.ts.map +1 -0
  820. package/dist/services/compression/engines/filter-definitions.js +45 -0
  821. package/dist/services/compression/engines/filter-definitions.js.map +1 -0
  822. package/dist/services/compression/engines/hard-budget.d.ts +2 -0
  823. package/dist/services/compression/engines/hard-budget.d.ts.map +1 -0
  824. package/dist/services/compression/engines/hard-budget.js +52 -0
  825. package/dist/services/compression/engines/hard-budget.js.map +1 -0
  826. package/dist/services/compression/engines/index.d.ts +9 -0
  827. package/dist/services/compression/engines/index.d.ts.map +1 -0
  828. package/dist/services/compression/engines/index.js +9 -0
  829. package/dist/services/compression/engines/index.js.map +1 -0
  830. package/dist/services/compression/engines/jsoncompact.d.ts +3 -0
  831. package/dist/services/compression/engines/jsoncompact.d.ts.map +1 -0
  832. package/dist/services/compression/engines/jsoncompact.js +145 -0
  833. package/dist/services/compression/engines/jsoncompact.js.map +1 -0
  834. package/dist/services/compression/engines/lite.d.ts +2 -0
  835. package/dist/services/compression/engines/lite.d.ts.map +1 -0
  836. package/dist/services/compression/engines/lite.js +34 -0
  837. package/dist/services/compression/engines/lite.js.map +1 -0
  838. package/dist/services/compression/engines/read-lifecycle.d.ts +2 -0
  839. package/dist/services/compression/engines/read-lifecycle.d.ts.map +1 -0
  840. package/dist/services/compression/engines/read-lifecycle.js +70 -0
  841. package/dist/services/compression/engines/read-lifecycle.js.map +1 -0
  842. package/dist/services/compression/engines/relevance.d.ts +2 -0
  843. package/dist/services/compression/engines/relevance.d.ts.map +1 -0
  844. package/dist/services/compression/engines/relevance.js +74 -0
  845. package/dist/services/compression/engines/relevance.js.map +1 -0
  846. package/dist/services/compression/engines/toolfilter.d.ts +2 -0
  847. package/dist/services/compression/engines/toolfilter.d.ts.map +1 -0
  848. package/dist/services/compression/engines/toolfilter.js +195 -0
  849. package/dist/services/compression/engines/toolfilter.js.map +1 -0
  850. package/dist/services/compression/fidelity-gate.d.ts +12 -0
  851. package/dist/services/compression/fidelity-gate.d.ts.map +1 -0
  852. package/dist/services/compression/fidelity-gate.js +90 -0
  853. package/dist/services/compression/fidelity-gate.js.map +1 -0
  854. package/dist/services/compression/helpers.d.ts +9 -0
  855. package/dist/services/compression/helpers.d.ts.map +1 -0
  856. package/dist/services/compression/helpers.js +52 -0
  857. package/dist/services/compression/helpers.js.map +1 -0
  858. package/dist/services/compression/pipeline.d.ts +7 -0
  859. package/dist/services/compression/pipeline.d.ts.map +1 -0
  860. package/dist/services/compression/pipeline.js +245 -0
  861. package/dist/services/compression/pipeline.js.map +1 -0
  862. package/dist/services/compression/preservation.d.ts +27 -0
  863. package/dist/services/compression/preservation.d.ts.map +1 -0
  864. package/dist/services/compression/preservation.js +109 -0
  865. package/dist/services/compression/preservation.js.map +1 -0
  866. package/dist/services/compression/registry.d.ts +6 -0
  867. package/dist/services/compression/registry.d.ts.map +1 -0
  868. package/dist/services/compression/registry.js +16 -0
  869. package/dist/services/compression/registry.js.map +1 -0
  870. package/dist/services/compression/stats.d.ts +29 -0
  871. package/dist/services/compression/stats.d.ts.map +1 -0
  872. package/dist/services/compression/stats.js +55 -0
  873. package/dist/services/compression/stats.js.map +1 -0
  874. package/dist/services/compression/types.d.ts +90 -0
  875. package/dist/services/compression/types.d.ts.map +1 -0
  876. package/dist/services/compression/types.js +2 -0
  877. package/dist/services/compression/types.js.map +1 -0
  878. package/dist/services/context-handoff.d.ts +22 -0
  879. package/dist/services/context-handoff.d.ts.map +1 -0
  880. package/dist/services/context-handoff.js +164 -0
  881. package/dist/services/context-handoff.js.map +1 -0
  882. package/dist/services/cooldown-probe.d.ts +24 -0
  883. package/dist/services/cooldown-probe.d.ts.map +1 -0
  884. package/dist/services/cooldown-probe.js +181 -0
  885. package/dist/services/cooldown-probe.js.map +1 -0
  886. package/dist/services/custom-endpoint.d.ts +58 -0
  887. package/dist/services/custom-endpoint.d.ts.map +1 -0
  888. package/dist/services/custom-endpoint.js +167 -0
  889. package/dist/services/custom-endpoint.js.map +1 -0
  890. package/dist/services/custom-media-register.d.ts +22 -0
  891. package/dist/services/custom-media-register.d.ts.map +1 -0
  892. package/dist/services/custom-media-register.js +45 -0
  893. package/dist/services/custom-media-register.js.map +1 -0
  894. package/dist/services/custom-model-register.d.ts +50 -0
  895. package/dist/services/custom-model-register.d.ts.map +1 -0
  896. package/dist/services/custom-model-register.js +103 -0
  897. package/dist/services/custom-model-register.js.map +1 -0
  898. package/dist/services/custom-model-seed.d.ts +18 -0
  899. package/dist/services/custom-model-seed.d.ts.map +1 -0
  900. package/dist/services/custom-model-seed.js +42 -0
  901. package/dist/services/custom-model-seed.js.map +1 -0
  902. package/dist/services/custom-model-sync.d.ts +37 -0
  903. package/dist/services/custom-model-sync.d.ts.map +1 -0
  904. package/dist/services/custom-model-sync.js +118 -0
  905. package/dist/services/custom-model-sync.js.map +1 -0
  906. package/dist/services/custom-model-tombstone.d.ts +5 -0
  907. package/dist/services/custom-model-tombstone.d.ts.map +1 -0
  908. package/dist/services/custom-model-tombstone.js +30 -0
  909. package/dist/services/custom-model-tombstone.js.map +1 -0
  910. package/dist/services/declarative-config.d.ts +356 -0
  911. package/dist/services/declarative-config.d.ts.map +1 -0
  912. package/dist/services/declarative-config.js +427 -0
  913. package/dist/services/declarative-config.js.map +1 -0
  914. package/dist/services/degradation.d.ts +40 -0
  915. package/dist/services/degradation.d.ts.map +1 -0
  916. package/dist/services/degradation.js +119 -0
  917. package/dist/services/degradation.js.map +1 -0
  918. package/dist/services/embeddings.d.ts +76 -0
  919. package/dist/services/embeddings.d.ts.map +1 -0
  920. package/dist/services/embeddings.js +319 -0
  921. package/dist/services/embeddings.js.map +1 -0
  922. package/dist/services/fusion.d.ts +162 -0
  923. package/dist/services/fusion.d.ts.map +1 -0
  924. package/dist/services/fusion.js +789 -0
  925. package/dist/services/fusion.js.map +1 -0
  926. package/dist/services/gemini-map.d.ts +14 -0
  927. package/dist/services/gemini-map.d.ts.map +1 -0
  928. package/dist/services/gemini-map.js +78 -0
  929. package/dist/services/gemini-map.js.map +1 -0
  930. package/dist/services/health.d.ts +52 -0
  931. package/dist/services/health.d.ts.map +1 -0
  932. package/dist/services/health.js +352 -0
  933. package/dist/services/health.js.map +1 -0
  934. package/dist/services/idempotency.d.ts +49 -0
  935. package/dist/services/idempotency.d.ts.map +1 -0
  936. package/dist/services/idempotency.js +140 -0
  937. package/dist/services/idempotency.js.map +1 -0
  938. package/dist/services/key-budget.d.ts +74 -0
  939. package/dist/services/key-budget.d.ts.map +1 -0
  940. package/dist/services/key-budget.js +200 -0
  941. package/dist/services/key-budget.js.map +1 -0
  942. package/dist/services/media.d.ts +123 -0
  943. package/dist/services/media.d.ts.map +1 -0
  944. package/dist/services/media.js +954 -0
  945. package/dist/services/media.js.map +1 -0
  946. package/dist/services/model-discovery.d.ts +126 -0
  947. package/dist/services/model-discovery.d.ts.map +1 -0
  948. package/dist/services/model-discovery.js +662 -0
  949. package/dist/services/model-discovery.js.map +1 -0
  950. package/dist/services/model-groups.d.ts +149 -0
  951. package/dist/services/model-groups.d.ts.map +1 -0
  952. package/dist/services/model-groups.js +314 -0
  953. package/dist/services/model-groups.js.map +1 -0
  954. package/dist/services/model-listing.d.ts +18 -0
  955. package/dist/services/model-listing.d.ts.map +1 -0
  956. package/dist/services/model-listing.js +94 -0
  957. package/dist/services/model-listing.js.map +1 -0
  958. package/dist/services/model-retirement.d.ts +29 -0
  959. package/dist/services/model-retirement.d.ts.map +1 -0
  960. package/dist/services/model-retirement.js +86 -0
  961. package/dist/services/model-retirement.js.map +1 -0
  962. package/dist/services/model-state.d.ts +90 -0
  963. package/dist/services/model-state.d.ts.map +1 -0
  964. package/dist/services/model-state.js +342 -0
  965. package/dist/services/model-state.js.map +1 -0
  966. package/dist/services/model-weight-overrides.d.ts +48 -0
  967. package/dist/services/model-weight-overrides.d.ts.map +1 -0
  968. package/dist/services/model-weight-overrides.js +127 -0
  969. package/dist/services/model-weight-overrides.js.map +1 -0
  970. package/dist/services/notification-emitter.d.ts +50 -0
  971. package/dist/services/notification-emitter.d.ts.map +1 -0
  972. package/dist/services/notification-emitter.js +177 -0
  973. package/dist/services/notification-emitter.js.map +1 -0
  974. package/dist/services/opencode-free-sync.d.ts +29 -0
  975. package/dist/services/opencode-free-sync.d.ts.map +1 -0
  976. package/dist/services/opencode-free-sync.js +206 -0
  977. package/dist/services/opencode-free-sync.js.map +1 -0
  978. package/dist/services/penalty-inspector.d.ts +56 -0
  979. package/dist/services/penalty-inspector.d.ts.map +1 -0
  980. package/dist/services/penalty-inspector.js +167 -0
  981. package/dist/services/penalty-inspector.js.map +1 -0
  982. package/dist/services/profile-models.d.ts +5 -0
  983. package/dist/services/profile-models.d.ts.map +1 -0
  984. package/dist/services/profile-models.js +62 -0
  985. package/dist/services/profile-models.js.map +1 -0
  986. package/dist/services/provider-credential.d.ts +19 -0
  987. package/dist/services/provider-credential.d.ts.map +1 -0
  988. package/dist/services/provider-credential.js +51 -0
  989. package/dist/services/provider-credential.js.map +1 -0
  990. package/dist/services/provider-quota.d.ts +64 -0
  991. package/dist/services/provider-quota.d.ts.map +1 -0
  992. package/dist/services/provider-quota.js +563 -0
  993. package/dist/services/provider-quota.js.map +1 -0
  994. package/dist/services/quirks.d.ts +0 -0
  995. package/dist/services/quirks.d.ts.map +1 -0
  996. package/dist/services/quirks.js +0 -0
  997. package/dist/services/quirks.js.map +1 -0
  998. package/dist/services/quota-forecast.d.ts +50 -0
  999. package/dist/services/quota-forecast.d.ts.map +1 -0
  1000. package/dist/services/quota-forecast.js +150 -0
  1001. package/dist/services/quota-forecast.js.map +1 -0
  1002. package/dist/services/quota-outlook.d.ts +4 -0
  1003. package/dist/services/quota-outlook.d.ts.map +1 -0
  1004. package/dist/services/quota-outlook.js +92 -0
  1005. package/dist/services/quota-outlook.js.map +1 -0
  1006. package/dist/services/ratelimit.d.ts +249 -0
  1007. package/dist/services/ratelimit.d.ts.map +1 -0
  1008. package/dist/services/ratelimit.js +1182 -0
  1009. package/dist/services/ratelimit.js.map +1 -0
  1010. package/dist/services/request-retention.d.ts +60 -0
  1011. package/dist/services/request-retention.d.ts.map +1 -0
  1012. package/dist/services/request-retention.js +229 -0
  1013. package/dist/services/request-retention.js.map +1 -0
  1014. package/dist/services/router.d.ts +384 -0
  1015. package/dist/services/router.d.ts.map +1 -0
  1016. package/dist/services/router.js +1986 -0
  1017. package/dist/services/router.js.map +1 -0
  1018. package/dist/services/scoring.d.ts +156 -0
  1019. package/dist/services/scoring.d.ts.map +1 -0
  1020. package/dist/services/scoring.js +435 -0
  1021. package/dist/services/scoring.js.map +1 -0
  1022. package/dist/services/url-tokens.d.ts +15 -0
  1023. package/dist/services/url-tokens.d.ts.map +1 -0
  1024. package/dist/services/url-tokens.js +55 -0
  1025. package/dist/services/url-tokens.js.map +1 -0
  1026. package/package.json +77 -4
  1027. package/README.md +0 -3
@@ -0,0 +1,2793 @@
1
+ import crypto from 'crypto';
2
+ import { Router } from 'express';
3
+ import { z } from 'zod';
4
+ import { routeRequest, resolveRoutingChain, resolveModelGroupCandidates, resolveStickyPreference, hasEnabledVisionModel, hasEnabledToolsModel, routingReserveTokens } from '../services/router.js';
5
+ import { secondsUntilNextMonth } from '../services/key-budget.js';
6
+ import { runEmbeddings, EmbeddingsError } from '../services/embeddings.js';
7
+ import { retryAfterSeconds } from '../lib/retry-hint.js';
8
+ import { runImageGeneration, runVideoGeneration, runSpeech, runTranscription, MediaError, MAX_TRANSCRIPTION_BYTES } from '../services/media.js';
9
+ import multer from 'multer';
10
+ import { getDb } from '../db/index.js';
11
+ import { resolveAuth, prependSystemPrompt } from '../lib/system-prompt.js';
12
+ import { contentToString, estimateInputTokens, messageHasImage, normalizeOutboundContent, sanitizeResponse, truncateMessagesForGithub } from '../lib/content.js';
13
+ import { routeOutputBudget } from '../lib/output-cap.js';
14
+ import { resolveTaskType } from '../lib/task-type.js';
15
+ import { normalizeMessageImages } from '../lib/image-normalize.js';
16
+ import { repairToolArguments, toolSchemaMap } from '../lib/tool-args.js';
17
+ import { invalidToolArgumentsError, invalidToolCallReasons, isToolArgumentValidationEnabled } from '../lib/tool-validate.js';
18
+ import { sanitizeProviderErrorMessage } from '../lib/error-redaction.js';
19
+ import { rescueInlineToolCalls, startsWithDialectMarker, couldBecomeDialectMarker, containsDialectMarker } from '../lib/tool-call-rescue.js';
20
+ import { getContextHandoffMode, recordIncomingMessages, maybeInjectContextHandoff, recordSuccessfulModel, hasPriorModel, HANDOFF_MAX_TOKENS } from '../services/context-handoff.js';
21
+ import { isFusionModel, runFusion, fusionConfigSchema, FusionError, FUSION_MODEL_ID } from '../services/fusion.js';
22
+ import { isRetryableError, isPaymentRequiredError, isModelNotFoundError, isModelAccessForbiddenError, isClientAbortError, newClientAbortError, newHedgeAbortError, isUpstreamClassificationOutput } from '../lib/error-classify.js';
23
+ import { logRequest } from '../lib/request-log.js';
24
+ import { observeServedModel } from '../lib/served-model.js';
25
+ import { parseCacheDirective, cacheActive, isCacheableTemperature, computeCacheKey, getCachedResponse, storeCachedResponse, getCachedStreamResponse, storeCachedStreamResponse, STREAM_CACHE_MAX_BYTES } from '../services/cache.js';
26
+ import { normalizeIdempotencyKey, hashIdempotencyKey, computeIdempotencyFingerprint, lookupIdempotencyReplay, storeIdempotencyResult } from '../services/idempotency.js';
27
+ import { runFallbackLoop, newFallbackState, fallbackRoutingTokens, recordUpstreamSuccess, exhaustedRetryError, setFallbackHeaders, exhaustionErrorPayload, setExhaustionHeaders } from '../lib/fallback-loop.js';
28
+ import { routedViaValue, safeHeaderValue } from '../lib/header-value.js';
29
+ import { applyTokenBudget, tokenBudgetMessage } from '../lib/guardrails.js';
30
+ import { samplingParamSchemaFields, pickSamplingParams, supportedParametersForPlatforms } from '../lib/sampling-params.js';
31
+ import { enforceJsonContent } from '../lib/structured-output.js';
32
+ import { inferQuotaPoolKey } from '../services/provider-quota.js';
33
+ import { isUnifyEnabled, getModelGroups, resolveRequestedIdForDispatch } from '../services/model-groups.js';
34
+ import { buildModelListing } from '../services/model-listing.js';
35
+ import { claudeFamilyDiscoveryEntries } from '../services/anthropic-map.js';
36
+ import { compressRequest, formatCompressionHeader } from '../services/compression/pipeline.js';
37
+ import { recordRouteTrace } from '../lib/route-live.js';
38
+ export const proxyRouter = Router();
39
+ // Virtual "auto" model. Clients like Hermes require a non-empty `model` field
40
+ // on every request, but srouter's whole point is to pick the model itself.
41
+ // Requesting this id means "let the router decide" — identical to omitting
42
+ // `model` entirely.
43
+ const AUTO_MODEL_ID = 'auto';
44
+ function isAutoModel(modelId) {
45
+ if (!modelId)
46
+ return true;
47
+ const lower = modelId.toLowerCase();
48
+ return lower === AUTO_MODEL_ID || lower.startsWith(`${AUTO_MODEL_ID}:`);
49
+ }
50
+ // timingSafeStringEqual moved to lib/system-prompt.ts (resolveAuth needs it
51
+ // and importing it back from this route would be a cycle). Re-exported here
52
+ // for existing importers (anthropic, gemini, mcp, ollama, status, url-tokens).
53
+ export { timingSafeStringEqual } from '../lib/system-prompt.js';
54
+ // Shared auth gate for the /v1 inference endpoints (#411): accepts the unified
55
+ // key (default behavior, no enforced prompt) or an enabled client-profile key
56
+ // (which may carry a server-enforced system prompt). Profile keys are ONLY
57
+ // valid here — never on the /api dashboard surface. Writes the 401 itself so
58
+ // call sites can simply bail on null.
59
+ function requireInferenceAuth(req, res) {
60
+ const auth = resolveAuth(extractApiToken(req));
61
+ if (!auth) {
62
+ res.status(401).json({ error: { message: 'Invalid API key', type: 'authentication_error' } });
63
+ return null;
64
+ }
65
+ return auth;
66
+ }
67
+ // Extract the unified API key from an incoming request. Accepts both the
68
+ // OpenAI Bearer, Anthropic x-api-key, and Gemini x-goog-api-key headers.
69
+ // Query credentials remain scoped to the Gemini router.
70
+ export function extractApiToken(req) {
71
+ const bearer = req.headers.authorization?.replace(/^Bearer\s+/i, '').trim();
72
+ if (bearer)
73
+ return bearer;
74
+ const apiKeyHeader = req.headers['x-api-key'];
75
+ const xApiKey = Array.isArray(apiKeyHeader) ? apiKeyHeader[0] : apiKeyHeader;
76
+ const trimmed = xApiKey?.trim();
77
+ if (trimmed)
78
+ return trimmed;
79
+ const googleHeader = req.headers['x-goog-api-key'];
80
+ const googleKey = Array.isArray(googleHeader) ? googleHeader[0] : googleHeader;
81
+ return googleKey?.trim() || undefined;
82
+ }
83
+ function quotaContextForRoute(route, endpoint) {
84
+ return {
85
+ platform: route.platform,
86
+ keyId: route.keyId,
87
+ modelId: route.modelId,
88
+ quotaPoolKey: inferQuotaPoolKey(route.platform, route.modelId),
89
+ endpoint,
90
+ origin: 'proxy',
91
+ };
92
+ }
93
+ export function getRequestGroupId(req) {
94
+ const raw = req.headers['x-request-id'];
95
+ const value = Array.isArray(raw) ? raw[0] : raw;
96
+ const trimmed = value?.trim();
97
+ return trimmed || crypto.randomUUID();
98
+ }
99
+ function shortRequestId(requestId) {
100
+ return requestId.replace(/-/g, '').slice(0, 6);
101
+ }
102
+ /**
103
+ * Stamp the execution id onto a body that was stored by an EARLIER request
104
+ * (response cache hit, idempotency replay). The id goes on the outbound copy
105
+ * only — never on the stored entry — so a replay reports the id of THIS
106
+ * request instead of the one that first filled the store, which is what makes
107
+ * the field safe to trust on every response. Stored bodies are typed
108
+ * `unknown`; a non-object entry (corrupt) is passed through untouched rather
109
+ * than spread into indexed characters.
110
+ */
111
+ function withExecutionId(body, executionId) {
112
+ if (!body || typeof body !== 'object' || Array.isArray(body))
113
+ return body;
114
+ return { ...body, execution_id: executionId };
115
+ }
116
+ export function traceRouteEvent(scope, opts) {
117
+ const parts = [
118
+ `[${scope}]`,
119
+ new Date().toISOString().slice(11, 19),
120
+ opts.event,
121
+ shortRequestId(opts.requestId),
122
+ `a${opts.attempt}`,
123
+ opts.platform,
124
+ '-',
125
+ opts.model,
126
+ ];
127
+ if (opts.requestedModel)
128
+ parts.push(`req=${opts.requestedModel}`);
129
+ if (opts.latencyMs != null)
130
+ parts.push(`lat=${opts.latencyMs}ms`);
131
+ if (opts.inputTokens != null)
132
+ parts.push(`in=${opts.inputTokens}`);
133
+ if (opts.outputTokens != null)
134
+ parts.push(`out=${opts.outputTokens}`);
135
+ if (opts.error)
136
+ parts.push(`err=${JSON.stringify(opts.error)}`);
137
+ // Mirror the trace into the in-memory live tracker the dashboard's routing
138
+ // strip reads (lib/route-live). One call here covers every Proxy/Responses
139
+ // start/next/ok/fail/canceled without touching the request path.
140
+ recordRouteTrace(opts);
141
+ console.log(parts.join(' '));
142
+ }
143
+ // exhaustedRetryError moved to lib/fallback-loop.ts (the shared retry loop needs
144
+ // it and importing it back from a route would be a cycle). Re-exported here for
145
+ // existing importers (routes/responses.ts, proxy-retry.test.ts historically).
146
+ export { exhaustedRetryError };
147
+ // Sticky sessions: track which model served each "session"
148
+ // Key: hash of first user message → model_db_id
149
+ // This prevents model switching mid-conversation which causes hallucination
150
+ const stickySessionMap = new Map();
151
+ const STICKY_TTL_MS = 30 * 60 * 1000; // 30 min session TTL
152
+ // #797: per-session memory of the last assistant turn's thinking trace.
153
+ // DeepSeek thinking models on OpenCode Zen 400 on a follow-up turn unless the
154
+ // prior `reasoning_content` is replayed; opencode (and other AI-SDK clients)
155
+ // strip the field when re-serializing history, so the proxy restores what it
156
+ // itself returned last turn. Non-thinking sessions never record an entry.
157
+ // The trace is stored WITH the model key that produced it: a remembered trace
158
+ // is only ever replayed to that same platform+model, so a session that fails
159
+ // over (or auto-routes elsewhere on the next turn) never carries one model's
160
+ // thinking into another provider's payload.
161
+ const reasoningMemory = new Map();
162
+ const REASONING_TTL_MS = 30 * 60 * 1000; // 30 min, matching sticky sessions
163
+ // Platforms that reject an assistant turn WITHOUT `reasoning_content` once the
164
+ // conversation is in thinking mode, i.e. where the field has to be present on
165
+ // every assistant message and older turns need an empty-string filler. Only
166
+ // OpenCode Zen is on record for this (the DeepSeek thinking semantics behind
167
+ // #255/#797); everywhere else only the turn we actually have a trace for is
168
+ // touched, so no other provider's bytes change.
169
+ const PLATFORMS_REQUIRING_REASONING_ECHO = new Set(['opencode']);
170
+ function rememberReasoning(sessionKey, modelKey, reasoning) {
171
+ if (!sessionKey || !reasoning)
172
+ return;
173
+ reasoningMemory.set(sessionKey, { reasoning, modelKey, lastUsed: Date.now() });
174
+ if (reasoningMemory.size > 500) {
175
+ const now = Date.now();
176
+ for (const [key, entry] of reasoningMemory) {
177
+ if (now - entry.lastUsed > REASONING_TTL_MS)
178
+ reasoningMemory.delete(key);
179
+ }
180
+ // Hard cap: traces are far bigger than a sticky entry, so an all-fresh map
181
+ // must not grow without bound. Evict oldest by lastUsed, as setStickyModel does.
182
+ if (reasoningMemory.size > 1000) {
183
+ const entries = [...reasoningMemory.entries()].sort((a, b) => a[1].lastUsed - b[1].lastUsed);
184
+ const toEvict = reasoningMemory.size - 1000;
185
+ for (let i = 0; i < toEvict; i++)
186
+ reasoningMemory.delete(entries[i][0]);
187
+ }
188
+ }
189
+ }
190
+ // The remembered trace for this session, or undefined when there is none, it
191
+ // expired, or it came from a different model than the one about to be called.
192
+ // An expired entry is dropped on read rather than left for the size sweep.
193
+ export function clearReasoningMemory() {
194
+ reasoningMemory.clear();
195
+ stickySessionMap.clear();
196
+ }
197
+ function rememberedReasoningFor(sessionKey, modelKey) {
198
+ if (!sessionKey)
199
+ return undefined;
200
+ const entry = reasoningMemory.get(sessionKey);
201
+ if (!entry)
202
+ return undefined;
203
+ if (Date.now() - entry.lastUsed > REASONING_TTL_MS) {
204
+ reasoningMemory.delete(sessionKey);
205
+ return undefined;
206
+ }
207
+ return entry.modelKey === modelKey ? entry.reasoning : undefined;
208
+ }
209
+ // Put the remembered trace back on the newest assistant turn that lost it.
210
+ // Returns a NEW array (with new objects for the messages it changes) whenever
211
+ // it changes anything — the caller's `messages` are what handoff recording,
212
+ // logging, compression and the response cache already saw, and must not move
213
+ // under them. Returns the input untouched when there is nothing to restore.
214
+ function restoreSessionReasoning(messages, reasoning, platform) {
215
+ // Older assistant turns get "" — DeepSeek requires the field on every
216
+ // assistant message, and an empty string satisfies it (see opencode issue
217
+ // #24104). Only for platforms that actually enforce that; elsewhere just the
218
+ // one turn we have a real trace for is touched.
219
+ const fillOlderTurns = PLATFORMS_REQUIRING_REASONING_ECHO.has(platform);
220
+ let restored;
221
+ let restoredLatest = false;
222
+ for (let i = messages.length - 1; i >= 0; i -= 1) {
223
+ const m = messages[i];
224
+ if (m.role !== 'assistant')
225
+ continue;
226
+ // The client kept the field — nothing was dropped, leave it alone.
227
+ if (typeof m.reasoning_content === 'string' && m.reasoning_content.length > 0)
228
+ continue;
229
+ restored ??= [...messages];
230
+ restored[i] = { ...m, reasoning_content: restoredLatest ? '' : reasoning };
231
+ if (!restoredLatest && !fillOlderTurns)
232
+ break;
233
+ restoredLatest = true;
234
+ }
235
+ return restored ?? messages;
236
+ }
237
+ function getSessionKey(messages, sessionIdHeader, strategyKey) {
238
+ if (sessionIdHeader) {
239
+ return strategyKey ? `hdr:${sessionIdHeader}::${strategyKey}` : `hdr:${sessionIdHeader}`;
240
+ }
241
+ const firstUser = messages.find(m => m.role === 'user');
242
+ if (!firstUser)
243
+ return '';
244
+ const text = contentToString(firstUser.content ?? '');
245
+ if (!text)
246
+ return '';
247
+ const payload = strategyKey ? `${text}::${strategyKey}` : text;
248
+ return crypto.createHash('sha1').update(payload).digest('hex');
249
+ }
250
+ export function getStickyModel(messages, sessionIdHeader, strategyKey) {
251
+ const hasAssistant = messages.some(m => m.role === 'assistant');
252
+ if (!hasAssistant)
253
+ return undefined;
254
+ const key = getSessionKey(messages, sessionIdHeader, strategyKey);
255
+ if (!key)
256
+ return undefined;
257
+ const entry = stickySessionMap.get(key);
258
+ if (!entry)
259
+ return undefined;
260
+ if (Date.now() - entry.lastUsed > STICKY_TTL_MS) {
261
+ stickySessionMap.delete(key);
262
+ return undefined;
263
+ }
264
+ return entry.modelDbId;
265
+ }
266
+ export function setStickyModel(messages, modelDbId, sessionIdHeader, strategyKey) {
267
+ const key = getSessionKey(messages, sessionIdHeader, strategyKey);
268
+ if (!key)
269
+ return;
270
+ stickySessionMap.set(key, { modelDbId, lastUsed: Date.now() });
271
+ // Cleanup old entries
272
+ if (stickySessionMap.size > 500) {
273
+ const now = Date.now();
274
+ for (const [k, v] of stickySessionMap) {
275
+ if (now - v.lastUsed > STICKY_TTL_MS)
276
+ stickySessionMap.delete(k);
277
+ }
278
+ // Hard cap: if still over 1000 after pruning expired entries, evict oldest by lastUsed
279
+ if (stickySessionMap.size > 1000) {
280
+ const entries = [...stickySessionMap.entries()].sort((a, b) => a[1].lastUsed - b[1].lastUsed);
281
+ const toEvict = stickySessionMap.size - 1000;
282
+ for (let i = 0; i < toEvict; i++) {
283
+ stickySessionMap.delete(entries[i][0]);
284
+ }
285
+ }
286
+ }
287
+ }
288
+ // OpenAI-compatible /models endpoint (used by Hermes for metadata)
289
+ // shows API models which is linked by the user
290
+ proxyRouter.get('/models', (req, res) => {
291
+ if (!requireInferenceAuth(req, res))
292
+ return;
293
+ // By default we return the WHOLE catalog (one row per model id), each tagged
294
+ // with whether it is currently usable, so a client can see everything and know
295
+ // what's connected vs. disabled/keyless (#242). `?available=true` (aliases
296
+ // `?connected=true`, `?ready=true`) narrows the list to only models that can
297
+ // serve a request right now — the previous default behavior. The `ready`
298
+ // alias is the machine-readable filter a meta-gateway uses (#433) to ask
299
+ // "which models can this instance actually serve now". `available` is computed as
300
+ // "enabled AND an enabled key can serve it"; dedup prefers an available
301
+ // instance of a model id over a disabled/keyless one.
302
+ // Shared catalog listing (one source of truth for the OpenAI and Anthropic
303
+ // /v1/models endpoints — see services/model-listing.ts). `autoContextWindow`
304
+ // is the honest ceiling for the virtual "auto" model: the largest context
305
+ // window among models that can serve a request right now. Advertising null
306
+ // makes OpenAI-compatible clients (opencode, Continue) fall back to their own
307
+ // conservative default and truncate long inputs before they reach us (#282).
308
+ const { models: allListed, autoContextWindow } = buildModelListing();
309
+ const q = String(req.query.available ?? req.query.connected ?? req.query.ready ?? '').toLowerCase();
310
+ const onlyAvailable = q === '1' || q === 'true' || q === 'yes';
311
+ const listed = onlyAvailable ? allListed.filter(m => m.available === 1) : allListed;
312
+ // Named fallback chains (#960/#895): every user-defined profile is exposed
313
+ // as an `auto:<name>` model so a client can pick a specific fallback chain
314
+ // per request (auto:my-group) instead of only the active one. Available iff
315
+ // at least one model in that profile's chain can serve a request right now.
316
+ const profileRows = getDb().prepare(`
317
+ SELECT p.id, p.name,
318
+ EXISTS (
319
+ SELECT 1
320
+ FROM profile_models pm
321
+ JOIN models m ON m.id = pm.model_db_id AND m.enabled = 1
322
+ WHERE pm.profile_id = p.id AND pm.enabled = 1
323
+ AND EXISTS (
324
+ SELECT 1 FROM api_keys k
325
+ WHERE k.platform = m.platform AND k.enabled = 1
326
+ AND (m.key_id IS NULL OR k.id = m.key_id)
327
+ )
328
+ ) AS usable,
329
+ (SELECT MAX(m2.context_window)
330
+ FROM profile_models pm2
331
+ JOIN models m2 ON m2.id = pm2.model_db_id AND m2.enabled = 1
332
+ WHERE pm2.profile_id = p.id AND pm2.enabled = 1) AS max_ctx
333
+ FROM profiles p
334
+ WHERE p.type = 'custom'
335
+ ORDER BY p.sort_order, p.id
336
+ `).all();
337
+ // Claude-family discovery entries (#880). The Anthropic-shaped GET /v1/models
338
+ // in routes/anthropic.ts already lists one id per Claude family so clients
339
+ // that only accept Claude-looking ids can discover anything at all — but that
340
+ // handler only answers when the caller sends an `anthropic-version` header.
341
+ // Claude Desktop's gateway picker fetches this path WITHOUT that header, so
342
+ // it fell through to the OpenAI-shaped listing below and still saw zero
343
+ // Claude-shaped ids. Emit the same entries here, from the same builder, so
344
+ // both shapes agree on what the gateway will serve. Listed only when
345
+ // something can actually serve them, and never when the id would collide
346
+ // with a real catalog row.
347
+ const listedIds = new Set(listed.map(m => m.id));
348
+ const claudeFamilyEntries = allListed.some(m => m.available === 1)
349
+ ? claudeFamilyDiscoveryEntries()
350
+ .filter(a => !listedIds.has(a.id))
351
+ .map(a => ({
352
+ id: a.id,
353
+ object: 'model',
354
+ created: 0,
355
+ owned_by: 'srouter',
356
+ name: a.displayName,
357
+ context_window: autoContextWindow,
358
+ context_length: autoContextWindow,
359
+ available: true,
360
+ unavailable_reason: null,
361
+ }))
362
+ : [];
363
+ // Machine-readable execution filter: `?execution_status=ready` narrows to
364
+ // models a request can actually serve RIGHT NOW (a key that is not scoped
365
+ // away, cooling down or out of window). `ready` implies available;
366
+ // `needsKey`/`exhausted` select the rest. Matched case-insensitively because
367
+ // the value itself is camelCase; an unrecognised value filters nothing, the
368
+ // same way an unrecognised `?available=` does. Like `?available=`, this
369
+ // narrows only the catalog rows: `auto`, `fusion` and the named chains are
370
+ // router entries, not models with keys of their own.
371
+ const esValues = {
372
+ ready: 'ready', needskey: 'needsKey', exhausted: 'exhausted',
373
+ };
374
+ const es = esValues[String(req.query.execution_status ?? '').toLowerCase()];
375
+ const esFiltered = es ? listed.filter(m => m.executionStatus === es) : listed;
376
+ res.json({
377
+ object: 'list',
378
+ data: [
379
+ {
380
+ id: AUTO_MODEL_ID,
381
+ object: 'model',
382
+ created: 0,
383
+ owned_by: 'srouter',
384
+ name: 'Auto (router picks the best available model)',
385
+ context_window: autoContextWindow,
386
+ // `context_length` is OpenRouter's field name and the one most
387
+ // OpenAI-compatible clients read; emit both so whichever a client
388
+ // looks for is populated. Additive — clients ignore unknown fields.
389
+ context_length: autoContextWindow,
390
+ available: true,
391
+ unavailable_reason: null,
392
+ },
393
+ {
394
+ id: FUSION_MODEL_ID,
395
+ object: 'model',
396
+ created: 0,
397
+ owned_by: 'srouter',
398
+ name: 'Fusion (panel of models answer in parallel, a judge synthesizes one answer)',
399
+ context_window: autoContextWindow,
400
+ context_length: autoContextWindow,
401
+ // Available whenever auto is — fusion needs at least one routable model.
402
+ available: autoContextWindow != null,
403
+ unavailable_reason: autoContextWindow != null ? null : 'no_models',
404
+ },
405
+ ...claudeFamilyEntries,
406
+ ...profileRows.map(p => ({
407
+ id: `auto:${p.name.toLowerCase()}`,
408
+ object: 'model',
409
+ created: 0,
410
+ owned_by: 'srouter',
411
+ name: `Auto: ${p.name} (named fallback chain)`,
412
+ context_window: p.max_ctx,
413
+ context_length: p.max_ctx,
414
+ available: p.usable === 1,
415
+ unavailable_reason: p.usable === 1 ? null : 'no_models',
416
+ })),
417
+ ...esFiltered.map(m => ({
418
+ id: m.id,
419
+ object: 'model',
420
+ created: 0,
421
+ owned_by: m.ownedBy,
422
+ name: m.name,
423
+ context_window: m.contextWindow,
424
+ context_length: m.contextWindow,
425
+ // Non-standard but additive: OpenAI clients ignore unknown fields.
426
+ available: m.available === 1,
427
+ unavailable_reason: m.available === 1 ? null : (m.enabled === 1 ? 'no_key' : 'disabled'),
428
+ // Machine-readable dynamic status: 'ready' | 'needsKey' | 'exhausted'
429
+ // (see services/model-listing.ts). Agents can filter with
430
+ // ?execution_status=ready and route around exhausted models.
431
+ execution_status: m.executionStatus,
432
+ // OpenRouter's field name; agents use it to pick knobs per model. For
433
+ // a unify group this is the intersection over member platforms — a
434
+ // param is only advertised when every platform the router might pick
435
+ // honors it.
436
+ supported_parameters: supportedParametersForPlatforms(m.platforms, { tools: m.supportsTools }),
437
+ })),
438
+ ],
439
+ });
440
+ });
441
+ const MAX_RETRIES = 20;
442
+ // Echo-tolerant tool calls: agents replay OUR responses back as history, and
443
+ // not all of them preserve the strict OpenAI shape. `type` may be dropped
444
+ // (re-added on forward), Gemini-lineage agents (Qwen Code, AionUI) often
445
+ // send `arguments` as a parsed object instead of a JSON string, and `id` may
446
+ // be missing or empty (ids aren't a Gemini concept) — all get normalized
447
+ // below rather than 400-ing the whole session. Missing ids are synthesized
448
+ // and paired with their tool-result messages by order. (#200)
449
+ const toolCallSchema = z.object({
450
+ id: z.string().optional(),
451
+ type: z.literal('function').optional(),
452
+ function: z.object({
453
+ name: z.string().min(1),
454
+ arguments: z.union([z.string(), z.record(z.string(), z.unknown())]),
455
+ }),
456
+ thought_signature: z.string().optional(),
457
+ });
458
+ const toolCallArgsToString = (args) => typeof args === 'string' ? args : JSON.stringify(args);
459
+ // OpenAI multimodal envelope. Clients like opencode / continue.dev send
460
+ // content as an array of typed blocks even when only text is present, and
461
+ // Gemini-lineage agents send part-style blocks like `{ "text": "..." }` with
462
+ // no `type` at all. Accept any object (or bare string) as a block; flatten to
463
+ // string for providers that don't support arrays (Cohere, Cloudflare).
464
+ // Non-text blocks pass z validation but get dropped by contentToString —
465
+ // vision/audio still isn't supported. (#200)
466
+ const contentBlockSchema = z.union([z.string(), z.record(z.string(), z.unknown())]);
467
+ const contentSchema = z.union([z.string(), z.array(contentBlockSchema)]);
468
+ const systemMessageSchema = z.object({
469
+ role: z.literal('system'),
470
+ content: contentSchema,
471
+ name: z.string().optional(),
472
+ });
473
+ // OpenAI's newer SDKs send the system prompt as role:"developer"; accept it
474
+ // and forward as "system" — none of the routed providers know the developer
475
+ // role. (#200)
476
+ const developerMessageSchema = z.object({
477
+ role: z.literal('developer'),
478
+ content: contentSchema,
479
+ name: z.string().optional(),
480
+ });
481
+ const userMessageSchema = z.object({
482
+ role: z.literal('user'),
483
+ content: contentSchema,
484
+ name: z.string().optional(),
485
+ });
486
+ // Assistant turns may carry empty/null content and no tool_calls — OpenAI
487
+ // accepts these in conversation history (a turn that produced no visible text,
488
+ // a placeholder, a tool turn whose content was emptied), and clients replay
489
+ // them verbatim. We accept them too and coerce empty/null content to "" before
490
+ // forwarding (see message build below) rather than 400-ing a payload OpenAI
491
+ // would take. (#165)
492
+ const assistantMessageSchema = z.object({
493
+ role: z.literal('assistant'),
494
+ content: z.union([contentSchema, z.null()]).optional(),
495
+ name: z.string().optional(),
496
+ // tool_calls: null (not just missing) is what several agents replay for
497
+ // no-tool assistant turns — aionrs (AionUI's engine) writes it into every
498
+ // session-resumed assistant echo. Treated as absent. (#200)
499
+ tool_calls: z.array(toolCallSchema).nullable().optional(),
500
+ // Thinking trace echoed back by a client. DeepSeek thinking models on
501
+ // OpenCode Zen 400 ("reasoning_content in thinking mode must be passed back")
502
+ // unless the prior turn's reasoning_content is replayed, so keep it through
503
+ // validation instead of stripping it. See issue #255.
504
+ reasoning_content: z.string().nullable().optional(),
505
+ // Moonshot's "partial" prefill flag. A plain z.object (no .passthrough())
506
+ // would silently strip it; keep it through validation so it can be forwarded
507
+ // to Moonshot/Kimi models, which document it. See issue #1038.
508
+ partial: z.boolean().optional(),
509
+ });
510
+ // Tool results may arrive with null/missing content (a tool that returned
511
+ // nothing) and a missing/empty tool_call_id (Gemini-lineage agents) — coerced
512
+ // to "" and paired by order with the preceding tool_calls respectively. (#200)
513
+ const toolMessageSchema = z.object({
514
+ role: z.literal('tool'),
515
+ content: z.union([contentSchema, z.null()]).optional(),
516
+ tool_call_id: z.string().optional(),
517
+ name: z.string().optional(),
518
+ });
519
+ // Legacy function-calling shape (pre-tools OpenAI API). Old clients still
520
+ // replay these in history; forwarded as a tool message. (#200)
521
+ const functionMessageSchema = z.object({
522
+ role: z.literal('function'),
523
+ name: z.string().min(1),
524
+ content: z.union([contentSchema, z.null()]).optional(),
525
+ });
526
+ const toolDefinitionSchema = z.object({
527
+ // Some agents omit `type` on tool definitions; re-defaulted to 'function'
528
+ // on forward. (#200)
529
+ type: z.literal('function').optional(),
530
+ function: z.object({
531
+ name: z.string().min(1),
532
+ description: z.string().optional(),
533
+ parameters: z.record(z.string(), z.unknown()).optional(),
534
+ strict: z.boolean().optional(),
535
+ }),
536
+ });
537
+ const toolChoiceSchema = z.union([
538
+ // 'any' is the Mistral/Gemini wording for OpenAI's 'required'; mapped on
539
+ // forward. (#200)
540
+ z.enum(['none', 'auto', 'required', 'any']),
541
+ z.object({
542
+ type: z.literal('function'),
543
+ function: z.object({
544
+ name: z.string().min(1),
545
+ }),
546
+ }),
547
+ ]);
548
+ const stopSchema = z.union([z.string(), z.array(z.string()).min(1).max(64)]);
549
+ function providerSafeStop(stop) {
550
+ if (!Array.isArray(stop))
551
+ return stop;
552
+ return stop.slice(0, 4);
553
+ }
554
+ const chatCompletionSchema = z.object({
555
+ messages: z.array(z.union([
556
+ systemMessageSchema,
557
+ developerMessageSchema,
558
+ userMessageSchema,
559
+ assistantMessageSchema,
560
+ toolMessageSchema,
561
+ functionMessageSchema,
562
+ ])).min(1),
563
+ model: z.string().optional(),
564
+ temperature: z.number().min(0).max(2).optional(),
565
+ // Some clients send max_tokens <= 0 (or -1) to mean "no limit"; accepted and
566
+ // treated as unset on forward. (#200)
567
+ max_tokens: z.number().int().optional(),
568
+ top_p: z.number().min(0).max(1).optional(),
569
+ stop: stopSchema.optional(),
570
+ stream: z.boolean().optional(),
571
+ stream_options: z.object({
572
+ include_usage: z.boolean().optional(),
573
+ }).optional(),
574
+ // Top-level tool knobs may arrive as explicit nulls from clients that
575
+ // serialize every field of their request struct; all treated as absent
576
+ // and never forwarded as null. (#200)
577
+ tools: z.array(toolDefinitionSchema).nullable().optional(),
578
+ tool_choice: toolChoiceSchema.nullable().optional(),
579
+ parallel_tool_calls: z.boolean().nullable().optional(),
580
+ // Fusion config — only meaningful when `model` is the virtual "fusion" id.
581
+ // Ignored for every other model. See services/fusion.ts.
582
+ fusion: fusionConfigSchema.optional(),
583
+ // Extended sampling + structured-output params (top_k, seed, penalties,
584
+ // logit_bias, logprobs, response_format, max_completion_tokens…), forwarded
585
+ // per the platform policy in lib/sampling-params.ts.
586
+ ...samplingParamSchemaFields,
587
+ });
588
+ // Upstream-error classifiers live in lib/error-classify.ts so the fusion
589
+ // service can share them without an import cycle; imported above for internal
590
+ // use and re-exported here for existing importers (routes/responses.ts,
591
+ // proxy-retry.test.ts) that pull them from this module.
592
+ export { isRetryableError, isPaymentRequiredError, isModelNotFoundError, isModelAccessForbiddenError };
593
+ // Pull the incremental text out of a streaming chunk for token counting.
594
+ // Must tolerate chunks that carry no `choices` array at all: some providers
595
+ // (e.g. Groq) emit usage/keepalive frames shaped like `{usage:{...}}` with no
596
+ // `choices`. Indexing `chunk.choices[0]` on those throws "Cannot read
597
+ // properties of undefined (reading '0')", which — once the SSE stream has
598
+ // started — aborts the response mid-flight with no chance to fall back.
599
+ export function streamChunkText(chunk) {
600
+ return chunk?.choices?.[0]?.delta?.content ?? '';
601
+ }
602
+ // Pull the incremental reasoning text out of a streaming chunk. Reasoning
603
+ // models stream thinking via `reasoning_content` (Z.ai, DeepSeek-style — the
604
+ // <think> extractor in base.ts normalizes inline tags into the same field) or
605
+ // `reasoning` (Ollama-style) before the first visible answer token; both
606
+ // spellings must count for ttfb and output-token estimates. Same shape
607
+ // tolerance as streamChunkText. (#764)
608
+ export function streamReasoningText(chunk) {
609
+ const delta = chunk?.choices?.[0]?.delta;
610
+ const r = delta?.reasoning_content ?? delta?.reasoning;
611
+ return typeof r === 'string' ? r : '';
612
+ }
613
+ // OpenAI-compatible embeddings endpoint, routed through the embeddings family
614
+ // catalog: `model: "auto"` (or omitted) → the configured default family; a
615
+ // family name or provider model id → that family's provider chain. Failover
616
+ // only happens WITHIN a family (same model on another provider) — never across
617
+ // models, since vectors from different models are incompatible.
618
+ const EmbeddingsBody = z.object({
619
+ model: z.string().optional(),
620
+ input: z.union([z.string(), z.array(z.string())]),
621
+ // Optional output-dimension override forwarded to providers that support MRL
622
+ // truncation (NVIDIA NeMo NIM, Google Gemini Embedding, OpenAI v3). Validation
623
+ // only — bounds checking happens upstream (the provider rejects out-of-range
624
+ // values with a clear 400).
625
+ dimensions: z.number().int().positive().optional(),
626
+ });
627
+ proxyRouter.post('/embeddings', async (req, res) => {
628
+ if (!requireInferenceAuth(req, res))
629
+ return;
630
+ const parsed = EmbeddingsBody.safeParse(req.body);
631
+ if (!parsed.success) {
632
+ res.status(400).json({ error: { message: 'Invalid request: `input` is required', type: 'invalid_request_error' } });
633
+ return;
634
+ }
635
+ const inputs = Array.isArray(parsed.data.input) ? parsed.data.input : [parsed.data.input];
636
+ try {
637
+ const result = await runEmbeddings(parsed.data.model, inputs, parsed.data.dimensions);
638
+ res.json({
639
+ object: 'list',
640
+ data: result.vectors.map((values, i) => ({ object: 'embedding', index: i, embedding: values })),
641
+ model: result.family,
642
+ provider: result.platform,
643
+ usage: { prompt_tokens: result.inputTokens, total_tokens: result.inputTokens },
644
+ });
645
+ }
646
+ catch (err) {
647
+ const status = err instanceof EmbeddingsError ? err.status : 502;
648
+ const code = err instanceof EmbeddingsError ? inferenceBudgetCode(err, res) : {};
649
+ const type = status === 400 ? 'invalid_request_error' : status === 429 ? 'rate_limit_error' : 'server_error';
650
+ res.status(status).json({ error: { message: `embedding error: ${err?.message ?? 'unknown'}`, type, ...code } });
651
+ }
652
+ });
653
+ // OpenAI-compatible image generation. Routed through the media catalog (its own
654
+ // table, never the chat router): `model: "auto"` (or omitted) tries every enabled
655
+ // image provider in order; a provider model id pins to that one. Failover is
656
+ // across providers, never across modalities. See services/media.ts.
657
+ const ImageBody = z.object({
658
+ model: z.string().optional(),
659
+ prompt: z.string().min(1),
660
+ n: z.number().int().positive().max(4).optional(),
661
+ size: z.string().optional(),
662
+ response_format: z.enum(['url', 'b64_json']).optional(),
663
+ });
664
+ function inferenceBudgetCode(error, res) {
665
+ // retryAfterMs is set by the embeddings/media services only when the whole
666
+ // chain was rate limited (soonest stated back-off, budget resets included),
667
+ // so it wins over the month-long budget reset when a sibling returns sooner.
668
+ if (error.retryAfterMs !== undefined)
669
+ res.setHeader('Retry-After', retryAfterSeconds(error.retryAfterMs));
670
+ else if (error.code === 'quota_exceeded')
671
+ res.setHeader('Retry-After', secondsUntilNextMonth());
672
+ return error.code ? { code: error.code } : {};
673
+ }
674
+ function mediaErrorType(status) {
675
+ if (status === 400 || status === 413)
676
+ return 'invalid_request_error';
677
+ if (status === 401)
678
+ return 'authentication_error';
679
+ if (status === 429)
680
+ return 'rate_limit_error';
681
+ return 'server_error';
682
+ }
683
+ proxyRouter.post('/images/generations', async (req, res) => {
684
+ if (!requireInferenceAuth(req, res))
685
+ return;
686
+ const parsed = ImageBody.safeParse(req.body);
687
+ if (!parsed.success) {
688
+ res.status(400).json({ error: { message: 'Invalid request: `prompt` is required', type: 'invalid_request_error' } });
689
+ return;
690
+ }
691
+ try {
692
+ const result = await runImageGeneration(parsed.data.model, {
693
+ prompt: parsed.data.prompt, n: parsed.data.n, size: parsed.data.size,
694
+ });
695
+ res.json({
696
+ created: Math.floor(Date.now() / 1000),
697
+ data: result.images,
698
+ model: result.modelId,
699
+ provider: result.platform,
700
+ });
701
+ }
702
+ catch (err) {
703
+ const status = err instanceof MediaError ? err.status : 502;
704
+ const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
705
+ const httpStatus = status >= 400 && status < 600 ? status : 502;
706
+ res.status(httpStatus).json({ error: { message: `image generation error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
707
+ }
708
+ });
709
+ // Text-to-video generation. Providers may use a synchronous binary response
710
+ // (Pollinations) or an asynchronous queue internally (Hugging Face/fal.ai), but
711
+ // this gateway presents one bounded request and returns the completed MP4.
712
+ const VideoBody = z.object({
713
+ model: z.string().optional(),
714
+ prompt: z.string().min(1),
715
+ duration: z.number().int().min(1).max(120).optional(),
716
+ aspect_ratio: z.enum(['16:9', '9:16']).optional(),
717
+ image: z.string().url().optional(),
718
+ seed: z.number().int().min(-1).max(2_147_483_647).optional(),
719
+ audio: z.boolean().optional(),
720
+ });
721
+ proxyRouter.post('/videos/generations', async (req, res) => {
722
+ if (!requireInferenceAuth(req, res))
723
+ return;
724
+ const parsed = VideoBody.safeParse(req.body);
725
+ if (!parsed.success) {
726
+ res.status(400).json({
727
+ error: {
728
+ message: 'Invalid request: `prompt` is required and video options must use supported values',
729
+ type: 'invalid_request_error',
730
+ },
731
+ });
732
+ return;
733
+ }
734
+ // A video job runs for minutes, so a caller that hangs up must actually stop
735
+ // the work: without this the gateway would keep polling the provider and then
736
+ // fail over to a second one, both charged to the operator, for a response
737
+ // nobody is waiting for. 'close' also fires on normal completion, which
738
+ // writableEnded distinguishes.
739
+ const clientAbort = new AbortController();
740
+ res.on('close', () => {
741
+ if (!res.writableEnded)
742
+ clientAbort.abort();
743
+ });
744
+ try {
745
+ const result = await runVideoGeneration(parsed.data.model, {
746
+ prompt: parsed.data.prompt,
747
+ duration: parsed.data.duration,
748
+ aspectRatio: parsed.data.aspect_ratio,
749
+ image: parsed.data.image,
750
+ seed: parsed.data.seed,
751
+ audio: parsed.data.audio,
752
+ }, clientAbort.signal);
753
+ res.setHeader('Content-Type', result.contentType);
754
+ res.setHeader('Cache-Control', 'no-store');
755
+ res.setHeader('X-Provider', safeHeaderValue(result.platform));
756
+ res.setHeader('X-Model', safeHeaderValue(result.modelId));
757
+ res.send(result.video);
758
+ }
759
+ catch (err) {
760
+ // Nothing to report to a socket that is already gone.
761
+ if (clientAbort.signal.aborted || res.writableEnded)
762
+ return;
763
+ const status = err instanceof MediaError ? err.status : 502;
764
+ const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
765
+ const httpStatus = status >= 400 && status < 600 ? status : 502;
766
+ res.status(httpStatus).json({
767
+ error: { message: `video generation error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code },
768
+ });
769
+ }
770
+ });
771
+ // OpenAI-compatible text-to-speech. Returns raw audio bytes (OpenAI's /audio/speech
772
+ // shape). Same media-catalog routing as images.
773
+ const SpeechBody = z.object({
774
+ model: z.string().optional(),
775
+ input: z.string().min(1),
776
+ voice: z.string().optional(),
777
+ response_format: z.string().optional(),
778
+ });
779
+ proxyRouter.post('/audio/speech', async (req, res) => {
780
+ if (!requireInferenceAuth(req, res))
781
+ return;
782
+ const parsed = SpeechBody.safeParse(req.body);
783
+ if (!parsed.success) {
784
+ res.status(400).json({ error: { message: 'Invalid request: `input` is required', type: 'invalid_request_error' } });
785
+ return;
786
+ }
787
+ try {
788
+ const result = await runSpeech(parsed.data.model, {
789
+ input: parsed.data.input, voice: parsed.data.voice, format: parsed.data.response_format,
790
+ });
791
+ res.setHeader('Content-Type', result.contentType);
792
+ res.setHeader('X-Provider', safeHeaderValue(result.platform));
793
+ res.send(result.audio);
794
+ }
795
+ catch (err) {
796
+ const status = err instanceof MediaError ? err.status : 502;
797
+ const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
798
+ const httpStatus = status >= 400 && status < 600 ? status : 502;
799
+ res.status(httpStatus).json({ error: { message: `speech error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
800
+ }
801
+ });
802
+ // OpenAI-compatible speech-to-text (/v1/audio/transcriptions). Multipart form
803
+ // upload, held in memory only (multer memoryStorage — audio bytes never touch
804
+ // disk), routed through the STT provider chain in services/media.ts with the
805
+ // same key/failover/cooldown machinery as the other media endpoints. The STT
806
+ // registry (media_models, modality='transcription') is maintained by the
807
+ // published catalog's `transcriptionModels` array via catalog-sync, plus any
808
+ // OpenAI-compatible endpoint the operator registered themselves through
809
+ // POST /api/media/custom; on an install that has never synced one and has no
810
+ // custom row, the endpoint answers 503 with code 'no_transcription_models'.
811
+ //
812
+ // response_format: 'json' (default, {"text": ...}), 'text' (plain string),
813
+ // 'verbose_json' (OpenAI verbose shape when the provider returns segments,
814
+ // graceful fallback to the plain json shape otherwise), 'vtt' (only from
815
+ // providers that produce it natively — Cloudflare whisper). 'srt' is not
816
+ // produced natively by any configured provider and is refused with 400
817
+ // unsupported_format rather than synthesized.
818
+ const transcriptionUpload = multer({
819
+ storage: multer.memoryStorage(),
820
+ limits: { fileSize: MAX_TRANSCRIPTION_BYTES, files: 1 },
821
+ });
822
+ const TRANSCRIPTION_FORMATS = new Set(['json', 'text', 'verbose_json', 'srt', 'vtt']);
823
+ function transcriptionBadRequest(res, message, code) {
824
+ res.status(400).json({ error: { message, type: 'invalid_request_error', ...(code ? { code } : {}) } });
825
+ }
826
+ proxyRouter.post('/audio/transcriptions', (req, res, next) => {
827
+ // Auth before the multipart body is parsed: an unauthenticated caller's
828
+ // upload is never buffered.
829
+ if (!requireInferenceAuth(req, res))
830
+ return;
831
+ transcriptionUpload.single('file')(req, res, (err) => {
832
+ if (err) {
833
+ if (err instanceof multer.MulterError && err.code === 'LIMIT_FILE_SIZE') {
834
+ res.status(413).json({
835
+ error: {
836
+ message: `Audio file too large: the maximum upload size is ${MAX_TRANSCRIPTION_BYTES / (1024 * 1024)} MB.`,
837
+ type: 'invalid_request_error',
838
+ code: 'file_too_large',
839
+ },
840
+ });
841
+ return;
842
+ }
843
+ transcriptionBadRequest(res, 'Malformed multipart/form-data upload.');
844
+ return;
845
+ }
846
+ next();
847
+ });
848
+ }, async (req, res) => {
849
+ const file = req.file;
850
+ if (!file || !file.buffer?.length) {
851
+ transcriptionBadRequest(res, 'Invalid request: `file` is required (multipart/form-data audio upload).');
852
+ return;
853
+ }
854
+ const model = typeof req.body?.model === 'string' ? req.body.model.trim() : '';
855
+ if (!model) {
856
+ transcriptionBadRequest(res, "Invalid request: `model` is required (use 'whisper-1' or 'auto' to let the router decide).");
857
+ return;
858
+ }
859
+ const rawFormat = typeof req.body?.response_format === 'string' ? req.body.response_format.trim() : '';
860
+ const responseFormat = rawFormat || 'json';
861
+ if (!TRANSCRIPTION_FORMATS.has(responseFormat)) {
862
+ transcriptionBadRequest(res, `Invalid response_format '${responseFormat}'. Supported: json, text, verbose_json, vtt.`);
863
+ return;
864
+ }
865
+ if (responseFormat === 'srt') {
866
+ transcriptionBadRequest(res, "response_format 'srt' is not supported: no configured provider produces srt natively. Use json, text, verbose_json, or vtt.", 'unsupported_format');
867
+ return;
868
+ }
869
+ let temperature;
870
+ if (req.body?.temperature !== undefined && req.body.temperature !== '') {
871
+ temperature = Number(req.body.temperature);
872
+ if (!Number.isFinite(temperature) || temperature < 0 || temperature > 1) {
873
+ transcriptionBadRequest(res, 'Invalid temperature: must be a number between 0 and 1.');
874
+ return;
875
+ }
876
+ }
877
+ const language = typeof req.body?.language === 'string' && req.body.language.trim() ? req.body.language.trim() : undefined;
878
+ const prompt = typeof req.body?.prompt === 'string' && req.body.prompt ? req.body.prompt : undefined;
879
+ try {
880
+ const result = await runTranscription(model, {
881
+ file: file.buffer,
882
+ filename: file.originalname || 'audio',
883
+ mimeType: file.mimetype,
884
+ language,
885
+ prompt,
886
+ temperature,
887
+ responseFormat,
888
+ });
889
+ res.setHeader('X-Provider', safeHeaderValue(result.platform));
890
+ res.setHeader('X-Model', safeHeaderValue(result.modelId));
891
+ if (responseFormat === 'text') {
892
+ res.type('text/plain').send(result.text);
893
+ return;
894
+ }
895
+ if (responseFormat === 'vtt') {
896
+ res.type('text/vtt').send(result.vtt ?? '');
897
+ return;
898
+ }
899
+ if (responseFormat === 'verbose_json' && Array.isArray(result.segments) && result.segments.length > 0) {
900
+ res.json({
901
+ task: 'transcribe',
902
+ language: result.language ?? null,
903
+ duration: result.duration ?? null,
904
+ text: result.text,
905
+ segments: result.segments,
906
+ });
907
+ return;
908
+ }
909
+ res.json({ text: result.text });
910
+ }
911
+ catch (err) {
912
+ const status = err instanceof MediaError ? err.status : 502;
913
+ const code = err instanceof MediaError ? inferenceBudgetCode(err, res) : {};
914
+ const httpStatus = status >= 400 && status < 600 ? status : 502;
915
+ res.status(httpStatus).json({ error: { message: `transcription error: ${err?.message ?? 'unknown'}`, type: mediaErrorType(status), ...code } });
916
+ }
917
+ });
918
+ const CompletionBody = z.object({
919
+ model: z.string().optional(),
920
+ prompt: z.string(),
921
+ suffix: z.string().optional(),
922
+ temperature: z.number().min(0).max(2).optional(),
923
+ max_tokens: z.number().int().optional(),
924
+ top_p: z.number().min(0).max(1).optional(),
925
+ stop: stopSchema.optional(),
926
+ stream: z.boolean().optional(),
927
+ });
928
+ function completionPromptToMessages(prompt, suffix) {
929
+ const hasSuffix = suffix !== undefined && suffix.length > 0;
930
+ return [
931
+ {
932
+ role: 'system',
933
+ content: [
934
+ 'You are a code autocomplete engine.',
935
+ 'Complete at the cursor and return only the text to insert.',
936
+ 'Do not include markdown fences, explanations, or repeat surrounding code.',
937
+ ].join(' '),
938
+ },
939
+ {
940
+ role: 'user',
941
+ content: hasSuffix
942
+ ? `Prefix before cursor:\n${prompt}\n\nSuffix after cursor:\n${suffix}\n\nCompletion to insert:`
943
+ : `Prefix before cursor:\n${prompt}\n\nCompletion to insert:`,
944
+ },
945
+ ];
946
+ }
947
+ function completionTextFromChat(result) {
948
+ return contentToString(result?.choices?.[0]?.message?.content ?? '');
949
+ }
950
+ // Non-streaming counterpart of streamReasoningText: reasoning models attach
951
+ // thinking to the completed message as `reasoning_content` or `reasoning`.
952
+ // Included in the chars/4 output estimate so analytics and the rate-limit
953
+ // ledger aren't undercounted for thinking models. (#764)
954
+ export function completionReasoningText(result) {
955
+ const msg = result?.choices?.[0]?.message;
956
+ const r = msg?.reasoning_content ?? msg?.reasoning;
957
+ return typeof r === 'string' ? r : '';
958
+ }
959
+ function completionIdFromChat(id) {
960
+ if (!id)
961
+ return `cmpl-${Date.now()}`;
962
+ return id.startsWith('cmpl-') ? id : `cmpl-${id}`;
963
+ }
964
+ function legacyCompletionChunk(route, chunk, text) {
965
+ return {
966
+ id: completionIdFromChat(chunk?.id),
967
+ object: 'text_completion',
968
+ created: chunk?.created ?? Math.floor(Date.now() / 1000),
969
+ model: route.modelId,
970
+ choices: [{
971
+ text,
972
+ index: chunk?.choices?.[0]?.index ?? 0,
973
+ logprobs: null,
974
+ finish_reason: chunk?.choices?.[0]?.finish_reason ?? null,
975
+ }],
976
+ };
977
+ }
978
+ // OpenAI-compatible legacy completions endpoint. Editor ghost-text clients
979
+ // (notably Continue autocomplete) still send prompt/suffix requests here; route
980
+ // those through chat models while preserving the legacy text_completion shape.
981
+ proxyRouter.post('/completions', async (req, res) => {
982
+ const start = Date.now();
983
+ const requestGroupId = getRequestGroupId(req);
984
+ res.setHeader('X-Request-ID', requestGroupId);
985
+ const auth = requireInferenceAuth(req, res);
986
+ if (!auth)
987
+ return;
988
+ const parsed = CompletionBody.safeParse(req.body);
989
+ if (!parsed.success) {
990
+ const detail = parsed.error.errors
991
+ .map(e => (e.path.length ? `${e.path.join('.')}: ${e.message}` : e.message))
992
+ .slice(0, 5)
993
+ .join(', ');
994
+ res.status(400).json({
995
+ error: { message: `Invalid request: ${detail}`, type: 'invalid_request_error' },
996
+ execution_id: requestGroupId,
997
+ });
998
+ return;
999
+ }
1000
+ const { model: requestedModel, prompt, suffix, temperature, top_p, stream } = parsed.data;
1001
+ const requestedModelLabel = requestedModel ?? 'auto';
1002
+ const max_tokens = parsed.data.max_tokens != null && parsed.data.max_tokens > 0
1003
+ ? parsed.data.max_tokens : 128;
1004
+ const stop = providerSafeStop(parsed.data.stop);
1005
+ // A profile's enforced prompt goes ahead of the autocomplete system message.
1006
+ const messages = prependSystemPrompt(completionPromptToMessages(prompt, suffix), auth.systemPrompt);
1007
+ const estimatedInputTokens = messages.reduce((sum, m) => sum + Math.ceil(contentToString(m.content).length / 4), 0);
1008
+ // Cap the reserved output so a huge client-set max_tokens doesn't falsely
1009
+ // exclude the whole model pool (#470); input is still counted in full. The
1010
+ // reserve is passed to the router separately: it is an exact count and must
1011
+ // not be inflated by the context-window safety margin (#956 review).
1012
+ const outputReserve = routingReserveTokens(max_tokens);
1013
+ const estimatedTotal = estimatedInputTokens + outputReserve;
1014
+ // Guardrail: per-request token budget (request_max_tokens_budget, default
1015
+ // off). max_tokens always has a value on this surface (default 128), so a
1016
+ // violation can only reject — no capping branch.
1017
+ const budgetCheck = applyTokenBudget(estimatedInputTokens, max_tokens);
1018
+ if (budgetCheck.rejection) {
1019
+ res.status(413).json({
1020
+ error: { message: tokenBudgetMessage(budgetCheck.rejection), type: 'invalid_request_error', code: 'request_token_budget' },
1021
+ execution_id: requestGroupId,
1022
+ });
1023
+ return;
1024
+ }
1025
+ let resolvedChain;
1026
+ if (isAutoModel(requestedModel)) {
1027
+ resolvedChain = resolveRoutingChain(requestedModel);
1028
+ }
1029
+ let preferredModel;
1030
+ let groupChain;
1031
+ if (!isAutoModel(requestedModel) && requestedModel) {
1032
+ const db = getDb();
1033
+ const resolved = isUnifyEnabled() ? resolveRequestedIdForDispatch(requestedModel, getModelGroups()) : null;
1034
+ const members = resolved?.memberDbIds ?? null;
1035
+ if (members && members.length > 0) {
1036
+ groupChain = resolveModelGroupCandidates(members, resolved.demotedDbIds);
1037
+ if (groupChain.length === 0) {
1038
+ const placeholders = members.map(() => '?').join(',');
1039
+ const anyEnabled = db.prepare(`SELECT 1 FROM models WHERE id IN (${placeholders}) AND enabled = 1 LIMIT 1`).get(...members);
1040
+ // Honest statuses: a model whose providers exist but have no usable key
1041
+ // is a server-side configuration gap (503), not a client mistake; a
1042
+ // disabled/unknown model is a 404 model_not_found (OpenAI semantics).
1043
+ if (anyEnabled) {
1044
+ res.status(503).json({
1045
+ error: {
1046
+ message: `Model '${requestedModel}' has no providers with an enabled key. Add a provider API key for it, use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1047
+ type: 'service_unavailable',
1048
+ code: 'no_providers_configured',
1049
+ },
1050
+ execution_id: requestGroupId,
1051
+ });
1052
+ }
1053
+ else {
1054
+ res.status(404).json({
1055
+ error: {
1056
+ message: `Model '${requestedModel}' is disabled. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1057
+ type: 'invalid_request_error',
1058
+ code: 'model_not_found',
1059
+ },
1060
+ execution_id: requestGroupId,
1061
+ });
1062
+ }
1063
+ return;
1064
+ }
1065
+ }
1066
+ else {
1067
+ const enabled = db.prepare('SELECT id FROM models WHERE model_id = ? AND enabled = 1').get(requestedModel);
1068
+ if (enabled) {
1069
+ preferredModel = enabled.id;
1070
+ }
1071
+ else {
1072
+ const disabled = db.prepare('SELECT id FROM models WHERE model_id = ?').get(requestedModel);
1073
+ const reason = disabled ? 'is disabled' : 'is not in the catalog';
1074
+ res.status(404).json({
1075
+ error: {
1076
+ message: `Model '${requestedModel}' ${reason}. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1077
+ type: 'invalid_request_error',
1078
+ code: 'model_not_found',
1079
+ },
1080
+ execution_id: requestGroupId,
1081
+ });
1082
+ return;
1083
+ }
1084
+ }
1085
+ }
1086
+ const pinnedModelId = requestedModel && !isAutoModel(requestedModel) ? requestedModel : null;
1087
+ const state = newFallbackState();
1088
+ const attemptLog = [];
1089
+ // Client-disconnect fan-out: the flag stops the loop before the NEXT
1090
+ // attempt; the AbortController (threaded to the provider as
1091
+ // CompletionOptions.signal) additionally cancels the IN-FLIGHT upstream
1092
+ // fetch and any body/stream read, so tokens stop burning and the in-flight
1093
+ // lease frees immediately. 'close' also fires on normal completion —
1094
+ // writableEnded distinguishes a real disconnect.
1095
+ let clientGone = false;
1096
+ const clientAbort = new AbortController();
1097
+ // Fallback-v2 hedging: the loop aborts this controller (via abortInFlight)
1098
+ // when the wall-clock retry budget expires mid-attempt, canceling the
1099
+ // in-flight upstream instead of waiting for a stalled attempt to time out.
1100
+ const hedgeAbort = new AbortController();
1101
+ res.on('close', () => {
1102
+ if (!res.writableEnded) {
1103
+ clientGone = true;
1104
+ clientAbort.abort(newClientAbortError());
1105
+ }
1106
+ });
1107
+ // Legacy /completions is a thin adapter over the shared fallback loop
1108
+ // (lib/fallback-loop.ts): the cooldown/skip/penalty/exhaustion machinery is
1109
+ // shared; only the text_completion request/stream translation lives here.
1110
+ await runFallbackLoop({
1111
+ maxRetries: MAX_RETRIES,
1112
+ state,
1113
+ attemptLog,
1114
+ logIdentity: { surface: 'legacy completions', requestId: requestGroupId, requestedModel: requestedModelLabel },
1115
+ clientGone: () => clientGone,
1116
+ abortInFlight: () => hedgeAbort.abort(newHedgeAbortError()),
1117
+ route: () => {
1118
+ // #507: see inbound-chat.ts — inflate the routing estimate from any
1119
+ // provider-reported REQUESTED size latched onto state so the existing
1120
+ // size gates in router.ts skip low-TPM / small-context models on retry.
1121
+ const routingTotal = fallbackRoutingTokens(state, estimatedTotal, outputReserve);
1122
+ return routeRequest(routingTotal, state.skipKeys.size > 0 ? state.skipKeys : undefined, preferredModel, false, false, state.skipModels.size > 0 ? state.skipModels : undefined, groupChain ?? resolvedChain?.chain, false, state.skipPlatforms.size > 0 ? state.skipPlatforms : undefined, outputReserve);
1123
+ },
1124
+ dispatch: async (route, attempt, ctx) => {
1125
+ const contextBudget = routeOutputBudget(route, estimatedInputTokens);
1126
+ traceRouteEvent('Proxy', {
1127
+ event: attempt === 0 ? 'start' : 'next',
1128
+ requestId: requestGroupId,
1129
+ attempt,
1130
+ platform: route.platform,
1131
+ model: route.modelId,
1132
+ requestedModel: attempt === 0 ? requestedModelLabel : undefined,
1133
+ });
1134
+ // Same GitHub input ceiling as /chat/completions below: trim the
1135
+ // dispatched copy so a long legacy prompt doesn't 413 the github hop.
1136
+ const dispatchMessages = route.platform === 'github'
1137
+ ? truncateMessagesForGithub(messages)
1138
+ : messages;
1139
+ if (stream) {
1140
+ let totalOutputTokens = 0;
1141
+ let headerSent = false;
1142
+ let ttfbMs = null;
1143
+ let sawText = false;
1144
+ let upstreamFinish = null;
1145
+ const buffered = [];
1146
+ const flushHeaders = () => {
1147
+ if (headerSent)
1148
+ return;
1149
+ // #764: ttfb is recorded on the first token of ANY kind (content or
1150
+ // reasoning) in the pump loop below; this call only backfills streams
1151
+ // that reached the commit point without one.
1152
+ if (ttfbMs === null)
1153
+ ttfbMs = Date.now() - start;
1154
+ res.setHeader('Content-Type', 'text/event-stream');
1155
+ res.setHeader('Cache-Control', 'no-cache');
1156
+ res.setHeader('Connection', 'keep-alive');
1157
+ res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
1158
+ setFallbackHeaders(res, attempt, attemptLog);
1159
+ headerSent = true;
1160
+ // Committed: the answer is on its way, so the retry budget must no
1161
+ // longer cancel this attempt (it could not fail over now anyway).
1162
+ ctx.disarmHedge();
1163
+ for (const frame of buffered)
1164
+ res.write(`data: ${JSON.stringify(frame)}\n\n`);
1165
+ buffered.length = 0;
1166
+ };
1167
+ try {
1168
+ const gen = route.provider.streamChatCompletion(route.apiKey, dispatchMessages, route.modelId, { temperature, max_tokens, top_p, stop, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
1169
+ for await (const chunk of gen) {
1170
+ if (clientGone)
1171
+ break; // client hung up: stop pulling; reader.cancel() aborts upstream
1172
+ const text = streamChunkText(chunk);
1173
+ if (text.length > 0)
1174
+ sawText = true;
1175
+ // #764: reasoning models stream thinking before any visible text.
1176
+ // ttfb must count the first token of ANY kind — otherwise the speed
1177
+ // shown is the thinking tail, or NULL when headers never flush.
1178
+ const reasoning = streamReasoningText(chunk);
1179
+ if (ttfbMs === null && (text.length > 0 || reasoning.length > 0)) {
1180
+ ttfbMs = Date.now() - start;
1181
+ }
1182
+ const finish = chunk?.choices?.[0]?.finish_reason;
1183
+ if (finish)
1184
+ upstreamFinish = finish;
1185
+ // #764: reasoning tokens are real output consumption — count them
1186
+ // so analytics and the rate-limit ledger aren't undercounted.
1187
+ totalOutputTokens += Math.ceil((text.length + reasoning.length) / 4);
1188
+ const frame = legacyCompletionChunk(route, chunk, text);
1189
+ // Commit point: hold headers until the first real text, so a stream
1190
+ // that dies before producing any fails over invisibly.
1191
+ if (!headerSent && !sawText) {
1192
+ buffered.push(frame);
1193
+ continue;
1194
+ }
1195
+ flushHeaders();
1196
+ res.write(`data: ${JSON.stringify(frame)}\n\n`);
1197
+ }
1198
+ // Disconnect before the commit point: the break above fired with no
1199
+ // text seen, which is indistinguishable from an empty completion
1200
+ // below — but it is CLIENT behavior, not a provider failure. Without
1201
+ // this check every Ctrl-C during a reasoning model's TTFB window
1202
+ // benched the healthy model+key for 90s and logged a provider error.
1203
+ if (clientGone && !headerSent && !sawText) {
1204
+ console.log(`[Proxy] client disconnected before first token from ${route.displayName} — dropping attempt without benching`);
1205
+ traceRouteEvent('Proxy', {
1206
+ event: 'canceled',
1207
+ requestId: requestGroupId,
1208
+ attempt,
1209
+ platform: route.platform,
1210
+ model: route.modelId,
1211
+ });
1212
+ return 'committed';
1213
+ }
1214
+ if (!sawText) {
1215
+ // finish_reason 'length' means the model spent the whole output
1216
+ // budget before any visible text (hidden reasoning) — fail over,
1217
+ // but skip the cooldown/penalty: not a provider-health signal.
1218
+ throw Object.assign(new Error(`empty completion from ${route.displayName} (legacy stream produced no text)`), upstreamFinish === 'length' ? { skipBench: true } : {});
1219
+ }
1220
+ flushHeaders();
1221
+ res.write('data: [DONE]\n\n');
1222
+ res.end();
1223
+ recordUpstreamSuccess(route, estimatedInputTokens + totalOutputTokens);
1224
+ traceRouteEvent('Proxy', {
1225
+ event: 'ok',
1226
+ requestId: requestGroupId,
1227
+ attempt,
1228
+ platform: route.platform,
1229
+ model: route.modelId,
1230
+ latencyMs: Date.now() - start,
1231
+ inputTokens: estimatedInputTokens,
1232
+ outputTokens: totalOutputTokens,
1233
+ });
1234
+ logRequest(route.platform, route.modelId, route.keyId, 'success', estimatedInputTokens, totalOutputTokens, Date.now() - start, null, ttfbMs, pinnedModelId, null, 'http');
1235
+ return 'done';
1236
+ }
1237
+ catch (streamErr) {
1238
+ // Client abort mid-stream: the pump's own `if (clientGone) break`
1239
+ // can lose the race against the fetch-signal rejection, so the
1240
+ // abort may surface here instead. Rethrow — the shared loop's
1241
+ // client-abort branch stops the ladder without benching or an
1242
+ // error log row (the socket is gone; nothing to render).
1243
+ if (isClientAbortError(streamErr))
1244
+ throw streamErr;
1245
+ if (headerSent) {
1246
+ console.error(`[Proxy] Mid-stream legacy completion error from ${route.displayName}:`, streamErr.message);
1247
+ const payload = { error: { message: `Provider error (${route.displayName}): stream interrupted`, type: 'stream_error' } };
1248
+ try {
1249
+ res.write(`data: ${JSON.stringify(payload)}\n\n`);
1250
+ }
1251
+ catch { /* socket gone */ }
1252
+ try {
1253
+ res.write('data: [DONE]\n\n');
1254
+ res.end();
1255
+ }
1256
+ catch { /* socket gone */ }
1257
+ traceRouteEvent('Proxy', {
1258
+ event: 'fail',
1259
+ requestId: requestGroupId,
1260
+ attempt,
1261
+ platform: route.platform,
1262
+ model: route.modelId,
1263
+ latencyMs: Date.now() - start,
1264
+ error: sanitizeProviderErrorMessage(streamErr.message),
1265
+ });
1266
+ logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, sanitizeProviderErrorMessage(streamErr.message), ttfbMs, pinnedModelId, null, 'http');
1267
+ return 'committed';
1268
+ }
1269
+ throw streamErr;
1270
+ }
1271
+ }
1272
+ const result = await route.provider.chatCompletion(route.apiKey, dispatchMessages, route.modelId, { temperature, max_tokens, top_p, stop, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
1273
+ const text = completionTextFromChat(result);
1274
+ if (!text) {
1275
+ // finish_reason 'length' = output budget consumed by hidden reasoning
1276
+ // before any visible text: fail over without a cooldown/penalty.
1277
+ throw Object.assign(new Error(`empty completion from ${route.displayName}`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
1278
+ }
1279
+ // #809: a bare "safe"/"unsafe" classification word from a relay is an
1280
+ // upstream filter, not the requested model — fail over like an empty
1281
+ // completion.
1282
+ if (isUpstreamClassificationOutput(text, route.platform)) {
1283
+ throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
1284
+ }
1285
+ // Usage fallback: providers that omit `usage` used to be logged as 0
1286
+ // tokens, silently undercounting analytics and the rate-limit ledger.
1287
+ // Fall back to the same chars/4 estimate the streaming path uses,
1288
+ // including reasoning tokens (thinking models). (#764)
1289
+ const promptTokens = result.usage?.prompt_tokens ?? estimatedInputTokens;
1290
+ const completionTokens = result.usage?.completion_tokens
1291
+ ?? Math.ceil((text.length + completionReasoningText(result).length) / 4);
1292
+ const totalTokens = result.usage?.total_tokens ?? (promptTokens + completionTokens);
1293
+ recordUpstreamSuccess(route, totalTokens);
1294
+ res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
1295
+ setFallbackHeaders(res, attempt, attemptLog);
1296
+ res.json({
1297
+ id: completionIdFromChat(result.id),
1298
+ object: 'text_completion',
1299
+ created: result.created ?? Math.floor(Date.now() / 1000),
1300
+ model: route.modelId,
1301
+ choices: [{
1302
+ text,
1303
+ index: result.choices?.[0]?.index ?? 0,
1304
+ logprobs: null,
1305
+ finish_reason: result.choices?.[0]?.finish_reason ?? 'stop',
1306
+ }],
1307
+ // `usage` is required by the OpenAI completions spec. The fallback
1308
+ // counts computed just above (chars/4 when the provider omits usage,
1309
+ // #764) existed but were dropped here — clients like editor
1310
+ // ghost-text plugins read usage to throttle and saw `undefined`.
1311
+ usage: result.usage ?? {
1312
+ prompt_tokens: promptTokens,
1313
+ completion_tokens: completionTokens,
1314
+ total_tokens: totalTokens,
1315
+ estimated: true,
1316
+ },
1317
+ execution_id: requestGroupId,
1318
+ });
1319
+ traceRouteEvent('Proxy', {
1320
+ event: 'ok',
1321
+ requestId: requestGroupId,
1322
+ attempt,
1323
+ platform: route.platform,
1324
+ model: route.modelId,
1325
+ latencyMs: Date.now() - start,
1326
+ inputTokens: promptTokens,
1327
+ outputTokens: completionTokens,
1328
+ });
1329
+ logRequest(route.platform, route.modelId, route.keyId, 'success', promptTokens, completionTokens, Date.now() - start, null, null, pinnedModelId, null, 'http');
1330
+ return 'done';
1331
+ },
1332
+ logFailure: (route, err, attempt) => {
1333
+ const latency = Date.now() - start;
1334
+ const safeError = sanitizeProviderErrorMessage(err.message);
1335
+ traceRouteEvent('Proxy', {
1336
+ event: 'fail',
1337
+ requestId: requestGroupId,
1338
+ attempt,
1339
+ platform: route.platform,
1340
+ model: route.modelId,
1341
+ latencyMs: latency,
1342
+ error: safeError,
1343
+ });
1344
+ logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, 0, latency, safeError, null, pinnedModelId, null, 'http');
1345
+ },
1346
+ onFatal: (route, err, attempt) => {
1347
+ setFallbackHeaders(res, attempt, attemptLog);
1348
+ res.status(502).json({
1349
+ error: {
1350
+ message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(err.message)}`,
1351
+ type: 'provider_error',
1352
+ },
1353
+ execution_id: requestGroupId,
1354
+ });
1355
+ },
1356
+ onRoutingExhausted: (lastError, routeErr, exhaustion, info) => {
1357
+ setFallbackHeaders(res, info.attempts.length, info.attempts);
1358
+ setExhaustionHeaders(res, exhaustion);
1359
+ res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
1360
+ },
1361
+ onExhausted: (exhaustion, info) => {
1362
+ setFallbackHeaders(res, info.attempts.length, info.attempts);
1363
+ setExhaustionHeaders(res, exhaustion);
1364
+ res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
1365
+ },
1366
+ });
1367
+ });
1368
+ proxyRouter.post('/chat/completions', async (req, res) => {
1369
+ const start = Date.now();
1370
+ const requestGroupId = getRequestGroupId(req);
1371
+ res.setHeader('X-Request-ID', requestGroupId);
1372
+ // Authenticate every proxy request, including loopback callers. Browser
1373
+ // pages can reach localhost, so socket locality is not a reliable
1374
+ // authorization boundary. Client-profile keys resolve here too, carrying
1375
+ // their server-enforced system prompt (#411).
1376
+ const auth = requireInferenceAuth(req, res);
1377
+ if (!auth)
1378
+ return;
1379
+ // Validate request
1380
+ const parsed = chatCompletionSchema.safeParse(req.body);
1381
+ if (!parsed.success) {
1382
+ // Path-qualified issues ("messages.1.content: Invalid input" beats a bare
1383
+ // "Invalid input") and a server-side breadcrumb — these rejections never
1384
+ // reach the request log, which made #200 nearly undebuggable.
1385
+ const detail = parsed.error.errors
1386
+ .map(e => (e.path.length ? `${e.path.join('.')}: ${e.message}` : e.message))
1387
+ .slice(0, 5)
1388
+ .join(', ');
1389
+ console.warn(`[proxy] 400 invalid /chat/completions request: ${detail}`);
1390
+ res.status(400).json({
1391
+ error: {
1392
+ message: `Invalid request: ${detail}`,
1393
+ type: 'invalid_request_error',
1394
+ },
1395
+ execution_id: requestGroupId,
1396
+ });
1397
+ return;
1398
+ }
1399
+ const { model: requestedModel, temperature, top_p, stream } = parsed.data;
1400
+ const requestedModelLabel = requestedModel ?? 'auto';
1401
+ // Agent-tolerant knob normalization (#200): max_tokens <= 0 means "no
1402
+ // limit" in several clients → unset; tool_choice 'any' is OpenAI's
1403
+ // 'required'; tool definitions get their 'function' type re-defaulted.
1404
+ // `max_completion_tokens` is OpenAI's newer alias — honored when max_tokens
1405
+ // itself is absent. `let`: the token-budget guardrail below may cap an
1406
+ // absent max_tokens to the budget remainder before the options objects are
1407
+ // built from it.
1408
+ const requestedMaxTokens = parsed.data.max_tokens ?? parsed.data.max_completion_tokens;
1409
+ let max_tokens = requestedMaxTokens != null && requestedMaxTokens > 0
1410
+ ? requestedMaxTokens : undefined;
1411
+ // Extended sampling/output params (seed, penalties, response_format…),
1412
+ // spread into every options object below — including fusion fan-out.
1413
+ const samplingParams = pickSamplingParams(parsed.data);
1414
+ const stop = providerSafeStop(parsed.data.stop);
1415
+ const tool_choice = parsed.data.tool_choice === 'any' ? 'required' : parsed.data.tool_choice ?? undefined;
1416
+ const tools = parsed.data.tools?.map(t => ({ ...t, type: 'function' }));
1417
+ const parallel_tool_calls = parsed.data.parallel_tool_calls ?? undefined;
1418
+ // Pairing state for id-less tool calls (#200): every tool_call id (given or
1419
+ // synthesized) queues up here; a tool message without a tool_call_id takes
1420
+ // the oldest unanswered one, which matches the single-call-per-turn flow
1421
+ // Gemini-lineage agents produce.
1422
+ const pendingToolCallIds = [];
1423
+ let syntheticIdCounter = 0;
1424
+ const takeToolCallId = (given) => {
1425
+ if (given && given.length > 0) {
1426
+ const qi = pendingToolCallIds.indexOf(given);
1427
+ if (qi !== -1)
1428
+ pendingToolCallIds.splice(qi, 1);
1429
+ return given;
1430
+ }
1431
+ return pendingToolCallIds.shift() ?? `call_auto_${++syntheticIdCounter}`;
1432
+ };
1433
+ let messages = parsed.data.messages.map((m) => {
1434
+ if (m.role === 'assistant') {
1435
+ const hasToolCalls = (m.tool_calls?.length ?? 0) > 0;
1436
+ // With tool_calls, content: null is the correct OpenAI shape — keep it.
1437
+ // Without tool_calls, coerce empty/null content to "" so strict upstreams
1438
+ // don't choke on a null-content assistant turn we just accepted. (#165)
1439
+ const isEmptyContent = m.content == null
1440
+ || (typeof m.content === 'string' && m.content.length === 0)
1441
+ || (Array.isArray(m.content) && m.content.length === 0);
1442
+ const assistantContent = hasToolCalls
1443
+ ? (m.content ?? null)
1444
+ : (isEmptyContent ? '' : m.content);
1445
+ return {
1446
+ role: 'assistant',
1447
+ content: assistantContent,
1448
+ ...(m.name ? { name: m.name } : {}),
1449
+ // Replay the thinking trace verbatim. DeepSeek thinking models on
1450
+ // OpenCode Zen reject a follow-up turn that drops it; other providers
1451
+ // ignore the unknown field. Same round-trip rationale as
1452
+ // thought_signature below. (#255)
1453
+ ...(typeof m.reasoning_content === 'string' && m.reasoning_content.length > 0
1454
+ ? { reasoning_content: m.reasoning_content }
1455
+ : {}),
1456
+ // Moonshot's "partial" prefill flag: keep it through the message build
1457
+ // (the schema already preserves it); the provider layer decides whether
1458
+ // the routed model understands it and strips it otherwise. (#1038)
1459
+ ...(m.partial === true ? { partial: true } : {}),
1460
+ // hasToolCalls (not a bare truthiness check) so null AND empty-array
1461
+ // tool_calls are dropped rather than forwarded — strict upstreams
1462
+ // reject both shapes. (#200)
1463
+ ...(hasToolCalls ? { tool_calls: m.tool_calls.map(tc => {
1464
+ // Normalize echo-tolerant inputs back to the strict OpenAI shape
1465
+ // before forwarding (see toolCallSchema); synthesize missing ids
1466
+ // and queue every id for order-based tool-result pairing. (#200)
1467
+ const id = tc.id && tc.id.length > 0 ? tc.id : `call_auto_${++syntheticIdCounter}`;
1468
+ pendingToolCallIds.push(id);
1469
+ return {
1470
+ id,
1471
+ type: 'function',
1472
+ function: { name: tc.function.name, arguments: toolCallArgsToString(tc.function.arguments) },
1473
+ thought_signature: tc.thought_signature,
1474
+ };
1475
+ }) } : {}),
1476
+ };
1477
+ }
1478
+ if (m.role === 'tool') {
1479
+ return {
1480
+ role: 'tool',
1481
+ // Null/missing content (a tool that returned nothing) → "". (#200)
1482
+ content: m.content ?? '',
1483
+ tool_call_id: takeToolCallId(m.tool_call_id),
1484
+ ...(m.name ? { name: m.name } : {}),
1485
+ };
1486
+ }
1487
+ // Legacy function-calling result → forward as a tool message, paired by
1488
+ // order like an id-less tool message. (#200)
1489
+ if (m.role === 'function') {
1490
+ return {
1491
+ role: 'tool',
1492
+ content: m.content ?? '',
1493
+ tool_call_id: takeToolCallId(undefined),
1494
+ name: m.name,
1495
+ };
1496
+ }
1497
+ return {
1498
+ // 'developer' is OpenAI's newer name for the system role — providers
1499
+ // downstream only know 'system'. (#200)
1500
+ role: m.role === 'developer' ? 'system' : m.role,
1501
+ content: m.content,
1502
+ ...(m.name ? { name: m.name } : {}),
1503
+ };
1504
+ });
1505
+ let cacheControlPrefixLength = 0;
1506
+ parsed.data.messages.forEach((message, index) => {
1507
+ const content = message.content;
1508
+ if (Array.isArray(content)
1509
+ && content.some(block => block && typeof block === 'object' && 'cache_control' in block)) {
1510
+ cacheControlPrefixLength = index + 1;
1511
+ }
1512
+ });
1513
+ const compressionResult = compressRequest(messages, {
1514
+ header: req.headers['x-freellm-compress'],
1515
+ tools,
1516
+ cacheControlPrefixLength,
1517
+ });
1518
+ messages = compressionResult.messages;
1519
+ res.setHeader('X-FreeLLM-Compress', formatCompressionHeader(compressionResult));
1520
+ // Server-enforced system prompt (#411): injected AFTER compression so it is
1521
+ // never compressed away, and FIRST in the list so a caller-supplied system
1522
+ // message follows it and cannot override it. Constant per profile, so the
1523
+ // provider-side cache prefix stays stable across requests. Neutral no-op for
1524
+ // the unified key and for profiles without a prompt.
1525
+ messages = prependSystemPrompt(messages, auth.systemPrompt);
1526
+ // Downscale over-threshold inline images before estimation/routing so the
1527
+ // token budget, payload limits, and upstream transfer all see the shrunk
1528
+ // bytes (see lib/image-normalize.ts). Mutates the image blocks in place.
1529
+ await normalizeMessageImages(messages);
1530
+ // Token estimation is intentionally a heuristic (~4 chars per token). Used
1531
+ // for routing decisions (skip a model whose budget is too small) and for
1532
+ // streaming bookkeeping where the provider doesn't echo a final usage count.
1533
+ // Non-streaming requests reconcile against the provider's real `usage` block;
1534
+ // streaming does the same when stream_options.include_usage produces a final
1535
+ // usage frame, and otherwise falls back to this estimate.
1536
+ const estimatedInputTokens = estimateInputTokens(messages, tools);
1537
+ // Image requests must route to a vision-capable model. Reject up front with a
1538
+ // clear message when none is enabled, rather than silently dropping the image
1539
+ // or surfacing the generic "all models exhausted" error (#118, #125). Add a
1540
+ // rough per-image token cost so budget routing isn't skewed by content the
1541
+ // heuristic above (text-only) can't see.
1542
+ const hasImage = messageHasImage(messages);
1543
+ if (hasImage && !hasEnabledVisionModel()) {
1544
+ res.status(422).json({
1545
+ error: {
1546
+ message: 'This request includes an image, but no vision-capable model is enabled. Enable a vision model (e.g. Gemini 2.5 Flash, Llama 4 Scout) in the Fallback Chain.',
1547
+ type: 'invalid_request_error',
1548
+ code: 'no_vision_model',
1549
+ },
1550
+ execution_id: requestGroupId,
1551
+ });
1552
+ return;
1553
+ }
1554
+ const IMAGE_TOKEN_ESTIMATE = 1000;
1555
+ const imageCount = messages.reduce((n, m) => n + (Array.isArray(m.content) ? m.content.filter(b => b?.type === 'image_url' || b?.type === 'image').length : 0), 0);
1556
+ // The reserved output is capped (routingReserveTokens, #470) so an oversized
1557
+ // client max_tokens can't starve routing; input + images count in full. The
1558
+ // reserve is threaded to the router separately: it is exact and must not be
1559
+ // inflated by the context-window safety margin (#956 review).
1560
+ const outputReserve = routingReserveTokens(max_tokens);
1561
+ const estimatedTotal = estimatedInputTokens + imageCount * IMAGE_TOKEN_ESTIMATE + outputReserve;
1562
+ // Tool-bearing requests must route to a model that emits STRUCTURED
1563
+ // tool_calls. A model without real function-calling support serializes the
1564
+ // call into its text answer — the request "succeeds" but the client's tool
1565
+ // loop sees nothing, which is strictly worse than an error. Same up-front
1566
+ // gate pattern as vision above.
1567
+ const wantsTools = (tools?.length ?? 0) > 0;
1568
+ if (wantsTools && !hasEnabledToolsModel()) {
1569
+ res.status(422).json({
1570
+ error: {
1571
+ message: 'This request includes tools, but no tool-capable model is enabled. Enable a tool-calling model (e.g. GPT-OSS 120B, Gemini 3.5 Flash, GLM-4.7) in the Fallback Chain.',
1572
+ type: 'invalid_request_error',
1573
+ code: 'no_tools_model',
1574
+ },
1575
+ execution_id: requestGroupId,
1576
+ });
1577
+ return;
1578
+ }
1579
+ // Guardrail: per-request token budget (request_max_tokens_budget, default
1580
+ // off). Estimated input (incl. images) + requested output must fit the
1581
+ // ceiling; a request with no max_tokens gets its output capped to the
1582
+ // remainder instead. Sits before the Fusion branch so fan-out inherits the
1583
+ // capped max_tokens too.
1584
+ const budgetCheck = applyTokenBudget(estimatedInputTokens + imageCount * IMAGE_TOKEN_ESTIMATE, max_tokens);
1585
+ if (budgetCheck.rejection) {
1586
+ res.status(413).json({
1587
+ error: { message: tokenBudgetMessage(budgetCheck.rejection), type: 'invalid_request_error', code: 'request_token_budget' },
1588
+ execution_id: requestGroupId,
1589
+ });
1590
+ return;
1591
+ }
1592
+ max_tokens = budgetCheck.maxTokens;
1593
+ // Client-disconnect fan-out: the flag stops the loop before the NEXT
1594
+ // attempt; the AbortController (threaded to the provider as
1595
+ // CompletionOptions.signal) additionally cancels the IN-FLIGHT upstream
1596
+ // fetch and any body/stream read, so tokens stop burning and the in-flight
1597
+ // lease frees immediately. 'close' also fires on normal completion —
1598
+ // writableEnded distinguishes a real disconnect.
1599
+ //
1600
+ // This controller is declared before the Fusion branch because Fusion uses
1601
+ // the same cancellation signal as the ordinary fallback loop below.
1602
+ let clientGone = false;
1603
+ const clientAbort = new AbortController();
1604
+ res.on('close', () => {
1605
+ if (!res.writableEnded) {
1606
+ clientGone = true;
1607
+ clientAbort.abort(newClientAbortError());
1608
+ }
1609
+ });
1610
+ // ── Fusion: multi-model synthesis ──────────────────────────────────────────
1611
+ // The virtual "fusion" model fans the prompt out to a panel of diverse models
1612
+ // in parallel, then a judge synthesizes one answer. It routes each panel/judge
1613
+ // sub-call through the normal path (cooldowns, quotas, analytics), so it
1614
+ // behaves like a normal model from the client's side — just K+1x the tokens.
1615
+ // Image requests run on vision-capable panel members; tool requests run on
1616
+ // tool-capable members and return the first structured tool call directly.
1617
+ if (isFusionModel(requestedModel)) {
1618
+ // Keep Fusion's sequential tool fallback tied to the caller's socket. The
1619
+ // Responses route already threads this signal; the legacy Chat surface
1620
+ // needs the same cancellation so a disconnected client does not leave a
1621
+ // bounded-but-expensive provider call running in the background.
1622
+ const fusionOptions = { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, ...samplingParams, signal: clientAbort.signal };
1623
+ const fusionConfig = parsed.data.fusion ?? {};
1624
+ if (stream) {
1625
+ // Streaming fusion: open the SSE response immediately and emit additive
1626
+ // `_fusion` frames (no `choices`, so standard OpenAI clients skip them) as
1627
+ // each panel model settles and when the judge runs — the Playground shows
1628
+ // these arriving in a collapsible trace. The final synthesized answer is
1629
+ // then streamed as normal content deltas, so plain clients still get it.
1630
+ res.setHeader('Content-Type', 'text/event-stream');
1631
+ res.setHeader('Cache-Control', 'no-cache');
1632
+ res.setHeader('Connection', 'keep-alive');
1633
+ const writeFrame = (o) => { try {
1634
+ res.write(`data: ${JSON.stringify(o)}\n\n`);
1635
+ }
1636
+ catch { /* socket gone */ } };
1637
+ const streamId = `fusion-${Date.now()}-${crypto.randomBytes(3).toString('hex')}`;
1638
+ const base = { id: streamId, object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: FUSION_MODEL_ID };
1639
+ // Track whether the judge already streamed content so we don't re-emit it.
1640
+ let answerStarted = false;
1641
+ try {
1642
+ const { response } = await runFusion({
1643
+ messages,
1644
+ config: fusionConfig,
1645
+ options: fusionOptions,
1646
+ estimatedTokens: estimatedTotal,
1647
+ vision: hasImage,
1648
+ hooks: {
1649
+ // `a` already carries a sanitized error for failed slots; content is
1650
+ // the model's own answer and is forwarded as-is.
1651
+ onPanel: (a) => writeFrame({
1652
+ ...base,
1653
+ choices: [{ index: 0, delta: {}, finish_reason: null }],
1654
+ _fusion: { event: 'panel', ...a },
1655
+ }),
1656
+ onJudge: (j) => writeFrame({
1657
+ ...base,
1658
+ choices: [{ index: 0, delta: {}, finish_reason: null }],
1659
+ _fusion: { event: 'judge', ...j },
1660
+ }),
1661
+ // Stream the judge's synthesis live as standard content deltas, so
1662
+ // the final answer appears as it's written instead of after the wait.
1663
+ onJudgeDelta: (delta) => {
1664
+ if (!answerStarted) {
1665
+ writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
1666
+ answerStarted = true;
1667
+ }
1668
+ writeFrame({ ...base, choices: [{ index: 0, delta: { content: delta }, finish_reason: null }] });
1669
+ },
1670
+ },
1671
+ });
1672
+ // best_of / single-survivor / judge-fell-back-to-best-of never streamed
1673
+ // a delta — emit the final answer as one chunk in that case.
1674
+ const finalMsg = response.choices[0]?.message;
1675
+ const finalToolCalls = finalMsg?.tool_calls;
1676
+ const hasFinalToolCalls = Array.isArray(finalToolCalls) && finalToolCalls.length > 0;
1677
+ if (hasFinalToolCalls) {
1678
+ writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
1679
+ writeFrame({ ...base, choices: [{ index: 0, delta: { tool_calls: finalToolCalls }, finish_reason: null }] });
1680
+ writeFrame({ ...base, choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }], usage: response.usage });
1681
+ }
1682
+ else {
1683
+ if (!answerStarted) {
1684
+ const finalText = contentToString(finalMsg?.content ?? '');
1685
+ writeFrame({ ...base, choices: [{ index: 0, delta: { role: 'assistant' }, finish_reason: null }] });
1686
+ writeFrame({ ...base, choices: [{ index: 0, delta: { content: finalText }, finish_reason: null }] });
1687
+ }
1688
+ writeFrame({ ...base, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }], usage: response.usage });
1689
+ }
1690
+ }
1691
+ catch (err) {
1692
+ const message = err instanceof FusionError ? err.message : `fusion error: ${sanitizeProviderErrorMessage(err?.message)}`;
1693
+ const type = err instanceof FusionError
1694
+ ? (err.status === 429 ? 'rate_limit_error' : err.status >= 500 ? 'server_error' : 'invalid_request_error')
1695
+ : 'server_error';
1696
+ writeFrame({ error: { message, type } });
1697
+ }
1698
+ try {
1699
+ res.write('data: [DONE]\n\n');
1700
+ res.end();
1701
+ }
1702
+ catch { /* socket gone */ }
1703
+ return;
1704
+ }
1705
+ try {
1706
+ const { response, routedVia } = await runFusion({
1707
+ messages,
1708
+ config: fusionConfig,
1709
+ options: fusionOptions,
1710
+ estimatedTokens: estimatedTotal,
1711
+ vision: hasImage,
1712
+ });
1713
+ // Structured-output enforcement for fusion (#516 scope gap): the panel/
1714
+ // judge output got no format check, so model:"fusion" could hand back
1715
+ // prose as a "success" for a json_schema request. Fusion has no failover
1716
+ // machinery to hand this to — heal what's healable, otherwise answer
1717
+ // honestly instead of pretending. (Streaming fusion stays unenforced,
1718
+ // same boundary as every other streamed response.)
1719
+ const fusionMsg = response?.choices?.[0]?.message;
1720
+ if (samplingParams.response_format && fusionMsg && !fusionMsg.tool_calls?.length) {
1721
+ const fusionText = contentToString(fusionMsg.content ?? '');
1722
+ if (fusionText) {
1723
+ const enforced = enforceJsonContent(fusionText);
1724
+ if (!enforced.ok) {
1725
+ res.status(502).json({ error: { message: `fusion produced non-JSON output despite response_format=${samplingParams.response_format.type} — retry, or pin a structured-output-capable model instead of "fusion"`, type: 'server_error' }, execution_id: requestGroupId });
1726
+ return;
1727
+ }
1728
+ if (enforced.healed)
1729
+ fusionMsg.content = enforced.content;
1730
+ }
1731
+ }
1732
+ res.setHeader('X-Routed-Via', safeHeaderValue(routedVia));
1733
+ res.json({ ...response, execution_id: requestGroupId });
1734
+ }
1735
+ catch (err) {
1736
+ if (err instanceof FusionError) {
1737
+ res.status(err.status).json({
1738
+ error: {
1739
+ message: err.message,
1740
+ type: err.status === 429 ? 'rate_limit_error' : err.status >= 500 ? 'server_error' : 'invalid_request_error',
1741
+ },
1742
+ execution_id: requestGroupId,
1743
+ });
1744
+ }
1745
+ else {
1746
+ res.status(502).json({ error: { message: `fusion error: ${sanitizeProviderErrorMessage(err?.message)}`, type: 'server_error' }, execution_id: requestGroupId });
1747
+ }
1748
+ }
1749
+ return;
1750
+ }
1751
+ // ── Response cache (services/cache.ts) ──
1752
+ // Opt-in exact-match cache. An identical earlier request is replayed from an
1753
+ // in-memory LRU without spending any provider quota. Computed here, after
1754
+ // message + sampling-param normalization but before any routing/session work,
1755
+ // so a hit short-circuits the whole pipeline. Only NON-streaming requests at a
1756
+ // cacheable temperature are eligible (v1 scope: streaming always bypasses); a
1757
+ // per-request `X-FreeLLM-Cache` header can force or bypass. Off unless enabled
1758
+ // via the RESPONSE_CACHE env var or the response_cache_enabled setting.
1759
+ const cacheDirective = parseCacheDirective(req.headers['x-freellm-cache'], req.headers['cache-control']);
1760
+ // Streaming requests participate in the cache too: same canonical key (the
1761
+ // request content is identical), but hits are looked up in the streaming
1762
+ // store and replayed as SSE rather than JSON (see below).
1763
+ const cacheKey = (cacheActive(cacheDirective) && isCacheableTemperature(temperature))
1764
+ ? computeCacheKey({
1765
+ model: requestedModel, messages, temperature, top_p, max_tokens, tools, tool_choice,
1766
+ // Normalized stop (providerSafeStop), i.e. what is actually forwarded.
1767
+ stop,
1768
+ // The knobs below are NOT in chatCompletionSchema, so zod strips them
1769
+ // from parsed.data; read them from the raw body. They still change what
1770
+ // answer the client is asking for, so requests differing only in one of
1771
+ // them must never collide on a cached entry. Explicit null is coerced
1772
+ // to undefined (dropped from the key) to match how the proxy treats
1773
+ // null-valued optional knobs as absent.
1774
+ response_format: req.body?.response_format ?? undefined,
1775
+ n: req.body?.n ?? undefined,
1776
+ seed: req.body?.seed ?? undefined,
1777
+ presence_penalty: req.body?.presence_penalty ?? undefined,
1778
+ frequency_penalty: req.body?.frequency_penalty ?? undefined,
1779
+ logit_bias: req.body?.logit_bias ?? undefined,
1780
+ logprobs: req.body?.logprobs ?? undefined,
1781
+ top_logprobs: req.body?.top_logprobs ?? undefined,
1782
+ // Normalized reasoning knob (flat field or object form) — a different
1783
+ // effort asks for a different answer, so it must never collide.
1784
+ reasoning_effort: samplingParams.reasoning_effort ?? undefined,
1785
+ compression: compressionResult.cacheKey,
1786
+ })
1787
+ : null;
1788
+ if (cacheKey) {
1789
+ if (stream) {
1790
+ // Streaming hit: replay the captured SSE frame sequence verbatim —
1791
+ // same zero-quota rationale as a JSON hit, with first-byte semantics
1792
+ // preserved (the frames include the leading content chunks, so the
1793
+ // first token arrives immediately).
1794
+ const streamHit = getCachedStreamResponse(cacheKey);
1795
+ if (streamHit) {
1796
+ res.setHeader('Content-Type', 'text/event-stream');
1797
+ res.setHeader('Cache-Control', 'no-cache');
1798
+ res.setHeader('Connection', 'keep-alive');
1799
+ res.setHeader('X-Routed-Via', 'cache');
1800
+ res.setHeader('X-FreeLLM-Cache', 'HIT');
1801
+ res.write(streamHit.sse);
1802
+ res.end();
1803
+ return;
1804
+ }
1805
+ }
1806
+ else {
1807
+ const hit = getCachedResponse(cacheKey);
1808
+ if (hit) {
1809
+ // A hit consumes NO provider quota, so recordRequest/recordTokens are
1810
+ // deliberately skipped and the reply is not re-logged as provider usage.
1811
+ // The savings are reported separately by GET /api/cache/stats.
1812
+ res.setHeader('X-Routed-Via', 'cache');
1813
+ res.setHeader('X-FreeLLM-Cache', 'HIT');
1814
+ res.json(withExecutionId(hit.body, requestGroupId));
1815
+ return;
1816
+ }
1817
+ }
1818
+ }
1819
+ // ── Idempotency-Key (services/idempotency.ts) ──
1820
+ // Optional caller-scoped dedup for NON-streaming requests: a client that
1821
+ // times out and retries with the same Idempotency-Key gets the ORIGINAL
1822
+ // response replayed (zero provider cost) instead of burning a second
1823
+ // free-tier slot. Only a SHA-256 hash of the key is stored. Reusing a key
1824
+ // with different request content is a 409 conflict. Streaming always
1825
+ // bypasses (like the response cache) — a stream cannot be replayed as a
1826
+ // unit, and the open connection is itself the retry signal.
1827
+ const idemKeyRaw = req.headers['idempotency-key'] ?? req.headers['Idempotency-Key'];
1828
+ const idemKey = !stream ? normalizeIdempotencyKey(idemKeyRaw) : null;
1829
+ const idemFingerprint = idemKey
1830
+ ? computeIdempotencyFingerprint({
1831
+ model: requestedModel,
1832
+ messages,
1833
+ temperature,
1834
+ top_p,
1835
+ max_tokens,
1836
+ tools,
1837
+ tool_choice,
1838
+ })
1839
+ : null;
1840
+ if (idemKey && idemFingerprint) {
1841
+ const keyHash = hashIdempotencyKey(idemKey);
1842
+ const claim = lookupIdempotencyReplay(keyHash, idemFingerprint);
1843
+ if (claim.kind === 'replay') {
1844
+ // Replay consumes NO provider quota — same zero-cost rationale as a
1845
+ // cache hit, so request/usage bookkeeping is skipped here too.
1846
+ res.setHeader('X-Routed-Via', 'idempotency');
1847
+ res.status(claim.status).json(withExecutionId(claim.body, requestGroupId));
1848
+ return;
1849
+ }
1850
+ if (claim.kind === 'conflict') {
1851
+ res.status(409).json({
1852
+ error: {
1853
+ message: 'idempotency_key_conflict',
1854
+ type: 'invalid_request_error',
1855
+ },
1856
+ execution_id: requestGroupId,
1857
+ });
1858
+ return;
1859
+ }
1860
+ // kind === 'miss': no prior claim (or it expired) — proceed normally
1861
+ // and persist the result on success below.
1862
+ }
1863
+ // Optional client-managed session affinity (see getSessionKey). Express
1864
+ // lower-cases header names; a repeated header arrives as an array — take
1865
+ // the first value.
1866
+ const rawSessionId = req.headers['x-session-id'];
1867
+ const sessionIdHeader = Array.isArray(rawSessionId) ? rawSessionId[0] : rawSessionId;
1868
+ let resolvedChain;
1869
+ let strategyKey;
1870
+ if (isAutoModel(requestedModel)) {
1871
+ resolvedChain = resolveRoutingChain(requestedModel);
1872
+ strategyKey = resolvedChain.strategyKey;
1873
+ }
1874
+ // Context handoff only applies to auto-routed requests. Pinned-model requests
1875
+ // are deliberate client choices; injecting "you are taking over" there would
1876
+ // be semantically wrong.
1877
+ const isAutoRouted = !requestedModel || isAutoModel(requestedModel);
1878
+ const handoffMode = isAutoRouted ? getContextHandoffMode() : 'off';
1879
+ const sessionKey = handoffMode !== 'off' ? getSessionKey(messages, sessionIdHeader, strategyKey) : '';
1880
+ if (handoffMode !== 'off' && sessionKey) {
1881
+ recordIncomingMessages(sessionKey, messages);
1882
+ }
1883
+ // #797: key for the per-session thinking-trace memory. Read and written
1884
+ // inside the dispatch loop, where the routed platform/model is known — the
1885
+ // restore only ever touches the OUTBOUND copy, never `messages`, so handoff
1886
+ // recording above, request logging, compression and the response cache all
1887
+ // keep seeing exactly what the client sent.
1888
+ //
1889
+ // Header-less clients are covered: without x-session-id, getSessionKey hashes
1890
+ // the FIRST user message, which does not change as the conversation grows —
1891
+ // the same stability sticky sessions already rely on. Its known weakness is
1892
+ // shared here: two conversations opening with identical text share a key, so
1893
+ // the same-model + field-actually-missing gates below are what keep a
1894
+ // mis-keyed restore from reaching a payload it does not belong in.
1895
+ const reasoningSessionKey = getSessionKey(messages, sessionIdHeader, strategyKey);
1896
+ // A handoff can only fire when a prior model is on record for this session.
1897
+ // Check after recordIncomingMessages, which clears the prior model on a
1898
+ // fresh conversation. Stable across the retry loop (the prior model only
1899
+ // changes on a success, which returns), so compute it once here.
1900
+ const handoffPossible = handoffMode !== 'off' && !!sessionKey && hasPriorModel(sessionKey);
1901
+ // Explicit `model` field pins routing. If the catalog has no enabled row
1902
+ // matching the requested id, return 400 — silently auto-routing to a
1903
+ // different model would be surprising to OpenAI-compatible clients.
1904
+ // Sticky-session is the fallback when no `model` field was sent at all.
1905
+ let preferredModel;
1906
+ // When the pinned model is a unified group, this holds the group's ordered
1907
+ // members and is passed to routeRequest as the STRICT chain (no other model
1908
+ // is ever reached). Undefined for auto and legacy single-row pins.
1909
+ let groupChain;
1910
+ // Sticky scope: auto requests bucket by routing strategy; a unified group pin
1911
+ // buckets by the canonical id the client sent, so the group prefers its last
1912
+ // successful provider without leaking stickiness across groups.
1913
+ let stickyStrategyKey = strategyKey;
1914
+ if (isAutoModel(requestedModel)) {
1915
+ preferredModel = resolveStickyPreference(getStickyModel(messages, sessionIdHeader, strategyKey), resolvedChain?.chain);
1916
+ }
1917
+ else if (requestedModel) {
1918
+ const db = getDb();
1919
+ // Unify ON: a requested id (canonical slug OR any provider's model_id) maps
1920
+ // to the whole logical-model group, and we route STRICTLY across only its
1921
+ // providers — failing over between them, never to a different model (#335).
1922
+ const resolved = isUnifyEnabled() ? resolveRequestedIdForDispatch(requestedModel, getModelGroups()) : null;
1923
+ const members = resolved?.memberDbIds ?? null;
1924
+ if (members && members.length > 0) {
1925
+ groupChain = resolveModelGroupCandidates(members, resolved.demotedDbIds);
1926
+ if (groupChain.length === 0) {
1927
+ // Distinguish a catalog-disabled model (404 model_not_found, OpenAI
1928
+ // semantics) from one whose providers are present but unusable
1929
+ // (chain-disabled / no key) — the latter is a server-side
1930
+ // configuration gap, so it renders an honest 503.
1931
+ const placeholders = members.map(() => '?').join(',');
1932
+ const anyEnabled = db.prepare(`SELECT 1 FROM models WHERE id IN (${placeholders}) AND enabled = 1 LIMIT 1`).get(...members);
1933
+ if (anyEnabled) {
1934
+ res.status(503).json({
1935
+ error: {
1936
+ message: `Model '${requestedModel}' has no providers with an enabled key. Add a provider API key for it, use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1937
+ type: 'service_unavailable',
1938
+ code: 'no_providers_configured',
1939
+ },
1940
+ execution_id: requestGroupId,
1941
+ });
1942
+ }
1943
+ else {
1944
+ res.status(404).json({
1945
+ error: {
1946
+ message: `Model '${requestedModel}' is disabled. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1947
+ type: 'invalid_request_error',
1948
+ code: 'model_not_found',
1949
+ },
1950
+ execution_id: requestGroupId,
1951
+ });
1952
+ }
1953
+ return;
1954
+ }
1955
+ stickyStrategyKey = requestedModel;
1956
+ const sticky = getStickyModel(messages, sessionIdHeader, stickyStrategyKey);
1957
+ // Only prefer the sticky member if it's actually IN this group — passing a
1958
+ // non-member as preferredModelDbId would make routeRequest inject an
1959
+ // off-group model and break strict pinning.
1960
+ preferredModel = (sticky != null && groupChain.some(r => r.model_db_id === sticky)) ? sticky : undefined;
1961
+ }
1962
+ else {
1963
+ // Unify OFF, or an id that isn't in the catalog: legacy single-row pin.
1964
+ const enabled = db.prepare('SELECT id FROM models WHERE model_id = ? AND enabled = 1').get(requestedModel);
1965
+ if (enabled) {
1966
+ preferredModel = enabled.id;
1967
+ }
1968
+ else {
1969
+ const disabled = db.prepare('SELECT id FROM models WHERE model_id = ?').get(requestedModel);
1970
+ const reason = disabled ? 'is disabled' : 'is not in the catalog';
1971
+ res.status(404).json({
1972
+ error: {
1973
+ message: `Model '${requestedModel}' ${reason}. Use 'auto' (or omit the 'model' field) to auto-route, or call /v1/models for the available list.`,
1974
+ type: 'invalid_request_error',
1975
+ code: 'model_not_found',
1976
+ },
1977
+ execution_id: requestGroupId,
1978
+ });
1979
+ return;
1980
+ }
1981
+ }
1982
+ }
1983
+ else {
1984
+ preferredModel = resolveStickyPreference(getStickyModel(messages, sessionIdHeader, strategyKey), resolvedChain?.chain);
1985
+ }
1986
+ // For analytics: the model id the client pinned, null when auto-routed
1987
+ // ('auto' or omitted). Logged with every request row so pinned vs auto
1988
+ // traffic and failover overrides are visible.
1989
+ const pinnedModelId = requestedModel && !isAutoModel(requestedModel) ? requestedModel : null;
1990
+ // Retry loop: on 429/rate limit, skip that model+key and try the next one.
1991
+ // The attempt iteration, cooldown/skip/penalty bookkeeping, and exhaustion
1992
+ // rendering are the shared fallback loop (lib/fallback-loop.ts). What stays
1993
+ // here is /chat/completions-specific: the response-cache MISS store, the
1994
+ // context-handoff injection, group/unified-chain routing, and the OpenAI
1995
+ // stream turn-integrity framing.
1996
+ const state = newFallbackState();
1997
+ // Lets the failover loop learn which models reject tool calls (#1230).
1998
+ state.wantsTools = wantsTools;
1999
+ const attemptLog = [];
2000
+ // Fallback-v2 hedging: the loop aborts this controller (via abortInFlight)
2001
+ // when the wall-clock retry budget expires mid-attempt, canceling the
2002
+ // in-flight upstream instead of waiting for a stalled attempt to time out.
2003
+ const hedgeAbort = new AbortController();
2004
+ await runFallbackLoop({
2005
+ maxRetries: MAX_RETRIES,
2006
+ state,
2007
+ attemptLog,
2008
+ logIdentity: { surface: 'chat completions', requestId: requestGroupId, requestedModel: requestedModelLabel },
2009
+ clientGone: () => clientGone,
2010
+ abortInFlight: () => hedgeAbort.abort(newHedgeAbortError()),
2011
+ route: () => {
2012
+ // When a handoff could fire this turn, pad the token estimate so the router's
2013
+ // context-window and TPM checks account for the extra system message overhead.
2014
+ // We don't know the selected model key until after routeRequest() returns, so
2015
+ // the padding is conservative on turns where injection is *possible* (a prior
2016
+ // model is on record). Turns where injection can't happen — every turn 1, and
2017
+ // sessions that never switched — pay no headroom tax.
2018
+ const routingEstimate = fallbackRoutingTokens(state, estimatedTotal, outputReserve) + (handoffPossible ? HANDOFF_MAX_TOKENS : 0);
2019
+ // Task-type routing (#1127): the client can declare code/chat intent via
2020
+ // header; otherwise a bounded rule derives it (tools present / code
2021
+ // markers). undefined keeps the preset weights untouched.
2022
+ const taskType = resolveTaskType(req, tools, messages);
2023
+ return routeRequest(routingEstimate, state.skipKeys.size > 0 ? state.skipKeys : undefined, preferredModel, hasImage, wantsTools, state.skipModels.size > 0 ? state.skipModels : undefined, groupChain ?? resolvedChain?.chain, samplingParams.response_format !== undefined, state.skipPlatforms.size > 0 ? state.skipPlatforms : undefined, outputReserve, taskType);
2024
+ },
2025
+ dispatch: async (route, attempt, ctx) => {
2026
+ const contextBudget = routeOutputBudget(route, estimatedInputTokens);
2027
+ const modelKey = `${route.platform}:${route.modelId}`;
2028
+ traceRouteEvent('Proxy', {
2029
+ event: attempt === 0 ? 'start' : 'next',
2030
+ requestId: requestGroupId,
2031
+ attempt,
2032
+ platform: route.platform,
2033
+ model: route.modelId,
2034
+ requestedModel: attempt === 0 ? requestedModelLabel : undefined,
2035
+ });
2036
+ let outboundMessages = messages;
2037
+ // #797: thinking trace accumulated from this turn's streamed deltas, then
2038
+ // remembered per-session so a follow-up whose client stripped the field
2039
+ // can have it restored (see restore block above).
2040
+ let streamReasoning = '';
2041
+ // Extra input tokens the injected handoff adds on this turn (0 when not
2042
+ // injected). Folded into the streaming success accounting, where token
2043
+ // counts are estimated; the non-stream path uses the provider's usage,
2044
+ // which already counts the injected message.
2045
+ let injectedHandoffTokens = 0;
2046
+ if (handoffMode !== 'off' && sessionKey) {
2047
+ const handoff = maybeInjectContextHandoff({ mode: handoffMode, sessionKey, messages, selectedModelKey: modelKey });
2048
+ if (handoff.injected)
2049
+ console.log(`[Proxy] Context handoff injected (session ${sessionKey.slice(0, 8)}…, model switch detected)`);
2050
+ outboundMessages = handoff.messages;
2051
+ injectedHandoffTokens = handoff.injectedTokens;
2052
+ }
2053
+ // #797: restore the thinking trace this proxy emitted last turn, for THIS
2054
+ // model, when the client dropped it on replay. Scoped to the outbound copy
2055
+ // and to the model that produced the trace, so a failover hop and every
2056
+ // provider that never needed the field send the client's bytes unchanged.
2057
+ const rememberedReasoning = rememberedReasoningFor(reasoningSessionKey, modelKey);
2058
+ if (rememberedReasoning) {
2059
+ outboundMessages = restoreSessionReasoning(outboundMessages, rememberedReasoning, route.platform);
2060
+ }
2061
+ // GitHub Models 413s a history above its input ceiling instead of
2062
+ // truncating it, so a long conversation burns the github hop of every
2063
+ // chain it appears in. Trim the outbound copy to what the platform will
2064
+ // accept — scoped to this attempt, so the next candidate still sees the
2065
+ // client's full history. A no-op returning the same array when the
2066
+ // request already fits.
2067
+ if (route.platform === 'github') {
2068
+ outboundMessages = truncateMessagesForGithub(outboundMessages);
2069
+ }
2070
+ if (stream) {
2071
+ // — Stream turn-integrity (#231 audit) —
2072
+ // The old loop forwarded upstream chunks verbatim and called any
2073
+ // stream that produced bytes a success. Live failure modes that
2074
+ // slipped through: in-band `{"error":...}` frames delivered as dead
2075
+ // turns, tool calls with no terminal finish_reason, inline tool-call
2076
+ // dialect emitted as text, truncations logged as success. This loop
2077
+ // validates the TURN, not the transport:
2078
+ // - headers are held until the first real payload, so anything that
2079
+ // dies before producing one fails over invisibly;
2080
+ // - text that starts with an inline tool-call dialect marker is held
2081
+ // and rescued into structured tool_calls (or failed over);
2082
+ // - tool_call deltas are buffered, argument-repaired, and emitted as
2083
+ // one complete chunk, always followed by finish_reason
2084
+ // "tool_calls" — agents never see calls without a terminal reason;
2085
+ // - a stream that ends with neither content nor calls is an empty
2086
+ // completion and fails over like the non-stream path.
2087
+ let totalOutputTokens = 0;
2088
+ let headerSent = false;
2089
+ let ttfbMs = null;
2090
+ // Hold-window state: 'undecided' until the first text either matches
2091
+ // a dialect marker (→ 'dialect': buffer everything, rescue at end),
2092
+ // carries a structured-output request (→ 'json': buffer everything,
2093
+ // enforce JSON at end) or provably cannot (→ 'passthrough': flush and
2094
+ // stream normally).
2095
+ let mode = 'undecided';
2096
+ let heldText = '';
2097
+ const preamble = []; // role-only chunks held until flush
2098
+ const toolCallAcc = new Map();
2099
+ let upstreamFinish = null;
2100
+ let usageChunk = null;
2101
+ let lastMeta = {};
2102
+ // Raw upstream-reported model, captured off the first frame that
2103
+ // carries one — BEFORE the per-frame overwrite below destroys it.
2104
+ // Only evidence when a provider serves a different model than routed
2105
+ // (#534); compared/persisted on success via observeServedModel.
2106
+ let upstreamModel = null;
2107
+ // Every `data: ...` frame the client sees, captured for a possible
2108
+ // streaming cache store on success (exact SSE replay on a later hit).
2109
+ // Collected ONLY when this request is actually cacheable: with the
2110
+ // cache off — the default — a stream must retain nothing, so the
2111
+ // buffer stays null and every frame is written straight through.
2112
+ const streamFrames = cacheKey ? [] : null;
2113
+ let streamFrameBytes = 0;
2114
+ // Flipped off once the answer outgrows what is worth holding; the
2115
+ // buffer is dropped and this response is not stored.
2116
+ let streamCacheable = cacheKey !== null;
2117
+ const collectFrame = (frame) => {
2118
+ if (!streamCacheable || !streamFrames)
2119
+ return;
2120
+ streamFrameBytes += Buffer.byteLength(frame);
2121
+ if (streamFrameBytes > STREAM_CACHE_MAX_BYTES) {
2122
+ streamCacheable = false;
2123
+ streamFrames.length = 0;
2124
+ return;
2125
+ }
2126
+ streamFrames.push(frame);
2127
+ };
2128
+ const flushHeaders = () => {
2129
+ if (headerSent)
2130
+ return;
2131
+ // #764: backfill only — the pump loop already records ttfb on the
2132
+ // first token (content or reasoning) it sees.
2133
+ if (ttfbMs === null)
2134
+ ttfbMs = Date.now() - start;
2135
+ res.setHeader('Content-Type', 'text/event-stream');
2136
+ res.setHeader('Cache-Control', 'no-cache');
2137
+ res.setHeader('Connection', 'keep-alive');
2138
+ res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
2139
+ res.setHeader('X-FreeLLM-Cache', cacheKey ? 'MISS' : 'OFF');
2140
+ setFallbackHeaders(res, attempt, attemptLog);
2141
+ headerSent = true;
2142
+ // Committed: the answer is on its way, so the retry budget must no
2143
+ // longer cancel this attempt (it could not fail over now anyway).
2144
+ ctx.disarmHedge();
2145
+ for (const p of preamble) {
2146
+ const frame = `data: ${JSON.stringify(p)}\n\n`;
2147
+ collectFrame(frame);
2148
+ res.write(frame);
2149
+ }
2150
+ preamble.length = 0;
2151
+ };
2152
+ const mkChunk = (delta, finish) => ({
2153
+ id: lastMeta.id ?? `chatcmpl-${Date.now()}`,
2154
+ object: 'chat.completion.chunk',
2155
+ created: lastMeta.created ?? Math.floor(Date.now() / 1000),
2156
+ model: lastMeta.model ?? route.modelId,
2157
+ choices: [{ index: 0, delta, finish_reason: finish }],
2158
+ });
2159
+ const writeChunk = (c) => {
2160
+ const frame = `data: ${JSON.stringify(c)}\n\n`;
2161
+ collectFrame(frame);
2162
+ res.write(frame);
2163
+ };
2164
+ try {
2165
+ const gen = route.provider.streamChatCompletion(route.apiKey, outboundMessages, route.modelId, { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, stream_options: parsed.data.stream_options, ...samplingParams, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
2166
+ for await (const chunk of gen) {
2167
+ if (clientGone)
2168
+ break; // client hung up: stop pulling; reader.cancel() aborts upstream
2169
+ // Provider metadata is not authoritative for the public gateway
2170
+ // response. Some OpenAI-compatible providers (notably Reka) return
2171
+ // the literal model name "default" even when a concrete model was
2172
+ // requested. Normalize every streamed frame at the proxy boundary
2173
+ // so clients consistently see the model that was actually routed.
2174
+ const rawChunkModel = chunk.model;
2175
+ if (upstreamModel == null && typeof rawChunkModel === 'string' && rawChunkModel.length > 0) {
2176
+ upstreamModel = rawChunkModel;
2177
+ }
2178
+ const anyChunk = { ...chunk, model: route.modelId };
2179
+ // In-band upstream error frame (observed live: Groq emits
2180
+ // {"error":{...,"code":"tool_use_failed"}} inside a 200 SSE
2181
+ // stream). Before headers: retryable, the next model gets the
2182
+ // request. After: surface an error frame instead of pretending
2183
+ // the turn succeeded.
2184
+ if (anyChunk.error && !anyChunk.choices) {
2185
+ const msg = anyChunk.error.message ?? JSON.stringify(anyChunk.error).slice(0, 200);
2186
+ if (!headerSent)
2187
+ throw new Error(`in-band provider error from ${route.displayName}: ${msg}`);
2188
+ console.error(`[Proxy] In-band error frame from ${route.displayName} mid-stream:`, msg);
2189
+ writeChunk({ error: { message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(String(msg))}`, type: 'stream_error' } });
2190
+ try {
2191
+ res.write('data: [DONE]\n\n');
2192
+ res.end();
2193
+ }
2194
+ catch { /* socket gone */ }
2195
+ traceRouteEvent('Proxy', {
2196
+ event: 'fail',
2197
+ requestId: requestGroupId,
2198
+ attempt,
2199
+ platform: route.platform,
2200
+ model: route.modelId,
2201
+ latencyMs: Date.now() - start,
2202
+ error: sanitizeProviderErrorMessage(String(msg)),
2203
+ });
2204
+ logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, `in-band error frame: ${sanitizeProviderErrorMessage(String(msg))}`, ttfbMs, pinnedModelId, null, 'http');
2205
+ return 'committed';
2206
+ }
2207
+ if (anyChunk.id)
2208
+ lastMeta = { id: anyChunk.id, model: anyChunk.model, created: anyChunk.created };
2209
+ // Usage arrives either on its own frame (OpenAI's
2210
+ // stream_options.include_usage shape) or bundled onto the last
2211
+ // choice-bearing frame — several providers do the latter, and
2212
+ // reading it only off choice-less frames threw their real token
2213
+ // counts away and left accounting on the chars/4 estimate. Capture
2214
+ // it wherever it lands, reduced to a usage-only frame: the frame's
2215
+ // deltas are re-emitted through our own framing below, so holding
2216
+ // the original verbatim would duplicate content (or a finish_reason
2217
+ // the client already saw) when it is written back after our finish
2218
+ // chunk to preserve OpenAI ordering.
2219
+ if (anyChunk.usage)
2220
+ usageChunk = { ...anyChunk, choices: [], usage: anyChunk.usage };
2221
+ const choice = anyChunk.choices?.[0];
2222
+ if (!choice)
2223
+ continue;
2224
+ if (choice.finish_reason)
2225
+ upstreamFinish = choice.finish_reason;
2226
+ // #797: accumulate this turn's thinking trace (native reasoning
2227
+ // deltas; the <think> extractor in base.ts has already normalized
2228
+ // inline tags into reasoning_content) so a follow-up request whose
2229
+ // client stripped it can have it restored from session memory.
2230
+ // Shared with the #764 ttfb/token accounting below.
2231
+ const reasoning = streamReasoningText(anyChunk);
2232
+ if (reasoning.length > 0)
2233
+ streamReasoning += reasoning;
2234
+ // Buffer tool_call deltas — emitted complete + repaired at end.
2235
+ for (const tc of choice.delta?.tool_calls ?? []) {
2236
+ const idx = tc.index ?? 0;
2237
+ if (!toolCallAcc.has(idx))
2238
+ toolCallAcc.set(idx, { id: undefined, name: '', args: '' });
2239
+ const acc = toolCallAcc.get(idx);
2240
+ if (tc.id && !acc.id)
2241
+ acc.id = tc.id;
2242
+ if (tc.function?.name)
2243
+ acc.name += tc.function.name;
2244
+ if (tc.function?.arguments)
2245
+ acc.args += tc.function.arguments;
2246
+ }
2247
+ normalizeOutboundContent(anyChunk);
2248
+ sanitizeResponse(anyChunk);
2249
+ const text = typeof choice.delta?.content === 'string' ? choice.delta.content : '';
2250
+ // #764: ttfb = first token of ANY kind, not just visible content —
2251
+ // reasoning models stream thinking long before the first answer
2252
+ // token, and the old code deferred ttfb until header flush (or left
2253
+ // it NULL on long-thinking turns that never flushed).
2254
+ if (ttfbMs === null && (text.length > 0 || reasoning.length > 0)) {
2255
+ ttfbMs = Date.now() - start;
2256
+ }
2257
+ if (text.length === 0) {
2258
+ // Role preamble / keep-alive: hold until first payload decides
2259
+ // the mode, forward afterwards. tool_calls and finish_reason are
2260
+ // stripped — both are re-emitted complete at the end (OpenRouter
2261
+ // attaches tool_call deltas to chunks that also carry role/
2262
+ // reasoning keys; forwarding them raw would duplicate the call).
2263
+ // #764: thinking-only chunks still consumed tokens — count them.
2264
+ if (reasoning.length > 0)
2265
+ totalOutputTokens += Math.ceil(reasoning.length / 4);
2266
+ if (choice.delta && Object.keys(choice.delta).some(k => k !== 'content' && k !== 'tool_calls' && choice.delta[k] != null)) {
2267
+ // `usage: undefined` (dropped by JSON.stringify): it was held
2268
+ // above and is re-emitted once, after our finish chunk.
2269
+ const cleaned = { ...anyChunk, usage: undefined, choices: [{ ...choice, delta: { ...choice.delta, tool_calls: undefined }, finish_reason: null }] };
2270
+ if (headerSent)
2271
+ writeChunk(cleaned);
2272
+ else
2273
+ preamble.push(cleaned);
2274
+ }
2275
+ continue;
2276
+ }
2277
+ // #764: count reasoning tokens with the same chars/4 estimate so
2278
+ // analytics and rate-limit reflect real consumption of thinking
2279
+ // models (a chunk can carry both reasoning and text).
2280
+ totalOutputTokens += Math.ceil((text.length + reasoning.length) / 4);
2281
+ if (mode === 'passthrough') {
2282
+ // Same rule as the preamble path: usage rides the held frame,
2283
+ // not this one, so the client sees it exactly once and last.
2284
+ writeChunk({ ...anyChunk, usage: undefined, choices: [{ ...choice, delta: { ...choice.delta, tool_calls: undefined }, finish_reason: null }] });
2285
+ continue;
2286
+ }
2287
+ heldText += text;
2288
+ if (mode === 'dialect' || mode === 'json')
2289
+ continue;
2290
+ const probe = heldText.trimStart();
2291
+ if (wantsTools && startsWithDialectMarker(probe)) {
2292
+ mode = 'dialect';
2293
+ }
2294
+ else if (samplingParams.response_format) {
2295
+ // Structured-output request (#933): hold ALL text until the
2296
+ // stream ends, then enforce JSON (mirrors the non-stream check
2297
+ // below). Streaming bytes are already committed once headers
2298
+ // flush, so a model that answers in prose despite the forwarded
2299
+ // response_format must be caught here, before any byte leaves —
2300
+ // the client asked for machine-readable output, not an essay.
2301
+ mode = 'json';
2302
+ }
2303
+ else if (!wantsTools || !couldBecomeDialectMarker(probe) || probe.length > 256) {
2304
+ mode = 'passthrough';
2305
+ flushHeaders();
2306
+ writeChunk(mkChunk({ content: heldText }, null));
2307
+ heldText = '';
2308
+ }
2309
+ // else: still a strict prefix of a marker — keep holding.
2310
+ }
2311
+ // — Stream ended cleanly (provider saw [DONE] or a finish_reason) —
2312
+ // Assemble buffered tool calls: synthesize missing ids, repair
2313
+ // double-encoded arguments against the request's schemas, drop
2314
+ // calls whose args still aren't valid JSON.
2315
+ const schemas = toolSchemaMap(tools);
2316
+ let syntheticStreamIds = 0;
2317
+ const completedCalls = [...toolCallAcc.entries()]
2318
+ .sort((a, b) => a[0] - b[0])
2319
+ .map(([, acc]) => ({
2320
+ id: acc.id && acc.id.length > 0 ? acc.id : `call_stream_${++syntheticStreamIds}`,
2321
+ type: 'function',
2322
+ function: { name: acc.name, arguments: repairToolArguments(acc.args || '{}', schemas.get(acc.name)) },
2323
+ }))
2324
+ .filter(c => { try {
2325
+ JSON.parse(c.function.arguments);
2326
+ return c.function.name.length > 0;
2327
+ }
2328
+ catch {
2329
+ return false;
2330
+ } });
2331
+ // Dialect rescue: the held text is an inline tool call in some
2332
+ // model's private syntax. Parse it into structured calls or treat
2333
+ // the turn as dead (headers were never sent in dialect mode, so
2334
+ // failing over is free).
2335
+ if (wantsTools && (mode === 'dialect' || (mode === 'undecided' && heldText.length > 0 && containsDialectMarker(heldText)))) {
2336
+ const rescue = rescueInlineToolCalls(heldText, new Set((tools ?? []).map(t => t.function.name)));
2337
+ if (rescue.detected) {
2338
+ if (!rescue.calls)
2339
+ throw new Error(`unparseable inline tool-call dialect from ${route.displayName}: ${heldText.slice(0, 120)}`);
2340
+ let rescuedIds = 0;
2341
+ for (const c of rescue.calls) {
2342
+ completedCalls.push({ id: `call_rescued_${++rescuedIds}`, type: 'function', function: { name: c.name, arguments: repairToolArguments(c.arguments, schemas.get(c.name)) } });
2343
+ }
2344
+ heldText = rescue.cleanText;
2345
+ console.log(`[Proxy] Rescued ${rescuedIds} inline tool call(s) from ${route.displayName} into structured tool_calls`);
2346
+ }
2347
+ }
2348
+ // Opt-in schema verdict, taken AFTER the rescue so it covers the
2349
+ // calls the rescue reconstructed from prose — those are the ones
2350
+ // most likely to be malformed, and running first exempted exactly
2351
+ // them. `!headerSent` is the whole licence to throw here: the commit
2352
+ // point is held until the first meaningful content, so the common
2353
+ // tool-call turn (no prose before the call) has sent no bytes yet and
2354
+ // can still fail over invisibly. A turn that already flushed prose is
2355
+ // past the point of no return — the catch below would have to tear
2356
+ // the SSE stream down with a `stream_error`, which is strictly worse
2357
+ // for the client than forwarding a tool call the schema dislikes.
2358
+ // Off-by-default or not, this check must never turn a served answer
2359
+ // into a broken one.
2360
+ if (isToolArgumentValidationEnabled() && !headerSent && completedCalls.length > 0) {
2361
+ const invalid = invalidToolCallReasons(completedCalls, schemas);
2362
+ if (invalid.length > 0)
2363
+ throw invalidToolArgumentsError(route.displayName, invalid);
2364
+ }
2365
+ // Disconnect before the commit point: nothing usable was (or will
2366
+ // be) delivered, and that is CLIENT behavior, not a provider
2367
+ // failure — do not let it fall through to the empty-completion
2368
+ // throw below, which would bench a healthy model+key for 90s and
2369
+ // log a provider error for every Ctrl-C during a reasoning model's
2370
+ // TTFB window.
2371
+ if (clientGone && !headerSent && heldText.trim().length === 0 && completedCalls.length === 0) {
2372
+ console.log(`[Proxy] client disconnected before first token from ${route.displayName} — dropping attempt without benching`);
2373
+ traceRouteEvent('Proxy', {
2374
+ event: 'canceled',
2375
+ requestId: requestGroupId,
2376
+ attempt,
2377
+ platform: route.platform,
2378
+ model: route.modelId,
2379
+ });
2380
+ return 'committed';
2381
+ }
2382
+ const hasText = headerSent || heldText.trim().length > 0;
2383
+ if (!hasText && completedCalls.length === 0) {
2384
+ // Nothing usable came out — same failover semantics as the
2385
+ // non-stream empty-completion path. Headers can't have been sent
2386
+ // (header flush requires payload), so the client never notices.
2387
+ // finish_reason 'length' = the model spent the whole output budget
2388
+ // on hidden reasoning before any visible text: fail over, but skip
2389
+ // the cooldown/penalty (not a provider-health signal).
2390
+ throw Object.assign(new Error(`empty completion from ${route.displayName} (stream produced no content and no tool calls)`), upstreamFinish === 'length' ? { skipBench: true } : {});
2391
+ }
2392
+ // #809: a bare "safe"/"unsafe" classification word streamed by a
2393
+ // relay is an upstream filter, not the requested model — fail over
2394
+ // like an empty completion.
2395
+ if (isUpstreamClassificationOutput(heldText, route.platform) && completedCalls.length === 0) {
2396
+ throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), upstreamFinish === 'length' ? { skipBench: true } : {});
2397
+ }
2398
+ // Structured-output enforcement for streams (#933): the non-stream
2399
+ // path checks JSON before returning; the stream path must too, or a
2400
+ // model that answers in prose despite the forwarded response_format
2401
+ // ships the essay as a "success" — the worst case for a
2402
+ // machine-readable request. json mode held every byte (headers never
2403
+ // flushed), so failing over here is free: skipBench (provider
2404
+ // healthy, the MODEL misbehaved) + skipModelForRequest (a sibling
2405
+ // key would misbehave identically). Mirrors proxy.ts non-stream.
2406
+ if (mode === 'json' && samplingParams.response_format && completedCalls.length === 0) {
2407
+ const enforced = enforceJsonContent(heldText);
2408
+ if (!enforced.ok) {
2409
+ const truncated = upstreamFinish === 'length';
2410
+ throw Object.assign(new Error(truncated
2411
+ ? `truncated JSON from ${route.displayName} (finish_reason=length — raise max_tokens for this ${samplingParams.response_format.type} request)`
2412
+ : `${route.displayName} ignored response_format (returned non-JSON despite ${samplingParams.response_format.type})`), { skipBench: true, skipModelForRequest: true });
2413
+ }
2414
+ if (enforced.healed)
2415
+ heldText = enforced.content;
2416
+ }
2417
+ flushHeaders();
2418
+ if (heldText.length > 0) {
2419
+ writeChunk(mkChunk({ content: heldText }, null));
2420
+ }
2421
+ if (completedCalls.length > 0) {
2422
+ writeChunk(mkChunk({ tool_calls: completedCalls.map((c, i) => ({ index: i, ...c })) }, null));
2423
+ totalOutputTokens += Math.ceil(completedCalls.reduce((n, c) => n + c.function.arguments.length, 0) / 4);
2424
+ }
2425
+ // Terminal finish_reason, ALWAYS present: calls win over a sloppy
2426
+ // upstream 'stop'; 'length'/'content_filter' survive for pure-text
2427
+ // turns; missing upstream reason is synthesized.
2428
+ const finish = completedCalls.length > 0
2429
+ ? 'tool_calls'
2430
+ : (upstreamFinish && upstreamFinish !== 'tool_calls' ? upstreamFinish : 'stop');
2431
+ writeChunk(mkChunk({}, finish));
2432
+ // One prompt-token estimate for both the injected usage frame below
2433
+ // and the accounting fallback after it, so a client that reads the
2434
+ // frame and the row this request writes can never disagree. Images
2435
+ // are billed at the same flat per-image estimate the routing budget
2436
+ // uses (the chars/4 pass sees text only).
2437
+ const estimatedPromptTokens = estimatedInputTokens + injectedHandoffTokens + imageCount * IMAGE_TOKEN_ESTIMATE;
2438
+ if (usageChunk) {
2439
+ writeChunk(usageChunk);
2440
+ }
2441
+ else {
2442
+ // Some OpenAI-compatible upstreams never echo a final usage
2443
+ // frame — neither when stream_options.include_usage is requested
2444
+ // nor otherwise. Strict clients (Hermes, Cline, Continue) treat a
2445
+ // missing usage block as "no accounting happened" and skip
2446
+ // per-call token/cost/billing_provider writes entirely; agents
2447
+ // that read usage for context-window display (e.g. #1084) show 0.
2448
+ //
2449
+ // So inject the estimate whenever the upstream never sent one —
2450
+ // regardless of whether the client asked for include_usage. The
2451
+ // numbers are this gateway's own chars/4 estimate (the same total
2452
+ // the accounting below records), never the upstream's accounting,
2453
+ // so the block is flagged `estimated: true` rather than passed
2454
+ // off as real counts.
2455
+ const completionTokens = totalOutputTokens;
2456
+ writeChunk({
2457
+ id: lastMeta.id ?? `chatcmpl-${Date.now()}`,
2458
+ object: 'chat.completion.chunk',
2459
+ created: lastMeta.created ?? Math.floor(Date.now() / 1000),
2460
+ model: lastMeta.model ?? route.modelId,
2461
+ choices: [],
2462
+ usage: {
2463
+ prompt_tokens: estimatedPromptTokens,
2464
+ completion_tokens: completionTokens,
2465
+ total_tokens: estimatedPromptTokens + completionTokens,
2466
+ estimated: true,
2467
+ },
2468
+ });
2469
+ }
2470
+ const doneFrame = 'data: [DONE]\n\n';
2471
+ collectFrame(doneFrame);
2472
+ res.write(doneFrame);
2473
+ res.end();
2474
+ const upstreamUsage = usageChunk?.usage;
2475
+ const inputTokens = upstreamUsage?.prompt_tokens ?? estimatedPromptTokens;
2476
+ const outputTokens = upstreamUsage?.completion_tokens ?? totalOutputTokens;
2477
+ const totalTokens = upstreamUsage?.total_tokens ?? (inputTokens + outputTokens);
2478
+ recordUpstreamSuccess(route, totalTokens, state);
2479
+ // Cache the freshly-generated SSE sequence so an identical later
2480
+ // stream request is replayed without spending another free-tier
2481
+ // slot. A truncated turn (finish 'length') is NOT cached, matching
2482
+ // the JSON cache policy — replaying a cut-off answer would be worse
2483
+ // than regenerating — and neither is one that outgrew the buffer
2484
+ // ceiling. A stream that errored or was aborted mid-flight never
2485
+ // reaches here at all (the catch below owns that path).
2486
+ if (cacheKey && streamFrames && streamCacheable && finish !== 'length') {
2487
+ storeCachedStreamResponse(cacheKey, {
2488
+ frames: streamFrames,
2489
+ platform: route.platform,
2490
+ modelId: route.modelId,
2491
+ keyId: route.keyId,
2492
+ promptTokens: inputTokens,
2493
+ completionTokens: outputTokens,
2494
+ });
2495
+ }
2496
+ setStickyModel(messages, route.modelDbId, sessionIdHeader, stickyStrategyKey);
2497
+ if (handoffMode !== 'off' && sessionKey)
2498
+ recordSuccessfulModel({ sessionKey, modelKey });
2499
+ // #797: remember this turn's thinking trace so the next request from
2500
+ // the same session can restore it (clients strip it on replay).
2501
+ if (streamReasoning.length > 0)
2502
+ rememberReasoning(reasoningSessionKey, modelKey, streamReasoning);
2503
+ traceRouteEvent('Proxy', {
2504
+ event: 'ok',
2505
+ requestId: requestGroupId,
2506
+ attempt,
2507
+ platform: route.platform,
2508
+ model: route.modelId,
2509
+ latencyMs: Date.now() - start,
2510
+ inputTokens,
2511
+ outputTokens,
2512
+ });
2513
+ logRequest(route.platform, route.modelId, route.keyId, 'success', inputTokens, outputTokens, Date.now() - start, null, ttfbMs, pinnedModelId, observeServedModel({ platform: route.platform, requestedModel: route.modelId, servedModel: upstreamModel }), 'http');
2514
+ return 'done';
2515
+ }
2516
+ catch (streamErr) {
2517
+ // Client abort mid-stream: the pump's own `if (clientGone) break`
2518
+ // can lose the race against the fetch-signal rejection, so the
2519
+ // abort may surface here instead. Rethrow — the shared loop's
2520
+ // client-abort branch stops the ladder without benching or an
2521
+ // error log row (the socket is gone; nothing to render).
2522
+ if (isClientAbortError(streamErr))
2523
+ throw streamErr;
2524
+ if (headerSent) {
2525
+ // Mid-stream error after real payload reached the client — finish
2526
+ // the SSE response honestly instead of leaving the client hanging.
2527
+ console.error(`[Proxy] Mid-stream error from ${route.displayName}:`, streamErr.message);
2528
+ const payload = { error: { message: `Provider error (${route.displayName}): stream interrupted`, type: 'stream_error' } };
2529
+ try {
2530
+ res.write(`data: ${JSON.stringify(payload)}\n\n`);
2531
+ }
2532
+ catch { /* socket gone */ }
2533
+ try {
2534
+ res.write('data: [DONE]\n\n');
2535
+ res.end();
2536
+ }
2537
+ catch { /* socket gone */ }
2538
+ traceRouteEvent('Proxy', {
2539
+ event: 'fail',
2540
+ requestId: requestGroupId,
2541
+ attempt,
2542
+ platform: route.platform,
2543
+ model: route.modelId,
2544
+ latencyMs: Date.now() - start,
2545
+ error: sanitizeProviderErrorMessage(streamErr.message),
2546
+ });
2547
+ logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, totalOutputTokens, Date.now() - start, sanitizeProviderErrorMessage(streamErr.message), ttfbMs, pinnedModelId, null, 'http');
2548
+ return 'committed';
2549
+ }
2550
+ // Headers never sent — bubble to the shared loop, which cooldowns this
2551
+ // model+key and tries the next one. Covers upstream HTTP errors, in-band
2552
+ // error frames, abrupt EOF, stalls, empty completions, and unparseable
2553
+ // dialect turns alike.
2554
+ throw streamErr;
2555
+ }
2556
+ }
2557
+ else {
2558
+ const result = await route.provider.chatCompletion(route.apiKey, outboundMessages, route.modelId, { temperature, max_tokens, top_p, stop, tools, tool_choice, parallel_tool_calls, ...samplingParams, contextBudget, signal: AbortSignal.any([clientAbort.signal, hedgeAbort.signal]) }, quotaContextForRoute(route, 'chat/completions'));
2559
+ // Raw upstream-reported model, captured BEFORE the contract overwrite
2560
+ // below destroys it — the only evidence when a provider silently
2561
+ // serves a different model than requested (#534). The OpenAI-compat,
2562
+ // cohere, and cloudflare adapters pass the upstream body through, so
2563
+ // result.model here is still the provider's own claim; google/aihorde
2564
+ // synthesize their responses with the routed id (no upstream signal).
2565
+ const upstreamModel = typeof result.model === 'string' ? result.model : null;
2566
+ // Upstream `model` fields are provider-controlled and can be a generic
2567
+ // placeholder such as Reka's "default". The gateway contract exposes
2568
+ // the concrete routed model, consistently across every provider.
2569
+ result.model = route.modelId;
2570
+ // Empty completion (no text, no tool calls) → fail over rather than
2571
+ // return a transport-level "success" the caller can't act on. Mirrors
2572
+ // the zero-chunk streaming case above. Throwing hands it to the shared
2573
+ // loop, which classifies "empty completion" as retryable and applies the
2574
+ // same cooldown/skip/penalty bookkeeping as every other failure.
2575
+ const respMsg = result.choices?.[0]?.message;
2576
+ const respText = contentToString(respMsg?.content ?? '');
2577
+ if (!respText && (respMsg?.tool_calls?.length ?? 0) === 0) {
2578
+ // finish_reason 'length' = the model spent the whole output budget on
2579
+ // hidden reasoning before any visible text (observed live: 5 of 11
2580
+ // hops in one chain). Still fail over, but skipBench tells the shared
2581
+ // loop not to cooldown/penalize a healthy model for a truncated turn.
2582
+ throw Object.assign(new Error(`empty completion from ${route.displayName}`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
2583
+ }
2584
+ // #809: a bare "safe"/"unsafe" classification word from a relay is an
2585
+ // upstream filter, not the requested model — fail over like an empty
2586
+ // completion.
2587
+ if (isUpstreamClassificationOutput(respText, route.platform) && (respMsg?.tool_calls?.length ?? 0) === 0) {
2588
+ throw Object.assign(new Error(`empty completion from ${route.displayName} (upstream classification output)`), result.choices?.[0]?.finish_reason === 'length' ? { skipBench: true } : {});
2589
+ }
2590
+ // Inline tool-call dialect rescue (#231 audit): a tool-bearing
2591
+ // request answered with the call serialized as TEXT (a mid-
2592
+ // conversation model switch makes the new model imitate the previous
2593
+ // model's private syntax). Re-parse it into structured tool_calls so
2594
+ // the client's agent loop keeps working; a detected-but-unparseable
2595
+ // dialect is a dead turn and fails over like an empty completion.
2596
+ if (wantsTools && respMsg && (respMsg.tool_calls?.length ?? 0) === 0 && respText) {
2597
+ const rescue = rescueInlineToolCalls(respText, new Set((tools ?? []).map(t => t.function.name)));
2598
+ if (rescue.detected) {
2599
+ if (!rescue.calls) {
2600
+ throw new Error(`unparseable inline tool-call dialect from ${route.displayName}: ${respText.slice(0, 120)}`);
2601
+ }
2602
+ const schemas = toolSchemaMap(tools);
2603
+ respMsg.tool_calls = rescue.calls.map((c, i) => ({
2604
+ id: `call_rescued_${i + 1}`,
2605
+ type: 'function',
2606
+ function: { name: c.name, arguments: repairToolArguments(c.arguments, schemas.get(c.name)) },
2607
+ }));
2608
+ respMsg.content = rescue.cleanText.length > 0 ? rescue.cleanText : null;
2609
+ if (result.choices?.[0])
2610
+ result.choices[0].finish_reason = 'tool_calls';
2611
+ console.log(`[Proxy] Rescued ${rescue.calls.length} inline tool call(s) from ${route.displayName} into structured tool_calls`);
2612
+ }
2613
+ }
2614
+ // Structured-output enforcement (#514 follow-up): the client asked for
2615
+ // JSON; a model that answered in prose despite the forwarded
2616
+ // response_format must not be returned as a "success". Heal the common
2617
+ // almost-right shapes (fenced block, prose-wrapped JSON) in place;
2618
+ // otherwise fail over. Deliberately AFTER the dialect rescue (matching
2619
+ // responses.ts): an inline tool-call turn isn't JSON either, and
2620
+ // gating it first burned a failover hop on turns the rescue converts.
2621
+ // skipBench: the provider is healthy — the MODEL misbehaved — so no
2622
+ // cooldown/penalty; skipModelForRequest: a sibling key would misbehave
2623
+ // identically, so rule out the whole model for this request.
2624
+ if (samplingParams.response_format && respText && (respMsg?.tool_calls?.length ?? 0) === 0) {
2625
+ const enforced = enforceJsonContent(respText);
2626
+ if (!enforced.ok) {
2627
+ // finish_reason 'length' = the JSON was CUT OFF by max_tokens, not
2628
+ // ignored — same failover (a terser model may fit the budget), but
2629
+ // an honest error class/trail instead of "ignored response_format".
2630
+ const truncated = result.choices?.[0]?.finish_reason === 'length';
2631
+ throw Object.assign(new Error(truncated
2632
+ ? `truncated JSON from ${route.displayName} (finish_reason=length — raise max_tokens for this ${samplingParams.response_format.type} request)`
2633
+ : `${route.displayName} ignored response_format (returned non-JSON despite ${samplingParams.response_format.type})`), { skipBench: true, skipModelForRequest: true });
2634
+ }
2635
+ if (enforced.healed && respMsg) {
2636
+ respMsg.content = enforced.content;
2637
+ }
2638
+ }
2639
+ // Repair double-encoded tool arguments against the request's tool
2640
+ // schemas (e.g. GLM emitting an array parameter as a JSON string),
2641
+ // so strict clients don't reject the call. Schema-gated — a true
2642
+ // string parameter is never touched. See lib/tool-args.ts.
2643
+ //
2644
+ // Deliberately BEFORE the success bookkeeping below: the opt-in schema
2645
+ // verdict that follows can still fail this attempt over, and crediting
2646
+ // recordUpstreamSuccess / rememberReasoning / setStickyModel to an
2647
+ // attempt we are about to discard would bill a model that never served
2648
+ // the turn and pin the session to it for the next one.
2649
+ if (respMsg?.tool_calls?.length) {
2650
+ const schemas = toolSchemaMap(tools);
2651
+ for (const tc of respMsg.tool_calls) {
2652
+ if (tc?.function?.arguments != null) {
2653
+ tc.function.arguments = repairToolArguments(tc.function.arguments, schemas.get(tc.function.name));
2654
+ }
2655
+ }
2656
+ // Whatever the repair could not fix is still broken. Opt-in, and
2657
+ // thrown before anything is written, so failover is invisible.
2658
+ if (isToolArgumentValidationEnabled()) {
2659
+ const invalid = invalidToolCallReasons(respMsg.tool_calls, schemas);
2660
+ if (invalid.length > 0)
2661
+ throw invalidToolArgumentsError(route.displayName, invalid);
2662
+ }
2663
+ }
2664
+ // Usage fallback: providers that omit `usage` used to be logged as 0
2665
+ // tokens, silently undercounting analytics and the rate-limit ledger.
2666
+ // Fall back to the same chars/4 estimate the streaming path uses (tool
2667
+ // arguments included, mirroring the stream accounting; counted after
2668
+ // the repair above, so it measures the bytes actually sent, and
2669
+ // reasoning tokens included too, so thinking models aren't
2670
+ // undercounted — #764).
2671
+ const respToolArgChars = (respMsg?.tool_calls ?? []).reduce((n, tc) => n + (tc?.function?.arguments?.length ?? 0), 0);
2672
+ const promptTokens = result.usage?.prompt_tokens ?? estimatedInputTokens;
2673
+ const completionTokens = result.usage?.completion_tokens
2674
+ ?? Math.ceil((contentToString(respMsg?.content ?? '').length + completionReasoningText(result).length + respToolArgChars) / 4);
2675
+ const totalTokens = result.usage?.total_tokens ?? (promptTokens + completionTokens);
2676
+ recordUpstreamSuccess(route, totalTokens, state);
2677
+ // #797: remember this turn's thinking trace so the next request from
2678
+ // the same session can restore it (clients strip it on replay).
2679
+ // normalizeChoices keeps reasoning_content on the message even when it
2680
+ // folds the trace into an otherwise-empty content field.
2681
+ if (typeof respMsg?.reasoning_content === 'string' && respMsg.reasoning_content.length > 0) {
2682
+ rememberReasoning(reasoningSessionKey, modelKey, respMsg.reasoning_content);
2683
+ }
2684
+ // Use stickyStrategyKey (not the global strategyKey) so a group-pinned
2685
+ // request writes its sticky entry under the SAME key the next turn reads
2686
+ // from (set to the requested model id at the top of the loop). Matches the
2687
+ // streaming success path; without it, "prefer last successful provider"
2688
+ // is lost for non-streaming group-pinned sessions. (#341 review)
2689
+ setStickyModel(messages, route.modelDbId, sessionIdHeader, stickyStrategyKey);
2690
+ if (handoffMode !== 'off' && sessionKey)
2691
+ recordSuccessfulModel({ sessionKey, modelKey });
2692
+ res.setHeader('X-Routed-Via', routedViaValue(route.platform, route.modelId));
2693
+ setFallbackHeaders(res, attempt, attemptLog);
2694
+ // Normalize array-shaped message.content to a string on the way out (#166).
2695
+ const outboundBody = sanitizeResponse(normalizeOutboundContent(result));
2696
+ res.setHeader('X-FreeLLM-Cache', cacheKey ? 'MISS' : 'OFF');
2697
+ // #1084: agents show zero context usage when the upstream omits
2698
+ // `usage` (many free-tier providers do). Fall back to the same
2699
+ // chars/4 estimate used for accounting above, flagged `estimated:
2700
+ // true` so a cost-accounting client can tell it apart from the
2701
+ // upstream's real counts.
2702
+ if (!outboundBody.usage) {
2703
+ outboundBody.usage = {
2704
+ prompt_tokens: promptTokens,
2705
+ completion_tokens: completionTokens,
2706
+ total_tokens: totalTokens,
2707
+ estimated: true,
2708
+ };
2709
+ }
2710
+ // Persist the completed response for Idempotency-Key replays. Only
2711
+ // non-streaming requests with a valid key reach here; a truncated turn
2712
+ // (finish_reason 'length') is NOT stored — replaying a cut-off answer
2713
+ // would be worse than regenerating, matching the cache policy below.
2714
+ if (idemKey
2715
+ && idemFingerprint
2716
+ && result.choices?.[0]?.finish_reason !== 'length') {
2717
+ storeIdempotencyResult(hashIdempotencyKey(idemKey), idemFingerprint, 200, outboundBody, requestGroupId);
2718
+ }
2719
+ // #1102: expose the execution id in the body so clients (incl. AI
2720
+ // agents that ignore headers) can trace this response to its analytics
2721
+ // attempt trail. Additive — OpenAI clients ignore unknown fields.
2722
+ res.json({ execution_id: requestGroupId, ...outboundBody });
2723
+ // Cache the freshly-generated answer so an identical later request is
2724
+ // served from memory without spending another free-tier slot. A
2725
+ // truncated turn (finish_reason 'length') is NOT cached: replaying a
2726
+ // cut-off answer forever would be worse than regenerating.
2727
+ if (cacheKey && result.choices?.[0]?.finish_reason !== 'length') {
2728
+ storeCachedResponse(cacheKey, {
2729
+ body: outboundBody,
2730
+ platform: route.platform,
2731
+ modelId: route.modelId,
2732
+ keyId: route.keyId,
2733
+ promptTokens,
2734
+ completionTokens,
2735
+ });
2736
+ }
2737
+ traceRouteEvent('Proxy', {
2738
+ event: 'ok',
2739
+ requestId: requestGroupId,
2740
+ attempt,
2741
+ platform: route.platform,
2742
+ model: route.modelId,
2743
+ latencyMs: Date.now() - start,
2744
+ inputTokens: promptTokens,
2745
+ outputTokens: completionTokens,
2746
+ });
2747
+ logRequest(route.platform, route.modelId, route.keyId, 'success', promptTokens, completionTokens, Date.now() - start, null, null, pinnedModelId, observeServedModel({ platform: route.platform, requestedModel: route.modelId, servedModel: upstreamModel }), 'http');
2748
+ return 'done';
2749
+ }
2750
+ },
2751
+ logFailure: (route, err, attempt) => {
2752
+ const latency = Date.now() - start;
2753
+ const safeError = sanitizeProviderErrorMessage(err.message);
2754
+ traceRouteEvent('Proxy', {
2755
+ event: 'fail',
2756
+ requestId: requestGroupId,
2757
+ attempt,
2758
+ platform: route.platform,
2759
+ model: route.modelId,
2760
+ latencyMs: latency,
2761
+ error: safeError,
2762
+ });
2763
+ logRequest(route.platform, route.modelId, route.keyId, 'error', estimatedInputTokens, 0, latency, safeError, null, pinnedModelId, null, 'http');
2764
+ },
2765
+ onFatal: (route, err, attempt) => {
2766
+ // Non-retryable error (bare 4xx, etc.): don't retry.
2767
+ setFallbackHeaders(res, attempt, attemptLog);
2768
+ res.status(502).json({
2769
+ error: {
2770
+ message: `Provider error (${route.displayName}): ${sanitizeProviderErrorMessage(err.message)}`,
2771
+ type: 'provider_error',
2772
+ },
2773
+ execution_id: requestGroupId,
2774
+ });
2775
+ },
2776
+ onRoutingExhausted: (lastError, routeErr, exhaustion, info) => {
2777
+ // No more models available.
2778
+ setFallbackHeaders(res, info.attempts.length, info.attempts);
2779
+ setExhaustionHeaders(res, exhaustion);
2780
+ res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
2781
+ },
2782
+ onExhausted: (exhaustion, info) => {
2783
+ setFallbackHeaders(res, info.attempts.length, info.attempts);
2784
+ setExhaustionHeaders(res, exhaustion);
2785
+ res.status(exhaustion.status).json({ error: exhaustionErrorPayload(exhaustion), execution_id: requestGroupId });
2786
+ },
2787
+ });
2788
+ });
2789
+ // logRequest moved to lib/request-log.ts (shared with the fusion service to
2790
+ // avoid an import cycle); imported above for internal use and re-exported here
2791
+ // for routes/responses.ts which imports it from this module.
2792
+ export { logRequest };
2793
+ //# sourceMappingURL=proxy.js.map