@homeflare/distilled-litellm 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (390) hide show
  1. package/README.md +48 -12
  2. package/dist/services/a2a.js +1 -1
  3. package/dist/services/a2a_registration.js +1 -1
  4. package/dist/services/access_groups.d.ts +18 -0
  5. package/dist/services/access_groups.d.ts.map +1 -1
  6. package/dist/services/access_groups.js +15 -1
  7. package/dist/services/access_groups.js.map +1 -1
  8. package/dist/services/adaptive_router.js +1 -1
  9. package/dist/services/agents.d.ts +10 -0
  10. package/dist/services/agents.d.ts.map +1 -1
  11. package/dist/services/agents.js +10 -1
  12. package/dist/services/agents.js.map +1 -1
  13. package/dist/services/alerting.js +1 -1
  14. package/dist/services/anthropic_passthrough.js +1 -1
  15. package/dist/services/anthropic_skills.d.ts +7 -1
  16. package/dist/services/anthropic_skills.d.ts.map +1 -1
  17. package/dist/services/anthropic_skills.js +6 -2
  18. package/dist/services/anthropic_skills.js.map +1 -1
  19. package/dist/services/assistants.js +1 -1
  20. package/dist/services/audio.js +1 -1
  21. package/dist/services/audit_logging.d.ts +2 -0
  22. package/dist/services/audit_logging.d.ts.map +1 -1
  23. package/dist/services/audit_logging.js +2 -1
  24. package/dist/services/audit_logging.js.map +1 -1
  25. package/dist/services/auto_router.d.ts +162 -62
  26. package/dist/services/auto_router.d.ts.map +1 -1
  27. package/dist/services/auto_router.js +92 -31
  28. package/dist/services/auto_router.js.map +1 -1
  29. package/dist/services/batch.js +1 -1
  30. package/dist/services/beta_agents.js +1 -1
  31. package/dist/services/beta_mcp.js +1 -1
  32. package/dist/services/budget_management.d.ts +8 -3
  33. package/dist/services/budget_management.d.ts.map +1 -1
  34. package/dist/services/budget_management.js +7 -4
  35. package/dist/services/budget_management.js.map +1 -1
  36. package/dist/services/budget_spend_tracking.d.ts +24 -5
  37. package/dist/services/budget_spend_tracking.d.ts.map +1 -1
  38. package/dist/services/budget_spend_tracking.js +17 -4
  39. package/dist/services/budget_spend_tracking.js.map +1 -1
  40. package/dist/services/cache_settings.js +1 -1
  41. package/dist/services/caching.js +1 -1
  42. package/dist/services/chat_completions.js +1 -1
  43. package/dist/services/claude_code_marketplace.d.ts +11 -10
  44. package/dist/services/claude_code_marketplace.d.ts.map +1 -1
  45. package/dist/services/claude_code_marketplace.js +8 -6
  46. package/dist/services/claude_code_marketplace.js.map +1 -1
  47. package/dist/services/cloudzero.js +1 -1
  48. package/dist/services/completions.js +1 -1
  49. package/dist/services/compliance.js +1 -1
  50. package/dist/services/config_overrides.d.ts +5 -1
  51. package/dist/services/config_overrides.d.ts.map +1 -1
  52. package/dist/services/config_overrides.js +3 -1
  53. package/dist/services/config_overrides.js.map +1 -1
  54. package/dist/services/config_yaml.d.ts +120 -6
  55. package/dist/services/config_yaml.d.ts.map +1 -1
  56. package/dist/services/config_yaml.js +67 -1
  57. package/dist/services/config_yaml.js.map +1 -1
  58. package/dist/services/containers.d.ts +8 -2
  59. package/dist/services/containers.d.ts.map +1 -1
  60. package/dist/services/containers.js +13 -5
  61. package/dist/services/containers.js.map +1 -1
  62. package/dist/services/coordination_redis_settings.js +1 -1
  63. package/dist/services/cost_tracking.d.ts +91 -1
  64. package/dist/services/cost_tracking.d.ts.map +1 -1
  65. package/dist/services/cost_tracking.js +82 -2
  66. package/dist/services/cost_tracking.js.map +1 -1
  67. package/dist/services/credential_management.d.ts +16 -3
  68. package/dist/services/credential_management.d.ts.map +1 -1
  69. package/dist/services/credential_management.js +13 -4
  70. package/dist/services/credential_management.js.map +1 -1
  71. package/dist/services/customer_management.d.ts +25 -2
  72. package/dist/services/customer_management.d.ts.map +1 -1
  73. package/dist/services/customer_management.js +22 -3
  74. package/dist/services/customer_management.js.map +1 -1
  75. package/dist/services/email_management.js +1 -1
  76. package/dist/services/embeddings.js +1 -1
  77. package/dist/services/evals.js +1 -1
  78. package/dist/services/experimental.js +1 -1
  79. package/dist/services/fallback_management.js +1 -1
  80. package/dist/services/files.js +1 -1
  81. package/dist/services/fine_tuning.js +1 -1
  82. package/dist/services/gemini_agents.js +1 -1
  83. package/dist/services/google_genai_endpoints.js +1 -1
  84. package/dist/services/guardrails.d.ts +80 -7
  85. package/dist/services/guardrails.d.ts.map +1 -1
  86. package/dist/services/guardrails.js +37 -1
  87. package/dist/services/guardrails.js.map +1 -1
  88. package/dist/services/health.d.ts +2 -2
  89. package/dist/services/health.d.ts.map +1 -1
  90. package/dist/services/health.js +2 -2
  91. package/dist/services/health.js.map +1 -1
  92. package/dist/services/images.js +1 -1
  93. package/dist/services/index.d.ts +1 -15
  94. package/dist/services/index.d.ts.map +1 -1
  95. package/dist/services/index.js +1 -15
  96. package/dist/services/index.js.map +1 -1
  97. package/dist/services/internal_user_management.d.ts +227 -39
  98. package/dist/services/internal_user_management.d.ts.map +1 -1
  99. package/dist/services/internal_user_management.js +194 -37
  100. package/dist/services/internal_user_management.js.map +1 -1
  101. package/dist/services/invite_links.js +1 -1
  102. package/dist/services/jwt_mappings.d.ts +3 -0
  103. package/dist/services/jwt_mappings.d.ts.map +1 -1
  104. package/dist/services/jwt_mappings.js +4 -1
  105. package/dist/services/jwt_mappings.js.map +1 -1
  106. package/dist/services/key_management.d.ts +116 -50
  107. package/dist/services/key_management.d.ts.map +1 -1
  108. package/dist/services/key_management.js +95 -40
  109. package/dist/services/key_management.js.map +1 -1
  110. package/dist/services/langfuse_passthrough.js +1 -1
  111. package/dist/services/llm_passthrough.d.ts +1199 -0
  112. package/dist/services/llm_passthrough.d.ts.map +1 -0
  113. package/dist/services/llm_passthrough.js +2150 -0
  114. package/dist/services/llm_passthrough.js.map +1 -0
  115. package/dist/services/llm_utils.js +1 -1
  116. package/dist/services/logging_callbacks.js +1 -1
  117. package/dist/services/mcp_byok_oauth.d.ts +18 -11
  118. package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
  119. package/dist/services/mcp_byok_oauth.js +27 -24
  120. package/dist/services/mcp_byok_oauth.js.map +1 -1
  121. package/dist/services/mcp_discoverable.d.ts +34 -0
  122. package/dist/services/mcp_discoverable.d.ts.map +1 -1
  123. package/dist/services/mcp_discoverable.js +51 -1
  124. package/dist/services/mcp_discoverable.js.map +1 -1
  125. package/dist/services/mcp_management.d.ts +90 -2
  126. package/dist/services/mcp_management.d.ts.map +1 -1
  127. package/dist/services/mcp_management.js +115 -3
  128. package/dist/services/mcp_management.js.map +1 -1
  129. package/dist/services/mcp_rest.d.ts +13 -0
  130. package/dist/services/mcp_rest.d.ts.map +1 -1
  131. package/dist/services/mcp_rest.js +10 -2
  132. package/dist/services/mcp_rest.js.map +1 -1
  133. package/dist/services/memory_management.d.ts +2 -0
  134. package/dist/services/memory_management.d.ts.map +1 -1
  135. package/dist/services/memory_management.js +2 -1
  136. package/dist/services/memory_management.js.map +1 -1
  137. package/dist/services/misc.d.ts +86 -1
  138. package/dist/services/misc.d.ts.map +1 -1
  139. package/dist/services/misc.js +126 -2
  140. package/dist/services/misc.js.map +1 -1
  141. package/dist/services/model_management.d.ts +310 -19
  142. package/dist/services/model_management.d.ts.map +1 -1
  143. package/dist/services/model_management.js +240 -7
  144. package/dist/services/model_management.js.map +1 -1
  145. package/dist/services/moderations.js +1 -1
  146. package/dist/services/ocr.js +1 -1
  147. package/dist/services/open_ai_pass_through.d.ts +35 -90
  148. package/dist/services/open_ai_pass_through.d.ts.map +1 -1
  149. package/dist/services/open_ai_pass_through.js +41 -139
  150. package/dist/services/open_ai_pass_through.js.map +1 -1
  151. package/dist/services/organization_management.d.ts +21 -1
  152. package/dist/services/organization_management.d.ts.map +1 -1
  153. package/dist/services/organization_management.js +20 -2
  154. package/dist/services/organization_management.js.map +1 -1
  155. package/dist/services/plugins.js +1 -1
  156. package/dist/services/policies.d.ts +2 -0
  157. package/dist/services/policies.d.ts.map +1 -1
  158. package/dist/services/policies.js +3 -1
  159. package/dist/services/policies.js.map +1 -1
  160. package/dist/services/policy_engine.d.ts +6 -0
  161. package/dist/services/policy_engine.d.ts.map +1 -1
  162. package/dist/services/policy_engine.js +4 -1
  163. package/dist/services/policy_engine.js.map +1 -1
  164. package/dist/services/project_management.d.ts +15 -0
  165. package/dist/services/project_management.d.ts.map +1 -1
  166. package/dist/services/project_management.js +14 -1
  167. package/dist/services/project_management.js.map +1 -1
  168. package/dist/services/prompts.js +1 -1
  169. package/dist/services/public.d.ts +124 -0
  170. package/dist/services/public.d.ts.map +1 -1
  171. package/dist/services/public.js +158 -1
  172. package/dist/services/public.js.map +1 -1
  173. package/dist/services/rag.js +1 -1
  174. package/dist/services/realtime.d.ts +4 -26
  175. package/dist/services/realtime.d.ts.map +1 -1
  176. package/dist/services/realtime.js +7 -57
  177. package/dist/services/realtime.js.map +1 -1
  178. package/dist/services/rerank.d.ts +72 -0
  179. package/dist/services/rerank.d.ts.map +1 -1
  180. package/dist/services/rerank.js +61 -4
  181. package/dist/services/rerank.js.map +1 -1
  182. package/dist/services/responses.d.ts +30 -0
  183. package/dist/services/responses.d.ts.map +1 -1
  184. package/dist/services/responses.js +59 -1
  185. package/dist/services/responses.js.map +1 -1
  186. package/dist/services/router_settings.js +1 -1
  187. package/dist/services/rust_control_plane.js +1 -1
  188. package/dist/services/scim.d.ts +41 -1
  189. package/dist/services/scim.d.ts.map +1 -1
  190. package/dist/services/scim.js +60 -2
  191. package/dist/services/scim.js.map +1 -1
  192. package/dist/services/search.js +1 -1
  193. package/dist/services/search_tools.js +1 -1
  194. package/dist/services/settings.d.ts +88 -0
  195. package/dist/services/settings.d.ts.map +1 -1
  196. package/dist/services/settings.js +113 -1
  197. package/dist/services/settings.js.map +1 -1
  198. package/dist/services/sso_settings.d.ts +1 -1
  199. package/dist/services/sso_settings.d.ts.map +1 -1
  200. package/dist/services/sso_settings.js +1 -1
  201. package/dist/services/sso_settings.js.map +1 -1
  202. package/dist/services/tag_management.d.ts +7 -0
  203. package/dist/services/tag_management.d.ts.map +1 -1
  204. package/dist/services/tag_management.js +8 -1
  205. package/dist/services/tag_management.js.map +1 -1
  206. package/dist/services/team_management.d.ts +265 -17
  207. package/dist/services/team_management.d.ts.map +1 -1
  208. package/dist/services/team_management.js +247 -12
  209. package/dist/services/team_management.js.map +1 -1
  210. package/dist/services/tools.js +1 -1
  211. package/dist/services/ui_settings.js +1 -1
  212. package/dist/services/ui_theme_settings.js +1 -1
  213. package/dist/services/usage_ai.js +1 -1
  214. package/dist/services/vantage.js +1 -1
  215. package/dist/services/vector_store_management.js +1 -1
  216. package/dist/services/vector_stores.js +1 -1
  217. package/dist/services/videos.js +1 -1
  218. package/dist/services/web_socket.d.ts +0 -18
  219. package/dist/services/web_socket.d.ts.map +1 -1
  220. package/dist/services/web_socket.js +1 -33
  221. package/dist/services/web_socket.js.map +1 -1
  222. package/dist/services/workflow_management.js +1 -1
  223. package/package.json +1 -1
  224. package/src/services/a2a.ts +1 -1
  225. package/src/services/a2a_registration.ts +1 -1
  226. package/src/services/access_groups.ts +44 -1
  227. package/src/services/adaptive_router.ts +1 -1
  228. package/src/services/agents.ts +22 -1
  229. package/src/services/alerting.ts +1 -1
  230. package/src/services/anthropic_passthrough.ts +1 -1
  231. package/src/services/anthropic_skills.ts +12 -2
  232. package/src/services/assistants.ts +1 -1
  233. package/src/services/audio.ts +1 -1
  234. package/src/services/audit_logging.ts +4 -1
  235. package/src/services/auto_router.ts +303 -93
  236. package/src/services/batch.ts +1 -1
  237. package/src/services/beta_agents.ts +1 -1
  238. package/src/services/beta_mcp.ts +1 -1
  239. package/src/services/budget_management.ts +12 -4
  240. package/src/services/budget_spend_tracking.ts +38 -6
  241. package/src/services/cache_settings.ts +1 -1
  242. package/src/services/caching.ts +1 -1
  243. package/src/services/chat_completions.ts +1 -1
  244. package/src/services/claude_code_marketplace.ts +20 -14
  245. package/src/services/cloudzero.ts +1 -1
  246. package/src/services/completions.ts +1 -1
  247. package/src/services/compliance.ts +1 -1
  248. package/src/services/config_overrides.ts +8 -2
  249. package/src/services/config_yaml.ts +251 -6
  250. package/src/services/containers.ts +29 -11
  251. package/src/services/coordination_redis_settings.ts +1 -1
  252. package/src/services/cost_tracking.ts +198 -2
  253. package/src/services/credential_management.ts +25 -6
  254. package/src/services/customer_management.ts +49 -3
  255. package/src/services/email_management.ts +1 -1
  256. package/src/services/embeddings.ts +1 -1
  257. package/src/services/evals.ts +1 -1
  258. package/src/services/experimental.ts +1 -1
  259. package/src/services/fallback_management.ts +1 -1
  260. package/src/services/files.ts +1 -1
  261. package/src/services/fine_tuning.ts +1 -1
  262. package/src/services/gemini_agents.ts +1 -1
  263. package/src/services/google_genai_endpoints.ts +1 -1
  264. package/src/services/guardrails.ts +198 -8
  265. package/src/services/health.ts +3 -2
  266. package/src/services/images.ts +1 -1
  267. package/src/services/index.ts +1 -15
  268. package/src/services/internal_user_management.ts +565 -103
  269. package/src/services/invite_links.ts +1 -1
  270. package/src/services/jwt_mappings.ts +7 -1
  271. package/src/services/key_management.ts +257 -127
  272. package/src/services/langfuse_passthrough.ts +1 -1
  273. package/src/services/llm_passthrough.ts +4542 -0
  274. package/src/services/llm_utils.ts +1 -1
  275. package/src/services/logging_callbacks.ts +1 -1
  276. package/src/services/mcp_byok_oauth.ts +65 -45
  277. package/src/services/mcp_discoverable.ts +109 -1
  278. package/src/services/mcp_management.ts +258 -3
  279. package/src/services/mcp_rest.ts +30 -5
  280. package/src/services/memory_management.ts +4 -1
  281. package/src/services/misc.ts +269 -2
  282. package/src/services/model_management.ts +666 -21
  283. package/src/services/moderations.ts +1 -1
  284. package/src/services/ocr.ts +1 -1
  285. package/src/services/open_ai_pass_through.ts +81 -286
  286. package/src/services/organization_management.ts +44 -2
  287. package/src/services/plugins.ts +1 -1
  288. package/src/services/policies.ts +5 -1
  289. package/src/services/policy_engine.ts +10 -1
  290. package/src/services/project_management.ts +33 -1
  291. package/src/services/prompts.ts +1 -1
  292. package/src/services/public.ts +356 -1
  293. package/src/services/rag.ts +1 -1
  294. package/src/services/realtime.ts +11 -111
  295. package/src/services/rerank.ts +187 -7
  296. package/src/services/responses.ts +117 -1
  297. package/src/services/router_settings.ts +1 -1
  298. package/src/services/rust_control_plane.ts +1 -1
  299. package/src/services/scim.ts +134 -3
  300. package/src/services/search.ts +1 -1
  301. package/src/services/search_tools.ts +1 -1
  302. package/src/services/settings.ts +278 -1
  303. package/src/services/sso_settings.ts +2 -1
  304. package/src/services/tag_management.ts +15 -1
  305. package/src/services/team_management.ts +676 -37
  306. package/src/services/tools.ts +1 -1
  307. package/src/services/ui_settings.ts +1 -1
  308. package/src/services/ui_theme_settings.ts +1 -1
  309. package/src/services/usage_ai.ts +1 -1
  310. package/src/services/vantage.ts +1 -1
  311. package/src/services/vector_store_management.ts +1 -1
  312. package/src/services/vector_stores.ts +1 -1
  313. package/src/services/videos.ts +1 -1
  314. package/src/services/web_socket.ts +1 -61
  315. package/src/services/workflow_management.ts +1 -1
  316. package/dist/services/anthropic_pass_through.d.ts +0 -70
  317. package/dist/services/anthropic_pass_through.d.ts.map +0 -1
  318. package/dist/services/anthropic_pass_through.js +0 -114
  319. package/dist/services/anthropic_pass_through.js.map +0 -1
  320. package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
  321. package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
  322. package/dist/services/assembly_ai_eu_pass_through.js +0 -114
  323. package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
  324. package/dist/services/assembly_ai_pass_through.d.ts +0 -70
  325. package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
  326. package/dist/services/assembly_ai_pass_through.js +0 -114
  327. package/dist/services/assembly_ai_pass_through.js.map +0 -1
  328. package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
  329. package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
  330. package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
  331. package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
  332. package/dist/services/azure_ai_pass_through.d.ts +0 -70
  333. package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
  334. package/dist/services/azure_ai_pass_through.js +0 -112
  335. package/dist/services/azure_ai_pass_through.js.map +0 -1
  336. package/dist/services/azure_pass_through.d.ts +0 -70
  337. package/dist/services/azure_pass_through.d.ts.map +0 -1
  338. package/dist/services/azure_pass_through.js +0 -107
  339. package/dist/services/azure_pass_through.js.map +0 -1
  340. package/dist/services/bedrock_pass_through.d.ts +0 -70
  341. package/dist/services/bedrock_pass_through.d.ts.map +0 -1
  342. package/dist/services/bedrock_pass_through.js +0 -114
  343. package/dist/services/bedrock_pass_through.js.map +0 -1
  344. package/dist/services/cohere_pass_through.d.ts +0 -70
  345. package/dist/services/cohere_pass_through.d.ts.map +0 -1
  346. package/dist/services/cohere_pass_through.js +0 -112
  347. package/dist/services/cohere_pass_through.js.map +0 -1
  348. package/dist/services/cursor_pass_through.d.ts +0 -70
  349. package/dist/services/cursor_pass_through.d.ts.map +0 -1
  350. package/dist/services/cursor_pass_through.js +0 -112
  351. package/dist/services/cursor_pass_through.js.map +0 -1
  352. package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
  353. package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
  354. package/dist/services/google_ai_studio_pass_through.js +0 -112
  355. package/dist/services/google_ai_studio_pass_through.js.map +0 -1
  356. package/dist/services/milvus_pass_through.d.ts +0 -70
  357. package/dist/services/milvus_pass_through.d.ts.map +0 -1
  358. package/dist/services/milvus_pass_through.js +0 -112
  359. package/dist/services/milvus_pass_through.js.map +0 -1
  360. package/dist/services/mistral_pass_through.d.ts +0 -70
  361. package/dist/services/mistral_pass_through.d.ts.map +0 -1
  362. package/dist/services/mistral_pass_through.js +0 -114
  363. package/dist/services/mistral_pass_through.js.map +0 -1
  364. package/dist/services/vertex_ai_pass_through.d.ts +0 -180
  365. package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
  366. package/dist/services/vertex_ai_pass_through.js +0 -334
  367. package/dist/services/vertex_ai_pass_through.js.map +0 -1
  368. package/dist/services/vllm_pass_through.d.ts +0 -70
  369. package/dist/services/vllm_pass_through.d.ts.map +0 -1
  370. package/dist/services/vllm_pass_through.js +0 -104
  371. package/dist/services/vllm_pass_through.js.map +0 -1
  372. package/dist/services/watsonx_pass_through.d.ts +0 -70
  373. package/dist/services/watsonx_pass_through.d.ts.map +0 -1
  374. package/dist/services/watsonx_pass_through.js +0 -114
  375. package/dist/services/watsonx_pass_through.js.map +0 -1
  376. package/src/services/anthropic_pass_through.ts +0 -233
  377. package/src/services/assembly_ai_eu_pass_through.ts +0 -237
  378. package/src/services/assembly_ai_pass_through.ts +0 -237
  379. package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
  380. package/src/services/azure_ai_pass_through.ts +0 -231
  381. package/src/services/azure_pass_through.ts +0 -227
  382. package/src/services/bedrock_pass_through.ts +0 -229
  383. package/src/services/cohere_pass_through.ts +0 -227
  384. package/src/services/cursor_pass_through.ts +0 -227
  385. package/src/services/google_ai_studio_pass_through.ts +0 -227
  386. package/src/services/milvus_pass_through.ts +0 -227
  387. package/src/services/mistral_pass_through.ts +0 -229
  388. package/src/services/vertex_ai_pass_through.ts +0 -684
  389. package/src/services/vllm_pass_through.ts +0 -227
  390. package/src/services/watsonx_pass_through.ts +0 -229
@@ -26,6 +26,13 @@ export type LiteLLMParamsAdaptiveRouterConfigMap = {
26
26
  [key: string]: unknown | undefined;
27
27
  };
28
28
  export declare const LiteLLMParamsAdaptiveRouterConfigMap: S.Schema<LiteLLMParamsAdaptiveRouterConfigMap>;
29
+ export interface AwsSessionTag {
30
+ Key: string;
31
+ Value: string;
32
+ }
33
+ export declare const AwsSessionTag: S.Schema<AwsSessionTag>;
34
+ export type LiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
35
+ export declare const LiteLLMParamsAwsSessionTagsList: S.Schema<LiteLLMParamsAwsSessionTagsList>;
29
36
  export type LiteLLMParamsBedrockTagsList = Array<unknown>;
30
37
  export declare const LiteLLMParamsBedrockTagsList: S.Schema<LiteLLMParamsBedrockTagsList>;
31
38
  export type LiteLLMParamsComplexityRouterConfigMap = {
@@ -40,6 +47,8 @@ export type LiteLLMParamsConfigurableClientsideAuthParamsItem = string | Configu
40
47
  export declare const LiteLLMParamsConfigurableClientsideAuthParamsItem: S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsItem>;
41
48
  export type LiteLLMParamsConfigurableClientsideAuthParamsList = Array<LiteLLMParamsConfigurableClientsideAuthParamsItem>;
42
49
  export declare const LiteLLMParamsConfigurableClientsideAuthParamsList: S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsList>;
50
+ export type LiteLLMParamsDropParams = boolean | string;
51
+ export declare const LiteLLMParamsDropParams: S.Schema<LiteLLMParamsDropParams>;
43
52
  export type LiteLLMParamsMilvusPartitionNamesList = Array<string>;
44
53
  export declare const LiteLLMParamsMilvusPartitionNamesList: S.Schema<LiteLLMParamsMilvusPartitionNamesList>;
45
54
  export type ChoicesFinishReason = "stop" | "content_filter" | "function_call" | "tool_calls" | "length" | "guardrail_intervened" | "eos" | "finish_reason_unspecified" | "malformed_function_call";
@@ -265,6 +274,7 @@ export interface LiteLLMParams {
265
274
  adaptive_router_default_model?: string | null;
266
275
  allow_client_keepalive_override?: boolean | null;
267
276
  annotation_cost_per_page?: number | null;
277
+ annotation_cost_per_page_batches?: number | null;
268
278
  api_base?: string | null;
269
279
  api_key?: string | null;
270
280
  api_version?: string | null;
@@ -273,6 +283,8 @@ export interface LiteLLMParams {
273
283
  auto_router_default_model?: string | null;
274
284
  auto_router_embedding_model?: string | null;
275
285
  auto_router_max_input_chars?: number | null;
286
+ auto_router_model_compression?: string | null;
287
+ auto_router_routing_compression?: string | null;
276
288
  aws_access_key_id?: string | null;
277
289
  aws_batch_role_arn?: string | null;
278
290
  aws_bedrock_project_id?: string | null;
@@ -283,10 +295,14 @@ export interface LiteLLMParams {
283
295
  aws_role_name?: string | null;
284
296
  aws_secret_access_key?: string | null;
285
297
  aws_session_name?: string | null;
298
+ aws_session_tags?: LiteLLMParamsAwsSessionTagsList | null;
286
299
  aws_session_token?: string | null;
287
300
  aws_sts_endpoint?: string | null;
288
301
  aws_web_identity_token?: string | null;
289
302
  azure_ad_token?: string | null;
303
+ azure_password?: string | null;
304
+ azure_scope?: string | null;
305
+ azure_username?: string | null;
290
306
  bedrock_tags?: LiteLLMParamsBedrockTagsList | null;
291
307
  budget_duration?: string | null;
292
308
  cache_creation_input_audio_token_cost?: number | null;
@@ -311,22 +327,27 @@ export interface LiteLLMParams {
311
327
  cache_read_input_token_cost_priority?: number | null;
312
328
  cache_read_input_token_cost_ultrafast?: number | null;
313
329
  citation_cost_per_token?: number | null;
330
+ client_id?: string | null;
331
+ client_secret?: string | null;
314
332
  complexity_router_config?: LiteLLMParamsComplexityRouterConfigMap | null;
315
333
  complexity_router_default_model?: string | null;
316
334
  configurable_clientside_auth_params?: LiteLLMParamsConfigurableClientsideAuthParamsList | null;
317
335
  custom_llm_provider?: string | null;
318
336
  default_api_key_rpm_limit?: number | null;
319
337
  default_api_key_tpm_limit?: number | null;
338
+ drop_params?: LiteLLMParamsDropParams | null;
320
339
  gcs_bucket_name?: string | null;
321
340
  google_maps_grounding_cost_per_query?: number | null;
322
341
  input_cost_per_audio_per_second?: number | null;
323
342
  input_cost_per_audio_per_second_above_128k_tokens?: number | null;
324
343
  input_cost_per_audio_token?: number | null;
344
+ input_cost_per_audio_token_batches?: number | null;
325
345
  input_cost_per_character?: number | null;
326
346
  input_cost_per_character_above_128k_tokens?: number | null;
327
347
  input_cost_per_image?: number | null;
328
348
  input_cost_per_image_above_128k_tokens?: number | null;
329
349
  input_cost_per_image_token?: number | null;
350
+ input_cost_per_image_token_batches?: number | null;
330
351
  input_cost_per_pixel?: number | null;
331
352
  input_cost_per_query?: number | null;
332
353
  input_cost_per_second?: number | null;
@@ -348,6 +369,7 @@ export interface LiteLLMParams {
348
369
  input_cost_per_video_per_second_above_15s_interval?: number | null;
349
370
  input_cost_per_video_per_second_above_8s_interval?: number | null;
350
371
  input_cost_per_video_token?: number | null;
372
+ input_cost_per_video_token_batches?: number | null;
351
373
  itpm?: number | null;
352
374
  keepalive_seconds?: number | null;
353
375
  litellm_credential_name?: string | null;
@@ -364,6 +386,7 @@ export interface LiteLLMParams {
364
386
  model_info?: LiteLLMParamsModelInfoMap | null;
365
387
  ocr_cost_per_credit?: number | null;
366
388
  ocr_cost_per_page?: number | null;
389
+ ocr_cost_per_page_batches?: number | null;
367
390
  organization?: string | null;
368
391
  otpm?: number | null;
369
392
  output_cost_per_audio_per_second?: number | null;
@@ -380,6 +403,7 @@ export interface LiteLLMParams {
380
403
  output_cost_per_second_1080p?: number | null;
381
404
  output_cost_per_second_480p?: number | null;
382
405
  output_cost_per_second_4k?: number | null;
406
+ output_cost_per_second_720p?: number | null;
383
407
  output_cost_per_token?: number | null;
384
408
  output_cost_per_token_above_128k_tokens?: number | null;
385
409
  output_cost_per_token_above_200k_tokens?: number | null;
@@ -404,12 +428,14 @@ export interface LiteLLMParams {
404
428
  rpm?: number | null;
405
429
  s3_bucket_name?: string | null;
406
430
  s3_encryption_key_id?: string | null;
431
+ s3_endpoint_url?: string | null;
407
432
  s3_output_bucket_name?: string | null;
408
433
  s3_region_name?: string | null;
409
434
  search_context_cost_per_query?: LiteLLMParamsSearchContextCostPerQueryMap | null;
410
435
  stream_timeout?: LiteLLMParamsStreamTimeout | null;
411
436
  tag_regex?: LiteLLMParamsTagRegexList | null;
412
437
  tags?: LiteLLMParamsTagsList | null;
438
+ tenant_id?: string | null;
413
439
  tiered_pricing?: LiteLLMParamsTieredPricingList | null;
414
440
  timeout?: LiteLLMParamsTimeout | null;
415
441
  tpm?: number | null;
@@ -453,6 +479,8 @@ export interface LitellmTypesRouterModelInfo {
453
479
  id: string | null;
454
480
  input_cost_per_character?: number | null;
455
481
  input_cost_per_token?: number | null;
482
+ internal_router_model?: boolean | null;
483
+ member_auto_router?: boolean;
456
484
  output_cost_per_character?: number | null;
457
485
  output_cost_per_token?: number | null;
458
486
  ptu_count?: number | null;
@@ -732,6 +760,10 @@ export interface GetModelInfoV2V2ModelInfoRequest {
732
760
  sortOrder?: string;
733
761
  /** Omit auto-router deployments (litellm model prefixed `auto_router/`). They select among deployments rather than being deployments themselves, so a caller rendering a deployment list can leave them out. Defaults to false, so existing callers are unaffected */
734
762
  exclude_auto_routers?: boolean;
763
+ /** Only return deployments whose `model_info.access_groups` contains this access group */
764
+ access_group?: string;
765
+ /** Only return wildcard deployments, i.e. those whose `model_name` contains `*` */
766
+ wildcard_only?: boolean;
735
767
  }
736
768
  export declare const GetModelInfoV2V2ModelInfoRequest: S.Schema<GetModelInfoV2V2ModelInfoRequest>;
737
769
  export interface GetModelInfoV2V2ModelInfoResponse {
@@ -834,6 +866,8 @@ export type UpdateLiteLLMParamsAdaptiveRouterConfigMap = {
834
866
  [key: string]: unknown | undefined;
835
867
  };
836
868
  export declare const UpdateLiteLLMParamsAdaptiveRouterConfigMap: S.Schema<UpdateLiteLLMParamsAdaptiveRouterConfigMap>;
869
+ export type UpdateLiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
870
+ export declare const UpdateLiteLLMParamsAwsSessionTagsList: S.Schema<UpdateLiteLLMParamsAwsSessionTagsList>;
837
871
  export type UpdateLiteLLMParamsBedrockTagsList = Array<unknown>;
838
872
  export declare const UpdateLiteLLMParamsBedrockTagsList: S.Schema<UpdateLiteLLMParamsBedrockTagsList>;
839
873
  export type UpdateLiteLLMParamsComplexityRouterConfigMap = {
@@ -844,6 +878,8 @@ export type UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem = string | C
844
878
  export declare const UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem: S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem>;
845
879
  export type UpdateLiteLLMParamsConfigurableClientsideAuthParamsList = Array<UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem>;
846
880
  export declare const UpdateLiteLLMParamsConfigurableClientsideAuthParamsList: S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsList>;
881
+ export type UpdateLiteLLMParamsDropParams = boolean | string;
882
+ export declare const UpdateLiteLLMParamsDropParams: S.Schema<UpdateLiteLLMParamsDropParams>;
847
883
  export type UpdateLiteLLMParamsMilvusPartitionNamesList = Array<string>;
848
884
  export declare const UpdateLiteLLMParamsMilvusPartitionNamesList: S.Schema<UpdateLiteLLMParamsMilvusPartitionNamesList>;
849
885
  export type UpdateLiteLLMParamsMockResponse = string | ModelResponse | unknown;
@@ -885,6 +921,7 @@ export interface UpdateLiteLLMParams {
885
921
  adaptive_router_default_model?: string | null;
886
922
  allow_client_keepalive_override?: boolean | null;
887
923
  annotation_cost_per_page?: number | null;
924
+ annotation_cost_per_page_batches?: number | null;
888
925
  api_base?: string | null;
889
926
  api_key?: string | null;
890
927
  api_version?: string | null;
@@ -893,6 +930,8 @@ export interface UpdateLiteLLMParams {
893
930
  auto_router_default_model?: string | null;
894
931
  auto_router_embedding_model?: string | null;
895
932
  auto_router_max_input_chars?: number | null;
933
+ auto_router_model_compression?: string | null;
934
+ auto_router_routing_compression?: string | null;
896
935
  aws_access_key_id?: string | null;
897
936
  aws_batch_role_arn?: string | null;
898
937
  aws_bedrock_project_id?: string | null;
@@ -903,10 +942,14 @@ export interface UpdateLiteLLMParams {
903
942
  aws_role_name?: string | null;
904
943
  aws_secret_access_key?: string | null;
905
944
  aws_session_name?: string | null;
945
+ aws_session_tags?: UpdateLiteLLMParamsAwsSessionTagsList | null;
906
946
  aws_session_token?: string | null;
907
947
  aws_sts_endpoint?: string | null;
908
948
  aws_web_identity_token?: string | null;
909
949
  azure_ad_token?: string | null;
950
+ azure_password?: string | null;
951
+ azure_scope?: string | null;
952
+ azure_username?: string | null;
910
953
  bedrock_tags?: UpdateLiteLLMParamsBedrockTagsList | null;
911
954
  budget_duration?: string | null;
912
955
  cache_creation_input_audio_token_cost?: number | null;
@@ -931,22 +974,27 @@ export interface UpdateLiteLLMParams {
931
974
  cache_read_input_token_cost_priority?: number | null;
932
975
  cache_read_input_token_cost_ultrafast?: number | null;
933
976
  citation_cost_per_token?: number | null;
977
+ client_id?: string | null;
978
+ client_secret?: string | null;
934
979
  complexity_router_config?: UpdateLiteLLMParamsComplexityRouterConfigMap | null;
935
980
  complexity_router_default_model?: string | null;
936
981
  configurable_clientside_auth_params?: UpdateLiteLLMParamsConfigurableClientsideAuthParamsList | null;
937
982
  custom_llm_provider?: string | null;
938
983
  default_api_key_rpm_limit?: number | null;
939
984
  default_api_key_tpm_limit?: number | null;
985
+ drop_params?: UpdateLiteLLMParamsDropParams | null;
940
986
  gcs_bucket_name?: string | null;
941
987
  google_maps_grounding_cost_per_query?: number | null;
942
988
  input_cost_per_audio_per_second?: number | null;
943
989
  input_cost_per_audio_per_second_above_128k_tokens?: number | null;
944
990
  input_cost_per_audio_token?: number | null;
991
+ input_cost_per_audio_token_batches?: number | null;
945
992
  input_cost_per_character?: number | null;
946
993
  input_cost_per_character_above_128k_tokens?: number | null;
947
994
  input_cost_per_image?: number | null;
948
995
  input_cost_per_image_above_128k_tokens?: number | null;
949
996
  input_cost_per_image_token?: number | null;
997
+ input_cost_per_image_token_batches?: number | null;
950
998
  input_cost_per_pixel?: number | null;
951
999
  input_cost_per_query?: number | null;
952
1000
  input_cost_per_second?: number | null;
@@ -968,6 +1016,7 @@ export interface UpdateLiteLLMParams {
968
1016
  input_cost_per_video_per_second_above_15s_interval?: number | null;
969
1017
  input_cost_per_video_per_second_above_8s_interval?: number | null;
970
1018
  input_cost_per_video_token?: number | null;
1019
+ input_cost_per_video_token_batches?: number | null;
971
1020
  itpm?: number | null;
972
1021
  keepalive_seconds?: number | null;
973
1022
  litellm_credential_name?: string | null;
@@ -984,6 +1033,7 @@ export interface UpdateLiteLLMParams {
984
1033
  model_info?: UpdateLiteLLMParamsModelInfoMap | null;
985
1034
  ocr_cost_per_credit?: number | null;
986
1035
  ocr_cost_per_page?: number | null;
1036
+ ocr_cost_per_page_batches?: number | null;
987
1037
  organization?: string | null;
988
1038
  otpm?: number | null;
989
1039
  output_cost_per_audio_per_second?: number | null;
@@ -1000,6 +1050,7 @@ export interface UpdateLiteLLMParams {
1000
1050
  output_cost_per_second_1080p?: number | null;
1001
1051
  output_cost_per_second_480p?: number | null;
1002
1052
  output_cost_per_second_4k?: number | null;
1053
+ output_cost_per_second_720p?: number | null;
1003
1054
  output_cost_per_token?: number | null;
1004
1055
  output_cost_per_token_above_128k_tokens?: number | null;
1005
1056
  output_cost_per_token_above_200k_tokens?: number | null;
@@ -1024,12 +1075,14 @@ export interface UpdateLiteLLMParams {
1024
1075
  rpm?: number | null;
1025
1076
  s3_bucket_name?: string | null;
1026
1077
  s3_encryption_key_id?: string | null;
1078
+ s3_endpoint_url?: string | null;
1027
1079
  s3_output_bucket_name?: string | null;
1028
1080
  s3_region_name?: string | null;
1029
1081
  search_context_cost_per_query?: UpdateLiteLLMParamsSearchContextCostPerQueryMap | null;
1030
1082
  stream_timeout?: UpdateLiteLLMParamsStreamTimeout | null;
1031
1083
  tag_regex?: UpdateLiteLLMParamsTagRegexList | null;
1032
1084
  tags?: UpdateLiteLLMParamsTagsList | null;
1085
+ tenant_id?: string | null;
1033
1086
  tiered_pricing?: UpdateLiteLLMParamsTieredPricingList | null;
1034
1087
  timeout?: UpdateLiteLLMParamsTimeout | null;
1035
1088
  tpm?: number | null;
@@ -1093,7 +1146,7 @@ export interface PostBlockModelModelBlockResponse {
1093
1146
  export declare const PostBlockModelModelBlockResponse: S.Schema<PostBlockModelModelBlockResponse>;
1094
1147
  /** An operator-defined tier: the name the LLM classifier must return and its rubric description. */
1095
1148
  export interface TierDefinition {
1096
- /** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (SIMPLE/MEDIUM/COMPLEX/REASONING), which inherits the built-in criteria when omitted */
1149
+ /** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (NON_REASONING, SIMPLE, MEDIUM, COMPLEX, REASONING), which inherits the built-in criteria when omitted */
1097
1150
  description?: string | null;
1098
1151
  /** Tier name; becomes a value the LLM classifier can return and a key of `tiers` */
1099
1152
  name: string;
@@ -1101,10 +1154,17 @@ export interface TierDefinition {
1101
1154
  export declare const TierDefinition: S.Schema<TierDefinition>;
1102
1155
  export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList = Array<TierDefinition>;
1103
1156
  export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList>;
1157
+ export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap = {
1158
+ [key: string]: string | undefined;
1159
+ };
1160
+ export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap>;
1104
1161
  export interface PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest {
1162
+ classification_examples?: string | null;
1105
1163
  classification_prompt?: string | null;
1164
+ classification_rubric?: ClassificationRubric | (string & {}) | null;
1106
1165
  context_window_size?: number;
1107
- tier_definitions: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList;
1166
+ tier_definitions?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList | null;
1167
+ tier_labels?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap | null;
1108
1168
  }
1109
1169
  export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest>;
1110
1170
  /** When adaptive=True: 'all' scores every pool model with a tier-distance penalty (soft floors); 'classified_tier' Thompson-samples only inside the classified tier's pool */
@@ -1115,26 +1175,93 @@ export interface AdaptiveRouterWeights {
1115
1175
  quality?: number;
1116
1176
  }
1117
1177
  export declare const AdaptiveRouterWeights: S.Schema<AdaptiveRouterWeights>;
1178
+ export interface CapabilityCalibrationConfig {
1179
+ intercept: number;
1180
+ slope: number;
1181
+ version: string;
1182
+ }
1183
+ export declare const CapabilityCalibrationConfig: S.Schema<CapabilityCalibrationConfig>;
1184
+ /** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
1185
+ export type CapabilityClassifierConfigResponseFormat = "json_schema" | "json_object";
1186
+ export declare const CapabilityClassifierConfigResponseFormat: any;
1187
+ /** Switchyard-compatible probability threshold policy for two model tiers. */
1188
+ export interface CapabilityClassifierConfig {
1189
+ /** Lowest p_solve that routes a supported task to efficient_tier */
1190
+ base_threshold: number;
1191
+ /** Optional versioned sigmoid calibration fitted for this judge, capability card, efficient model, and execution setup. Applies sigmoid(slope * logit(clip(p_solve, 1e-6, 1-1e-6)) + intercept) before the threshold policy. Omit to route on the raw forecast. */
1192
+ calibration?: CapabilityCalibrationConfig | null;
1193
+ /** Higher, fail-closed tier used below the adjusted threshold or when the classifier verdict is unavailable */
1194
+ capable_tier: string;
1195
+ /** Tier used when the efficient model's forecasted solve probability meets the adjusted threshold */
1196
+ efficient_tier: string;
1197
+ /** Maximum completion tokens available to the capability classifier verdict */
1198
+ max_output_tokens?: number;
1199
+ /** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
1200
+ response_format?: CapabilityClassifierConfigResponseFormat | (string & {});
1201
+ /** Amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts */
1202
+ threshold_step?: number;
1203
+ }
1204
+ export declare const CapabilityClassifierConfig: S.Schema<CapabilityClassifierConfig>;
1205
+ /** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
1206
+ export type RequestComplexityRouterConfigClassificationMode = "every_request" | "user_turn";
1207
+ export declare const RequestComplexityRouterConfigClassificationMode: any;
1118
1208
  /** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
1119
1209
  export type RequestComplexityRouterConfigClassifierFallback = "heuristic" | "default_model";
1120
1210
  export declare const RequestComplexityRouterConfigClassifierFallback: any;
1211
+ export type ClassifierLLMConfigReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
1212
+ export declare const ClassifierLLMConfigReasoningEffort: any;
1213
+ /** Whether the LLM classifier sees the images on the request it is classifying. Off by default because images cost far more than the text ask they arrive with, and the classifier runs on every request. A turn whose complexity lives in the image ("what is wrong in this stack trace screenshot") is invisible to a text-only classifier, which is what this buys. */
1214
+ export interface ClassifierVisionConfig {
1215
+ /** Forward image content to the classifier. Requires a classifier model declared supports_vision, on the deployment's model_info or in the model cost map; images stay stripped otherwise, so a classifier that cannot read them is never sent one. Declare model_info.supports_vision on the deployment to enable a model the cost map does not describe. Only inline data: URIs are forwarded. A request whose images are http(s) URLs still classifies on its text alone, because some providers fetch such a URL from the proxy rather than the provider, which would let a caller aim a proxy-side request at an address of their choosing. */
1216
+ enabled?: boolean;
1217
+ /** How many images from the newest user turn to forward, in wire order. Bounds the added cost of a turn that attaches many images. Images on earlier turns are never forwarded. */
1218
+ max_images?: number;
1219
+ }
1220
+ export declare const ClassifierVisionConfig: S.Schema<ClassifierVisionConfig>;
1121
1221
  /** Configuration for the LLM-based complexity classifier. */
1122
1222
  export interface ClassifierLLMConfig {
1223
+ /** How long to skip this router's LLM classifier after a classification call times out. Requests use classifier_fallback during the cooldown. When it expires, one request probes the classifier while concurrent requests keep using the fallback; a successful probe closes the circuit and a failed probe restarts the cooldown. */
1224
+ circuit_breaker_cooldown_seconds?: number;
1225
+ /** Whether one classifier timeout temporarily sends requests through classifier_fallback. Enabled by default so an unhealthy classifier cannot repeat its timeout across sessions. */
1226
+ circuit_breaker_enabled?: boolean;
1123
1227
  /** Which calibration examples the built-in rubric carries. 'agentic' anchors routine installs, builds, multi-file edits, and standard debugging at MEDIUM, so ordinary engineering does not route to the most expensive tier; it suits agent, terminal, and coding-assistant traffic as well as mixed traffic. 'chat' omits those engineering anchors, for a deployment serving only conversational traffic. 'business' carries business/sales anchors and business-flavored tier criteria that keep routine drafting and summarizing off the expensive tiers and reserve the top tier for committing to decisions under tradeoffs; it suits sales, support, and go-to-market traffic. Every preset keeps the same four tiers, so this moves where the boundary sits without changing the taxonomy. Leave unset for 'legacy', the rubric as it shipped before calibration examples existed, so an existing router's tier decisions and spend do not move on upgrade. Mutually exclusive with system_prompt, which replaces the rubric this would select. Only applies when classifier_type is 'llm'. */
1124
1228
  classification_rubric?: ClassificationRubric | (string & {}) | null;
1125
1229
  /** Model name (from the router's model_list) to call for classification */
1126
1230
  model: string;
1231
+ /** Reasoning effort override for classifier calls. Leave unset to use the classifier deployment or provider default. */
1232
+ reasoning_effort?: ClassifierLLMConfigReasoningEffort | (string & {}) | null;
1127
1233
  /** Replaces the built-in complexity rubric as the classifier's entire system role. When set, neither the default rubric nor the context-window closing line is appended, so the prompt owns the whole taxonomy and the tier names SIMPLE/MEDIUM/COMPLEX/REASONING become whatever buckets it defines: a prompt that classifies data sensitivity routes on that instead of on difficulty. Two consequences of full replacement. The default rubric's closing paragraph is the classifier's prompt-injection defense, telling it that the caller's quoted system prompt and prior turns are material to judge and never instructions; a replacement that omits it lets a caller ask for a tier and get it. And the heuristic fallback still scores complexity, so a router on some other taxonomy wants classifier_fallback='default_model'. Leave unset for the built-in rubric. Only applies when classifier_type is 'llm'. */
1128
1234
  system_prompt?: string | null;
1129
1235
  /** Timeout budget for the classification call, in milliseconds */
1130
1236
  timeout_ms?: number;
1237
+ /** Whether the classifier sees images on the request, and how many */
1238
+ vision?: ClassifierVisionConfig;
1131
1239
  }
1132
1240
  export declare const ClassifierLLMConfig: S.Schema<ClassifierLLMConfig>;
1133
- /** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin, or 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier */
1134
- export type RequestComplexityRouterConfigClassifierType = "heuristic" | "llm" | "custom" | "heuristic_first";
1241
+ /** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
1242
+ export type RequestComplexityRouterConfigClassifierType = "heuristic" | "heuristic_v2" | "llm" | "capability" | "llm_v2" | "custom" | "heuristic_first" | "hybrid" | "jev";
1135
1243
  export declare const RequestComplexityRouterConfigClassifierType: any;
1136
1244
  export type RequestComplexityRouterConfigCodeKeywordsList = Array<string>;
1137
1245
  export declare const RequestComplexityRouterConfigCodeKeywordsList: S.Schema<RequestComplexityRouterConfigCodeKeywordsList>;
1246
+ export type CustomDimensionKeywordsList = Array<string>;
1247
+ export declare const CustomDimensionKeywordsList: S.Schema<CustomDimensionKeywordsList>;
1248
+ export type CustomDimensionPatternsList = Array<string>;
1249
+ export declare const CustomDimensionPatternsList: S.Schema<CustomDimensionPatternsList>;
1250
+ /** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
1251
+ export type CustomDimensionScoringMode = "binary" | "match_count";
1252
+ export declare const CustomDimensionScoringMode: any;
1253
+ export interface CustomDimension {
1254
+ keywords?: CustomDimensionKeywordsList;
1255
+ name: string;
1256
+ patterns?: CustomDimensionPatternsList;
1257
+ /** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
1258
+ scoring_mode?: CustomDimensionScoringMode | (string & {});
1259
+ weight: number;
1260
+ }
1261
+ export declare const CustomDimension: S.Schema<CustomDimension>;
1262
+ /** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
1263
+ export type RequestComplexityRouterConfigCustomDimensionsList = Array<CustomDimension>;
1264
+ export declare const RequestComplexityRouterConfigCustomDimensionsList: S.Schema<RequestComplexityRouterConfigCustomDimensionsList>;
1138
1265
  export type RequestComplexityRouterConfigCustomTechnicalKeywordsList = Array<string>;
1139
1266
  export declare const RequestComplexityRouterConfigCustomTechnicalKeywordsList: S.Schema<RequestComplexityRouterConfigCustomTechnicalKeywordsList>;
1140
1267
  /** Weights for each scoring dimension */
@@ -1144,8 +1271,76 @@ export type RequestComplexityRouterConfigDimensionWeightsMap = {
1144
1271
  export declare const RequestComplexityRouterConfigDimensionWeightsMap: S.Schema<RequestComplexityRouterConfigDimensionWeightsMap>;
1145
1272
  export type RequestComplexityRouterConfigEscalationKeywordsList = Array<string>;
1146
1273
  export declare const RequestComplexityRouterConfigEscalationKeywordsList: S.Schema<RequestComplexityRouterConfigEscalationKeywordsList>;
1274
+ export interface TierCohortStatistic {
1275
+ cohort: string;
1276
+ observations: number;
1277
+ successes: number;
1278
+ tier: number;
1279
+ }
1280
+ export declare const TierCohortStatistic: S.Schema<TierCohortStatistic>;
1281
+ export type TrainedTierArtifactCohortStatisticsList = Array<TierCohortStatistic>;
1282
+ export declare const TrainedTierArtifactCohortStatisticsList: S.Schema<TrainedTierArtifactCohortStatisticsList>;
1283
+ export interface TierDataset {
1284
+ license: string;
1285
+ name: string;
1286
+ rows: number;
1287
+ success_definition?: string;
1288
+ url: string;
1289
+ }
1290
+ export declare const TierDataset: S.Schema<TierDataset>;
1291
+ export type TrainedTierArtifactDatasetsList = Array<TierDataset>;
1292
+ export declare const TrainedTierArtifactDatasetsList: S.Schema<TrainedTierArtifactDatasetsList>;
1293
+ /** Fixed v0 taxonomy. User-extensible types come in v1. */
1294
+ export type RequestType = "code_generation" | "code_understanding" | "technical_design" | "analytical_reasoning" | "writing" | "factual_lookup" | "general";
1295
+ export declare const RequestType: any;
1296
+ export interface TierDomainStatistic {
1297
+ observations: number;
1298
+ request_type: RequestType | (string & {});
1299
+ successes: number;
1300
+ tier: number;
1301
+ }
1302
+ export declare const TierDomainStatistic: S.Schema<TierDomainStatistic>;
1303
+ export type TrainedTierArtifactDomainStatisticsList = Array<TierDomainStatistic>;
1304
+ export declare const TrainedTierArtifactDomainStatisticsList: S.Schema<TrainedTierArtifactDomainStatisticsList>;
1305
+ export interface TierGlobalStatistic {
1306
+ observations: number;
1307
+ successes: number;
1308
+ tier: number;
1309
+ }
1310
+ export declare const TierGlobalStatistic: S.Schema<TierGlobalStatistic>;
1311
+ export type TrainedTierArtifactGlobalStatisticsList = Array<TierGlobalStatistic>;
1312
+ export declare const TrainedTierArtifactGlobalStatisticsList: S.Schema<TrainedTierArtifactGlobalStatisticsList>;
1313
+ export interface TrainedTierArtifact {
1314
+ cohort_prior_mass?: number;
1315
+ cohort_statistics?: TrainedTierArtifactCohortStatisticsList;
1316
+ datasets?: TrainedTierArtifactDatasetsList;
1317
+ domain_prior_mass?: number;
1318
+ domain_statistics?: TrainedTierArtifactDomainStatisticsList;
1319
+ global_statistics: TrainedTierArtifactGlobalStatisticsList;
1320
+ routing_threshold?: number;
1321
+ schema_version?: number;
1322
+ split_method?: string;
1323
+ success_definition?: string;
1324
+ }
1325
+ export declare const TrainedTierArtifact: S.Schema<TrainedTierArtifact>;
1326
+ /** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
1327
+ export type RequestComplexityRouterConfigHeuristicV2Artifact = TrainedTierArtifact | string;
1328
+ export declare const RequestComplexityRouterConfigHeuristicV2Artifact: S.Schema<RequestComplexityRouterConfigHeuristicV2Artifact>;
1147
1329
  export type RequestComplexityRouterConfigHousekeepingPatternsList = Array<string>;
1148
1330
  export declare const RequestComplexityRouterConfigHousekeepingPatternsList: S.Schema<RequestComplexityRouterConfigHousekeepingPatternsList>;
1331
+ export interface JevClassifierConfig {
1332
+ /** TypeSafe API base, falling back to TYPESAFE_API_BASE and then https://api.typesafe.ai */
1333
+ api_base?: string | null;
1334
+ /** TypeSafe API key, falling back to TYPESAFE_API_KEY */
1335
+ api_key?: string | null;
1336
+ circuit_breaker_cooldown_seconds?: number;
1337
+ circuit_breaker_enabled?: boolean;
1338
+ /** Replaces the built-in Jev question instructions */
1339
+ instructions?: string | null;
1340
+ model?: string;
1341
+ timeout_ms?: number;
1342
+ }
1343
+ export declare const JevClassifierConfig: S.Schema<JevClassifierConfig>;
1149
1344
  /** Keywords/phrases that trigger this rule (lexical or semantic match) */
1150
1345
  export type KeywordTierRuleKeywordsList = Array<string>;
1151
1346
  export declare const KeywordTierRuleKeywordsList: S.Schema<KeywordTierRuleKeywordsList>;
@@ -1159,6 +1354,36 @@ export interface KeywordTierRule {
1159
1354
  export declare const KeywordTierRule: S.Schema<KeywordTierRule>;
1160
1355
  export type RequestComplexityRouterConfigKeywordTierRulesList = Array<KeywordTierRule>;
1161
1356
  export declare const RequestComplexityRouterConfigKeywordTierRulesList: S.Schema<RequestComplexityRouterConfigKeywordTierRulesList>;
1357
+ export interface LLMV2ProbabilityCalibration {
1358
+ intercept: number;
1359
+ slope: number;
1360
+ }
1361
+ export declare const LLMV2ProbabilityCalibration: S.Schema<LLMV2ProbabilityCalibration>;
1362
+ export interface LLMV2Calibration {
1363
+ capable: LLMV2ProbabilityCalibration;
1364
+ efficient: LLMV2ProbabilityCalibration;
1365
+ prompt_version: string;
1366
+ version: string;
1367
+ }
1368
+ export declare const LLMV2Calibration: S.Schema<LLMV2Calibration>;
1369
+ export type LLMV2ConfigResponseFormat = "json_schema" | "json_object";
1370
+ export declare const LLMV2ConfigResponseFormat: any;
1371
+ export interface LLMV2Config {
1372
+ calibration?: LLMV2Calibration | null;
1373
+ capable_profile?: string | null;
1374
+ capable_profile_preset?: string | null;
1375
+ capable_tier?: string;
1376
+ efficient_profile?: string | null;
1377
+ efficient_profile_preset?: string | null;
1378
+ efficient_tier?: string;
1379
+ harness?: string | null;
1380
+ harness_preset?: string | null;
1381
+ max_output_tokens?: number;
1382
+ /** Maximum estimated success loss allowed for efficient. */
1383
+ max_quality_gap: number;
1384
+ response_format?: LLMV2ConfigResponseFormat | (string & {});
1385
+ }
1386
+ export declare const LLMV2Config: S.Schema<LLMV2Config>;
1162
1387
  export type RequestComplexityRouterConfigPlanModePatternsList = Array<string>;
1163
1388
  export declare const RequestComplexityRouterConfigPlanModePatternsList: S.Schema<RequestComplexityRouterConfigPlanModePatternsList>;
1164
1389
  export type RequestComplexityRouterConfigReasoningKeywordsList = Array<string>;
@@ -1226,50 +1451,77 @@ export interface RequestComplexityRouterConfig {
1226
1451
  adaptive_eligible?: RequestComplexityRouterConfigAdaptiveEligible | (string & {});
1227
1452
  /** Quality vs cost weights for adaptive selection (used when adaptive=True) */
1228
1453
  adaptive_weights?: AdaptiveRouterWeights;
1229
- /** Replaces the opening instructions of the LLM classifier rubric (the judging-criteria prose) for a custom tier set. The per-tier bullets and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text are always appended after it and cannot be overridden. Requires tier_definitions; a built-in-tier router customizes its prompt via classifier_llm_config.system_prompt or classification_rubric instead. */
1454
+ /** Probability threshold policy required when classifier_type is 'capability'. The classifier forecasts p_solve for efficient_tier, adjusts base_threshold using the capability-card boundary, and otherwise routes to capable_tier */
1455
+ capability_classifier_config?: CapabilityClassifierConfig | null;
1456
+ /** Replaces the calibration examples of the LLM classifier rubric, and nothing else. Written as example lines only: the router renders the 'Calibration examples:' heading above them, after the per-tier bullets. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_prompt replaces them, the classification instructions; a custom tier set ships no examples of its own, so the section renders only when this is set. */
1457
+ classification_examples?: string | null;
1458
+ /** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
1459
+ classification_mode?: RequestComplexityRouterConfigClassificationMode | (string & {});
1460
+ /** Replaces the classification instructions that open the LLM classifier rubric, and nothing else. The per-tier bullets follow it, the calibration examples follow those, and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text is always appended after them and cannot be overridden. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_examples replaces them, the calibration examples. */
1230
1461
  classification_prompt?: string | null;
1231
- /** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the caller's system prompt sit outside this budget and are always sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
1462
+ /** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
1232
1463
  classifier_context_budget_chars?: number;
1233
1464
  /** Include assistant turns in the classifier context window, so difficulty stated by the model rather than by the user stays visible: a plan the assistant calls complex, which the user approves with 'yes', is classified on the work being approved instead of on the word 'yes'. When enabled, classifier_context_window_size counts the last N turns of the conversation across both roles rather than the last N user turns, and assistant text is sent to the classifier model, which may be a different deployment or provider than the routed completion model. Assistant replies spend classifier_context_budget_chars alongside user turns, so raise it if the oldest turns stop being quoted once replies join the window. Off by default because enabling it shifts tier decisions, and therefore spend, for an already-deployed router. Only applies when classifier_type is 'llm'. */
1234
1465
  classifier_context_include_assistant_turns?: boolean;
1235
1466
  /** Optional cap on each individual prior turn's text, applied before classifier_context_budget_chars bounds the block. Unset by default, so one long turn may spend the whole budget, which is usually what a follow-up needs; set it when no single turn should dominate the context the classifier sees. A capped turn keeps its opening and its ending with the middle elided. Only applies when classifier_type is 'llm'. */
1236
1467
  classifier_context_per_turn_chars?: number | null;
1237
- /** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call already carries the current user ask and the caller's system prompt in full. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
1468
+ /** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
1238
1469
  classifier_context_window_size?: number;
1239
1470
  /** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
1240
1471
  classifier_fallback?: RequestComplexityRouterConfigClassifierFallback | (string & {});
1241
- /** Configuration for the LLM classifier; required when classifier_type is 'llm' or 'heuristic_first' */
1472
+ /** Configuration for the LLM classifier; required when classifier_type is 'llm', 'capability', 'heuristic_first' or 'hybrid' */
1242
1473
  classifier_llm_config?: ClassifierLLMConfig | null;
1243
1474
  /** Not settable over HTTP; the classifier plugin is a runtime object */
1244
1475
  classifier_plugin?: unknown | null;
1245
1476
  /** Timeout budget for the classifier plugin call, in milliseconds. On expiry the fallback path decides the tier. Only applies when classifier_type is 'custom'. */
1246
1477
  classifier_plugin_timeout_ms?: number;
1247
- /** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin, or 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier */
1478
+ /** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
1248
1479
  classifier_type?: RequestComplexityRouterConfigClassifierType | (string & {});
1249
1480
  /** Keywords indicating code-related content */
1250
1481
  code_keywords?: RequestComplexityRouterConfigCodeKeywordsList | null;
1482
+ /** Fraction of a model's declared context window the estimated prompt must fit within. The token count is an estimate, so fitting against the full window would dispatch prompts that the provider's own tokenizer then rejects; 0.95 leaves room for that drift plus the response tokens. */
1483
+ context_window_escalation_buffer?: number;
1484
+ /** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
1485
+ custom_dimensions?: RequestComplexityRouterConfigCustomDimensionsList;
1251
1486
  /** Domain-specific technical keywords appended to the effective base list (technical_keywords if set, otherwise DEFAULT_TECHNICAL_KEYWORDS). Order is preserved; duplicates are removed case-insensitively against the base list and within this list. */
1252
1487
  custom_technical_keywords?: RequestComplexityRouterConfigCustomTechnicalKeywordsList | null;
1253
1488
  /** Default model to use if tier cannot be determined */
1254
1489
  default_model?: string | null;
1255
- /** When True and a session_id is resolvable on the request, pin the deployment chosen inside each routed model group and reuse it whenever the session returns to that group, without pinning which group the session routes to. Independent of session_affinity, which pins the model group instead (and always carries this deployment pin with it): with session_affinity off, every turn is still classified on its own merits while a session that escalates to a stronger tier and comes back still lands on the deployment it used before, which is what keeps a provider prompt cache warm. Pins are held per model group, so switching tiers does not disturb the pin left behind in the previous group. On by default because re-shuffling a conversation across deployments of the same model discards that cache for no benefit; set False to keep every turn load-balanced across the group, which is what a deployment set with tight per-deployment rate limits wants. Inert when no session_id is resolvable, since there is nothing to key a pin on, and suppressed when plugins are configured, for the same reason session_affinity is. */
1490
+ /** When True and a client session_id is resolvable, reuse the session's chosen model for each classified tier and its deployment within each model group. With session_affinity off, every turn is still classified: moving to another tier leaves the previous tier's model pin intact for a later return. Pins yield to current candidate, context, modality, and availability constraints. Adaptive selection chooses the initial model from its eligible pool, then reuses that choice per tier. This reduces avoidable provider prompt-cache misses; it does not guarantee cache hits. Set False to select models and load-balance deployments on every turn, unless session_affinity or user_turn classification requires a pin. Inert without a client session_id and suppressed when plugins are configured. */
1256
1491
  deployment_affinity?: boolean;
1257
1492
  /** Weights for each scoring dimension */
1258
1493
  dimension_weights?: RequestComplexityRouterConfigDimensionWeightsMap;
1259
1494
  /** Embedding model (LiteLLM model name) used when semantic_keyword_matching is enabled */
1260
1495
  embedding_model?: string | null;
1496
+ /** Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Set false to dispatch on complexity alone, as before. */
1497
+ enable_context_window_escalation?: boolean;
1498
+ /** Add NON_REASONING as a fifth built-in tier below SIMPLE, for operational agent traffic that relays or reformats information rather than reasoning about it. Off by default: turning it on adds a rung to this router's ladder, a bullet to the LLM classifier's rubric, and a value the classifier may return, all of which move tier decisions and spend on an already-deployed router. Requires an LLM, Jev, or custom classifier plugin, since the heuristic scorers cannot produce the tier, and a model in `tiers` under the NON_REASONING key. Escalation still walks up from it, and it is never the savings baseline or a `heuristic_v2` prediction. */
1499
+ enable_non_reasoning_tier?: boolean;
1261
1500
  /** Case-sensitive phrases a user can include to force a bump to the next-higher complexity tier when they aren't satisfied with results (they can force a stronger model, but not choose which one). Defaults to ['LITELLM ESCALATE'] when unset; set to an empty list to disable. */
1262
1501
  escalation_keywords?: RequestComplexityRouterConfigEscalationKeywordsList | null;
1263
1502
  /** Tier routed to when the LLM classifier fails (timeout, provider error, or an unparseable reply). Required with tier_definitions and must name a defined tier; the heuristic scorer cannot produce custom tiers, so this replaces the heuristic fallback for custom tier sets. */
1264
1503
  fallback_tier?: string | null;
1265
1504
  /** The highest tier the local scorer may decide on its own; required when classifier_type is 'heuristic_first' and rejected otherwise. A request whose heuristic tier is at or below this one skips the LLM classifier and routes straight to that heuristic tier, so the classifier call is only paid for on traffic the scorer could not place cheaply. The scorer must also have produced at least one signal: a prompt where no dimension fired scores 0.0 and would otherwise land SIMPLE by default rather than by evidence, which is how a chained router would silently send unclassified traffic to the cheapest model. Names a built-in tier, and may not name the highest one, since that would make the LLM classifier unreachable. */
1266
1505
  heuristic_first_max_tier?: string | null;
1506
+ /** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
1507
+ heuristic_v2_artifact?: RequestComplexityRouterConfigHeuristicV2Artifact;
1267
1508
  /** Additional case-sensitive literal sentinels that mark a request as client housekeeping, on top of the built-in conversation-title ones. For clients whose wording the built-ins don't cover, or after a client release changes its strings. */
1268
1509
  housekeeping_patterns?: RequestComplexityRouterConfigHousekeepingPatternsList | null;
1510
+ /** How close to a tier boundary a heuristic score has to land before the LLM classifier breaks the tie; required when classifier_type is 'hybrid' and rejected otherwise. Everything further than this from every active boundary routes on the scorer's own tier with no classifier call, at any tier, which is what separates 'hybrid' from 'heuristic_first' and its cheap-tier ceiling. A prompt where no dimension fired still goes to the classifier, since the scorer has no opinion to be near a boundary with. 0 escalates only scores sitting exactly on a boundary. */
1511
+ hybrid_boundary_margin?: number | null;
1512
+ jev_classifier_config?: JevClassifierConfig | null;
1269
1513
  /** Rules that force a specific tier when their keywords match the prompt */
1270
1514
  keyword_tier_rules?: RequestComplexityRouterConfigKeywordTierRulesList | null;
1515
+ /** Experimental joint task-demand and solver-capability forecasting for classifier_type llm_v2. */
1516
+ llm_v2_config?: LLMV2Config | null;
1271
1517
  /** Minimum cosine similarity for a semantic keyword match */
1272
1518
  match_threshold?: number;
1519
+ /** Set max_tokens on every routed request to the output ceiling of the tier model it lands on, replacing whatever the caller sent. A caller behind an auto-router cannot pick one value that fits every tier: the smallest tier's ceiling starves a bigger tier's thinking budget, and a bigger tier's ceiling is rejected by the smallest. The ceiling is the smallest max_output_tokens across the tier model's deployments, read from each deployment's model_info and then the model cost map; a tier model with a deployment whose ceiling is unknown keeps the caller's value. A max_tokens, max_completion_tokens or max_output_tokens in the tier's own litellm_params still wins. Set false to forward the caller's value unchanged. */
1520
+ max_tokens_from_tier_model?: boolean;
1521
+ /** Let modality_routing replace a kept session-affinity pin on the turns that carry an image. Without this, a session pinned to a text-only model fails every image turn with a provider 400, since the pin is exempt from the modality gate. When enabled, such a turn routes to a capable model for that request only and the stored pin is left untouched, so the next text turn replays the session's own model; the override is reported as cause modality_pin_override and is never itself pinned. Inert unless modality_routing is also enabled. */
1522
+ modality_pin_override?: boolean;
1523
+ /** Route image-bearing requests only to models that can accept image input. The classifier reads text alone, so an image request whose text classifies cheap otherwise lands on a text-only model and fails with a provider 400. When enabled, a routed model explicitly declared supports_vision false (deployment model_info or the model cost map; unmapped names stay routable) is replaced by the nearest HIGHER tier holding a capable model, then default_model, else a clear 400. A kept session-affinity pin still wins even when an image arrives, unless modality_pin_override is also enabled. */
1524
+ modality_routing?: boolean;
1273
1525
  /** When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to at least this tier: the classified tier still wins when it is higher, and the floor also overrides a session-affinity pin to a lower tier for exactly the turns carrying the sentinel, without rewriting the pin -- the first turn after plan mode exits routes as if plan mode had never happened. Names a built-in tier, or with tier_definitions set, one of the defined tier names (list order is ascending severity, same as keyword_tier_rules). Unset disables detection entirely. The sentinels ride in client-injected prompt text, so a caller who pastes one can spend up to this tier's models -- never down, and never outside the configured pools. */
1274
1526
  plan_mode_min_tier?: string | null;
1275
1527
  /** Additional case-sensitive literal sentinels that mark a request as plan mode, on top of the built-in Claude Code and Copilot ones. For clients whose plan-mode wording the built-ins don't cover, or after a client release changes its strings. */
@@ -1280,7 +1532,7 @@ export interface RequestComplexityRouterConfig {
1280
1532
  reasoning_keywords?: RequestComplexityRouterConfigReasoningKeywordsList | null;
1281
1533
  /** Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a request the scorer placed in the cheapest tier; 0 restores the unconditional override */
1282
1534
  reasoning_override_min_score?: number | null;
1283
- /** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in default of ('<system-reminder>', '</system-reminder>'), so a harness that also emits that pair lists it too. Matching is case-insensitive. */
1535
+ /** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in system-reminder pair and the Codex envelope pairs enabled for Codex user agents, so list every built-in pair your harness also emits. Matching is case-insensitive. */
1284
1536
  reminder_markers?: RequestComplexityRouterConfigReminderMarkersList | null;
1285
1537
  /** Return the resolved raw model name in the response model field instead of the client-requested complexity-router alias */
1286
1538
  return_raw_model_name?: boolean;
@@ -1290,15 +1542,21 @@ export interface RequestComplexityRouterConfig {
1290
1542
  semantic_keyword_matching?: boolean;
1291
1543
  /** When True and a session_id is resolvable on the request, pin the model chosen on the session's first turn and reuse it for every later turn, skipping re-classification. Off by default so every turn is classified on its own merits and routed to the cheapest adequate tier. Set True to keep a multi-turn session on one model, which preserves provider prompt caches and avoids cross-model conversation-history errors. Always implies the deployment pin regardless of deployment_affinity: the session sticks to one deployment of the pinned model, since freezing the model while re-shuffling its deployments would still go cache-cold. */
1292
1544
  session_affinity?: boolean;
1293
- /** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity deployment pin, so it measures idle time for the session's routing decisions rather than total session length */
1545
+ /** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity per-tier model and deployment pins, so it measures idle time for the session's routing decisions rather than total session length */
1294
1546
  session_affinity_ttl_seconds?: number;
1295
1547
  /** Keywords indicating simple/basic queries */
1296
1548
  simple_keywords?: RequestComplexityRouterConfigSimpleKeywordsList | null;
1549
+ /** Escalate mid-task to the next-higher configured tier when the assistant's own recent tool calls look stuck: the newest tool call repeats, or errors, at least stall_escalation_repeat_threshold times across the last stall_escalation_window calls. Both tests are anchored on the newest call, so a task that tried the same thing a few times and then moved on is not escalated on the strength of those older calls alone, while a retry loop broken up by an unrelated lookup still counts. One tier at most, on the same ladder escalation_keywords bumps along, and never above the highest configured tier. Detection re-runs on every classified turn from the tool calls visible in that request, so it needs no state and nothing survives past the task. Mutually exclusive with session_affinity and classification_mode='user_turn', which both replay a held routing decision instead of classifying most turns, so this would never see the tool calls to look at. Off by default. */
1550
+ stall_escalation_enabled?: boolean;
1551
+ /** How many of the last stall_escalation_window tool calls must repeat the newest call, or must have errored alongside it, before the task counts as stalled. Must not exceed stall_escalation_window, or the condition could never be reached. */
1552
+ stall_escalation_repeat_threshold?: number;
1553
+ /** How many of the assistant's most recent tool calls stall detection looks at, oldest ones dropped as new calls happen. Counted across the whole visible conversation rather than reset at the newest human ask, so evidence from before a plain follow-up message like 'try again' is still visible on the turn after it. */
1554
+ stall_escalation_window?: number;
1297
1555
  /** Keywords indicating technical content */
1298
1556
  technical_keywords?: RequestComplexityRouterConfigTechnicalKeywordsList | null;
1299
1557
  /** Score boundaries between tiers. These keys (simple_medium, medium_complex, complex_reasoning) name the gaps between the default tier names and are not renameable by tier_labels; they are scorer knobs persisted by name on every routing decision */
1300
1558
  tier_boundaries?: RequestComplexityRouterConfigTierBoundariesMap;
1301
- /** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
1559
+ /** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm', 'jev' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
1302
1560
  tier_definitions?: RequestComplexityRouterConfigTierDefinitionsList | null;
1303
1561
  /** Score penalty per tier-step away from the classified tier when adaptive=True */
1304
1562
  tier_distance_penalty?: number;
@@ -1351,8 +1609,23 @@ export interface PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest {
1351
1609
  tools?: PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequestToolsList | null;
1352
1610
  }
1353
1611
  export declare const PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest: S.Schema<PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest>;
1354
- export type StandardLoggingRoutingDecisionCause = "heuristic_scorer" | "reasoning_override" | "llm_classifier" | "heuristic_first_short_circuit" | "classifier_plugin" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "session_affinity_pin" | "session_affinity_escalation" | "default_fallback" | "keyword" | "quality_tier" | "bandit";
1612
+ export type StandardLoggingRoutingDecisionCause = "heuristic_scorer" | "heuristic_v2" | "reasoning_override" | "llm_classifier" | "capability_classifier" | "jev_classifier" | "llm_v2_classifier" | "llm_v2_fallback" | "heuristic_first_short_circuit" | "hybrid_short_circuit" | "classifier_plugin" | "classifier_fallback" | "capability_classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "modality_escalation" | "modality_pin_override" | "health_failover" | "health_default_fallback" | "session_affinity_pin" | "session_affinity_escalation" | "user_turn_continuation" | "default_fallback" | "keyword" | "quality_tier" | "bandit";
1355
1613
  export declare const StandardLoggingRoutingDecisionCause: any;
1614
+ export type StandardLoggingRoutingDecisionClassifierProbabilitiesMap = {
1615
+ [key: string]: number | undefined;
1616
+ };
1617
+ export declare const StandardLoggingRoutingDecisionClassifierProbabilitiesMap: S.Schema<StandardLoggingRoutingDecisionClassifierProbabilitiesMap>;
1618
+ export type StandardLoggingHeuristicV2ForecastProbabilitiesMap = {
1619
+ [key: string]: number | undefined;
1620
+ };
1621
+ export declare const StandardLoggingHeuristicV2ForecastProbabilitiesMap: S.Schema<StandardLoggingHeuristicV2ForecastProbabilitiesMap>;
1622
+ export interface StandardLoggingHeuristicV2Forecast {
1623
+ predicted_tier: string;
1624
+ probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap;
1625
+ request_type: string;
1626
+ threshold: number;
1627
+ }
1628
+ export declare const StandardLoggingHeuristicV2Forecast: S.Schema<StandardLoggingHeuristicV2Forecast>;
1356
1629
  export type StandardLoggingRoutingDecisionRouterType = "complexity" | "adaptive" | "quality";
1357
1630
  export declare const StandardLoggingRoutingDecisionRouterType: any;
1358
1631
  export type StandardLoggingRoutingDecisionSignalsList = Array<string>;
@@ -1371,11 +1644,29 @@ export declare const StandardLoggingRoutingDecisionTierLitellmParamsMap: S.Schem
1371
1644
  /** Per-request provenance for a pre-routing strategy (auto-router) decision. */
1372
1645
  export interface StandardLoggingRoutingDecision {
1373
1646
  cause?: StandardLoggingRoutingDecisionCause;
1647
+ classifier_calibrated_capable_p_solve?: number;
1648
+ classifier_calibrated_efficient_p_solve?: number;
1649
+ classifier_calibrated_p_solve?: number;
1650
+ classifier_calibration_version?: string;
1651
+ classifier_capability_boundary?: string;
1652
+ classifier_capable_p_solve?: number;
1653
+ classifier_confidence?: number;
1374
1654
  classifier_cost?: number;
1655
+ classifier_crux?: string;
1656
+ classifier_efficient_p_solve?: number;
1657
+ classifier_max_quality_gap?: number;
1375
1658
  classifier_model?: string;
1659
+ classifier_p_solve?: number;
1660
+ classifier_primary_rule?: string;
1661
+ classifier_probabilities?: StandardLoggingRoutingDecisionClassifierProbabilitiesMap;
1662
+ classifier_prompt_version?: string;
1663
+ classifier_threshold?: number;
1664
+ context_escalated?: boolean;
1665
+ context_escalation_original_tier?: string;
1376
1666
  conversation_continuing?: boolean;
1377
1667
  escalated?: boolean;
1378
1668
  escalation_keyword?: string;
1669
+ heuristic_v2_forecast?: StandardLoggingHeuristicV2Forecast;
1379
1670
  matched_keyword?: string;
1380
1671
  reasoning_override_min_score?: number;
1381
1672
  request_type?: string;
@@ -1551,7 +1842,7 @@ export type GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetError
1551
1842
  /** Get Model Cost Map Reload Status ADMIN ONLY / MASTER KEY Only Endpoint Get the status of the scheduled model cost map reload job. */
1552
1843
  export declare const getModelCostMapReloadStatusScheduleModelCostMapReloadStatusGet: API.OperationMethod<GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetRequest, GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetResponse, GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetError, LitellmOpContext>;
1553
1844
  export type GetModelCostMapSourceModelCostMapSourceGetError = LitellmOpError;
1554
- /** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - model_count: number of models in the currently loaded cost map */
1845
+ /** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - loaded_at: when this pod last loaded the map - source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it - etag: the ETag of the remote fetch (null for the bundled backup) - model_count: number of models in the currently loaded cost map */
1555
1846
  export declare const getModelCostMapSourceModelCostMapSourceGet: API.OperationMethod<GetModelCostMapSourceModelCostMapSourceGetRequest, GetModelCostMapSourceModelCostMapSourceGetResponse, GetModelCostMapSourceModelCostMapSourceGetError, LitellmOpContext>;
1556
1847
  export type GetModelDeprecationsModelDeprecationError = UnprocessableEntity | LitellmOpError;
1557
1848
  /** Model Deprecations List models with known deprecation/sunset dates, bucketed by urgency. Reads `deprecation_date` metadata from `model_prices_and_context_window.json` (and any per-deployment `model_info.deprecation_date` overrides) for the models configured on this proxy. Parameters: warn_within_days: Window (in days) used to bucket "imminent" models, 30 by default. Returns: A payload with three lists of `ModelDeprecationInfo` entries: - `deprecated`: deprecation date is in the past, so these requests may fail at any time. - `imminent`: deprecation date is within `warn_within_days` from today. - `upcoming`: deprecation date is further out. Example: ```shell curl -X GET 'http://localhost:4000/model/deprecations' \ -H 'Authorization: Bearer sk-1234' ``` */
@@ -1566,16 +1857,16 @@ export type GetModelInfoModelsModelIdError = UnprocessableEntity | LitellmOpErro
1566
1857
  /** Model Info Retrieve information about a specific model accessible to your API key. Returns model details only if the model is available to your API key/team. Returns 404 if the model doesn't exist or is not accessible. Follows OpenAI API specification for individual model retrieval. https://platform.openai.com/docs/api-reference/models/retrieve Query parameters mirror `/v1/models` so the same caller context (team scoping, health filtering, paused deployments) drives both endpoints; the listing's public id must resolve to the same internal deployment here. */
1567
1858
  export declare const getModelInfoModelsModelId: API.OperationMethod<GetModelInfoModelsModelIdRequest, GetModelInfoModelsModelIdResponse, GetModelInfoModelsModelIdError, LitellmOpContext>;
1568
1859
  export type GetModelInfoV1ModelInfoError = UnprocessableEntity | LitellmOpError;
1569
- /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: Returns a dictionary containing information about each model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
1860
+ /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
1570
1861
  export declare const getModelInfoV1ModelInfo: API.OperationMethod<GetModelInfoV1ModelInfoRequest, GetModelInfoV1ModelInfoResponse, GetModelInfoV1ModelInfoError, LitellmOpContext>;
1571
1862
  export type GetModelInfoV1ModelsModelIdError = UnprocessableEntity | LitellmOpError;
1572
1863
  /** Model Info Retrieve information about a specific model accessible to your API key. Returns model details only if the model is available to your API key/team. Returns 404 if the model doesn't exist or is not accessible. Follows OpenAI API specification for individual model retrieval. https://platform.openai.com/docs/api-reference/models/retrieve Query parameters mirror `/v1/models` so the same caller context (team scoping, health filtering, paused deployments) drives both endpoints; the listing's public id must resolve to the same internal deployment here. */
1573
1864
  export declare const getModelInfoV1ModelsModelId: API.OperationMethod<GetModelInfoV1ModelsModelIdRequest, GetModelInfoV1ModelsModelIdResponse, GetModelInfoV1ModelsModelIdError, LitellmOpContext>;
1574
1865
  export type GetModelInfoV1V1ModelInfoError = UnprocessableEntity | LitellmOpError;
1575
- /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: Returns a dictionary containing information about each model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
1866
+ /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
1576
1867
  export declare const getModelInfoV1V1ModelInfo: API.OperationMethod<GetModelInfoV1V1ModelInfoRequest, GetModelInfoV1V1ModelInfoResponse, GetModelInfoV1V1ModelInfoError, LitellmOpContext>;
1577
1868
  export type GetModelInfoV2V2ModelInfoError = UnprocessableEntity | LitellmOpError;
1578
- /** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
1869
+ /** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. access_group: Only return deployments in this model access group. wildcard_only: Only return deployments whose `model_name` contains `*`. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
1579
1870
  export declare const getModelInfoV2V2ModelInfo: API.OperationMethod<GetModelInfoV2V2ModelInfoRequest, GetModelInfoV2V2ModelInfoResponse, GetModelInfoV2V2ModelInfoError, LitellmOpContext>;
1580
1871
  export type GetModelMetricsExceptionsModelMetricsExceptionError = UnprocessableEntity | LitellmOpError;
1581
1872
  /** Model Metrics Exceptions View number of failed requests per model on config.yaml */
@@ -1644,6 +1935,6 @@ export type UpdateUsefulLinksModelHubUpdateUsefulLinksPostError = UnprocessableE
1644
1935
  /** Update Useful Links Update useful links */
1645
1936
  export declare const updateUsefulLinksModelHubUpdateUsefulLinksPost: API.OperationMethod<UpdateUsefulLinksModelHubUpdateUsefulLinksPostRequest, UpdateUsefulLinksModelHubUpdateUsefulLinksPostResponse, UpdateUsefulLinksModelHubUpdateUsefulLinksPostError, LitellmOpContext>;
1646
1937
  export type ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError = UnprocessableEntity | LitellmOpError;
1647
- /** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Gated exactly like the save it rehearses: a proxy admin, or a team admin naming their own team. Nothing is created, routed, or billed. */
1938
+ /** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Uses the same team opt-in and model-access checks as configuration writes for members. Nothing is created, routed, or billed. */
1648
1939
  export declare const validateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPost: API.OperationMethod<ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostRequest, ComplexityRouterConfigValidationResponse, ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError, LitellmOpContext>;
1649
1940
  //# sourceMappingURL=model_management.d.ts.map