@homeflare/distilled-litellm 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (390) hide show
  1. package/README.md +41 -12
  2. package/dist/services/a2a.js +1 -1
  3. package/dist/services/a2a_registration.js +1 -1
  4. package/dist/services/access_groups.d.ts +18 -0
  5. package/dist/services/access_groups.d.ts.map +1 -1
  6. package/dist/services/access_groups.js +15 -1
  7. package/dist/services/access_groups.js.map +1 -1
  8. package/dist/services/adaptive_router.js +1 -1
  9. package/dist/services/agents.d.ts +10 -0
  10. package/dist/services/agents.d.ts.map +1 -1
  11. package/dist/services/agents.js +10 -1
  12. package/dist/services/agents.js.map +1 -1
  13. package/dist/services/alerting.js +1 -1
  14. package/dist/services/anthropic_passthrough.js +1 -1
  15. package/dist/services/anthropic_skills.d.ts +7 -1
  16. package/dist/services/anthropic_skills.d.ts.map +1 -1
  17. package/dist/services/anthropic_skills.js +6 -2
  18. package/dist/services/anthropic_skills.js.map +1 -1
  19. package/dist/services/assistants.js +1 -1
  20. package/dist/services/audio.js +1 -1
  21. package/dist/services/audit_logging.d.ts +2 -0
  22. package/dist/services/audit_logging.d.ts.map +1 -1
  23. package/dist/services/audit_logging.js +2 -1
  24. package/dist/services/audit_logging.js.map +1 -1
  25. package/dist/services/auto_router.d.ts +162 -62
  26. package/dist/services/auto_router.d.ts.map +1 -1
  27. package/dist/services/auto_router.js +92 -31
  28. package/dist/services/auto_router.js.map +1 -1
  29. package/dist/services/batch.js +1 -1
  30. package/dist/services/beta_agents.js +1 -1
  31. package/dist/services/beta_mcp.js +1 -1
  32. package/dist/services/budget_management.d.ts +8 -3
  33. package/dist/services/budget_management.d.ts.map +1 -1
  34. package/dist/services/budget_management.js +7 -4
  35. package/dist/services/budget_management.js.map +1 -1
  36. package/dist/services/budget_spend_tracking.d.ts +24 -5
  37. package/dist/services/budget_spend_tracking.d.ts.map +1 -1
  38. package/dist/services/budget_spend_tracking.js +17 -4
  39. package/dist/services/budget_spend_tracking.js.map +1 -1
  40. package/dist/services/cache_settings.js +1 -1
  41. package/dist/services/caching.js +1 -1
  42. package/dist/services/chat_completions.js +1 -1
  43. package/dist/services/claude_code_marketplace.d.ts +11 -10
  44. package/dist/services/claude_code_marketplace.d.ts.map +1 -1
  45. package/dist/services/claude_code_marketplace.js +8 -6
  46. package/dist/services/claude_code_marketplace.js.map +1 -1
  47. package/dist/services/cloudzero.js +1 -1
  48. package/dist/services/completions.js +1 -1
  49. package/dist/services/compliance.js +1 -1
  50. package/dist/services/config_overrides.d.ts +5 -1
  51. package/dist/services/config_overrides.d.ts.map +1 -1
  52. package/dist/services/config_overrides.js +3 -1
  53. package/dist/services/config_overrides.js.map +1 -1
  54. package/dist/services/config_yaml.d.ts +120 -6
  55. package/dist/services/config_yaml.d.ts.map +1 -1
  56. package/dist/services/config_yaml.js +67 -1
  57. package/dist/services/config_yaml.js.map +1 -1
  58. package/dist/services/containers.d.ts +8 -2
  59. package/dist/services/containers.d.ts.map +1 -1
  60. package/dist/services/containers.js +13 -5
  61. package/dist/services/containers.js.map +1 -1
  62. package/dist/services/coordination_redis_settings.js +1 -1
  63. package/dist/services/cost_tracking.d.ts +91 -1
  64. package/dist/services/cost_tracking.d.ts.map +1 -1
  65. package/dist/services/cost_tracking.js +82 -2
  66. package/dist/services/cost_tracking.js.map +1 -1
  67. package/dist/services/credential_management.d.ts +2 -1
  68. package/dist/services/credential_management.d.ts.map +1 -1
  69. package/dist/services/credential_management.js +3 -2
  70. package/dist/services/credential_management.js.map +1 -1
  71. package/dist/services/customer_management.d.ts +25 -2
  72. package/dist/services/customer_management.d.ts.map +1 -1
  73. package/dist/services/customer_management.js +22 -3
  74. package/dist/services/customer_management.js.map +1 -1
  75. package/dist/services/email_management.js +1 -1
  76. package/dist/services/embeddings.js +1 -1
  77. package/dist/services/evals.js +1 -1
  78. package/dist/services/experimental.js +1 -1
  79. package/dist/services/fallback_management.js +1 -1
  80. package/dist/services/files.js +1 -1
  81. package/dist/services/fine_tuning.js +1 -1
  82. package/dist/services/gemini_agents.js +1 -1
  83. package/dist/services/google_genai_endpoints.js +1 -1
  84. package/dist/services/guardrails.d.ts +80 -7
  85. package/dist/services/guardrails.d.ts.map +1 -1
  86. package/dist/services/guardrails.js +37 -1
  87. package/dist/services/guardrails.js.map +1 -1
  88. package/dist/services/health.d.ts +2 -2
  89. package/dist/services/health.d.ts.map +1 -1
  90. package/dist/services/health.js +2 -2
  91. package/dist/services/health.js.map +1 -1
  92. package/dist/services/images.js +1 -1
  93. package/dist/services/index.d.ts +1 -15
  94. package/dist/services/index.d.ts.map +1 -1
  95. package/dist/services/index.js +1 -15
  96. package/dist/services/index.js.map +1 -1
  97. package/dist/services/internal_user_management.d.ts +227 -39
  98. package/dist/services/internal_user_management.d.ts.map +1 -1
  99. package/dist/services/internal_user_management.js +194 -37
  100. package/dist/services/internal_user_management.js.map +1 -1
  101. package/dist/services/invite_links.js +1 -1
  102. package/dist/services/jwt_mappings.d.ts +3 -0
  103. package/dist/services/jwt_mappings.d.ts.map +1 -1
  104. package/dist/services/jwt_mappings.js +4 -1
  105. package/dist/services/jwt_mappings.js.map +1 -1
  106. package/dist/services/key_management.d.ts +93 -49
  107. package/dist/services/key_management.d.ts.map +1 -1
  108. package/dist/services/key_management.js +75 -39
  109. package/dist/services/key_management.js.map +1 -1
  110. package/dist/services/langfuse_passthrough.js +1 -1
  111. package/dist/services/llm_passthrough.d.ts +1199 -0
  112. package/dist/services/llm_passthrough.d.ts.map +1 -0
  113. package/dist/services/llm_passthrough.js +2150 -0
  114. package/dist/services/llm_passthrough.js.map +1 -0
  115. package/dist/services/llm_utils.js +1 -1
  116. package/dist/services/logging_callbacks.js +1 -1
  117. package/dist/services/mcp_byok_oauth.d.ts +18 -11
  118. package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
  119. package/dist/services/mcp_byok_oauth.js +27 -24
  120. package/dist/services/mcp_byok_oauth.js.map +1 -1
  121. package/dist/services/mcp_discoverable.d.ts +34 -0
  122. package/dist/services/mcp_discoverable.d.ts.map +1 -1
  123. package/dist/services/mcp_discoverable.js +51 -1
  124. package/dist/services/mcp_discoverable.js.map +1 -1
  125. package/dist/services/mcp_management.d.ts +90 -2
  126. package/dist/services/mcp_management.d.ts.map +1 -1
  127. package/dist/services/mcp_management.js +115 -3
  128. package/dist/services/mcp_management.js.map +1 -1
  129. package/dist/services/mcp_rest.d.ts +13 -0
  130. package/dist/services/mcp_rest.d.ts.map +1 -1
  131. package/dist/services/mcp_rest.js +10 -2
  132. package/dist/services/mcp_rest.js.map +1 -1
  133. package/dist/services/memory_management.d.ts +2 -0
  134. package/dist/services/memory_management.d.ts.map +1 -1
  135. package/dist/services/memory_management.js +2 -1
  136. package/dist/services/memory_management.js.map +1 -1
  137. package/dist/services/misc.d.ts +86 -1
  138. package/dist/services/misc.d.ts.map +1 -1
  139. package/dist/services/misc.js +126 -2
  140. package/dist/services/misc.js.map +1 -1
  141. package/dist/services/model_management.d.ts +310 -19
  142. package/dist/services/model_management.d.ts.map +1 -1
  143. package/dist/services/model_management.js +240 -7
  144. package/dist/services/model_management.js.map +1 -1
  145. package/dist/services/moderations.js +1 -1
  146. package/dist/services/ocr.js +1 -1
  147. package/dist/services/open_ai_pass_through.d.ts +35 -90
  148. package/dist/services/open_ai_pass_through.d.ts.map +1 -1
  149. package/dist/services/open_ai_pass_through.js +41 -139
  150. package/dist/services/open_ai_pass_through.js.map +1 -1
  151. package/dist/services/organization_management.d.ts +21 -1
  152. package/dist/services/organization_management.d.ts.map +1 -1
  153. package/dist/services/organization_management.js +20 -2
  154. package/dist/services/organization_management.js.map +1 -1
  155. package/dist/services/plugins.js +1 -1
  156. package/dist/services/policies.d.ts +2 -0
  157. package/dist/services/policies.d.ts.map +1 -1
  158. package/dist/services/policies.js +3 -1
  159. package/dist/services/policies.js.map +1 -1
  160. package/dist/services/policy_engine.d.ts +6 -0
  161. package/dist/services/policy_engine.d.ts.map +1 -1
  162. package/dist/services/policy_engine.js +4 -1
  163. package/dist/services/policy_engine.js.map +1 -1
  164. package/dist/services/project_management.d.ts +15 -0
  165. package/dist/services/project_management.d.ts.map +1 -1
  166. package/dist/services/project_management.js +14 -1
  167. package/dist/services/project_management.js.map +1 -1
  168. package/dist/services/prompts.js +1 -1
  169. package/dist/services/public.d.ts +124 -0
  170. package/dist/services/public.d.ts.map +1 -1
  171. package/dist/services/public.js +158 -1
  172. package/dist/services/public.js.map +1 -1
  173. package/dist/services/rag.js +1 -1
  174. package/dist/services/realtime.d.ts +4 -26
  175. package/dist/services/realtime.d.ts.map +1 -1
  176. package/dist/services/realtime.js +7 -57
  177. package/dist/services/realtime.js.map +1 -1
  178. package/dist/services/rerank.d.ts +72 -0
  179. package/dist/services/rerank.d.ts.map +1 -1
  180. package/dist/services/rerank.js +61 -4
  181. package/dist/services/rerank.js.map +1 -1
  182. package/dist/services/responses.d.ts +30 -0
  183. package/dist/services/responses.d.ts.map +1 -1
  184. package/dist/services/responses.js +59 -1
  185. package/dist/services/responses.js.map +1 -1
  186. package/dist/services/router_settings.js +1 -1
  187. package/dist/services/rust_control_plane.js +1 -1
  188. package/dist/services/scim.d.ts +41 -1
  189. package/dist/services/scim.d.ts.map +1 -1
  190. package/dist/services/scim.js +60 -2
  191. package/dist/services/scim.js.map +1 -1
  192. package/dist/services/search.js +1 -1
  193. package/dist/services/search_tools.js +1 -1
  194. package/dist/services/settings.d.ts +88 -0
  195. package/dist/services/settings.d.ts.map +1 -1
  196. package/dist/services/settings.js +113 -1
  197. package/dist/services/settings.js.map +1 -1
  198. package/dist/services/sso_settings.d.ts +1 -1
  199. package/dist/services/sso_settings.d.ts.map +1 -1
  200. package/dist/services/sso_settings.js +1 -1
  201. package/dist/services/sso_settings.js.map +1 -1
  202. package/dist/services/tag_management.d.ts +7 -0
  203. package/dist/services/tag_management.d.ts.map +1 -1
  204. package/dist/services/tag_management.js +8 -1
  205. package/dist/services/tag_management.js.map +1 -1
  206. package/dist/services/team_management.d.ts +265 -17
  207. package/dist/services/team_management.d.ts.map +1 -1
  208. package/dist/services/team_management.js +247 -12
  209. package/dist/services/team_management.js.map +1 -1
  210. package/dist/services/tools.js +1 -1
  211. package/dist/services/ui_settings.js +1 -1
  212. package/dist/services/ui_theme_settings.js +1 -1
  213. package/dist/services/usage_ai.js +1 -1
  214. package/dist/services/vantage.js +1 -1
  215. package/dist/services/vector_store_management.js +1 -1
  216. package/dist/services/vector_stores.js +1 -1
  217. package/dist/services/videos.js +1 -1
  218. package/dist/services/web_socket.d.ts +0 -18
  219. package/dist/services/web_socket.d.ts.map +1 -1
  220. package/dist/services/web_socket.js +1 -33
  221. package/dist/services/web_socket.js.map +1 -1
  222. package/dist/services/workflow_management.js +1 -1
  223. package/package.json +1 -1
  224. package/src/services/a2a.ts +1 -1
  225. package/src/services/a2a_registration.ts +1 -1
  226. package/src/services/access_groups.ts +44 -1
  227. package/src/services/adaptive_router.ts +1 -1
  228. package/src/services/agents.ts +22 -1
  229. package/src/services/alerting.ts +1 -1
  230. package/src/services/anthropic_passthrough.ts +1 -1
  231. package/src/services/anthropic_skills.ts +12 -2
  232. package/src/services/assistants.ts +1 -1
  233. package/src/services/audio.ts +1 -1
  234. package/src/services/audit_logging.ts +4 -1
  235. package/src/services/auto_router.ts +303 -93
  236. package/src/services/batch.ts +1 -1
  237. package/src/services/beta_agents.ts +1 -1
  238. package/src/services/beta_mcp.ts +1 -1
  239. package/src/services/budget_management.ts +12 -4
  240. package/src/services/budget_spend_tracking.ts +38 -6
  241. package/src/services/cache_settings.ts +1 -1
  242. package/src/services/caching.ts +1 -1
  243. package/src/services/chat_completions.ts +1 -1
  244. package/src/services/claude_code_marketplace.ts +20 -14
  245. package/src/services/cloudzero.ts +1 -1
  246. package/src/services/completions.ts +1 -1
  247. package/src/services/compliance.ts +1 -1
  248. package/src/services/config_overrides.ts +8 -2
  249. package/src/services/config_yaml.ts +251 -6
  250. package/src/services/containers.ts +29 -11
  251. package/src/services/coordination_redis_settings.ts +1 -1
  252. package/src/services/cost_tracking.ts +198 -2
  253. package/src/services/credential_management.ts +9 -4
  254. package/src/services/customer_management.ts +49 -3
  255. package/src/services/email_management.ts +1 -1
  256. package/src/services/embeddings.ts +1 -1
  257. package/src/services/evals.ts +1 -1
  258. package/src/services/experimental.ts +1 -1
  259. package/src/services/fallback_management.ts +1 -1
  260. package/src/services/files.ts +1 -1
  261. package/src/services/fine_tuning.ts +1 -1
  262. package/src/services/gemini_agents.ts +1 -1
  263. package/src/services/google_genai_endpoints.ts +1 -1
  264. package/src/services/guardrails.ts +198 -8
  265. package/src/services/health.ts +3 -2
  266. package/src/services/images.ts +1 -1
  267. package/src/services/index.ts +1 -15
  268. package/src/services/internal_user_management.ts +565 -103
  269. package/src/services/invite_links.ts +1 -1
  270. package/src/services/jwt_mappings.ts +7 -1
  271. package/src/services/key_management.ts +229 -126
  272. package/src/services/langfuse_passthrough.ts +1 -1
  273. package/src/services/llm_passthrough.ts +4542 -0
  274. package/src/services/llm_utils.ts +1 -1
  275. package/src/services/logging_callbacks.ts +1 -1
  276. package/src/services/mcp_byok_oauth.ts +65 -45
  277. package/src/services/mcp_discoverable.ts +109 -1
  278. package/src/services/mcp_management.ts +258 -3
  279. package/src/services/mcp_rest.ts +30 -5
  280. package/src/services/memory_management.ts +4 -1
  281. package/src/services/misc.ts +269 -2
  282. package/src/services/model_management.ts +666 -21
  283. package/src/services/moderations.ts +1 -1
  284. package/src/services/ocr.ts +1 -1
  285. package/src/services/open_ai_pass_through.ts +81 -286
  286. package/src/services/organization_management.ts +44 -2
  287. package/src/services/plugins.ts +1 -1
  288. package/src/services/policies.ts +5 -1
  289. package/src/services/policy_engine.ts +10 -1
  290. package/src/services/project_management.ts +33 -1
  291. package/src/services/prompts.ts +1 -1
  292. package/src/services/public.ts +356 -1
  293. package/src/services/rag.ts +1 -1
  294. package/src/services/realtime.ts +11 -111
  295. package/src/services/rerank.ts +187 -7
  296. package/src/services/responses.ts +117 -1
  297. package/src/services/router_settings.ts +1 -1
  298. package/src/services/rust_control_plane.ts +1 -1
  299. package/src/services/scim.ts +134 -3
  300. package/src/services/search.ts +1 -1
  301. package/src/services/search_tools.ts +1 -1
  302. package/src/services/settings.ts +278 -1
  303. package/src/services/sso_settings.ts +2 -1
  304. package/src/services/tag_management.ts +15 -1
  305. package/src/services/team_management.ts +676 -37
  306. package/src/services/tools.ts +1 -1
  307. package/src/services/ui_settings.ts +1 -1
  308. package/src/services/ui_theme_settings.ts +1 -1
  309. package/src/services/usage_ai.ts +1 -1
  310. package/src/services/vantage.ts +1 -1
  311. package/src/services/vector_store_management.ts +1 -1
  312. package/src/services/vector_stores.ts +1 -1
  313. package/src/services/videos.ts +1 -1
  314. package/src/services/web_socket.ts +1 -61
  315. package/src/services/workflow_management.ts +1 -1
  316. package/dist/services/anthropic_pass_through.d.ts +0 -70
  317. package/dist/services/anthropic_pass_through.d.ts.map +0 -1
  318. package/dist/services/anthropic_pass_through.js +0 -114
  319. package/dist/services/anthropic_pass_through.js.map +0 -1
  320. package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
  321. package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
  322. package/dist/services/assembly_ai_eu_pass_through.js +0 -114
  323. package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
  324. package/dist/services/assembly_ai_pass_through.d.ts +0 -70
  325. package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
  326. package/dist/services/assembly_ai_pass_through.js +0 -114
  327. package/dist/services/assembly_ai_pass_through.js.map +0 -1
  328. package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
  329. package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
  330. package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
  331. package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
  332. package/dist/services/azure_ai_pass_through.d.ts +0 -70
  333. package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
  334. package/dist/services/azure_ai_pass_through.js +0 -112
  335. package/dist/services/azure_ai_pass_through.js.map +0 -1
  336. package/dist/services/azure_pass_through.d.ts +0 -70
  337. package/dist/services/azure_pass_through.d.ts.map +0 -1
  338. package/dist/services/azure_pass_through.js +0 -107
  339. package/dist/services/azure_pass_through.js.map +0 -1
  340. package/dist/services/bedrock_pass_through.d.ts +0 -70
  341. package/dist/services/bedrock_pass_through.d.ts.map +0 -1
  342. package/dist/services/bedrock_pass_through.js +0 -114
  343. package/dist/services/bedrock_pass_through.js.map +0 -1
  344. package/dist/services/cohere_pass_through.d.ts +0 -70
  345. package/dist/services/cohere_pass_through.d.ts.map +0 -1
  346. package/dist/services/cohere_pass_through.js +0 -112
  347. package/dist/services/cohere_pass_through.js.map +0 -1
  348. package/dist/services/cursor_pass_through.d.ts +0 -70
  349. package/dist/services/cursor_pass_through.d.ts.map +0 -1
  350. package/dist/services/cursor_pass_through.js +0 -112
  351. package/dist/services/cursor_pass_through.js.map +0 -1
  352. package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
  353. package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
  354. package/dist/services/google_ai_studio_pass_through.js +0 -112
  355. package/dist/services/google_ai_studio_pass_through.js.map +0 -1
  356. package/dist/services/milvus_pass_through.d.ts +0 -70
  357. package/dist/services/milvus_pass_through.d.ts.map +0 -1
  358. package/dist/services/milvus_pass_through.js +0 -112
  359. package/dist/services/milvus_pass_through.js.map +0 -1
  360. package/dist/services/mistral_pass_through.d.ts +0 -70
  361. package/dist/services/mistral_pass_through.d.ts.map +0 -1
  362. package/dist/services/mistral_pass_through.js +0 -114
  363. package/dist/services/mistral_pass_through.js.map +0 -1
  364. package/dist/services/vertex_ai_pass_through.d.ts +0 -180
  365. package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
  366. package/dist/services/vertex_ai_pass_through.js +0 -334
  367. package/dist/services/vertex_ai_pass_through.js.map +0 -1
  368. package/dist/services/vllm_pass_through.d.ts +0 -70
  369. package/dist/services/vllm_pass_through.d.ts.map +0 -1
  370. package/dist/services/vllm_pass_through.js +0 -104
  371. package/dist/services/vllm_pass_through.js.map +0 -1
  372. package/dist/services/watsonx_pass_through.d.ts +0 -70
  373. package/dist/services/watsonx_pass_through.d.ts.map +0 -1
  374. package/dist/services/watsonx_pass_through.js +0 -114
  375. package/dist/services/watsonx_pass_through.js.map +0 -1
  376. package/src/services/anthropic_pass_through.ts +0 -233
  377. package/src/services/assembly_ai_eu_pass_through.ts +0 -237
  378. package/src/services/assembly_ai_pass_through.ts +0 -237
  379. package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
  380. package/src/services/azure_ai_pass_through.ts +0 -231
  381. package/src/services/azure_pass_through.ts +0 -227
  382. package/src/services/bedrock_pass_through.ts +0 -229
  383. package/src/services/cohere_pass_through.ts +0 -227
  384. package/src/services/cursor_pass_through.ts +0 -227
  385. package/src/services/google_ai_studio_pass_through.ts +0 -227
  386. package/src/services/milvus_pass_through.ts +0 -227
  387. package/src/services/mistral_pass_through.ts +0 -229
  388. package/src/services/vertex_ai_pass_through.ts +0 -684
  389. package/src/services/vllm_pass_through.ts +0 -227
  390. package/src/services/watsonx_pass_through.ts +0 -229
@@ -1,4 +1,4 @@
1
- // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.100.0 — see README). Do not edit.
1
+ // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
2
2
  import * as S from "@distilled.cloud/core/schema";
3
3
  import * as API from "@distilled.cloud/core/api";
4
4
  import * as C from "@distilled.cloud/core/category";
@@ -38,6 +38,22 @@ export const LiteLLMParamsAdaptiveRouterConfigMap = /*@__PURE__*/ S.Record(
38
38
  S.Unknown,
39
39
  ) as any as S.Schema<LiteLLMParamsAdaptiveRouterConfigMap>;
40
40
 
41
+ export interface AwsSessionTag {
42
+ Key: string;
43
+ Value: string;
44
+ }
45
+ export const AwsSessionTag = /*@__PURE__*/ S.suspend(() =>
46
+ S.Struct({
47
+ Key: S.String,
48
+ Value: S.String,
49
+ }),
50
+ ).annotate({ identifier: "AwsSessionTag" }) as any as S.Schema<AwsSessionTag>;
51
+
52
+ export type LiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
53
+ export const LiteLLMParamsAwsSessionTagsList = /*@__PURE__*/ S.Array(
54
+ AwsSessionTag,
55
+ ) as any as S.Schema<LiteLLMParamsAwsSessionTagsList>;
56
+
41
57
  export type LiteLLMParamsBedrockTagsList = Array<unknown>;
42
58
  export const LiteLLMParamsBedrockTagsList = /*@__PURE__*/ S.Array(
43
59
  S.Unknown,
@@ -76,6 +92,10 @@ export const LiteLLMParamsConfigurableClientsideAuthParamsList =
76
92
  LiteLLMParamsConfigurableClientsideAuthParamsItem,
77
93
  ) as any as S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsList>;
78
94
 
95
+ export type LiteLLMParamsDropParams = boolean | string;
96
+ export const LiteLLMParamsDropParams =
97
+ S.Unknown as any as S.Schema<LiteLLMParamsDropParams>;
98
+
79
99
  export type LiteLLMParamsMilvusPartitionNamesList = Array<string>;
80
100
  export const LiteLLMParamsMilvusPartitionNamesList = /*@__PURE__*/ S.Array(
81
101
  S.String,
@@ -603,6 +623,7 @@ export interface LiteLLMParams {
603
623
  adaptive_router_default_model?: string | null;
604
624
  allow_client_keepalive_override?: boolean | null;
605
625
  annotation_cost_per_page?: number | null;
626
+ annotation_cost_per_page_batches?: number | null;
606
627
  api_base?: string | null;
607
628
  api_key?: string | null;
608
629
  api_version?: string | null;
@@ -611,6 +632,8 @@ export interface LiteLLMParams {
611
632
  auto_router_default_model?: string | null;
612
633
  auto_router_embedding_model?: string | null;
613
634
  auto_router_max_input_chars?: number | null;
635
+ auto_router_model_compression?: string | null;
636
+ auto_router_routing_compression?: string | null;
614
637
  aws_access_key_id?: string | null;
615
638
  aws_batch_role_arn?: string | null;
616
639
  aws_bedrock_project_id?: string | null;
@@ -621,10 +644,14 @@ export interface LiteLLMParams {
621
644
  aws_role_name?: string | null;
622
645
  aws_secret_access_key?: string | null;
623
646
  aws_session_name?: string | null;
647
+ aws_session_tags?: LiteLLMParamsAwsSessionTagsList | null;
624
648
  aws_session_token?: string | null;
625
649
  aws_sts_endpoint?: string | null;
626
650
  aws_web_identity_token?: string | null;
627
651
  azure_ad_token?: string | null;
652
+ azure_password?: string | null;
653
+ azure_scope?: string | null;
654
+ azure_username?: string | null;
628
655
  bedrock_tags?: LiteLLMParamsBedrockTagsList | null;
629
656
  budget_duration?: string | null;
630
657
  cache_creation_input_audio_token_cost?: number | null;
@@ -649,22 +676,27 @@ export interface LiteLLMParams {
649
676
  cache_read_input_token_cost_priority?: number | null;
650
677
  cache_read_input_token_cost_ultrafast?: number | null;
651
678
  citation_cost_per_token?: number | null;
679
+ client_id?: string | null;
680
+ client_secret?: string | null;
652
681
  complexity_router_config?: LiteLLMParamsComplexityRouterConfigMap | null;
653
682
  complexity_router_default_model?: string | null;
654
683
  configurable_clientside_auth_params?: LiteLLMParamsConfigurableClientsideAuthParamsList | null;
655
684
  custom_llm_provider?: string | null;
656
685
  default_api_key_rpm_limit?: number | null;
657
686
  default_api_key_tpm_limit?: number | null;
687
+ drop_params?: LiteLLMParamsDropParams | null;
658
688
  gcs_bucket_name?: string | null;
659
689
  google_maps_grounding_cost_per_query?: number | null;
660
690
  input_cost_per_audio_per_second?: number | null;
661
691
  input_cost_per_audio_per_second_above_128k_tokens?: number | null;
662
692
  input_cost_per_audio_token?: number | null;
693
+ input_cost_per_audio_token_batches?: number | null;
663
694
  input_cost_per_character?: number | null;
664
695
  input_cost_per_character_above_128k_tokens?: number | null;
665
696
  input_cost_per_image?: number | null;
666
697
  input_cost_per_image_above_128k_tokens?: number | null;
667
698
  input_cost_per_image_token?: number | null;
699
+ input_cost_per_image_token_batches?: number | null;
668
700
  input_cost_per_pixel?: number | null;
669
701
  input_cost_per_query?: number | null;
670
702
  input_cost_per_second?: number | null;
@@ -686,6 +718,7 @@ export interface LiteLLMParams {
686
718
  input_cost_per_video_per_second_above_15s_interval?: number | null;
687
719
  input_cost_per_video_per_second_above_8s_interval?: number | null;
688
720
  input_cost_per_video_token?: number | null;
721
+ input_cost_per_video_token_batches?: number | null;
689
722
  itpm?: number | null;
690
723
  keepalive_seconds?: number | null;
691
724
  litellm_credential_name?: string | null;
@@ -702,6 +735,7 @@ export interface LiteLLMParams {
702
735
  model_info?: LiteLLMParamsModelInfoMap | null;
703
736
  ocr_cost_per_credit?: number | null;
704
737
  ocr_cost_per_page?: number | null;
738
+ ocr_cost_per_page_batches?: number | null;
705
739
  organization?: string | null;
706
740
  otpm?: number | null;
707
741
  output_cost_per_audio_per_second?: number | null;
@@ -718,6 +752,7 @@ export interface LiteLLMParams {
718
752
  output_cost_per_second_1080p?: number | null;
719
753
  output_cost_per_second_480p?: number | null;
720
754
  output_cost_per_second_4k?: number | null;
755
+ output_cost_per_second_720p?: number | null;
721
756
  output_cost_per_token?: number | null;
722
757
  output_cost_per_token_above_128k_tokens?: number | null;
723
758
  output_cost_per_token_above_200k_tokens?: number | null;
@@ -742,12 +777,14 @@ export interface LiteLLMParams {
742
777
  rpm?: number | null;
743
778
  s3_bucket_name?: string | null;
744
779
  s3_encryption_key_id?: string | null;
780
+ s3_endpoint_url?: string | null;
745
781
  s3_output_bucket_name?: string | null;
746
782
  s3_region_name?: string | null;
747
783
  search_context_cost_per_query?: LiteLLMParamsSearchContextCostPerQueryMap | null;
748
784
  stream_timeout?: LiteLLMParamsStreamTimeout | null;
749
785
  tag_regex?: LiteLLMParamsTagRegexList | null;
750
786
  tags?: LiteLLMParamsTagsList | null;
787
+ tenant_id?: string | null;
751
788
  tiered_pricing?: LiteLLMParamsTieredPricingList | null;
752
789
  timeout?: LiteLLMParamsTimeout | null;
753
790
  tpm?: number | null;
@@ -776,6 +813,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
776
813
  adaptive_router_default_model: S.optional(S.NullOr(S.String)),
777
814
  allow_client_keepalive_override: S.optional(S.NullOr(S.Boolean)),
778
815
  annotation_cost_per_page: S.optional(S.NullOr(S.Number)),
816
+ annotation_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
779
817
  api_base: S.optional(S.NullOr(S.String)),
780
818
  api_key: S.optional(S.NullOr(S.String)),
781
819
  api_version: S.optional(S.NullOr(S.String)),
@@ -784,6 +822,8 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
784
822
  auto_router_default_model: S.optional(S.NullOr(S.String)),
785
823
  auto_router_embedding_model: S.optional(S.NullOr(S.String)),
786
824
  auto_router_max_input_chars: S.optional(S.NullOr(S.Number)),
825
+ auto_router_model_compression: S.optional(S.NullOr(S.String)),
826
+ auto_router_routing_compression: S.optional(S.NullOr(S.String)),
787
827
  aws_access_key_id: S.optional(S.NullOr(S.String)),
788
828
  aws_batch_role_arn: S.optional(S.NullOr(S.String)),
789
829
  aws_bedrock_project_id: S.optional(S.NullOr(S.String)),
@@ -794,10 +834,14 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
794
834
  aws_role_name: S.optional(S.NullOr(S.String)),
795
835
  aws_secret_access_key: S.optional(S.NullOr(S.String)),
796
836
  aws_session_name: S.optional(S.NullOr(S.String)),
837
+ aws_session_tags: S.optional(S.NullOr(LiteLLMParamsAwsSessionTagsList)),
797
838
  aws_session_token: S.optional(S.NullOr(S.String)),
798
839
  aws_sts_endpoint: S.optional(S.NullOr(S.String)),
799
840
  aws_web_identity_token: S.optional(S.NullOr(S.String)),
800
841
  azure_ad_token: S.optional(S.NullOr(S.String)),
842
+ azure_password: S.optional(S.NullOr(S.String)),
843
+ azure_scope: S.optional(S.NullOr(S.String)),
844
+ azure_username: S.optional(S.NullOr(S.String)),
801
845
  bedrock_tags: S.optional(S.NullOr(LiteLLMParamsBedrockTagsList)),
802
846
  budget_duration: S.optional(S.NullOr(S.String)),
803
847
  cache_creation_input_audio_token_cost: S.optional(S.NullOr(S.Number)),
@@ -842,6 +886,8 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
842
886
  cache_read_input_token_cost_priority: S.optional(S.NullOr(S.Number)),
843
887
  cache_read_input_token_cost_ultrafast: S.optional(S.NullOr(S.Number)),
844
888
  citation_cost_per_token: S.optional(S.NullOr(S.Number)),
889
+ client_id: S.optional(S.NullOr(S.String)),
890
+ client_secret: S.optional(S.NullOr(S.String)),
845
891
  complexity_router_config: S.optional(
846
892
  S.NullOr(LiteLLMParamsComplexityRouterConfigMap),
847
893
  ),
@@ -852,6 +898,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
852
898
  custom_llm_provider: S.optional(S.NullOr(S.String)),
853
899
  default_api_key_rpm_limit: S.optional(S.NullOr(S.Number)),
854
900
  default_api_key_tpm_limit: S.optional(S.NullOr(S.Number)),
901
+ drop_params: S.optional(S.NullOr(LiteLLMParamsDropParams)),
855
902
  gcs_bucket_name: S.optional(S.NullOr(S.String)),
856
903
  google_maps_grounding_cost_per_query: S.optional(S.NullOr(S.Number)),
857
904
  input_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
@@ -859,11 +906,13 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
859
906
  S.NullOr(S.Number),
860
907
  ),
861
908
  input_cost_per_audio_token: S.optional(S.NullOr(S.Number)),
909
+ input_cost_per_audio_token_batches: S.optional(S.NullOr(S.Number)),
862
910
  input_cost_per_character: S.optional(S.NullOr(S.Number)),
863
911
  input_cost_per_character_above_128k_tokens: S.optional(S.NullOr(S.Number)),
864
912
  input_cost_per_image: S.optional(S.NullOr(S.Number)),
865
913
  input_cost_per_image_above_128k_tokens: S.optional(S.NullOr(S.Number)),
866
914
  input_cost_per_image_token: S.optional(S.NullOr(S.Number)),
915
+ input_cost_per_image_token_batches: S.optional(S.NullOr(S.Number)),
867
916
  input_cost_per_pixel: S.optional(S.NullOr(S.Number)),
868
917
  input_cost_per_query: S.optional(S.NullOr(S.Number)),
869
918
  input_cost_per_second: S.optional(S.NullOr(S.Number)),
@@ -895,6 +944,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
895
944
  S.NullOr(S.Number),
896
945
  ),
897
946
  input_cost_per_video_token: S.optional(S.NullOr(S.Number)),
947
+ input_cost_per_video_token_batches: S.optional(S.NullOr(S.Number)),
898
948
  itpm: S.optional(S.NullOr(S.Number)),
899
949
  keepalive_seconds: S.optional(S.NullOr(S.Number)),
900
950
  litellm_credential_name: S.optional(S.NullOr(S.String)),
@@ -913,6 +963,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
913
963
  model_info: S.optional(S.NullOr(LiteLLMParamsModelInfoMap)),
914
964
  ocr_cost_per_credit: S.optional(S.NullOr(S.Number)),
915
965
  ocr_cost_per_page: S.optional(S.NullOr(S.Number)),
966
+ ocr_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
916
967
  organization: S.optional(S.NullOr(S.String)),
917
968
  otpm: S.optional(S.NullOr(S.Number)),
918
969
  output_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
@@ -929,6 +980,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
929
980
  output_cost_per_second_1080p: S.optional(S.NullOr(S.Number)),
930
981
  output_cost_per_second_480p: S.optional(S.NullOr(S.Number)),
931
982
  output_cost_per_second_4k: S.optional(S.NullOr(S.Number)),
983
+ output_cost_per_second_720p: S.optional(S.NullOr(S.Number)),
932
984
  output_cost_per_token: S.optional(S.NullOr(S.Number)),
933
985
  output_cost_per_token_above_128k_tokens: S.optional(S.NullOr(S.Number)),
934
986
  output_cost_per_token_above_200k_tokens: S.optional(S.NullOr(S.Number)),
@@ -961,6 +1013,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
961
1013
  rpm: S.optional(S.NullOr(S.Number)),
962
1014
  s3_bucket_name: S.optional(S.NullOr(S.String)),
963
1015
  s3_encryption_key_id: S.optional(S.NullOr(S.String)),
1016
+ s3_endpoint_url: S.optional(S.NullOr(S.String)),
964
1017
  s3_output_bucket_name: S.optional(S.NullOr(S.String)),
965
1018
  s3_region_name: S.optional(S.NullOr(S.String)),
966
1019
  search_context_cost_per_query: S.optional(
@@ -969,6 +1022,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
969
1022
  stream_timeout: S.optional(S.NullOr(LiteLLMParamsStreamTimeout)),
970
1023
  tag_regex: S.optional(S.NullOr(LiteLLMParamsTagRegexList)),
971
1024
  tags: S.optional(S.NullOr(LiteLLMParamsTagsList)),
1025
+ tenant_id: S.optional(S.NullOr(S.String)),
972
1026
  tiered_pricing: S.optional(S.NullOr(LiteLLMParamsTieredPricingList)),
973
1027
  timeout: S.optional(S.NullOr(LiteLLMParamsTimeout)),
974
1028
  tpm: S.optional(S.NullOr(S.Number)),
@@ -1023,6 +1077,8 @@ export interface LitellmTypesRouterModelInfo {
1023
1077
  id: string | null;
1024
1078
  input_cost_per_character?: number | null;
1025
1079
  input_cost_per_token?: number | null;
1080
+ internal_router_model?: boolean | null;
1081
+ member_auto_router?: boolean;
1026
1082
  output_cost_per_character?: number | null;
1027
1083
  output_cost_per_token?: number | null;
1028
1084
  ptu_count?: number | null;
@@ -1050,6 +1106,8 @@ export const LitellmTypesRouterModelInfo = /*@__PURE__*/ S.suspend(() =>
1050
1106
  id: S.NullOr(S.String),
1051
1107
  input_cost_per_character: S.optional(S.NullOr(S.Number)),
1052
1108
  input_cost_per_token: S.optional(S.NullOr(S.Number)),
1109
+ internal_router_model: S.optional(S.NullOr(S.Boolean)),
1110
+ member_auto_router: S.optional(S.Boolean),
1053
1111
  output_cost_per_character: S.optional(S.NullOr(S.Number)),
1054
1112
  output_cost_per_token: S.optional(S.NullOr(S.Number)),
1055
1113
  ptu_count: S.optional(S.NullOr(S.Number)),
@@ -1772,6 +1830,10 @@ export interface GetModelInfoV2V2ModelInfoRequest {
1772
1830
  sortOrder?: string;
1773
1831
  /** Omit auto-router deployments (litellm model prefixed `auto_router/`). They select among deployments rather than being deployments themselves, so a caller rendering a deployment list can leave them out. Defaults to false, so existing callers are unaffected */
1774
1832
  exclude_auto_routers?: boolean;
1833
+ /** Only return deployments whose `model_info.access_groups` contains this access group */
1834
+ access_group?: string;
1835
+ /** Only return wildcard deployments, i.e. those whose `model_name` contains `*` */
1836
+ wildcard_only?: boolean;
1775
1837
  }
1776
1838
  export const GetModelInfoV2V2ModelInfoRequest = /*@__PURE__*/ S.suspend(() =>
1777
1839
  S.Struct({
@@ -1787,6 +1849,8 @@ export const GetModelInfoV2V2ModelInfoRequest = /*@__PURE__*/ S.suspend(() =>
1787
1849
  sortBy: S.optional(S.String.pipe(T.Query())),
1788
1850
  sortOrder: S.optional(S.String.pipe(T.Query())),
1789
1851
  exclude_auto_routers: S.optional(S.Boolean.pipe(T.Query())),
1852
+ access_group: S.optional(S.String.pipe(T.Query())),
1853
+ wildcard_only: S.optional(S.Boolean.pipe(T.Query())),
1790
1854
  }).pipe(T.Http({ method: "GET", uri: "/v2/model/info", code: 200 })),
1791
1855
  ).annotate({
1792
1856
  identifier: "GetModelInfoV2V2ModelInfoRequest",
@@ -2064,6 +2128,11 @@ export const UpdateLiteLLMParamsAdaptiveRouterConfigMap =
2064
2128
  S.Unknown,
2065
2129
  ) as any as S.Schema<UpdateLiteLLMParamsAdaptiveRouterConfigMap>;
2066
2130
 
2131
+ export type UpdateLiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
2132
+ export const UpdateLiteLLMParamsAwsSessionTagsList = /*@__PURE__*/ S.Array(
2133
+ AwsSessionTag,
2134
+ ) as any as S.Schema<UpdateLiteLLMParamsAwsSessionTagsList>;
2135
+
2067
2136
  export type UpdateLiteLLMParamsBedrockTagsList = Array<unknown>;
2068
2137
  export const UpdateLiteLLMParamsBedrockTagsList = /*@__PURE__*/ S.Array(
2069
2138
  S.Unknown,
@@ -2091,6 +2160,10 @@ export const UpdateLiteLLMParamsConfigurableClientsideAuthParamsList =
2091
2160
  UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem,
2092
2161
  ) as any as S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsList>;
2093
2162
 
2163
+ export type UpdateLiteLLMParamsDropParams = boolean | string;
2164
+ export const UpdateLiteLLMParamsDropParams =
2165
+ S.Unknown as any as S.Schema<UpdateLiteLLMParamsDropParams>;
2166
+
2094
2167
  export type UpdateLiteLLMParamsMilvusPartitionNamesList = Array<string>;
2095
2168
  export const UpdateLiteLLMParamsMilvusPartitionNamesList =
2096
2169
  /*@__PURE__*/ S.Array(
@@ -2178,6 +2251,7 @@ export interface UpdateLiteLLMParams {
2178
2251
  adaptive_router_default_model?: string | null;
2179
2252
  allow_client_keepalive_override?: boolean | null;
2180
2253
  annotation_cost_per_page?: number | null;
2254
+ annotation_cost_per_page_batches?: number | null;
2181
2255
  api_base?: string | null;
2182
2256
  api_key?: string | null;
2183
2257
  api_version?: string | null;
@@ -2186,6 +2260,8 @@ export interface UpdateLiteLLMParams {
2186
2260
  auto_router_default_model?: string | null;
2187
2261
  auto_router_embedding_model?: string | null;
2188
2262
  auto_router_max_input_chars?: number | null;
2263
+ auto_router_model_compression?: string | null;
2264
+ auto_router_routing_compression?: string | null;
2189
2265
  aws_access_key_id?: string | null;
2190
2266
  aws_batch_role_arn?: string | null;
2191
2267
  aws_bedrock_project_id?: string | null;
@@ -2196,10 +2272,14 @@ export interface UpdateLiteLLMParams {
2196
2272
  aws_role_name?: string | null;
2197
2273
  aws_secret_access_key?: string | null;
2198
2274
  aws_session_name?: string | null;
2275
+ aws_session_tags?: UpdateLiteLLMParamsAwsSessionTagsList | null;
2199
2276
  aws_session_token?: string | null;
2200
2277
  aws_sts_endpoint?: string | null;
2201
2278
  aws_web_identity_token?: string | null;
2202
2279
  azure_ad_token?: string | null;
2280
+ azure_password?: string | null;
2281
+ azure_scope?: string | null;
2282
+ azure_username?: string | null;
2203
2283
  bedrock_tags?: UpdateLiteLLMParamsBedrockTagsList | null;
2204
2284
  budget_duration?: string | null;
2205
2285
  cache_creation_input_audio_token_cost?: number | null;
@@ -2224,22 +2304,27 @@ export interface UpdateLiteLLMParams {
2224
2304
  cache_read_input_token_cost_priority?: number | null;
2225
2305
  cache_read_input_token_cost_ultrafast?: number | null;
2226
2306
  citation_cost_per_token?: number | null;
2307
+ client_id?: string | null;
2308
+ client_secret?: string | null;
2227
2309
  complexity_router_config?: UpdateLiteLLMParamsComplexityRouterConfigMap | null;
2228
2310
  complexity_router_default_model?: string | null;
2229
2311
  configurable_clientside_auth_params?: UpdateLiteLLMParamsConfigurableClientsideAuthParamsList | null;
2230
2312
  custom_llm_provider?: string | null;
2231
2313
  default_api_key_rpm_limit?: number | null;
2232
2314
  default_api_key_tpm_limit?: number | null;
2315
+ drop_params?: UpdateLiteLLMParamsDropParams | null;
2233
2316
  gcs_bucket_name?: string | null;
2234
2317
  google_maps_grounding_cost_per_query?: number | null;
2235
2318
  input_cost_per_audio_per_second?: number | null;
2236
2319
  input_cost_per_audio_per_second_above_128k_tokens?: number | null;
2237
2320
  input_cost_per_audio_token?: number | null;
2321
+ input_cost_per_audio_token_batches?: number | null;
2238
2322
  input_cost_per_character?: number | null;
2239
2323
  input_cost_per_character_above_128k_tokens?: number | null;
2240
2324
  input_cost_per_image?: number | null;
2241
2325
  input_cost_per_image_above_128k_tokens?: number | null;
2242
2326
  input_cost_per_image_token?: number | null;
2327
+ input_cost_per_image_token_batches?: number | null;
2243
2328
  input_cost_per_pixel?: number | null;
2244
2329
  input_cost_per_query?: number | null;
2245
2330
  input_cost_per_second?: number | null;
@@ -2261,6 +2346,7 @@ export interface UpdateLiteLLMParams {
2261
2346
  input_cost_per_video_per_second_above_15s_interval?: number | null;
2262
2347
  input_cost_per_video_per_second_above_8s_interval?: number | null;
2263
2348
  input_cost_per_video_token?: number | null;
2349
+ input_cost_per_video_token_batches?: number | null;
2264
2350
  itpm?: number | null;
2265
2351
  keepalive_seconds?: number | null;
2266
2352
  litellm_credential_name?: string | null;
@@ -2277,6 +2363,7 @@ export interface UpdateLiteLLMParams {
2277
2363
  model_info?: UpdateLiteLLMParamsModelInfoMap | null;
2278
2364
  ocr_cost_per_credit?: number | null;
2279
2365
  ocr_cost_per_page?: number | null;
2366
+ ocr_cost_per_page_batches?: number | null;
2280
2367
  organization?: string | null;
2281
2368
  otpm?: number | null;
2282
2369
  output_cost_per_audio_per_second?: number | null;
@@ -2293,6 +2380,7 @@ export interface UpdateLiteLLMParams {
2293
2380
  output_cost_per_second_1080p?: number | null;
2294
2381
  output_cost_per_second_480p?: number | null;
2295
2382
  output_cost_per_second_4k?: number | null;
2383
+ output_cost_per_second_720p?: number | null;
2296
2384
  output_cost_per_token?: number | null;
2297
2385
  output_cost_per_token_above_128k_tokens?: number | null;
2298
2386
  output_cost_per_token_above_200k_tokens?: number | null;
@@ -2317,12 +2405,14 @@ export interface UpdateLiteLLMParams {
2317
2405
  rpm?: number | null;
2318
2406
  s3_bucket_name?: string | null;
2319
2407
  s3_encryption_key_id?: string | null;
2408
+ s3_endpoint_url?: string | null;
2320
2409
  s3_output_bucket_name?: string | null;
2321
2410
  s3_region_name?: string | null;
2322
2411
  search_context_cost_per_query?: UpdateLiteLLMParamsSearchContextCostPerQueryMap | null;
2323
2412
  stream_timeout?: UpdateLiteLLMParamsStreamTimeout | null;
2324
2413
  tag_regex?: UpdateLiteLLMParamsTagRegexList | null;
2325
2414
  tags?: UpdateLiteLLMParamsTagsList | null;
2415
+ tenant_id?: string | null;
2326
2416
  tiered_pricing?: UpdateLiteLLMParamsTieredPricingList | null;
2327
2417
  timeout?: UpdateLiteLLMParamsTimeout | null;
2328
2418
  tpm?: number | null;
@@ -2351,6 +2441,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2351
2441
  adaptive_router_default_model: S.optional(S.NullOr(S.String)),
2352
2442
  allow_client_keepalive_override: S.optional(S.NullOr(S.Boolean)),
2353
2443
  annotation_cost_per_page: S.optional(S.NullOr(S.Number)),
2444
+ annotation_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
2354
2445
  api_base: S.optional(S.NullOr(S.String)),
2355
2446
  api_key: S.optional(S.NullOr(S.String)),
2356
2447
  api_version: S.optional(S.NullOr(S.String)),
@@ -2359,6 +2450,8 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2359
2450
  auto_router_default_model: S.optional(S.NullOr(S.String)),
2360
2451
  auto_router_embedding_model: S.optional(S.NullOr(S.String)),
2361
2452
  auto_router_max_input_chars: S.optional(S.NullOr(S.Number)),
2453
+ auto_router_model_compression: S.optional(S.NullOr(S.String)),
2454
+ auto_router_routing_compression: S.optional(S.NullOr(S.String)),
2362
2455
  aws_access_key_id: S.optional(S.NullOr(S.String)),
2363
2456
  aws_batch_role_arn: S.optional(S.NullOr(S.String)),
2364
2457
  aws_bedrock_project_id: S.optional(S.NullOr(S.String)),
@@ -2369,10 +2462,16 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2369
2462
  aws_role_name: S.optional(S.NullOr(S.String)),
2370
2463
  aws_secret_access_key: S.optional(S.NullOr(S.String)),
2371
2464
  aws_session_name: S.optional(S.NullOr(S.String)),
2465
+ aws_session_tags: S.optional(
2466
+ S.NullOr(UpdateLiteLLMParamsAwsSessionTagsList),
2467
+ ),
2372
2468
  aws_session_token: S.optional(S.NullOr(S.String)),
2373
2469
  aws_sts_endpoint: S.optional(S.NullOr(S.String)),
2374
2470
  aws_web_identity_token: S.optional(S.NullOr(S.String)),
2375
2471
  azure_ad_token: S.optional(S.NullOr(S.String)),
2472
+ azure_password: S.optional(S.NullOr(S.String)),
2473
+ azure_scope: S.optional(S.NullOr(S.String)),
2474
+ azure_username: S.optional(S.NullOr(S.String)),
2376
2475
  bedrock_tags: S.optional(S.NullOr(UpdateLiteLLMParamsBedrockTagsList)),
2377
2476
  budget_duration: S.optional(S.NullOr(S.String)),
2378
2477
  cache_creation_input_audio_token_cost: S.optional(S.NullOr(S.Number)),
@@ -2417,6 +2516,8 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2417
2516
  cache_read_input_token_cost_priority: S.optional(S.NullOr(S.Number)),
2418
2517
  cache_read_input_token_cost_ultrafast: S.optional(S.NullOr(S.Number)),
2419
2518
  citation_cost_per_token: S.optional(S.NullOr(S.Number)),
2519
+ client_id: S.optional(S.NullOr(S.String)),
2520
+ client_secret: S.optional(S.NullOr(S.String)),
2420
2521
  complexity_router_config: S.optional(
2421
2522
  S.NullOr(UpdateLiteLLMParamsComplexityRouterConfigMap),
2422
2523
  ),
@@ -2427,6 +2528,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2427
2528
  custom_llm_provider: S.optional(S.NullOr(S.String)),
2428
2529
  default_api_key_rpm_limit: S.optional(S.NullOr(S.Number)),
2429
2530
  default_api_key_tpm_limit: S.optional(S.NullOr(S.Number)),
2531
+ drop_params: S.optional(S.NullOr(UpdateLiteLLMParamsDropParams)),
2430
2532
  gcs_bucket_name: S.optional(S.NullOr(S.String)),
2431
2533
  google_maps_grounding_cost_per_query: S.optional(S.NullOr(S.Number)),
2432
2534
  input_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
@@ -2434,11 +2536,13 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2434
2536
  S.NullOr(S.Number),
2435
2537
  ),
2436
2538
  input_cost_per_audio_token: S.optional(S.NullOr(S.Number)),
2539
+ input_cost_per_audio_token_batches: S.optional(S.NullOr(S.Number)),
2437
2540
  input_cost_per_character: S.optional(S.NullOr(S.Number)),
2438
2541
  input_cost_per_character_above_128k_tokens: S.optional(S.NullOr(S.Number)),
2439
2542
  input_cost_per_image: S.optional(S.NullOr(S.Number)),
2440
2543
  input_cost_per_image_above_128k_tokens: S.optional(S.NullOr(S.Number)),
2441
2544
  input_cost_per_image_token: S.optional(S.NullOr(S.Number)),
2545
+ input_cost_per_image_token_batches: S.optional(S.NullOr(S.Number)),
2442
2546
  input_cost_per_pixel: S.optional(S.NullOr(S.Number)),
2443
2547
  input_cost_per_query: S.optional(S.NullOr(S.Number)),
2444
2548
  input_cost_per_second: S.optional(S.NullOr(S.Number)),
@@ -2470,6 +2574,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2470
2574
  S.NullOr(S.Number),
2471
2575
  ),
2472
2576
  input_cost_per_video_token: S.optional(S.NullOr(S.Number)),
2577
+ input_cost_per_video_token_batches: S.optional(S.NullOr(S.Number)),
2473
2578
  itpm: S.optional(S.NullOr(S.Number)),
2474
2579
  keepalive_seconds: S.optional(S.NullOr(S.Number)),
2475
2580
  litellm_credential_name: S.optional(S.NullOr(S.String)),
@@ -2488,6 +2593,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2488
2593
  model_info: S.optional(S.NullOr(UpdateLiteLLMParamsModelInfoMap)),
2489
2594
  ocr_cost_per_credit: S.optional(S.NullOr(S.Number)),
2490
2595
  ocr_cost_per_page: S.optional(S.NullOr(S.Number)),
2596
+ ocr_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
2491
2597
  organization: S.optional(S.NullOr(S.String)),
2492
2598
  otpm: S.optional(S.NullOr(S.Number)),
2493
2599
  output_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
@@ -2504,6 +2610,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2504
2610
  output_cost_per_second_1080p: S.optional(S.NullOr(S.Number)),
2505
2611
  output_cost_per_second_480p: S.optional(S.NullOr(S.Number)),
2506
2612
  output_cost_per_second_4k: S.optional(S.NullOr(S.Number)),
2613
+ output_cost_per_second_720p: S.optional(S.NullOr(S.Number)),
2507
2614
  output_cost_per_token: S.optional(S.NullOr(S.Number)),
2508
2615
  output_cost_per_token_above_128k_tokens: S.optional(S.NullOr(S.Number)),
2509
2616
  output_cost_per_token_above_200k_tokens: S.optional(S.NullOr(S.Number)),
@@ -2536,6 +2643,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2536
2643
  rpm: S.optional(S.NullOr(S.Number)),
2537
2644
  s3_bucket_name: S.optional(S.NullOr(S.String)),
2538
2645
  s3_encryption_key_id: S.optional(S.NullOr(S.String)),
2646
+ s3_endpoint_url: S.optional(S.NullOr(S.String)),
2539
2647
  s3_output_bucket_name: S.optional(S.NullOr(S.String)),
2540
2648
  s3_region_name: S.optional(S.NullOr(S.String)),
2541
2649
  search_context_cost_per_query: S.optional(
@@ -2544,6 +2652,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
2544
2652
  stream_timeout: S.optional(S.NullOr(UpdateLiteLLMParamsStreamTimeout)),
2545
2653
  tag_regex: S.optional(S.NullOr(UpdateLiteLLMParamsTagRegexList)),
2546
2654
  tags: S.optional(S.NullOr(UpdateLiteLLMParamsTagsList)),
2655
+ tenant_id: S.optional(S.NullOr(S.String)),
2547
2656
  tiered_pricing: S.optional(S.NullOr(UpdateLiteLLMParamsTieredPricingList)),
2548
2657
  timeout: S.optional(S.NullOr(UpdateLiteLLMParamsTimeout)),
2549
2658
  tpm: S.optional(S.NullOr(S.Number)),
@@ -2670,7 +2779,7 @@ export const PostBlockModelModelBlockResponse = /*@__PURE__*/ S.suspend(() =>
2670
2779
 
2671
2780
  /** An operator-defined tier: the name the LLM classifier must return and its rubric description. */
2672
2781
  export interface TierDefinition {
2673
- /** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (SIMPLE/MEDIUM/COMPLEX/REASONING), which inherits the built-in criteria when omitted */
2782
+ /** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (NON_REASONING, SIMPLE, MEDIUM, COMPLEX, REASONING), which inherits the built-in criteria when omitted */
2674
2783
  description?: string | null;
2675
2784
  /** Tier name; becomes a value the LLM classifier can return and a key of `tiers` */
2676
2785
  name: string;
@@ -2689,18 +2798,39 @@ export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPro
2689
2798
  TierDefinition,
2690
2799
  ) as any as S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList>;
2691
2800
 
2801
+ export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap =
2802
+ { [key: string]: string | undefined };
2803
+ export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap =
2804
+ /*@__PURE__*/ S.Record(
2805
+ S.String,
2806
+ S.String,
2807
+ ) as any as S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap>;
2808
+
2692
2809
  export interface PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest {
2810
+ classification_examples?: string | null;
2693
2811
  classification_prompt?: string | null;
2812
+ classification_rubric?: ClassificationRubric | (string & {}) | null;
2694
2813
  context_window_size?: number;
2695
- tier_definitions: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList;
2814
+ tier_definitions?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList | null;
2815
+ tier_labels?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap | null;
2696
2816
  }
2697
2817
  export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest =
2698
2818
  /*@__PURE__*/ S.suspend(() =>
2699
2819
  S.Struct({
2820
+ classification_examples: S.optional(S.NullOr(S.String)),
2700
2821
  classification_prompt: S.optional(S.NullOr(S.String)),
2822
+ classification_rubric: S.optional(S.NullOr(ClassificationRubric)),
2701
2823
  context_window_size: S.optional(S.Number),
2702
- tier_definitions:
2703
- PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList,
2824
+ tier_definitions: S.optional(
2825
+ S.NullOr(
2826
+ PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList,
2827
+ ),
2828
+ ),
2829
+ tier_labels: S.optional(
2830
+ S.NullOr(
2831
+ PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap,
2832
+ ),
2833
+ ),
2704
2834
  }).pipe(
2705
2835
  T.Http({
2706
2836
  method: "POST",
@@ -2732,40 +2862,141 @@ export const AdaptiveRouterWeights = /*@__PURE__*/ S.suspend(() =>
2732
2862
  identifier: "AdaptiveRouterWeights",
2733
2863
  }) as any as S.Schema<AdaptiveRouterWeights>;
2734
2864
 
2865
+ export interface CapabilityCalibrationConfig {
2866
+ intercept: number;
2867
+ slope: number;
2868
+ version: string;
2869
+ }
2870
+ export const CapabilityCalibrationConfig = /*@__PURE__*/ S.suspend(() =>
2871
+ S.Struct({
2872
+ intercept: S.Number,
2873
+ slope: S.Number,
2874
+ version: S.String,
2875
+ }),
2876
+ ).annotate({
2877
+ identifier: "CapabilityCalibrationConfig",
2878
+ }) as any as S.Schema<CapabilityCalibrationConfig>;
2879
+
2880
+ /** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
2881
+ export type CapabilityClassifierConfigResponseFormat =
2882
+ | "json_schema"
2883
+ | "json_object";
2884
+ export const CapabilityClassifierConfigResponseFormat = S.String;
2885
+
2886
+ /** Switchyard-compatible probability threshold policy for two model tiers. */
2887
+ export interface CapabilityClassifierConfig {
2888
+ /** Lowest p_solve that routes a supported task to efficient_tier */
2889
+ base_threshold: number;
2890
+ /** Optional versioned sigmoid calibration fitted for this judge, capability card, efficient model, and execution setup. Applies sigmoid(slope * logit(clip(p_solve, 1e-6, 1-1e-6)) + intercept) before the threshold policy. Omit to route on the raw forecast. */
2891
+ calibration?: CapabilityCalibrationConfig | null;
2892
+ /** Higher, fail-closed tier used below the adjusted threshold or when the classifier verdict is unavailable */
2893
+ capable_tier: string;
2894
+ /** Tier used when the efficient model's forecasted solve probability meets the adjusted threshold */
2895
+ efficient_tier: string;
2896
+ /** Maximum completion tokens available to the capability classifier verdict */
2897
+ max_output_tokens?: number;
2898
+ /** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
2899
+ response_format?: CapabilityClassifierConfigResponseFormat | (string & {});
2900
+ /** Amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts */
2901
+ threshold_step?: number;
2902
+ }
2903
+ export const CapabilityClassifierConfig = /*@__PURE__*/ S.suspend(() =>
2904
+ S.Struct({
2905
+ base_threshold: S.Number,
2906
+ calibration: S.optional(S.NullOr(CapabilityCalibrationConfig)),
2907
+ capable_tier: S.String,
2908
+ efficient_tier: S.String,
2909
+ max_output_tokens: S.optional(S.Number),
2910
+ response_format: S.optional(CapabilityClassifierConfigResponseFormat),
2911
+ threshold_step: S.optional(S.Number),
2912
+ }),
2913
+ ).annotate({
2914
+ identifier: "CapabilityClassifierConfig",
2915
+ }) as any as S.Schema<CapabilityClassifierConfig>;
2916
+
2917
+ /** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
2918
+ export type RequestComplexityRouterConfigClassificationMode =
2919
+ | "every_request"
2920
+ | "user_turn";
2921
+ export const RequestComplexityRouterConfigClassificationMode = S.String;
2922
+
2735
2923
  /** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
2736
2924
  export type RequestComplexityRouterConfigClassifierFallback =
2737
2925
  | "heuristic"
2738
2926
  | "default_model";
2739
2927
  export const RequestComplexityRouterConfigClassifierFallback = S.String;
2740
2928
 
2929
+ export type ClassifierLLMConfigReasoningEffort =
2930
+ | "none"
2931
+ | "minimal"
2932
+ | "low"
2933
+ | "medium"
2934
+ | "high"
2935
+ | "xhigh"
2936
+ | "max";
2937
+ export const ClassifierLLMConfigReasoningEffort = S.String;
2938
+
2939
+ /** Whether the LLM classifier sees the images on the request it is classifying. Off by default because images cost far more than the text ask they arrive with, and the classifier runs on every request. A turn whose complexity lives in the image ("what is wrong in this stack trace screenshot") is invisible to a text-only classifier, which is what this buys. */
2940
+ export interface ClassifierVisionConfig {
2941
+ /** Forward image content to the classifier. Requires a classifier model declared supports_vision, on the deployment's model_info or in the model cost map; images stay stripped otherwise, so a classifier that cannot read them is never sent one. Declare model_info.supports_vision on the deployment to enable a model the cost map does not describe. Only inline data: URIs are forwarded. A request whose images are http(s) URLs still classifies on its text alone, because some providers fetch such a URL from the proxy rather than the provider, which would let a caller aim a proxy-side request at an address of their choosing. */
2942
+ enabled?: boolean;
2943
+ /** How many images from the newest user turn to forward, in wire order. Bounds the added cost of a turn that attaches many images. Images on earlier turns are never forwarded. */
2944
+ max_images?: number;
2945
+ }
2946
+ export const ClassifierVisionConfig = /*@__PURE__*/ S.suspend(() =>
2947
+ S.Struct({
2948
+ enabled: S.optional(S.Boolean),
2949
+ max_images: S.optional(S.Number),
2950
+ }),
2951
+ ).annotate({
2952
+ identifier: "ClassifierVisionConfig",
2953
+ }) as any as S.Schema<ClassifierVisionConfig>;
2954
+
2741
2955
  /** Configuration for the LLM-based complexity classifier. */
2742
2956
  export interface ClassifierLLMConfig {
2957
+ /** How long to skip this router's LLM classifier after a classification call times out. Requests use classifier_fallback during the cooldown. When it expires, one request probes the classifier while concurrent requests keep using the fallback; a successful probe closes the circuit and a failed probe restarts the cooldown. */
2958
+ circuit_breaker_cooldown_seconds?: number;
2959
+ /** Whether one classifier timeout temporarily sends requests through classifier_fallback. Enabled by default so an unhealthy classifier cannot repeat its timeout across sessions. */
2960
+ circuit_breaker_enabled?: boolean;
2743
2961
  /** Which calibration examples the built-in rubric carries. 'agentic' anchors routine installs, builds, multi-file edits, and standard debugging at MEDIUM, so ordinary engineering does not route to the most expensive tier; it suits agent, terminal, and coding-assistant traffic as well as mixed traffic. 'chat' omits those engineering anchors, for a deployment serving only conversational traffic. 'business' carries business/sales anchors and business-flavored tier criteria that keep routine drafting and summarizing off the expensive tiers and reserve the top tier for committing to decisions under tradeoffs; it suits sales, support, and go-to-market traffic. Every preset keeps the same four tiers, so this moves where the boundary sits without changing the taxonomy. Leave unset for 'legacy', the rubric as it shipped before calibration examples existed, so an existing router's tier decisions and spend do not move on upgrade. Mutually exclusive with system_prompt, which replaces the rubric this would select. Only applies when classifier_type is 'llm'. */
2744
2962
  classification_rubric?: ClassificationRubric | (string & {}) | null;
2745
2963
  /** Model name (from the router's model_list) to call for classification */
2746
2964
  model: string;
2965
+ /** Reasoning effort override for classifier calls. Leave unset to use the classifier deployment or provider default. */
2966
+ reasoning_effort?: ClassifierLLMConfigReasoningEffort | (string & {}) | null;
2747
2967
  /** Replaces the built-in complexity rubric as the classifier's entire system role. When set, neither the default rubric nor the context-window closing line is appended, so the prompt owns the whole taxonomy and the tier names SIMPLE/MEDIUM/COMPLEX/REASONING become whatever buckets it defines: a prompt that classifies data sensitivity routes on that instead of on difficulty. Two consequences of full replacement. The default rubric's closing paragraph is the classifier's prompt-injection defense, telling it that the caller's quoted system prompt and prior turns are material to judge and never instructions; a replacement that omits it lets a caller ask for a tier and get it. And the heuristic fallback still scores complexity, so a router on some other taxonomy wants classifier_fallback='default_model'. Leave unset for the built-in rubric. Only applies when classifier_type is 'llm'. */
2748
2968
  system_prompt?: string | null;
2749
2969
  /** Timeout budget for the classification call, in milliseconds */
2750
2970
  timeout_ms?: number;
2971
+ /** Whether the classifier sees images on the request, and how many */
2972
+ vision?: ClassifierVisionConfig;
2751
2973
  }
2752
2974
  export const ClassifierLLMConfig = /*@__PURE__*/ S.suspend(() =>
2753
2975
  S.Struct({
2976
+ circuit_breaker_cooldown_seconds: S.optional(S.Number),
2977
+ circuit_breaker_enabled: S.optional(S.Boolean),
2754
2978
  classification_rubric: S.optional(S.NullOr(ClassificationRubric)),
2755
2979
  model: S.String,
2980
+ reasoning_effort: S.optional(S.NullOr(ClassifierLLMConfigReasoningEffort)),
2756
2981
  system_prompt: S.optional(S.NullOr(S.String)),
2757
2982
  timeout_ms: S.optional(S.Number),
2983
+ vision: S.optional(ClassifierVisionConfig),
2758
2984
  }),
2759
2985
  ).annotate({
2760
2986
  identifier: "ClassifierLLMConfig",
2761
2987
  }) as any as S.Schema<ClassifierLLMConfig>;
2762
2988
 
2763
- /** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin, or 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier */
2989
+ /** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
2764
2990
  export type RequestComplexityRouterConfigClassifierType =
2765
2991
  | "heuristic"
2992
+ | "heuristic_v2"
2766
2993
  | "llm"
2994
+ | "capability"
2995
+ | "llm_v2"
2767
2996
  | "custom"
2768
- | "heuristic_first";
2997
+ | "heuristic_first"
2998
+ | "hybrid"
2999
+ | "jev";
2769
3000
  export const RequestComplexityRouterConfigClassifierType = S.String;
2770
3001
 
2771
3002
  export type RequestComplexityRouterConfigCodeKeywordsList = Array<string>;
@@ -2774,6 +3005,48 @@ export const RequestComplexityRouterConfigCodeKeywordsList =
2774
3005
  S.String,
2775
3006
  ) as any as S.Schema<RequestComplexityRouterConfigCodeKeywordsList>;
2776
3007
 
3008
+ export type CustomDimensionKeywordsList = Array<string>;
3009
+ export const CustomDimensionKeywordsList = /*@__PURE__*/ S.Array(
3010
+ S.String,
3011
+ ) as any as S.Schema<CustomDimensionKeywordsList>;
3012
+
3013
+ export type CustomDimensionPatternsList = Array<string>;
3014
+ export const CustomDimensionPatternsList = /*@__PURE__*/ S.Array(
3015
+ S.String,
3016
+ ) as any as S.Schema<CustomDimensionPatternsList>;
3017
+
3018
+ /** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
3019
+ export type CustomDimensionScoringMode = "binary" | "match_count";
3020
+ export const CustomDimensionScoringMode = S.String;
3021
+
3022
+ export interface CustomDimension {
3023
+ keywords?: CustomDimensionKeywordsList;
3024
+ name: string;
3025
+ patterns?: CustomDimensionPatternsList;
3026
+ /** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
3027
+ scoring_mode?: CustomDimensionScoringMode | (string & {});
3028
+ weight: number;
3029
+ }
3030
+ export const CustomDimension = /*@__PURE__*/ S.suspend(() =>
3031
+ S.Struct({
3032
+ keywords: S.optional(CustomDimensionKeywordsList),
3033
+ name: S.String,
3034
+ patterns: S.optional(CustomDimensionPatternsList),
3035
+ scoring_mode: S.optional(CustomDimensionScoringMode),
3036
+ weight: S.Number,
3037
+ }),
3038
+ ).annotate({
3039
+ identifier: "CustomDimension",
3040
+ }) as any as S.Schema<CustomDimension>;
3041
+
3042
+ /** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
3043
+ export type RequestComplexityRouterConfigCustomDimensionsList =
3044
+ Array<CustomDimension>;
3045
+ export const RequestComplexityRouterConfigCustomDimensionsList =
3046
+ /*@__PURE__*/ S.Array(
3047
+ CustomDimension,
3048
+ ) as any as S.Schema<RequestComplexityRouterConfigCustomDimensionsList>;
3049
+
2777
3050
  export type RequestComplexityRouterConfigCustomTechnicalKeywordsList =
2778
3051
  Array<string>;
2779
3052
  export const RequestComplexityRouterConfigCustomTechnicalKeywordsList =
@@ -2797,6 +3070,142 @@ export const RequestComplexityRouterConfigEscalationKeywordsList =
2797
3070
  S.String,
2798
3071
  ) as any as S.Schema<RequestComplexityRouterConfigEscalationKeywordsList>;
2799
3072
 
3073
+ export interface TierCohortStatistic {
3074
+ cohort: string;
3075
+ observations: number;
3076
+ successes: number;
3077
+ tier: number;
3078
+ }
3079
+ export const TierCohortStatistic = /*@__PURE__*/ S.suspend(() =>
3080
+ S.Struct({
3081
+ cohort: S.String,
3082
+ observations: S.Number,
3083
+ successes: S.Number,
3084
+ tier: S.Number,
3085
+ }),
3086
+ ).annotate({
3087
+ identifier: "TierCohortStatistic",
3088
+ }) as any as S.Schema<TierCohortStatistic>;
3089
+
3090
+ export type TrainedTierArtifactCohortStatisticsList =
3091
+ Array<TierCohortStatistic>;
3092
+ export const TrainedTierArtifactCohortStatisticsList = /*@__PURE__*/ S.Array(
3093
+ TierCohortStatistic,
3094
+ ) as any as S.Schema<TrainedTierArtifactCohortStatisticsList>;
3095
+
3096
+ export interface TierDataset {
3097
+ license: string;
3098
+ name: string;
3099
+ rows: number;
3100
+ success_definition?: string;
3101
+ url: string;
3102
+ }
3103
+ export const TierDataset = /*@__PURE__*/ S.suspend(() =>
3104
+ S.Struct({
3105
+ license: S.String,
3106
+ name: S.String,
3107
+ rows: S.Number,
3108
+ success_definition: S.optional(S.String),
3109
+ url: S.String,
3110
+ }),
3111
+ ).annotate({ identifier: "TierDataset" }) as any as S.Schema<TierDataset>;
3112
+
3113
+ export type TrainedTierArtifactDatasetsList = Array<TierDataset>;
3114
+ export const TrainedTierArtifactDatasetsList = /*@__PURE__*/ S.Array(
3115
+ TierDataset,
3116
+ ) as any as S.Schema<TrainedTierArtifactDatasetsList>;
3117
+
3118
+ /** Fixed v0 taxonomy. User-extensible types come in v1. */
3119
+ export type RequestType =
3120
+ | "code_generation"
3121
+ | "code_understanding"
3122
+ | "technical_design"
3123
+ | "analytical_reasoning"
3124
+ | "writing"
3125
+ | "factual_lookup"
3126
+ | "general";
3127
+ export const RequestType = S.String;
3128
+
3129
+ export interface TierDomainStatistic {
3130
+ observations: number;
3131
+ request_type: RequestType | (string & {});
3132
+ successes: number;
3133
+ tier: number;
3134
+ }
3135
+ export const TierDomainStatistic = /*@__PURE__*/ S.suspend(() =>
3136
+ S.Struct({
3137
+ observations: S.Number,
3138
+ request_type: RequestType,
3139
+ successes: S.Number,
3140
+ tier: S.Number,
3141
+ }),
3142
+ ).annotate({
3143
+ identifier: "TierDomainStatistic",
3144
+ }) as any as S.Schema<TierDomainStatistic>;
3145
+
3146
+ export type TrainedTierArtifactDomainStatisticsList =
3147
+ Array<TierDomainStatistic>;
3148
+ export const TrainedTierArtifactDomainStatisticsList = /*@__PURE__*/ S.Array(
3149
+ TierDomainStatistic,
3150
+ ) as any as S.Schema<TrainedTierArtifactDomainStatisticsList>;
3151
+
3152
+ export interface TierGlobalStatistic {
3153
+ observations: number;
3154
+ successes: number;
3155
+ tier: number;
3156
+ }
3157
+ export const TierGlobalStatistic = /*@__PURE__*/ S.suspend(() =>
3158
+ S.Struct({
3159
+ observations: S.Number,
3160
+ successes: S.Number,
3161
+ tier: S.Number,
3162
+ }),
3163
+ ).annotate({
3164
+ identifier: "TierGlobalStatistic",
3165
+ }) as any as S.Schema<TierGlobalStatistic>;
3166
+
3167
+ export type TrainedTierArtifactGlobalStatisticsList =
3168
+ Array<TierGlobalStatistic>;
3169
+ export const TrainedTierArtifactGlobalStatisticsList = /*@__PURE__*/ S.Array(
3170
+ TierGlobalStatistic,
3171
+ ) as any as S.Schema<TrainedTierArtifactGlobalStatisticsList>;
3172
+
3173
+ export interface TrainedTierArtifact {
3174
+ cohort_prior_mass?: number;
3175
+ cohort_statistics?: TrainedTierArtifactCohortStatisticsList;
3176
+ datasets?: TrainedTierArtifactDatasetsList;
3177
+ domain_prior_mass?: number;
3178
+ domain_statistics?: TrainedTierArtifactDomainStatisticsList;
3179
+ global_statistics: TrainedTierArtifactGlobalStatisticsList;
3180
+ routing_threshold?: number;
3181
+ schema_version?: number;
3182
+ split_method?: string;
3183
+ success_definition?: string;
3184
+ }
3185
+ export const TrainedTierArtifact = /*@__PURE__*/ S.suspend(() =>
3186
+ S.Struct({
3187
+ cohort_prior_mass: S.optional(S.Number),
3188
+ cohort_statistics: S.optional(TrainedTierArtifactCohortStatisticsList),
3189
+ datasets: S.optional(TrainedTierArtifactDatasetsList),
3190
+ domain_prior_mass: S.optional(S.Number),
3191
+ domain_statistics: S.optional(TrainedTierArtifactDomainStatisticsList),
3192
+ global_statistics: TrainedTierArtifactGlobalStatisticsList,
3193
+ routing_threshold: S.optional(S.Number),
3194
+ schema_version: S.optional(S.Number),
3195
+ split_method: S.optional(S.String),
3196
+ success_definition: S.optional(S.String),
3197
+ }),
3198
+ ).annotate({
3199
+ identifier: "TrainedTierArtifact",
3200
+ }) as any as S.Schema<TrainedTierArtifact>;
3201
+
3202
+ /** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
3203
+ export type RequestComplexityRouterConfigHeuristicV2Artifact =
3204
+ | TrainedTierArtifact
3205
+ | string;
3206
+ export const RequestComplexityRouterConfigHeuristicV2Artifact =
3207
+ S.Unknown as any as S.Schema<RequestComplexityRouterConfigHeuristicV2Artifact>;
3208
+
2800
3209
  export type RequestComplexityRouterConfigHousekeepingPatternsList =
2801
3210
  Array<string>;
2802
3211
  export const RequestComplexityRouterConfigHousekeepingPatternsList =
@@ -2804,6 +3213,32 @@ export const RequestComplexityRouterConfigHousekeepingPatternsList =
2804
3213
  S.String,
2805
3214
  ) as any as S.Schema<RequestComplexityRouterConfigHousekeepingPatternsList>;
2806
3215
 
3216
+ export interface JevClassifierConfig {
3217
+ /** TypeSafe API base, falling back to TYPESAFE_API_BASE and then https://api.typesafe.ai */
3218
+ api_base?: string | null;
3219
+ /** TypeSafe API key, falling back to TYPESAFE_API_KEY */
3220
+ api_key?: string | null;
3221
+ circuit_breaker_cooldown_seconds?: number;
3222
+ circuit_breaker_enabled?: boolean;
3223
+ /** Replaces the built-in Jev question instructions */
3224
+ instructions?: string | null;
3225
+ model?: string;
3226
+ timeout_ms?: number;
3227
+ }
3228
+ export const JevClassifierConfig = /*@__PURE__*/ S.suspend(() =>
3229
+ S.Struct({
3230
+ api_base: S.optional(S.NullOr(S.String)),
3231
+ api_key: S.optional(S.NullOr(S.String)),
3232
+ circuit_breaker_cooldown_seconds: S.optional(S.Number),
3233
+ circuit_breaker_enabled: S.optional(S.Boolean),
3234
+ instructions: S.optional(S.NullOr(S.String)),
3235
+ model: S.optional(S.String),
3236
+ timeout_ms: S.optional(S.Number),
3237
+ }),
3238
+ ).annotate({
3239
+ identifier: "JevClassifierConfig",
3240
+ }) as any as S.Schema<JevClassifierConfig>;
3241
+
2807
3242
  /** Keywords/phrases that trigger this rule (lexical or semantic match) */
2808
3243
  export type KeywordTierRuleKeywordsList = Array<string>;
2809
3244
  export const KeywordTierRuleKeywordsList = /*@__PURE__*/ S.Array(
@@ -2833,6 +3268,71 @@ export const RequestComplexityRouterConfigKeywordTierRulesList =
2833
3268
  KeywordTierRule,
2834
3269
  ) as any as S.Schema<RequestComplexityRouterConfigKeywordTierRulesList>;
2835
3270
 
3271
+ export interface LLMV2ProbabilityCalibration {
3272
+ intercept: number;
3273
+ slope: number;
3274
+ }
3275
+ export const LLMV2ProbabilityCalibration = /*@__PURE__*/ S.suspend(() =>
3276
+ S.Struct({
3277
+ intercept: S.Number,
3278
+ slope: S.Number,
3279
+ }),
3280
+ ).annotate({
3281
+ identifier: "LLMV2ProbabilityCalibration",
3282
+ }) as any as S.Schema<LLMV2ProbabilityCalibration>;
3283
+
3284
+ export interface LLMV2Calibration {
3285
+ capable: LLMV2ProbabilityCalibration;
3286
+ efficient: LLMV2ProbabilityCalibration;
3287
+ prompt_version: string;
3288
+ version: string;
3289
+ }
3290
+ export const LLMV2Calibration = /*@__PURE__*/ S.suspend(() =>
3291
+ S.Struct({
3292
+ capable: LLMV2ProbabilityCalibration,
3293
+ efficient: LLMV2ProbabilityCalibration,
3294
+ prompt_version: S.String,
3295
+ version: S.String,
3296
+ }),
3297
+ ).annotate({
3298
+ identifier: "LLMV2Calibration",
3299
+ }) as any as S.Schema<LLMV2Calibration>;
3300
+
3301
+ export type LLMV2ConfigResponseFormat = "json_schema" | "json_object";
3302
+ export const LLMV2ConfigResponseFormat = S.String;
3303
+
3304
+ export interface LLMV2Config {
3305
+ calibration?: LLMV2Calibration | null;
3306
+ capable_profile?: string | null;
3307
+ capable_profile_preset?: string | null;
3308
+ capable_tier?: string;
3309
+ efficient_profile?: string | null;
3310
+ efficient_profile_preset?: string | null;
3311
+ efficient_tier?: string;
3312
+ harness?: string | null;
3313
+ harness_preset?: string | null;
3314
+ max_output_tokens?: number;
3315
+ /** Maximum estimated success loss allowed for efficient. */
3316
+ max_quality_gap: number;
3317
+ response_format?: LLMV2ConfigResponseFormat | (string & {});
3318
+ }
3319
+ export const LLMV2Config = /*@__PURE__*/ S.suspend(() =>
3320
+ S.Struct({
3321
+ calibration: S.optional(S.NullOr(LLMV2Calibration)),
3322
+ capable_profile: S.optional(S.NullOr(S.String)),
3323
+ capable_profile_preset: S.optional(S.NullOr(S.String)),
3324
+ capable_tier: S.optional(S.String),
3325
+ efficient_profile: S.optional(S.NullOr(S.String)),
3326
+ efficient_profile_preset: S.optional(S.NullOr(S.String)),
3327
+ efficient_tier: S.optional(S.String),
3328
+ harness: S.optional(S.NullOr(S.String)),
3329
+ harness_preset: S.optional(S.NullOr(S.String)),
3330
+ max_output_tokens: S.optional(S.Number),
3331
+ max_quality_gap: S.Number,
3332
+ response_format: S.optional(LLMV2ConfigResponseFormat),
3333
+ }),
3334
+ ).annotate({ identifier: "LLMV2Config" }) as any as S.Schema<LLMV2Config>;
3335
+
2836
3336
  export type RequestComplexityRouterConfigPlanModePatternsList = Array<string>;
2837
3337
  export const RequestComplexityRouterConfigPlanModePatternsList =
2838
3338
  /*@__PURE__*/ S.Array(
@@ -2987,52 +3487,81 @@ export interface RequestComplexityRouterConfig {
2987
3487
  | (string & {});
2988
3488
  /** Quality vs cost weights for adaptive selection (used when adaptive=True) */
2989
3489
  adaptive_weights?: AdaptiveRouterWeights;
2990
- /** Replaces the opening instructions of the LLM classifier rubric (the judging-criteria prose) for a custom tier set. The per-tier bullets and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text are always appended after it and cannot be overridden. Requires tier_definitions; a built-in-tier router customizes its prompt via classifier_llm_config.system_prompt or classification_rubric instead. */
3490
+ /** Probability threshold policy required when classifier_type is 'capability'. The classifier forecasts p_solve for efficient_tier, adjusts base_threshold using the capability-card boundary, and otherwise routes to capable_tier */
3491
+ capability_classifier_config?: CapabilityClassifierConfig | null;
3492
+ /** Replaces the calibration examples of the LLM classifier rubric, and nothing else. Written as example lines only: the router renders the 'Calibration examples:' heading above them, after the per-tier bullets. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_prompt replaces them, the classification instructions; a custom tier set ships no examples of its own, so the section renders only when this is set. */
3493
+ classification_examples?: string | null;
3494
+ /** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
3495
+ classification_mode?:
3496
+ | RequestComplexityRouterConfigClassificationMode
3497
+ | (string & {});
3498
+ /** Replaces the classification instructions that open the LLM classifier rubric, and nothing else. The per-tier bullets follow it, the calibration examples follow those, and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text is always appended after them and cannot be overridden. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_examples replaces them, the calibration examples. */
2991
3499
  classification_prompt?: string | null;
2992
- /** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the caller's system prompt sit outside this budget and are always sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
3500
+ /** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
2993
3501
  classifier_context_budget_chars?: number;
2994
3502
  /** Include assistant turns in the classifier context window, so difficulty stated by the model rather than by the user stays visible: a plan the assistant calls complex, which the user approves with 'yes', is classified on the work being approved instead of on the word 'yes'. When enabled, classifier_context_window_size counts the last N turns of the conversation across both roles rather than the last N user turns, and assistant text is sent to the classifier model, which may be a different deployment or provider than the routed completion model. Assistant replies spend classifier_context_budget_chars alongside user turns, so raise it if the oldest turns stop being quoted once replies join the window. Off by default because enabling it shifts tier decisions, and therefore spend, for an already-deployed router. Only applies when classifier_type is 'llm'. */
2995
3503
  classifier_context_include_assistant_turns?: boolean;
2996
3504
  /** Optional cap on each individual prior turn's text, applied before classifier_context_budget_chars bounds the block. Unset by default, so one long turn may spend the whole budget, which is usually what a follow-up needs; set it when no single turn should dominate the context the classifier sees. A capped turn keeps its opening and its ending with the middle elided. Only applies when classifier_type is 'llm'. */
2997
3505
  classifier_context_per_turn_chars?: number | null;
2998
- /** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call already carries the current user ask and the caller's system prompt in full. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
3506
+ /** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
2999
3507
  classifier_context_window_size?: number;
3000
3508
  /** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
3001
3509
  classifier_fallback?:
3002
3510
  | RequestComplexityRouterConfigClassifierFallback
3003
3511
  | (string & {});
3004
- /** Configuration for the LLM classifier; required when classifier_type is 'llm' or 'heuristic_first' */
3512
+ /** Configuration for the LLM classifier; required when classifier_type is 'llm', 'capability', 'heuristic_first' or 'hybrid' */
3005
3513
  classifier_llm_config?: ClassifierLLMConfig | null;
3006
3514
  /** Not settable over HTTP; the classifier plugin is a runtime object */
3007
3515
  classifier_plugin?: unknown | null;
3008
3516
  /** Timeout budget for the classifier plugin call, in milliseconds. On expiry the fallback path decides the tier. Only applies when classifier_type is 'custom'. */
3009
3517
  classifier_plugin_timeout_ms?: number;
3010
- /** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin, or 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier */
3518
+ /** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
3011
3519
  classifier_type?: RequestComplexityRouterConfigClassifierType | (string & {});
3012
3520
  /** Keywords indicating code-related content */
3013
3521
  code_keywords?: RequestComplexityRouterConfigCodeKeywordsList | null;
3522
+ /** Fraction of a model's declared context window the estimated prompt must fit within. The token count is an estimate, so fitting against the full window would dispatch prompts that the provider's own tokenizer then rejects; 0.95 leaves room for that drift plus the response tokens. */
3523
+ context_window_escalation_buffer?: number;
3524
+ /** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
3525
+ custom_dimensions?: RequestComplexityRouterConfigCustomDimensionsList;
3014
3526
  /** Domain-specific technical keywords appended to the effective base list (technical_keywords if set, otherwise DEFAULT_TECHNICAL_KEYWORDS). Order is preserved; duplicates are removed case-insensitively against the base list and within this list. */
3015
3527
  custom_technical_keywords?: RequestComplexityRouterConfigCustomTechnicalKeywordsList | null;
3016
3528
  /** Default model to use if tier cannot be determined */
3017
3529
  default_model?: string | null;
3018
- /** When True and a session_id is resolvable on the request, pin the deployment chosen inside each routed model group and reuse it whenever the session returns to that group, without pinning which group the session routes to. Independent of session_affinity, which pins the model group instead (and always carries this deployment pin with it): with session_affinity off, every turn is still classified on its own merits while a session that escalates to a stronger tier and comes back still lands on the deployment it used before, which is what keeps a provider prompt cache warm. Pins are held per model group, so switching tiers does not disturb the pin left behind in the previous group. On by default because re-shuffling a conversation across deployments of the same model discards that cache for no benefit; set False to keep every turn load-balanced across the group, which is what a deployment set with tight per-deployment rate limits wants. Inert when no session_id is resolvable, since there is nothing to key a pin on, and suppressed when plugins are configured, for the same reason session_affinity is. */
3530
+ /** When True and a client session_id is resolvable, reuse the session's chosen model for each classified tier and its deployment within each model group. With session_affinity off, every turn is still classified: moving to another tier leaves the previous tier's model pin intact for a later return. Pins yield to current candidate, context, modality, and availability constraints. Adaptive selection chooses the initial model from its eligible pool, then reuses that choice per tier. This reduces avoidable provider prompt-cache misses; it does not guarantee cache hits. Set False to select models and load-balance deployments on every turn, unless session_affinity or user_turn classification requires a pin. Inert without a client session_id and suppressed when plugins are configured. */
3019
3531
  deployment_affinity?: boolean;
3020
3532
  /** Weights for each scoring dimension */
3021
3533
  dimension_weights?: RequestComplexityRouterConfigDimensionWeightsMap;
3022
3534
  /** Embedding model (LiteLLM model name) used when semantic_keyword_matching is enabled */
3023
3535
  embedding_model?: string | null;
3536
+ /** Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Set false to dispatch on complexity alone, as before. */
3537
+ enable_context_window_escalation?: boolean;
3538
+ /** Add NON_REASONING as a fifth built-in tier below SIMPLE, for operational agent traffic that relays or reformats information rather than reasoning about it. Off by default: turning it on adds a rung to this router's ladder, a bullet to the LLM classifier's rubric, and a value the classifier may return, all of which move tier decisions and spend on an already-deployed router. Requires an LLM, Jev, or custom classifier plugin, since the heuristic scorers cannot produce the tier, and a model in `tiers` under the NON_REASONING key. Escalation still walks up from it, and it is never the savings baseline or a `heuristic_v2` prediction. */
3539
+ enable_non_reasoning_tier?: boolean;
3024
3540
  /** Case-sensitive phrases a user can include to force a bump to the next-higher complexity tier when they aren't satisfied with results (they can force a stronger model, but not choose which one). Defaults to ['LITELLM ESCALATE'] when unset; set to an empty list to disable. */
3025
3541
  escalation_keywords?: RequestComplexityRouterConfigEscalationKeywordsList | null;
3026
3542
  /** Tier routed to when the LLM classifier fails (timeout, provider error, or an unparseable reply). Required with tier_definitions and must name a defined tier; the heuristic scorer cannot produce custom tiers, so this replaces the heuristic fallback for custom tier sets. */
3027
3543
  fallback_tier?: string | null;
3028
3544
  /** The highest tier the local scorer may decide on its own; required when classifier_type is 'heuristic_first' and rejected otherwise. A request whose heuristic tier is at or below this one skips the LLM classifier and routes straight to that heuristic tier, so the classifier call is only paid for on traffic the scorer could not place cheaply. The scorer must also have produced at least one signal: a prompt where no dimension fired scores 0.0 and would otherwise land SIMPLE by default rather than by evidence, which is how a chained router would silently send unclassified traffic to the cheapest model. Names a built-in tier, and may not name the highest one, since that would make the LLM classifier unreachable. */
3029
3545
  heuristic_first_max_tier?: string | null;
3546
+ /** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
3547
+ heuristic_v2_artifact?: RequestComplexityRouterConfigHeuristicV2Artifact;
3030
3548
  /** Additional case-sensitive literal sentinels that mark a request as client housekeeping, on top of the built-in conversation-title ones. For clients whose wording the built-ins don't cover, or after a client release changes its strings. */
3031
3549
  housekeeping_patterns?: RequestComplexityRouterConfigHousekeepingPatternsList | null;
3550
+ /** How close to a tier boundary a heuristic score has to land before the LLM classifier breaks the tie; required when classifier_type is 'hybrid' and rejected otherwise. Everything further than this from every active boundary routes on the scorer's own tier with no classifier call, at any tier, which is what separates 'hybrid' from 'heuristic_first' and its cheap-tier ceiling. A prompt where no dimension fired still goes to the classifier, since the scorer has no opinion to be near a boundary with. 0 escalates only scores sitting exactly on a boundary. */
3551
+ hybrid_boundary_margin?: number | null;
3552
+ jev_classifier_config?: JevClassifierConfig | null;
3032
3553
  /** Rules that force a specific tier when their keywords match the prompt */
3033
3554
  keyword_tier_rules?: RequestComplexityRouterConfigKeywordTierRulesList | null;
3555
+ /** Experimental joint task-demand and solver-capability forecasting for classifier_type llm_v2. */
3556
+ llm_v2_config?: LLMV2Config | null;
3034
3557
  /** Minimum cosine similarity for a semantic keyword match */
3035
3558
  match_threshold?: number;
3559
+ /** Set max_tokens on every routed request to the output ceiling of the tier model it lands on, replacing whatever the caller sent. A caller behind an auto-router cannot pick one value that fits every tier: the smallest tier's ceiling starves a bigger tier's thinking budget, and a bigger tier's ceiling is rejected by the smallest. The ceiling is the smallest max_output_tokens across the tier model's deployments, read from each deployment's model_info and then the model cost map; a tier model with a deployment whose ceiling is unknown keeps the caller's value. A max_tokens, max_completion_tokens or max_output_tokens in the tier's own litellm_params still wins. Set false to forward the caller's value unchanged. */
3560
+ max_tokens_from_tier_model?: boolean;
3561
+ /** Let modality_routing replace a kept session-affinity pin on the turns that carry an image. Without this, a session pinned to a text-only model fails every image turn with a provider 400, since the pin is exempt from the modality gate. When enabled, such a turn routes to a capable model for that request only and the stored pin is left untouched, so the next text turn replays the session's own model; the override is reported as cause modality_pin_override and is never itself pinned. Inert unless modality_routing is also enabled. */
3562
+ modality_pin_override?: boolean;
3563
+ /** Route image-bearing requests only to models that can accept image input. The classifier reads text alone, so an image request whose text classifies cheap otherwise lands on a text-only model and fails with a provider 400. When enabled, a routed model explicitly declared supports_vision false (deployment model_info or the model cost map; unmapped names stay routable) is replaced by the nearest HIGHER tier holding a capable model, then default_model, else a clear 400. A kept session-affinity pin still wins even when an image arrives, unless modality_pin_override is also enabled. */
3564
+ modality_routing?: boolean;
3036
3565
  /** When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to at least this tier: the classified tier still wins when it is higher, and the floor also overrides a session-affinity pin to a lower tier for exactly the turns carrying the sentinel, without rewriting the pin -- the first turn after plan mode exits routes as if plan mode had never happened. Names a built-in tier, or with tier_definitions set, one of the defined tier names (list order is ascending severity, same as keyword_tier_rules). Unset disables detection entirely. The sentinels ride in client-injected prompt text, so a caller who pastes one can spend up to this tier's models -- never down, and never outside the configured pools. */
3037
3566
  plan_mode_min_tier?: string | null;
3038
3567
  /** Additional case-sensitive literal sentinels that mark a request as plan mode, on top of the built-in Claude Code and Copilot ones. For clients whose plan-mode wording the built-ins don't cover, or after a client release changes its strings. */
@@ -3043,7 +3572,7 @@ export interface RequestComplexityRouterConfig {
3043
3572
  reasoning_keywords?: RequestComplexityRouterConfigReasoningKeywordsList | null;
3044
3573
  /** Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a request the scorer placed in the cheapest tier; 0 restores the unconditional override */
3045
3574
  reasoning_override_min_score?: number | null;
3046
- /** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in default of ('<system-reminder>', '</system-reminder>'), so a harness that also emits that pair lists it too. Matching is case-insensitive. */
3575
+ /** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in system-reminder pair and the Codex envelope pairs enabled for Codex user agents, so list every built-in pair your harness also emits. Matching is case-insensitive. */
3047
3576
  reminder_markers?: RequestComplexityRouterConfigReminderMarkersList | null;
3048
3577
  /** Return the resolved raw model name in the response model field instead of the client-requested complexity-router alias */
3049
3578
  return_raw_model_name?: boolean;
@@ -3053,15 +3582,21 @@ export interface RequestComplexityRouterConfig {
3053
3582
  semantic_keyword_matching?: boolean;
3054
3583
  /** When True and a session_id is resolvable on the request, pin the model chosen on the session's first turn and reuse it for every later turn, skipping re-classification. Off by default so every turn is classified on its own merits and routed to the cheapest adequate tier. Set True to keep a multi-turn session on one model, which preserves provider prompt caches and avoids cross-model conversation-history errors. Always implies the deployment pin regardless of deployment_affinity: the session sticks to one deployment of the pinned model, since freezing the model while re-shuffling its deployments would still go cache-cold. */
3055
3584
  session_affinity?: boolean;
3056
- /** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity deployment pin, so it measures idle time for the session's routing decisions rather than total session length */
3585
+ /** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity per-tier model and deployment pins, so it measures idle time for the session's routing decisions rather than total session length */
3057
3586
  session_affinity_ttl_seconds?: number;
3058
3587
  /** Keywords indicating simple/basic queries */
3059
3588
  simple_keywords?: RequestComplexityRouterConfigSimpleKeywordsList | null;
3589
+ /** Escalate mid-task to the next-higher configured tier when the assistant's own recent tool calls look stuck: the newest tool call repeats, or errors, at least stall_escalation_repeat_threshold times across the last stall_escalation_window calls. Both tests are anchored on the newest call, so a task that tried the same thing a few times and then moved on is not escalated on the strength of those older calls alone, while a retry loop broken up by an unrelated lookup still counts. One tier at most, on the same ladder escalation_keywords bumps along, and never above the highest configured tier. Detection re-runs on every classified turn from the tool calls visible in that request, so it needs no state and nothing survives past the task. Mutually exclusive with session_affinity and classification_mode='user_turn', which both replay a held routing decision instead of classifying most turns, so this would never see the tool calls to look at. Off by default. */
3590
+ stall_escalation_enabled?: boolean;
3591
+ /** How many of the last stall_escalation_window tool calls must repeat the newest call, or must have errored alongside it, before the task counts as stalled. Must not exceed stall_escalation_window, or the condition could never be reached. */
3592
+ stall_escalation_repeat_threshold?: number;
3593
+ /** How many of the assistant's most recent tool calls stall detection looks at, oldest ones dropped as new calls happen. Counted across the whole visible conversation rather than reset at the newest human ask, so evidence from before a plain follow-up message like 'try again' is still visible on the turn after it. */
3594
+ stall_escalation_window?: number;
3060
3595
  /** Keywords indicating technical content */
3061
3596
  technical_keywords?: RequestComplexityRouterConfigTechnicalKeywordsList | null;
3062
3597
  /** Score boundaries between tiers. These keys (simple_medium, medium_complex, complex_reasoning) name the gaps between the default tier names and are not renameable by tier_labels; they are scorer knobs persisted by name on every routing decision */
3063
3598
  tier_boundaries?: RequestComplexityRouterConfigTierBoundariesMap;
3064
- /** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
3599
+ /** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm', 'jev' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
3065
3600
  tier_definitions?: RequestComplexityRouterConfigTierDefinitionsList | null;
3066
3601
  /** Score penalty per tier-step away from the classified tier when adaptive=True */
3067
3602
  tier_distance_penalty?: number;
@@ -3080,6 +3615,13 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
3080
3615
  RequestComplexityRouterConfigAdaptiveEligible,
3081
3616
  ),
3082
3617
  adaptive_weights: S.optional(AdaptiveRouterWeights),
3618
+ capability_classifier_config: S.optional(
3619
+ S.NullOr(CapabilityClassifierConfig),
3620
+ ),
3621
+ classification_examples: S.optional(S.NullOr(S.String)),
3622
+ classification_mode: S.optional(
3623
+ RequestComplexityRouterConfigClassificationMode,
3624
+ ),
3083
3625
  classification_prompt: S.optional(S.NullOr(S.String)),
3084
3626
  classifier_context_budget_chars: S.optional(S.Number),
3085
3627
  classifier_context_include_assistant_turns: S.optional(S.Boolean),
@@ -3095,6 +3637,10 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
3095
3637
  code_keywords: S.optional(
3096
3638
  S.NullOr(RequestComplexityRouterConfigCodeKeywordsList),
3097
3639
  ),
3640
+ context_window_escalation_buffer: S.optional(S.Number),
3641
+ custom_dimensions: S.optional(
3642
+ RequestComplexityRouterConfigCustomDimensionsList,
3643
+ ),
3098
3644
  custom_technical_keywords: S.optional(
3099
3645
  S.NullOr(RequestComplexityRouterConfigCustomTechnicalKeywordsList),
3100
3646
  ),
@@ -3104,18 +3650,29 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
3104
3650
  RequestComplexityRouterConfigDimensionWeightsMap,
3105
3651
  ),
3106
3652
  embedding_model: S.optional(S.NullOr(S.String)),
3653
+ enable_context_window_escalation: S.optional(S.Boolean),
3654
+ enable_non_reasoning_tier: S.optional(S.Boolean),
3107
3655
  escalation_keywords: S.optional(
3108
3656
  S.NullOr(RequestComplexityRouterConfigEscalationKeywordsList),
3109
3657
  ),
3110
3658
  fallback_tier: S.optional(S.NullOr(S.String)),
3111
3659
  heuristic_first_max_tier: S.optional(S.NullOr(S.String)),
3660
+ heuristic_v2_artifact: S.optional(
3661
+ RequestComplexityRouterConfigHeuristicV2Artifact,
3662
+ ),
3112
3663
  housekeeping_patterns: S.optional(
3113
3664
  S.NullOr(RequestComplexityRouterConfigHousekeepingPatternsList),
3114
3665
  ),
3666
+ hybrid_boundary_margin: S.optional(S.NullOr(S.Number)),
3667
+ jev_classifier_config: S.optional(S.NullOr(JevClassifierConfig)),
3115
3668
  keyword_tier_rules: S.optional(
3116
3669
  S.NullOr(RequestComplexityRouterConfigKeywordTierRulesList),
3117
3670
  ),
3671
+ llm_v2_config: S.optional(S.NullOr(LLMV2Config)),
3118
3672
  match_threshold: S.optional(S.Number),
3673
+ max_tokens_from_tier_model: S.optional(S.Boolean),
3674
+ modality_pin_override: S.optional(S.Boolean),
3675
+ modality_routing: S.optional(S.Boolean),
3119
3676
  plan_mode_min_tier: S.optional(S.NullOr(S.String)),
3120
3677
  plan_mode_patterns: S.optional(
3121
3678
  S.NullOr(RequestComplexityRouterConfigPlanModePatternsList),
@@ -3136,6 +3693,9 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
3136
3693
  simple_keywords: S.optional(
3137
3694
  S.NullOr(RequestComplexityRouterConfigSimpleKeywordsList),
3138
3695
  ),
3696
+ stall_escalation_enabled: S.optional(S.Boolean),
3697
+ stall_escalation_repeat_threshold: S.optional(S.Number),
3698
+ stall_escalation_window: S.optional(S.Number),
3139
3699
  technical_keywords: S.optional(
3140
3700
  S.NullOr(RequestComplexityRouterConfigTechnicalKeywordsList),
3141
3701
  ),
@@ -3259,24 +3819,71 @@ export const PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest =
3259
3819
 
3260
3820
  export type StandardLoggingRoutingDecisionCause =
3261
3821
  | "heuristic_scorer"
3822
+ | "heuristic_v2"
3262
3823
  | "reasoning_override"
3263
3824
  | "llm_classifier"
3825
+ | "capability_classifier"
3826
+ | "jev_classifier"
3827
+ | "llm_v2_classifier"
3828
+ | "llm_v2_fallback"
3264
3829
  | "heuristic_first_short_circuit"
3830
+ | "hybrid_short_circuit"
3265
3831
  | "classifier_plugin"
3266
3832
  | "classifier_fallback"
3833
+ | "capability_classifier_fallback"
3267
3834
  | "default_model_fallback"
3268
3835
  | "literal_keyword_match"
3269
3836
  | "semantic_keyword_match"
3270
3837
  | "plan_mode"
3271
3838
  | "housekeeping"
3839
+ | "modality_escalation"
3840
+ | "modality_pin_override"
3841
+ | "health_failover"
3842
+ | "health_default_fallback"
3272
3843
  | "session_affinity_pin"
3273
3844
  | "session_affinity_escalation"
3845
+ | "user_turn_continuation"
3274
3846
  | "default_fallback"
3275
3847
  | "keyword"
3276
3848
  | "quality_tier"
3277
3849
  | "bandit";
3278
3850
  export const StandardLoggingRoutingDecisionCause = S.String;
3279
3851
 
3852
+ export type StandardLoggingRoutingDecisionClassifierProbabilitiesMap = {
3853
+ [key: string]: number | undefined;
3854
+ };
3855
+ export const StandardLoggingRoutingDecisionClassifierProbabilitiesMap =
3856
+ /*@__PURE__*/ S.Record(
3857
+ S.String,
3858
+ S.Number,
3859
+ ) as any as S.Schema<StandardLoggingRoutingDecisionClassifierProbabilitiesMap>;
3860
+
3861
+ export type StandardLoggingHeuristicV2ForecastProbabilitiesMap = {
3862
+ [key: string]: number | undefined;
3863
+ };
3864
+ export const StandardLoggingHeuristicV2ForecastProbabilitiesMap =
3865
+ /*@__PURE__*/ S.Record(
3866
+ S.String,
3867
+ S.Number,
3868
+ ) as any as S.Schema<StandardLoggingHeuristicV2ForecastProbabilitiesMap>;
3869
+
3870
+ export interface StandardLoggingHeuristicV2Forecast {
3871
+ predicted_tier: string;
3872
+ probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap;
3873
+ request_type: string;
3874
+ threshold: number;
3875
+ }
3876
+ export const StandardLoggingHeuristicV2Forecast = /*@__PURE__*/ S.suspend(() =>
3877
+ S.Struct({
3878
+ predicted_tier: S.String,
3879
+ probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap,
3880
+ request_type: S.String,
3881
+ threshold: S.Number,
3882
+ }),
3883
+ ).annotate({
3884
+ identifier: "StandardLoggingHeuristicV2Forecast",
3885
+ }) as any as S.Schema<StandardLoggingHeuristicV2Forecast>;
3886
+
3280
3887
  export type StandardLoggingRoutingDecisionRouterType =
3281
3888
  | "complexity"
3282
3889
  | "adaptive"
@@ -3317,11 +3924,29 @@ export const StandardLoggingRoutingDecisionTierLitellmParamsMap =
3317
3924
  /** Per-request provenance for a pre-routing strategy (auto-router) decision. */
3318
3925
  export interface StandardLoggingRoutingDecision {
3319
3926
  cause?: StandardLoggingRoutingDecisionCause;
3927
+ classifier_calibrated_capable_p_solve?: number;
3928
+ classifier_calibrated_efficient_p_solve?: number;
3929
+ classifier_calibrated_p_solve?: number;
3930
+ classifier_calibration_version?: string;
3931
+ classifier_capability_boundary?: string;
3932
+ classifier_capable_p_solve?: number;
3933
+ classifier_confidence?: number;
3320
3934
  classifier_cost?: number;
3935
+ classifier_crux?: string;
3936
+ classifier_efficient_p_solve?: number;
3937
+ classifier_max_quality_gap?: number;
3321
3938
  classifier_model?: string;
3939
+ classifier_p_solve?: number;
3940
+ classifier_primary_rule?: string;
3941
+ classifier_probabilities?: StandardLoggingRoutingDecisionClassifierProbabilitiesMap;
3942
+ classifier_prompt_version?: string;
3943
+ classifier_threshold?: number;
3944
+ context_escalated?: boolean;
3945
+ context_escalation_original_tier?: string;
3322
3946
  conversation_continuing?: boolean;
3323
3947
  escalated?: boolean;
3324
3948
  escalation_keyword?: string;
3949
+ heuristic_v2_forecast?: StandardLoggingHeuristicV2Forecast;
3325
3950
  matched_keyword?: string;
3326
3951
  reasoning_override_min_score?: number;
3327
3952
  request_type?: string;
@@ -3340,11 +3965,31 @@ export interface StandardLoggingRoutingDecision {
3340
3965
  export const StandardLoggingRoutingDecision = /*@__PURE__*/ S.suspend(() =>
3341
3966
  S.Struct({
3342
3967
  cause: S.optional(StandardLoggingRoutingDecisionCause),
3968
+ classifier_calibrated_capable_p_solve: S.optional(S.Number),
3969
+ classifier_calibrated_efficient_p_solve: S.optional(S.Number),
3970
+ classifier_calibrated_p_solve: S.optional(S.Number),
3971
+ classifier_calibration_version: S.optional(S.String),
3972
+ classifier_capability_boundary: S.optional(S.String),
3973
+ classifier_capable_p_solve: S.optional(S.Number),
3974
+ classifier_confidence: S.optional(S.Number),
3343
3975
  classifier_cost: S.optional(S.Number),
3976
+ classifier_crux: S.optional(S.String),
3977
+ classifier_efficient_p_solve: S.optional(S.Number),
3978
+ classifier_max_quality_gap: S.optional(S.Number),
3344
3979
  classifier_model: S.optional(S.String),
3980
+ classifier_p_solve: S.optional(S.Number),
3981
+ classifier_primary_rule: S.optional(S.String),
3982
+ classifier_probabilities: S.optional(
3983
+ StandardLoggingRoutingDecisionClassifierProbabilitiesMap,
3984
+ ),
3985
+ classifier_prompt_version: S.optional(S.String),
3986
+ classifier_threshold: S.optional(S.Number),
3987
+ context_escalated: S.optional(S.Boolean),
3988
+ context_escalation_original_tier: S.optional(S.String),
3345
3989
  conversation_continuing: S.optional(S.Boolean),
3346
3990
  escalated: S.optional(S.Boolean),
3347
3991
  escalation_keyword: S.optional(S.String),
3992
+ heuristic_v2_forecast: S.optional(StandardLoggingHeuristicV2Forecast),
3348
3993
  matched_keyword: S.optional(S.String),
3349
3994
  reasoning_override_min_score: S.optional(S.Number),
3350
3995
  request_type: S.optional(S.String),
@@ -3964,7 +4609,7 @@ export const getModelCostMapReloadStatusScheduleModelCostMapReloadStatusGet: API
3964
4609
  }));
3965
4610
 
3966
4611
  export type GetModelCostMapSourceModelCostMapSourceGetError = LitellmOpError;
3967
- /** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - model_count: number of models in the currently loaded cost map */
4612
+ /** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - loaded_at: when this pod last loaded the map - source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it - etag: the ETag of the remote fetch (null for the bundled backup) - model_count: number of models in the currently loaded cost map */
3968
4613
  export const getModelCostMapSourceModelCostMapSourceGet: API.OperationMethod<
3969
4614
  GetModelCostMapSourceModelCostMapSourceGetRequest,
3970
4615
  GetModelCostMapSourceModelCostMapSourceGetResponse,
@@ -4047,7 +4692,7 @@ export const getModelInfoModelsModelId: API.OperationMethod<
4047
4692
  }));
4048
4693
 
4049
4694
  export type GetModelInfoV1ModelInfoError = UnprocessableEntity | LitellmOpError;
4050
- /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: Returns a dictionary containing information about each model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
4695
+ /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
4051
4696
  export const getModelInfoV1ModelInfo: API.OperationMethod<
4052
4697
  GetModelInfoV1ModelInfoRequest,
4053
4698
  GetModelInfoV1ModelInfoResponse,
@@ -4081,7 +4726,7 @@ export const getModelInfoV1ModelsModelId: API.OperationMethod<
4081
4726
  export type GetModelInfoV1V1ModelInfoError =
4082
4727
  | UnprocessableEntity
4083
4728
  | LitellmOpError;
4084
- /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: Returns a dictionary containing information about each model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
4729
+ /** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
4085
4730
  export const getModelInfoV1V1ModelInfo: API.OperationMethod<
4086
4731
  GetModelInfoV1V1ModelInfoRequest,
4087
4732
  GetModelInfoV1V1ModelInfoResponse,
@@ -4098,7 +4743,7 @@ export const getModelInfoV1V1ModelInfo: API.OperationMethod<
4098
4743
  export type GetModelInfoV2V2ModelInfoError =
4099
4744
  | UnprocessableEntity
4100
4745
  | LitellmOpError;
4101
- /** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
4746
+ /** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. access_group: Only return deployments in this model access group. wildcard_only: Only return deployments whose `model_name` contains `*`. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
4102
4747
  export const getModelInfoV2V2ModelInfo: API.OperationMethod<
4103
4748
  GetModelInfoV2V2ModelInfoRequest,
4104
4749
  GetModelInfoV2V2ModelInfoResponse,
@@ -4485,7 +5130,7 @@ export const updateUsefulLinksModelHubUpdateUsefulLinksPost: API.OperationMethod
4485
5130
  export type ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError =
4486
5131
  | UnprocessableEntity
4487
5132
  | LitellmOpError;
4488
- /** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Gated exactly like the save it rehearses: a proxy admin, or a team admin naming their own team. Nothing is created, routed, or billed. */
5133
+ /** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Uses the same team opt-in and model-access checks as configuration writes for members. Nothing is created, routed, or billed. */
4489
5134
  export const validateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPost: API.OperationMethod<
4490
5135
  ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostRequest,
4491
5136
  ComplexityRouterConfigValidationResponse,