@homeflare/distilled-litellm 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (390) hide show
  1. package/README.md +48 -12
  2. package/dist/services/a2a.js +1 -1
  3. package/dist/services/a2a_registration.js +1 -1
  4. package/dist/services/access_groups.d.ts +18 -0
  5. package/dist/services/access_groups.d.ts.map +1 -1
  6. package/dist/services/access_groups.js +15 -1
  7. package/dist/services/access_groups.js.map +1 -1
  8. package/dist/services/adaptive_router.js +1 -1
  9. package/dist/services/agents.d.ts +10 -0
  10. package/dist/services/agents.d.ts.map +1 -1
  11. package/dist/services/agents.js +10 -1
  12. package/dist/services/agents.js.map +1 -1
  13. package/dist/services/alerting.js +1 -1
  14. package/dist/services/anthropic_passthrough.js +1 -1
  15. package/dist/services/anthropic_skills.d.ts +7 -1
  16. package/dist/services/anthropic_skills.d.ts.map +1 -1
  17. package/dist/services/anthropic_skills.js +6 -2
  18. package/dist/services/anthropic_skills.js.map +1 -1
  19. package/dist/services/assistants.js +1 -1
  20. package/dist/services/audio.js +1 -1
  21. package/dist/services/audit_logging.d.ts +2 -0
  22. package/dist/services/audit_logging.d.ts.map +1 -1
  23. package/dist/services/audit_logging.js +2 -1
  24. package/dist/services/audit_logging.js.map +1 -1
  25. package/dist/services/auto_router.d.ts +162 -62
  26. package/dist/services/auto_router.d.ts.map +1 -1
  27. package/dist/services/auto_router.js +92 -31
  28. package/dist/services/auto_router.js.map +1 -1
  29. package/dist/services/batch.js +1 -1
  30. package/dist/services/beta_agents.js +1 -1
  31. package/dist/services/beta_mcp.js +1 -1
  32. package/dist/services/budget_management.d.ts +8 -3
  33. package/dist/services/budget_management.d.ts.map +1 -1
  34. package/dist/services/budget_management.js +7 -4
  35. package/dist/services/budget_management.js.map +1 -1
  36. package/dist/services/budget_spend_tracking.d.ts +24 -5
  37. package/dist/services/budget_spend_tracking.d.ts.map +1 -1
  38. package/dist/services/budget_spend_tracking.js +17 -4
  39. package/dist/services/budget_spend_tracking.js.map +1 -1
  40. package/dist/services/cache_settings.js +1 -1
  41. package/dist/services/caching.js +1 -1
  42. package/dist/services/chat_completions.js +1 -1
  43. package/dist/services/claude_code_marketplace.d.ts +11 -10
  44. package/dist/services/claude_code_marketplace.d.ts.map +1 -1
  45. package/dist/services/claude_code_marketplace.js +8 -6
  46. package/dist/services/claude_code_marketplace.js.map +1 -1
  47. package/dist/services/cloudzero.js +1 -1
  48. package/dist/services/completions.js +1 -1
  49. package/dist/services/compliance.js +1 -1
  50. package/dist/services/config_overrides.d.ts +5 -1
  51. package/dist/services/config_overrides.d.ts.map +1 -1
  52. package/dist/services/config_overrides.js +3 -1
  53. package/dist/services/config_overrides.js.map +1 -1
  54. package/dist/services/config_yaml.d.ts +120 -6
  55. package/dist/services/config_yaml.d.ts.map +1 -1
  56. package/dist/services/config_yaml.js +67 -1
  57. package/dist/services/config_yaml.js.map +1 -1
  58. package/dist/services/containers.d.ts +8 -2
  59. package/dist/services/containers.d.ts.map +1 -1
  60. package/dist/services/containers.js +13 -5
  61. package/dist/services/containers.js.map +1 -1
  62. package/dist/services/coordination_redis_settings.js +1 -1
  63. package/dist/services/cost_tracking.d.ts +91 -1
  64. package/dist/services/cost_tracking.d.ts.map +1 -1
  65. package/dist/services/cost_tracking.js +82 -2
  66. package/dist/services/cost_tracking.js.map +1 -1
  67. package/dist/services/credential_management.d.ts +16 -3
  68. package/dist/services/credential_management.d.ts.map +1 -1
  69. package/dist/services/credential_management.js +13 -4
  70. package/dist/services/credential_management.js.map +1 -1
  71. package/dist/services/customer_management.d.ts +25 -2
  72. package/dist/services/customer_management.d.ts.map +1 -1
  73. package/dist/services/customer_management.js +22 -3
  74. package/dist/services/customer_management.js.map +1 -1
  75. package/dist/services/email_management.js +1 -1
  76. package/dist/services/embeddings.js +1 -1
  77. package/dist/services/evals.js +1 -1
  78. package/dist/services/experimental.js +1 -1
  79. package/dist/services/fallback_management.js +1 -1
  80. package/dist/services/files.js +1 -1
  81. package/dist/services/fine_tuning.js +1 -1
  82. package/dist/services/gemini_agents.js +1 -1
  83. package/dist/services/google_genai_endpoints.js +1 -1
  84. package/dist/services/guardrails.d.ts +80 -7
  85. package/dist/services/guardrails.d.ts.map +1 -1
  86. package/dist/services/guardrails.js +37 -1
  87. package/dist/services/guardrails.js.map +1 -1
  88. package/dist/services/health.d.ts +2 -2
  89. package/dist/services/health.d.ts.map +1 -1
  90. package/dist/services/health.js +2 -2
  91. package/dist/services/health.js.map +1 -1
  92. package/dist/services/images.js +1 -1
  93. package/dist/services/index.d.ts +1 -15
  94. package/dist/services/index.d.ts.map +1 -1
  95. package/dist/services/index.js +1 -15
  96. package/dist/services/index.js.map +1 -1
  97. package/dist/services/internal_user_management.d.ts +227 -39
  98. package/dist/services/internal_user_management.d.ts.map +1 -1
  99. package/dist/services/internal_user_management.js +194 -37
  100. package/dist/services/internal_user_management.js.map +1 -1
  101. package/dist/services/invite_links.js +1 -1
  102. package/dist/services/jwt_mappings.d.ts +3 -0
  103. package/dist/services/jwt_mappings.d.ts.map +1 -1
  104. package/dist/services/jwt_mappings.js +4 -1
  105. package/dist/services/jwt_mappings.js.map +1 -1
  106. package/dist/services/key_management.d.ts +116 -50
  107. package/dist/services/key_management.d.ts.map +1 -1
  108. package/dist/services/key_management.js +95 -40
  109. package/dist/services/key_management.js.map +1 -1
  110. package/dist/services/langfuse_passthrough.js +1 -1
  111. package/dist/services/llm_passthrough.d.ts +1199 -0
  112. package/dist/services/llm_passthrough.d.ts.map +1 -0
  113. package/dist/services/llm_passthrough.js +2150 -0
  114. package/dist/services/llm_passthrough.js.map +1 -0
  115. package/dist/services/llm_utils.js +1 -1
  116. package/dist/services/logging_callbacks.js +1 -1
  117. package/dist/services/mcp_byok_oauth.d.ts +18 -11
  118. package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
  119. package/dist/services/mcp_byok_oauth.js +27 -24
  120. package/dist/services/mcp_byok_oauth.js.map +1 -1
  121. package/dist/services/mcp_discoverable.d.ts +34 -0
  122. package/dist/services/mcp_discoverable.d.ts.map +1 -1
  123. package/dist/services/mcp_discoverable.js +51 -1
  124. package/dist/services/mcp_discoverable.js.map +1 -1
  125. package/dist/services/mcp_management.d.ts +90 -2
  126. package/dist/services/mcp_management.d.ts.map +1 -1
  127. package/dist/services/mcp_management.js +115 -3
  128. package/dist/services/mcp_management.js.map +1 -1
  129. package/dist/services/mcp_rest.d.ts +13 -0
  130. package/dist/services/mcp_rest.d.ts.map +1 -1
  131. package/dist/services/mcp_rest.js +10 -2
  132. package/dist/services/mcp_rest.js.map +1 -1
  133. package/dist/services/memory_management.d.ts +2 -0
  134. package/dist/services/memory_management.d.ts.map +1 -1
  135. package/dist/services/memory_management.js +2 -1
  136. package/dist/services/memory_management.js.map +1 -1
  137. package/dist/services/misc.d.ts +86 -1
  138. package/dist/services/misc.d.ts.map +1 -1
  139. package/dist/services/misc.js +126 -2
  140. package/dist/services/misc.js.map +1 -1
  141. package/dist/services/model_management.d.ts +310 -19
  142. package/dist/services/model_management.d.ts.map +1 -1
  143. package/dist/services/model_management.js +240 -7
  144. package/dist/services/model_management.js.map +1 -1
  145. package/dist/services/moderations.js +1 -1
  146. package/dist/services/ocr.js +1 -1
  147. package/dist/services/open_ai_pass_through.d.ts +35 -90
  148. package/dist/services/open_ai_pass_through.d.ts.map +1 -1
  149. package/dist/services/open_ai_pass_through.js +41 -139
  150. package/dist/services/open_ai_pass_through.js.map +1 -1
  151. package/dist/services/organization_management.d.ts +21 -1
  152. package/dist/services/organization_management.d.ts.map +1 -1
  153. package/dist/services/organization_management.js +20 -2
  154. package/dist/services/organization_management.js.map +1 -1
  155. package/dist/services/plugins.js +1 -1
  156. package/dist/services/policies.d.ts +2 -0
  157. package/dist/services/policies.d.ts.map +1 -1
  158. package/dist/services/policies.js +3 -1
  159. package/dist/services/policies.js.map +1 -1
  160. package/dist/services/policy_engine.d.ts +6 -0
  161. package/dist/services/policy_engine.d.ts.map +1 -1
  162. package/dist/services/policy_engine.js +4 -1
  163. package/dist/services/policy_engine.js.map +1 -1
  164. package/dist/services/project_management.d.ts +15 -0
  165. package/dist/services/project_management.d.ts.map +1 -1
  166. package/dist/services/project_management.js +14 -1
  167. package/dist/services/project_management.js.map +1 -1
  168. package/dist/services/prompts.js +1 -1
  169. package/dist/services/public.d.ts +124 -0
  170. package/dist/services/public.d.ts.map +1 -1
  171. package/dist/services/public.js +158 -1
  172. package/dist/services/public.js.map +1 -1
  173. package/dist/services/rag.js +1 -1
  174. package/dist/services/realtime.d.ts +4 -26
  175. package/dist/services/realtime.d.ts.map +1 -1
  176. package/dist/services/realtime.js +7 -57
  177. package/dist/services/realtime.js.map +1 -1
  178. package/dist/services/rerank.d.ts +72 -0
  179. package/dist/services/rerank.d.ts.map +1 -1
  180. package/dist/services/rerank.js +61 -4
  181. package/dist/services/rerank.js.map +1 -1
  182. package/dist/services/responses.d.ts +30 -0
  183. package/dist/services/responses.d.ts.map +1 -1
  184. package/dist/services/responses.js +59 -1
  185. package/dist/services/responses.js.map +1 -1
  186. package/dist/services/router_settings.js +1 -1
  187. package/dist/services/rust_control_plane.js +1 -1
  188. package/dist/services/scim.d.ts +41 -1
  189. package/dist/services/scim.d.ts.map +1 -1
  190. package/dist/services/scim.js +60 -2
  191. package/dist/services/scim.js.map +1 -1
  192. package/dist/services/search.js +1 -1
  193. package/dist/services/search_tools.js +1 -1
  194. package/dist/services/settings.d.ts +88 -0
  195. package/dist/services/settings.d.ts.map +1 -1
  196. package/dist/services/settings.js +113 -1
  197. package/dist/services/settings.js.map +1 -1
  198. package/dist/services/sso_settings.d.ts +1 -1
  199. package/dist/services/sso_settings.d.ts.map +1 -1
  200. package/dist/services/sso_settings.js +1 -1
  201. package/dist/services/sso_settings.js.map +1 -1
  202. package/dist/services/tag_management.d.ts +7 -0
  203. package/dist/services/tag_management.d.ts.map +1 -1
  204. package/dist/services/tag_management.js +8 -1
  205. package/dist/services/tag_management.js.map +1 -1
  206. package/dist/services/team_management.d.ts +265 -17
  207. package/dist/services/team_management.d.ts.map +1 -1
  208. package/dist/services/team_management.js +247 -12
  209. package/dist/services/team_management.js.map +1 -1
  210. package/dist/services/tools.js +1 -1
  211. package/dist/services/ui_settings.js +1 -1
  212. package/dist/services/ui_theme_settings.js +1 -1
  213. package/dist/services/usage_ai.js +1 -1
  214. package/dist/services/vantage.js +1 -1
  215. package/dist/services/vector_store_management.js +1 -1
  216. package/dist/services/vector_stores.js +1 -1
  217. package/dist/services/videos.js +1 -1
  218. package/dist/services/web_socket.d.ts +0 -18
  219. package/dist/services/web_socket.d.ts.map +1 -1
  220. package/dist/services/web_socket.js +1 -33
  221. package/dist/services/web_socket.js.map +1 -1
  222. package/dist/services/workflow_management.js +1 -1
  223. package/package.json +1 -1
  224. package/src/services/a2a.ts +1 -1
  225. package/src/services/a2a_registration.ts +1 -1
  226. package/src/services/access_groups.ts +44 -1
  227. package/src/services/adaptive_router.ts +1 -1
  228. package/src/services/agents.ts +22 -1
  229. package/src/services/alerting.ts +1 -1
  230. package/src/services/anthropic_passthrough.ts +1 -1
  231. package/src/services/anthropic_skills.ts +12 -2
  232. package/src/services/assistants.ts +1 -1
  233. package/src/services/audio.ts +1 -1
  234. package/src/services/audit_logging.ts +4 -1
  235. package/src/services/auto_router.ts +303 -93
  236. package/src/services/batch.ts +1 -1
  237. package/src/services/beta_agents.ts +1 -1
  238. package/src/services/beta_mcp.ts +1 -1
  239. package/src/services/budget_management.ts +12 -4
  240. package/src/services/budget_spend_tracking.ts +38 -6
  241. package/src/services/cache_settings.ts +1 -1
  242. package/src/services/caching.ts +1 -1
  243. package/src/services/chat_completions.ts +1 -1
  244. package/src/services/claude_code_marketplace.ts +20 -14
  245. package/src/services/cloudzero.ts +1 -1
  246. package/src/services/completions.ts +1 -1
  247. package/src/services/compliance.ts +1 -1
  248. package/src/services/config_overrides.ts +8 -2
  249. package/src/services/config_yaml.ts +251 -6
  250. package/src/services/containers.ts +29 -11
  251. package/src/services/coordination_redis_settings.ts +1 -1
  252. package/src/services/cost_tracking.ts +198 -2
  253. package/src/services/credential_management.ts +25 -6
  254. package/src/services/customer_management.ts +49 -3
  255. package/src/services/email_management.ts +1 -1
  256. package/src/services/embeddings.ts +1 -1
  257. package/src/services/evals.ts +1 -1
  258. package/src/services/experimental.ts +1 -1
  259. package/src/services/fallback_management.ts +1 -1
  260. package/src/services/files.ts +1 -1
  261. package/src/services/fine_tuning.ts +1 -1
  262. package/src/services/gemini_agents.ts +1 -1
  263. package/src/services/google_genai_endpoints.ts +1 -1
  264. package/src/services/guardrails.ts +198 -8
  265. package/src/services/health.ts +3 -2
  266. package/src/services/images.ts +1 -1
  267. package/src/services/index.ts +1 -15
  268. package/src/services/internal_user_management.ts +565 -103
  269. package/src/services/invite_links.ts +1 -1
  270. package/src/services/jwt_mappings.ts +7 -1
  271. package/src/services/key_management.ts +257 -127
  272. package/src/services/langfuse_passthrough.ts +1 -1
  273. package/src/services/llm_passthrough.ts +4542 -0
  274. package/src/services/llm_utils.ts +1 -1
  275. package/src/services/logging_callbacks.ts +1 -1
  276. package/src/services/mcp_byok_oauth.ts +65 -45
  277. package/src/services/mcp_discoverable.ts +109 -1
  278. package/src/services/mcp_management.ts +258 -3
  279. package/src/services/mcp_rest.ts +30 -5
  280. package/src/services/memory_management.ts +4 -1
  281. package/src/services/misc.ts +269 -2
  282. package/src/services/model_management.ts +666 -21
  283. package/src/services/moderations.ts +1 -1
  284. package/src/services/ocr.ts +1 -1
  285. package/src/services/open_ai_pass_through.ts +81 -286
  286. package/src/services/organization_management.ts +44 -2
  287. package/src/services/plugins.ts +1 -1
  288. package/src/services/policies.ts +5 -1
  289. package/src/services/policy_engine.ts +10 -1
  290. package/src/services/project_management.ts +33 -1
  291. package/src/services/prompts.ts +1 -1
  292. package/src/services/public.ts +356 -1
  293. package/src/services/rag.ts +1 -1
  294. package/src/services/realtime.ts +11 -111
  295. package/src/services/rerank.ts +187 -7
  296. package/src/services/responses.ts +117 -1
  297. package/src/services/router_settings.ts +1 -1
  298. package/src/services/rust_control_plane.ts +1 -1
  299. package/src/services/scim.ts +134 -3
  300. package/src/services/search.ts +1 -1
  301. package/src/services/search_tools.ts +1 -1
  302. package/src/services/settings.ts +278 -1
  303. package/src/services/sso_settings.ts +2 -1
  304. package/src/services/tag_management.ts +15 -1
  305. package/src/services/team_management.ts +676 -37
  306. package/src/services/tools.ts +1 -1
  307. package/src/services/ui_settings.ts +1 -1
  308. package/src/services/ui_theme_settings.ts +1 -1
  309. package/src/services/usage_ai.ts +1 -1
  310. package/src/services/vantage.ts +1 -1
  311. package/src/services/vector_store_management.ts +1 -1
  312. package/src/services/vector_stores.ts +1 -1
  313. package/src/services/videos.ts +1 -1
  314. package/src/services/web_socket.ts +1 -61
  315. package/src/services/workflow_management.ts +1 -1
  316. package/dist/services/anthropic_pass_through.d.ts +0 -70
  317. package/dist/services/anthropic_pass_through.d.ts.map +0 -1
  318. package/dist/services/anthropic_pass_through.js +0 -114
  319. package/dist/services/anthropic_pass_through.js.map +0 -1
  320. package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
  321. package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
  322. package/dist/services/assembly_ai_eu_pass_through.js +0 -114
  323. package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
  324. package/dist/services/assembly_ai_pass_through.d.ts +0 -70
  325. package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
  326. package/dist/services/assembly_ai_pass_through.js +0 -114
  327. package/dist/services/assembly_ai_pass_through.js.map +0 -1
  328. package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
  329. package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
  330. package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
  331. package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
  332. package/dist/services/azure_ai_pass_through.d.ts +0 -70
  333. package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
  334. package/dist/services/azure_ai_pass_through.js +0 -112
  335. package/dist/services/azure_ai_pass_through.js.map +0 -1
  336. package/dist/services/azure_pass_through.d.ts +0 -70
  337. package/dist/services/azure_pass_through.d.ts.map +0 -1
  338. package/dist/services/azure_pass_through.js +0 -107
  339. package/dist/services/azure_pass_through.js.map +0 -1
  340. package/dist/services/bedrock_pass_through.d.ts +0 -70
  341. package/dist/services/bedrock_pass_through.d.ts.map +0 -1
  342. package/dist/services/bedrock_pass_through.js +0 -114
  343. package/dist/services/bedrock_pass_through.js.map +0 -1
  344. package/dist/services/cohere_pass_through.d.ts +0 -70
  345. package/dist/services/cohere_pass_through.d.ts.map +0 -1
  346. package/dist/services/cohere_pass_through.js +0 -112
  347. package/dist/services/cohere_pass_through.js.map +0 -1
  348. package/dist/services/cursor_pass_through.d.ts +0 -70
  349. package/dist/services/cursor_pass_through.d.ts.map +0 -1
  350. package/dist/services/cursor_pass_through.js +0 -112
  351. package/dist/services/cursor_pass_through.js.map +0 -1
  352. package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
  353. package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
  354. package/dist/services/google_ai_studio_pass_through.js +0 -112
  355. package/dist/services/google_ai_studio_pass_through.js.map +0 -1
  356. package/dist/services/milvus_pass_through.d.ts +0 -70
  357. package/dist/services/milvus_pass_through.d.ts.map +0 -1
  358. package/dist/services/milvus_pass_through.js +0 -112
  359. package/dist/services/milvus_pass_through.js.map +0 -1
  360. package/dist/services/mistral_pass_through.d.ts +0 -70
  361. package/dist/services/mistral_pass_through.d.ts.map +0 -1
  362. package/dist/services/mistral_pass_through.js +0 -114
  363. package/dist/services/mistral_pass_through.js.map +0 -1
  364. package/dist/services/vertex_ai_pass_through.d.ts +0 -180
  365. package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
  366. package/dist/services/vertex_ai_pass_through.js +0 -334
  367. package/dist/services/vertex_ai_pass_through.js.map +0 -1
  368. package/dist/services/vllm_pass_through.d.ts +0 -70
  369. package/dist/services/vllm_pass_through.d.ts.map +0 -1
  370. package/dist/services/vllm_pass_through.js +0 -104
  371. package/dist/services/vllm_pass_through.js.map +0 -1
  372. package/dist/services/watsonx_pass_through.d.ts +0 -70
  373. package/dist/services/watsonx_pass_through.d.ts.map +0 -1
  374. package/dist/services/watsonx_pass_through.js +0 -114
  375. package/dist/services/watsonx_pass_through.js.map +0 -1
  376. package/src/services/anthropic_pass_through.ts +0 -233
  377. package/src/services/assembly_ai_eu_pass_through.ts +0 -237
  378. package/src/services/assembly_ai_pass_through.ts +0 -237
  379. package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
  380. package/src/services/azure_ai_pass_through.ts +0 -231
  381. package/src/services/azure_pass_through.ts +0 -227
  382. package/src/services/bedrock_pass_through.ts +0 -229
  383. package/src/services/cohere_pass_through.ts +0 -227
  384. package/src/services/cursor_pass_through.ts +0 -227
  385. package/src/services/google_ai_studio_pass_through.ts +0 -227
  386. package/src/services/milvus_pass_through.ts +0 -227
  387. package/src/services/mistral_pass_through.ts +0 -229
  388. package/src/services/vertex_ai_pass_through.ts +0 -684
  389. package/src/services/vllm_pass_through.ts +0 -227
  390. package/src/services/watsonx_pass_through.ts +0 -229
@@ -1,4 +1,4 @@
1
- // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.100.0 — see README). Do not edit.
1
+ // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
2
2
  import * as S from "@distilled.cloud/core/schema";
3
3
  import * as API from "@distilled.cloud/core/api";
4
4
  import * as C from "@distilled.cloud/core/category";
@@ -26,12 +26,15 @@ export interface GetAutoRouterBenchmarksAutoRouterBenchmarksGetRequest {
26
26
  start_date?: string;
27
27
  /** YYYY-MM-DD UTC, inclusive (defaults to today) */
28
28
  end_date?: string;
29
+ /** Filter to one virtual key token hash */
30
+ api_key?: string;
29
31
  }
30
32
  export const GetAutoRouterBenchmarksAutoRouterBenchmarksGetRequest =
31
33
  /*@__PURE__*/ S.suspend(() =>
32
34
  S.Struct({
33
35
  start_date: S.optional(S.String.pipe(T.Query())),
34
36
  end_date: S.optional(S.String.pipe(T.Query())),
37
+ api_key: S.optional(S.String.pipe(T.Query())),
35
38
  }).pipe(
36
39
  T.Http({ method: "GET", uri: "/auto_router/benchmarks", code: 200 }),
37
40
  ),
@@ -112,18 +115,25 @@ export interface AutoRouterBenchmarkGroup {
112
115
  avg_session_seconds: number;
113
116
  avg_tokens_per_session: number;
114
117
  avg_turns_per_session: number;
115
- /** spend plus saved_spend: the estimated single-model cost */
116
- baseline_spend: number;
118
+ /** Estimated single-model cost for covered turns only */
119
+ baseline_spend: number | null;
117
120
  cache: AutoRouterCacheStats;
121
+ /** Recorded LLM classifier cost already included in spend; null when any session turns predate subtotal recording, and zero for an empty window */
122
+ classifier_cost: number | null;
118
123
  /** The auto-router alias requests were sent to */
119
124
  router_name: string;
120
125
  /** complexity, adaptive or quality */
121
126
  router_type: string;
122
- /** saved_spend over baseline_spend, as a percentage */
123
- saved_pct: number;
124
- saved_per_session: number;
125
- /** Signed dollars saved versus each router's savings baseline (derived from its hardest tier, or the configured override), from the same per-request savings record the usage tab reads */
126
- saved_spend: number;
127
+ /** Covered savings over covered baseline spend, as a percentage */
128
+ saved_pct: number | null;
129
+ /** Average session savings; unavailable unless every turn is covered */
130
+ saved_per_session: number | null;
131
+ /** Signed savings for covered turns only; null when traffic has no current estimates */
132
+ saved_spend: number | null;
133
+ /** Actual spend, including classifier cost, for covered turns only */
134
+ savings_estimated_actual_spend: number;
135
+ /** Turns covered by the current savings estimator; legacy estimates are excluded */
136
+ savings_estimated_turns: number;
127
137
  sessions: number;
128
138
  /** What the routed traffic actually cost */
129
139
  spend: number;
@@ -136,13 +146,16 @@ export const AutoRouterBenchmarkGroup = /*@__PURE__*/ S.suspend(() =>
136
146
  avg_session_seconds: S.Number,
137
147
  avg_tokens_per_session: S.Number,
138
148
  avg_turns_per_session: S.Number,
139
- baseline_spend: S.Number,
149
+ baseline_spend: S.NullOr(S.Number),
140
150
  cache: AutoRouterCacheStats,
151
+ classifier_cost: S.NullOr(S.Number),
141
152
  router_name: S.String,
142
153
  router_type: S.String,
143
- saved_pct: S.Number,
144
- saved_per_session: S.Number,
145
- saved_spend: S.Number,
154
+ saved_pct: S.NullOr(S.Number),
155
+ saved_per_session: S.NullOr(S.Number),
156
+ saved_spend: S.NullOr(S.Number),
157
+ savings_estimated_actual_spend: S.Number,
158
+ savings_estimated_turns: S.Number,
146
159
  sessions: S.Number,
147
160
  spend: S.Number,
148
161
  tier_turns: S.optional(AutoRouterBenchmarkGroupTierTurnsMap),
@@ -164,14 +177,21 @@ export interface AutoRouterBenchmarkTotals {
164
177
  avg_session_seconds: number;
165
178
  avg_tokens_per_session: number;
166
179
  avg_turns_per_session: number;
167
- /** spend plus saved_spend: the estimated single-model cost */
168
- baseline_spend: number;
180
+ /** Estimated single-model cost for covered turns only */
181
+ baseline_spend: number | null;
169
182
  cache: AutoRouterCacheStats;
170
- /** saved_spend over baseline_spend, as a percentage */
171
- saved_pct: number;
172
- saved_per_session: number;
173
- /** Signed dollars saved versus each router's savings baseline (derived from its hardest tier, or the configured override), from the same per-request savings record the usage tab reads */
174
- saved_spend: number;
183
+ /** Recorded LLM classifier cost already included in spend; null when any session turns predate subtotal recording, and zero for an empty window */
184
+ classifier_cost: number | null;
185
+ /** Covered savings over covered baseline spend, as a percentage */
186
+ saved_pct: number | null;
187
+ /** Average session savings; unavailable unless every turn is covered */
188
+ saved_per_session: number | null;
189
+ /** Signed savings for covered turns only; null when traffic has no current estimates */
190
+ saved_spend: number | null;
191
+ /** Actual spend, including classifier cost, for covered turns only */
192
+ savings_estimated_actual_spend: number;
193
+ /** Turns covered by the current savings estimator; legacy estimates are excluded */
194
+ savings_estimated_turns: number;
175
195
  sessions: number;
176
196
  /** What the routed traffic actually cost */
177
197
  spend: number;
@@ -182,11 +202,14 @@ export const AutoRouterBenchmarkTotals = /*@__PURE__*/ S.suspend(() =>
182
202
  avg_session_seconds: S.Number,
183
203
  avg_tokens_per_session: S.Number,
184
204
  avg_turns_per_session: S.Number,
185
- baseline_spend: S.Number,
205
+ baseline_spend: S.NullOr(S.Number),
186
206
  cache: AutoRouterCacheStats,
187
- saved_pct: S.Number,
188
- saved_per_session: S.Number,
189
- saved_spend: S.Number,
207
+ classifier_cost: S.NullOr(S.Number),
208
+ saved_pct: S.NullOr(S.Number),
209
+ saved_per_session: S.NullOr(S.Number),
210
+ saved_spend: S.NullOr(S.Number),
211
+ savings_estimated_actual_spend: S.Number,
212
+ savings_estimated_turns: S.Number,
190
213
  sessions: S.Number,
191
214
  spend: S.Number,
192
215
  turns: S.Number,
@@ -219,6 +242,77 @@ export const AutoRouterBenchmarksResponse = /*@__PURE__*/ S.suspend(() =>
219
242
  identifier: "AutoRouterBenchmarksResponse",
220
243
  }) as any as S.Schema<AutoRouterBenchmarksResponse>;
221
244
 
245
+ export interface GetAutoRouterSessionAutoRouterSessionGetRequest {
246
+ /** The client session id (x-*-session-id header) the turns were sent under */
247
+ session_id: string;
248
+ }
249
+ export const GetAutoRouterSessionAutoRouterSessionGetRequest =
250
+ /*@__PURE__*/ S.suspend(() =>
251
+ S.Struct({
252
+ session_id: S.String.pipe(T.Query()),
253
+ }).pipe(T.Http({ method: "GET", uri: "/auto_router/session", code: 200 })),
254
+ ).annotate({
255
+ identifier: "GetAutoRouterSessionAutoRouterSessionGetRequest",
256
+ }) as any as S.Schema<GetAutoRouterSessionAutoRouterSessionGetRequest>;
257
+
258
+ /** Covered turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both */
259
+ export type AutoRouterSessionResponseBaselineModelsMap = {
260
+ [key: string]: number | undefined;
261
+ };
262
+ export const AutoRouterSessionResponseBaselineModelsMap =
263
+ /*@__PURE__*/ S.Record(
264
+ S.String,
265
+ S.Number,
266
+ ) as any as S.Schema<AutoRouterSessionResponseBaselineModelsMap>;
267
+
268
+ /** One auto-routed session as its own key sees it: what the last turn ran on, and what the session cost against the router's savings baseline (the priciest model in its hardest tier). */
269
+ export interface AutoRouterSessionResponse {
270
+ /** The savings baseline most covered turns were priced against, recorded turn by turn, so it still names the counterfactual after the router is reconfigured or removed. None when no turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, which derive no baseline and so report no savings */
271
+ baseline_model: string | null;
272
+ /** Covered turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both */
273
+ baseline_models: AutoRouterSessionResponseBaselineModelsMap;
274
+ /** Estimated single-model cost; unavailable unless every turn is covered */
275
+ baseline_spend: number | null;
276
+ /** The deployment model the most recent turn was routed to */
277
+ last_model: string;
278
+ /** The auto-router alias the session's requests were sent to */
279
+ router_name: string;
280
+ /** complexity, adaptive or quality */
281
+ router_type: string;
282
+ /** Estimated savings for covered turns only, net of classifier cost */
283
+ saved_spend: number | null;
284
+ /** Actual spend, including classifier cost, for covered turns only */
285
+ savings_estimated_actual_spend: number;
286
+ /** Estimated single-model cost for covered turns only */
287
+ savings_estimated_baseline_spend: number | null;
288
+ /** Turns covered by the current savings estimator; legacy estimates are excluded */
289
+ savings_estimated_turns: number;
290
+ session_id: string;
291
+ /** What the session's routed traffic actually cost, classifier calls included */
292
+ spend: number;
293
+ /** Auto-routed turns the rollup has recorded for this session so far */
294
+ turns: number;
295
+ }
296
+ export const AutoRouterSessionResponse = /*@__PURE__*/ S.suspend(() =>
297
+ S.Struct({
298
+ baseline_model: S.NullOr(S.String),
299
+ baseline_models: AutoRouterSessionResponseBaselineModelsMap,
300
+ baseline_spend: S.NullOr(S.Number),
301
+ last_model: S.String,
302
+ router_name: S.String,
303
+ router_type: S.String,
304
+ saved_spend: S.NullOr(S.Number),
305
+ savings_estimated_actual_spend: S.Number,
306
+ savings_estimated_baseline_spend: S.NullOr(S.Number),
307
+ savings_estimated_turns: S.Number,
308
+ session_id: S.String,
309
+ spend: S.Number,
310
+ turns: S.Number,
311
+ }),
312
+ ).annotate({
313
+ identifier: "AutoRouterSessionResponse",
314
+ }) as any as S.Schema<AutoRouterSessionResponse>;
315
+
222
316
  export interface GetShadowEvalJobAutoRouterShadowEvalJobIdGetRequest {
223
317
  job_id: string;
224
318
  }
@@ -240,47 +334,13 @@ export const GetShadowEvalJobAutoRouterShadowEvalJobIdGetRequest =
240
334
  export type ShadowEvalJobResponseDirection = "forward" | "reverse";
241
335
  export const ShadowEvalJobResponseDirection = S.String;
242
336
 
243
- /** One key a job shadows, with its own budget and stop state. */
244
- export interface ShadowEvalJobKeyResponse {
245
- /** The hashed virtual key whose traffic this entry scopes */
246
- api_key_id: string;
247
- /** This key's sampled attempts so far, judged and errored alike, the same count the sampler budgets against max_turns; populated on list and detail responses. Frozen at stopped_at once the key is stamped, so in-flight attempts landing after a stop never reclassify it */
248
- attempt_count?: number | null;
249
- /** Alias of the shadowed key, resolved from the key row at read time; None when unset or deleted */
250
- key_alias?: string | null;
251
- /** Masked display name (sk-...) of the shadowed key, resolved at read time like key_alias */
252
- key_name?: string | null;
253
- /** This key's own USD budget for the eval's shadow and judge spend, independent of its siblings'; None on jobs created before spend budgets existed, which max_turns alone bounds */
254
- max_budget?: number | null;
255
- /** This key's sample-count ceiling: the whole budget for jobs created before max_budget existed, and the error-loop safety valve otherwise */
256
- max_turns: number;
257
- /** This key's recorded shadow plus judge spend in USD, the same figure the sampler budgets against max_budget; populated on list and detail responses and frozen at stopped_at exactly like attempt_count */
258
- spend?: number | null;
259
- /** When this key's slot was stamped free, whether its own budget ran out, the window closed, or an operator stopped the job; status is derived, so a spent budget reads completed even while this is still unset */
260
- stopped_at?: string | null;
261
- }
262
- export const ShadowEvalJobKeyResponse = /*@__PURE__*/ S.suspend(() =>
263
- S.Struct({
264
- api_key_id: S.String,
265
- attempt_count: S.optional(S.NullOr(S.Number)),
266
- key_alias: S.optional(S.NullOr(S.String)),
267
- key_name: S.optional(S.NullOr(S.String)),
268
- max_budget: S.optional(S.NullOr(S.Number)),
269
- max_turns: S.Number,
270
- spend: S.optional(S.NullOr(S.Number)),
271
- stopped_at: S.optional(S.NullOr(S.String)),
272
- }),
273
- ).annotate({
274
- identifier: "ShadowEvalJobKeyResponse",
275
- }) as any as S.Schema<ShadowEvalJobKeyResponse>;
276
-
277
- /** The keys whose traffic this job evaluates, and only those keys', each with its own budget */
278
- export type ShadowEvalJobResponseKeysList = Array<ShadowEvalJobKeyResponse>;
279
- export const ShadowEvalJobResponseKeysList = /*@__PURE__*/ S.Array(
280
- ShadowEvalJobKeyResponse,
281
- ) as any as S.Schema<ShadowEvalJobResponseKeysList>;
337
+ /** Model groups the sampled traffic is narrowed to; empty means every model the targets use */
338
+ export type ShadowEvalJobResponseModelsList = Array<string>;
339
+ export const ShadowEvalJobResponseModelsList = /*@__PURE__*/ S.Array(
340
+ S.String,
341
+ ) as any as S.Schema<ShadowEvalJobResponseModelsList>;
282
342
 
283
- /** Judge outcomes for one slice of a job's verdicts (a router tier, or one of the models that served the real arm). */
343
+ /** Judge outcomes for one slice of a job's verdicts: a router tier, one of the models that served the real arm, or one scoped target (embedded on that target's own entry, so slices never need re-joining to a target by id). */
284
344
  export interface ShadowEvalSlice {
285
345
  avg_judge_confidence: number;
286
346
  /** Judged turns litellm's response cache served, excluded from both spends: an adopted router would be served by the same cache, so those turns cost the same either way */
@@ -319,11 +379,11 @@ export const ShadowEvalResultByCurrentModelList = /*@__PURE__*/ S.Array(
319
379
  ShadowEvalSlice,
320
380
  ) as any as S.Schema<ShadowEvalResultByCurrentModelList>;
321
381
 
322
- /** One slice per scoped key that has judged verdicts, grouped on the raw key hash. Keys the job scopes but has not judged a turn for yet are absent rather than reported as zero */
323
- export type ShadowEvalResultByKeyList = Array<ShadowEvalSlice>;
324
- export const ShadowEvalResultByKeyList = /*@__PURE__*/ S.Array(
382
+ /** One slice per router arm, grouped on the router name. Every arm of a multi-router job is judged against the same real responses over the same sampled requests, so these slices compare routers head-to-head: like-for-like win rates and spends on identical traffic. Verdicts from before arm stamping existed count toward the job's own router */
383
+ export type ShadowEvalResultByRouterList = Array<ShadowEvalSlice>;
384
+ export const ShadowEvalResultByRouterList = /*@__PURE__*/ S.Array(
325
385
  ShadowEvalSlice,
326
- ) as any as S.Schema<ShadowEvalResultByKeyList>;
386
+ ) as any as S.Schema<ShadowEvalResultByRouterList>;
327
387
 
328
388
  export type ShadowEvalResultByTierList = Array<ShadowEvalSlice>;
329
389
  export const ShadowEvalResultByTierList = /*@__PURE__*/ S.Array(
@@ -334,16 +394,16 @@ export const ShadowEvalResultByTierList = /*@__PURE__*/ S.Array(
334
394
  export interface ShadowEvalResult {
335
395
  /** Sliced by the model that served the real arm: the keys' incumbent models in forward mode, and in reverse the models the router itself picked */
336
396
  by_current_model: ShadowEvalResultByCurrentModelList;
337
- /** One slice per scoped key that has judged verdicts, grouped on the raw key hash. Keys the job scopes but has not judged a turn for yet are absent rather than reported as zero */
338
- by_key: ShadowEvalResultByKeyList;
397
+ /** One slice per router arm, grouped on the router name. Every arm of a multi-router job is judged against the same real responses over the same sampled requests, so these slices compare routers head-to-head: like-for-like win rates and spends on identical traffic. Verdicts from before arm stamping existed count toward the job's own router */
398
+ by_router?: ShadowEvalResultByRouterList;
339
399
  by_tier: ShadowEvalResultByTierList;
340
400
  /** Eligible requests the sampling dice skipped, summed over legs: the judged rows stand for judged + this many requests. None for jobs from before the funnel existed */
341
401
  not_sampled_count?: number | null;
342
402
  overall_shadow_win_rate_pct: number;
343
403
  overall_tie_rate_pct: number;
344
- /** USD the real arm billed across all judged turns, cache-served turns excluded */
404
+ /** USD the real arm billed across all judged turns, cache-served turns excluded. A judged turn is one (request, router arm) verdict, so a multi-router job counts the real response once per arm it was judged against; per-router comparisons read by_router */
345
405
  sampled_real_spend?: number;
346
- /** USD the shadow arm billed across the same turns, judge excluded, like for like */
406
+ /** USD the shadow arms billed across the same turns, judge excluded, like for like */
347
407
  sampled_shadow_spend?: number;
348
408
  /** Sampled requests dropped by the per-pod concurrency cap, so quiet periods are overweighted */
349
409
  shed_count?: number | null;
@@ -355,7 +415,7 @@ export interface ShadowEvalResult {
355
415
  export const ShadowEvalResult = /*@__PURE__*/ S.suspend(() =>
356
416
  S.Struct({
357
417
  by_current_model: ShadowEvalResultByCurrentModelList,
358
- by_key: ShadowEvalResultByKeyList,
418
+ by_router: S.optional(ShadowEvalResultByRouterList),
359
419
  by_tier: ShadowEvalResultByTierList,
360
420
  not_sampled_count: S.optional(S.NullOr(S.Number)),
361
421
  overall_shadow_win_rate_pct: S.Number,
@@ -370,11 +430,68 @@ export const ShadowEvalResult = /*@__PURE__*/ S.suspend(() =>
370
430
  identifier: "ShadowEvalResult",
371
431
  }) as any as S.Schema<ShadowEvalResult>;
372
432
 
373
- /** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every key spending its budget, and anything else is running. The all-keys-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
433
+ /** Every auto-router this job runs as a shadow arm. Multi-router jobs sample one slice of traffic and judge every arm against the same real responses */
434
+ export type ShadowEvalJobResponseRouterNamesList = Array<string>;
435
+ export const ShadowEvalJobResponseRouterNamesList = /*@__PURE__*/ S.Array(
436
+ S.String,
437
+ ) as any as S.Schema<ShadowEvalJobResponseRouterNamesList>;
438
+
439
+ /** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every target spending its budget, and anything else is running. The all-targets-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
374
440
  export type ShadowEvalJobResponseStatus = "running" | "completed" | "stopped";
375
441
  export const ShadowEvalJobResponseStatus = S.String;
376
442
 
377
- /** A shadow-eval job over one or more keys, each with its own budget and stop state; status is derived from stopped_by, the keys' stop and budget state, and ends_at, never stored, so no writer anywhere can produce an inconsistent one. Aggregate fields are populated by the detail endpoint only and stay None on list responses. */
443
+ /** What kind of entity this entry scopes */
444
+ export type ShadowEvalJobTargetResponseTargetType = "key" | "team" | "user";
445
+ export const ShadowEvalJobTargetResponseTargetType = S.String;
446
+
447
+ /** One target a job shadows (a key, team, or user), with its own budget and stop state. */
448
+ export interface ShadowEvalJobTargetResponse {
449
+ /** This target's sampled attempts so far, judged and errored alike, the same count the sampler budgets against max_turns; populated on list and detail responses. Frozen at stopped_at once the target is stamped, so in-flight attempts landing after a stop never reclassify it */
450
+ attempt_count?: number | null;
451
+ /** Masked display name (sk-...) for key targets, resolved at read time; None for teams and users */
452
+ key_name?: string | null;
453
+ /** This target's own USD budget for the eval's shadow and judge spend, independent of its siblings'; None on jobs created before spend budgets existed, which max_turns alone bounds */
454
+ max_budget?: number | null;
455
+ /** This target's sample-count ceiling: the whole budget for jobs created before max_budget existed, and the error-loop safety valve otherwise */
456
+ max_turns: number;
457
+ /** This target's recorded shadow plus judge spend in USD, the same figure the sampler budgets against max_budget; populated on list and detail responses and frozen at stopped_at exactly like attempt_count */
458
+ spend?: number | null;
459
+ /** When this target's slot was stamped free, whether its own budget ran out, the window closed, or an operator stopped the job; status is derived, so a spent budget reads completed even while this is still unset */
460
+ stopped_at?: string | null;
461
+ /** Display label resolved from the target's own row at read time: the key's alias, the team's alias, or the user's email; None when unset or deleted */
462
+ target_alias?: string | null;
463
+ /** The hashed virtual key, team id, or user id whose traffic this entry scopes */
464
+ target_id: string;
465
+ /** What kind of entity this entry scopes */
466
+ target_type: ShadowEvalJobTargetResponseTargetType;
467
+ /** This target's own judged-verdict slice; detail endpoint only, None until a turn is judged */
468
+ verdicts?: ShadowEvalSlice | null;
469
+ }
470
+ export const ShadowEvalJobTargetResponse = /*@__PURE__*/ S.suspend(() =>
471
+ S.Struct({
472
+ attempt_count: S.optional(S.NullOr(S.Number)),
473
+ key_name: S.optional(S.NullOr(S.String)),
474
+ max_budget: S.optional(S.NullOr(S.Number)),
475
+ max_turns: S.Number,
476
+ spend: S.optional(S.NullOr(S.Number)),
477
+ stopped_at: S.optional(S.NullOr(S.String)),
478
+ target_alias: S.optional(S.NullOr(S.String)),
479
+ target_id: S.String,
480
+ target_type: ShadowEvalJobTargetResponseTargetType,
481
+ verdicts: S.optional(S.NullOr(ShadowEvalSlice)),
482
+ }),
483
+ ).annotate({
484
+ identifier: "ShadowEvalJobTargetResponse",
485
+ }) as any as S.Schema<ShadowEvalJobTargetResponse>;
486
+
487
+ /** The targets whose traffic this job evaluates, and only theirs, each with its own budget */
488
+ export type ShadowEvalJobResponseTargetsList =
489
+ Array<ShadowEvalJobTargetResponse>;
490
+ export const ShadowEvalJobResponseTargetsList = /*@__PURE__*/ S.Array(
491
+ ShadowEvalJobTargetResponse,
492
+ ) as any as S.Schema<ShadowEvalJobResponseTargetsList>;
493
+
494
+ /** A shadow-eval job over one or more targets, each with its own budget and stop state; status is derived from stopped_by, the targets' stop and budget state, and ends_at, never stored, so no writer anywhere can produce an inconsistent one. Aggregate fields are populated by the detail endpoint only and stay None on list responses. */
378
495
  export interface ShadowEvalJobResponse {
379
496
  baseline_model?: string | null;
380
497
  created_at: string;
@@ -388,18 +505,23 @@ export interface ShadowEvalJobResponse {
388
505
  judge_spend?: number | null;
389
506
  /** Verdicts recorded; detail endpoint only */
390
507
  judged_count?: number | null;
391
- /** The keys whose traffic this job evaluates, and only those keys', each with its own budget */
392
- keys: ShadowEvalJobResponseKeysList;
393
508
  /** Most recent attempt error; detail endpoint only */
394
509
  last_error?: string | null;
510
+ /** Model groups the sampled traffic is narrowed to; empty means every model the targets use */
511
+ models?: ShadowEvalJobResponseModelsList;
395
512
  /** Stratified verdicts; detail endpoint only */
396
513
  results?: ShadowEvalResult | null;
514
+ /** The first router, kept for callers that predate router_names; derived so the two fields can never disagree. */
397
515
  router_name: string;
516
+ /** Every auto-router this job runs as a shadow arm. Multi-router jobs sample one slice of traffic and judge every arm against the same real responses */
517
+ router_names: ShadowEvalJobResponseRouterNamesList;
398
518
  shadow_percentage: number;
399
- /** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every key spending its budget, and anything else is running. The all-keys-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
519
+ /** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every target spending its budget, and anything else is running. The all-targets-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
400
520
  status: ShadowEvalJobResponseStatus;
401
521
  /** The operator who stopped the job early, recorded by the stop endpoint; 'unknown' backfilled by migration for jobs that displayed stopped when the column arrived; None when the job ended on its own. Its presence is what makes a job read stopped rather than completed */
402
522
  stopped_by?: string | null;
523
+ /** The targets whose traffic this job evaluates, and only theirs, each with its own budget */
524
+ targets: ShadowEvalJobResponseTargetsList;
403
525
  }
404
526
  export const ShadowEvalJobResponse = /*@__PURE__*/ S.suspend(() =>
405
527
  S.Struct({
@@ -412,28 +534,46 @@ export const ShadowEvalJobResponse = /*@__PURE__*/ S.suspend(() =>
412
534
  judge_model: S.String,
413
535
  judge_spend: S.optional(S.NullOr(S.Number)),
414
536
  judged_count: S.optional(S.NullOr(S.Number)),
415
- keys: ShadowEvalJobResponseKeysList,
416
537
  last_error: S.optional(S.NullOr(S.String)),
538
+ models: S.optional(ShadowEvalJobResponseModelsList),
417
539
  results: S.optional(S.NullOr(ShadowEvalResult)),
418
540
  router_name: S.String,
541
+ router_names: ShadowEvalJobResponseRouterNamesList,
419
542
  shadow_percentage: S.Number,
420
543
  status: ShadowEvalJobResponseStatus,
421
544
  stopped_by: S.optional(S.NullOr(S.String)),
545
+ targets: ShadowEvalJobResponseTargetsList,
422
546
  }),
423
547
  ).annotate({
424
548
  identifier: "ShadowEvalJobResponse",
425
549
  }) as any as S.Schema<ShadowEvalJobResponse>;
426
550
 
551
+ export type ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType =
552
+ | "key"
553
+ | "team"
554
+ | "user";
555
+ export const ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType =
556
+ S.String;
557
+
427
558
  export interface ListShadowEvalJobsAutoRouterShadowEvalGetRequest {
428
- /** Filter to jobs that shadow this key, alone or alongside others */
429
- api_key_id?: string;
559
+ /** Kind of target to filter on; requires target_id */
560
+ target_type?:
561
+ | ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType
562
+ | (string & {});
563
+ /** Filter to jobs that shadow this target, alone or alongside others */
564
+ target_id?: string;
430
565
  /** Newest jobs to return */
431
566
  limit?: number;
432
567
  }
433
568
  export const ListShadowEvalJobsAutoRouterShadowEvalGetRequest =
434
569
  /*@__PURE__*/ S.suspend(() =>
435
570
  S.Struct({
436
- api_key_id: S.optional(S.String.pipe(T.Query())),
571
+ target_type: S.optional(
572
+ ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType.pipe(
573
+ T.Query(),
574
+ ),
575
+ ),
576
+ target_id: S.optional(S.String.pipe(T.Query())),
437
577
  limit: S.optional(S.Number.pipe(T.Query())),
438
578
  }).pipe(
439
579
  T.Http({ method: "GET", uri: "/auto_router/shadow_eval", code: 200 }),
@@ -461,7 +601,7 @@ export const ListShadowEvalJobsAutoRouterShadowEvalGetResponse =
461
601
  identifier: "ListShadowEvalJobsAutoRouterShadowEvalGetResponse",
462
602
  }) as any as S.Schema<ListShadowEvalJobsAutoRouterShadowEvalGetResponse>;
463
603
 
464
- /** The hashed virtual keys whose traffic will be shadowed. Shadow evaluation runs ONLY on these keys' traffic; requests made with any other key are not sampled. Each key carries its own max_budget spend budget, so one key exhausting its budget leaves the others sampling. At most 100 keys per job, which also bounds every read the job's endpoints make. */
604
+ /** Hashed virtual keys whose traffic will be shadowed. Combined with team_ids and user_ids the job needs at least one target and at most 100, which also bounds every read the job's endpoints make. Each target carries its own max_budget spend budget, so one exhausting its budget leaves the others sampling. */
465
605
  export type StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList =
466
606
  Array<string>;
467
607
  export const StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList =
@@ -476,9 +616,41 @@ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection =
476
616
  export const StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection =
477
617
  S.String;
478
618
 
619
+ /** Model groups to narrow the sampled traffic to, matched on the group the caller requested and resolved through model_group_alias, so an alias and its target are one name. Empty samples every model the targets use. This ANDs with the targets: a job over a user and one model samples that user's requests on that model across every key they own, and none of their other traffic. Forward jobs only: a reverse job samples exactly the traffic its own router served, which no other model group can name */
620
+ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList =
621
+ Array<string>;
622
+ export const StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList =
623
+ /*@__PURE__*/ S.Array(
624
+ S.String,
625
+ ) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList>;
626
+
627
+ /** The auto-routers under evaluation, at most 4. Every sampled request runs through every router listed and each arm is judged independently against the same real response, so routers compare head-to-head on identical traffic. More than one router requires direction 'forward'. After validation this field always carries the full deduplicated set, whichever spelling the caller used */
628
+ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList =
629
+ Array<string>;
630
+ export const StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList =
631
+ /*@__PURE__*/ S.Array(
632
+ S.String,
633
+ ) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList>;
634
+
635
+ /** Teams whose traffic will be shadowed, matched on the team every authenticated request resolves to, so a team's JWT-auth and virtual-key traffic are both sampled */
636
+ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList =
637
+ Array<string>;
638
+ export const StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList =
639
+ /*@__PURE__*/ S.Array(
640
+ S.String,
641
+ ) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList>;
642
+
643
+ /** Users whose traffic will be shadowed, matched on the user every authenticated request resolves to across all their teams: JWT requests carrying their subject claim and virtual keys they own */
644
+ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList =
645
+ Array<string>;
646
+ export const StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList =
647
+ /*@__PURE__*/ S.Array(
648
+ S.String,
649
+ ) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList>;
650
+
479
651
  export interface StartShadowEvalAutoRouterShadowEvalStartPostRequest {
480
- /** The hashed virtual keys whose traffic will be shadowed. Shadow evaluation runs ONLY on these keys' traffic; requests made with any other key are not sampled. Each key carries its own max_budget spend budget, so one key exhausting its budget leaves the others sampling. At most 100 keys per job, which also bounds every read the job's endpoints make. */
481
- api_key_ids: StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList;
652
+ /** Hashed virtual keys whose traffic will be shadowed. Combined with team_ids and user_ids the job needs at least one target and at most 100, which also bounds every read the job's endpoints make. Each target carries its own max_budget spend budget, so one exhausting its budget leaves the others sampling. */
653
+ api_key_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList;
482
654
  /** Required when direction is reverse and rejected otherwise: the fixed model the router's own responses are judged against. Must be a plain model rather than another auto-router */
483
655
  baseline_model?: string | null;
484
656
  /** forward answers 'should this key adopt router_name': it samples the requests the key did NOT route through the router and duplicates them through it. reverse answers 'is the router still worth it for a key already on it': it samples the requests the router did serve and duplicates them against baseline_model. The response the caller received is always the real arm */
@@ -489,18 +661,27 @@ export interface StartShadowEvalAutoRouterShadowEvalStartPostRequest {
489
661
  duration_days?: number;
490
662
  /** Model used to blindly judge real vs. shadow responses. The judge only compares two answers, so a mid-tier model (Claude Sonnet or GPT-4o class) is the sweet spot: small/nano-class models produce unreliable or malformed verdicts, while frontier reasoning models add cost without changing outcomes. */
491
663
  judge_model?: string;
492
- /** Per-key USD budget for the eval's own overhead, the shadow-arm and judge calls, priced with the same figures the spend pipeline bills. EACH scoped key samples until its recorded eval spend reaches this, so a job over N keys spends at most about N times max_budget; in-flight samples can overshoot the cap by one sampling cache window */
664
+ /** Per-target USD budget for the eval's own overhead, the shadow-arm and judge calls, priced with the same figures the spend pipeline bills. EACH scoped target samples until its recorded eval spend reaches this, so a job over N targets spends at most about N times max_budget; in-flight samples can overshoot the cap by one sampling cache window. Every router arm draws from the same per-target budget, so a multi-router job reaches it proportionally sooner */
493
665
  max_budget?: number;
494
- /** The auto-router under evaluation, in either direction */
495
- router_name: string;
496
- /** Percentage of the key's requests to duplicate through the router */
666
+ /** Model groups to narrow the sampled traffic to, matched on the group the caller requested and resolved through model_group_alias, so an alias and its target are one name. Empty samples every model the targets use. This ANDs with the targets: a job over a user and one model samples that user's requests on that model across every key they own, and none of their other traffic. Forward jobs only: a reverse job samples exactly the traffic its own router served, which no other model group can name */
667
+ models?: StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList;
668
+ /** The auto-router under evaluation, in either direction: the single-router spelling of router_names. Provide exactly one of the two fields */
669
+ router_name?: string | null;
670
+ /** The auto-routers under evaluation, at most 4. Every sampled request runs through every router listed and each arm is judged independently against the same real response, so routers compare head-to-head on identical traffic. More than one router requires direction 'forward'. After validation this field always carries the full deduplicated set, whichever spelling the caller used */
671
+ router_names?: StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList;
672
+ /** Percentage of each target's requests to duplicate through the router */
497
673
  shadow_percentage: number;
674
+ /** Teams whose traffic will be shadowed, matched on the team every authenticated request resolves to, so a team's JWT-auth and virtual-key traffic are both sampled */
675
+ team_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList;
676
+ /** Users whose traffic will be shadowed, matched on the user every authenticated request resolves to across all their teams: JWT requests carrying their subject claim and virtual keys they own */
677
+ user_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList;
498
678
  }
499
679
  export const StartShadowEvalAutoRouterShadowEvalStartPostRequest =
500
680
  /*@__PURE__*/ S.suspend(() =>
501
681
  S.Struct({
502
- api_key_ids:
682
+ api_key_ids: S.optional(
503
683
  StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList,
684
+ ),
504
685
  baseline_model: S.optional(S.NullOr(S.String)),
505
686
  direction: S.optional(
506
687
  StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection,
@@ -508,8 +689,20 @@ export const StartShadowEvalAutoRouterShadowEvalStartPostRequest =
508
689
  duration_days: S.optional(S.Number),
509
690
  judge_model: S.optional(S.String),
510
691
  max_budget: S.optional(S.Number),
511
- router_name: S.String,
692
+ models: S.optional(
693
+ StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList,
694
+ ),
695
+ router_name: S.optional(S.NullOr(S.String)),
696
+ router_names: S.optional(
697
+ StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList,
698
+ ),
512
699
  shadow_percentage: S.Number,
700
+ team_ids: S.optional(
701
+ StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList,
702
+ ),
703
+ user_ids: S.optional(
704
+ StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList,
705
+ ),
513
706
  }).pipe(
514
707
  T.Http({
515
708
  method: "POST",
@@ -556,6 +749,23 @@ export const getAutoRouterBenchmarksAutoRouterBenchmarksGet: API.OperationMethod
556
749
  retry: Retry.Retry,
557
750
  }));
558
751
 
752
+ export type GetAutoRouterSessionAutoRouterSessionGetError =
753
+ | UnprocessableEntity
754
+ | LitellmOpError;
755
+ /** Get Auto Router Session One auto-routed session, for the key that ran it: the model its last turn was routed to and the session's spend against the router's savings baseline. Built for a coding agent's status line or stop hook, so any virtual key may call it and only ever sees rows written under its own key hash. Reads the LiteLLM_AutoRouterSession rollup, which the asynchronous spend flush fills a moment after each turn; a session with no flushed auto-routed turn yet is a 404. The id is bounded the way the writer bounded it, so an oversized client id still finds its row. */
756
+ export const getAutoRouterSessionAutoRouterSessionGet: API.OperationMethod<
757
+ GetAutoRouterSessionAutoRouterSessionGetRequest,
758
+ AutoRouterSessionResponse,
759
+ GetAutoRouterSessionAutoRouterSessionGetError,
760
+ LitellmOpContext
761
+ > = /*@__PURE__*/ API.make(() => ({
762
+ input: GetAutoRouterSessionAutoRouterSessionGetRequest,
763
+ output: AutoRouterSessionResponse,
764
+ errors: [UnprocessableEntity],
765
+ protocol: LitellmProtocol,
766
+ retry: Retry.Retry,
767
+ }));
768
+
559
769
  export type GetShadowEvalJobAutoRouterShadowEvalJobIdGetError =
560
770
  | UnprocessableEntity
561
771
  | LitellmOpError;
@@ -576,7 +786,7 @@ export const getShadowEvalJobAutoRouterShadowEvalJobIdGet: API.OperationMethod<
576
786
  export type ListShadowEvalJobsAutoRouterShadowEvalGetError =
577
787
  | UnprocessableEntity
578
788
  | LitellmOpError;
579
- /** List Shadow Eval Jobs List shadow eval jobs, newest first, each key with its attempt count so status is accurate. Judged counts, spend, and results ride the detail endpoint only. */
789
+ /** List Shadow Eval Jobs List shadow eval jobs, newest first, each target with its attempt count so status is accurate. Judged counts, spend, and results ride the detail endpoint only. */
580
790
  export const listShadowEvalJobsAutoRouterShadowEvalGet: API.OperationMethod<
581
791
  ListShadowEvalJobsAutoRouterShadowEvalGetRequest,
582
792
  ListShadowEvalJobsAutoRouterShadowEvalGetResponse,
@@ -593,7 +803,7 @@ export const listShadowEvalJobsAutoRouterShadowEvalGet: API.OperationMethod<
593
803
  export type StartShadowEvalAutoRouterShadowEvalStartPostError =
594
804
  | UnprocessableEntity
595
805
  | LitellmOpError;
596
- /** Start Shadow Eval Start a shadow eval: duplicate a sampled slice of one or more keys' live traffic against a second arm, judge the two responses blind, and stratify win rates by tier, by the model that served the real arm, and by key. A forward job answers whether the keys should adopt router_name: it samples the requests the router did not serve and duplicates them through it. A reverse job answers whether a key already on the router still gains from it: it samples the requests the router did serve and duplicates them against baseline_model. A key can hold one active job per direction, so both questions can run at once. Shadow responses are never served to users. Each key samples until its recorded eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the job's window ends, or the job is stopped, so one key running out of budget does not end sampling for the others; sampling changes propagate to pods within about 10 seconds. Shadow and judge calls bill to the shadowed key but are excluded from request counts and auto-router adoption metrics. */
806
+ /** Start Shadow Eval Start a shadow eval: duplicate a sampled slice of one or more targets' live traffic against a second arm, judge the two responses blind, and stratify win rates by tier, by the model that served the real arm, and by target. A target is a virtual key, a team, or a user. Team and user targets match on the identity every request resolves to at auth time, so they cover JWT-authenticated traffic, which presents no virtual key; a user target samples that user's traffic across all their teams, whether it arrives on a JWT or a key they own. models narrows every target to requests for those model groups, so a user plus one model samples that user's traffic on that model across every key they own; it is forward-only, since a reverse job already samples exactly the traffic its own router served. A forward job answers whether the targets should adopt router_name: it samples the requests the router did not serve and duplicates them through it. A reverse job answers whether a target already on the router still gains from it: it samples the requests the router did serve and duplicates them against baseline_model. A target can hold one active job per direction, so both questions can run at once, and a request matching several jobs' targets (say its key and its team) is sampled by each, separately budgeted. Shadow responses are never served to users. Each target samples until its recorded eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the job's window ends, or the job is stopped, so one target running out of budget does not end sampling for the others; sampling changes propagate to pods within about 10 seconds. Shadow and judge calls bill to the sampled request's own identity but are excluded from request counts and auto-router adoption metrics. */
597
807
  export const startShadowEvalAutoRouterShadowEvalStartPost: API.OperationMethod<
598
808
  StartShadowEvalAutoRouterShadowEvalStartPostRequest,
599
809
  ShadowEvalJobResponse,
@@ -610,7 +820,7 @@ export const startShadowEvalAutoRouterShadowEvalStartPost: API.OperationMethod<
610
820
  export type StopShadowEvalJobAutoRouterShadowEvalJobIdStopPostError =
611
821
  | UnprocessableEntity
612
822
  | LitellmOpError;
613
- /** Stop Shadow Eval Job Stop an active shadow eval job, every key it scopes at once. Attempts are kept; sampling halts within ~10s. Keys that already stopped on their own budget keep the stopped_at they earned. The statement is the whole state machine: it claims the job only while a leg still samples inside the window with no stop recorded, so a racing operator, a same-instant budget spend, and a repeat stop all read the same 400 with the status the job actually holds. */
823
+ /** Stop Shadow Eval Job Stop an active shadow eval job, every target it scopes at once. Attempts are kept; sampling halts within ~10s. Targets that already stopped on their own budget keep the stopped_at they earned. The statement is the whole state machine: it claims the job only while a leg still samples inside the window with no stop recorded, so a racing operator, a same-instant budget spend, and a repeat stop all read the same 400 with the status the job actually holds. */
614
824
  export const stopShadowEvalJobAutoRouterShadowEvalJobIdStopPost: API.OperationMethod<
615
825
  StopShadowEvalJobAutoRouterShadowEvalJobIdStopPostRequest,
616
826
  ShadowEvalJobResponse,
@@ -1,4 +1,4 @@
1
- // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.100.0 — see README). Do not edit.
1
+ // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
2
2
  import * as S from "@distilled.cloud/core/schema";
3
3
  import * as API from "@distilled.cloud/core/api";
4
4
  import * as C from "@distilled.cloud/core/category";
@@ -1,4 +1,4 @@
1
- // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.100.0 — see README). Do not edit.
1
+ // AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
2
2
  import * as S from "@distilled.cloud/core/schema";
3
3
  import * as API from "@distilled.cloud/core/api";
4
4
  import * as T from "../traits.ts";