@homeflare/distilled-litellm 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -12
- package/dist/services/a2a.js +1 -1
- package/dist/services/a2a_registration.js +1 -1
- package/dist/services/access_groups.d.ts +18 -0
- package/dist/services/access_groups.d.ts.map +1 -1
- package/dist/services/access_groups.js +15 -1
- package/dist/services/access_groups.js.map +1 -1
- package/dist/services/adaptive_router.js +1 -1
- package/dist/services/agents.d.ts +10 -0
- package/dist/services/agents.d.ts.map +1 -1
- package/dist/services/agents.js +10 -1
- package/dist/services/agents.js.map +1 -1
- package/dist/services/alerting.js +1 -1
- package/dist/services/anthropic_passthrough.js +1 -1
- package/dist/services/anthropic_skills.d.ts +7 -1
- package/dist/services/anthropic_skills.d.ts.map +1 -1
- package/dist/services/anthropic_skills.js +6 -2
- package/dist/services/anthropic_skills.js.map +1 -1
- package/dist/services/assistants.js +1 -1
- package/dist/services/audio.js +1 -1
- package/dist/services/audit_logging.d.ts +2 -0
- package/dist/services/audit_logging.d.ts.map +1 -1
- package/dist/services/audit_logging.js +2 -1
- package/dist/services/audit_logging.js.map +1 -1
- package/dist/services/auto_router.d.ts +162 -62
- package/dist/services/auto_router.d.ts.map +1 -1
- package/dist/services/auto_router.js +92 -31
- package/dist/services/auto_router.js.map +1 -1
- package/dist/services/batch.js +1 -1
- package/dist/services/beta_agents.js +1 -1
- package/dist/services/beta_mcp.js +1 -1
- package/dist/services/budget_management.d.ts +8 -3
- package/dist/services/budget_management.d.ts.map +1 -1
- package/dist/services/budget_management.js +7 -4
- package/dist/services/budget_management.js.map +1 -1
- package/dist/services/budget_spend_tracking.d.ts +24 -5
- package/dist/services/budget_spend_tracking.d.ts.map +1 -1
- package/dist/services/budget_spend_tracking.js +17 -4
- package/dist/services/budget_spend_tracking.js.map +1 -1
- package/dist/services/cache_settings.js +1 -1
- package/dist/services/caching.js +1 -1
- package/dist/services/chat_completions.js +1 -1
- package/dist/services/claude_code_marketplace.d.ts +11 -10
- package/dist/services/claude_code_marketplace.d.ts.map +1 -1
- package/dist/services/claude_code_marketplace.js +8 -6
- package/dist/services/claude_code_marketplace.js.map +1 -1
- package/dist/services/cloudzero.js +1 -1
- package/dist/services/completions.js +1 -1
- package/dist/services/compliance.js +1 -1
- package/dist/services/config_overrides.d.ts +5 -1
- package/dist/services/config_overrides.d.ts.map +1 -1
- package/dist/services/config_overrides.js +3 -1
- package/dist/services/config_overrides.js.map +1 -1
- package/dist/services/config_yaml.d.ts +120 -6
- package/dist/services/config_yaml.d.ts.map +1 -1
- package/dist/services/config_yaml.js +67 -1
- package/dist/services/config_yaml.js.map +1 -1
- package/dist/services/containers.d.ts +8 -2
- package/dist/services/containers.d.ts.map +1 -1
- package/dist/services/containers.js +13 -5
- package/dist/services/containers.js.map +1 -1
- package/dist/services/coordination_redis_settings.js +1 -1
- package/dist/services/cost_tracking.d.ts +91 -1
- package/dist/services/cost_tracking.d.ts.map +1 -1
- package/dist/services/cost_tracking.js +82 -2
- package/dist/services/cost_tracking.js.map +1 -1
- package/dist/services/credential_management.d.ts +2 -1
- package/dist/services/credential_management.d.ts.map +1 -1
- package/dist/services/credential_management.js +3 -2
- package/dist/services/credential_management.js.map +1 -1
- package/dist/services/customer_management.d.ts +25 -2
- package/dist/services/customer_management.d.ts.map +1 -1
- package/dist/services/customer_management.js +22 -3
- package/dist/services/customer_management.js.map +1 -1
- package/dist/services/email_management.js +1 -1
- package/dist/services/embeddings.js +1 -1
- package/dist/services/evals.js +1 -1
- package/dist/services/experimental.js +1 -1
- package/dist/services/fallback_management.js +1 -1
- package/dist/services/files.js +1 -1
- package/dist/services/fine_tuning.js +1 -1
- package/dist/services/gemini_agents.js +1 -1
- package/dist/services/google_genai_endpoints.js +1 -1
- package/dist/services/guardrails.d.ts +80 -7
- package/dist/services/guardrails.d.ts.map +1 -1
- package/dist/services/guardrails.js +37 -1
- package/dist/services/guardrails.js.map +1 -1
- package/dist/services/health.d.ts +2 -2
- package/dist/services/health.d.ts.map +1 -1
- package/dist/services/health.js +2 -2
- package/dist/services/health.js.map +1 -1
- package/dist/services/images.js +1 -1
- package/dist/services/index.d.ts +1 -15
- package/dist/services/index.d.ts.map +1 -1
- package/dist/services/index.js +1 -15
- package/dist/services/index.js.map +1 -1
- package/dist/services/internal_user_management.d.ts +227 -39
- package/dist/services/internal_user_management.d.ts.map +1 -1
- package/dist/services/internal_user_management.js +194 -37
- package/dist/services/internal_user_management.js.map +1 -1
- package/dist/services/invite_links.js +1 -1
- package/dist/services/jwt_mappings.d.ts +3 -0
- package/dist/services/jwt_mappings.d.ts.map +1 -1
- package/dist/services/jwt_mappings.js +4 -1
- package/dist/services/jwt_mappings.js.map +1 -1
- package/dist/services/key_management.d.ts +93 -49
- package/dist/services/key_management.d.ts.map +1 -1
- package/dist/services/key_management.js +75 -39
- package/dist/services/key_management.js.map +1 -1
- package/dist/services/langfuse_passthrough.js +1 -1
- package/dist/services/llm_passthrough.d.ts +1199 -0
- package/dist/services/llm_passthrough.d.ts.map +1 -0
- package/dist/services/llm_passthrough.js +2150 -0
- package/dist/services/llm_passthrough.js.map +1 -0
- package/dist/services/llm_utils.js +1 -1
- package/dist/services/logging_callbacks.js +1 -1
- package/dist/services/mcp_byok_oauth.d.ts +18 -11
- package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
- package/dist/services/mcp_byok_oauth.js +27 -24
- package/dist/services/mcp_byok_oauth.js.map +1 -1
- package/dist/services/mcp_discoverable.d.ts +34 -0
- package/dist/services/mcp_discoverable.d.ts.map +1 -1
- package/dist/services/mcp_discoverable.js +51 -1
- package/dist/services/mcp_discoverable.js.map +1 -1
- package/dist/services/mcp_management.d.ts +90 -2
- package/dist/services/mcp_management.d.ts.map +1 -1
- package/dist/services/mcp_management.js +115 -3
- package/dist/services/mcp_management.js.map +1 -1
- package/dist/services/mcp_rest.d.ts +13 -0
- package/dist/services/mcp_rest.d.ts.map +1 -1
- package/dist/services/mcp_rest.js +10 -2
- package/dist/services/mcp_rest.js.map +1 -1
- package/dist/services/memory_management.d.ts +2 -0
- package/dist/services/memory_management.d.ts.map +1 -1
- package/dist/services/memory_management.js +2 -1
- package/dist/services/memory_management.js.map +1 -1
- package/dist/services/misc.d.ts +86 -1
- package/dist/services/misc.d.ts.map +1 -1
- package/dist/services/misc.js +126 -2
- package/dist/services/misc.js.map +1 -1
- package/dist/services/model_management.d.ts +310 -19
- package/dist/services/model_management.d.ts.map +1 -1
- package/dist/services/model_management.js +240 -7
- package/dist/services/model_management.js.map +1 -1
- package/dist/services/moderations.js +1 -1
- package/dist/services/ocr.js +1 -1
- package/dist/services/open_ai_pass_through.d.ts +35 -90
- package/dist/services/open_ai_pass_through.d.ts.map +1 -1
- package/dist/services/open_ai_pass_through.js +41 -139
- package/dist/services/open_ai_pass_through.js.map +1 -1
- package/dist/services/organization_management.d.ts +21 -1
- package/dist/services/organization_management.d.ts.map +1 -1
- package/dist/services/organization_management.js +20 -2
- package/dist/services/organization_management.js.map +1 -1
- package/dist/services/plugins.js +1 -1
- package/dist/services/policies.d.ts +2 -0
- package/dist/services/policies.d.ts.map +1 -1
- package/dist/services/policies.js +3 -1
- package/dist/services/policies.js.map +1 -1
- package/dist/services/policy_engine.d.ts +6 -0
- package/dist/services/policy_engine.d.ts.map +1 -1
- package/dist/services/policy_engine.js +4 -1
- package/dist/services/policy_engine.js.map +1 -1
- package/dist/services/project_management.d.ts +15 -0
- package/dist/services/project_management.d.ts.map +1 -1
- package/dist/services/project_management.js +14 -1
- package/dist/services/project_management.js.map +1 -1
- package/dist/services/prompts.js +1 -1
- package/dist/services/public.d.ts +124 -0
- package/dist/services/public.d.ts.map +1 -1
- package/dist/services/public.js +158 -1
- package/dist/services/public.js.map +1 -1
- package/dist/services/rag.js +1 -1
- package/dist/services/realtime.d.ts +4 -26
- package/dist/services/realtime.d.ts.map +1 -1
- package/dist/services/realtime.js +7 -57
- package/dist/services/realtime.js.map +1 -1
- package/dist/services/rerank.d.ts +72 -0
- package/dist/services/rerank.d.ts.map +1 -1
- package/dist/services/rerank.js +61 -4
- package/dist/services/rerank.js.map +1 -1
- package/dist/services/responses.d.ts +30 -0
- package/dist/services/responses.d.ts.map +1 -1
- package/dist/services/responses.js +59 -1
- package/dist/services/responses.js.map +1 -1
- package/dist/services/router_settings.js +1 -1
- package/dist/services/rust_control_plane.js +1 -1
- package/dist/services/scim.d.ts +41 -1
- package/dist/services/scim.d.ts.map +1 -1
- package/dist/services/scim.js +60 -2
- package/dist/services/scim.js.map +1 -1
- package/dist/services/search.js +1 -1
- package/dist/services/search_tools.js +1 -1
- package/dist/services/settings.d.ts +88 -0
- package/dist/services/settings.d.ts.map +1 -1
- package/dist/services/settings.js +113 -1
- package/dist/services/settings.js.map +1 -1
- package/dist/services/sso_settings.d.ts +1 -1
- package/dist/services/sso_settings.d.ts.map +1 -1
- package/dist/services/sso_settings.js +1 -1
- package/dist/services/sso_settings.js.map +1 -1
- package/dist/services/tag_management.d.ts +7 -0
- package/dist/services/tag_management.d.ts.map +1 -1
- package/dist/services/tag_management.js +8 -1
- package/dist/services/tag_management.js.map +1 -1
- package/dist/services/team_management.d.ts +265 -17
- package/dist/services/team_management.d.ts.map +1 -1
- package/dist/services/team_management.js +247 -12
- package/dist/services/team_management.js.map +1 -1
- package/dist/services/tools.js +1 -1
- package/dist/services/ui_settings.js +1 -1
- package/dist/services/ui_theme_settings.js +1 -1
- package/dist/services/usage_ai.js +1 -1
- package/dist/services/vantage.js +1 -1
- package/dist/services/vector_store_management.js +1 -1
- package/dist/services/vector_stores.js +1 -1
- package/dist/services/videos.js +1 -1
- package/dist/services/web_socket.d.ts +0 -18
- package/dist/services/web_socket.d.ts.map +1 -1
- package/dist/services/web_socket.js +1 -33
- package/dist/services/web_socket.js.map +1 -1
- package/dist/services/workflow_management.js +1 -1
- package/package.json +1 -1
- package/src/services/a2a.ts +1 -1
- package/src/services/a2a_registration.ts +1 -1
- package/src/services/access_groups.ts +44 -1
- package/src/services/adaptive_router.ts +1 -1
- package/src/services/agents.ts +22 -1
- package/src/services/alerting.ts +1 -1
- package/src/services/anthropic_passthrough.ts +1 -1
- package/src/services/anthropic_skills.ts +12 -2
- package/src/services/assistants.ts +1 -1
- package/src/services/audio.ts +1 -1
- package/src/services/audit_logging.ts +4 -1
- package/src/services/auto_router.ts +303 -93
- package/src/services/batch.ts +1 -1
- package/src/services/beta_agents.ts +1 -1
- package/src/services/beta_mcp.ts +1 -1
- package/src/services/budget_management.ts +12 -4
- package/src/services/budget_spend_tracking.ts +38 -6
- package/src/services/cache_settings.ts +1 -1
- package/src/services/caching.ts +1 -1
- package/src/services/chat_completions.ts +1 -1
- package/src/services/claude_code_marketplace.ts +20 -14
- package/src/services/cloudzero.ts +1 -1
- package/src/services/completions.ts +1 -1
- package/src/services/compliance.ts +1 -1
- package/src/services/config_overrides.ts +8 -2
- package/src/services/config_yaml.ts +251 -6
- package/src/services/containers.ts +29 -11
- package/src/services/coordination_redis_settings.ts +1 -1
- package/src/services/cost_tracking.ts +198 -2
- package/src/services/credential_management.ts +9 -4
- package/src/services/customer_management.ts +49 -3
- package/src/services/email_management.ts +1 -1
- package/src/services/embeddings.ts +1 -1
- package/src/services/evals.ts +1 -1
- package/src/services/experimental.ts +1 -1
- package/src/services/fallback_management.ts +1 -1
- package/src/services/files.ts +1 -1
- package/src/services/fine_tuning.ts +1 -1
- package/src/services/gemini_agents.ts +1 -1
- package/src/services/google_genai_endpoints.ts +1 -1
- package/src/services/guardrails.ts +198 -8
- package/src/services/health.ts +3 -2
- package/src/services/images.ts +1 -1
- package/src/services/index.ts +1 -15
- package/src/services/internal_user_management.ts +565 -103
- package/src/services/invite_links.ts +1 -1
- package/src/services/jwt_mappings.ts +7 -1
- package/src/services/key_management.ts +229 -126
- package/src/services/langfuse_passthrough.ts +1 -1
- package/src/services/llm_passthrough.ts +4542 -0
- package/src/services/llm_utils.ts +1 -1
- package/src/services/logging_callbacks.ts +1 -1
- package/src/services/mcp_byok_oauth.ts +65 -45
- package/src/services/mcp_discoverable.ts +109 -1
- package/src/services/mcp_management.ts +258 -3
- package/src/services/mcp_rest.ts +30 -5
- package/src/services/memory_management.ts +4 -1
- package/src/services/misc.ts +269 -2
- package/src/services/model_management.ts +666 -21
- package/src/services/moderations.ts +1 -1
- package/src/services/ocr.ts +1 -1
- package/src/services/open_ai_pass_through.ts +81 -286
- package/src/services/organization_management.ts +44 -2
- package/src/services/plugins.ts +1 -1
- package/src/services/policies.ts +5 -1
- package/src/services/policy_engine.ts +10 -1
- package/src/services/project_management.ts +33 -1
- package/src/services/prompts.ts +1 -1
- package/src/services/public.ts +356 -1
- package/src/services/rag.ts +1 -1
- package/src/services/realtime.ts +11 -111
- package/src/services/rerank.ts +187 -7
- package/src/services/responses.ts +117 -1
- package/src/services/router_settings.ts +1 -1
- package/src/services/rust_control_plane.ts +1 -1
- package/src/services/scim.ts +134 -3
- package/src/services/search.ts +1 -1
- package/src/services/search_tools.ts +1 -1
- package/src/services/settings.ts +278 -1
- package/src/services/sso_settings.ts +2 -1
- package/src/services/tag_management.ts +15 -1
- package/src/services/team_management.ts +676 -37
- package/src/services/tools.ts +1 -1
- package/src/services/ui_settings.ts +1 -1
- package/src/services/ui_theme_settings.ts +1 -1
- package/src/services/usage_ai.ts +1 -1
- package/src/services/vantage.ts +1 -1
- package/src/services/vector_store_management.ts +1 -1
- package/src/services/vector_stores.ts +1 -1
- package/src/services/videos.ts +1 -1
- package/src/services/web_socket.ts +1 -61
- package/src/services/workflow_management.ts +1 -1
- package/dist/services/anthropic_pass_through.d.ts +0 -70
- package/dist/services/anthropic_pass_through.d.ts.map +0 -1
- package/dist/services/anthropic_pass_through.js +0 -114
- package/dist/services/anthropic_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.js +0 -114
- package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_pass_through.js +0 -114
- package/dist/services/assembly_ai_pass_through.js.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
- package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
- package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
- package/dist/services/azure_ai_pass_through.d.ts +0 -70
- package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
- package/dist/services/azure_ai_pass_through.js +0 -112
- package/dist/services/azure_ai_pass_through.js.map +0 -1
- package/dist/services/azure_pass_through.d.ts +0 -70
- package/dist/services/azure_pass_through.d.ts.map +0 -1
- package/dist/services/azure_pass_through.js +0 -107
- package/dist/services/azure_pass_through.js.map +0 -1
- package/dist/services/bedrock_pass_through.d.ts +0 -70
- package/dist/services/bedrock_pass_through.d.ts.map +0 -1
- package/dist/services/bedrock_pass_through.js +0 -114
- package/dist/services/bedrock_pass_through.js.map +0 -1
- package/dist/services/cohere_pass_through.d.ts +0 -70
- package/dist/services/cohere_pass_through.d.ts.map +0 -1
- package/dist/services/cohere_pass_through.js +0 -112
- package/dist/services/cohere_pass_through.js.map +0 -1
- package/dist/services/cursor_pass_through.d.ts +0 -70
- package/dist/services/cursor_pass_through.d.ts.map +0 -1
- package/dist/services/cursor_pass_through.js +0 -112
- package/dist/services/cursor_pass_through.js.map +0 -1
- package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
- package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
- package/dist/services/google_ai_studio_pass_through.js +0 -112
- package/dist/services/google_ai_studio_pass_through.js.map +0 -1
- package/dist/services/milvus_pass_through.d.ts +0 -70
- package/dist/services/milvus_pass_through.d.ts.map +0 -1
- package/dist/services/milvus_pass_through.js +0 -112
- package/dist/services/milvus_pass_through.js.map +0 -1
- package/dist/services/mistral_pass_through.d.ts +0 -70
- package/dist/services/mistral_pass_through.d.ts.map +0 -1
- package/dist/services/mistral_pass_through.js +0 -114
- package/dist/services/mistral_pass_through.js.map +0 -1
- package/dist/services/vertex_ai_pass_through.d.ts +0 -180
- package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
- package/dist/services/vertex_ai_pass_through.js +0 -334
- package/dist/services/vertex_ai_pass_through.js.map +0 -1
- package/dist/services/vllm_pass_through.d.ts +0 -70
- package/dist/services/vllm_pass_through.d.ts.map +0 -1
- package/dist/services/vllm_pass_through.js +0 -104
- package/dist/services/vllm_pass_through.js.map +0 -1
- package/dist/services/watsonx_pass_through.d.ts +0 -70
- package/dist/services/watsonx_pass_through.d.ts.map +0 -1
- package/dist/services/watsonx_pass_through.js +0 -114
- package/dist/services/watsonx_pass_through.js.map +0 -1
- package/src/services/anthropic_pass_through.ts +0 -233
- package/src/services/assembly_ai_eu_pass_through.ts +0 -237
- package/src/services/assembly_ai_pass_through.ts +0 -237
- package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
- package/src/services/azure_ai_pass_through.ts +0 -231
- package/src/services/azure_pass_through.ts +0 -227
- package/src/services/bedrock_pass_through.ts +0 -229
- package/src/services/cohere_pass_through.ts +0 -227
- package/src/services/cursor_pass_through.ts +0 -227
- package/src/services/google_ai_studio_pass_through.ts +0 -227
- package/src/services/milvus_pass_through.ts +0 -227
- package/src/services/mistral_pass_through.ts +0 -229
- package/src/services/vertex_ai_pass_through.ts +0 -684
- package/src/services/vllm_pass_through.ts +0 -227
- package/src/services/watsonx_pass_through.ts +0 -229
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.
|
|
1
|
+
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
|
|
2
2
|
import * as S from "@distilled.cloud/core/schema";
|
|
3
3
|
import * as API from "@distilled.cloud/core/api";
|
|
4
4
|
import * as C from "@distilled.cloud/core/category";
|
|
@@ -38,6 +38,22 @@ export const LiteLLMParamsAdaptiveRouterConfigMap = /*@__PURE__*/ S.Record(
|
|
|
38
38
|
S.Unknown,
|
|
39
39
|
) as any as S.Schema<LiteLLMParamsAdaptiveRouterConfigMap>;
|
|
40
40
|
|
|
41
|
+
export interface AwsSessionTag {
|
|
42
|
+
Key: string;
|
|
43
|
+
Value: string;
|
|
44
|
+
}
|
|
45
|
+
export const AwsSessionTag = /*@__PURE__*/ S.suspend(() =>
|
|
46
|
+
S.Struct({
|
|
47
|
+
Key: S.String,
|
|
48
|
+
Value: S.String,
|
|
49
|
+
}),
|
|
50
|
+
).annotate({ identifier: "AwsSessionTag" }) as any as S.Schema<AwsSessionTag>;
|
|
51
|
+
|
|
52
|
+
export type LiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
|
|
53
|
+
export const LiteLLMParamsAwsSessionTagsList = /*@__PURE__*/ S.Array(
|
|
54
|
+
AwsSessionTag,
|
|
55
|
+
) as any as S.Schema<LiteLLMParamsAwsSessionTagsList>;
|
|
56
|
+
|
|
41
57
|
export type LiteLLMParamsBedrockTagsList = Array<unknown>;
|
|
42
58
|
export const LiteLLMParamsBedrockTagsList = /*@__PURE__*/ S.Array(
|
|
43
59
|
S.Unknown,
|
|
@@ -76,6 +92,10 @@ export const LiteLLMParamsConfigurableClientsideAuthParamsList =
|
|
|
76
92
|
LiteLLMParamsConfigurableClientsideAuthParamsItem,
|
|
77
93
|
) as any as S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsList>;
|
|
78
94
|
|
|
95
|
+
export type LiteLLMParamsDropParams = boolean | string;
|
|
96
|
+
export const LiteLLMParamsDropParams =
|
|
97
|
+
S.Unknown as any as S.Schema<LiteLLMParamsDropParams>;
|
|
98
|
+
|
|
79
99
|
export type LiteLLMParamsMilvusPartitionNamesList = Array<string>;
|
|
80
100
|
export const LiteLLMParamsMilvusPartitionNamesList = /*@__PURE__*/ S.Array(
|
|
81
101
|
S.String,
|
|
@@ -603,6 +623,7 @@ export interface LiteLLMParams {
|
|
|
603
623
|
adaptive_router_default_model?: string | null;
|
|
604
624
|
allow_client_keepalive_override?: boolean | null;
|
|
605
625
|
annotation_cost_per_page?: number | null;
|
|
626
|
+
annotation_cost_per_page_batches?: number | null;
|
|
606
627
|
api_base?: string | null;
|
|
607
628
|
api_key?: string | null;
|
|
608
629
|
api_version?: string | null;
|
|
@@ -611,6 +632,8 @@ export interface LiteLLMParams {
|
|
|
611
632
|
auto_router_default_model?: string | null;
|
|
612
633
|
auto_router_embedding_model?: string | null;
|
|
613
634
|
auto_router_max_input_chars?: number | null;
|
|
635
|
+
auto_router_model_compression?: string | null;
|
|
636
|
+
auto_router_routing_compression?: string | null;
|
|
614
637
|
aws_access_key_id?: string | null;
|
|
615
638
|
aws_batch_role_arn?: string | null;
|
|
616
639
|
aws_bedrock_project_id?: string | null;
|
|
@@ -621,10 +644,14 @@ export interface LiteLLMParams {
|
|
|
621
644
|
aws_role_name?: string | null;
|
|
622
645
|
aws_secret_access_key?: string | null;
|
|
623
646
|
aws_session_name?: string | null;
|
|
647
|
+
aws_session_tags?: LiteLLMParamsAwsSessionTagsList | null;
|
|
624
648
|
aws_session_token?: string | null;
|
|
625
649
|
aws_sts_endpoint?: string | null;
|
|
626
650
|
aws_web_identity_token?: string | null;
|
|
627
651
|
azure_ad_token?: string | null;
|
|
652
|
+
azure_password?: string | null;
|
|
653
|
+
azure_scope?: string | null;
|
|
654
|
+
azure_username?: string | null;
|
|
628
655
|
bedrock_tags?: LiteLLMParamsBedrockTagsList | null;
|
|
629
656
|
budget_duration?: string | null;
|
|
630
657
|
cache_creation_input_audio_token_cost?: number | null;
|
|
@@ -649,22 +676,27 @@ export interface LiteLLMParams {
|
|
|
649
676
|
cache_read_input_token_cost_priority?: number | null;
|
|
650
677
|
cache_read_input_token_cost_ultrafast?: number | null;
|
|
651
678
|
citation_cost_per_token?: number | null;
|
|
679
|
+
client_id?: string | null;
|
|
680
|
+
client_secret?: string | null;
|
|
652
681
|
complexity_router_config?: LiteLLMParamsComplexityRouterConfigMap | null;
|
|
653
682
|
complexity_router_default_model?: string | null;
|
|
654
683
|
configurable_clientside_auth_params?: LiteLLMParamsConfigurableClientsideAuthParamsList | null;
|
|
655
684
|
custom_llm_provider?: string | null;
|
|
656
685
|
default_api_key_rpm_limit?: number | null;
|
|
657
686
|
default_api_key_tpm_limit?: number | null;
|
|
687
|
+
drop_params?: LiteLLMParamsDropParams | null;
|
|
658
688
|
gcs_bucket_name?: string | null;
|
|
659
689
|
google_maps_grounding_cost_per_query?: number | null;
|
|
660
690
|
input_cost_per_audio_per_second?: number | null;
|
|
661
691
|
input_cost_per_audio_per_second_above_128k_tokens?: number | null;
|
|
662
692
|
input_cost_per_audio_token?: number | null;
|
|
693
|
+
input_cost_per_audio_token_batches?: number | null;
|
|
663
694
|
input_cost_per_character?: number | null;
|
|
664
695
|
input_cost_per_character_above_128k_tokens?: number | null;
|
|
665
696
|
input_cost_per_image?: number | null;
|
|
666
697
|
input_cost_per_image_above_128k_tokens?: number | null;
|
|
667
698
|
input_cost_per_image_token?: number | null;
|
|
699
|
+
input_cost_per_image_token_batches?: number | null;
|
|
668
700
|
input_cost_per_pixel?: number | null;
|
|
669
701
|
input_cost_per_query?: number | null;
|
|
670
702
|
input_cost_per_second?: number | null;
|
|
@@ -686,6 +718,7 @@ export interface LiteLLMParams {
|
|
|
686
718
|
input_cost_per_video_per_second_above_15s_interval?: number | null;
|
|
687
719
|
input_cost_per_video_per_second_above_8s_interval?: number | null;
|
|
688
720
|
input_cost_per_video_token?: number | null;
|
|
721
|
+
input_cost_per_video_token_batches?: number | null;
|
|
689
722
|
itpm?: number | null;
|
|
690
723
|
keepalive_seconds?: number | null;
|
|
691
724
|
litellm_credential_name?: string | null;
|
|
@@ -702,6 +735,7 @@ export interface LiteLLMParams {
|
|
|
702
735
|
model_info?: LiteLLMParamsModelInfoMap | null;
|
|
703
736
|
ocr_cost_per_credit?: number | null;
|
|
704
737
|
ocr_cost_per_page?: number | null;
|
|
738
|
+
ocr_cost_per_page_batches?: number | null;
|
|
705
739
|
organization?: string | null;
|
|
706
740
|
otpm?: number | null;
|
|
707
741
|
output_cost_per_audio_per_second?: number | null;
|
|
@@ -718,6 +752,7 @@ export interface LiteLLMParams {
|
|
|
718
752
|
output_cost_per_second_1080p?: number | null;
|
|
719
753
|
output_cost_per_second_480p?: number | null;
|
|
720
754
|
output_cost_per_second_4k?: number | null;
|
|
755
|
+
output_cost_per_second_720p?: number | null;
|
|
721
756
|
output_cost_per_token?: number | null;
|
|
722
757
|
output_cost_per_token_above_128k_tokens?: number | null;
|
|
723
758
|
output_cost_per_token_above_200k_tokens?: number | null;
|
|
@@ -742,12 +777,14 @@ export interface LiteLLMParams {
|
|
|
742
777
|
rpm?: number | null;
|
|
743
778
|
s3_bucket_name?: string | null;
|
|
744
779
|
s3_encryption_key_id?: string | null;
|
|
780
|
+
s3_endpoint_url?: string | null;
|
|
745
781
|
s3_output_bucket_name?: string | null;
|
|
746
782
|
s3_region_name?: string | null;
|
|
747
783
|
search_context_cost_per_query?: LiteLLMParamsSearchContextCostPerQueryMap | null;
|
|
748
784
|
stream_timeout?: LiteLLMParamsStreamTimeout | null;
|
|
749
785
|
tag_regex?: LiteLLMParamsTagRegexList | null;
|
|
750
786
|
tags?: LiteLLMParamsTagsList | null;
|
|
787
|
+
tenant_id?: string | null;
|
|
751
788
|
tiered_pricing?: LiteLLMParamsTieredPricingList | null;
|
|
752
789
|
timeout?: LiteLLMParamsTimeout | null;
|
|
753
790
|
tpm?: number | null;
|
|
@@ -776,6 +813,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
776
813
|
adaptive_router_default_model: S.optional(S.NullOr(S.String)),
|
|
777
814
|
allow_client_keepalive_override: S.optional(S.NullOr(S.Boolean)),
|
|
778
815
|
annotation_cost_per_page: S.optional(S.NullOr(S.Number)),
|
|
816
|
+
annotation_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
|
|
779
817
|
api_base: S.optional(S.NullOr(S.String)),
|
|
780
818
|
api_key: S.optional(S.NullOr(S.String)),
|
|
781
819
|
api_version: S.optional(S.NullOr(S.String)),
|
|
@@ -784,6 +822,8 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
784
822
|
auto_router_default_model: S.optional(S.NullOr(S.String)),
|
|
785
823
|
auto_router_embedding_model: S.optional(S.NullOr(S.String)),
|
|
786
824
|
auto_router_max_input_chars: S.optional(S.NullOr(S.Number)),
|
|
825
|
+
auto_router_model_compression: S.optional(S.NullOr(S.String)),
|
|
826
|
+
auto_router_routing_compression: S.optional(S.NullOr(S.String)),
|
|
787
827
|
aws_access_key_id: S.optional(S.NullOr(S.String)),
|
|
788
828
|
aws_batch_role_arn: S.optional(S.NullOr(S.String)),
|
|
789
829
|
aws_bedrock_project_id: S.optional(S.NullOr(S.String)),
|
|
@@ -794,10 +834,14 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
794
834
|
aws_role_name: S.optional(S.NullOr(S.String)),
|
|
795
835
|
aws_secret_access_key: S.optional(S.NullOr(S.String)),
|
|
796
836
|
aws_session_name: S.optional(S.NullOr(S.String)),
|
|
837
|
+
aws_session_tags: S.optional(S.NullOr(LiteLLMParamsAwsSessionTagsList)),
|
|
797
838
|
aws_session_token: S.optional(S.NullOr(S.String)),
|
|
798
839
|
aws_sts_endpoint: S.optional(S.NullOr(S.String)),
|
|
799
840
|
aws_web_identity_token: S.optional(S.NullOr(S.String)),
|
|
800
841
|
azure_ad_token: S.optional(S.NullOr(S.String)),
|
|
842
|
+
azure_password: S.optional(S.NullOr(S.String)),
|
|
843
|
+
azure_scope: S.optional(S.NullOr(S.String)),
|
|
844
|
+
azure_username: S.optional(S.NullOr(S.String)),
|
|
801
845
|
bedrock_tags: S.optional(S.NullOr(LiteLLMParamsBedrockTagsList)),
|
|
802
846
|
budget_duration: S.optional(S.NullOr(S.String)),
|
|
803
847
|
cache_creation_input_audio_token_cost: S.optional(S.NullOr(S.Number)),
|
|
@@ -842,6 +886,8 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
842
886
|
cache_read_input_token_cost_priority: S.optional(S.NullOr(S.Number)),
|
|
843
887
|
cache_read_input_token_cost_ultrafast: S.optional(S.NullOr(S.Number)),
|
|
844
888
|
citation_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
889
|
+
client_id: S.optional(S.NullOr(S.String)),
|
|
890
|
+
client_secret: S.optional(S.NullOr(S.String)),
|
|
845
891
|
complexity_router_config: S.optional(
|
|
846
892
|
S.NullOr(LiteLLMParamsComplexityRouterConfigMap),
|
|
847
893
|
),
|
|
@@ -852,6 +898,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
852
898
|
custom_llm_provider: S.optional(S.NullOr(S.String)),
|
|
853
899
|
default_api_key_rpm_limit: S.optional(S.NullOr(S.Number)),
|
|
854
900
|
default_api_key_tpm_limit: S.optional(S.NullOr(S.Number)),
|
|
901
|
+
drop_params: S.optional(S.NullOr(LiteLLMParamsDropParams)),
|
|
855
902
|
gcs_bucket_name: S.optional(S.NullOr(S.String)),
|
|
856
903
|
google_maps_grounding_cost_per_query: S.optional(S.NullOr(S.Number)),
|
|
857
904
|
input_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -859,11 +906,13 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
859
906
|
S.NullOr(S.Number),
|
|
860
907
|
),
|
|
861
908
|
input_cost_per_audio_token: S.optional(S.NullOr(S.Number)),
|
|
909
|
+
input_cost_per_audio_token_batches: S.optional(S.NullOr(S.Number)),
|
|
862
910
|
input_cost_per_character: S.optional(S.NullOr(S.Number)),
|
|
863
911
|
input_cost_per_character_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
864
912
|
input_cost_per_image: S.optional(S.NullOr(S.Number)),
|
|
865
913
|
input_cost_per_image_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
866
914
|
input_cost_per_image_token: S.optional(S.NullOr(S.Number)),
|
|
915
|
+
input_cost_per_image_token_batches: S.optional(S.NullOr(S.Number)),
|
|
867
916
|
input_cost_per_pixel: S.optional(S.NullOr(S.Number)),
|
|
868
917
|
input_cost_per_query: S.optional(S.NullOr(S.Number)),
|
|
869
918
|
input_cost_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -895,6 +944,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
895
944
|
S.NullOr(S.Number),
|
|
896
945
|
),
|
|
897
946
|
input_cost_per_video_token: S.optional(S.NullOr(S.Number)),
|
|
947
|
+
input_cost_per_video_token_batches: S.optional(S.NullOr(S.Number)),
|
|
898
948
|
itpm: S.optional(S.NullOr(S.Number)),
|
|
899
949
|
keepalive_seconds: S.optional(S.NullOr(S.Number)),
|
|
900
950
|
litellm_credential_name: S.optional(S.NullOr(S.String)),
|
|
@@ -913,6 +963,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
913
963
|
model_info: S.optional(S.NullOr(LiteLLMParamsModelInfoMap)),
|
|
914
964
|
ocr_cost_per_credit: S.optional(S.NullOr(S.Number)),
|
|
915
965
|
ocr_cost_per_page: S.optional(S.NullOr(S.Number)),
|
|
966
|
+
ocr_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
|
|
916
967
|
organization: S.optional(S.NullOr(S.String)),
|
|
917
968
|
otpm: S.optional(S.NullOr(S.Number)),
|
|
918
969
|
output_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -929,6 +980,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
929
980
|
output_cost_per_second_1080p: S.optional(S.NullOr(S.Number)),
|
|
930
981
|
output_cost_per_second_480p: S.optional(S.NullOr(S.Number)),
|
|
931
982
|
output_cost_per_second_4k: S.optional(S.NullOr(S.Number)),
|
|
983
|
+
output_cost_per_second_720p: S.optional(S.NullOr(S.Number)),
|
|
932
984
|
output_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
933
985
|
output_cost_per_token_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
934
986
|
output_cost_per_token_above_200k_tokens: S.optional(S.NullOr(S.Number)),
|
|
@@ -961,6 +1013,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
961
1013
|
rpm: S.optional(S.NullOr(S.Number)),
|
|
962
1014
|
s3_bucket_name: S.optional(S.NullOr(S.String)),
|
|
963
1015
|
s3_encryption_key_id: S.optional(S.NullOr(S.String)),
|
|
1016
|
+
s3_endpoint_url: S.optional(S.NullOr(S.String)),
|
|
964
1017
|
s3_output_bucket_name: S.optional(S.NullOr(S.String)),
|
|
965
1018
|
s3_region_name: S.optional(S.NullOr(S.String)),
|
|
966
1019
|
search_context_cost_per_query: S.optional(
|
|
@@ -969,6 +1022,7 @@ export const LiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
969
1022
|
stream_timeout: S.optional(S.NullOr(LiteLLMParamsStreamTimeout)),
|
|
970
1023
|
tag_regex: S.optional(S.NullOr(LiteLLMParamsTagRegexList)),
|
|
971
1024
|
tags: S.optional(S.NullOr(LiteLLMParamsTagsList)),
|
|
1025
|
+
tenant_id: S.optional(S.NullOr(S.String)),
|
|
972
1026
|
tiered_pricing: S.optional(S.NullOr(LiteLLMParamsTieredPricingList)),
|
|
973
1027
|
timeout: S.optional(S.NullOr(LiteLLMParamsTimeout)),
|
|
974
1028
|
tpm: S.optional(S.NullOr(S.Number)),
|
|
@@ -1023,6 +1077,8 @@ export interface LitellmTypesRouterModelInfo {
|
|
|
1023
1077
|
id: string | null;
|
|
1024
1078
|
input_cost_per_character?: number | null;
|
|
1025
1079
|
input_cost_per_token?: number | null;
|
|
1080
|
+
internal_router_model?: boolean | null;
|
|
1081
|
+
member_auto_router?: boolean;
|
|
1026
1082
|
output_cost_per_character?: number | null;
|
|
1027
1083
|
output_cost_per_token?: number | null;
|
|
1028
1084
|
ptu_count?: number | null;
|
|
@@ -1050,6 +1106,8 @@ export const LitellmTypesRouterModelInfo = /*@__PURE__*/ S.suspend(() =>
|
|
|
1050
1106
|
id: S.NullOr(S.String),
|
|
1051
1107
|
input_cost_per_character: S.optional(S.NullOr(S.Number)),
|
|
1052
1108
|
input_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
1109
|
+
internal_router_model: S.optional(S.NullOr(S.Boolean)),
|
|
1110
|
+
member_auto_router: S.optional(S.Boolean),
|
|
1053
1111
|
output_cost_per_character: S.optional(S.NullOr(S.Number)),
|
|
1054
1112
|
output_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
1055
1113
|
ptu_count: S.optional(S.NullOr(S.Number)),
|
|
@@ -1772,6 +1830,10 @@ export interface GetModelInfoV2V2ModelInfoRequest {
|
|
|
1772
1830
|
sortOrder?: string;
|
|
1773
1831
|
/** Omit auto-router deployments (litellm model prefixed `auto_router/`). They select among deployments rather than being deployments themselves, so a caller rendering a deployment list can leave them out. Defaults to false, so existing callers are unaffected */
|
|
1774
1832
|
exclude_auto_routers?: boolean;
|
|
1833
|
+
/** Only return deployments whose `model_info.access_groups` contains this access group */
|
|
1834
|
+
access_group?: string;
|
|
1835
|
+
/** Only return wildcard deployments, i.e. those whose `model_name` contains `*` */
|
|
1836
|
+
wildcard_only?: boolean;
|
|
1775
1837
|
}
|
|
1776
1838
|
export const GetModelInfoV2V2ModelInfoRequest = /*@__PURE__*/ S.suspend(() =>
|
|
1777
1839
|
S.Struct({
|
|
@@ -1787,6 +1849,8 @@ export const GetModelInfoV2V2ModelInfoRequest = /*@__PURE__*/ S.suspend(() =>
|
|
|
1787
1849
|
sortBy: S.optional(S.String.pipe(T.Query())),
|
|
1788
1850
|
sortOrder: S.optional(S.String.pipe(T.Query())),
|
|
1789
1851
|
exclude_auto_routers: S.optional(S.Boolean.pipe(T.Query())),
|
|
1852
|
+
access_group: S.optional(S.String.pipe(T.Query())),
|
|
1853
|
+
wildcard_only: S.optional(S.Boolean.pipe(T.Query())),
|
|
1790
1854
|
}).pipe(T.Http({ method: "GET", uri: "/v2/model/info", code: 200 })),
|
|
1791
1855
|
).annotate({
|
|
1792
1856
|
identifier: "GetModelInfoV2V2ModelInfoRequest",
|
|
@@ -2064,6 +2128,11 @@ export const UpdateLiteLLMParamsAdaptiveRouterConfigMap =
|
|
|
2064
2128
|
S.Unknown,
|
|
2065
2129
|
) as any as S.Schema<UpdateLiteLLMParamsAdaptiveRouterConfigMap>;
|
|
2066
2130
|
|
|
2131
|
+
export type UpdateLiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
|
|
2132
|
+
export const UpdateLiteLLMParamsAwsSessionTagsList = /*@__PURE__*/ S.Array(
|
|
2133
|
+
AwsSessionTag,
|
|
2134
|
+
) as any as S.Schema<UpdateLiteLLMParamsAwsSessionTagsList>;
|
|
2135
|
+
|
|
2067
2136
|
export type UpdateLiteLLMParamsBedrockTagsList = Array<unknown>;
|
|
2068
2137
|
export const UpdateLiteLLMParamsBedrockTagsList = /*@__PURE__*/ S.Array(
|
|
2069
2138
|
S.Unknown,
|
|
@@ -2091,6 +2160,10 @@ export const UpdateLiteLLMParamsConfigurableClientsideAuthParamsList =
|
|
|
2091
2160
|
UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem,
|
|
2092
2161
|
) as any as S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsList>;
|
|
2093
2162
|
|
|
2163
|
+
export type UpdateLiteLLMParamsDropParams = boolean | string;
|
|
2164
|
+
export const UpdateLiteLLMParamsDropParams =
|
|
2165
|
+
S.Unknown as any as S.Schema<UpdateLiteLLMParamsDropParams>;
|
|
2166
|
+
|
|
2094
2167
|
export type UpdateLiteLLMParamsMilvusPartitionNamesList = Array<string>;
|
|
2095
2168
|
export const UpdateLiteLLMParamsMilvusPartitionNamesList =
|
|
2096
2169
|
/*@__PURE__*/ S.Array(
|
|
@@ -2178,6 +2251,7 @@ export interface UpdateLiteLLMParams {
|
|
|
2178
2251
|
adaptive_router_default_model?: string | null;
|
|
2179
2252
|
allow_client_keepalive_override?: boolean | null;
|
|
2180
2253
|
annotation_cost_per_page?: number | null;
|
|
2254
|
+
annotation_cost_per_page_batches?: number | null;
|
|
2181
2255
|
api_base?: string | null;
|
|
2182
2256
|
api_key?: string | null;
|
|
2183
2257
|
api_version?: string | null;
|
|
@@ -2186,6 +2260,8 @@ export interface UpdateLiteLLMParams {
|
|
|
2186
2260
|
auto_router_default_model?: string | null;
|
|
2187
2261
|
auto_router_embedding_model?: string | null;
|
|
2188
2262
|
auto_router_max_input_chars?: number | null;
|
|
2263
|
+
auto_router_model_compression?: string | null;
|
|
2264
|
+
auto_router_routing_compression?: string | null;
|
|
2189
2265
|
aws_access_key_id?: string | null;
|
|
2190
2266
|
aws_batch_role_arn?: string | null;
|
|
2191
2267
|
aws_bedrock_project_id?: string | null;
|
|
@@ -2196,10 +2272,14 @@ export interface UpdateLiteLLMParams {
|
|
|
2196
2272
|
aws_role_name?: string | null;
|
|
2197
2273
|
aws_secret_access_key?: string | null;
|
|
2198
2274
|
aws_session_name?: string | null;
|
|
2275
|
+
aws_session_tags?: UpdateLiteLLMParamsAwsSessionTagsList | null;
|
|
2199
2276
|
aws_session_token?: string | null;
|
|
2200
2277
|
aws_sts_endpoint?: string | null;
|
|
2201
2278
|
aws_web_identity_token?: string | null;
|
|
2202
2279
|
azure_ad_token?: string | null;
|
|
2280
|
+
azure_password?: string | null;
|
|
2281
|
+
azure_scope?: string | null;
|
|
2282
|
+
azure_username?: string | null;
|
|
2203
2283
|
bedrock_tags?: UpdateLiteLLMParamsBedrockTagsList | null;
|
|
2204
2284
|
budget_duration?: string | null;
|
|
2205
2285
|
cache_creation_input_audio_token_cost?: number | null;
|
|
@@ -2224,22 +2304,27 @@ export interface UpdateLiteLLMParams {
|
|
|
2224
2304
|
cache_read_input_token_cost_priority?: number | null;
|
|
2225
2305
|
cache_read_input_token_cost_ultrafast?: number | null;
|
|
2226
2306
|
citation_cost_per_token?: number | null;
|
|
2307
|
+
client_id?: string | null;
|
|
2308
|
+
client_secret?: string | null;
|
|
2227
2309
|
complexity_router_config?: UpdateLiteLLMParamsComplexityRouterConfigMap | null;
|
|
2228
2310
|
complexity_router_default_model?: string | null;
|
|
2229
2311
|
configurable_clientside_auth_params?: UpdateLiteLLMParamsConfigurableClientsideAuthParamsList | null;
|
|
2230
2312
|
custom_llm_provider?: string | null;
|
|
2231
2313
|
default_api_key_rpm_limit?: number | null;
|
|
2232
2314
|
default_api_key_tpm_limit?: number | null;
|
|
2315
|
+
drop_params?: UpdateLiteLLMParamsDropParams | null;
|
|
2233
2316
|
gcs_bucket_name?: string | null;
|
|
2234
2317
|
google_maps_grounding_cost_per_query?: number | null;
|
|
2235
2318
|
input_cost_per_audio_per_second?: number | null;
|
|
2236
2319
|
input_cost_per_audio_per_second_above_128k_tokens?: number | null;
|
|
2237
2320
|
input_cost_per_audio_token?: number | null;
|
|
2321
|
+
input_cost_per_audio_token_batches?: number | null;
|
|
2238
2322
|
input_cost_per_character?: number | null;
|
|
2239
2323
|
input_cost_per_character_above_128k_tokens?: number | null;
|
|
2240
2324
|
input_cost_per_image?: number | null;
|
|
2241
2325
|
input_cost_per_image_above_128k_tokens?: number | null;
|
|
2242
2326
|
input_cost_per_image_token?: number | null;
|
|
2327
|
+
input_cost_per_image_token_batches?: number | null;
|
|
2243
2328
|
input_cost_per_pixel?: number | null;
|
|
2244
2329
|
input_cost_per_query?: number | null;
|
|
2245
2330
|
input_cost_per_second?: number | null;
|
|
@@ -2261,6 +2346,7 @@ export interface UpdateLiteLLMParams {
|
|
|
2261
2346
|
input_cost_per_video_per_second_above_15s_interval?: number | null;
|
|
2262
2347
|
input_cost_per_video_per_second_above_8s_interval?: number | null;
|
|
2263
2348
|
input_cost_per_video_token?: number | null;
|
|
2349
|
+
input_cost_per_video_token_batches?: number | null;
|
|
2264
2350
|
itpm?: number | null;
|
|
2265
2351
|
keepalive_seconds?: number | null;
|
|
2266
2352
|
litellm_credential_name?: string | null;
|
|
@@ -2277,6 +2363,7 @@ export interface UpdateLiteLLMParams {
|
|
|
2277
2363
|
model_info?: UpdateLiteLLMParamsModelInfoMap | null;
|
|
2278
2364
|
ocr_cost_per_credit?: number | null;
|
|
2279
2365
|
ocr_cost_per_page?: number | null;
|
|
2366
|
+
ocr_cost_per_page_batches?: number | null;
|
|
2280
2367
|
organization?: string | null;
|
|
2281
2368
|
otpm?: number | null;
|
|
2282
2369
|
output_cost_per_audio_per_second?: number | null;
|
|
@@ -2293,6 +2380,7 @@ export interface UpdateLiteLLMParams {
|
|
|
2293
2380
|
output_cost_per_second_1080p?: number | null;
|
|
2294
2381
|
output_cost_per_second_480p?: number | null;
|
|
2295
2382
|
output_cost_per_second_4k?: number | null;
|
|
2383
|
+
output_cost_per_second_720p?: number | null;
|
|
2296
2384
|
output_cost_per_token?: number | null;
|
|
2297
2385
|
output_cost_per_token_above_128k_tokens?: number | null;
|
|
2298
2386
|
output_cost_per_token_above_200k_tokens?: number | null;
|
|
@@ -2317,12 +2405,14 @@ export interface UpdateLiteLLMParams {
|
|
|
2317
2405
|
rpm?: number | null;
|
|
2318
2406
|
s3_bucket_name?: string | null;
|
|
2319
2407
|
s3_encryption_key_id?: string | null;
|
|
2408
|
+
s3_endpoint_url?: string | null;
|
|
2320
2409
|
s3_output_bucket_name?: string | null;
|
|
2321
2410
|
s3_region_name?: string | null;
|
|
2322
2411
|
search_context_cost_per_query?: UpdateLiteLLMParamsSearchContextCostPerQueryMap | null;
|
|
2323
2412
|
stream_timeout?: UpdateLiteLLMParamsStreamTimeout | null;
|
|
2324
2413
|
tag_regex?: UpdateLiteLLMParamsTagRegexList | null;
|
|
2325
2414
|
tags?: UpdateLiteLLMParamsTagsList | null;
|
|
2415
|
+
tenant_id?: string | null;
|
|
2326
2416
|
tiered_pricing?: UpdateLiteLLMParamsTieredPricingList | null;
|
|
2327
2417
|
timeout?: UpdateLiteLLMParamsTimeout | null;
|
|
2328
2418
|
tpm?: number | null;
|
|
@@ -2351,6 +2441,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2351
2441
|
adaptive_router_default_model: S.optional(S.NullOr(S.String)),
|
|
2352
2442
|
allow_client_keepalive_override: S.optional(S.NullOr(S.Boolean)),
|
|
2353
2443
|
annotation_cost_per_page: S.optional(S.NullOr(S.Number)),
|
|
2444
|
+
annotation_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
|
|
2354
2445
|
api_base: S.optional(S.NullOr(S.String)),
|
|
2355
2446
|
api_key: S.optional(S.NullOr(S.String)),
|
|
2356
2447
|
api_version: S.optional(S.NullOr(S.String)),
|
|
@@ -2359,6 +2450,8 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2359
2450
|
auto_router_default_model: S.optional(S.NullOr(S.String)),
|
|
2360
2451
|
auto_router_embedding_model: S.optional(S.NullOr(S.String)),
|
|
2361
2452
|
auto_router_max_input_chars: S.optional(S.NullOr(S.Number)),
|
|
2453
|
+
auto_router_model_compression: S.optional(S.NullOr(S.String)),
|
|
2454
|
+
auto_router_routing_compression: S.optional(S.NullOr(S.String)),
|
|
2362
2455
|
aws_access_key_id: S.optional(S.NullOr(S.String)),
|
|
2363
2456
|
aws_batch_role_arn: S.optional(S.NullOr(S.String)),
|
|
2364
2457
|
aws_bedrock_project_id: S.optional(S.NullOr(S.String)),
|
|
@@ -2369,10 +2462,16 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2369
2462
|
aws_role_name: S.optional(S.NullOr(S.String)),
|
|
2370
2463
|
aws_secret_access_key: S.optional(S.NullOr(S.String)),
|
|
2371
2464
|
aws_session_name: S.optional(S.NullOr(S.String)),
|
|
2465
|
+
aws_session_tags: S.optional(
|
|
2466
|
+
S.NullOr(UpdateLiteLLMParamsAwsSessionTagsList),
|
|
2467
|
+
),
|
|
2372
2468
|
aws_session_token: S.optional(S.NullOr(S.String)),
|
|
2373
2469
|
aws_sts_endpoint: S.optional(S.NullOr(S.String)),
|
|
2374
2470
|
aws_web_identity_token: S.optional(S.NullOr(S.String)),
|
|
2375
2471
|
azure_ad_token: S.optional(S.NullOr(S.String)),
|
|
2472
|
+
azure_password: S.optional(S.NullOr(S.String)),
|
|
2473
|
+
azure_scope: S.optional(S.NullOr(S.String)),
|
|
2474
|
+
azure_username: S.optional(S.NullOr(S.String)),
|
|
2376
2475
|
bedrock_tags: S.optional(S.NullOr(UpdateLiteLLMParamsBedrockTagsList)),
|
|
2377
2476
|
budget_duration: S.optional(S.NullOr(S.String)),
|
|
2378
2477
|
cache_creation_input_audio_token_cost: S.optional(S.NullOr(S.Number)),
|
|
@@ -2417,6 +2516,8 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2417
2516
|
cache_read_input_token_cost_priority: S.optional(S.NullOr(S.Number)),
|
|
2418
2517
|
cache_read_input_token_cost_ultrafast: S.optional(S.NullOr(S.Number)),
|
|
2419
2518
|
citation_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
2519
|
+
client_id: S.optional(S.NullOr(S.String)),
|
|
2520
|
+
client_secret: S.optional(S.NullOr(S.String)),
|
|
2420
2521
|
complexity_router_config: S.optional(
|
|
2421
2522
|
S.NullOr(UpdateLiteLLMParamsComplexityRouterConfigMap),
|
|
2422
2523
|
),
|
|
@@ -2427,6 +2528,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2427
2528
|
custom_llm_provider: S.optional(S.NullOr(S.String)),
|
|
2428
2529
|
default_api_key_rpm_limit: S.optional(S.NullOr(S.Number)),
|
|
2429
2530
|
default_api_key_tpm_limit: S.optional(S.NullOr(S.Number)),
|
|
2531
|
+
drop_params: S.optional(S.NullOr(UpdateLiteLLMParamsDropParams)),
|
|
2430
2532
|
gcs_bucket_name: S.optional(S.NullOr(S.String)),
|
|
2431
2533
|
google_maps_grounding_cost_per_query: S.optional(S.NullOr(S.Number)),
|
|
2432
2534
|
input_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -2434,11 +2536,13 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2434
2536
|
S.NullOr(S.Number),
|
|
2435
2537
|
),
|
|
2436
2538
|
input_cost_per_audio_token: S.optional(S.NullOr(S.Number)),
|
|
2539
|
+
input_cost_per_audio_token_batches: S.optional(S.NullOr(S.Number)),
|
|
2437
2540
|
input_cost_per_character: S.optional(S.NullOr(S.Number)),
|
|
2438
2541
|
input_cost_per_character_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
2439
2542
|
input_cost_per_image: S.optional(S.NullOr(S.Number)),
|
|
2440
2543
|
input_cost_per_image_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
2441
2544
|
input_cost_per_image_token: S.optional(S.NullOr(S.Number)),
|
|
2545
|
+
input_cost_per_image_token_batches: S.optional(S.NullOr(S.Number)),
|
|
2442
2546
|
input_cost_per_pixel: S.optional(S.NullOr(S.Number)),
|
|
2443
2547
|
input_cost_per_query: S.optional(S.NullOr(S.Number)),
|
|
2444
2548
|
input_cost_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -2470,6 +2574,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2470
2574
|
S.NullOr(S.Number),
|
|
2471
2575
|
),
|
|
2472
2576
|
input_cost_per_video_token: S.optional(S.NullOr(S.Number)),
|
|
2577
|
+
input_cost_per_video_token_batches: S.optional(S.NullOr(S.Number)),
|
|
2473
2578
|
itpm: S.optional(S.NullOr(S.Number)),
|
|
2474
2579
|
keepalive_seconds: S.optional(S.NullOr(S.Number)),
|
|
2475
2580
|
litellm_credential_name: S.optional(S.NullOr(S.String)),
|
|
@@ -2488,6 +2593,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2488
2593
|
model_info: S.optional(S.NullOr(UpdateLiteLLMParamsModelInfoMap)),
|
|
2489
2594
|
ocr_cost_per_credit: S.optional(S.NullOr(S.Number)),
|
|
2490
2595
|
ocr_cost_per_page: S.optional(S.NullOr(S.Number)),
|
|
2596
|
+
ocr_cost_per_page_batches: S.optional(S.NullOr(S.Number)),
|
|
2491
2597
|
organization: S.optional(S.NullOr(S.String)),
|
|
2492
2598
|
otpm: S.optional(S.NullOr(S.Number)),
|
|
2493
2599
|
output_cost_per_audio_per_second: S.optional(S.NullOr(S.Number)),
|
|
@@ -2504,6 +2610,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2504
2610
|
output_cost_per_second_1080p: S.optional(S.NullOr(S.Number)),
|
|
2505
2611
|
output_cost_per_second_480p: S.optional(S.NullOr(S.Number)),
|
|
2506
2612
|
output_cost_per_second_4k: S.optional(S.NullOr(S.Number)),
|
|
2613
|
+
output_cost_per_second_720p: S.optional(S.NullOr(S.Number)),
|
|
2507
2614
|
output_cost_per_token: S.optional(S.NullOr(S.Number)),
|
|
2508
2615
|
output_cost_per_token_above_128k_tokens: S.optional(S.NullOr(S.Number)),
|
|
2509
2616
|
output_cost_per_token_above_200k_tokens: S.optional(S.NullOr(S.Number)),
|
|
@@ -2536,6 +2643,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2536
2643
|
rpm: S.optional(S.NullOr(S.Number)),
|
|
2537
2644
|
s3_bucket_name: S.optional(S.NullOr(S.String)),
|
|
2538
2645
|
s3_encryption_key_id: S.optional(S.NullOr(S.String)),
|
|
2646
|
+
s3_endpoint_url: S.optional(S.NullOr(S.String)),
|
|
2539
2647
|
s3_output_bucket_name: S.optional(S.NullOr(S.String)),
|
|
2540
2648
|
s3_region_name: S.optional(S.NullOr(S.String)),
|
|
2541
2649
|
search_context_cost_per_query: S.optional(
|
|
@@ -2544,6 +2652,7 @@ export const UpdateLiteLLMParams = /*@__PURE__*/ S.suspend(() =>
|
|
|
2544
2652
|
stream_timeout: S.optional(S.NullOr(UpdateLiteLLMParamsStreamTimeout)),
|
|
2545
2653
|
tag_regex: S.optional(S.NullOr(UpdateLiteLLMParamsTagRegexList)),
|
|
2546
2654
|
tags: S.optional(S.NullOr(UpdateLiteLLMParamsTagsList)),
|
|
2655
|
+
tenant_id: S.optional(S.NullOr(S.String)),
|
|
2547
2656
|
tiered_pricing: S.optional(S.NullOr(UpdateLiteLLMParamsTieredPricingList)),
|
|
2548
2657
|
timeout: S.optional(S.NullOr(UpdateLiteLLMParamsTimeout)),
|
|
2549
2658
|
tpm: S.optional(S.NullOr(S.Number)),
|
|
@@ -2670,7 +2779,7 @@ export const PostBlockModelModelBlockResponse = /*@__PURE__*/ S.suspend(() =>
|
|
|
2670
2779
|
|
|
2671
2780
|
/** An operator-defined tier: the name the LLM classifier must return and its rubric description. */
|
|
2672
2781
|
export interface TierDefinition {
|
|
2673
|
-
/** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (SIMPLE
|
|
2782
|
+
/** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (NON_REASONING, SIMPLE, MEDIUM, COMPLEX, REASONING), which inherits the built-in criteria when omitted */
|
|
2674
2783
|
description?: string | null;
|
|
2675
2784
|
/** Tier name; becomes a value the LLM classifier can return and a key of `tiers` */
|
|
2676
2785
|
name: string;
|
|
@@ -2689,18 +2798,39 @@ export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPro
|
|
|
2689
2798
|
TierDefinition,
|
|
2690
2799
|
) as any as S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList>;
|
|
2691
2800
|
|
|
2801
|
+
export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap =
|
|
2802
|
+
{ [key: string]: string | undefined };
|
|
2803
|
+
export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap =
|
|
2804
|
+
/*@__PURE__*/ S.Record(
|
|
2805
|
+
S.String,
|
|
2806
|
+
S.String,
|
|
2807
|
+
) as any as S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap>;
|
|
2808
|
+
|
|
2692
2809
|
export interface PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest {
|
|
2810
|
+
classification_examples?: string | null;
|
|
2693
2811
|
classification_prompt?: string | null;
|
|
2812
|
+
classification_rubric?: ClassificationRubric | (string & {}) | null;
|
|
2694
2813
|
context_window_size?: number;
|
|
2695
|
-
tier_definitions
|
|
2814
|
+
tier_definitions?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList | null;
|
|
2815
|
+
tier_labels?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap | null;
|
|
2696
2816
|
}
|
|
2697
2817
|
export const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest =
|
|
2698
2818
|
/*@__PURE__*/ S.suspend(() =>
|
|
2699
2819
|
S.Struct({
|
|
2820
|
+
classification_examples: S.optional(S.NullOr(S.String)),
|
|
2700
2821
|
classification_prompt: S.optional(S.NullOr(S.String)),
|
|
2822
|
+
classification_rubric: S.optional(S.NullOr(ClassificationRubric)),
|
|
2701
2823
|
context_window_size: S.optional(S.Number),
|
|
2702
|
-
tier_definitions:
|
|
2703
|
-
|
|
2824
|
+
tier_definitions: S.optional(
|
|
2825
|
+
S.NullOr(
|
|
2826
|
+
PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList,
|
|
2827
|
+
),
|
|
2828
|
+
),
|
|
2829
|
+
tier_labels: S.optional(
|
|
2830
|
+
S.NullOr(
|
|
2831
|
+
PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap,
|
|
2832
|
+
),
|
|
2833
|
+
),
|
|
2704
2834
|
}).pipe(
|
|
2705
2835
|
T.Http({
|
|
2706
2836
|
method: "POST",
|
|
@@ -2732,40 +2862,141 @@ export const AdaptiveRouterWeights = /*@__PURE__*/ S.suspend(() =>
|
|
|
2732
2862
|
identifier: "AdaptiveRouterWeights",
|
|
2733
2863
|
}) as any as S.Schema<AdaptiveRouterWeights>;
|
|
2734
2864
|
|
|
2865
|
+
export interface CapabilityCalibrationConfig {
|
|
2866
|
+
intercept: number;
|
|
2867
|
+
slope: number;
|
|
2868
|
+
version: string;
|
|
2869
|
+
}
|
|
2870
|
+
export const CapabilityCalibrationConfig = /*@__PURE__*/ S.suspend(() =>
|
|
2871
|
+
S.Struct({
|
|
2872
|
+
intercept: S.Number,
|
|
2873
|
+
slope: S.Number,
|
|
2874
|
+
version: S.String,
|
|
2875
|
+
}),
|
|
2876
|
+
).annotate({
|
|
2877
|
+
identifier: "CapabilityCalibrationConfig",
|
|
2878
|
+
}) as any as S.Schema<CapabilityCalibrationConfig>;
|
|
2879
|
+
|
|
2880
|
+
/** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
|
|
2881
|
+
export type CapabilityClassifierConfigResponseFormat =
|
|
2882
|
+
| "json_schema"
|
|
2883
|
+
| "json_object";
|
|
2884
|
+
export const CapabilityClassifierConfigResponseFormat = S.String;
|
|
2885
|
+
|
|
2886
|
+
/** Switchyard-compatible probability threshold policy for two model tiers. */
|
|
2887
|
+
export interface CapabilityClassifierConfig {
|
|
2888
|
+
/** Lowest p_solve that routes a supported task to efficient_tier */
|
|
2889
|
+
base_threshold: number;
|
|
2890
|
+
/** Optional versioned sigmoid calibration fitted for this judge, capability card, efficient model, and execution setup. Applies sigmoid(slope * logit(clip(p_solve, 1e-6, 1-1e-6)) + intercept) before the threshold policy. Omit to route on the raw forecast. */
|
|
2891
|
+
calibration?: CapabilityCalibrationConfig | null;
|
|
2892
|
+
/** Higher, fail-closed tier used below the adjusted threshold or when the classifier verdict is unavailable */
|
|
2893
|
+
capable_tier: string;
|
|
2894
|
+
/** Tier used when the efficient model's forecasted solve probability meets the adjusted threshold */
|
|
2895
|
+
efficient_tier: string;
|
|
2896
|
+
/** Maximum completion tokens available to the capability classifier verdict */
|
|
2897
|
+
max_output_tokens?: number;
|
|
2898
|
+
/** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
|
|
2899
|
+
response_format?: CapabilityClassifierConfigResponseFormat | (string & {});
|
|
2900
|
+
/** Amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts */
|
|
2901
|
+
threshold_step?: number;
|
|
2902
|
+
}
|
|
2903
|
+
export const CapabilityClassifierConfig = /*@__PURE__*/ S.suspend(() =>
|
|
2904
|
+
S.Struct({
|
|
2905
|
+
base_threshold: S.Number,
|
|
2906
|
+
calibration: S.optional(S.NullOr(CapabilityCalibrationConfig)),
|
|
2907
|
+
capable_tier: S.String,
|
|
2908
|
+
efficient_tier: S.String,
|
|
2909
|
+
max_output_tokens: S.optional(S.Number),
|
|
2910
|
+
response_format: S.optional(CapabilityClassifierConfigResponseFormat),
|
|
2911
|
+
threshold_step: S.optional(S.Number),
|
|
2912
|
+
}),
|
|
2913
|
+
).annotate({
|
|
2914
|
+
identifier: "CapabilityClassifierConfig",
|
|
2915
|
+
}) as any as S.Schema<CapabilityClassifierConfig>;
|
|
2916
|
+
|
|
2917
|
+
/** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
|
|
2918
|
+
export type RequestComplexityRouterConfigClassificationMode =
|
|
2919
|
+
| "every_request"
|
|
2920
|
+
| "user_turn";
|
|
2921
|
+
export const RequestComplexityRouterConfigClassificationMode = S.String;
|
|
2922
|
+
|
|
2735
2923
|
/** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
|
|
2736
2924
|
export type RequestComplexityRouterConfigClassifierFallback =
|
|
2737
2925
|
| "heuristic"
|
|
2738
2926
|
| "default_model";
|
|
2739
2927
|
export const RequestComplexityRouterConfigClassifierFallback = S.String;
|
|
2740
2928
|
|
|
2929
|
+
export type ClassifierLLMConfigReasoningEffort =
|
|
2930
|
+
| "none"
|
|
2931
|
+
| "minimal"
|
|
2932
|
+
| "low"
|
|
2933
|
+
| "medium"
|
|
2934
|
+
| "high"
|
|
2935
|
+
| "xhigh"
|
|
2936
|
+
| "max";
|
|
2937
|
+
export const ClassifierLLMConfigReasoningEffort = S.String;
|
|
2938
|
+
|
|
2939
|
+
/** Whether the LLM classifier sees the images on the request it is classifying. Off by default because images cost far more than the text ask they arrive with, and the classifier runs on every request. A turn whose complexity lives in the image ("what is wrong in this stack trace screenshot") is invisible to a text-only classifier, which is what this buys. */
|
|
2940
|
+
export interface ClassifierVisionConfig {
|
|
2941
|
+
/** Forward image content to the classifier. Requires a classifier model declared supports_vision, on the deployment's model_info or in the model cost map; images stay stripped otherwise, so a classifier that cannot read them is never sent one. Declare model_info.supports_vision on the deployment to enable a model the cost map does not describe. Only inline data: URIs are forwarded. A request whose images are http(s) URLs still classifies on its text alone, because some providers fetch such a URL from the proxy rather than the provider, which would let a caller aim a proxy-side request at an address of their choosing. */
|
|
2942
|
+
enabled?: boolean;
|
|
2943
|
+
/** How many images from the newest user turn to forward, in wire order. Bounds the added cost of a turn that attaches many images. Images on earlier turns are never forwarded. */
|
|
2944
|
+
max_images?: number;
|
|
2945
|
+
}
|
|
2946
|
+
export const ClassifierVisionConfig = /*@__PURE__*/ S.suspend(() =>
|
|
2947
|
+
S.Struct({
|
|
2948
|
+
enabled: S.optional(S.Boolean),
|
|
2949
|
+
max_images: S.optional(S.Number),
|
|
2950
|
+
}),
|
|
2951
|
+
).annotate({
|
|
2952
|
+
identifier: "ClassifierVisionConfig",
|
|
2953
|
+
}) as any as S.Schema<ClassifierVisionConfig>;
|
|
2954
|
+
|
|
2741
2955
|
/** Configuration for the LLM-based complexity classifier. */
|
|
2742
2956
|
export interface ClassifierLLMConfig {
|
|
2957
|
+
/** How long to skip this router's LLM classifier after a classification call times out. Requests use classifier_fallback during the cooldown. When it expires, one request probes the classifier while concurrent requests keep using the fallback; a successful probe closes the circuit and a failed probe restarts the cooldown. */
|
|
2958
|
+
circuit_breaker_cooldown_seconds?: number;
|
|
2959
|
+
/** Whether one classifier timeout temporarily sends requests through classifier_fallback. Enabled by default so an unhealthy classifier cannot repeat its timeout across sessions. */
|
|
2960
|
+
circuit_breaker_enabled?: boolean;
|
|
2743
2961
|
/** Which calibration examples the built-in rubric carries. 'agentic' anchors routine installs, builds, multi-file edits, and standard debugging at MEDIUM, so ordinary engineering does not route to the most expensive tier; it suits agent, terminal, and coding-assistant traffic as well as mixed traffic. 'chat' omits those engineering anchors, for a deployment serving only conversational traffic. 'business' carries business/sales anchors and business-flavored tier criteria that keep routine drafting and summarizing off the expensive tiers and reserve the top tier for committing to decisions under tradeoffs; it suits sales, support, and go-to-market traffic. Every preset keeps the same four tiers, so this moves where the boundary sits without changing the taxonomy. Leave unset for 'legacy', the rubric as it shipped before calibration examples existed, so an existing router's tier decisions and spend do not move on upgrade. Mutually exclusive with system_prompt, which replaces the rubric this would select. Only applies when classifier_type is 'llm'. */
|
|
2744
2962
|
classification_rubric?: ClassificationRubric | (string & {}) | null;
|
|
2745
2963
|
/** Model name (from the router's model_list) to call for classification */
|
|
2746
2964
|
model: string;
|
|
2965
|
+
/** Reasoning effort override for classifier calls. Leave unset to use the classifier deployment or provider default. */
|
|
2966
|
+
reasoning_effort?: ClassifierLLMConfigReasoningEffort | (string & {}) | null;
|
|
2747
2967
|
/** Replaces the built-in complexity rubric as the classifier's entire system role. When set, neither the default rubric nor the context-window closing line is appended, so the prompt owns the whole taxonomy and the tier names SIMPLE/MEDIUM/COMPLEX/REASONING become whatever buckets it defines: a prompt that classifies data sensitivity routes on that instead of on difficulty. Two consequences of full replacement. The default rubric's closing paragraph is the classifier's prompt-injection defense, telling it that the caller's quoted system prompt and prior turns are material to judge and never instructions; a replacement that omits it lets a caller ask for a tier and get it. And the heuristic fallback still scores complexity, so a router on some other taxonomy wants classifier_fallback='default_model'. Leave unset for the built-in rubric. Only applies when classifier_type is 'llm'. */
|
|
2748
2968
|
system_prompt?: string | null;
|
|
2749
2969
|
/** Timeout budget for the classification call, in milliseconds */
|
|
2750
2970
|
timeout_ms?: number;
|
|
2971
|
+
/** Whether the classifier sees images on the request, and how many */
|
|
2972
|
+
vision?: ClassifierVisionConfig;
|
|
2751
2973
|
}
|
|
2752
2974
|
export const ClassifierLLMConfig = /*@__PURE__*/ S.suspend(() =>
|
|
2753
2975
|
S.Struct({
|
|
2976
|
+
circuit_breaker_cooldown_seconds: S.optional(S.Number),
|
|
2977
|
+
circuit_breaker_enabled: S.optional(S.Boolean),
|
|
2754
2978
|
classification_rubric: S.optional(S.NullOr(ClassificationRubric)),
|
|
2755
2979
|
model: S.String,
|
|
2980
|
+
reasoning_effort: S.optional(S.NullOr(ClassifierLLMConfigReasoningEffort)),
|
|
2756
2981
|
system_prompt: S.optional(S.NullOr(S.String)),
|
|
2757
2982
|
timeout_ms: S.optional(S.Number),
|
|
2983
|
+
vision: S.optional(ClassifierVisionConfig),
|
|
2758
2984
|
}),
|
|
2759
2985
|
).annotate({
|
|
2760
2986
|
identifier: "ClassifierLLMConfig",
|
|
2761
2987
|
}) as any as S.Schema<ClassifierLLMConfig>;
|
|
2762
2988
|
|
|
2763
|
-
/** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin,
|
|
2989
|
+
/** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
|
|
2764
2990
|
export type RequestComplexityRouterConfigClassifierType =
|
|
2765
2991
|
| "heuristic"
|
|
2992
|
+
| "heuristic_v2"
|
|
2766
2993
|
| "llm"
|
|
2994
|
+
| "capability"
|
|
2995
|
+
| "llm_v2"
|
|
2767
2996
|
| "custom"
|
|
2768
|
-
| "heuristic_first"
|
|
2997
|
+
| "heuristic_first"
|
|
2998
|
+
| "hybrid"
|
|
2999
|
+
| "jev";
|
|
2769
3000
|
export const RequestComplexityRouterConfigClassifierType = S.String;
|
|
2770
3001
|
|
|
2771
3002
|
export type RequestComplexityRouterConfigCodeKeywordsList = Array<string>;
|
|
@@ -2774,6 +3005,48 @@ export const RequestComplexityRouterConfigCodeKeywordsList =
|
|
|
2774
3005
|
S.String,
|
|
2775
3006
|
) as any as S.Schema<RequestComplexityRouterConfigCodeKeywordsList>;
|
|
2776
3007
|
|
|
3008
|
+
export type CustomDimensionKeywordsList = Array<string>;
|
|
3009
|
+
export const CustomDimensionKeywordsList = /*@__PURE__*/ S.Array(
|
|
3010
|
+
S.String,
|
|
3011
|
+
) as any as S.Schema<CustomDimensionKeywordsList>;
|
|
3012
|
+
|
|
3013
|
+
export type CustomDimensionPatternsList = Array<string>;
|
|
3014
|
+
export const CustomDimensionPatternsList = /*@__PURE__*/ S.Array(
|
|
3015
|
+
S.String,
|
|
3016
|
+
) as any as S.Schema<CustomDimensionPatternsList>;
|
|
3017
|
+
|
|
3018
|
+
/** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
|
|
3019
|
+
export type CustomDimensionScoringMode = "binary" | "match_count";
|
|
3020
|
+
export const CustomDimensionScoringMode = S.String;
|
|
3021
|
+
|
|
3022
|
+
export interface CustomDimension {
|
|
3023
|
+
keywords?: CustomDimensionKeywordsList;
|
|
3024
|
+
name: string;
|
|
3025
|
+
patterns?: CustomDimensionPatternsList;
|
|
3026
|
+
/** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
|
|
3027
|
+
scoring_mode?: CustomDimensionScoringMode | (string & {});
|
|
3028
|
+
weight: number;
|
|
3029
|
+
}
|
|
3030
|
+
export const CustomDimension = /*@__PURE__*/ S.suspend(() =>
|
|
3031
|
+
S.Struct({
|
|
3032
|
+
keywords: S.optional(CustomDimensionKeywordsList),
|
|
3033
|
+
name: S.String,
|
|
3034
|
+
patterns: S.optional(CustomDimensionPatternsList),
|
|
3035
|
+
scoring_mode: S.optional(CustomDimensionScoringMode),
|
|
3036
|
+
weight: S.Number,
|
|
3037
|
+
}),
|
|
3038
|
+
).annotate({
|
|
3039
|
+
identifier: "CustomDimension",
|
|
3040
|
+
}) as any as S.Schema<CustomDimension>;
|
|
3041
|
+
|
|
3042
|
+
/** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
|
|
3043
|
+
export type RequestComplexityRouterConfigCustomDimensionsList =
|
|
3044
|
+
Array<CustomDimension>;
|
|
3045
|
+
export const RequestComplexityRouterConfigCustomDimensionsList =
|
|
3046
|
+
/*@__PURE__*/ S.Array(
|
|
3047
|
+
CustomDimension,
|
|
3048
|
+
) as any as S.Schema<RequestComplexityRouterConfigCustomDimensionsList>;
|
|
3049
|
+
|
|
2777
3050
|
export type RequestComplexityRouterConfigCustomTechnicalKeywordsList =
|
|
2778
3051
|
Array<string>;
|
|
2779
3052
|
export const RequestComplexityRouterConfigCustomTechnicalKeywordsList =
|
|
@@ -2797,6 +3070,142 @@ export const RequestComplexityRouterConfigEscalationKeywordsList =
|
|
|
2797
3070
|
S.String,
|
|
2798
3071
|
) as any as S.Schema<RequestComplexityRouterConfigEscalationKeywordsList>;
|
|
2799
3072
|
|
|
3073
|
+
export interface TierCohortStatistic {
|
|
3074
|
+
cohort: string;
|
|
3075
|
+
observations: number;
|
|
3076
|
+
successes: number;
|
|
3077
|
+
tier: number;
|
|
3078
|
+
}
|
|
3079
|
+
export const TierCohortStatistic = /*@__PURE__*/ S.suspend(() =>
|
|
3080
|
+
S.Struct({
|
|
3081
|
+
cohort: S.String,
|
|
3082
|
+
observations: S.Number,
|
|
3083
|
+
successes: S.Number,
|
|
3084
|
+
tier: S.Number,
|
|
3085
|
+
}),
|
|
3086
|
+
).annotate({
|
|
3087
|
+
identifier: "TierCohortStatistic",
|
|
3088
|
+
}) as any as S.Schema<TierCohortStatistic>;
|
|
3089
|
+
|
|
3090
|
+
export type TrainedTierArtifactCohortStatisticsList =
|
|
3091
|
+
Array<TierCohortStatistic>;
|
|
3092
|
+
export const TrainedTierArtifactCohortStatisticsList = /*@__PURE__*/ S.Array(
|
|
3093
|
+
TierCohortStatistic,
|
|
3094
|
+
) as any as S.Schema<TrainedTierArtifactCohortStatisticsList>;
|
|
3095
|
+
|
|
3096
|
+
export interface TierDataset {
|
|
3097
|
+
license: string;
|
|
3098
|
+
name: string;
|
|
3099
|
+
rows: number;
|
|
3100
|
+
success_definition?: string;
|
|
3101
|
+
url: string;
|
|
3102
|
+
}
|
|
3103
|
+
export const TierDataset = /*@__PURE__*/ S.suspend(() =>
|
|
3104
|
+
S.Struct({
|
|
3105
|
+
license: S.String,
|
|
3106
|
+
name: S.String,
|
|
3107
|
+
rows: S.Number,
|
|
3108
|
+
success_definition: S.optional(S.String),
|
|
3109
|
+
url: S.String,
|
|
3110
|
+
}),
|
|
3111
|
+
).annotate({ identifier: "TierDataset" }) as any as S.Schema<TierDataset>;
|
|
3112
|
+
|
|
3113
|
+
export type TrainedTierArtifactDatasetsList = Array<TierDataset>;
|
|
3114
|
+
export const TrainedTierArtifactDatasetsList = /*@__PURE__*/ S.Array(
|
|
3115
|
+
TierDataset,
|
|
3116
|
+
) as any as S.Schema<TrainedTierArtifactDatasetsList>;
|
|
3117
|
+
|
|
3118
|
+
/** Fixed v0 taxonomy. User-extensible types come in v1. */
|
|
3119
|
+
export type RequestType =
|
|
3120
|
+
| "code_generation"
|
|
3121
|
+
| "code_understanding"
|
|
3122
|
+
| "technical_design"
|
|
3123
|
+
| "analytical_reasoning"
|
|
3124
|
+
| "writing"
|
|
3125
|
+
| "factual_lookup"
|
|
3126
|
+
| "general";
|
|
3127
|
+
export const RequestType = S.String;
|
|
3128
|
+
|
|
3129
|
+
export interface TierDomainStatistic {
|
|
3130
|
+
observations: number;
|
|
3131
|
+
request_type: RequestType | (string & {});
|
|
3132
|
+
successes: number;
|
|
3133
|
+
tier: number;
|
|
3134
|
+
}
|
|
3135
|
+
export const TierDomainStatistic = /*@__PURE__*/ S.suspend(() =>
|
|
3136
|
+
S.Struct({
|
|
3137
|
+
observations: S.Number,
|
|
3138
|
+
request_type: RequestType,
|
|
3139
|
+
successes: S.Number,
|
|
3140
|
+
tier: S.Number,
|
|
3141
|
+
}),
|
|
3142
|
+
).annotate({
|
|
3143
|
+
identifier: "TierDomainStatistic",
|
|
3144
|
+
}) as any as S.Schema<TierDomainStatistic>;
|
|
3145
|
+
|
|
3146
|
+
export type TrainedTierArtifactDomainStatisticsList =
|
|
3147
|
+
Array<TierDomainStatistic>;
|
|
3148
|
+
export const TrainedTierArtifactDomainStatisticsList = /*@__PURE__*/ S.Array(
|
|
3149
|
+
TierDomainStatistic,
|
|
3150
|
+
) as any as S.Schema<TrainedTierArtifactDomainStatisticsList>;
|
|
3151
|
+
|
|
3152
|
+
export interface TierGlobalStatistic {
|
|
3153
|
+
observations: number;
|
|
3154
|
+
successes: number;
|
|
3155
|
+
tier: number;
|
|
3156
|
+
}
|
|
3157
|
+
export const TierGlobalStatistic = /*@__PURE__*/ S.suspend(() =>
|
|
3158
|
+
S.Struct({
|
|
3159
|
+
observations: S.Number,
|
|
3160
|
+
successes: S.Number,
|
|
3161
|
+
tier: S.Number,
|
|
3162
|
+
}),
|
|
3163
|
+
).annotate({
|
|
3164
|
+
identifier: "TierGlobalStatistic",
|
|
3165
|
+
}) as any as S.Schema<TierGlobalStatistic>;
|
|
3166
|
+
|
|
3167
|
+
export type TrainedTierArtifactGlobalStatisticsList =
|
|
3168
|
+
Array<TierGlobalStatistic>;
|
|
3169
|
+
export const TrainedTierArtifactGlobalStatisticsList = /*@__PURE__*/ S.Array(
|
|
3170
|
+
TierGlobalStatistic,
|
|
3171
|
+
) as any as S.Schema<TrainedTierArtifactGlobalStatisticsList>;
|
|
3172
|
+
|
|
3173
|
+
export interface TrainedTierArtifact {
|
|
3174
|
+
cohort_prior_mass?: number;
|
|
3175
|
+
cohort_statistics?: TrainedTierArtifactCohortStatisticsList;
|
|
3176
|
+
datasets?: TrainedTierArtifactDatasetsList;
|
|
3177
|
+
domain_prior_mass?: number;
|
|
3178
|
+
domain_statistics?: TrainedTierArtifactDomainStatisticsList;
|
|
3179
|
+
global_statistics: TrainedTierArtifactGlobalStatisticsList;
|
|
3180
|
+
routing_threshold?: number;
|
|
3181
|
+
schema_version?: number;
|
|
3182
|
+
split_method?: string;
|
|
3183
|
+
success_definition?: string;
|
|
3184
|
+
}
|
|
3185
|
+
export const TrainedTierArtifact = /*@__PURE__*/ S.suspend(() =>
|
|
3186
|
+
S.Struct({
|
|
3187
|
+
cohort_prior_mass: S.optional(S.Number),
|
|
3188
|
+
cohort_statistics: S.optional(TrainedTierArtifactCohortStatisticsList),
|
|
3189
|
+
datasets: S.optional(TrainedTierArtifactDatasetsList),
|
|
3190
|
+
domain_prior_mass: S.optional(S.Number),
|
|
3191
|
+
domain_statistics: S.optional(TrainedTierArtifactDomainStatisticsList),
|
|
3192
|
+
global_statistics: TrainedTierArtifactGlobalStatisticsList,
|
|
3193
|
+
routing_threshold: S.optional(S.Number),
|
|
3194
|
+
schema_version: S.optional(S.Number),
|
|
3195
|
+
split_method: S.optional(S.String),
|
|
3196
|
+
success_definition: S.optional(S.String),
|
|
3197
|
+
}),
|
|
3198
|
+
).annotate({
|
|
3199
|
+
identifier: "TrainedTierArtifact",
|
|
3200
|
+
}) as any as S.Schema<TrainedTierArtifact>;
|
|
3201
|
+
|
|
3202
|
+
/** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
|
|
3203
|
+
export type RequestComplexityRouterConfigHeuristicV2Artifact =
|
|
3204
|
+
| TrainedTierArtifact
|
|
3205
|
+
| string;
|
|
3206
|
+
export const RequestComplexityRouterConfigHeuristicV2Artifact =
|
|
3207
|
+
S.Unknown as any as S.Schema<RequestComplexityRouterConfigHeuristicV2Artifact>;
|
|
3208
|
+
|
|
2800
3209
|
export type RequestComplexityRouterConfigHousekeepingPatternsList =
|
|
2801
3210
|
Array<string>;
|
|
2802
3211
|
export const RequestComplexityRouterConfigHousekeepingPatternsList =
|
|
@@ -2804,6 +3213,32 @@ export const RequestComplexityRouterConfigHousekeepingPatternsList =
|
|
|
2804
3213
|
S.String,
|
|
2805
3214
|
) as any as S.Schema<RequestComplexityRouterConfigHousekeepingPatternsList>;
|
|
2806
3215
|
|
|
3216
|
+
export interface JevClassifierConfig {
|
|
3217
|
+
/** TypeSafe API base, falling back to TYPESAFE_API_BASE and then https://api.typesafe.ai */
|
|
3218
|
+
api_base?: string | null;
|
|
3219
|
+
/** TypeSafe API key, falling back to TYPESAFE_API_KEY */
|
|
3220
|
+
api_key?: string | null;
|
|
3221
|
+
circuit_breaker_cooldown_seconds?: number;
|
|
3222
|
+
circuit_breaker_enabled?: boolean;
|
|
3223
|
+
/** Replaces the built-in Jev question instructions */
|
|
3224
|
+
instructions?: string | null;
|
|
3225
|
+
model?: string;
|
|
3226
|
+
timeout_ms?: number;
|
|
3227
|
+
}
|
|
3228
|
+
export const JevClassifierConfig = /*@__PURE__*/ S.suspend(() =>
|
|
3229
|
+
S.Struct({
|
|
3230
|
+
api_base: S.optional(S.NullOr(S.String)),
|
|
3231
|
+
api_key: S.optional(S.NullOr(S.String)),
|
|
3232
|
+
circuit_breaker_cooldown_seconds: S.optional(S.Number),
|
|
3233
|
+
circuit_breaker_enabled: S.optional(S.Boolean),
|
|
3234
|
+
instructions: S.optional(S.NullOr(S.String)),
|
|
3235
|
+
model: S.optional(S.String),
|
|
3236
|
+
timeout_ms: S.optional(S.Number),
|
|
3237
|
+
}),
|
|
3238
|
+
).annotate({
|
|
3239
|
+
identifier: "JevClassifierConfig",
|
|
3240
|
+
}) as any as S.Schema<JevClassifierConfig>;
|
|
3241
|
+
|
|
2807
3242
|
/** Keywords/phrases that trigger this rule (lexical or semantic match) */
|
|
2808
3243
|
export type KeywordTierRuleKeywordsList = Array<string>;
|
|
2809
3244
|
export const KeywordTierRuleKeywordsList = /*@__PURE__*/ S.Array(
|
|
@@ -2833,6 +3268,71 @@ export const RequestComplexityRouterConfigKeywordTierRulesList =
|
|
|
2833
3268
|
KeywordTierRule,
|
|
2834
3269
|
) as any as S.Schema<RequestComplexityRouterConfigKeywordTierRulesList>;
|
|
2835
3270
|
|
|
3271
|
+
export interface LLMV2ProbabilityCalibration {
|
|
3272
|
+
intercept: number;
|
|
3273
|
+
slope: number;
|
|
3274
|
+
}
|
|
3275
|
+
export const LLMV2ProbabilityCalibration = /*@__PURE__*/ S.suspend(() =>
|
|
3276
|
+
S.Struct({
|
|
3277
|
+
intercept: S.Number,
|
|
3278
|
+
slope: S.Number,
|
|
3279
|
+
}),
|
|
3280
|
+
).annotate({
|
|
3281
|
+
identifier: "LLMV2ProbabilityCalibration",
|
|
3282
|
+
}) as any as S.Schema<LLMV2ProbabilityCalibration>;
|
|
3283
|
+
|
|
3284
|
+
export interface LLMV2Calibration {
|
|
3285
|
+
capable: LLMV2ProbabilityCalibration;
|
|
3286
|
+
efficient: LLMV2ProbabilityCalibration;
|
|
3287
|
+
prompt_version: string;
|
|
3288
|
+
version: string;
|
|
3289
|
+
}
|
|
3290
|
+
export const LLMV2Calibration = /*@__PURE__*/ S.suspend(() =>
|
|
3291
|
+
S.Struct({
|
|
3292
|
+
capable: LLMV2ProbabilityCalibration,
|
|
3293
|
+
efficient: LLMV2ProbabilityCalibration,
|
|
3294
|
+
prompt_version: S.String,
|
|
3295
|
+
version: S.String,
|
|
3296
|
+
}),
|
|
3297
|
+
).annotate({
|
|
3298
|
+
identifier: "LLMV2Calibration",
|
|
3299
|
+
}) as any as S.Schema<LLMV2Calibration>;
|
|
3300
|
+
|
|
3301
|
+
export type LLMV2ConfigResponseFormat = "json_schema" | "json_object";
|
|
3302
|
+
export const LLMV2ConfigResponseFormat = S.String;
|
|
3303
|
+
|
|
3304
|
+
export interface LLMV2Config {
|
|
3305
|
+
calibration?: LLMV2Calibration | null;
|
|
3306
|
+
capable_profile?: string | null;
|
|
3307
|
+
capable_profile_preset?: string | null;
|
|
3308
|
+
capable_tier?: string;
|
|
3309
|
+
efficient_profile?: string | null;
|
|
3310
|
+
efficient_profile_preset?: string | null;
|
|
3311
|
+
efficient_tier?: string;
|
|
3312
|
+
harness?: string | null;
|
|
3313
|
+
harness_preset?: string | null;
|
|
3314
|
+
max_output_tokens?: number;
|
|
3315
|
+
/** Maximum estimated success loss allowed for efficient. */
|
|
3316
|
+
max_quality_gap: number;
|
|
3317
|
+
response_format?: LLMV2ConfigResponseFormat | (string & {});
|
|
3318
|
+
}
|
|
3319
|
+
export const LLMV2Config = /*@__PURE__*/ S.suspend(() =>
|
|
3320
|
+
S.Struct({
|
|
3321
|
+
calibration: S.optional(S.NullOr(LLMV2Calibration)),
|
|
3322
|
+
capable_profile: S.optional(S.NullOr(S.String)),
|
|
3323
|
+
capable_profile_preset: S.optional(S.NullOr(S.String)),
|
|
3324
|
+
capable_tier: S.optional(S.String),
|
|
3325
|
+
efficient_profile: S.optional(S.NullOr(S.String)),
|
|
3326
|
+
efficient_profile_preset: S.optional(S.NullOr(S.String)),
|
|
3327
|
+
efficient_tier: S.optional(S.String),
|
|
3328
|
+
harness: S.optional(S.NullOr(S.String)),
|
|
3329
|
+
harness_preset: S.optional(S.NullOr(S.String)),
|
|
3330
|
+
max_output_tokens: S.optional(S.Number),
|
|
3331
|
+
max_quality_gap: S.Number,
|
|
3332
|
+
response_format: S.optional(LLMV2ConfigResponseFormat),
|
|
3333
|
+
}),
|
|
3334
|
+
).annotate({ identifier: "LLMV2Config" }) as any as S.Schema<LLMV2Config>;
|
|
3335
|
+
|
|
2836
3336
|
export type RequestComplexityRouterConfigPlanModePatternsList = Array<string>;
|
|
2837
3337
|
export const RequestComplexityRouterConfigPlanModePatternsList =
|
|
2838
3338
|
/*@__PURE__*/ S.Array(
|
|
@@ -2987,52 +3487,81 @@ export interface RequestComplexityRouterConfig {
|
|
|
2987
3487
|
| (string & {});
|
|
2988
3488
|
/** Quality vs cost weights for adaptive selection (used when adaptive=True) */
|
|
2989
3489
|
adaptive_weights?: AdaptiveRouterWeights;
|
|
2990
|
-
/**
|
|
3490
|
+
/** Probability threshold policy required when classifier_type is 'capability'. The classifier forecasts p_solve for efficient_tier, adjusts base_threshold using the capability-card boundary, and otherwise routes to capable_tier */
|
|
3491
|
+
capability_classifier_config?: CapabilityClassifierConfig | null;
|
|
3492
|
+
/** Replaces the calibration examples of the LLM classifier rubric, and nothing else. Written as example lines only: the router renders the 'Calibration examples:' heading above them, after the per-tier bullets. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_prompt replaces them, the classification instructions; a custom tier set ships no examples of its own, so the section renders only when this is set. */
|
|
3493
|
+
classification_examples?: string | null;
|
|
3494
|
+
/** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
|
|
3495
|
+
classification_mode?:
|
|
3496
|
+
| RequestComplexityRouterConfigClassificationMode
|
|
3497
|
+
| (string & {});
|
|
3498
|
+
/** Replaces the classification instructions that open the LLM classifier rubric, and nothing else. The per-tier bullets follow it, the calibration examples follow those, and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text is always appended after them and cannot be overridden. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_examples replaces them, the calibration examples. */
|
|
2991
3499
|
classification_prompt?: string | null;
|
|
2992
|
-
/** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the
|
|
3500
|
+
/** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
|
|
2993
3501
|
classifier_context_budget_chars?: number;
|
|
2994
3502
|
/** Include assistant turns in the classifier context window, so difficulty stated by the model rather than by the user stays visible: a plan the assistant calls complex, which the user approves with 'yes', is classified on the work being approved instead of on the word 'yes'. When enabled, classifier_context_window_size counts the last N turns of the conversation across both roles rather than the last N user turns, and assistant text is sent to the classifier model, which may be a different deployment or provider than the routed completion model. Assistant replies spend classifier_context_budget_chars alongside user turns, so raise it if the oldest turns stop being quoted once replies join the window. Off by default because enabling it shifts tier decisions, and therefore spend, for an already-deployed router. Only applies when classifier_type is 'llm'. */
|
|
2995
3503
|
classifier_context_include_assistant_turns?: boolean;
|
|
2996
3504
|
/** Optional cap on each individual prior turn's text, applied before classifier_context_budget_chars bounds the block. Unset by default, so one long turn may spend the whole budget, which is usually what a follow-up needs; set it when no single turn should dominate the context the classifier sees. A capped turn keeps its opening and its ending with the middle elided. Only applies when classifier_type is 'llm'. */
|
|
2997
3505
|
classifier_context_per_turn_chars?: number | null;
|
|
2998
|
-
/** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call
|
|
3506
|
+
/** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
|
|
2999
3507
|
classifier_context_window_size?: number;
|
|
3000
3508
|
/** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
|
|
3001
3509
|
classifier_fallback?:
|
|
3002
3510
|
| RequestComplexityRouterConfigClassifierFallback
|
|
3003
3511
|
| (string & {});
|
|
3004
|
-
/** Configuration for the LLM classifier; required when classifier_type is 'llm'
|
|
3512
|
+
/** Configuration for the LLM classifier; required when classifier_type is 'llm', 'capability', 'heuristic_first' or 'hybrid' */
|
|
3005
3513
|
classifier_llm_config?: ClassifierLLMConfig | null;
|
|
3006
3514
|
/** Not settable over HTTP; the classifier plugin is a runtime object */
|
|
3007
3515
|
classifier_plugin?: unknown | null;
|
|
3008
3516
|
/** Timeout budget for the classifier plugin call, in milliseconds. On expiry the fallback path decides the tier. Only applies when classifier_type is 'custom'. */
|
|
3009
3517
|
classifier_plugin_timeout_ms?: number;
|
|
3010
|
-
/** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin,
|
|
3518
|
+
/** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
|
|
3011
3519
|
classifier_type?: RequestComplexityRouterConfigClassifierType | (string & {});
|
|
3012
3520
|
/** Keywords indicating code-related content */
|
|
3013
3521
|
code_keywords?: RequestComplexityRouterConfigCodeKeywordsList | null;
|
|
3522
|
+
/** Fraction of a model's declared context window the estimated prompt must fit within. The token count is an estimate, so fitting against the full window would dispatch prompts that the provider's own tokenizer then rejects; 0.95 leaves room for that drift plus the response tokens. */
|
|
3523
|
+
context_window_escalation_buffer?: number;
|
|
3524
|
+
/** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
|
|
3525
|
+
custom_dimensions?: RequestComplexityRouterConfigCustomDimensionsList;
|
|
3014
3526
|
/** Domain-specific technical keywords appended to the effective base list (technical_keywords if set, otherwise DEFAULT_TECHNICAL_KEYWORDS). Order is preserved; duplicates are removed case-insensitively against the base list and within this list. */
|
|
3015
3527
|
custom_technical_keywords?: RequestComplexityRouterConfigCustomTechnicalKeywordsList | null;
|
|
3016
3528
|
/** Default model to use if tier cannot be determined */
|
|
3017
3529
|
default_model?: string | null;
|
|
3018
|
-
/** When True and a session_id is resolvable
|
|
3530
|
+
/** When True and a client session_id is resolvable, reuse the session's chosen model for each classified tier and its deployment within each model group. With session_affinity off, every turn is still classified: moving to another tier leaves the previous tier's model pin intact for a later return. Pins yield to current candidate, context, modality, and availability constraints. Adaptive selection chooses the initial model from its eligible pool, then reuses that choice per tier. This reduces avoidable provider prompt-cache misses; it does not guarantee cache hits. Set False to select models and load-balance deployments on every turn, unless session_affinity or user_turn classification requires a pin. Inert without a client session_id and suppressed when plugins are configured. */
|
|
3019
3531
|
deployment_affinity?: boolean;
|
|
3020
3532
|
/** Weights for each scoring dimension */
|
|
3021
3533
|
dimension_weights?: RequestComplexityRouterConfigDimensionWeightsMap;
|
|
3022
3534
|
/** Embedding model (LiteLLM model name) used when semantic_keyword_matching is enabled */
|
|
3023
3535
|
embedding_model?: string | null;
|
|
3536
|
+
/** Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Set false to dispatch on complexity alone, as before. */
|
|
3537
|
+
enable_context_window_escalation?: boolean;
|
|
3538
|
+
/** Add NON_REASONING as a fifth built-in tier below SIMPLE, for operational agent traffic that relays or reformats information rather than reasoning about it. Off by default: turning it on adds a rung to this router's ladder, a bullet to the LLM classifier's rubric, and a value the classifier may return, all of which move tier decisions and spend on an already-deployed router. Requires an LLM, Jev, or custom classifier plugin, since the heuristic scorers cannot produce the tier, and a model in `tiers` under the NON_REASONING key. Escalation still walks up from it, and it is never the savings baseline or a `heuristic_v2` prediction. */
|
|
3539
|
+
enable_non_reasoning_tier?: boolean;
|
|
3024
3540
|
/** Case-sensitive phrases a user can include to force a bump to the next-higher complexity tier when they aren't satisfied with results (they can force a stronger model, but not choose which one). Defaults to ['LITELLM ESCALATE'] when unset; set to an empty list to disable. */
|
|
3025
3541
|
escalation_keywords?: RequestComplexityRouterConfigEscalationKeywordsList | null;
|
|
3026
3542
|
/** Tier routed to when the LLM classifier fails (timeout, provider error, or an unparseable reply). Required with tier_definitions and must name a defined tier; the heuristic scorer cannot produce custom tiers, so this replaces the heuristic fallback for custom tier sets. */
|
|
3027
3543
|
fallback_tier?: string | null;
|
|
3028
3544
|
/** The highest tier the local scorer may decide on its own; required when classifier_type is 'heuristic_first' and rejected otherwise. A request whose heuristic tier is at or below this one skips the LLM classifier and routes straight to that heuristic tier, so the classifier call is only paid for on traffic the scorer could not place cheaply. The scorer must also have produced at least one signal: a prompt where no dimension fired scores 0.0 and would otherwise land SIMPLE by default rather than by evidence, which is how a chained router would silently send unclassified traffic to the cheapest model. Names a built-in tier, and may not name the highest one, since that would make the LLM classifier unreachable. */
|
|
3029
3545
|
heuristic_first_max_tier?: string | null;
|
|
3546
|
+
/** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
|
|
3547
|
+
heuristic_v2_artifact?: RequestComplexityRouterConfigHeuristicV2Artifact;
|
|
3030
3548
|
/** Additional case-sensitive literal sentinels that mark a request as client housekeeping, on top of the built-in conversation-title ones. For clients whose wording the built-ins don't cover, or after a client release changes its strings. */
|
|
3031
3549
|
housekeeping_patterns?: RequestComplexityRouterConfigHousekeepingPatternsList | null;
|
|
3550
|
+
/** How close to a tier boundary a heuristic score has to land before the LLM classifier breaks the tie; required when classifier_type is 'hybrid' and rejected otherwise. Everything further than this from every active boundary routes on the scorer's own tier with no classifier call, at any tier, which is what separates 'hybrid' from 'heuristic_first' and its cheap-tier ceiling. A prompt where no dimension fired still goes to the classifier, since the scorer has no opinion to be near a boundary with. 0 escalates only scores sitting exactly on a boundary. */
|
|
3551
|
+
hybrid_boundary_margin?: number | null;
|
|
3552
|
+
jev_classifier_config?: JevClassifierConfig | null;
|
|
3032
3553
|
/** Rules that force a specific tier when their keywords match the prompt */
|
|
3033
3554
|
keyword_tier_rules?: RequestComplexityRouterConfigKeywordTierRulesList | null;
|
|
3555
|
+
/** Experimental joint task-demand and solver-capability forecasting for classifier_type llm_v2. */
|
|
3556
|
+
llm_v2_config?: LLMV2Config | null;
|
|
3034
3557
|
/** Minimum cosine similarity for a semantic keyword match */
|
|
3035
3558
|
match_threshold?: number;
|
|
3559
|
+
/** Set max_tokens on every routed request to the output ceiling of the tier model it lands on, replacing whatever the caller sent. A caller behind an auto-router cannot pick one value that fits every tier: the smallest tier's ceiling starves a bigger tier's thinking budget, and a bigger tier's ceiling is rejected by the smallest. The ceiling is the smallest max_output_tokens across the tier model's deployments, read from each deployment's model_info and then the model cost map; a tier model with a deployment whose ceiling is unknown keeps the caller's value. A max_tokens, max_completion_tokens or max_output_tokens in the tier's own litellm_params still wins. Set false to forward the caller's value unchanged. */
|
|
3560
|
+
max_tokens_from_tier_model?: boolean;
|
|
3561
|
+
/** Let modality_routing replace a kept session-affinity pin on the turns that carry an image. Without this, a session pinned to a text-only model fails every image turn with a provider 400, since the pin is exempt from the modality gate. When enabled, such a turn routes to a capable model for that request only and the stored pin is left untouched, so the next text turn replays the session's own model; the override is reported as cause modality_pin_override and is never itself pinned. Inert unless modality_routing is also enabled. */
|
|
3562
|
+
modality_pin_override?: boolean;
|
|
3563
|
+
/** Route image-bearing requests only to models that can accept image input. The classifier reads text alone, so an image request whose text classifies cheap otherwise lands on a text-only model and fails with a provider 400. When enabled, a routed model explicitly declared supports_vision false (deployment model_info or the model cost map; unmapped names stay routable) is replaced by the nearest HIGHER tier holding a capable model, then default_model, else a clear 400. A kept session-affinity pin still wins even when an image arrives, unless modality_pin_override is also enabled. */
|
|
3564
|
+
modality_routing?: boolean;
|
|
3036
3565
|
/** When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to at least this tier: the classified tier still wins when it is higher, and the floor also overrides a session-affinity pin to a lower tier for exactly the turns carrying the sentinel, without rewriting the pin -- the first turn after plan mode exits routes as if plan mode had never happened. Names a built-in tier, or with tier_definitions set, one of the defined tier names (list order is ascending severity, same as keyword_tier_rules). Unset disables detection entirely. The sentinels ride in client-injected prompt text, so a caller who pastes one can spend up to this tier's models -- never down, and never outside the configured pools. */
|
|
3037
3566
|
plan_mode_min_tier?: string | null;
|
|
3038
3567
|
/** Additional case-sensitive literal sentinels that mark a request as plan mode, on top of the built-in Claude Code and Copilot ones. For clients whose plan-mode wording the built-ins don't cover, or after a client release changes its strings. */
|
|
@@ -3043,7 +3572,7 @@ export interface RequestComplexityRouterConfig {
|
|
|
3043
3572
|
reasoning_keywords?: RequestComplexityRouterConfigReasoningKeywordsList | null;
|
|
3044
3573
|
/** Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a request the scorer placed in the cheapest tier; 0 restores the unconditional override */
|
|
3045
3574
|
reasoning_override_min_score?: number | null;
|
|
3046
|
-
/** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in
|
|
3575
|
+
/** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in system-reminder pair and the Codex envelope pairs enabled for Codex user agents, so list every built-in pair your harness also emits. Matching is case-insensitive. */
|
|
3047
3576
|
reminder_markers?: RequestComplexityRouterConfigReminderMarkersList | null;
|
|
3048
3577
|
/** Return the resolved raw model name in the response model field instead of the client-requested complexity-router alias */
|
|
3049
3578
|
return_raw_model_name?: boolean;
|
|
@@ -3053,15 +3582,21 @@ export interface RequestComplexityRouterConfig {
|
|
|
3053
3582
|
semantic_keyword_matching?: boolean;
|
|
3054
3583
|
/** When True and a session_id is resolvable on the request, pin the model chosen on the session's first turn and reuse it for every later turn, skipping re-classification. Off by default so every turn is classified on its own merits and routed to the cheapest adequate tier. Set True to keep a multi-turn session on one model, which preserves provider prompt caches and avoids cross-model conversation-history errors. Always implies the deployment pin regardless of deployment_affinity: the session sticks to one deployment of the pinned model, since freezing the model while re-shuffling its deployments would still go cache-cold. */
|
|
3055
3584
|
session_affinity?: boolean;
|
|
3056
|
-
/** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity deployment
|
|
3585
|
+
/** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity per-tier model and deployment pins, so it measures idle time for the session's routing decisions rather than total session length */
|
|
3057
3586
|
session_affinity_ttl_seconds?: number;
|
|
3058
3587
|
/** Keywords indicating simple/basic queries */
|
|
3059
3588
|
simple_keywords?: RequestComplexityRouterConfigSimpleKeywordsList | null;
|
|
3589
|
+
/** Escalate mid-task to the next-higher configured tier when the assistant's own recent tool calls look stuck: the newest tool call repeats, or errors, at least stall_escalation_repeat_threshold times across the last stall_escalation_window calls. Both tests are anchored on the newest call, so a task that tried the same thing a few times and then moved on is not escalated on the strength of those older calls alone, while a retry loop broken up by an unrelated lookup still counts. One tier at most, on the same ladder escalation_keywords bumps along, and never above the highest configured tier. Detection re-runs on every classified turn from the tool calls visible in that request, so it needs no state and nothing survives past the task. Mutually exclusive with session_affinity and classification_mode='user_turn', which both replay a held routing decision instead of classifying most turns, so this would never see the tool calls to look at. Off by default. */
|
|
3590
|
+
stall_escalation_enabled?: boolean;
|
|
3591
|
+
/** How many of the last stall_escalation_window tool calls must repeat the newest call, or must have errored alongside it, before the task counts as stalled. Must not exceed stall_escalation_window, or the condition could never be reached. */
|
|
3592
|
+
stall_escalation_repeat_threshold?: number;
|
|
3593
|
+
/** How many of the assistant's most recent tool calls stall detection looks at, oldest ones dropped as new calls happen. Counted across the whole visible conversation rather than reset at the newest human ask, so evidence from before a plain follow-up message like 'try again' is still visible on the turn after it. */
|
|
3594
|
+
stall_escalation_window?: number;
|
|
3060
3595
|
/** Keywords indicating technical content */
|
|
3061
3596
|
technical_keywords?: RequestComplexityRouterConfigTechnicalKeywordsList | null;
|
|
3062
3597
|
/** Score boundaries between tiers. These keys (simple_medium, medium_complex, complex_reasoning) name the gaps between the default tier names and are not renameable by tier_labels; they are scorer knobs persisted by name on every routing decision */
|
|
3063
3598
|
tier_boundaries?: RequestComplexityRouterConfigTierBoundariesMap;
|
|
3064
|
-
/** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
|
|
3599
|
+
/** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm', 'jev' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
|
|
3065
3600
|
tier_definitions?: RequestComplexityRouterConfigTierDefinitionsList | null;
|
|
3066
3601
|
/** Score penalty per tier-step away from the classified tier when adaptive=True */
|
|
3067
3602
|
tier_distance_penalty?: number;
|
|
@@ -3080,6 +3615,13 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
|
|
|
3080
3615
|
RequestComplexityRouterConfigAdaptiveEligible,
|
|
3081
3616
|
),
|
|
3082
3617
|
adaptive_weights: S.optional(AdaptiveRouterWeights),
|
|
3618
|
+
capability_classifier_config: S.optional(
|
|
3619
|
+
S.NullOr(CapabilityClassifierConfig),
|
|
3620
|
+
),
|
|
3621
|
+
classification_examples: S.optional(S.NullOr(S.String)),
|
|
3622
|
+
classification_mode: S.optional(
|
|
3623
|
+
RequestComplexityRouterConfigClassificationMode,
|
|
3624
|
+
),
|
|
3083
3625
|
classification_prompt: S.optional(S.NullOr(S.String)),
|
|
3084
3626
|
classifier_context_budget_chars: S.optional(S.Number),
|
|
3085
3627
|
classifier_context_include_assistant_turns: S.optional(S.Boolean),
|
|
@@ -3095,6 +3637,10 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
|
|
|
3095
3637
|
code_keywords: S.optional(
|
|
3096
3638
|
S.NullOr(RequestComplexityRouterConfigCodeKeywordsList),
|
|
3097
3639
|
),
|
|
3640
|
+
context_window_escalation_buffer: S.optional(S.Number),
|
|
3641
|
+
custom_dimensions: S.optional(
|
|
3642
|
+
RequestComplexityRouterConfigCustomDimensionsList,
|
|
3643
|
+
),
|
|
3098
3644
|
custom_technical_keywords: S.optional(
|
|
3099
3645
|
S.NullOr(RequestComplexityRouterConfigCustomTechnicalKeywordsList),
|
|
3100
3646
|
),
|
|
@@ -3104,18 +3650,29 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
|
|
|
3104
3650
|
RequestComplexityRouterConfigDimensionWeightsMap,
|
|
3105
3651
|
),
|
|
3106
3652
|
embedding_model: S.optional(S.NullOr(S.String)),
|
|
3653
|
+
enable_context_window_escalation: S.optional(S.Boolean),
|
|
3654
|
+
enable_non_reasoning_tier: S.optional(S.Boolean),
|
|
3107
3655
|
escalation_keywords: S.optional(
|
|
3108
3656
|
S.NullOr(RequestComplexityRouterConfigEscalationKeywordsList),
|
|
3109
3657
|
),
|
|
3110
3658
|
fallback_tier: S.optional(S.NullOr(S.String)),
|
|
3111
3659
|
heuristic_first_max_tier: S.optional(S.NullOr(S.String)),
|
|
3660
|
+
heuristic_v2_artifact: S.optional(
|
|
3661
|
+
RequestComplexityRouterConfigHeuristicV2Artifact,
|
|
3662
|
+
),
|
|
3112
3663
|
housekeeping_patterns: S.optional(
|
|
3113
3664
|
S.NullOr(RequestComplexityRouterConfigHousekeepingPatternsList),
|
|
3114
3665
|
),
|
|
3666
|
+
hybrid_boundary_margin: S.optional(S.NullOr(S.Number)),
|
|
3667
|
+
jev_classifier_config: S.optional(S.NullOr(JevClassifierConfig)),
|
|
3115
3668
|
keyword_tier_rules: S.optional(
|
|
3116
3669
|
S.NullOr(RequestComplexityRouterConfigKeywordTierRulesList),
|
|
3117
3670
|
),
|
|
3671
|
+
llm_v2_config: S.optional(S.NullOr(LLMV2Config)),
|
|
3118
3672
|
match_threshold: S.optional(S.Number),
|
|
3673
|
+
max_tokens_from_tier_model: S.optional(S.Boolean),
|
|
3674
|
+
modality_pin_override: S.optional(S.Boolean),
|
|
3675
|
+
modality_routing: S.optional(S.Boolean),
|
|
3119
3676
|
plan_mode_min_tier: S.optional(S.NullOr(S.String)),
|
|
3120
3677
|
plan_mode_patterns: S.optional(
|
|
3121
3678
|
S.NullOr(RequestComplexityRouterConfigPlanModePatternsList),
|
|
@@ -3136,6 +3693,9 @@ export const RequestComplexityRouterConfig = /*@__PURE__*/ S.suspend(() =>
|
|
|
3136
3693
|
simple_keywords: S.optional(
|
|
3137
3694
|
S.NullOr(RequestComplexityRouterConfigSimpleKeywordsList),
|
|
3138
3695
|
),
|
|
3696
|
+
stall_escalation_enabled: S.optional(S.Boolean),
|
|
3697
|
+
stall_escalation_repeat_threshold: S.optional(S.Number),
|
|
3698
|
+
stall_escalation_window: S.optional(S.Number),
|
|
3139
3699
|
technical_keywords: S.optional(
|
|
3140
3700
|
S.NullOr(RequestComplexityRouterConfigTechnicalKeywordsList),
|
|
3141
3701
|
),
|
|
@@ -3259,24 +3819,71 @@ export const PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest =
|
|
|
3259
3819
|
|
|
3260
3820
|
export type StandardLoggingRoutingDecisionCause =
|
|
3261
3821
|
| "heuristic_scorer"
|
|
3822
|
+
| "heuristic_v2"
|
|
3262
3823
|
| "reasoning_override"
|
|
3263
3824
|
| "llm_classifier"
|
|
3825
|
+
| "capability_classifier"
|
|
3826
|
+
| "jev_classifier"
|
|
3827
|
+
| "llm_v2_classifier"
|
|
3828
|
+
| "llm_v2_fallback"
|
|
3264
3829
|
| "heuristic_first_short_circuit"
|
|
3830
|
+
| "hybrid_short_circuit"
|
|
3265
3831
|
| "classifier_plugin"
|
|
3266
3832
|
| "classifier_fallback"
|
|
3833
|
+
| "capability_classifier_fallback"
|
|
3267
3834
|
| "default_model_fallback"
|
|
3268
3835
|
| "literal_keyword_match"
|
|
3269
3836
|
| "semantic_keyword_match"
|
|
3270
3837
|
| "plan_mode"
|
|
3271
3838
|
| "housekeeping"
|
|
3839
|
+
| "modality_escalation"
|
|
3840
|
+
| "modality_pin_override"
|
|
3841
|
+
| "health_failover"
|
|
3842
|
+
| "health_default_fallback"
|
|
3272
3843
|
| "session_affinity_pin"
|
|
3273
3844
|
| "session_affinity_escalation"
|
|
3845
|
+
| "user_turn_continuation"
|
|
3274
3846
|
| "default_fallback"
|
|
3275
3847
|
| "keyword"
|
|
3276
3848
|
| "quality_tier"
|
|
3277
3849
|
| "bandit";
|
|
3278
3850
|
export const StandardLoggingRoutingDecisionCause = S.String;
|
|
3279
3851
|
|
|
3852
|
+
export type StandardLoggingRoutingDecisionClassifierProbabilitiesMap = {
|
|
3853
|
+
[key: string]: number | undefined;
|
|
3854
|
+
};
|
|
3855
|
+
export const StandardLoggingRoutingDecisionClassifierProbabilitiesMap =
|
|
3856
|
+
/*@__PURE__*/ S.Record(
|
|
3857
|
+
S.String,
|
|
3858
|
+
S.Number,
|
|
3859
|
+
) as any as S.Schema<StandardLoggingRoutingDecisionClassifierProbabilitiesMap>;
|
|
3860
|
+
|
|
3861
|
+
export type StandardLoggingHeuristicV2ForecastProbabilitiesMap = {
|
|
3862
|
+
[key: string]: number | undefined;
|
|
3863
|
+
};
|
|
3864
|
+
export const StandardLoggingHeuristicV2ForecastProbabilitiesMap =
|
|
3865
|
+
/*@__PURE__*/ S.Record(
|
|
3866
|
+
S.String,
|
|
3867
|
+
S.Number,
|
|
3868
|
+
) as any as S.Schema<StandardLoggingHeuristicV2ForecastProbabilitiesMap>;
|
|
3869
|
+
|
|
3870
|
+
export interface StandardLoggingHeuristicV2Forecast {
|
|
3871
|
+
predicted_tier: string;
|
|
3872
|
+
probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap;
|
|
3873
|
+
request_type: string;
|
|
3874
|
+
threshold: number;
|
|
3875
|
+
}
|
|
3876
|
+
export const StandardLoggingHeuristicV2Forecast = /*@__PURE__*/ S.suspend(() =>
|
|
3877
|
+
S.Struct({
|
|
3878
|
+
predicted_tier: S.String,
|
|
3879
|
+
probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap,
|
|
3880
|
+
request_type: S.String,
|
|
3881
|
+
threshold: S.Number,
|
|
3882
|
+
}),
|
|
3883
|
+
).annotate({
|
|
3884
|
+
identifier: "StandardLoggingHeuristicV2Forecast",
|
|
3885
|
+
}) as any as S.Schema<StandardLoggingHeuristicV2Forecast>;
|
|
3886
|
+
|
|
3280
3887
|
export type StandardLoggingRoutingDecisionRouterType =
|
|
3281
3888
|
| "complexity"
|
|
3282
3889
|
| "adaptive"
|
|
@@ -3317,11 +3924,29 @@ export const StandardLoggingRoutingDecisionTierLitellmParamsMap =
|
|
|
3317
3924
|
/** Per-request provenance for a pre-routing strategy (auto-router) decision. */
|
|
3318
3925
|
export interface StandardLoggingRoutingDecision {
|
|
3319
3926
|
cause?: StandardLoggingRoutingDecisionCause;
|
|
3927
|
+
classifier_calibrated_capable_p_solve?: number;
|
|
3928
|
+
classifier_calibrated_efficient_p_solve?: number;
|
|
3929
|
+
classifier_calibrated_p_solve?: number;
|
|
3930
|
+
classifier_calibration_version?: string;
|
|
3931
|
+
classifier_capability_boundary?: string;
|
|
3932
|
+
classifier_capable_p_solve?: number;
|
|
3933
|
+
classifier_confidence?: number;
|
|
3320
3934
|
classifier_cost?: number;
|
|
3935
|
+
classifier_crux?: string;
|
|
3936
|
+
classifier_efficient_p_solve?: number;
|
|
3937
|
+
classifier_max_quality_gap?: number;
|
|
3321
3938
|
classifier_model?: string;
|
|
3939
|
+
classifier_p_solve?: number;
|
|
3940
|
+
classifier_primary_rule?: string;
|
|
3941
|
+
classifier_probabilities?: StandardLoggingRoutingDecisionClassifierProbabilitiesMap;
|
|
3942
|
+
classifier_prompt_version?: string;
|
|
3943
|
+
classifier_threshold?: number;
|
|
3944
|
+
context_escalated?: boolean;
|
|
3945
|
+
context_escalation_original_tier?: string;
|
|
3322
3946
|
conversation_continuing?: boolean;
|
|
3323
3947
|
escalated?: boolean;
|
|
3324
3948
|
escalation_keyword?: string;
|
|
3949
|
+
heuristic_v2_forecast?: StandardLoggingHeuristicV2Forecast;
|
|
3325
3950
|
matched_keyword?: string;
|
|
3326
3951
|
reasoning_override_min_score?: number;
|
|
3327
3952
|
request_type?: string;
|
|
@@ -3340,11 +3965,31 @@ export interface StandardLoggingRoutingDecision {
|
|
|
3340
3965
|
export const StandardLoggingRoutingDecision = /*@__PURE__*/ S.suspend(() =>
|
|
3341
3966
|
S.Struct({
|
|
3342
3967
|
cause: S.optional(StandardLoggingRoutingDecisionCause),
|
|
3968
|
+
classifier_calibrated_capable_p_solve: S.optional(S.Number),
|
|
3969
|
+
classifier_calibrated_efficient_p_solve: S.optional(S.Number),
|
|
3970
|
+
classifier_calibrated_p_solve: S.optional(S.Number),
|
|
3971
|
+
classifier_calibration_version: S.optional(S.String),
|
|
3972
|
+
classifier_capability_boundary: S.optional(S.String),
|
|
3973
|
+
classifier_capable_p_solve: S.optional(S.Number),
|
|
3974
|
+
classifier_confidence: S.optional(S.Number),
|
|
3343
3975
|
classifier_cost: S.optional(S.Number),
|
|
3976
|
+
classifier_crux: S.optional(S.String),
|
|
3977
|
+
classifier_efficient_p_solve: S.optional(S.Number),
|
|
3978
|
+
classifier_max_quality_gap: S.optional(S.Number),
|
|
3344
3979
|
classifier_model: S.optional(S.String),
|
|
3980
|
+
classifier_p_solve: S.optional(S.Number),
|
|
3981
|
+
classifier_primary_rule: S.optional(S.String),
|
|
3982
|
+
classifier_probabilities: S.optional(
|
|
3983
|
+
StandardLoggingRoutingDecisionClassifierProbabilitiesMap,
|
|
3984
|
+
),
|
|
3985
|
+
classifier_prompt_version: S.optional(S.String),
|
|
3986
|
+
classifier_threshold: S.optional(S.Number),
|
|
3987
|
+
context_escalated: S.optional(S.Boolean),
|
|
3988
|
+
context_escalation_original_tier: S.optional(S.String),
|
|
3345
3989
|
conversation_continuing: S.optional(S.Boolean),
|
|
3346
3990
|
escalated: S.optional(S.Boolean),
|
|
3347
3991
|
escalation_keyword: S.optional(S.String),
|
|
3992
|
+
heuristic_v2_forecast: S.optional(StandardLoggingHeuristicV2Forecast),
|
|
3348
3993
|
matched_keyword: S.optional(S.String),
|
|
3349
3994
|
reasoning_override_min_score: S.optional(S.Number),
|
|
3350
3995
|
request_type: S.optional(S.String),
|
|
@@ -3964,7 +4609,7 @@ export const getModelCostMapReloadStatusScheduleModelCostMapReloadStatusGet: API
|
|
|
3964
4609
|
}));
|
|
3965
4610
|
|
|
3966
4611
|
export type GetModelCostMapSourceModelCostMapSourceGetError = LitellmOpError;
|
|
3967
|
-
/** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - model_count: number of models in the currently loaded cost map */
|
|
4612
|
+
/** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - loaded_at: when this pod last loaded the map - source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it - etag: the ETag of the remote fetch (null for the bundled backup) - model_count: number of models in the currently loaded cost map */
|
|
3968
4613
|
export const getModelCostMapSourceModelCostMapSourceGet: API.OperationMethod<
|
|
3969
4614
|
GetModelCostMapSourceModelCostMapSourceGetRequest,
|
|
3970
4615
|
GetModelCostMapSourceModelCostMapSourceGetResponse,
|
|
@@ -4047,7 +4692,7 @@ export const getModelInfoModelsModelId: API.OperationMethod<
|
|
|
4047
4692
|
}));
|
|
4048
4693
|
|
|
4049
4694
|
export type GetModelInfoV1ModelInfoError = UnprocessableEntity | LitellmOpError;
|
|
4050
|
-
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns:
|
|
4695
|
+
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
|
|
4051
4696
|
export const getModelInfoV1ModelInfo: API.OperationMethod<
|
|
4052
4697
|
GetModelInfoV1ModelInfoRequest,
|
|
4053
4698
|
GetModelInfoV1ModelInfoResponse,
|
|
@@ -4081,7 +4726,7 @@ export const getModelInfoV1ModelsModelId: API.OperationMethod<
|
|
|
4081
4726
|
export type GetModelInfoV1V1ModelInfoError =
|
|
4082
4727
|
| UnprocessableEntity
|
|
4083
4728
|
| LitellmOpError;
|
|
4084
|
-
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns:
|
|
4729
|
+
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
|
|
4085
4730
|
export const getModelInfoV1V1ModelInfo: API.OperationMethod<
|
|
4086
4731
|
GetModelInfoV1V1ModelInfoRequest,
|
|
4087
4732
|
GetModelInfoV1V1ModelInfoResponse,
|
|
@@ -4098,7 +4743,7 @@ export const getModelInfoV1V1ModelInfo: API.OperationMethod<
|
|
|
4098
4743
|
export type GetModelInfoV2V2ModelInfoError =
|
|
4099
4744
|
| UnprocessableEntity
|
|
4100
4745
|
| LitellmOpError;
|
|
4101
|
-
/** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
|
|
4746
|
+
/** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. access_group: Only return deployments in this model access group. wildcard_only: Only return deployments whose `model_name` contains `*`. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
|
|
4102
4747
|
export const getModelInfoV2V2ModelInfo: API.OperationMethod<
|
|
4103
4748
|
GetModelInfoV2V2ModelInfoRequest,
|
|
4104
4749
|
GetModelInfoV2V2ModelInfoResponse,
|
|
@@ -4485,7 +5130,7 @@ export const updateUsefulLinksModelHubUpdateUsefulLinksPost: API.OperationMethod
|
|
|
4485
5130
|
export type ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError =
|
|
4486
5131
|
| UnprocessableEntity
|
|
4487
5132
|
| LitellmOpError;
|
|
4488
|
-
/** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save.
|
|
5133
|
+
/** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Uses the same team opt-in and model-access checks as configuration writes for members. Nothing is created, routed, or billed. */
|
|
4489
5134
|
export const validateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPost: API.OperationMethod<
|
|
4490
5135
|
ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostRequest,
|
|
4491
5136
|
ComplexityRouterConfigValidationResponse,
|