@homeflare/distilled-litellm 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -12
- package/dist/services/a2a.js +1 -1
- package/dist/services/a2a_registration.js +1 -1
- package/dist/services/access_groups.d.ts +18 -0
- package/dist/services/access_groups.d.ts.map +1 -1
- package/dist/services/access_groups.js +15 -1
- package/dist/services/access_groups.js.map +1 -1
- package/dist/services/adaptive_router.js +1 -1
- package/dist/services/agents.d.ts +10 -0
- package/dist/services/agents.d.ts.map +1 -1
- package/dist/services/agents.js +10 -1
- package/dist/services/agents.js.map +1 -1
- package/dist/services/alerting.js +1 -1
- package/dist/services/anthropic_passthrough.js +1 -1
- package/dist/services/anthropic_skills.d.ts +7 -1
- package/dist/services/anthropic_skills.d.ts.map +1 -1
- package/dist/services/anthropic_skills.js +6 -2
- package/dist/services/anthropic_skills.js.map +1 -1
- package/dist/services/assistants.js +1 -1
- package/dist/services/audio.js +1 -1
- package/dist/services/audit_logging.d.ts +2 -0
- package/dist/services/audit_logging.d.ts.map +1 -1
- package/dist/services/audit_logging.js +2 -1
- package/dist/services/audit_logging.js.map +1 -1
- package/dist/services/auto_router.d.ts +162 -62
- package/dist/services/auto_router.d.ts.map +1 -1
- package/dist/services/auto_router.js +92 -31
- package/dist/services/auto_router.js.map +1 -1
- package/dist/services/batch.js +1 -1
- package/dist/services/beta_agents.js +1 -1
- package/dist/services/beta_mcp.js +1 -1
- package/dist/services/budget_management.d.ts +8 -3
- package/dist/services/budget_management.d.ts.map +1 -1
- package/dist/services/budget_management.js +7 -4
- package/dist/services/budget_management.js.map +1 -1
- package/dist/services/budget_spend_tracking.d.ts +24 -5
- package/dist/services/budget_spend_tracking.d.ts.map +1 -1
- package/dist/services/budget_spend_tracking.js +17 -4
- package/dist/services/budget_spend_tracking.js.map +1 -1
- package/dist/services/cache_settings.js +1 -1
- package/dist/services/caching.js +1 -1
- package/dist/services/chat_completions.js +1 -1
- package/dist/services/claude_code_marketplace.d.ts +11 -10
- package/dist/services/claude_code_marketplace.d.ts.map +1 -1
- package/dist/services/claude_code_marketplace.js +8 -6
- package/dist/services/claude_code_marketplace.js.map +1 -1
- package/dist/services/cloudzero.js +1 -1
- package/dist/services/completions.js +1 -1
- package/dist/services/compliance.js +1 -1
- package/dist/services/config_overrides.d.ts +5 -1
- package/dist/services/config_overrides.d.ts.map +1 -1
- package/dist/services/config_overrides.js +3 -1
- package/dist/services/config_overrides.js.map +1 -1
- package/dist/services/config_yaml.d.ts +120 -6
- package/dist/services/config_yaml.d.ts.map +1 -1
- package/dist/services/config_yaml.js +67 -1
- package/dist/services/config_yaml.js.map +1 -1
- package/dist/services/containers.d.ts +8 -2
- package/dist/services/containers.d.ts.map +1 -1
- package/dist/services/containers.js +13 -5
- package/dist/services/containers.js.map +1 -1
- package/dist/services/coordination_redis_settings.js +1 -1
- package/dist/services/cost_tracking.d.ts +91 -1
- package/dist/services/cost_tracking.d.ts.map +1 -1
- package/dist/services/cost_tracking.js +82 -2
- package/dist/services/cost_tracking.js.map +1 -1
- package/dist/services/credential_management.d.ts +2 -1
- package/dist/services/credential_management.d.ts.map +1 -1
- package/dist/services/credential_management.js +3 -2
- package/dist/services/credential_management.js.map +1 -1
- package/dist/services/customer_management.d.ts +25 -2
- package/dist/services/customer_management.d.ts.map +1 -1
- package/dist/services/customer_management.js +22 -3
- package/dist/services/customer_management.js.map +1 -1
- package/dist/services/email_management.js +1 -1
- package/dist/services/embeddings.js +1 -1
- package/dist/services/evals.js +1 -1
- package/dist/services/experimental.js +1 -1
- package/dist/services/fallback_management.js +1 -1
- package/dist/services/files.js +1 -1
- package/dist/services/fine_tuning.js +1 -1
- package/dist/services/gemini_agents.js +1 -1
- package/dist/services/google_genai_endpoints.js +1 -1
- package/dist/services/guardrails.d.ts +80 -7
- package/dist/services/guardrails.d.ts.map +1 -1
- package/dist/services/guardrails.js +37 -1
- package/dist/services/guardrails.js.map +1 -1
- package/dist/services/health.d.ts +2 -2
- package/dist/services/health.d.ts.map +1 -1
- package/dist/services/health.js +2 -2
- package/dist/services/health.js.map +1 -1
- package/dist/services/images.js +1 -1
- package/dist/services/index.d.ts +1 -15
- package/dist/services/index.d.ts.map +1 -1
- package/dist/services/index.js +1 -15
- package/dist/services/index.js.map +1 -1
- package/dist/services/internal_user_management.d.ts +227 -39
- package/dist/services/internal_user_management.d.ts.map +1 -1
- package/dist/services/internal_user_management.js +194 -37
- package/dist/services/internal_user_management.js.map +1 -1
- package/dist/services/invite_links.js +1 -1
- package/dist/services/jwt_mappings.d.ts +3 -0
- package/dist/services/jwt_mappings.d.ts.map +1 -1
- package/dist/services/jwt_mappings.js +4 -1
- package/dist/services/jwt_mappings.js.map +1 -1
- package/dist/services/key_management.d.ts +93 -49
- package/dist/services/key_management.d.ts.map +1 -1
- package/dist/services/key_management.js +75 -39
- package/dist/services/key_management.js.map +1 -1
- package/dist/services/langfuse_passthrough.js +1 -1
- package/dist/services/llm_passthrough.d.ts +1199 -0
- package/dist/services/llm_passthrough.d.ts.map +1 -0
- package/dist/services/llm_passthrough.js +2150 -0
- package/dist/services/llm_passthrough.js.map +1 -0
- package/dist/services/llm_utils.js +1 -1
- package/dist/services/logging_callbacks.js +1 -1
- package/dist/services/mcp_byok_oauth.d.ts +18 -11
- package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
- package/dist/services/mcp_byok_oauth.js +27 -24
- package/dist/services/mcp_byok_oauth.js.map +1 -1
- package/dist/services/mcp_discoverable.d.ts +34 -0
- package/dist/services/mcp_discoverable.d.ts.map +1 -1
- package/dist/services/mcp_discoverable.js +51 -1
- package/dist/services/mcp_discoverable.js.map +1 -1
- package/dist/services/mcp_management.d.ts +90 -2
- package/dist/services/mcp_management.d.ts.map +1 -1
- package/dist/services/mcp_management.js +115 -3
- package/dist/services/mcp_management.js.map +1 -1
- package/dist/services/mcp_rest.d.ts +13 -0
- package/dist/services/mcp_rest.d.ts.map +1 -1
- package/dist/services/mcp_rest.js +10 -2
- package/dist/services/mcp_rest.js.map +1 -1
- package/dist/services/memory_management.d.ts +2 -0
- package/dist/services/memory_management.d.ts.map +1 -1
- package/dist/services/memory_management.js +2 -1
- package/dist/services/memory_management.js.map +1 -1
- package/dist/services/misc.d.ts +86 -1
- package/dist/services/misc.d.ts.map +1 -1
- package/dist/services/misc.js +126 -2
- package/dist/services/misc.js.map +1 -1
- package/dist/services/model_management.d.ts +310 -19
- package/dist/services/model_management.d.ts.map +1 -1
- package/dist/services/model_management.js +240 -7
- package/dist/services/model_management.js.map +1 -1
- package/dist/services/moderations.js +1 -1
- package/dist/services/ocr.js +1 -1
- package/dist/services/open_ai_pass_through.d.ts +35 -90
- package/dist/services/open_ai_pass_through.d.ts.map +1 -1
- package/dist/services/open_ai_pass_through.js +41 -139
- package/dist/services/open_ai_pass_through.js.map +1 -1
- package/dist/services/organization_management.d.ts +21 -1
- package/dist/services/organization_management.d.ts.map +1 -1
- package/dist/services/organization_management.js +20 -2
- package/dist/services/organization_management.js.map +1 -1
- package/dist/services/plugins.js +1 -1
- package/dist/services/policies.d.ts +2 -0
- package/dist/services/policies.d.ts.map +1 -1
- package/dist/services/policies.js +3 -1
- package/dist/services/policies.js.map +1 -1
- package/dist/services/policy_engine.d.ts +6 -0
- package/dist/services/policy_engine.d.ts.map +1 -1
- package/dist/services/policy_engine.js +4 -1
- package/dist/services/policy_engine.js.map +1 -1
- package/dist/services/project_management.d.ts +15 -0
- package/dist/services/project_management.d.ts.map +1 -1
- package/dist/services/project_management.js +14 -1
- package/dist/services/project_management.js.map +1 -1
- package/dist/services/prompts.js +1 -1
- package/dist/services/public.d.ts +124 -0
- package/dist/services/public.d.ts.map +1 -1
- package/dist/services/public.js +158 -1
- package/dist/services/public.js.map +1 -1
- package/dist/services/rag.js +1 -1
- package/dist/services/realtime.d.ts +4 -26
- package/dist/services/realtime.d.ts.map +1 -1
- package/dist/services/realtime.js +7 -57
- package/dist/services/realtime.js.map +1 -1
- package/dist/services/rerank.d.ts +72 -0
- package/dist/services/rerank.d.ts.map +1 -1
- package/dist/services/rerank.js +61 -4
- package/dist/services/rerank.js.map +1 -1
- package/dist/services/responses.d.ts +30 -0
- package/dist/services/responses.d.ts.map +1 -1
- package/dist/services/responses.js +59 -1
- package/dist/services/responses.js.map +1 -1
- package/dist/services/router_settings.js +1 -1
- package/dist/services/rust_control_plane.js +1 -1
- package/dist/services/scim.d.ts +41 -1
- package/dist/services/scim.d.ts.map +1 -1
- package/dist/services/scim.js +60 -2
- package/dist/services/scim.js.map +1 -1
- package/dist/services/search.js +1 -1
- package/dist/services/search_tools.js +1 -1
- package/dist/services/settings.d.ts +88 -0
- package/dist/services/settings.d.ts.map +1 -1
- package/dist/services/settings.js +113 -1
- package/dist/services/settings.js.map +1 -1
- package/dist/services/sso_settings.d.ts +1 -1
- package/dist/services/sso_settings.d.ts.map +1 -1
- package/dist/services/sso_settings.js +1 -1
- package/dist/services/sso_settings.js.map +1 -1
- package/dist/services/tag_management.d.ts +7 -0
- package/dist/services/tag_management.d.ts.map +1 -1
- package/dist/services/tag_management.js +8 -1
- package/dist/services/tag_management.js.map +1 -1
- package/dist/services/team_management.d.ts +265 -17
- package/dist/services/team_management.d.ts.map +1 -1
- package/dist/services/team_management.js +247 -12
- package/dist/services/team_management.js.map +1 -1
- package/dist/services/tools.js +1 -1
- package/dist/services/ui_settings.js +1 -1
- package/dist/services/ui_theme_settings.js +1 -1
- package/dist/services/usage_ai.js +1 -1
- package/dist/services/vantage.js +1 -1
- package/dist/services/vector_store_management.js +1 -1
- package/dist/services/vector_stores.js +1 -1
- package/dist/services/videos.js +1 -1
- package/dist/services/web_socket.d.ts +0 -18
- package/dist/services/web_socket.d.ts.map +1 -1
- package/dist/services/web_socket.js +1 -33
- package/dist/services/web_socket.js.map +1 -1
- package/dist/services/workflow_management.js +1 -1
- package/package.json +1 -1
- package/src/services/a2a.ts +1 -1
- package/src/services/a2a_registration.ts +1 -1
- package/src/services/access_groups.ts +44 -1
- package/src/services/adaptive_router.ts +1 -1
- package/src/services/agents.ts +22 -1
- package/src/services/alerting.ts +1 -1
- package/src/services/anthropic_passthrough.ts +1 -1
- package/src/services/anthropic_skills.ts +12 -2
- package/src/services/assistants.ts +1 -1
- package/src/services/audio.ts +1 -1
- package/src/services/audit_logging.ts +4 -1
- package/src/services/auto_router.ts +303 -93
- package/src/services/batch.ts +1 -1
- package/src/services/beta_agents.ts +1 -1
- package/src/services/beta_mcp.ts +1 -1
- package/src/services/budget_management.ts +12 -4
- package/src/services/budget_spend_tracking.ts +38 -6
- package/src/services/cache_settings.ts +1 -1
- package/src/services/caching.ts +1 -1
- package/src/services/chat_completions.ts +1 -1
- package/src/services/claude_code_marketplace.ts +20 -14
- package/src/services/cloudzero.ts +1 -1
- package/src/services/completions.ts +1 -1
- package/src/services/compliance.ts +1 -1
- package/src/services/config_overrides.ts +8 -2
- package/src/services/config_yaml.ts +251 -6
- package/src/services/containers.ts +29 -11
- package/src/services/coordination_redis_settings.ts +1 -1
- package/src/services/cost_tracking.ts +198 -2
- package/src/services/credential_management.ts +9 -4
- package/src/services/customer_management.ts +49 -3
- package/src/services/email_management.ts +1 -1
- package/src/services/embeddings.ts +1 -1
- package/src/services/evals.ts +1 -1
- package/src/services/experimental.ts +1 -1
- package/src/services/fallback_management.ts +1 -1
- package/src/services/files.ts +1 -1
- package/src/services/fine_tuning.ts +1 -1
- package/src/services/gemini_agents.ts +1 -1
- package/src/services/google_genai_endpoints.ts +1 -1
- package/src/services/guardrails.ts +198 -8
- package/src/services/health.ts +3 -2
- package/src/services/images.ts +1 -1
- package/src/services/index.ts +1 -15
- package/src/services/internal_user_management.ts +565 -103
- package/src/services/invite_links.ts +1 -1
- package/src/services/jwt_mappings.ts +7 -1
- package/src/services/key_management.ts +229 -126
- package/src/services/langfuse_passthrough.ts +1 -1
- package/src/services/llm_passthrough.ts +4542 -0
- package/src/services/llm_utils.ts +1 -1
- package/src/services/logging_callbacks.ts +1 -1
- package/src/services/mcp_byok_oauth.ts +65 -45
- package/src/services/mcp_discoverable.ts +109 -1
- package/src/services/mcp_management.ts +258 -3
- package/src/services/mcp_rest.ts +30 -5
- package/src/services/memory_management.ts +4 -1
- package/src/services/misc.ts +269 -2
- package/src/services/model_management.ts +666 -21
- package/src/services/moderations.ts +1 -1
- package/src/services/ocr.ts +1 -1
- package/src/services/open_ai_pass_through.ts +81 -286
- package/src/services/organization_management.ts +44 -2
- package/src/services/plugins.ts +1 -1
- package/src/services/policies.ts +5 -1
- package/src/services/policy_engine.ts +10 -1
- package/src/services/project_management.ts +33 -1
- package/src/services/prompts.ts +1 -1
- package/src/services/public.ts +356 -1
- package/src/services/rag.ts +1 -1
- package/src/services/realtime.ts +11 -111
- package/src/services/rerank.ts +187 -7
- package/src/services/responses.ts +117 -1
- package/src/services/router_settings.ts +1 -1
- package/src/services/rust_control_plane.ts +1 -1
- package/src/services/scim.ts +134 -3
- package/src/services/search.ts +1 -1
- package/src/services/search_tools.ts +1 -1
- package/src/services/settings.ts +278 -1
- package/src/services/sso_settings.ts +2 -1
- package/src/services/tag_management.ts +15 -1
- package/src/services/team_management.ts +676 -37
- package/src/services/tools.ts +1 -1
- package/src/services/ui_settings.ts +1 -1
- package/src/services/ui_theme_settings.ts +1 -1
- package/src/services/usage_ai.ts +1 -1
- package/src/services/vantage.ts +1 -1
- package/src/services/vector_store_management.ts +1 -1
- package/src/services/vector_stores.ts +1 -1
- package/src/services/videos.ts +1 -1
- package/src/services/web_socket.ts +1 -61
- package/src/services/workflow_management.ts +1 -1
- package/dist/services/anthropic_pass_through.d.ts +0 -70
- package/dist/services/anthropic_pass_through.d.ts.map +0 -1
- package/dist/services/anthropic_pass_through.js +0 -114
- package/dist/services/anthropic_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.js +0 -114
- package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_pass_through.js +0 -114
- package/dist/services/assembly_ai_pass_through.js.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
- package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
- package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
- package/dist/services/azure_ai_pass_through.d.ts +0 -70
- package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
- package/dist/services/azure_ai_pass_through.js +0 -112
- package/dist/services/azure_ai_pass_through.js.map +0 -1
- package/dist/services/azure_pass_through.d.ts +0 -70
- package/dist/services/azure_pass_through.d.ts.map +0 -1
- package/dist/services/azure_pass_through.js +0 -107
- package/dist/services/azure_pass_through.js.map +0 -1
- package/dist/services/bedrock_pass_through.d.ts +0 -70
- package/dist/services/bedrock_pass_through.d.ts.map +0 -1
- package/dist/services/bedrock_pass_through.js +0 -114
- package/dist/services/bedrock_pass_through.js.map +0 -1
- package/dist/services/cohere_pass_through.d.ts +0 -70
- package/dist/services/cohere_pass_through.d.ts.map +0 -1
- package/dist/services/cohere_pass_through.js +0 -112
- package/dist/services/cohere_pass_through.js.map +0 -1
- package/dist/services/cursor_pass_through.d.ts +0 -70
- package/dist/services/cursor_pass_through.d.ts.map +0 -1
- package/dist/services/cursor_pass_through.js +0 -112
- package/dist/services/cursor_pass_through.js.map +0 -1
- package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
- package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
- package/dist/services/google_ai_studio_pass_through.js +0 -112
- package/dist/services/google_ai_studio_pass_through.js.map +0 -1
- package/dist/services/milvus_pass_through.d.ts +0 -70
- package/dist/services/milvus_pass_through.d.ts.map +0 -1
- package/dist/services/milvus_pass_through.js +0 -112
- package/dist/services/milvus_pass_through.js.map +0 -1
- package/dist/services/mistral_pass_through.d.ts +0 -70
- package/dist/services/mistral_pass_through.d.ts.map +0 -1
- package/dist/services/mistral_pass_through.js +0 -114
- package/dist/services/mistral_pass_through.js.map +0 -1
- package/dist/services/vertex_ai_pass_through.d.ts +0 -180
- package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
- package/dist/services/vertex_ai_pass_through.js +0 -334
- package/dist/services/vertex_ai_pass_through.js.map +0 -1
- package/dist/services/vllm_pass_through.d.ts +0 -70
- package/dist/services/vllm_pass_through.d.ts.map +0 -1
- package/dist/services/vllm_pass_through.js +0 -104
- package/dist/services/vllm_pass_through.js.map +0 -1
- package/dist/services/watsonx_pass_through.d.ts +0 -70
- package/dist/services/watsonx_pass_through.d.ts.map +0 -1
- package/dist/services/watsonx_pass_through.js +0 -114
- package/dist/services/watsonx_pass_through.js.map +0 -1
- package/src/services/anthropic_pass_through.ts +0 -233
- package/src/services/assembly_ai_eu_pass_through.ts +0 -237
- package/src/services/assembly_ai_pass_through.ts +0 -237
- package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
- package/src/services/azure_ai_pass_through.ts +0 -231
- package/src/services/azure_pass_through.ts +0 -227
- package/src/services/bedrock_pass_through.ts +0 -229
- package/src/services/cohere_pass_through.ts +0 -227
- package/src/services/cursor_pass_through.ts +0 -227
- package/src/services/google_ai_studio_pass_through.ts +0 -227
- package/src/services/milvus_pass_through.ts +0 -227
- package/src/services/mistral_pass_through.ts +0 -229
- package/src/services/vertex_ai_pass_through.ts +0 -684
- package/src/services/vllm_pass_through.ts +0 -227
- package/src/services/watsonx_pass_through.ts +0 -229
|
@@ -26,6 +26,13 @@ export type LiteLLMParamsAdaptiveRouterConfigMap = {
|
|
|
26
26
|
[key: string]: unknown | undefined;
|
|
27
27
|
};
|
|
28
28
|
export declare const LiteLLMParamsAdaptiveRouterConfigMap: S.Schema<LiteLLMParamsAdaptiveRouterConfigMap>;
|
|
29
|
+
export interface AwsSessionTag {
|
|
30
|
+
Key: string;
|
|
31
|
+
Value: string;
|
|
32
|
+
}
|
|
33
|
+
export declare const AwsSessionTag: S.Schema<AwsSessionTag>;
|
|
34
|
+
export type LiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
|
|
35
|
+
export declare const LiteLLMParamsAwsSessionTagsList: S.Schema<LiteLLMParamsAwsSessionTagsList>;
|
|
29
36
|
export type LiteLLMParamsBedrockTagsList = Array<unknown>;
|
|
30
37
|
export declare const LiteLLMParamsBedrockTagsList: S.Schema<LiteLLMParamsBedrockTagsList>;
|
|
31
38
|
export type LiteLLMParamsComplexityRouterConfigMap = {
|
|
@@ -40,6 +47,8 @@ export type LiteLLMParamsConfigurableClientsideAuthParamsItem = string | Configu
|
|
|
40
47
|
export declare const LiteLLMParamsConfigurableClientsideAuthParamsItem: S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsItem>;
|
|
41
48
|
export type LiteLLMParamsConfigurableClientsideAuthParamsList = Array<LiteLLMParamsConfigurableClientsideAuthParamsItem>;
|
|
42
49
|
export declare const LiteLLMParamsConfigurableClientsideAuthParamsList: S.Schema<LiteLLMParamsConfigurableClientsideAuthParamsList>;
|
|
50
|
+
export type LiteLLMParamsDropParams = boolean | string;
|
|
51
|
+
export declare const LiteLLMParamsDropParams: S.Schema<LiteLLMParamsDropParams>;
|
|
43
52
|
export type LiteLLMParamsMilvusPartitionNamesList = Array<string>;
|
|
44
53
|
export declare const LiteLLMParamsMilvusPartitionNamesList: S.Schema<LiteLLMParamsMilvusPartitionNamesList>;
|
|
45
54
|
export type ChoicesFinishReason = "stop" | "content_filter" | "function_call" | "tool_calls" | "length" | "guardrail_intervened" | "eos" | "finish_reason_unspecified" | "malformed_function_call";
|
|
@@ -265,6 +274,7 @@ export interface LiteLLMParams {
|
|
|
265
274
|
adaptive_router_default_model?: string | null;
|
|
266
275
|
allow_client_keepalive_override?: boolean | null;
|
|
267
276
|
annotation_cost_per_page?: number | null;
|
|
277
|
+
annotation_cost_per_page_batches?: number | null;
|
|
268
278
|
api_base?: string | null;
|
|
269
279
|
api_key?: string | null;
|
|
270
280
|
api_version?: string | null;
|
|
@@ -273,6 +283,8 @@ export interface LiteLLMParams {
|
|
|
273
283
|
auto_router_default_model?: string | null;
|
|
274
284
|
auto_router_embedding_model?: string | null;
|
|
275
285
|
auto_router_max_input_chars?: number | null;
|
|
286
|
+
auto_router_model_compression?: string | null;
|
|
287
|
+
auto_router_routing_compression?: string | null;
|
|
276
288
|
aws_access_key_id?: string | null;
|
|
277
289
|
aws_batch_role_arn?: string | null;
|
|
278
290
|
aws_bedrock_project_id?: string | null;
|
|
@@ -283,10 +295,14 @@ export interface LiteLLMParams {
|
|
|
283
295
|
aws_role_name?: string | null;
|
|
284
296
|
aws_secret_access_key?: string | null;
|
|
285
297
|
aws_session_name?: string | null;
|
|
298
|
+
aws_session_tags?: LiteLLMParamsAwsSessionTagsList | null;
|
|
286
299
|
aws_session_token?: string | null;
|
|
287
300
|
aws_sts_endpoint?: string | null;
|
|
288
301
|
aws_web_identity_token?: string | null;
|
|
289
302
|
azure_ad_token?: string | null;
|
|
303
|
+
azure_password?: string | null;
|
|
304
|
+
azure_scope?: string | null;
|
|
305
|
+
azure_username?: string | null;
|
|
290
306
|
bedrock_tags?: LiteLLMParamsBedrockTagsList | null;
|
|
291
307
|
budget_duration?: string | null;
|
|
292
308
|
cache_creation_input_audio_token_cost?: number | null;
|
|
@@ -311,22 +327,27 @@ export interface LiteLLMParams {
|
|
|
311
327
|
cache_read_input_token_cost_priority?: number | null;
|
|
312
328
|
cache_read_input_token_cost_ultrafast?: number | null;
|
|
313
329
|
citation_cost_per_token?: number | null;
|
|
330
|
+
client_id?: string | null;
|
|
331
|
+
client_secret?: string | null;
|
|
314
332
|
complexity_router_config?: LiteLLMParamsComplexityRouterConfigMap | null;
|
|
315
333
|
complexity_router_default_model?: string | null;
|
|
316
334
|
configurable_clientside_auth_params?: LiteLLMParamsConfigurableClientsideAuthParamsList | null;
|
|
317
335
|
custom_llm_provider?: string | null;
|
|
318
336
|
default_api_key_rpm_limit?: number | null;
|
|
319
337
|
default_api_key_tpm_limit?: number | null;
|
|
338
|
+
drop_params?: LiteLLMParamsDropParams | null;
|
|
320
339
|
gcs_bucket_name?: string | null;
|
|
321
340
|
google_maps_grounding_cost_per_query?: number | null;
|
|
322
341
|
input_cost_per_audio_per_second?: number | null;
|
|
323
342
|
input_cost_per_audio_per_second_above_128k_tokens?: number | null;
|
|
324
343
|
input_cost_per_audio_token?: number | null;
|
|
344
|
+
input_cost_per_audio_token_batches?: number | null;
|
|
325
345
|
input_cost_per_character?: number | null;
|
|
326
346
|
input_cost_per_character_above_128k_tokens?: number | null;
|
|
327
347
|
input_cost_per_image?: number | null;
|
|
328
348
|
input_cost_per_image_above_128k_tokens?: number | null;
|
|
329
349
|
input_cost_per_image_token?: number | null;
|
|
350
|
+
input_cost_per_image_token_batches?: number | null;
|
|
330
351
|
input_cost_per_pixel?: number | null;
|
|
331
352
|
input_cost_per_query?: number | null;
|
|
332
353
|
input_cost_per_second?: number | null;
|
|
@@ -348,6 +369,7 @@ export interface LiteLLMParams {
|
|
|
348
369
|
input_cost_per_video_per_second_above_15s_interval?: number | null;
|
|
349
370
|
input_cost_per_video_per_second_above_8s_interval?: number | null;
|
|
350
371
|
input_cost_per_video_token?: number | null;
|
|
372
|
+
input_cost_per_video_token_batches?: number | null;
|
|
351
373
|
itpm?: number | null;
|
|
352
374
|
keepalive_seconds?: number | null;
|
|
353
375
|
litellm_credential_name?: string | null;
|
|
@@ -364,6 +386,7 @@ export interface LiteLLMParams {
|
|
|
364
386
|
model_info?: LiteLLMParamsModelInfoMap | null;
|
|
365
387
|
ocr_cost_per_credit?: number | null;
|
|
366
388
|
ocr_cost_per_page?: number | null;
|
|
389
|
+
ocr_cost_per_page_batches?: number | null;
|
|
367
390
|
organization?: string | null;
|
|
368
391
|
otpm?: number | null;
|
|
369
392
|
output_cost_per_audio_per_second?: number | null;
|
|
@@ -380,6 +403,7 @@ export interface LiteLLMParams {
|
|
|
380
403
|
output_cost_per_second_1080p?: number | null;
|
|
381
404
|
output_cost_per_second_480p?: number | null;
|
|
382
405
|
output_cost_per_second_4k?: number | null;
|
|
406
|
+
output_cost_per_second_720p?: number | null;
|
|
383
407
|
output_cost_per_token?: number | null;
|
|
384
408
|
output_cost_per_token_above_128k_tokens?: number | null;
|
|
385
409
|
output_cost_per_token_above_200k_tokens?: number | null;
|
|
@@ -404,12 +428,14 @@ export interface LiteLLMParams {
|
|
|
404
428
|
rpm?: number | null;
|
|
405
429
|
s3_bucket_name?: string | null;
|
|
406
430
|
s3_encryption_key_id?: string | null;
|
|
431
|
+
s3_endpoint_url?: string | null;
|
|
407
432
|
s3_output_bucket_name?: string | null;
|
|
408
433
|
s3_region_name?: string | null;
|
|
409
434
|
search_context_cost_per_query?: LiteLLMParamsSearchContextCostPerQueryMap | null;
|
|
410
435
|
stream_timeout?: LiteLLMParamsStreamTimeout | null;
|
|
411
436
|
tag_regex?: LiteLLMParamsTagRegexList | null;
|
|
412
437
|
tags?: LiteLLMParamsTagsList | null;
|
|
438
|
+
tenant_id?: string | null;
|
|
413
439
|
tiered_pricing?: LiteLLMParamsTieredPricingList | null;
|
|
414
440
|
timeout?: LiteLLMParamsTimeout | null;
|
|
415
441
|
tpm?: number | null;
|
|
@@ -453,6 +479,8 @@ export interface LitellmTypesRouterModelInfo {
|
|
|
453
479
|
id: string | null;
|
|
454
480
|
input_cost_per_character?: number | null;
|
|
455
481
|
input_cost_per_token?: number | null;
|
|
482
|
+
internal_router_model?: boolean | null;
|
|
483
|
+
member_auto_router?: boolean;
|
|
456
484
|
output_cost_per_character?: number | null;
|
|
457
485
|
output_cost_per_token?: number | null;
|
|
458
486
|
ptu_count?: number | null;
|
|
@@ -732,6 +760,10 @@ export interface GetModelInfoV2V2ModelInfoRequest {
|
|
|
732
760
|
sortOrder?: string;
|
|
733
761
|
/** Omit auto-router deployments (litellm model prefixed `auto_router/`). They select among deployments rather than being deployments themselves, so a caller rendering a deployment list can leave them out. Defaults to false, so existing callers are unaffected */
|
|
734
762
|
exclude_auto_routers?: boolean;
|
|
763
|
+
/** Only return deployments whose `model_info.access_groups` contains this access group */
|
|
764
|
+
access_group?: string;
|
|
765
|
+
/** Only return wildcard deployments, i.e. those whose `model_name` contains `*` */
|
|
766
|
+
wildcard_only?: boolean;
|
|
735
767
|
}
|
|
736
768
|
export declare const GetModelInfoV2V2ModelInfoRequest: S.Schema<GetModelInfoV2V2ModelInfoRequest>;
|
|
737
769
|
export interface GetModelInfoV2V2ModelInfoResponse {
|
|
@@ -834,6 +866,8 @@ export type UpdateLiteLLMParamsAdaptiveRouterConfigMap = {
|
|
|
834
866
|
[key: string]: unknown | undefined;
|
|
835
867
|
};
|
|
836
868
|
export declare const UpdateLiteLLMParamsAdaptiveRouterConfigMap: S.Schema<UpdateLiteLLMParamsAdaptiveRouterConfigMap>;
|
|
869
|
+
export type UpdateLiteLLMParamsAwsSessionTagsList = Array<AwsSessionTag>;
|
|
870
|
+
export declare const UpdateLiteLLMParamsAwsSessionTagsList: S.Schema<UpdateLiteLLMParamsAwsSessionTagsList>;
|
|
837
871
|
export type UpdateLiteLLMParamsBedrockTagsList = Array<unknown>;
|
|
838
872
|
export declare const UpdateLiteLLMParamsBedrockTagsList: S.Schema<UpdateLiteLLMParamsBedrockTagsList>;
|
|
839
873
|
export type UpdateLiteLLMParamsComplexityRouterConfigMap = {
|
|
@@ -844,6 +878,8 @@ export type UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem = string | C
|
|
|
844
878
|
export declare const UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem: S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem>;
|
|
845
879
|
export type UpdateLiteLLMParamsConfigurableClientsideAuthParamsList = Array<UpdateLiteLLMParamsConfigurableClientsideAuthParamsItem>;
|
|
846
880
|
export declare const UpdateLiteLLMParamsConfigurableClientsideAuthParamsList: S.Schema<UpdateLiteLLMParamsConfigurableClientsideAuthParamsList>;
|
|
881
|
+
export type UpdateLiteLLMParamsDropParams = boolean | string;
|
|
882
|
+
export declare const UpdateLiteLLMParamsDropParams: S.Schema<UpdateLiteLLMParamsDropParams>;
|
|
847
883
|
export type UpdateLiteLLMParamsMilvusPartitionNamesList = Array<string>;
|
|
848
884
|
export declare const UpdateLiteLLMParamsMilvusPartitionNamesList: S.Schema<UpdateLiteLLMParamsMilvusPartitionNamesList>;
|
|
849
885
|
export type UpdateLiteLLMParamsMockResponse = string | ModelResponse | unknown;
|
|
@@ -885,6 +921,7 @@ export interface UpdateLiteLLMParams {
|
|
|
885
921
|
adaptive_router_default_model?: string | null;
|
|
886
922
|
allow_client_keepalive_override?: boolean | null;
|
|
887
923
|
annotation_cost_per_page?: number | null;
|
|
924
|
+
annotation_cost_per_page_batches?: number | null;
|
|
888
925
|
api_base?: string | null;
|
|
889
926
|
api_key?: string | null;
|
|
890
927
|
api_version?: string | null;
|
|
@@ -893,6 +930,8 @@ export interface UpdateLiteLLMParams {
|
|
|
893
930
|
auto_router_default_model?: string | null;
|
|
894
931
|
auto_router_embedding_model?: string | null;
|
|
895
932
|
auto_router_max_input_chars?: number | null;
|
|
933
|
+
auto_router_model_compression?: string | null;
|
|
934
|
+
auto_router_routing_compression?: string | null;
|
|
896
935
|
aws_access_key_id?: string | null;
|
|
897
936
|
aws_batch_role_arn?: string | null;
|
|
898
937
|
aws_bedrock_project_id?: string | null;
|
|
@@ -903,10 +942,14 @@ export interface UpdateLiteLLMParams {
|
|
|
903
942
|
aws_role_name?: string | null;
|
|
904
943
|
aws_secret_access_key?: string | null;
|
|
905
944
|
aws_session_name?: string | null;
|
|
945
|
+
aws_session_tags?: UpdateLiteLLMParamsAwsSessionTagsList | null;
|
|
906
946
|
aws_session_token?: string | null;
|
|
907
947
|
aws_sts_endpoint?: string | null;
|
|
908
948
|
aws_web_identity_token?: string | null;
|
|
909
949
|
azure_ad_token?: string | null;
|
|
950
|
+
azure_password?: string | null;
|
|
951
|
+
azure_scope?: string | null;
|
|
952
|
+
azure_username?: string | null;
|
|
910
953
|
bedrock_tags?: UpdateLiteLLMParamsBedrockTagsList | null;
|
|
911
954
|
budget_duration?: string | null;
|
|
912
955
|
cache_creation_input_audio_token_cost?: number | null;
|
|
@@ -931,22 +974,27 @@ export interface UpdateLiteLLMParams {
|
|
|
931
974
|
cache_read_input_token_cost_priority?: number | null;
|
|
932
975
|
cache_read_input_token_cost_ultrafast?: number | null;
|
|
933
976
|
citation_cost_per_token?: number | null;
|
|
977
|
+
client_id?: string | null;
|
|
978
|
+
client_secret?: string | null;
|
|
934
979
|
complexity_router_config?: UpdateLiteLLMParamsComplexityRouterConfigMap | null;
|
|
935
980
|
complexity_router_default_model?: string | null;
|
|
936
981
|
configurable_clientside_auth_params?: UpdateLiteLLMParamsConfigurableClientsideAuthParamsList | null;
|
|
937
982
|
custom_llm_provider?: string | null;
|
|
938
983
|
default_api_key_rpm_limit?: number | null;
|
|
939
984
|
default_api_key_tpm_limit?: number | null;
|
|
985
|
+
drop_params?: UpdateLiteLLMParamsDropParams | null;
|
|
940
986
|
gcs_bucket_name?: string | null;
|
|
941
987
|
google_maps_grounding_cost_per_query?: number | null;
|
|
942
988
|
input_cost_per_audio_per_second?: number | null;
|
|
943
989
|
input_cost_per_audio_per_second_above_128k_tokens?: number | null;
|
|
944
990
|
input_cost_per_audio_token?: number | null;
|
|
991
|
+
input_cost_per_audio_token_batches?: number | null;
|
|
945
992
|
input_cost_per_character?: number | null;
|
|
946
993
|
input_cost_per_character_above_128k_tokens?: number | null;
|
|
947
994
|
input_cost_per_image?: number | null;
|
|
948
995
|
input_cost_per_image_above_128k_tokens?: number | null;
|
|
949
996
|
input_cost_per_image_token?: number | null;
|
|
997
|
+
input_cost_per_image_token_batches?: number | null;
|
|
950
998
|
input_cost_per_pixel?: number | null;
|
|
951
999
|
input_cost_per_query?: number | null;
|
|
952
1000
|
input_cost_per_second?: number | null;
|
|
@@ -968,6 +1016,7 @@ export interface UpdateLiteLLMParams {
|
|
|
968
1016
|
input_cost_per_video_per_second_above_15s_interval?: number | null;
|
|
969
1017
|
input_cost_per_video_per_second_above_8s_interval?: number | null;
|
|
970
1018
|
input_cost_per_video_token?: number | null;
|
|
1019
|
+
input_cost_per_video_token_batches?: number | null;
|
|
971
1020
|
itpm?: number | null;
|
|
972
1021
|
keepalive_seconds?: number | null;
|
|
973
1022
|
litellm_credential_name?: string | null;
|
|
@@ -984,6 +1033,7 @@ export interface UpdateLiteLLMParams {
|
|
|
984
1033
|
model_info?: UpdateLiteLLMParamsModelInfoMap | null;
|
|
985
1034
|
ocr_cost_per_credit?: number | null;
|
|
986
1035
|
ocr_cost_per_page?: number | null;
|
|
1036
|
+
ocr_cost_per_page_batches?: number | null;
|
|
987
1037
|
organization?: string | null;
|
|
988
1038
|
otpm?: number | null;
|
|
989
1039
|
output_cost_per_audio_per_second?: number | null;
|
|
@@ -1000,6 +1050,7 @@ export interface UpdateLiteLLMParams {
|
|
|
1000
1050
|
output_cost_per_second_1080p?: number | null;
|
|
1001
1051
|
output_cost_per_second_480p?: number | null;
|
|
1002
1052
|
output_cost_per_second_4k?: number | null;
|
|
1053
|
+
output_cost_per_second_720p?: number | null;
|
|
1003
1054
|
output_cost_per_token?: number | null;
|
|
1004
1055
|
output_cost_per_token_above_128k_tokens?: number | null;
|
|
1005
1056
|
output_cost_per_token_above_200k_tokens?: number | null;
|
|
@@ -1024,12 +1075,14 @@ export interface UpdateLiteLLMParams {
|
|
|
1024
1075
|
rpm?: number | null;
|
|
1025
1076
|
s3_bucket_name?: string | null;
|
|
1026
1077
|
s3_encryption_key_id?: string | null;
|
|
1078
|
+
s3_endpoint_url?: string | null;
|
|
1027
1079
|
s3_output_bucket_name?: string | null;
|
|
1028
1080
|
s3_region_name?: string | null;
|
|
1029
1081
|
search_context_cost_per_query?: UpdateLiteLLMParamsSearchContextCostPerQueryMap | null;
|
|
1030
1082
|
stream_timeout?: UpdateLiteLLMParamsStreamTimeout | null;
|
|
1031
1083
|
tag_regex?: UpdateLiteLLMParamsTagRegexList | null;
|
|
1032
1084
|
tags?: UpdateLiteLLMParamsTagsList | null;
|
|
1085
|
+
tenant_id?: string | null;
|
|
1033
1086
|
tiered_pricing?: UpdateLiteLLMParamsTieredPricingList | null;
|
|
1034
1087
|
timeout?: UpdateLiteLLMParamsTimeout | null;
|
|
1035
1088
|
tpm?: number | null;
|
|
@@ -1093,7 +1146,7 @@ export interface PostBlockModelModelBlockResponse {
|
|
|
1093
1146
|
export declare const PostBlockModelModelBlockResponse: S.Schema<PostBlockModelModelBlockResponse>;
|
|
1094
1147
|
/** An operator-defined tier: the name the LLM classifier must return and its rubric description. */
|
|
1095
1148
|
export interface TierDefinition {
|
|
1096
|
-
/** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (SIMPLE
|
|
1149
|
+
/** What belongs in this tier; rendered as this tier's bullet in the classifier rubric. Required unless the name is a built-in tier (NON_REASONING, SIMPLE, MEDIUM, COMPLEX, REASONING), which inherits the built-in criteria when omitted */
|
|
1097
1150
|
description?: string | null;
|
|
1098
1151
|
/** Tier name; becomes a value the LLM classifier can return and a key of `tiers` */
|
|
1099
1152
|
name: string;
|
|
@@ -1101,10 +1154,17 @@ export interface TierDefinition {
|
|
|
1101
1154
|
export declare const TierDefinition: S.Schema<TierDefinition>;
|
|
1102
1155
|
export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList = Array<TierDefinition>;
|
|
1103
1156
|
export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList>;
|
|
1157
|
+
export type PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap = {
|
|
1158
|
+
[key: string]: string | undefined;
|
|
1159
|
+
};
|
|
1160
|
+
export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap>;
|
|
1104
1161
|
export interface PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest {
|
|
1162
|
+
classification_examples?: string | null;
|
|
1105
1163
|
classification_prompt?: string | null;
|
|
1164
|
+
classification_rubric?: ClassificationRubric | (string & {}) | null;
|
|
1106
1165
|
context_window_size?: number;
|
|
1107
|
-
tier_definitions
|
|
1166
|
+
tier_definitions?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierDefinitionsList | null;
|
|
1167
|
+
tier_labels?: PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequestTierLabelsMap | null;
|
|
1108
1168
|
}
|
|
1109
1169
|
export declare const PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest: S.Schema<PostPreviewAutoRouterClassifierPromptAutoRouterClassifierDefaultPromptRequest>;
|
|
1110
1170
|
/** When adaptive=True: 'all' scores every pool model with a tier-distance penalty (soft floors); 'classified_tier' Thompson-samples only inside the classified tier's pool */
|
|
@@ -1115,26 +1175,93 @@ export interface AdaptiveRouterWeights {
|
|
|
1115
1175
|
quality?: number;
|
|
1116
1176
|
}
|
|
1117
1177
|
export declare const AdaptiveRouterWeights: S.Schema<AdaptiveRouterWeights>;
|
|
1178
|
+
export interface CapabilityCalibrationConfig {
|
|
1179
|
+
intercept: number;
|
|
1180
|
+
slope: number;
|
|
1181
|
+
version: string;
|
|
1182
|
+
}
|
|
1183
|
+
export declare const CapabilityCalibrationConfig: S.Schema<CapabilityCalibrationConfig>;
|
|
1184
|
+
/** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
|
|
1185
|
+
export type CapabilityClassifierConfigResponseFormat = "json_schema" | "json_object";
|
|
1186
|
+
export declare const CapabilityClassifierConfigResponseFormat: any;
|
|
1187
|
+
/** Switchyard-compatible probability threshold policy for two model tiers. */
|
|
1188
|
+
export interface CapabilityClassifierConfig {
|
|
1189
|
+
/** Lowest p_solve that routes a supported task to efficient_tier */
|
|
1190
|
+
base_threshold: number;
|
|
1191
|
+
/** Optional versioned sigmoid calibration fitted for this judge, capability card, efficient model, and execution setup. Applies sigmoid(slope * logit(clip(p_solve, 1e-6, 1-1e-6)) + intercept) before the threshold policy. Omit to route on the raw forecast. */
|
|
1192
|
+
calibration?: CapabilityCalibrationConfig | null;
|
|
1193
|
+
/** Higher, fail-closed tier used below the adjusted threshold or when the classifier verdict is unavailable */
|
|
1194
|
+
capable_tier: string;
|
|
1195
|
+
/** Tier used when the efficient model's forecasted solve probability meets the adjusted threshold */
|
|
1196
|
+
efficient_tier: string;
|
|
1197
|
+
/** Maximum completion tokens available to the capability classifier verdict */
|
|
1198
|
+
max_output_tokens?: number;
|
|
1199
|
+
/** Use json_object for judges without strict JSON Schema support. This appends the verdict schema to the packaged system prompt; both modes validate the returned verdict identically. */
|
|
1200
|
+
response_format?: CapabilityClassifierConfigResponseFormat | (string & {});
|
|
1201
|
+
/** Amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts */
|
|
1202
|
+
threshold_step?: number;
|
|
1203
|
+
}
|
|
1204
|
+
export declare const CapabilityClassifierConfig: S.Schema<CapabilityClassifierConfig>;
|
|
1205
|
+
/** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
|
|
1206
|
+
export type RequestComplexityRouterConfigClassificationMode = "every_request" | "user_turn";
|
|
1207
|
+
export declare const RequestComplexityRouterConfigClassificationMode: any;
|
|
1118
1208
|
/** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
|
|
1119
1209
|
export type RequestComplexityRouterConfigClassifierFallback = "heuristic" | "default_model";
|
|
1120
1210
|
export declare const RequestComplexityRouterConfigClassifierFallback: any;
|
|
1211
|
+
export type ClassifierLLMConfigReasoningEffort = "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
|
1212
|
+
export declare const ClassifierLLMConfigReasoningEffort: any;
|
|
1213
|
+
/** Whether the LLM classifier sees the images on the request it is classifying. Off by default because images cost far more than the text ask they arrive with, and the classifier runs on every request. A turn whose complexity lives in the image ("what is wrong in this stack trace screenshot") is invisible to a text-only classifier, which is what this buys. */
|
|
1214
|
+
export interface ClassifierVisionConfig {
|
|
1215
|
+
/** Forward image content to the classifier. Requires a classifier model declared supports_vision, on the deployment's model_info or in the model cost map; images stay stripped otherwise, so a classifier that cannot read them is never sent one. Declare model_info.supports_vision on the deployment to enable a model the cost map does not describe. Only inline data: URIs are forwarded. A request whose images are http(s) URLs still classifies on its text alone, because some providers fetch such a URL from the proxy rather than the provider, which would let a caller aim a proxy-side request at an address of their choosing. */
|
|
1216
|
+
enabled?: boolean;
|
|
1217
|
+
/** How many images from the newest user turn to forward, in wire order. Bounds the added cost of a turn that attaches many images. Images on earlier turns are never forwarded. */
|
|
1218
|
+
max_images?: number;
|
|
1219
|
+
}
|
|
1220
|
+
export declare const ClassifierVisionConfig: S.Schema<ClassifierVisionConfig>;
|
|
1121
1221
|
/** Configuration for the LLM-based complexity classifier. */
|
|
1122
1222
|
export interface ClassifierLLMConfig {
|
|
1223
|
+
/** How long to skip this router's LLM classifier after a classification call times out. Requests use classifier_fallback during the cooldown. When it expires, one request probes the classifier while concurrent requests keep using the fallback; a successful probe closes the circuit and a failed probe restarts the cooldown. */
|
|
1224
|
+
circuit_breaker_cooldown_seconds?: number;
|
|
1225
|
+
/** Whether one classifier timeout temporarily sends requests through classifier_fallback. Enabled by default so an unhealthy classifier cannot repeat its timeout across sessions. */
|
|
1226
|
+
circuit_breaker_enabled?: boolean;
|
|
1123
1227
|
/** Which calibration examples the built-in rubric carries. 'agentic' anchors routine installs, builds, multi-file edits, and standard debugging at MEDIUM, so ordinary engineering does not route to the most expensive tier; it suits agent, terminal, and coding-assistant traffic as well as mixed traffic. 'chat' omits those engineering anchors, for a deployment serving only conversational traffic. 'business' carries business/sales anchors and business-flavored tier criteria that keep routine drafting and summarizing off the expensive tiers and reserve the top tier for committing to decisions under tradeoffs; it suits sales, support, and go-to-market traffic. Every preset keeps the same four tiers, so this moves where the boundary sits without changing the taxonomy. Leave unset for 'legacy', the rubric as it shipped before calibration examples existed, so an existing router's tier decisions and spend do not move on upgrade. Mutually exclusive with system_prompt, which replaces the rubric this would select. Only applies when classifier_type is 'llm'. */
|
|
1124
1228
|
classification_rubric?: ClassificationRubric | (string & {}) | null;
|
|
1125
1229
|
/** Model name (from the router's model_list) to call for classification */
|
|
1126
1230
|
model: string;
|
|
1231
|
+
/** Reasoning effort override for classifier calls. Leave unset to use the classifier deployment or provider default. */
|
|
1232
|
+
reasoning_effort?: ClassifierLLMConfigReasoningEffort | (string & {}) | null;
|
|
1127
1233
|
/** Replaces the built-in complexity rubric as the classifier's entire system role. When set, neither the default rubric nor the context-window closing line is appended, so the prompt owns the whole taxonomy and the tier names SIMPLE/MEDIUM/COMPLEX/REASONING become whatever buckets it defines: a prompt that classifies data sensitivity routes on that instead of on difficulty. Two consequences of full replacement. The default rubric's closing paragraph is the classifier's prompt-injection defense, telling it that the caller's quoted system prompt and prior turns are material to judge and never instructions; a replacement that omits it lets a caller ask for a tier and get it. And the heuristic fallback still scores complexity, so a router on some other taxonomy wants classifier_fallback='default_model'. Leave unset for the built-in rubric. Only applies when classifier_type is 'llm'. */
|
|
1128
1234
|
system_prompt?: string | null;
|
|
1129
1235
|
/** Timeout budget for the classification call, in milliseconds */
|
|
1130
1236
|
timeout_ms?: number;
|
|
1237
|
+
/** Whether the classifier sees images on the request, and how many */
|
|
1238
|
+
vision?: ClassifierVisionConfig;
|
|
1131
1239
|
}
|
|
1132
1240
|
export declare const ClassifierLLMConfig: S.Schema<ClassifierLLMConfig>;
|
|
1133
|
-
/** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin,
|
|
1134
|
-
export type RequestComplexityRouterConfigClassifierType = "heuristic" | "llm" | "custom" | "heuristic_first";
|
|
1241
|
+
/** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
|
|
1242
|
+
export type RequestComplexityRouterConfigClassifierType = "heuristic" | "heuristic_v2" | "llm" | "capability" | "llm_v2" | "custom" | "heuristic_first" | "hybrid" | "jev";
|
|
1135
1243
|
export declare const RequestComplexityRouterConfigClassifierType: any;
|
|
1136
1244
|
export type RequestComplexityRouterConfigCodeKeywordsList = Array<string>;
|
|
1137
1245
|
export declare const RequestComplexityRouterConfigCodeKeywordsList: S.Schema<RequestComplexityRouterConfigCodeKeywordsList>;
|
|
1246
|
+
export type CustomDimensionKeywordsList = Array<string>;
|
|
1247
|
+
export declare const CustomDimensionKeywordsList: S.Schema<CustomDimensionKeywordsList>;
|
|
1248
|
+
export type CustomDimensionPatternsList = Array<string>;
|
|
1249
|
+
export declare const CustomDimensionPatternsList: S.Schema<CustomDimensionPatternsList>;
|
|
1250
|
+
/** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
|
|
1251
|
+
export type CustomDimensionScoringMode = "binary" | "match_count";
|
|
1252
|
+
export declare const CustomDimensionScoringMode: any;
|
|
1253
|
+
export interface CustomDimension {
|
|
1254
|
+
keywords?: CustomDimensionKeywordsList;
|
|
1255
|
+
name: string;
|
|
1256
|
+
patterns?: CustomDimensionPatternsList;
|
|
1257
|
+
/** 'binary' scores 1 when any matcher hits. 'match_count' scores 0.5 when one distinct matcher hits and 1 when two or more do; repeated occurrences of one matcher never raise it. Keywords are distinct case-insensitively, patterns by source, and a keyword and a pattern are always distinct from each other. */
|
|
1258
|
+
scoring_mode?: CustomDimensionScoringMode | (string & {});
|
|
1259
|
+
weight: number;
|
|
1260
|
+
}
|
|
1261
|
+
export declare const CustomDimension: S.Schema<CustomDimension>;
|
|
1262
|
+
/** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
|
|
1263
|
+
export type RequestComplexityRouterConfigCustomDimensionsList = Array<CustomDimension>;
|
|
1264
|
+
export declare const RequestComplexityRouterConfigCustomDimensionsList: S.Schema<RequestComplexityRouterConfigCustomDimensionsList>;
|
|
1138
1265
|
export type RequestComplexityRouterConfigCustomTechnicalKeywordsList = Array<string>;
|
|
1139
1266
|
export declare const RequestComplexityRouterConfigCustomTechnicalKeywordsList: S.Schema<RequestComplexityRouterConfigCustomTechnicalKeywordsList>;
|
|
1140
1267
|
/** Weights for each scoring dimension */
|
|
@@ -1144,8 +1271,76 @@ export type RequestComplexityRouterConfigDimensionWeightsMap = {
|
|
|
1144
1271
|
export declare const RequestComplexityRouterConfigDimensionWeightsMap: S.Schema<RequestComplexityRouterConfigDimensionWeightsMap>;
|
|
1145
1272
|
export type RequestComplexityRouterConfigEscalationKeywordsList = Array<string>;
|
|
1146
1273
|
export declare const RequestComplexityRouterConfigEscalationKeywordsList: S.Schema<RequestComplexityRouterConfigEscalationKeywordsList>;
|
|
1274
|
+
export interface TierCohortStatistic {
|
|
1275
|
+
cohort: string;
|
|
1276
|
+
observations: number;
|
|
1277
|
+
successes: number;
|
|
1278
|
+
tier: number;
|
|
1279
|
+
}
|
|
1280
|
+
export declare const TierCohortStatistic: S.Schema<TierCohortStatistic>;
|
|
1281
|
+
export type TrainedTierArtifactCohortStatisticsList = Array<TierCohortStatistic>;
|
|
1282
|
+
export declare const TrainedTierArtifactCohortStatisticsList: S.Schema<TrainedTierArtifactCohortStatisticsList>;
|
|
1283
|
+
export interface TierDataset {
|
|
1284
|
+
license: string;
|
|
1285
|
+
name: string;
|
|
1286
|
+
rows: number;
|
|
1287
|
+
success_definition?: string;
|
|
1288
|
+
url: string;
|
|
1289
|
+
}
|
|
1290
|
+
export declare const TierDataset: S.Schema<TierDataset>;
|
|
1291
|
+
export type TrainedTierArtifactDatasetsList = Array<TierDataset>;
|
|
1292
|
+
export declare const TrainedTierArtifactDatasetsList: S.Schema<TrainedTierArtifactDatasetsList>;
|
|
1293
|
+
/** Fixed v0 taxonomy. User-extensible types come in v1. */
|
|
1294
|
+
export type RequestType = "code_generation" | "code_understanding" | "technical_design" | "analytical_reasoning" | "writing" | "factual_lookup" | "general";
|
|
1295
|
+
export declare const RequestType: any;
|
|
1296
|
+
export interface TierDomainStatistic {
|
|
1297
|
+
observations: number;
|
|
1298
|
+
request_type: RequestType | (string & {});
|
|
1299
|
+
successes: number;
|
|
1300
|
+
tier: number;
|
|
1301
|
+
}
|
|
1302
|
+
export declare const TierDomainStatistic: S.Schema<TierDomainStatistic>;
|
|
1303
|
+
export type TrainedTierArtifactDomainStatisticsList = Array<TierDomainStatistic>;
|
|
1304
|
+
export declare const TrainedTierArtifactDomainStatisticsList: S.Schema<TrainedTierArtifactDomainStatisticsList>;
|
|
1305
|
+
export interface TierGlobalStatistic {
|
|
1306
|
+
observations: number;
|
|
1307
|
+
successes: number;
|
|
1308
|
+
tier: number;
|
|
1309
|
+
}
|
|
1310
|
+
export declare const TierGlobalStatistic: S.Schema<TierGlobalStatistic>;
|
|
1311
|
+
export type TrainedTierArtifactGlobalStatisticsList = Array<TierGlobalStatistic>;
|
|
1312
|
+
export declare const TrainedTierArtifactGlobalStatisticsList: S.Schema<TrainedTierArtifactGlobalStatisticsList>;
|
|
1313
|
+
export interface TrainedTierArtifact {
|
|
1314
|
+
cohort_prior_mass?: number;
|
|
1315
|
+
cohort_statistics?: TrainedTierArtifactCohortStatisticsList;
|
|
1316
|
+
datasets?: TrainedTierArtifactDatasetsList;
|
|
1317
|
+
domain_prior_mass?: number;
|
|
1318
|
+
domain_statistics?: TrainedTierArtifactDomainStatisticsList;
|
|
1319
|
+
global_statistics: TrainedTierArtifactGlobalStatisticsList;
|
|
1320
|
+
routing_threshold?: number;
|
|
1321
|
+
schema_version?: number;
|
|
1322
|
+
split_method?: string;
|
|
1323
|
+
success_definition?: string;
|
|
1324
|
+
}
|
|
1325
|
+
export declare const TrainedTierArtifact: S.Schema<TrainedTierArtifact>;
|
|
1326
|
+
/** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
|
|
1327
|
+
export type RequestComplexityRouterConfigHeuristicV2Artifact = TrainedTierArtifact | string;
|
|
1328
|
+
export declare const RequestComplexityRouterConfigHeuristicV2Artifact: S.Schema<RequestComplexityRouterConfigHeuristicV2Artifact>;
|
|
1147
1329
|
export type RequestComplexityRouterConfigHousekeepingPatternsList = Array<string>;
|
|
1148
1330
|
export declare const RequestComplexityRouterConfigHousekeepingPatternsList: S.Schema<RequestComplexityRouterConfigHousekeepingPatternsList>;
|
|
1331
|
+
export interface JevClassifierConfig {
|
|
1332
|
+
/** TypeSafe API base, falling back to TYPESAFE_API_BASE and then https://api.typesafe.ai */
|
|
1333
|
+
api_base?: string | null;
|
|
1334
|
+
/** TypeSafe API key, falling back to TYPESAFE_API_KEY */
|
|
1335
|
+
api_key?: string | null;
|
|
1336
|
+
circuit_breaker_cooldown_seconds?: number;
|
|
1337
|
+
circuit_breaker_enabled?: boolean;
|
|
1338
|
+
/** Replaces the built-in Jev question instructions */
|
|
1339
|
+
instructions?: string | null;
|
|
1340
|
+
model?: string;
|
|
1341
|
+
timeout_ms?: number;
|
|
1342
|
+
}
|
|
1343
|
+
export declare const JevClassifierConfig: S.Schema<JevClassifierConfig>;
|
|
1149
1344
|
/** Keywords/phrases that trigger this rule (lexical or semantic match) */
|
|
1150
1345
|
export type KeywordTierRuleKeywordsList = Array<string>;
|
|
1151
1346
|
export declare const KeywordTierRuleKeywordsList: S.Schema<KeywordTierRuleKeywordsList>;
|
|
@@ -1159,6 +1354,36 @@ export interface KeywordTierRule {
|
|
|
1159
1354
|
export declare const KeywordTierRule: S.Schema<KeywordTierRule>;
|
|
1160
1355
|
export type RequestComplexityRouterConfigKeywordTierRulesList = Array<KeywordTierRule>;
|
|
1161
1356
|
export declare const RequestComplexityRouterConfigKeywordTierRulesList: S.Schema<RequestComplexityRouterConfigKeywordTierRulesList>;
|
|
1357
|
+
export interface LLMV2ProbabilityCalibration {
|
|
1358
|
+
intercept: number;
|
|
1359
|
+
slope: number;
|
|
1360
|
+
}
|
|
1361
|
+
export declare const LLMV2ProbabilityCalibration: S.Schema<LLMV2ProbabilityCalibration>;
|
|
1362
|
+
export interface LLMV2Calibration {
|
|
1363
|
+
capable: LLMV2ProbabilityCalibration;
|
|
1364
|
+
efficient: LLMV2ProbabilityCalibration;
|
|
1365
|
+
prompt_version: string;
|
|
1366
|
+
version: string;
|
|
1367
|
+
}
|
|
1368
|
+
export declare const LLMV2Calibration: S.Schema<LLMV2Calibration>;
|
|
1369
|
+
export type LLMV2ConfigResponseFormat = "json_schema" | "json_object";
|
|
1370
|
+
export declare const LLMV2ConfigResponseFormat: any;
|
|
1371
|
+
export interface LLMV2Config {
|
|
1372
|
+
calibration?: LLMV2Calibration | null;
|
|
1373
|
+
capable_profile?: string | null;
|
|
1374
|
+
capable_profile_preset?: string | null;
|
|
1375
|
+
capable_tier?: string;
|
|
1376
|
+
efficient_profile?: string | null;
|
|
1377
|
+
efficient_profile_preset?: string | null;
|
|
1378
|
+
efficient_tier?: string;
|
|
1379
|
+
harness?: string | null;
|
|
1380
|
+
harness_preset?: string | null;
|
|
1381
|
+
max_output_tokens?: number;
|
|
1382
|
+
/** Maximum estimated success loss allowed for efficient. */
|
|
1383
|
+
max_quality_gap: number;
|
|
1384
|
+
response_format?: LLMV2ConfigResponseFormat | (string & {});
|
|
1385
|
+
}
|
|
1386
|
+
export declare const LLMV2Config: S.Schema<LLMV2Config>;
|
|
1162
1387
|
export type RequestComplexityRouterConfigPlanModePatternsList = Array<string>;
|
|
1163
1388
|
export declare const RequestComplexityRouterConfigPlanModePatternsList: S.Schema<RequestComplexityRouterConfigPlanModePatternsList>;
|
|
1164
1389
|
export type RequestComplexityRouterConfigReasoningKeywordsList = Array<string>;
|
|
@@ -1226,50 +1451,77 @@ export interface RequestComplexityRouterConfig {
|
|
|
1226
1451
|
adaptive_eligible?: RequestComplexityRouterConfigAdaptiveEligible | (string & {});
|
|
1227
1452
|
/** Quality vs cost weights for adaptive selection (used when adaptive=True) */
|
|
1228
1453
|
adaptive_weights?: AdaptiveRouterWeights;
|
|
1229
|
-
/**
|
|
1454
|
+
/** Probability threshold policy required when classifier_type is 'capability'. The classifier forecasts p_solve for efficient_tier, adjusts base_threshold using the capability-card boundary, and otherwise routes to capable_tier */
|
|
1455
|
+
capability_classifier_config?: CapabilityClassifierConfig | null;
|
|
1456
|
+
/** Replaces the calibration examples of the LLM classifier rubric, and nothing else. Written as example lines only: the router renders the 'Calibration examples:' heading above them, after the per-tier bullets. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_prompt replaces them, the classification instructions; a custom tier set ships no examples of its own, so the section renders only when this is set. */
|
|
1457
|
+
classification_examples?: string | null;
|
|
1458
|
+
/** When to run the complexity classifier. 'every_request' (the default) classifies every inference request, including the tool-result continuation turns of an agentic loop. 'user_turn' classifies only requests whose newest turn is a new human ask and replays the session's held routing decision on continuation turns, which cuts classifier spend and eliminates mid-loop model switches. Continuations with no held decision to replay (no resolvable session_id, expired pin, fresh restart) still classify. Unlike session_affinity, a new human ask always re-classifies, so a session can still move tiers between asks. Suppressed when plugins are configured, for the same reason session_affinity is: a replayed decision would bypass the plugin pipeline. */
|
|
1459
|
+
classification_mode?: RequestComplexityRouterConfigClassificationMode | (string & {});
|
|
1460
|
+
/** Replaces the classification instructions that open the LLM classifier rubric, and nothing else. The per-tier bullets follow it, the calibration examples follow those, and the trust-boundary paragraph telling the classifier to ignore tier requests embedded in quoted caller text is always appended after them and cannot be overridden. Requires an LLM classifier and cannot be combined with classifier_llm_config.system_prompt. With built-in tiers the rubric preset still supplies the tier criteria and, unless classification_examples replaces them, the calibration examples. */
|
|
1230
1461
|
classification_prompt?: string | null;
|
|
1231
|
-
/** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and the
|
|
1462
|
+
/** Maximum characters of prior-turn text quoted to the LLM classifier, across the whole context window, per classification call. Turns are taken newest first and quoted whole while they fit, so a conversation small enough to quote entirely is never cut; once the budget runs out the older turns are dropped whole and only the turn straddling the boundary is truncated, into whatever space is left. The current ask and, except for Claude Code requests, the extracted system-role text sit outside this budget and are sent in full, as does the numbering each quoted turn carries. A budget under 120 leaves no room to quote a turn and suppresses the block; set classifier_context_window_size to 0 to turn context off deliberately. Only applies when classifier_type is 'llm'. */
|
|
1232
1463
|
classifier_context_budget_chars?: number;
|
|
1233
1464
|
/** Include assistant turns in the classifier context window, so difficulty stated by the model rather than by the user stays visible: a plan the assistant calls complex, which the user approves with 'yes', is classified on the work being approved instead of on the word 'yes'. When enabled, classifier_context_window_size counts the last N turns of the conversation across both roles rather than the last N user turns, and assistant text is sent to the classifier model, which may be a different deployment or provider than the routed completion model. Assistant replies spend classifier_context_budget_chars alongside user turns, so raise it if the oldest turns stop being quoted once replies join the window. Off by default because enabling it shifts tier decisions, and therefore spend, for an already-deployed router. Only applies when classifier_type is 'llm'. */
|
|
1234
1465
|
classifier_context_include_assistant_turns?: boolean;
|
|
1235
1466
|
/** Optional cap on each individual prior turn's text, applied before classifier_context_budget_chars bounds the block. Unset by default, so one long turn may spend the whole budget, which is usually what a follow-up needs; set it when no single turn should dominate the context the classifier sees. A capped turn keeps its opening and its ending with the middle elided. Only applies when classifier_type is 'llm'. */
|
|
1236
1467
|
classifier_context_per_turn_chars?: number | null;
|
|
1237
|
-
/** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call
|
|
1468
|
+
/** Number of prior user turns (tool output and harness reminders excluded) to include as context in the LLM classifier prompt, so a follow-up like 'now do the same for the streaming path' is classified against what it refers to. Counts turns of both roles when classifier_context_include_assistant_turns is enabled. These turns are sent to the classifier model, which may be a different deployment or provider than the routed completion model; that call carries the current user ask and, except for Claude Code requests, the extracted system-role text in full. Claude Code system text is omitted to avoid classifying harness instructions; the routed completion still receives it. Set to 0 to send neither prior turns nor any conversation context beyond the current ask. Only applies when classifier_type is 'llm'. */
|
|
1238
1469
|
classifier_context_window_size?: number;
|
|
1239
1470
|
/** What classifies the request when the LLM classifier errors, times out, or returns an unparseable response. 'heuristic' runs the local complexity scorer, which is right when the classifier grades complexity too. 'default_model' skips scoring and routes to default_model, which is what a classifier on some other taxonomy wants: a prompt that grades data sensitivity has no use for a complexity score, and scoring one produces a tier unrelated to what the operator configured. Requires default_model when set to 'default_model'. Only applies when classifier_type is 'llm', 'custom', or 'heuristic_first'. */
|
|
1240
1471
|
classifier_fallback?: RequestComplexityRouterConfigClassifierFallback | (string & {});
|
|
1241
|
-
/** Configuration for the LLM classifier; required when classifier_type is 'llm'
|
|
1472
|
+
/** Configuration for the LLM classifier; required when classifier_type is 'llm', 'capability', 'heuristic_first' or 'hybrid' */
|
|
1242
1473
|
classifier_llm_config?: ClassifierLLMConfig | null;
|
|
1243
1474
|
/** Not settable over HTTP; the classifier plugin is a runtime object */
|
|
1244
1475
|
classifier_plugin?: unknown | null;
|
|
1245
1476
|
/** Timeout budget for the classifier plugin call, in milliseconds. On expiry the fallback path decides the tier. Only applies when classifier_type is 'custom'. */
|
|
1246
1477
|
classifier_plugin_timeout_ms?: number;
|
|
1247
|
-
/** Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier plugin,
|
|
1478
|
+
/** Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, an LLM tier-selection call, a Switchyard-compatible capability forecast, a joint Fuse V2 forecast, a custom classifier plugin, 'heuristic_first', which scores locally and only pays for the LLM classifier when the local scorer does not confidently land a cheap tier, or 'hybrid', which trusts the local scorer everywhere except when its score lands near a tier boundary, or 'jev', a TypeSafe AI Jev structured choice call */
|
|
1248
1479
|
classifier_type?: RequestComplexityRouterConfigClassifierType | (string & {});
|
|
1249
1480
|
/** Keywords indicating code-related content */
|
|
1250
1481
|
code_keywords?: RequestComplexityRouterConfigCodeKeywordsList | null;
|
|
1482
|
+
/** Fraction of a model's declared context window the estimated prompt must fit within. The token count is an estimate, so fitting against the full window would dispatch prompts that the provider's own tokenizer then rejects; 0.95 leaves room for that drift plus the response tokens. */
|
|
1483
|
+
context_window_escalation_buffer?: number;
|
|
1484
|
+
/** Named dimensions added to the heuristic-v1 score. Each contributes its inline weight once when any keyword matches the current ask or a case-insensitive regex matches its first 2048 characters; scoring_mode 'match_count' instead grades half weight for one distinct matcher and full for two or more. Regex quantifiers repeat one character or class at most 64 times. Unbounded quantifiers, repeated groups, backreferences and lookarounds are rejected. Conservative work limits include alternation paths, repeat lengths and subsequent matching: 2048 units per pattern, 8192 across the router. Only heuristic, heuristic_first and hybrid accept this field. Uses the existing heuristic tuning quota. */
|
|
1485
|
+
custom_dimensions?: RequestComplexityRouterConfigCustomDimensionsList;
|
|
1251
1486
|
/** Domain-specific technical keywords appended to the effective base list (technical_keywords if set, otherwise DEFAULT_TECHNICAL_KEYWORDS). Order is preserved; duplicates are removed case-insensitively against the base list and within this list. */
|
|
1252
1487
|
custom_technical_keywords?: RequestComplexityRouterConfigCustomTechnicalKeywordsList | null;
|
|
1253
1488
|
/** Default model to use if tier cannot be determined */
|
|
1254
1489
|
default_model?: string | null;
|
|
1255
|
-
/** When True and a session_id is resolvable
|
|
1490
|
+
/** When True and a client session_id is resolvable, reuse the session's chosen model for each classified tier and its deployment within each model group. With session_affinity off, every turn is still classified: moving to another tier leaves the previous tier's model pin intact for a later return. Pins yield to current candidate, context, modality, and availability constraints. Adaptive selection chooses the initial model from its eligible pool, then reuses that choice per tier. This reduces avoidable provider prompt-cache misses; it does not guarantee cache hits. Set False to select models and load-balance deployments on every turn, unless session_affinity or user_turn classification requires a pin. Inert without a client session_id and suppressed when plugins are configured. */
|
|
1256
1491
|
deployment_affinity?: boolean;
|
|
1257
1492
|
/** Weights for each scoring dimension */
|
|
1258
1493
|
dimension_weights?: RequestComplexityRouterConfigDimensionWeightsMap;
|
|
1259
1494
|
/** Embedding model (LiteLLM model name) used when semantic_keyword_matching is enabled */
|
|
1260
1495
|
embedding_model?: string | null;
|
|
1496
|
+
/** Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Set false to dispatch on complexity alone, as before. */
|
|
1497
|
+
enable_context_window_escalation?: boolean;
|
|
1498
|
+
/** Add NON_REASONING as a fifth built-in tier below SIMPLE, for operational agent traffic that relays or reformats information rather than reasoning about it. Off by default: turning it on adds a rung to this router's ladder, a bullet to the LLM classifier's rubric, and a value the classifier may return, all of which move tier decisions and spend on an already-deployed router. Requires an LLM, Jev, or custom classifier plugin, since the heuristic scorers cannot produce the tier, and a model in `tiers` under the NON_REASONING key. Escalation still walks up from it, and it is never the savings baseline or a `heuristic_v2` prediction. */
|
|
1499
|
+
enable_non_reasoning_tier?: boolean;
|
|
1261
1500
|
/** Case-sensitive phrases a user can include to force a bump to the next-higher complexity tier when they aren't satisfied with results (they can force a stronger model, but not choose which one). Defaults to ['LITELLM ESCALATE'] when unset; set to an empty list to disable. */
|
|
1262
1501
|
escalation_keywords?: RequestComplexityRouterConfigEscalationKeywordsList | null;
|
|
1263
1502
|
/** Tier routed to when the LLM classifier fails (timeout, provider error, or an unparseable reply). Required with tier_definitions and must name a defined tier; the heuristic scorer cannot produce custom tiers, so this replaces the heuristic fallback for custom tier sets. */
|
|
1264
1503
|
fallback_tier?: string | null;
|
|
1265
1504
|
/** The highest tier the local scorer may decide on its own; required when classifier_type is 'heuristic_first' and rejected otherwise. A request whose heuristic tier is at or below this one skips the LLM classifier and routes straight to that heuristic tier, so the classifier call is only paid for on traffic the scorer could not place cheaply. The scorer must also have produced at least one signal: a prompt where no dimension fired scores 0.0 and would otherwise land SIMPLE by default rather than by evidence, which is how a chained router would silently send unclassified traffic to the cheapest model. Names a built-in tier, and may not name the highest one, since that would make the LLM classifier unreachable. */
|
|
1266
1505
|
heuristic_first_max_tier?: string | null;
|
|
1506
|
+
/** Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled UltraFeedback artifact is selected by default; an inline trained artifact may replace it */
|
|
1507
|
+
heuristic_v2_artifact?: RequestComplexityRouterConfigHeuristicV2Artifact;
|
|
1267
1508
|
/** Additional case-sensitive literal sentinels that mark a request as client housekeeping, on top of the built-in conversation-title ones. For clients whose wording the built-ins don't cover, or after a client release changes its strings. */
|
|
1268
1509
|
housekeeping_patterns?: RequestComplexityRouterConfigHousekeepingPatternsList | null;
|
|
1510
|
+
/** How close to a tier boundary a heuristic score has to land before the LLM classifier breaks the tie; required when classifier_type is 'hybrid' and rejected otherwise. Everything further than this from every active boundary routes on the scorer's own tier with no classifier call, at any tier, which is what separates 'hybrid' from 'heuristic_first' and its cheap-tier ceiling. A prompt where no dimension fired still goes to the classifier, since the scorer has no opinion to be near a boundary with. 0 escalates only scores sitting exactly on a boundary. */
|
|
1511
|
+
hybrid_boundary_margin?: number | null;
|
|
1512
|
+
jev_classifier_config?: JevClassifierConfig | null;
|
|
1269
1513
|
/** Rules that force a specific tier when their keywords match the prompt */
|
|
1270
1514
|
keyword_tier_rules?: RequestComplexityRouterConfigKeywordTierRulesList | null;
|
|
1515
|
+
/** Experimental joint task-demand and solver-capability forecasting for classifier_type llm_v2. */
|
|
1516
|
+
llm_v2_config?: LLMV2Config | null;
|
|
1271
1517
|
/** Minimum cosine similarity for a semantic keyword match */
|
|
1272
1518
|
match_threshold?: number;
|
|
1519
|
+
/** Set max_tokens on every routed request to the output ceiling of the tier model it lands on, replacing whatever the caller sent. A caller behind an auto-router cannot pick one value that fits every tier: the smallest tier's ceiling starves a bigger tier's thinking budget, and a bigger tier's ceiling is rejected by the smallest. The ceiling is the smallest max_output_tokens across the tier model's deployments, read from each deployment's model_info and then the model cost map; a tier model with a deployment whose ceiling is unknown keeps the caller's value. A max_tokens, max_completion_tokens or max_output_tokens in the tier's own litellm_params still wins. Set false to forward the caller's value unchanged. */
|
|
1520
|
+
max_tokens_from_tier_model?: boolean;
|
|
1521
|
+
/** Let modality_routing replace a kept session-affinity pin on the turns that carry an image. Without this, a session pinned to a text-only model fails every image turn with a provider 400, since the pin is exempt from the modality gate. When enabled, such a turn routes to a capable model for that request only and the stored pin is left untouched, so the next text turn replays the session's own model; the override is reported as cause modality_pin_override and is never itself pinned. Inert unless modality_routing is also enabled. */
|
|
1522
|
+
modality_pin_override?: boolean;
|
|
1523
|
+
/** Route image-bearing requests only to models that can accept image input. The classifier reads text alone, so an image request whose text classifies cheap otherwise lands on a text-only model and fails with a provider 400. When enabled, a routed model explicitly declared supports_vision false (deployment model_info or the model cost map; unmapped names stay routable) is replaced by the nearest HIGHER tier holding a capable model, then default_model, else a clear 400. A kept session-affinity pin still wins even when an image arrives, unless modality_pin_override is also enabled. */
|
|
1524
|
+
modality_routing?: boolean;
|
|
1273
1525
|
/** When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to at least this tier: the classified tier still wins when it is higher, and the floor also overrides a session-affinity pin to a lower tier for exactly the turns carrying the sentinel, without rewriting the pin -- the first turn after plan mode exits routes as if plan mode had never happened. Names a built-in tier, or with tier_definitions set, one of the defined tier names (list order is ascending severity, same as keyword_tier_rules). Unset disables detection entirely. The sentinels ride in client-injected prompt text, so a caller who pastes one can spend up to this tier's models -- never down, and never outside the configured pools. */
|
|
1274
1526
|
plan_mode_min_tier?: string | null;
|
|
1275
1527
|
/** Additional case-sensitive literal sentinels that mark a request as plan mode, on top of the built-in Claude Code and Copilot ones. For clients whose plan-mode wording the built-ins don't cover, or after a client release changes its strings. */
|
|
@@ -1280,7 +1532,7 @@ export interface RequestComplexityRouterConfig {
|
|
|
1280
1532
|
reasoning_keywords?: RequestComplexityRouterConfigReasoningKeywordsList | null;
|
|
1281
1533
|
/** Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a request the scorer placed in the cheapest tier; 0 restores the unconditional override */
|
|
1282
1534
|
reasoning_override_min_score?: number | null;
|
|
1283
|
-
/** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in
|
|
1535
|
+
/** Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in system-reminder pair and the Codex envelope pairs enabled for Codex user agents, so list every built-in pair your harness also emits. Matching is case-insensitive. */
|
|
1284
1536
|
reminder_markers?: RequestComplexityRouterConfigReminderMarkersList | null;
|
|
1285
1537
|
/** Return the resolved raw model name in the response model field instead of the client-requested complexity-router alias */
|
|
1286
1538
|
return_raw_model_name?: boolean;
|
|
@@ -1290,15 +1542,21 @@ export interface RequestComplexityRouterConfig {
|
|
|
1290
1542
|
semantic_keyword_matching?: boolean;
|
|
1291
1543
|
/** When True and a session_id is resolvable on the request, pin the model chosen on the session's first turn and reuse it for every later turn, skipping re-classification. Off by default so every turn is classified on its own merits and routed to the cheapest adequate tier. Set True to keep a multi-turn session on one model, which preserves provider prompt caches and avoids cross-model conversation-history errors. Always implies the deployment pin regardless of deployment_affinity: the session sticks to one deployment of the pinned model, since freezing the model while re-shuffling its deployments would still go cache-cold. */
|
|
1292
1544
|
session_affinity?: boolean;
|
|
1293
|
-
/** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity deployment
|
|
1545
|
+
/** TTL for the session affinity pin; refreshed on every cache hit. Bounds both the session_affinity model pin and the deployment_affinity per-tier model and deployment pins, so it measures idle time for the session's routing decisions rather than total session length */
|
|
1294
1546
|
session_affinity_ttl_seconds?: number;
|
|
1295
1547
|
/** Keywords indicating simple/basic queries */
|
|
1296
1548
|
simple_keywords?: RequestComplexityRouterConfigSimpleKeywordsList | null;
|
|
1549
|
+
/** Escalate mid-task to the next-higher configured tier when the assistant's own recent tool calls look stuck: the newest tool call repeats, or errors, at least stall_escalation_repeat_threshold times across the last stall_escalation_window calls. Both tests are anchored on the newest call, so a task that tried the same thing a few times and then moved on is not escalated on the strength of those older calls alone, while a retry loop broken up by an unrelated lookup still counts. One tier at most, on the same ladder escalation_keywords bumps along, and never above the highest configured tier. Detection re-runs on every classified turn from the tool calls visible in that request, so it needs no state and nothing survives past the task. Mutually exclusive with session_affinity and classification_mode='user_turn', which both replay a held routing decision instead of classifying most turns, so this would never see the tool calls to look at. Off by default. */
|
|
1550
|
+
stall_escalation_enabled?: boolean;
|
|
1551
|
+
/** How many of the last stall_escalation_window tool calls must repeat the newest call, or must have errored alongside it, before the task counts as stalled. Must not exceed stall_escalation_window, or the condition could never be reached. */
|
|
1552
|
+
stall_escalation_repeat_threshold?: number;
|
|
1553
|
+
/** How many of the assistant's most recent tool calls stall detection looks at, oldest ones dropped as new calls happen. Counted across the whole visible conversation rather than reset at the newest human ask, so evidence from before a plain follow-up message like 'try again' is still visible on the turn after it. */
|
|
1554
|
+
stall_escalation_window?: number;
|
|
1297
1555
|
/** Keywords indicating technical content */
|
|
1298
1556
|
technical_keywords?: RequestComplexityRouterConfigTechnicalKeywordsList | null;
|
|
1299
1557
|
/** Score boundaries between tiers. These keys (simple_medium, medium_complex, complex_reasoning) name the gaps between the default tier names and are not renameable by tier_labels; they are scorer knobs persisted by name on every routing decision */
|
|
1300
1558
|
tier_boundaries?: RequestComplexityRouterConfigTierBoundariesMap;
|
|
1301
|
-
/** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
|
|
1559
|
+
/** Operator-defined tier set replacing the built-in SIMPLE/MEDIUM/COMPLEX/REASONING. Each entry's name becomes a value the LLM classifier can return and its description becomes that tier's rubric bullet; entries named after a built-in tier may omit the description and inherit the built-in criteria. List order is ascending severity and decides which tier wins when several keyword_tier_rules match. Requires classifier_type 'llm', 'jev' or 'custom', a fallback_tier, and `tiers` keys matching the defined names exactly. Escalation, adaptive selection, session affinity, plugins, tier_labels, and the calibration-example rubric presets are unavailable with a custom tier set: the first four are built on the built-in tier ladder, and the last two rename or exemplify tiers the set replaces. */
|
|
1302
1560
|
tier_definitions?: RequestComplexityRouterConfigTierDefinitionsList | null;
|
|
1303
1561
|
/** Score penalty per tier-step away from the classified tier when adaptive=True */
|
|
1304
1562
|
tier_distance_penalty?: number;
|
|
@@ -1351,8 +1609,23 @@ export interface PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest {
|
|
|
1351
1609
|
tools?: PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequestToolsList | null;
|
|
1352
1610
|
}
|
|
1353
1611
|
export declare const PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest: S.Schema<PostPreviewAutoRouterRoutingAutoRouterTestRoutingRequest>;
|
|
1354
|
-
export type StandardLoggingRoutingDecisionCause = "heuristic_scorer" | "reasoning_override" | "llm_classifier" | "heuristic_first_short_circuit" | "classifier_plugin" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "session_affinity_pin" | "session_affinity_escalation" | "default_fallback" | "keyword" | "quality_tier" | "bandit";
|
|
1612
|
+
export type StandardLoggingRoutingDecisionCause = "heuristic_scorer" | "heuristic_v2" | "reasoning_override" | "llm_classifier" | "capability_classifier" | "jev_classifier" | "llm_v2_classifier" | "llm_v2_fallback" | "heuristic_first_short_circuit" | "hybrid_short_circuit" | "classifier_plugin" | "classifier_fallback" | "capability_classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "housekeeping" | "modality_escalation" | "modality_pin_override" | "health_failover" | "health_default_fallback" | "session_affinity_pin" | "session_affinity_escalation" | "user_turn_continuation" | "default_fallback" | "keyword" | "quality_tier" | "bandit";
|
|
1355
1613
|
export declare const StandardLoggingRoutingDecisionCause: any;
|
|
1614
|
+
export type StandardLoggingRoutingDecisionClassifierProbabilitiesMap = {
|
|
1615
|
+
[key: string]: number | undefined;
|
|
1616
|
+
};
|
|
1617
|
+
export declare const StandardLoggingRoutingDecisionClassifierProbabilitiesMap: S.Schema<StandardLoggingRoutingDecisionClassifierProbabilitiesMap>;
|
|
1618
|
+
export type StandardLoggingHeuristicV2ForecastProbabilitiesMap = {
|
|
1619
|
+
[key: string]: number | undefined;
|
|
1620
|
+
};
|
|
1621
|
+
export declare const StandardLoggingHeuristicV2ForecastProbabilitiesMap: S.Schema<StandardLoggingHeuristicV2ForecastProbabilitiesMap>;
|
|
1622
|
+
export interface StandardLoggingHeuristicV2Forecast {
|
|
1623
|
+
predicted_tier: string;
|
|
1624
|
+
probabilities: StandardLoggingHeuristicV2ForecastProbabilitiesMap;
|
|
1625
|
+
request_type: string;
|
|
1626
|
+
threshold: number;
|
|
1627
|
+
}
|
|
1628
|
+
export declare const StandardLoggingHeuristicV2Forecast: S.Schema<StandardLoggingHeuristicV2Forecast>;
|
|
1356
1629
|
export type StandardLoggingRoutingDecisionRouterType = "complexity" | "adaptive" | "quality";
|
|
1357
1630
|
export declare const StandardLoggingRoutingDecisionRouterType: any;
|
|
1358
1631
|
export type StandardLoggingRoutingDecisionSignalsList = Array<string>;
|
|
@@ -1371,11 +1644,29 @@ export declare const StandardLoggingRoutingDecisionTierLitellmParamsMap: S.Schem
|
|
|
1371
1644
|
/** Per-request provenance for a pre-routing strategy (auto-router) decision. */
|
|
1372
1645
|
export interface StandardLoggingRoutingDecision {
|
|
1373
1646
|
cause?: StandardLoggingRoutingDecisionCause;
|
|
1647
|
+
classifier_calibrated_capable_p_solve?: number;
|
|
1648
|
+
classifier_calibrated_efficient_p_solve?: number;
|
|
1649
|
+
classifier_calibrated_p_solve?: number;
|
|
1650
|
+
classifier_calibration_version?: string;
|
|
1651
|
+
classifier_capability_boundary?: string;
|
|
1652
|
+
classifier_capable_p_solve?: number;
|
|
1653
|
+
classifier_confidence?: number;
|
|
1374
1654
|
classifier_cost?: number;
|
|
1655
|
+
classifier_crux?: string;
|
|
1656
|
+
classifier_efficient_p_solve?: number;
|
|
1657
|
+
classifier_max_quality_gap?: number;
|
|
1375
1658
|
classifier_model?: string;
|
|
1659
|
+
classifier_p_solve?: number;
|
|
1660
|
+
classifier_primary_rule?: string;
|
|
1661
|
+
classifier_probabilities?: StandardLoggingRoutingDecisionClassifierProbabilitiesMap;
|
|
1662
|
+
classifier_prompt_version?: string;
|
|
1663
|
+
classifier_threshold?: number;
|
|
1664
|
+
context_escalated?: boolean;
|
|
1665
|
+
context_escalation_original_tier?: string;
|
|
1376
1666
|
conversation_continuing?: boolean;
|
|
1377
1667
|
escalated?: boolean;
|
|
1378
1668
|
escalation_keyword?: string;
|
|
1669
|
+
heuristic_v2_forecast?: StandardLoggingHeuristicV2Forecast;
|
|
1379
1670
|
matched_keyword?: string;
|
|
1380
1671
|
reasoning_override_min_score?: number;
|
|
1381
1672
|
request_type?: string;
|
|
@@ -1551,7 +1842,7 @@ export type GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetError
|
|
|
1551
1842
|
/** Get Model Cost Map Reload Status ADMIN ONLY / MASTER KEY Only Endpoint Get the status of the scheduled model cost map reload job. */
|
|
1552
1843
|
export declare const getModelCostMapReloadStatusScheduleModelCostMapReloadStatusGet: API.OperationMethod<GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetRequest, GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetResponse, GetModelCostMapReloadStatusScheduleModelCostMapReloadStatusGetError, LitellmOpContext>;
|
|
1553
1844
|
export type GetModelCostMapSourceModelCostMapSourceGetError = LitellmOpError;
|
|
1554
|
-
/** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - model_count: number of models in the currently loaded cost map */
|
|
1845
|
+
/** Get Model Cost Map Source ADMIN ONLY / MASTER KEY Only Endpoint Returns information about where the current model cost/pricing data was loaded from. Response fields: - source: "local" (bundled backup) or "remote" (fetched from URL) - url: the remote URL that was attempted (null when env-forced local) - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage - fallback_reason: human-readable reason why remote failed (null on success) - loaded_at: when this pod last loaded the map - source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it - etag: the ETag of the remote fetch (null for the bundled backup) - model_count: number of models in the currently loaded cost map */
|
|
1555
1846
|
export declare const getModelCostMapSourceModelCostMapSourceGet: API.OperationMethod<GetModelCostMapSourceModelCostMapSourceGetRequest, GetModelCostMapSourceModelCostMapSourceGetResponse, GetModelCostMapSourceModelCostMapSourceGetError, LitellmOpContext>;
|
|
1556
1847
|
export type GetModelDeprecationsModelDeprecationError = UnprocessableEntity | LitellmOpError;
|
|
1557
1848
|
/** Model Deprecations List models with known deprecation/sunset dates, bucketed by urgency. Reads `deprecation_date` metadata from `model_prices_and_context_window.json` (and any per-deployment `model_info.deprecation_date` overrides) for the models configured on this proxy. Parameters: warn_within_days: Window (in days) used to bucket "imminent" models, 30 by default. Returns: A payload with three lists of `ModelDeprecationInfo` entries: - `deprecated`: deprecation date is in the past, so these requests may fail at any time. - `imminent`: deprecation date is within `warn_within_days` from today. - `upcoming`: deprecation date is further out. Example: ```shell curl -X GET 'http://localhost:4000/model/deprecations' \ -H 'Authorization: Bearer sk-1234' ``` */
|
|
@@ -1566,16 +1857,16 @@ export type GetModelInfoModelsModelIdError = UnprocessableEntity | LitellmOpErro
|
|
|
1566
1857
|
/** Model Info Retrieve information about a specific model accessible to your API key. Returns model details only if the model is available to your API key/team. Returns 404 if the model doesn't exist or is not accessible. Follows OpenAI API specification for individual model retrieval. https://platform.openai.com/docs/api-reference/models/retrieve Query parameters mirror `/v1/models` so the same caller context (team scoping, health filtering, paused deployments) drives both endpoints; the listing's public id must resolve to the same internal deployment here. */
|
|
1567
1858
|
export declare const getModelInfoModelsModelId: API.OperationMethod<GetModelInfoModelsModelIdRequest, GetModelInfoModelsModelIdResponse, GetModelInfoModelsModelIdError, LitellmOpContext>;
|
|
1568
1859
|
export type GetModelInfoV1ModelInfoError = UnprocessableEntity | LitellmOpError;
|
|
1569
|
-
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns:
|
|
1860
|
+
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
|
|
1570
1861
|
export declare const getModelInfoV1ModelInfo: API.OperationMethod<GetModelInfoV1ModelInfoRequest, GetModelInfoV1ModelInfoResponse, GetModelInfoV1ModelInfoError, LitellmOpContext>;
|
|
1571
1862
|
export type GetModelInfoV1ModelsModelIdError = UnprocessableEntity | LitellmOpError;
|
|
1572
1863
|
/** Model Info Retrieve information about a specific model accessible to your API key. Returns model details only if the model is available to your API key/team. Returns 404 if the model doesn't exist or is not accessible. Follows OpenAI API specification for individual model retrieval. https://platform.openai.com/docs/api-reference/models/retrieve Query parameters mirror `/v1/models` so the same caller context (team scoping, health filtering, paused deployments) drives both endpoints; the listing's public id must resolve to the same internal deployment here. */
|
|
1573
1864
|
export declare const getModelInfoV1ModelsModelId: API.OperationMethod<GetModelInfoV1ModelsModelIdRequest, GetModelInfoV1ModelsModelIdResponse, GetModelInfoV1ModelsModelIdError, LitellmOpContext>;
|
|
1574
1865
|
export type GetModelInfoV1V1ModelInfoError = UnprocessableEntity | LitellmOpError;
|
|
1575
|
-
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns:
|
|
1866
|
+
/** Model Info V1 Provides more info about each model in /models, including config.yaml descriptions (except api key and api base) Parameters: litellm_model_id: Optional[str] = None (this is the value of `x-litellm-model-id` returned in response headers) - When litellm_model_id is passed, it will return the info for that specific model - When litellm_model_id is not passed, it will return the info for all models - include_team_models: When true, filter to deployments the caller can use (same as /v2/model/info). - teamId: Filter to models accessible by the given team. - healthy_only: When true, hide models whose backing deployments are all marked unhealthy by background health checks, matching `/v1/models?healthy_only=true`. Set `general_settings.model_list_healthy_only: true` to apply this to every caller without the query parameter. Requires `background_health_checks: true`, plus either `model_list_healthy_only` or `enable_health_check_routing` to keep deployment health state cached; without health state the listing is returned unfiltered (fail open). Ignored when `litellm_model_id` is passed, since that is a direct lookup of one deployment rather than a listing. Hiding is presentation-only: a hidden model can still be called directly. Each model in the list response includes `model_info.access_via_team_ids` and `model_info.direct_access` when the proxy database is connected. Returns: A JSON response whose `data` list holds one entry per model. Example Response: ```json { "data": [ { "model_name": "fake-openai-endpoint", "litellm_params": { "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", "model": "openai/fake" }, "model_info": { "id": "112f74fab24a7a5245d2ced3536dd8f5f9192c57ee6e332af0f0512e08bed5af", "db_model": false } } ] } ``` */
|
|
1576
1867
|
export declare const getModelInfoV1V1ModelInfo: API.OperationMethod<GetModelInfoV1V1ModelInfoRequest, GetModelInfoV1V1ModelInfoResponse, GetModelInfoV1V1ModelInfoError, LitellmOpContext>;
|
|
1577
1868
|
export type GetModelInfoV2V2ModelInfoError = UnprocessableEntity | LitellmOpError;
|
|
1578
|
-
/** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
|
|
1869
|
+
/** Model Info V2 Paginated model metadata for proxy deployments (pricing, provider, team access). Returns configured router deployments with enriched `model_info` (costs, provider, context window, etc.). Sensitive fields such as API keys and api_base are omitted. Query parameters: model: Filter to a single public `model_name`. user_models_only: When true, only return models created by the calling user. include_team_models: When true, populate `access_via_team_ids` and `direct_access` on each model and filter to deployments the caller can use. page / size: Pagination controls (defaults: page=1, size=50). search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. access_group: Only return deployments in this model access group. wildcard_only: Only return deployments whose `model_name` contains `*`. Example request: ``` curl -X GET 'http://localhost:4000/v2/model/info?include_team_models=true&page=1&size=50' \ --header 'Authorization: Bearer sk-1234' ``` Example response: ```json { "data": [ { "model_name": "gpt-4", "litellm_params": {"model": "openai/gpt-4.1"}, "model_info": { "id": "abc123", "litellm_provider": "openai", "access_via_team_ids": ["team-1"], "direct_access": true } } ], "total_count": 1, "current_page": 1, "total_pages": 1, "size": 50 } ``` */
|
|
1579
1870
|
export declare const getModelInfoV2V2ModelInfo: API.OperationMethod<GetModelInfoV2V2ModelInfoRequest, GetModelInfoV2V2ModelInfoResponse, GetModelInfoV2V2ModelInfoError, LitellmOpContext>;
|
|
1580
1871
|
export type GetModelMetricsExceptionsModelMetricsExceptionError = UnprocessableEntity | LitellmOpError;
|
|
1581
1872
|
/** Model Metrics Exceptions View number of failed requests per model on config.yaml */
|
|
@@ -1644,6 +1935,6 @@ export type UpdateUsefulLinksModelHubUpdateUsefulLinksPostError = UnprocessableE
|
|
|
1644
1935
|
/** Update Useful Links Update useful links */
|
|
1645
1936
|
export declare const updateUsefulLinksModelHubUpdateUsefulLinksPost: API.OperationMethod<UpdateUsefulLinksModelHubUpdateUsefulLinksPostRequest, UpdateUsefulLinksModelHubUpdateUsefulLinksPostResponse, UpdateUsefulLinksModelHubUpdateUsefulLinksPostError, LitellmOpContext>;
|
|
1646
1937
|
export type ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError = UnprocessableEntity | LitellmOpError;
|
|
1647
|
-
/** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save.
|
|
1938
|
+
/** Validate Complexity Router Config Validate a complexity-router config without saving it. Runs the same check every write path runs (the router's own pydantic model), so a form can show the backend's exact verdict while the operator is still editing rather than after a rejected save. Uses the same team opt-in and model-access checks as configuration writes for members. Nothing is created, routed, or billed. */
|
|
1648
1939
|
export declare const validateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPost: API.OperationMethod<ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostRequest, ComplexityRouterConfigValidationResponse, ValidateComplexityRouterConfigAutoRouterValidateComplexityRouterConfigPostError, LitellmOpContext>;
|
|
1649
1940
|
//# sourceMappingURL=model_management.d.ts.map
|