@homeflare/distilled-litellm 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -12
- package/dist/services/a2a.js +1 -1
- package/dist/services/a2a_registration.js +1 -1
- package/dist/services/access_groups.d.ts +18 -0
- package/dist/services/access_groups.d.ts.map +1 -1
- package/dist/services/access_groups.js +15 -1
- package/dist/services/access_groups.js.map +1 -1
- package/dist/services/adaptive_router.js +1 -1
- package/dist/services/agents.d.ts +10 -0
- package/dist/services/agents.d.ts.map +1 -1
- package/dist/services/agents.js +10 -1
- package/dist/services/agents.js.map +1 -1
- package/dist/services/alerting.js +1 -1
- package/dist/services/anthropic_passthrough.js +1 -1
- package/dist/services/anthropic_skills.d.ts +7 -1
- package/dist/services/anthropic_skills.d.ts.map +1 -1
- package/dist/services/anthropic_skills.js +6 -2
- package/dist/services/anthropic_skills.js.map +1 -1
- package/dist/services/assistants.js +1 -1
- package/dist/services/audio.js +1 -1
- package/dist/services/audit_logging.d.ts +2 -0
- package/dist/services/audit_logging.d.ts.map +1 -1
- package/dist/services/audit_logging.js +2 -1
- package/dist/services/audit_logging.js.map +1 -1
- package/dist/services/auto_router.d.ts +162 -62
- package/dist/services/auto_router.d.ts.map +1 -1
- package/dist/services/auto_router.js +92 -31
- package/dist/services/auto_router.js.map +1 -1
- package/dist/services/batch.js +1 -1
- package/dist/services/beta_agents.js +1 -1
- package/dist/services/beta_mcp.js +1 -1
- package/dist/services/budget_management.d.ts +8 -3
- package/dist/services/budget_management.d.ts.map +1 -1
- package/dist/services/budget_management.js +7 -4
- package/dist/services/budget_management.js.map +1 -1
- package/dist/services/budget_spend_tracking.d.ts +24 -5
- package/dist/services/budget_spend_tracking.d.ts.map +1 -1
- package/dist/services/budget_spend_tracking.js +17 -4
- package/dist/services/budget_spend_tracking.js.map +1 -1
- package/dist/services/cache_settings.js +1 -1
- package/dist/services/caching.js +1 -1
- package/dist/services/chat_completions.js +1 -1
- package/dist/services/claude_code_marketplace.d.ts +11 -10
- package/dist/services/claude_code_marketplace.d.ts.map +1 -1
- package/dist/services/claude_code_marketplace.js +8 -6
- package/dist/services/claude_code_marketplace.js.map +1 -1
- package/dist/services/cloudzero.js +1 -1
- package/dist/services/completions.js +1 -1
- package/dist/services/compliance.js +1 -1
- package/dist/services/config_overrides.d.ts +5 -1
- package/dist/services/config_overrides.d.ts.map +1 -1
- package/dist/services/config_overrides.js +3 -1
- package/dist/services/config_overrides.js.map +1 -1
- package/dist/services/config_yaml.d.ts +120 -6
- package/dist/services/config_yaml.d.ts.map +1 -1
- package/dist/services/config_yaml.js +67 -1
- package/dist/services/config_yaml.js.map +1 -1
- package/dist/services/containers.d.ts +8 -2
- package/dist/services/containers.d.ts.map +1 -1
- package/dist/services/containers.js +13 -5
- package/dist/services/containers.js.map +1 -1
- package/dist/services/coordination_redis_settings.js +1 -1
- package/dist/services/cost_tracking.d.ts +91 -1
- package/dist/services/cost_tracking.d.ts.map +1 -1
- package/dist/services/cost_tracking.js +82 -2
- package/dist/services/cost_tracking.js.map +1 -1
- package/dist/services/credential_management.d.ts +16 -3
- package/dist/services/credential_management.d.ts.map +1 -1
- package/dist/services/credential_management.js +13 -4
- package/dist/services/credential_management.js.map +1 -1
- package/dist/services/customer_management.d.ts +25 -2
- package/dist/services/customer_management.d.ts.map +1 -1
- package/dist/services/customer_management.js +22 -3
- package/dist/services/customer_management.js.map +1 -1
- package/dist/services/email_management.js +1 -1
- package/dist/services/embeddings.js +1 -1
- package/dist/services/evals.js +1 -1
- package/dist/services/experimental.js +1 -1
- package/dist/services/fallback_management.js +1 -1
- package/dist/services/files.js +1 -1
- package/dist/services/fine_tuning.js +1 -1
- package/dist/services/gemini_agents.js +1 -1
- package/dist/services/google_genai_endpoints.js +1 -1
- package/dist/services/guardrails.d.ts +80 -7
- package/dist/services/guardrails.d.ts.map +1 -1
- package/dist/services/guardrails.js +37 -1
- package/dist/services/guardrails.js.map +1 -1
- package/dist/services/health.d.ts +2 -2
- package/dist/services/health.d.ts.map +1 -1
- package/dist/services/health.js +2 -2
- package/dist/services/health.js.map +1 -1
- package/dist/services/images.js +1 -1
- package/dist/services/index.d.ts +1 -15
- package/dist/services/index.d.ts.map +1 -1
- package/dist/services/index.js +1 -15
- package/dist/services/index.js.map +1 -1
- package/dist/services/internal_user_management.d.ts +227 -39
- package/dist/services/internal_user_management.d.ts.map +1 -1
- package/dist/services/internal_user_management.js +194 -37
- package/dist/services/internal_user_management.js.map +1 -1
- package/dist/services/invite_links.js +1 -1
- package/dist/services/jwt_mappings.d.ts +3 -0
- package/dist/services/jwt_mappings.d.ts.map +1 -1
- package/dist/services/jwt_mappings.js +4 -1
- package/dist/services/jwt_mappings.js.map +1 -1
- package/dist/services/key_management.d.ts +116 -50
- package/dist/services/key_management.d.ts.map +1 -1
- package/dist/services/key_management.js +95 -40
- package/dist/services/key_management.js.map +1 -1
- package/dist/services/langfuse_passthrough.js +1 -1
- package/dist/services/llm_passthrough.d.ts +1199 -0
- package/dist/services/llm_passthrough.d.ts.map +1 -0
- package/dist/services/llm_passthrough.js +2150 -0
- package/dist/services/llm_passthrough.js.map +1 -0
- package/dist/services/llm_utils.js +1 -1
- package/dist/services/logging_callbacks.js +1 -1
- package/dist/services/mcp_byok_oauth.d.ts +18 -11
- package/dist/services/mcp_byok_oauth.d.ts.map +1 -1
- package/dist/services/mcp_byok_oauth.js +27 -24
- package/dist/services/mcp_byok_oauth.js.map +1 -1
- package/dist/services/mcp_discoverable.d.ts +34 -0
- package/dist/services/mcp_discoverable.d.ts.map +1 -1
- package/dist/services/mcp_discoverable.js +51 -1
- package/dist/services/mcp_discoverable.js.map +1 -1
- package/dist/services/mcp_management.d.ts +90 -2
- package/dist/services/mcp_management.d.ts.map +1 -1
- package/dist/services/mcp_management.js +115 -3
- package/dist/services/mcp_management.js.map +1 -1
- package/dist/services/mcp_rest.d.ts +13 -0
- package/dist/services/mcp_rest.d.ts.map +1 -1
- package/dist/services/mcp_rest.js +10 -2
- package/dist/services/mcp_rest.js.map +1 -1
- package/dist/services/memory_management.d.ts +2 -0
- package/dist/services/memory_management.d.ts.map +1 -1
- package/dist/services/memory_management.js +2 -1
- package/dist/services/memory_management.js.map +1 -1
- package/dist/services/misc.d.ts +86 -1
- package/dist/services/misc.d.ts.map +1 -1
- package/dist/services/misc.js +126 -2
- package/dist/services/misc.js.map +1 -1
- package/dist/services/model_management.d.ts +310 -19
- package/dist/services/model_management.d.ts.map +1 -1
- package/dist/services/model_management.js +240 -7
- package/dist/services/model_management.js.map +1 -1
- package/dist/services/moderations.js +1 -1
- package/dist/services/ocr.js +1 -1
- package/dist/services/open_ai_pass_through.d.ts +35 -90
- package/dist/services/open_ai_pass_through.d.ts.map +1 -1
- package/dist/services/open_ai_pass_through.js +41 -139
- package/dist/services/open_ai_pass_through.js.map +1 -1
- package/dist/services/organization_management.d.ts +21 -1
- package/dist/services/organization_management.d.ts.map +1 -1
- package/dist/services/organization_management.js +20 -2
- package/dist/services/organization_management.js.map +1 -1
- package/dist/services/plugins.js +1 -1
- package/dist/services/policies.d.ts +2 -0
- package/dist/services/policies.d.ts.map +1 -1
- package/dist/services/policies.js +3 -1
- package/dist/services/policies.js.map +1 -1
- package/dist/services/policy_engine.d.ts +6 -0
- package/dist/services/policy_engine.d.ts.map +1 -1
- package/dist/services/policy_engine.js +4 -1
- package/dist/services/policy_engine.js.map +1 -1
- package/dist/services/project_management.d.ts +15 -0
- package/dist/services/project_management.d.ts.map +1 -1
- package/dist/services/project_management.js +14 -1
- package/dist/services/project_management.js.map +1 -1
- package/dist/services/prompts.js +1 -1
- package/dist/services/public.d.ts +124 -0
- package/dist/services/public.d.ts.map +1 -1
- package/dist/services/public.js +158 -1
- package/dist/services/public.js.map +1 -1
- package/dist/services/rag.js +1 -1
- package/dist/services/realtime.d.ts +4 -26
- package/dist/services/realtime.d.ts.map +1 -1
- package/dist/services/realtime.js +7 -57
- package/dist/services/realtime.js.map +1 -1
- package/dist/services/rerank.d.ts +72 -0
- package/dist/services/rerank.d.ts.map +1 -1
- package/dist/services/rerank.js +61 -4
- package/dist/services/rerank.js.map +1 -1
- package/dist/services/responses.d.ts +30 -0
- package/dist/services/responses.d.ts.map +1 -1
- package/dist/services/responses.js +59 -1
- package/dist/services/responses.js.map +1 -1
- package/dist/services/router_settings.js +1 -1
- package/dist/services/rust_control_plane.js +1 -1
- package/dist/services/scim.d.ts +41 -1
- package/dist/services/scim.d.ts.map +1 -1
- package/dist/services/scim.js +60 -2
- package/dist/services/scim.js.map +1 -1
- package/dist/services/search.js +1 -1
- package/dist/services/search_tools.js +1 -1
- package/dist/services/settings.d.ts +88 -0
- package/dist/services/settings.d.ts.map +1 -1
- package/dist/services/settings.js +113 -1
- package/dist/services/settings.js.map +1 -1
- package/dist/services/sso_settings.d.ts +1 -1
- package/dist/services/sso_settings.d.ts.map +1 -1
- package/dist/services/sso_settings.js +1 -1
- package/dist/services/sso_settings.js.map +1 -1
- package/dist/services/tag_management.d.ts +7 -0
- package/dist/services/tag_management.d.ts.map +1 -1
- package/dist/services/tag_management.js +8 -1
- package/dist/services/tag_management.js.map +1 -1
- package/dist/services/team_management.d.ts +265 -17
- package/dist/services/team_management.d.ts.map +1 -1
- package/dist/services/team_management.js +247 -12
- package/dist/services/team_management.js.map +1 -1
- package/dist/services/tools.js +1 -1
- package/dist/services/ui_settings.js +1 -1
- package/dist/services/ui_theme_settings.js +1 -1
- package/dist/services/usage_ai.js +1 -1
- package/dist/services/vantage.js +1 -1
- package/dist/services/vector_store_management.js +1 -1
- package/dist/services/vector_stores.js +1 -1
- package/dist/services/videos.js +1 -1
- package/dist/services/web_socket.d.ts +0 -18
- package/dist/services/web_socket.d.ts.map +1 -1
- package/dist/services/web_socket.js +1 -33
- package/dist/services/web_socket.js.map +1 -1
- package/dist/services/workflow_management.js +1 -1
- package/package.json +1 -1
- package/src/services/a2a.ts +1 -1
- package/src/services/a2a_registration.ts +1 -1
- package/src/services/access_groups.ts +44 -1
- package/src/services/adaptive_router.ts +1 -1
- package/src/services/agents.ts +22 -1
- package/src/services/alerting.ts +1 -1
- package/src/services/anthropic_passthrough.ts +1 -1
- package/src/services/anthropic_skills.ts +12 -2
- package/src/services/assistants.ts +1 -1
- package/src/services/audio.ts +1 -1
- package/src/services/audit_logging.ts +4 -1
- package/src/services/auto_router.ts +303 -93
- package/src/services/batch.ts +1 -1
- package/src/services/beta_agents.ts +1 -1
- package/src/services/beta_mcp.ts +1 -1
- package/src/services/budget_management.ts +12 -4
- package/src/services/budget_spend_tracking.ts +38 -6
- package/src/services/cache_settings.ts +1 -1
- package/src/services/caching.ts +1 -1
- package/src/services/chat_completions.ts +1 -1
- package/src/services/claude_code_marketplace.ts +20 -14
- package/src/services/cloudzero.ts +1 -1
- package/src/services/completions.ts +1 -1
- package/src/services/compliance.ts +1 -1
- package/src/services/config_overrides.ts +8 -2
- package/src/services/config_yaml.ts +251 -6
- package/src/services/containers.ts +29 -11
- package/src/services/coordination_redis_settings.ts +1 -1
- package/src/services/cost_tracking.ts +198 -2
- package/src/services/credential_management.ts +25 -6
- package/src/services/customer_management.ts +49 -3
- package/src/services/email_management.ts +1 -1
- package/src/services/embeddings.ts +1 -1
- package/src/services/evals.ts +1 -1
- package/src/services/experimental.ts +1 -1
- package/src/services/fallback_management.ts +1 -1
- package/src/services/files.ts +1 -1
- package/src/services/fine_tuning.ts +1 -1
- package/src/services/gemini_agents.ts +1 -1
- package/src/services/google_genai_endpoints.ts +1 -1
- package/src/services/guardrails.ts +198 -8
- package/src/services/health.ts +3 -2
- package/src/services/images.ts +1 -1
- package/src/services/index.ts +1 -15
- package/src/services/internal_user_management.ts +565 -103
- package/src/services/invite_links.ts +1 -1
- package/src/services/jwt_mappings.ts +7 -1
- package/src/services/key_management.ts +257 -127
- package/src/services/langfuse_passthrough.ts +1 -1
- package/src/services/llm_passthrough.ts +4542 -0
- package/src/services/llm_utils.ts +1 -1
- package/src/services/logging_callbacks.ts +1 -1
- package/src/services/mcp_byok_oauth.ts +65 -45
- package/src/services/mcp_discoverable.ts +109 -1
- package/src/services/mcp_management.ts +258 -3
- package/src/services/mcp_rest.ts +30 -5
- package/src/services/memory_management.ts +4 -1
- package/src/services/misc.ts +269 -2
- package/src/services/model_management.ts +666 -21
- package/src/services/moderations.ts +1 -1
- package/src/services/ocr.ts +1 -1
- package/src/services/open_ai_pass_through.ts +81 -286
- package/src/services/organization_management.ts +44 -2
- package/src/services/plugins.ts +1 -1
- package/src/services/policies.ts +5 -1
- package/src/services/policy_engine.ts +10 -1
- package/src/services/project_management.ts +33 -1
- package/src/services/prompts.ts +1 -1
- package/src/services/public.ts +356 -1
- package/src/services/rag.ts +1 -1
- package/src/services/realtime.ts +11 -111
- package/src/services/rerank.ts +187 -7
- package/src/services/responses.ts +117 -1
- package/src/services/router_settings.ts +1 -1
- package/src/services/rust_control_plane.ts +1 -1
- package/src/services/scim.ts +134 -3
- package/src/services/search.ts +1 -1
- package/src/services/search_tools.ts +1 -1
- package/src/services/settings.ts +278 -1
- package/src/services/sso_settings.ts +2 -1
- package/src/services/tag_management.ts +15 -1
- package/src/services/team_management.ts +676 -37
- package/src/services/tools.ts +1 -1
- package/src/services/ui_settings.ts +1 -1
- package/src/services/ui_theme_settings.ts +1 -1
- package/src/services/usage_ai.ts +1 -1
- package/src/services/vantage.ts +1 -1
- package/src/services/vector_store_management.ts +1 -1
- package/src/services/vector_stores.ts +1 -1
- package/src/services/videos.ts +1 -1
- package/src/services/web_socket.ts +1 -61
- package/src/services/workflow_management.ts +1 -1
- package/dist/services/anthropic_pass_through.d.ts +0 -70
- package/dist/services/anthropic_pass_through.d.ts.map +0 -1
- package/dist/services/anthropic_pass_through.js +0 -114
- package/dist/services/anthropic_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_eu_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_eu_pass_through.js +0 -114
- package/dist/services/assembly_ai_eu_pass_through.js.map +0 -1
- package/dist/services/assembly_ai_pass_through.d.ts +0 -70
- package/dist/services/assembly_ai_pass_through.d.ts.map +0 -1
- package/dist/services/assembly_ai_pass_through.js +0 -114
- package/dist/services/assembly_ai_pass_through.js.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.d.ts +0 -36
- package/dist/services/aws_comprehend_medical_pass_through.d.ts.map +0 -1
- package/dist/services/aws_comprehend_medical_pass_through.js +0 -56
- package/dist/services/aws_comprehend_medical_pass_through.js.map +0 -1
- package/dist/services/azure_ai_pass_through.d.ts +0 -70
- package/dist/services/azure_ai_pass_through.d.ts.map +0 -1
- package/dist/services/azure_ai_pass_through.js +0 -112
- package/dist/services/azure_ai_pass_through.js.map +0 -1
- package/dist/services/azure_pass_through.d.ts +0 -70
- package/dist/services/azure_pass_through.d.ts.map +0 -1
- package/dist/services/azure_pass_through.js +0 -107
- package/dist/services/azure_pass_through.js.map +0 -1
- package/dist/services/bedrock_pass_through.d.ts +0 -70
- package/dist/services/bedrock_pass_through.d.ts.map +0 -1
- package/dist/services/bedrock_pass_through.js +0 -114
- package/dist/services/bedrock_pass_through.js.map +0 -1
- package/dist/services/cohere_pass_through.d.ts +0 -70
- package/dist/services/cohere_pass_through.d.ts.map +0 -1
- package/dist/services/cohere_pass_through.js +0 -112
- package/dist/services/cohere_pass_through.js.map +0 -1
- package/dist/services/cursor_pass_through.d.ts +0 -70
- package/dist/services/cursor_pass_through.d.ts.map +0 -1
- package/dist/services/cursor_pass_through.js +0 -112
- package/dist/services/cursor_pass_through.js.map +0 -1
- package/dist/services/google_ai_studio_pass_through.d.ts +0 -70
- package/dist/services/google_ai_studio_pass_through.d.ts.map +0 -1
- package/dist/services/google_ai_studio_pass_through.js +0 -112
- package/dist/services/google_ai_studio_pass_through.js.map +0 -1
- package/dist/services/milvus_pass_through.d.ts +0 -70
- package/dist/services/milvus_pass_through.d.ts.map +0 -1
- package/dist/services/milvus_pass_through.js +0 -112
- package/dist/services/milvus_pass_through.js.map +0 -1
- package/dist/services/mistral_pass_through.d.ts +0 -70
- package/dist/services/mistral_pass_through.d.ts.map +0 -1
- package/dist/services/mistral_pass_through.js +0 -114
- package/dist/services/mistral_pass_through.js.map +0 -1
- package/dist/services/vertex_ai_pass_through.d.ts +0 -180
- package/dist/services/vertex_ai_pass_through.d.ts.map +0 -1
- package/dist/services/vertex_ai_pass_through.js +0 -334
- package/dist/services/vertex_ai_pass_through.js.map +0 -1
- package/dist/services/vllm_pass_through.d.ts +0 -70
- package/dist/services/vllm_pass_through.d.ts.map +0 -1
- package/dist/services/vllm_pass_through.js +0 -104
- package/dist/services/vllm_pass_through.js.map +0 -1
- package/dist/services/watsonx_pass_through.d.ts +0 -70
- package/dist/services/watsonx_pass_through.d.ts.map +0 -1
- package/dist/services/watsonx_pass_through.js +0 -114
- package/dist/services/watsonx_pass_through.js.map +0 -1
- package/src/services/anthropic_pass_through.ts +0 -233
- package/src/services/assembly_ai_eu_pass_through.ts +0 -237
- package/src/services/assembly_ai_pass_through.ts +0 -237
- package/src/services/aws_comprehend_medical_pass_through.ts +0 -109
- package/src/services/azure_ai_pass_through.ts +0 -231
- package/src/services/azure_pass_through.ts +0 -227
- package/src/services/bedrock_pass_through.ts +0 -229
- package/src/services/cohere_pass_through.ts +0 -227
- package/src/services/cursor_pass_through.ts +0 -227
- package/src/services/google_ai_studio_pass_through.ts +0 -227
- package/src/services/milvus_pass_through.ts +0 -227
- package/src/services/mistral_pass_through.ts +0 -229
- package/src/services/vertex_ai_pass_through.ts +0 -684
- package/src/services/vllm_pass_through.ts +0 -227
- package/src/services/watsonx_pass_through.ts +0 -229
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.
|
|
1
|
+
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
|
|
2
2
|
import * as S from "@distilled.cloud/core/schema";
|
|
3
3
|
import * as API from "@distilled.cloud/core/api";
|
|
4
4
|
import * as C from "@distilled.cloud/core/category";
|
|
@@ -26,12 +26,15 @@ export interface GetAutoRouterBenchmarksAutoRouterBenchmarksGetRequest {
|
|
|
26
26
|
start_date?: string;
|
|
27
27
|
/** YYYY-MM-DD UTC, inclusive (defaults to today) */
|
|
28
28
|
end_date?: string;
|
|
29
|
+
/** Filter to one virtual key token hash */
|
|
30
|
+
api_key?: string;
|
|
29
31
|
}
|
|
30
32
|
export const GetAutoRouterBenchmarksAutoRouterBenchmarksGetRequest =
|
|
31
33
|
/*@__PURE__*/ S.suspend(() =>
|
|
32
34
|
S.Struct({
|
|
33
35
|
start_date: S.optional(S.String.pipe(T.Query())),
|
|
34
36
|
end_date: S.optional(S.String.pipe(T.Query())),
|
|
37
|
+
api_key: S.optional(S.String.pipe(T.Query())),
|
|
35
38
|
}).pipe(
|
|
36
39
|
T.Http({ method: "GET", uri: "/auto_router/benchmarks", code: 200 }),
|
|
37
40
|
),
|
|
@@ -112,18 +115,25 @@ export interface AutoRouterBenchmarkGroup {
|
|
|
112
115
|
avg_session_seconds: number;
|
|
113
116
|
avg_tokens_per_session: number;
|
|
114
117
|
avg_turns_per_session: number;
|
|
115
|
-
/**
|
|
116
|
-
baseline_spend: number;
|
|
118
|
+
/** Estimated single-model cost for covered turns only */
|
|
119
|
+
baseline_spend: number | null;
|
|
117
120
|
cache: AutoRouterCacheStats;
|
|
121
|
+
/** Recorded LLM classifier cost already included in spend; null when any session turns predate subtotal recording, and zero for an empty window */
|
|
122
|
+
classifier_cost: number | null;
|
|
118
123
|
/** The auto-router alias requests were sent to */
|
|
119
124
|
router_name: string;
|
|
120
125
|
/** complexity, adaptive or quality */
|
|
121
126
|
router_type: string;
|
|
122
|
-
/**
|
|
123
|
-
saved_pct: number;
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
+
/** Covered savings over covered baseline spend, as a percentage */
|
|
128
|
+
saved_pct: number | null;
|
|
129
|
+
/** Average session savings; unavailable unless every turn is covered */
|
|
130
|
+
saved_per_session: number | null;
|
|
131
|
+
/** Signed savings for covered turns only; null when traffic has no current estimates */
|
|
132
|
+
saved_spend: number | null;
|
|
133
|
+
/** Actual spend, including classifier cost, for covered turns only */
|
|
134
|
+
savings_estimated_actual_spend: number;
|
|
135
|
+
/** Turns covered by the current savings estimator; legacy estimates are excluded */
|
|
136
|
+
savings_estimated_turns: number;
|
|
127
137
|
sessions: number;
|
|
128
138
|
/** What the routed traffic actually cost */
|
|
129
139
|
spend: number;
|
|
@@ -136,13 +146,16 @@ export const AutoRouterBenchmarkGroup = /*@__PURE__*/ S.suspend(() =>
|
|
|
136
146
|
avg_session_seconds: S.Number,
|
|
137
147
|
avg_tokens_per_session: S.Number,
|
|
138
148
|
avg_turns_per_session: S.Number,
|
|
139
|
-
baseline_spend: S.Number,
|
|
149
|
+
baseline_spend: S.NullOr(S.Number),
|
|
140
150
|
cache: AutoRouterCacheStats,
|
|
151
|
+
classifier_cost: S.NullOr(S.Number),
|
|
141
152
|
router_name: S.String,
|
|
142
153
|
router_type: S.String,
|
|
143
|
-
saved_pct: S.Number,
|
|
144
|
-
saved_per_session: S.Number,
|
|
145
|
-
saved_spend: S.Number,
|
|
154
|
+
saved_pct: S.NullOr(S.Number),
|
|
155
|
+
saved_per_session: S.NullOr(S.Number),
|
|
156
|
+
saved_spend: S.NullOr(S.Number),
|
|
157
|
+
savings_estimated_actual_spend: S.Number,
|
|
158
|
+
savings_estimated_turns: S.Number,
|
|
146
159
|
sessions: S.Number,
|
|
147
160
|
spend: S.Number,
|
|
148
161
|
tier_turns: S.optional(AutoRouterBenchmarkGroupTierTurnsMap),
|
|
@@ -164,14 +177,21 @@ export interface AutoRouterBenchmarkTotals {
|
|
|
164
177
|
avg_session_seconds: number;
|
|
165
178
|
avg_tokens_per_session: number;
|
|
166
179
|
avg_turns_per_session: number;
|
|
167
|
-
/**
|
|
168
|
-
baseline_spend: number;
|
|
180
|
+
/** Estimated single-model cost for covered turns only */
|
|
181
|
+
baseline_spend: number | null;
|
|
169
182
|
cache: AutoRouterCacheStats;
|
|
170
|
-
/**
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
183
|
+
/** Recorded LLM classifier cost already included in spend; null when any session turns predate subtotal recording, and zero for an empty window */
|
|
184
|
+
classifier_cost: number | null;
|
|
185
|
+
/** Covered savings over covered baseline spend, as a percentage */
|
|
186
|
+
saved_pct: number | null;
|
|
187
|
+
/** Average session savings; unavailable unless every turn is covered */
|
|
188
|
+
saved_per_session: number | null;
|
|
189
|
+
/** Signed savings for covered turns only; null when traffic has no current estimates */
|
|
190
|
+
saved_spend: number | null;
|
|
191
|
+
/** Actual spend, including classifier cost, for covered turns only */
|
|
192
|
+
savings_estimated_actual_spend: number;
|
|
193
|
+
/** Turns covered by the current savings estimator; legacy estimates are excluded */
|
|
194
|
+
savings_estimated_turns: number;
|
|
175
195
|
sessions: number;
|
|
176
196
|
/** What the routed traffic actually cost */
|
|
177
197
|
spend: number;
|
|
@@ -182,11 +202,14 @@ export const AutoRouterBenchmarkTotals = /*@__PURE__*/ S.suspend(() =>
|
|
|
182
202
|
avg_session_seconds: S.Number,
|
|
183
203
|
avg_tokens_per_session: S.Number,
|
|
184
204
|
avg_turns_per_session: S.Number,
|
|
185
|
-
baseline_spend: S.Number,
|
|
205
|
+
baseline_spend: S.NullOr(S.Number),
|
|
186
206
|
cache: AutoRouterCacheStats,
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
207
|
+
classifier_cost: S.NullOr(S.Number),
|
|
208
|
+
saved_pct: S.NullOr(S.Number),
|
|
209
|
+
saved_per_session: S.NullOr(S.Number),
|
|
210
|
+
saved_spend: S.NullOr(S.Number),
|
|
211
|
+
savings_estimated_actual_spend: S.Number,
|
|
212
|
+
savings_estimated_turns: S.Number,
|
|
190
213
|
sessions: S.Number,
|
|
191
214
|
spend: S.Number,
|
|
192
215
|
turns: S.Number,
|
|
@@ -219,6 +242,77 @@ export const AutoRouterBenchmarksResponse = /*@__PURE__*/ S.suspend(() =>
|
|
|
219
242
|
identifier: "AutoRouterBenchmarksResponse",
|
|
220
243
|
}) as any as S.Schema<AutoRouterBenchmarksResponse>;
|
|
221
244
|
|
|
245
|
+
export interface GetAutoRouterSessionAutoRouterSessionGetRequest {
|
|
246
|
+
/** The client session id (x-*-session-id header) the turns were sent under */
|
|
247
|
+
session_id: string;
|
|
248
|
+
}
|
|
249
|
+
export const GetAutoRouterSessionAutoRouterSessionGetRequest =
|
|
250
|
+
/*@__PURE__*/ S.suspend(() =>
|
|
251
|
+
S.Struct({
|
|
252
|
+
session_id: S.String.pipe(T.Query()),
|
|
253
|
+
}).pipe(T.Http({ method: "GET", uri: "/auto_router/session", code: 200 })),
|
|
254
|
+
).annotate({
|
|
255
|
+
identifier: "GetAutoRouterSessionAutoRouterSessionGetRequest",
|
|
256
|
+
}) as any as S.Schema<GetAutoRouterSessionAutoRouterSessionGetRequest>;
|
|
257
|
+
|
|
258
|
+
/** Covered turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both */
|
|
259
|
+
export type AutoRouterSessionResponseBaselineModelsMap = {
|
|
260
|
+
[key: string]: number | undefined;
|
|
261
|
+
};
|
|
262
|
+
export const AutoRouterSessionResponseBaselineModelsMap =
|
|
263
|
+
/*@__PURE__*/ S.Record(
|
|
264
|
+
S.String,
|
|
265
|
+
S.Number,
|
|
266
|
+
) as any as S.Schema<AutoRouterSessionResponseBaselineModelsMap>;
|
|
267
|
+
|
|
268
|
+
/** One auto-routed session as its own key sees it: what the last turn ran on, and what the session cost against the router's savings baseline (the priciest model in its hardest tier). */
|
|
269
|
+
export interface AutoRouterSessionResponse {
|
|
270
|
+
/** The savings baseline most covered turns were priced against, recorded turn by turn, so it still names the counterfactual after the router is reconfigured or removed. None when no turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, which derive no baseline and so report no savings */
|
|
271
|
+
baseline_model: string | null;
|
|
272
|
+
/** Covered turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both */
|
|
273
|
+
baseline_models: AutoRouterSessionResponseBaselineModelsMap;
|
|
274
|
+
/** Estimated single-model cost; unavailable unless every turn is covered */
|
|
275
|
+
baseline_spend: number | null;
|
|
276
|
+
/** The deployment model the most recent turn was routed to */
|
|
277
|
+
last_model: string;
|
|
278
|
+
/** The auto-router alias the session's requests were sent to */
|
|
279
|
+
router_name: string;
|
|
280
|
+
/** complexity, adaptive or quality */
|
|
281
|
+
router_type: string;
|
|
282
|
+
/** Estimated savings for covered turns only, net of classifier cost */
|
|
283
|
+
saved_spend: number | null;
|
|
284
|
+
/** Actual spend, including classifier cost, for covered turns only */
|
|
285
|
+
savings_estimated_actual_spend: number;
|
|
286
|
+
/** Estimated single-model cost for covered turns only */
|
|
287
|
+
savings_estimated_baseline_spend: number | null;
|
|
288
|
+
/** Turns covered by the current savings estimator; legacy estimates are excluded */
|
|
289
|
+
savings_estimated_turns: number;
|
|
290
|
+
session_id: string;
|
|
291
|
+
/** What the session's routed traffic actually cost, classifier calls included */
|
|
292
|
+
spend: number;
|
|
293
|
+
/** Auto-routed turns the rollup has recorded for this session so far */
|
|
294
|
+
turns: number;
|
|
295
|
+
}
|
|
296
|
+
export const AutoRouterSessionResponse = /*@__PURE__*/ S.suspend(() =>
|
|
297
|
+
S.Struct({
|
|
298
|
+
baseline_model: S.NullOr(S.String),
|
|
299
|
+
baseline_models: AutoRouterSessionResponseBaselineModelsMap,
|
|
300
|
+
baseline_spend: S.NullOr(S.Number),
|
|
301
|
+
last_model: S.String,
|
|
302
|
+
router_name: S.String,
|
|
303
|
+
router_type: S.String,
|
|
304
|
+
saved_spend: S.NullOr(S.Number),
|
|
305
|
+
savings_estimated_actual_spend: S.Number,
|
|
306
|
+
savings_estimated_baseline_spend: S.NullOr(S.Number),
|
|
307
|
+
savings_estimated_turns: S.Number,
|
|
308
|
+
session_id: S.String,
|
|
309
|
+
spend: S.Number,
|
|
310
|
+
turns: S.Number,
|
|
311
|
+
}),
|
|
312
|
+
).annotate({
|
|
313
|
+
identifier: "AutoRouterSessionResponse",
|
|
314
|
+
}) as any as S.Schema<AutoRouterSessionResponse>;
|
|
315
|
+
|
|
222
316
|
export interface GetShadowEvalJobAutoRouterShadowEvalJobIdGetRequest {
|
|
223
317
|
job_id: string;
|
|
224
318
|
}
|
|
@@ -240,47 +334,13 @@ export const GetShadowEvalJobAutoRouterShadowEvalJobIdGetRequest =
|
|
|
240
334
|
export type ShadowEvalJobResponseDirection = "forward" | "reverse";
|
|
241
335
|
export const ShadowEvalJobResponseDirection = S.String;
|
|
242
336
|
|
|
243
|
-
/**
|
|
244
|
-
export
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
attempt_count?: number | null;
|
|
249
|
-
/** Alias of the shadowed key, resolved from the key row at read time; None when unset or deleted */
|
|
250
|
-
key_alias?: string | null;
|
|
251
|
-
/** Masked display name (sk-...) of the shadowed key, resolved at read time like key_alias */
|
|
252
|
-
key_name?: string | null;
|
|
253
|
-
/** This key's own USD budget for the eval's shadow and judge spend, independent of its siblings'; None on jobs created before spend budgets existed, which max_turns alone bounds */
|
|
254
|
-
max_budget?: number | null;
|
|
255
|
-
/** This key's sample-count ceiling: the whole budget for jobs created before max_budget existed, and the error-loop safety valve otherwise */
|
|
256
|
-
max_turns: number;
|
|
257
|
-
/** This key's recorded shadow plus judge spend in USD, the same figure the sampler budgets against max_budget; populated on list and detail responses and frozen at stopped_at exactly like attempt_count */
|
|
258
|
-
spend?: number | null;
|
|
259
|
-
/** When this key's slot was stamped free, whether its own budget ran out, the window closed, or an operator stopped the job; status is derived, so a spent budget reads completed even while this is still unset */
|
|
260
|
-
stopped_at?: string | null;
|
|
261
|
-
}
|
|
262
|
-
export const ShadowEvalJobKeyResponse = /*@__PURE__*/ S.suspend(() =>
|
|
263
|
-
S.Struct({
|
|
264
|
-
api_key_id: S.String,
|
|
265
|
-
attempt_count: S.optional(S.NullOr(S.Number)),
|
|
266
|
-
key_alias: S.optional(S.NullOr(S.String)),
|
|
267
|
-
key_name: S.optional(S.NullOr(S.String)),
|
|
268
|
-
max_budget: S.optional(S.NullOr(S.Number)),
|
|
269
|
-
max_turns: S.Number,
|
|
270
|
-
spend: S.optional(S.NullOr(S.Number)),
|
|
271
|
-
stopped_at: S.optional(S.NullOr(S.String)),
|
|
272
|
-
}),
|
|
273
|
-
).annotate({
|
|
274
|
-
identifier: "ShadowEvalJobKeyResponse",
|
|
275
|
-
}) as any as S.Schema<ShadowEvalJobKeyResponse>;
|
|
276
|
-
|
|
277
|
-
/** The keys whose traffic this job evaluates, and only those keys', each with its own budget */
|
|
278
|
-
export type ShadowEvalJobResponseKeysList = Array<ShadowEvalJobKeyResponse>;
|
|
279
|
-
export const ShadowEvalJobResponseKeysList = /*@__PURE__*/ S.Array(
|
|
280
|
-
ShadowEvalJobKeyResponse,
|
|
281
|
-
) as any as S.Schema<ShadowEvalJobResponseKeysList>;
|
|
337
|
+
/** Model groups the sampled traffic is narrowed to; empty means every model the targets use */
|
|
338
|
+
export type ShadowEvalJobResponseModelsList = Array<string>;
|
|
339
|
+
export const ShadowEvalJobResponseModelsList = /*@__PURE__*/ S.Array(
|
|
340
|
+
S.String,
|
|
341
|
+
) as any as S.Schema<ShadowEvalJobResponseModelsList>;
|
|
282
342
|
|
|
283
|
-
/** Judge outcomes for one slice of a job's verdicts
|
|
343
|
+
/** Judge outcomes for one slice of a job's verdicts: a router tier, one of the models that served the real arm, or one scoped target (embedded on that target's own entry, so slices never need re-joining to a target by id). */
|
|
284
344
|
export interface ShadowEvalSlice {
|
|
285
345
|
avg_judge_confidence: number;
|
|
286
346
|
/** Judged turns litellm's response cache served, excluded from both spends: an adopted router would be served by the same cache, so those turns cost the same either way */
|
|
@@ -319,11 +379,11 @@ export const ShadowEvalResultByCurrentModelList = /*@__PURE__*/ S.Array(
|
|
|
319
379
|
ShadowEvalSlice,
|
|
320
380
|
) as any as S.Schema<ShadowEvalResultByCurrentModelList>;
|
|
321
381
|
|
|
322
|
-
/** One slice per
|
|
323
|
-
export type
|
|
324
|
-
export const
|
|
382
|
+
/** One slice per router arm, grouped on the router name. Every arm of a multi-router job is judged against the same real responses over the same sampled requests, so these slices compare routers head-to-head: like-for-like win rates and spends on identical traffic. Verdicts from before arm stamping existed count toward the job's own router */
|
|
383
|
+
export type ShadowEvalResultByRouterList = Array<ShadowEvalSlice>;
|
|
384
|
+
export const ShadowEvalResultByRouterList = /*@__PURE__*/ S.Array(
|
|
325
385
|
ShadowEvalSlice,
|
|
326
|
-
) as any as S.Schema<
|
|
386
|
+
) as any as S.Schema<ShadowEvalResultByRouterList>;
|
|
327
387
|
|
|
328
388
|
export type ShadowEvalResultByTierList = Array<ShadowEvalSlice>;
|
|
329
389
|
export const ShadowEvalResultByTierList = /*@__PURE__*/ S.Array(
|
|
@@ -334,16 +394,16 @@ export const ShadowEvalResultByTierList = /*@__PURE__*/ S.Array(
|
|
|
334
394
|
export interface ShadowEvalResult {
|
|
335
395
|
/** Sliced by the model that served the real arm: the keys' incumbent models in forward mode, and in reverse the models the router itself picked */
|
|
336
396
|
by_current_model: ShadowEvalResultByCurrentModelList;
|
|
337
|
-
/** One slice per
|
|
338
|
-
|
|
397
|
+
/** One slice per router arm, grouped on the router name. Every arm of a multi-router job is judged against the same real responses over the same sampled requests, so these slices compare routers head-to-head: like-for-like win rates and spends on identical traffic. Verdicts from before arm stamping existed count toward the job's own router */
|
|
398
|
+
by_router?: ShadowEvalResultByRouterList;
|
|
339
399
|
by_tier: ShadowEvalResultByTierList;
|
|
340
400
|
/** Eligible requests the sampling dice skipped, summed over legs: the judged rows stand for judged + this many requests. None for jobs from before the funnel existed */
|
|
341
401
|
not_sampled_count?: number | null;
|
|
342
402
|
overall_shadow_win_rate_pct: number;
|
|
343
403
|
overall_tie_rate_pct: number;
|
|
344
|
-
/** USD the real arm billed across all judged turns, cache-served turns excluded */
|
|
404
|
+
/** USD the real arm billed across all judged turns, cache-served turns excluded. A judged turn is one (request, router arm) verdict, so a multi-router job counts the real response once per arm it was judged against; per-router comparisons read by_router */
|
|
345
405
|
sampled_real_spend?: number;
|
|
346
|
-
/** USD the shadow
|
|
406
|
+
/** USD the shadow arms billed across the same turns, judge excluded, like for like */
|
|
347
407
|
sampled_shadow_spend?: number;
|
|
348
408
|
/** Sampled requests dropped by the per-pod concurrency cap, so quiet periods are overweighted */
|
|
349
409
|
shed_count?: number | null;
|
|
@@ -355,7 +415,7 @@ export interface ShadowEvalResult {
|
|
|
355
415
|
export const ShadowEvalResult = /*@__PURE__*/ S.suspend(() =>
|
|
356
416
|
S.Struct({
|
|
357
417
|
by_current_model: ShadowEvalResultByCurrentModelList,
|
|
358
|
-
|
|
418
|
+
by_router: S.optional(ShadowEvalResultByRouterList),
|
|
359
419
|
by_tier: ShadowEvalResultByTierList,
|
|
360
420
|
not_sampled_count: S.optional(S.NullOr(S.Number)),
|
|
361
421
|
overall_shadow_win_rate_pct: S.Number,
|
|
@@ -370,11 +430,68 @@ export const ShadowEvalResult = /*@__PURE__*/ S.suspend(() =>
|
|
|
370
430
|
identifier: "ShadowEvalResult",
|
|
371
431
|
}) as any as S.Schema<ShadowEvalResult>;
|
|
372
432
|
|
|
373
|
-
/**
|
|
433
|
+
/** Every auto-router this job runs as a shadow arm. Multi-router jobs sample one slice of traffic and judge every arm against the same real responses */
|
|
434
|
+
export type ShadowEvalJobResponseRouterNamesList = Array<string>;
|
|
435
|
+
export const ShadowEvalJobResponseRouterNamesList = /*@__PURE__*/ S.Array(
|
|
436
|
+
S.String,
|
|
437
|
+
) as any as S.Schema<ShadowEvalJobResponseRouterNamesList>;
|
|
438
|
+
|
|
439
|
+
/** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every target spending its budget, and anything else is running. The all-targets-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
|
|
374
440
|
export type ShadowEvalJobResponseStatus = "running" | "completed" | "stopped";
|
|
375
441
|
export const ShadowEvalJobResponseStatus = S.String;
|
|
376
442
|
|
|
377
|
-
/**
|
|
443
|
+
/** What kind of entity this entry scopes */
|
|
444
|
+
export type ShadowEvalJobTargetResponseTargetType = "key" | "team" | "user";
|
|
445
|
+
export const ShadowEvalJobTargetResponseTargetType = S.String;
|
|
446
|
+
|
|
447
|
+
/** One target a job shadows (a key, team, or user), with its own budget and stop state. */
|
|
448
|
+
export interface ShadowEvalJobTargetResponse {
|
|
449
|
+
/** This target's sampled attempts so far, judged and errored alike, the same count the sampler budgets against max_turns; populated on list and detail responses. Frozen at stopped_at once the target is stamped, so in-flight attempts landing after a stop never reclassify it */
|
|
450
|
+
attempt_count?: number | null;
|
|
451
|
+
/** Masked display name (sk-...) for key targets, resolved at read time; None for teams and users */
|
|
452
|
+
key_name?: string | null;
|
|
453
|
+
/** This target's own USD budget for the eval's shadow and judge spend, independent of its siblings'; None on jobs created before spend budgets existed, which max_turns alone bounds */
|
|
454
|
+
max_budget?: number | null;
|
|
455
|
+
/** This target's sample-count ceiling: the whole budget for jobs created before max_budget existed, and the error-loop safety valve otherwise */
|
|
456
|
+
max_turns: number;
|
|
457
|
+
/** This target's recorded shadow plus judge spend in USD, the same figure the sampler budgets against max_budget; populated on list and detail responses and frozen at stopped_at exactly like attempt_count */
|
|
458
|
+
spend?: number | null;
|
|
459
|
+
/** When this target's slot was stamped free, whether its own budget ran out, the window closed, or an operator stopped the job; status is derived, so a spent budget reads completed even while this is still unset */
|
|
460
|
+
stopped_at?: string | null;
|
|
461
|
+
/** Display label resolved from the target's own row at read time: the key's alias, the team's alias, or the user's email; None when unset or deleted */
|
|
462
|
+
target_alias?: string | null;
|
|
463
|
+
/** The hashed virtual key, team id, or user id whose traffic this entry scopes */
|
|
464
|
+
target_id: string;
|
|
465
|
+
/** What kind of entity this entry scopes */
|
|
466
|
+
target_type: ShadowEvalJobTargetResponseTargetType;
|
|
467
|
+
/** This target's own judged-verdict slice; detail endpoint only, None until a turn is judged */
|
|
468
|
+
verdicts?: ShadowEvalSlice | null;
|
|
469
|
+
}
|
|
470
|
+
export const ShadowEvalJobTargetResponse = /*@__PURE__*/ S.suspend(() =>
|
|
471
|
+
S.Struct({
|
|
472
|
+
attempt_count: S.optional(S.NullOr(S.Number)),
|
|
473
|
+
key_name: S.optional(S.NullOr(S.String)),
|
|
474
|
+
max_budget: S.optional(S.NullOr(S.Number)),
|
|
475
|
+
max_turns: S.Number,
|
|
476
|
+
spend: S.optional(S.NullOr(S.Number)),
|
|
477
|
+
stopped_at: S.optional(S.NullOr(S.String)),
|
|
478
|
+
target_alias: S.optional(S.NullOr(S.String)),
|
|
479
|
+
target_id: S.String,
|
|
480
|
+
target_type: ShadowEvalJobTargetResponseTargetType,
|
|
481
|
+
verdicts: S.optional(S.NullOr(ShadowEvalSlice)),
|
|
482
|
+
}),
|
|
483
|
+
).annotate({
|
|
484
|
+
identifier: "ShadowEvalJobTargetResponse",
|
|
485
|
+
}) as any as S.Schema<ShadowEvalJobTargetResponse>;
|
|
486
|
+
|
|
487
|
+
/** The targets whose traffic this job evaluates, and only theirs, each with its own budget */
|
|
488
|
+
export type ShadowEvalJobResponseTargetsList =
|
|
489
|
+
Array<ShadowEvalJobTargetResponse>;
|
|
490
|
+
export const ShadowEvalJobResponseTargetsList = /*@__PURE__*/ S.Array(
|
|
491
|
+
ShadowEvalJobTargetResponse,
|
|
492
|
+
) as any as S.Schema<ShadowEvalJobResponseTargetsList>;
|
|
493
|
+
|
|
494
|
+
/** A shadow-eval job over one or more targets, each with its own budget and stop state; status is derived from stopped_by, the targets' stop and budget state, and ends_at, never stored, so no writer anywhere can produce an inconsistent one. Aggregate fields are populated by the detail endpoint only and stay None on list responses. */
|
|
378
495
|
export interface ShadowEvalJobResponse {
|
|
379
496
|
baseline_model?: string | null;
|
|
380
497
|
created_at: string;
|
|
@@ -388,18 +505,23 @@ export interface ShadowEvalJobResponse {
|
|
|
388
505
|
judge_spend?: number | null;
|
|
389
506
|
/** Verdicts recorded; detail endpoint only */
|
|
390
507
|
judged_count?: number | null;
|
|
391
|
-
/** The keys whose traffic this job evaluates, and only those keys', each with its own budget */
|
|
392
|
-
keys: ShadowEvalJobResponseKeysList;
|
|
393
508
|
/** Most recent attempt error; detail endpoint only */
|
|
394
509
|
last_error?: string | null;
|
|
510
|
+
/** Model groups the sampled traffic is narrowed to; empty means every model the targets use */
|
|
511
|
+
models?: ShadowEvalJobResponseModelsList;
|
|
395
512
|
/** Stratified verdicts; detail endpoint only */
|
|
396
513
|
results?: ShadowEvalResult | null;
|
|
514
|
+
/** The first router, kept for callers that predate router_names; derived so the two fields can never disagree. */
|
|
397
515
|
router_name: string;
|
|
516
|
+
/** Every auto-router this job runs as a shadow arm. Multi-router jobs sample one slice of traffic and judge every arm against the same real responses */
|
|
517
|
+
router_names: ShadowEvalJobResponseRouterNamesList;
|
|
398
518
|
shadow_percentage: number;
|
|
399
|
-
/** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every
|
|
519
|
+
/** Three recorded facts, no history-guessing: a stop is stopped_by (the migration backfills it for every job that displayed stopped when the column arrived, so the pre-column population is closed), completion is the window passing or every target spending its budget, and anything else is running. The all-targets-stamped fallback covers only stops written by pre-column pods during a rolling deploy. */
|
|
400
520
|
status: ShadowEvalJobResponseStatus;
|
|
401
521
|
/** The operator who stopped the job early, recorded by the stop endpoint; 'unknown' backfilled by migration for jobs that displayed stopped when the column arrived; None when the job ended on its own. Its presence is what makes a job read stopped rather than completed */
|
|
402
522
|
stopped_by?: string | null;
|
|
523
|
+
/** The targets whose traffic this job evaluates, and only theirs, each with its own budget */
|
|
524
|
+
targets: ShadowEvalJobResponseTargetsList;
|
|
403
525
|
}
|
|
404
526
|
export const ShadowEvalJobResponse = /*@__PURE__*/ S.suspend(() =>
|
|
405
527
|
S.Struct({
|
|
@@ -412,28 +534,46 @@ export const ShadowEvalJobResponse = /*@__PURE__*/ S.suspend(() =>
|
|
|
412
534
|
judge_model: S.String,
|
|
413
535
|
judge_spend: S.optional(S.NullOr(S.Number)),
|
|
414
536
|
judged_count: S.optional(S.NullOr(S.Number)),
|
|
415
|
-
keys: ShadowEvalJobResponseKeysList,
|
|
416
537
|
last_error: S.optional(S.NullOr(S.String)),
|
|
538
|
+
models: S.optional(ShadowEvalJobResponseModelsList),
|
|
417
539
|
results: S.optional(S.NullOr(ShadowEvalResult)),
|
|
418
540
|
router_name: S.String,
|
|
541
|
+
router_names: ShadowEvalJobResponseRouterNamesList,
|
|
419
542
|
shadow_percentage: S.Number,
|
|
420
543
|
status: ShadowEvalJobResponseStatus,
|
|
421
544
|
stopped_by: S.optional(S.NullOr(S.String)),
|
|
545
|
+
targets: ShadowEvalJobResponseTargetsList,
|
|
422
546
|
}),
|
|
423
547
|
).annotate({
|
|
424
548
|
identifier: "ShadowEvalJobResponse",
|
|
425
549
|
}) as any as S.Schema<ShadowEvalJobResponse>;
|
|
426
550
|
|
|
551
|
+
export type ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType =
|
|
552
|
+
| "key"
|
|
553
|
+
| "team"
|
|
554
|
+
| "user";
|
|
555
|
+
export const ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType =
|
|
556
|
+
S.String;
|
|
557
|
+
|
|
427
558
|
export interface ListShadowEvalJobsAutoRouterShadowEvalGetRequest {
|
|
428
|
-
/**
|
|
429
|
-
|
|
559
|
+
/** Kind of target to filter on; requires target_id */
|
|
560
|
+
target_type?:
|
|
561
|
+
| ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType
|
|
562
|
+
| (string & {});
|
|
563
|
+
/** Filter to jobs that shadow this target, alone or alongside others */
|
|
564
|
+
target_id?: string;
|
|
430
565
|
/** Newest jobs to return */
|
|
431
566
|
limit?: number;
|
|
432
567
|
}
|
|
433
568
|
export const ListShadowEvalJobsAutoRouterShadowEvalGetRequest =
|
|
434
569
|
/*@__PURE__*/ S.suspend(() =>
|
|
435
570
|
S.Struct({
|
|
436
|
-
|
|
571
|
+
target_type: S.optional(
|
|
572
|
+
ListShadowEvalJobsAutoRouterShadowEvalGetRequestTargetType.pipe(
|
|
573
|
+
T.Query(),
|
|
574
|
+
),
|
|
575
|
+
),
|
|
576
|
+
target_id: S.optional(S.String.pipe(T.Query())),
|
|
437
577
|
limit: S.optional(S.Number.pipe(T.Query())),
|
|
438
578
|
}).pipe(
|
|
439
579
|
T.Http({ method: "GET", uri: "/auto_router/shadow_eval", code: 200 }),
|
|
@@ -461,7 +601,7 @@ export const ListShadowEvalJobsAutoRouterShadowEvalGetResponse =
|
|
|
461
601
|
identifier: "ListShadowEvalJobsAutoRouterShadowEvalGetResponse",
|
|
462
602
|
}) as any as S.Schema<ListShadowEvalJobsAutoRouterShadowEvalGetResponse>;
|
|
463
603
|
|
|
464
|
-
/**
|
|
604
|
+
/** Hashed virtual keys whose traffic will be shadowed. Combined with team_ids and user_ids the job needs at least one target and at most 100, which also bounds every read the job's endpoints make. Each target carries its own max_budget spend budget, so one exhausting its budget leaves the others sampling. */
|
|
465
605
|
export type StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList =
|
|
466
606
|
Array<string>;
|
|
467
607
|
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList =
|
|
@@ -476,9 +616,41 @@ export type StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection =
|
|
|
476
616
|
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection =
|
|
477
617
|
S.String;
|
|
478
618
|
|
|
619
|
+
/** Model groups to narrow the sampled traffic to, matched on the group the caller requested and resolved through model_group_alias, so an alias and its target are one name. Empty samples every model the targets use. This ANDs with the targets: a job over a user and one model samples that user's requests on that model across every key they own, and none of their other traffic. Forward jobs only: a reverse job samples exactly the traffic its own router served, which no other model group can name */
|
|
620
|
+
export type StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList =
|
|
621
|
+
Array<string>;
|
|
622
|
+
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList =
|
|
623
|
+
/*@__PURE__*/ S.Array(
|
|
624
|
+
S.String,
|
|
625
|
+
) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList>;
|
|
626
|
+
|
|
627
|
+
/** The auto-routers under evaluation, at most 4. Every sampled request runs through every router listed and each arm is judged independently against the same real response, so routers compare head-to-head on identical traffic. More than one router requires direction 'forward'. After validation this field always carries the full deduplicated set, whichever spelling the caller used */
|
|
628
|
+
export type StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList =
|
|
629
|
+
Array<string>;
|
|
630
|
+
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList =
|
|
631
|
+
/*@__PURE__*/ S.Array(
|
|
632
|
+
S.String,
|
|
633
|
+
) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList>;
|
|
634
|
+
|
|
635
|
+
/** Teams whose traffic will be shadowed, matched on the team every authenticated request resolves to, so a team's JWT-auth and virtual-key traffic are both sampled */
|
|
636
|
+
export type StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList =
|
|
637
|
+
Array<string>;
|
|
638
|
+
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList =
|
|
639
|
+
/*@__PURE__*/ S.Array(
|
|
640
|
+
S.String,
|
|
641
|
+
) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList>;
|
|
642
|
+
|
|
643
|
+
/** Users whose traffic will be shadowed, matched on the user every authenticated request resolves to across all their teams: JWT requests carrying their subject claim and virtual keys they own */
|
|
644
|
+
export type StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList =
|
|
645
|
+
Array<string>;
|
|
646
|
+
export const StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList =
|
|
647
|
+
/*@__PURE__*/ S.Array(
|
|
648
|
+
S.String,
|
|
649
|
+
) as any as S.Schema<StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList>;
|
|
650
|
+
|
|
479
651
|
export interface StartShadowEvalAutoRouterShadowEvalStartPostRequest {
|
|
480
|
-
/**
|
|
481
|
-
api_key_ids
|
|
652
|
+
/** Hashed virtual keys whose traffic will be shadowed. Combined with team_ids and user_ids the job needs at least one target and at most 100, which also bounds every read the job's endpoints make. Each target carries its own max_budget spend budget, so one exhausting its budget leaves the others sampling. */
|
|
653
|
+
api_key_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList;
|
|
482
654
|
/** Required when direction is reverse and rejected otherwise: the fixed model the router's own responses are judged against. Must be a plain model rather than another auto-router */
|
|
483
655
|
baseline_model?: string | null;
|
|
484
656
|
/** forward answers 'should this key adopt router_name': it samples the requests the key did NOT route through the router and duplicates them through it. reverse answers 'is the router still worth it for a key already on it': it samples the requests the router did serve and duplicates them against baseline_model. The response the caller received is always the real arm */
|
|
@@ -489,18 +661,27 @@ export interface StartShadowEvalAutoRouterShadowEvalStartPostRequest {
|
|
|
489
661
|
duration_days?: number;
|
|
490
662
|
/** Model used to blindly judge real vs. shadow responses. The judge only compares two answers, so a mid-tier model (Claude Sonnet or GPT-4o class) is the sweet spot: small/nano-class models produce unreliable or malformed verdicts, while frontier reasoning models add cost without changing outcomes. */
|
|
491
663
|
judge_model?: string;
|
|
492
|
-
/** Per-
|
|
664
|
+
/** Per-target USD budget for the eval's own overhead, the shadow-arm and judge calls, priced with the same figures the spend pipeline bills. EACH scoped target samples until its recorded eval spend reaches this, so a job over N targets spends at most about N times max_budget; in-flight samples can overshoot the cap by one sampling cache window. Every router arm draws from the same per-target budget, so a multi-router job reaches it proportionally sooner */
|
|
493
665
|
max_budget?: number;
|
|
494
|
-
/**
|
|
495
|
-
|
|
496
|
-
/**
|
|
666
|
+
/** Model groups to narrow the sampled traffic to, matched on the group the caller requested and resolved through model_group_alias, so an alias and its target are one name. Empty samples every model the targets use. This ANDs with the targets: a job over a user and one model samples that user's requests on that model across every key they own, and none of their other traffic. Forward jobs only: a reverse job samples exactly the traffic its own router served, which no other model group can name */
|
|
667
|
+
models?: StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList;
|
|
668
|
+
/** The auto-router under evaluation, in either direction: the single-router spelling of router_names. Provide exactly one of the two fields */
|
|
669
|
+
router_name?: string | null;
|
|
670
|
+
/** The auto-routers under evaluation, at most 4. Every sampled request runs through every router listed and each arm is judged independently against the same real response, so routers compare head-to-head on identical traffic. More than one router requires direction 'forward'. After validation this field always carries the full deduplicated set, whichever spelling the caller used */
|
|
671
|
+
router_names?: StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList;
|
|
672
|
+
/** Percentage of each target's requests to duplicate through the router */
|
|
497
673
|
shadow_percentage: number;
|
|
674
|
+
/** Teams whose traffic will be shadowed, matched on the team every authenticated request resolves to, so a team's JWT-auth and virtual-key traffic are both sampled */
|
|
675
|
+
team_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList;
|
|
676
|
+
/** Users whose traffic will be shadowed, matched on the user every authenticated request resolves to across all their teams: JWT requests carrying their subject claim and virtual keys they own */
|
|
677
|
+
user_ids?: StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList;
|
|
498
678
|
}
|
|
499
679
|
export const StartShadowEvalAutoRouterShadowEvalStartPostRequest =
|
|
500
680
|
/*@__PURE__*/ S.suspend(() =>
|
|
501
681
|
S.Struct({
|
|
502
|
-
api_key_ids:
|
|
682
|
+
api_key_ids: S.optional(
|
|
503
683
|
StartShadowEvalAutoRouterShadowEvalStartPostRequestApiKeyIdsList,
|
|
684
|
+
),
|
|
504
685
|
baseline_model: S.optional(S.NullOr(S.String)),
|
|
505
686
|
direction: S.optional(
|
|
506
687
|
StartShadowEvalAutoRouterShadowEvalStartPostRequestDirection,
|
|
@@ -508,8 +689,20 @@ export const StartShadowEvalAutoRouterShadowEvalStartPostRequest =
|
|
|
508
689
|
duration_days: S.optional(S.Number),
|
|
509
690
|
judge_model: S.optional(S.String),
|
|
510
691
|
max_budget: S.optional(S.Number),
|
|
511
|
-
|
|
692
|
+
models: S.optional(
|
|
693
|
+
StartShadowEvalAutoRouterShadowEvalStartPostRequestModelsList,
|
|
694
|
+
),
|
|
695
|
+
router_name: S.optional(S.NullOr(S.String)),
|
|
696
|
+
router_names: S.optional(
|
|
697
|
+
StartShadowEvalAutoRouterShadowEvalStartPostRequestRouterNamesList,
|
|
698
|
+
),
|
|
512
699
|
shadow_percentage: S.Number,
|
|
700
|
+
team_ids: S.optional(
|
|
701
|
+
StartShadowEvalAutoRouterShadowEvalStartPostRequestTeamIdsList,
|
|
702
|
+
),
|
|
703
|
+
user_ids: S.optional(
|
|
704
|
+
StartShadowEvalAutoRouterShadowEvalStartPostRequestUserIdsList,
|
|
705
|
+
),
|
|
513
706
|
}).pipe(
|
|
514
707
|
T.Http({
|
|
515
708
|
method: "POST",
|
|
@@ -556,6 +749,23 @@ export const getAutoRouterBenchmarksAutoRouterBenchmarksGet: API.OperationMethod
|
|
|
556
749
|
retry: Retry.Retry,
|
|
557
750
|
}));
|
|
558
751
|
|
|
752
|
+
export type GetAutoRouterSessionAutoRouterSessionGetError =
|
|
753
|
+
| UnprocessableEntity
|
|
754
|
+
| LitellmOpError;
|
|
755
|
+
/** Get Auto Router Session One auto-routed session, for the key that ran it: the model its last turn was routed to and the session's spend against the router's savings baseline. Built for a coding agent's status line or stop hook, so any virtual key may call it and only ever sees rows written under its own key hash. Reads the LiteLLM_AutoRouterSession rollup, which the asynchronous spend flush fills a moment after each turn; a session with no flushed auto-routed turn yet is a 404. The id is bounded the way the writer bounded it, so an oversized client id still finds its row. */
|
|
756
|
+
export const getAutoRouterSessionAutoRouterSessionGet: API.OperationMethod<
|
|
757
|
+
GetAutoRouterSessionAutoRouterSessionGetRequest,
|
|
758
|
+
AutoRouterSessionResponse,
|
|
759
|
+
GetAutoRouterSessionAutoRouterSessionGetError,
|
|
760
|
+
LitellmOpContext
|
|
761
|
+
> = /*@__PURE__*/ API.make(() => ({
|
|
762
|
+
input: GetAutoRouterSessionAutoRouterSessionGetRequest,
|
|
763
|
+
output: AutoRouterSessionResponse,
|
|
764
|
+
errors: [UnprocessableEntity],
|
|
765
|
+
protocol: LitellmProtocol,
|
|
766
|
+
retry: Retry.Retry,
|
|
767
|
+
}));
|
|
768
|
+
|
|
559
769
|
export type GetShadowEvalJobAutoRouterShadowEvalJobIdGetError =
|
|
560
770
|
| UnprocessableEntity
|
|
561
771
|
| LitellmOpError;
|
|
@@ -576,7 +786,7 @@ export const getShadowEvalJobAutoRouterShadowEvalJobIdGet: API.OperationMethod<
|
|
|
576
786
|
export type ListShadowEvalJobsAutoRouterShadowEvalGetError =
|
|
577
787
|
| UnprocessableEntity
|
|
578
788
|
| LitellmOpError;
|
|
579
|
-
/** List Shadow Eval Jobs List shadow eval jobs, newest first, each
|
|
789
|
+
/** List Shadow Eval Jobs List shadow eval jobs, newest first, each target with its attempt count so status is accurate. Judged counts, spend, and results ride the detail endpoint only. */
|
|
580
790
|
export const listShadowEvalJobsAutoRouterShadowEvalGet: API.OperationMethod<
|
|
581
791
|
ListShadowEvalJobsAutoRouterShadowEvalGetRequest,
|
|
582
792
|
ListShadowEvalJobsAutoRouterShadowEvalGetResponse,
|
|
@@ -593,7 +803,7 @@ export const listShadowEvalJobsAutoRouterShadowEvalGet: API.OperationMethod<
|
|
|
593
803
|
export type StartShadowEvalAutoRouterShadowEvalStartPostError =
|
|
594
804
|
| UnprocessableEntity
|
|
595
805
|
| LitellmOpError;
|
|
596
|
-
/** Start Shadow Eval Start a shadow eval: duplicate a sampled slice of one or more
|
|
806
|
+
/** Start Shadow Eval Start a shadow eval: duplicate a sampled slice of one or more targets' live traffic against a second arm, judge the two responses blind, and stratify win rates by tier, by the model that served the real arm, and by target. A target is a virtual key, a team, or a user. Team and user targets match on the identity every request resolves to at auth time, so they cover JWT-authenticated traffic, which presents no virtual key; a user target samples that user's traffic across all their teams, whether it arrives on a JWT or a key they own. models narrows every target to requests for those model groups, so a user plus one model samples that user's traffic on that model across every key they own; it is forward-only, since a reverse job already samples exactly the traffic its own router served. A forward job answers whether the targets should adopt router_name: it samples the requests the router did not serve and duplicates them through it. A reverse job answers whether a target already on the router still gains from it: it samples the requests the router did serve and duplicates them against baseline_model. A target can hold one active job per direction, so both questions can run at once, and a request matching several jobs' targets (say its key and its team) is sampled by each, separately budgeted. Shadow responses are never served to users. Each target samples until its recorded eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the job's window ends, or the job is stopped, so one target running out of budget does not end sampling for the others; sampling changes propagate to pods within about 10 seconds. Shadow and judge calls bill to the sampled request's own identity but are excluded from request counts and auto-router adoption metrics. */
|
|
597
807
|
export const startShadowEvalAutoRouterShadowEvalStartPost: API.OperationMethod<
|
|
598
808
|
StartShadowEvalAutoRouterShadowEvalStartPostRequest,
|
|
599
809
|
ShadowEvalJobResponse,
|
|
@@ -610,7 +820,7 @@ export const startShadowEvalAutoRouterShadowEvalStartPost: API.OperationMethod<
|
|
|
610
820
|
export type StopShadowEvalJobAutoRouterShadowEvalJobIdStopPostError =
|
|
611
821
|
| UnprocessableEntity
|
|
612
822
|
| LitellmOpError;
|
|
613
|
-
/** Stop Shadow Eval Job Stop an active shadow eval job, every
|
|
823
|
+
/** Stop Shadow Eval Job Stop an active shadow eval job, every target it scopes at once. Attempts are kept; sampling halts within ~10s. Targets that already stopped on their own budget keep the stopped_at they earned. The statement is the whole state machine: it claims the job only while a leg still samples inside the window with no stop recorded, so a racing operator, a same-instant budget spend, and a repeat stop all read the same 400 with the status the job actually holds. */
|
|
614
824
|
export const stopShadowEvalJobAutoRouterShadowEvalJobIdStopPost: API.OperationMethod<
|
|
615
825
|
StopShadowEvalJobAutoRouterShadowEvalJobIdStopPostRequest,
|
|
616
826
|
ShadowEvalJobResponse,
|
package/src/services/batch.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.
|
|
1
|
+
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
|
|
2
2
|
import * as S from "@distilled.cloud/core/schema";
|
|
3
3
|
import * as API from "@distilled.cloud/core/api";
|
|
4
4
|
import * as C from "@distilled.cloud/core/category";
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.
|
|
1
|
+
// AUTO-GENERATED by scripts/generate.ts from .generated-specs (pinned litellm-openapi @ 1.103.0 — see README). Do not edit.
|
|
2
2
|
import * as S from "@distilled.cloud/core/schema";
|
|
3
3
|
import * as API from "@distilled.cloud/core/api";
|
|
4
4
|
import * as T from "../traits.ts";
|