@oneciel-ai/ciel-runtime 0.1.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +132 -0
- package/ciel-runtime-menu.py +56 -6
- package/ciel_runtime.py +9379 -35909
- package/ciel_runtime_support/advisor_client.py +193 -0
- package/ciel_runtime_support/advisor_policy.py +320 -0
- package/ciel_runtime_support/advisor_refinement.py +160 -0
- package/ciel_runtime_support/advisor_request_builder.py +261 -0
- package/ciel_runtime_support/agy_installer.py +169 -0
- package/ciel_runtime_support/agy_mcp_restore.py +182 -0
- package/ciel_runtime_support/anthropic_model_policy.py +186 -0
- package/ciel_runtime_support/anthropic_response_writer.py +255 -0
- package/ciel_runtime_support/anthropic_tool_turns.py +130 -0
- package/ciel_runtime_support/api_key_cooldown.py +159 -0
- package/ciel_runtime_support/architecture.py +488 -1
- package/ciel_runtime_support/architecture_budget.py +42 -0
- package/ciel_runtime_support/channel_backlog.py +90 -0
- package/ciel_runtime_support/channel_cli.py +119 -0
- package/ciel_runtime_support/channel_compact_injection.py +82 -0
- package/ciel_runtime_support/channel_compact_poll.py +67 -0
- package/ciel_runtime_support/channel_compact_request_repository.py +113 -0
- package/ciel_runtime_support/channel_config_service.py +281 -0
- package/ciel_runtime_support/channel_connection_lifecycle.py +180 -0
- package/ciel_runtime_support/channel_connection_registry.py +128 -0
- package/ciel_runtime_support/channel_connection_worker.py +284 -0
- package/ciel_runtime_support/channel_cursor_recovery.py +92 -0
- package/ciel_runtime_support/channel_cursor_repository.py +89 -0
- package/ciel_runtime_support/channel_cursor_service.py +178 -0
- package/ciel_runtime_support/channel_event_identity.py +212 -0
- package/ciel_runtime_support/channel_event_projection.py +315 -0
- package/ciel_runtime_support/channel_inflight.py +127 -0
- package/ciel_runtime_support/channel_injection.py +115 -0
- package/ciel_runtime_support/channel_launch_guard_repository.py +58 -0
- package/ciel_runtime_support/channel_launch_policy.py +180 -0
- package/ciel_runtime_support/channel_llm_context.py +156 -0
- package/ciel_runtime_support/channel_mcp_discovery.py +186 -0
- package/ciel_runtime_support/channel_mcp_http_controller.py +239 -0
- package/ciel_runtime_support/channel_mcp_ownership.py +148 -0
- package/ciel_runtime_support/channel_mcp_tools.py +240 -0
- package/ciel_runtime_support/channel_mcp_transport.py +394 -0
- package/ciel_runtime_support/channel_message_dedupe.py +65 -0
- package/ciel_runtime_support/channel_message_policy.py +256 -0
- package/ciel_runtime_support/channel_message_prompt.py +305 -0
- package/ciel_runtime_support/channel_message_repository.py +234 -0
- package/ciel_runtime_support/channel_notification_projection.py +217 -0
- package/ciel_runtime_support/channel_panel.py +162 -0
- package/ciel_runtime_support/channel_pending_injection.py +209 -0
- package/ciel_runtime_support/channel_pending_poll.py +109 -0
- package/ciel_runtime_support/channel_probe_cache.py +433 -0
- package/ciel_runtime_support/channel_probe_report.py +101 -0
- package/ciel_runtime_support/channel_runtime_environment.py +181 -0
- package/ciel_runtime_support/channel_session_lifecycle.py +114 -0
- package/ciel_runtime_support/channel_session_repository.py +90 -0
- package/ciel_runtime_support/channel_terminal_dispatch.py +115 -0
- package/ciel_runtime_support/channel_terminal_input.py +277 -0
- package/ciel_runtime_support/channel_terminal_proxy.py +447 -0
- package/ciel_runtime_support/channel_tool_context.py +166 -0
- package/ciel_runtime_support/channel_transcript.py +414 -0
- package/ciel_runtime_support/channel_transcript_repository.py +96 -0
- package/ciel_runtime_support/channel_wake_claim_repository.py +126 -0
- package/ciel_runtime_support/channel_wake_delivery_repository.py +88 -0
- package/ciel_runtime_support/chat_files.py +138 -0
- package/ciel_runtime_support/chat_http_controller.py +235 -0
- package/ciel_runtime_support/claude_environment.py +375 -0
- package/ciel_runtime_support/claude_router.py +247 -193
- package/ciel_runtime_support/cli_dispatch.py +792 -0
- package/ciel_runtime_support/cli_parser.py +165 -0
- package/ciel_runtime_support/cli_usage.py +100 -0
- package/ciel_runtime_support/codex_app_server.py +20 -5
- package/ciel_runtime_support/codex_channel_sse_launch.py +87 -0
- package/ciel_runtime_support/codex_cli.py +42 -6
- package/ciel_runtime_support/codex_config.py +323 -0
- package/ciel_runtime_support/codex_launch_configuration.py +240 -0
- package/ciel_runtime_support/codex_launch_policy.py +66 -0
- package/ciel_runtime_support/codex_mcp_integration.py +195 -0
- package/ciel_runtime_support/codex_mcp_restore.py +304 -0
- package/ciel_runtime_support/codex_model_catalog.py +133 -0
- package/ciel_runtime_support/codex_process_lifecycle.py +271 -0
- package/ciel_runtime_support/codex_router.py +147 -1
- package/ciel_runtime_support/codex_session_repository.py +115 -0
- package/ciel_runtime_support/codex_session_selection.py +114 -0
- package/ciel_runtime_support/command_asset_installer.py +103 -0
- package/ciel_runtime_support/compatibility_probe.py +295 -0
- package/ciel_runtime_support/compatibility_protocol.py +251 -0
- package/ciel_runtime_support/compatibility_runtime.py +166 -0
- package/ciel_runtime_support/compatibility_test.py +370 -0
- package/ciel_runtime_support/config_migrations.py +307 -0
- package/ciel_runtime_support/config_repository.py +175 -0
- package/ciel_runtime_support/config_value_codec.py +64 -0
- package/ciel_runtime_support/configuration_cli.py +374 -0
- package/ciel_runtime_support/context_compaction.py +280 -0
- package/ciel_runtime_support/context_setup.py +208 -0
- package/ciel_runtime_support/context_summary_policy.py +392 -0
- package/ciel_runtime_support/credential_cli.py +104 -0
- package/ciel_runtime_support/credential_management.py +261 -0
- package/ciel_runtime_support/credentials.py +269 -0
- package/ciel_runtime_support/executable_discovery.py +141 -0
- package/ciel_runtime_support/github_copilot_oauth.py +335 -0
- package/ciel_runtime_support/github_copilot_oauth_runtime.py +213 -0
- package/ciel_runtime_support/header_forwarding.py +73 -0
- package/ciel_runtime_support/headless_config.py +221 -0
- package/ciel_runtime_support/http_response.py +129 -0
- package/ciel_runtime_support/install_diagnostics.py +149 -0
- package/ciel_runtime_support/kimi_identity.py +123 -0
- package/ciel_runtime_support/launch_diagnostics.py +204 -0
- package/ciel_runtime_support/launch_state.py +127 -0
- package/ciel_runtime_support/live_api_key_controller.py +58 -0
- package/ciel_runtime_support/llm_config_http.py +148 -0
- package/ciel_runtime_support/llm_option_config.py +259 -0
- package/ciel_runtime_support/llm_presentation_data.py +447 -0
- package/ciel_runtime_support/llm_presets.py +773 -0
- package/ciel_runtime_support/lm_studio_runtime.py +401 -0
- package/ciel_runtime_support/managed_mcp_config.py +144 -0
- package/ciel_runtime_support/managed_mcp_discovery.py +151 -0
- package/ciel_runtime_support/managed_service_cleanup.py +89 -0
- package/ciel_runtime_support/mcp_config_reader.py +230 -0
- package/ciel_runtime_support/mcp_http_proxy.py +607 -0
- package/ciel_runtime_support/mcp_inventory.py +59 -0
- package/ciel_runtime_support/mcp_notification_wait_policy.py +159 -0
- package/ciel_runtime_support/mcp_probe_codec.py +136 -0
- package/ciel_runtime_support/mcp_probe_transport.py +328 -0
- package/ciel_runtime_support/mcp_proxy_codec.py +345 -0
- package/ciel_runtime_support/mcp_proxy_config.py +107 -0
- package/ciel_runtime_support/mcp_proxy_notifications.py +261 -0
- package/ciel_runtime_support/mcp_proxy_process.py +560 -0
- package/ciel_runtime_support/mcp_split_proxy_http.py +165 -0
- package/ciel_runtime_support/mcp_stdio_probe.py +240 -0
- package/ciel_runtime_support/mcp_transport.py +146 -0
- package/ciel_runtime_support/model_cache_lifecycle.py +78 -0
- package/ciel_runtime_support/model_catalog_projection.py +61 -0
- package/ciel_runtime_support/model_context_hints.py +109 -0
- package/ciel_runtime_support/model_panel.py +147 -0
- package/ciel_runtime_support/model_registry_repository.py +231 -0
- package/ciel_runtime_support/npm_runtime.py +191 -0
- package/ciel_runtime_support/ollama_catalog.py +462 -0
- package/ciel_runtime_support/ollama_catalog_cli.py +41 -0
- package/ciel_runtime_support/ollama_catalog_repository.py +75 -0
- package/ciel_runtime_support/ollama_context_sync.py +87 -0
- package/ciel_runtime_support/ollama_forwarding.py +449 -0
- package/ciel_runtime_support/openai_chat_passthrough.py +67 -0
- package/ciel_runtime_support/openai_chat_router.py +64 -0
- package/ciel_runtime_support/openai_forwarding.py +194 -0
- package/ciel_runtime_support/openai_responses_router.py +291 -0
- package/ciel_runtime_support/openai_responses_stream.py +135 -0
- package/ciel_runtime_support/output_budget.py +89 -0
- package/ciel_runtime_support/package_lifecycle.py +217 -0
- package/ciel_runtime_support/plan_artifact_controller.py +104 -0
- package/ciel_runtime_support/prelaunch.py +959 -0
- package/ciel_runtime_support/prelaunch_launch_preference.py +75 -0
- package/ciel_runtime_support/prelaunch_panel_projection.py +396 -0
- package/ciel_runtime_support/prelaunch_terminal.py +764 -0
- package/ciel_runtime_support/process_control.py +708 -0
- package/ciel_runtime_support/prompt_compaction.py +322 -0
- package/ciel_runtime_support/prompt_injection.py +176 -0
- package/ciel_runtime_support/protocols/__init__.py +24 -0
- package/ciel_runtime_support/protocols/anthropic_content.py +29 -0
- package/ciel_runtime_support/protocols/anthropic_thinking_policy.py +336 -0
- package/ciel_runtime_support/protocols/chat_projection.py +315 -0
- package/ciel_runtime_support/protocols/conversation_policy.py +217 -0
- package/ciel_runtime_support/protocols/conversation_turn_policy.py +972 -0
- package/ciel_runtime_support/protocols/ollama_chat.py +111 -0
- package/ciel_runtime_support/protocols/ollama_response.py +231 -0
- package/ciel_runtime_support/protocols/openai_reasoning.py +75 -0
- package/ciel_runtime_support/protocols/openai_responses.py +271 -0
- package/ciel_runtime_support/protocols/pseudo_tool_history.py +248 -0
- package/ciel_runtime_support/protocols/tool_result_projection.py +112 -0
- package/ciel_runtime_support/provider_adapters.py +160 -0
- package/ciel_runtime_support/provider_catalog_sources.py +316 -0
- package/ciel_runtime_support/provider_choice.py +201 -0
- package/ciel_runtime_support/provider_compatibility.py +165 -0
- package/ciel_runtime_support/provider_config_mutations.py +361 -0
- package/ciel_runtime_support/provider_configuration_service.py +162 -0
- package/ciel_runtime_support/provider_context.py +308 -0
- package/ciel_runtime_support/provider_contract_projection.py +73 -0
- package/ciel_runtime_support/provider_descriptor.py +82 -0
- package/ciel_runtime_support/provider_endpoint_policy.py +130 -0
- package/ciel_runtime_support/provider_endpoint_probe.py +165 -0
- package/ciel_runtime_support/provider_launch_endpoint.py +109 -0
- package/ciel_runtime_support/provider_limits.py +457 -0
- package/ciel_runtime_support/provider_model_identity.py +139 -0
- package/ciel_runtime_support/provider_model_selection.py +431 -0
- package/ciel_runtime_support/provider_model_specs.py +142 -0
- package/ciel_runtime_support/provider_models.py +263 -0
- package/ciel_runtime_support/provider_network.py +176 -0
- package/ciel_runtime_support/provider_option_cli.py +238 -0
- package/ciel_runtime_support/provider_option_panel.py +275 -0
- package/ciel_runtime_support/provider_option_status.py +192 -0
- package/ciel_runtime_support/provider_policy.py +101 -0
- package/ciel_runtime_support/provider_query_policy.py +67 -0
- package/ciel_runtime_support/provider_readiness.py +112 -0
- package/ciel_runtime_support/provider_request_access.py +131 -0
- package/ciel_runtime_support/provider_request_builder.py +250 -0
- package/ciel_runtime_support/provider_responses_passthrough.py +81 -0
- package/ciel_runtime_support/provider_runtime_info.py +113 -0
- package/ciel_runtime_support/provider_runtime_modes.py +150 -0
- package/ciel_runtime_support/provider_sampling_policy.py +46 -0
- package/ciel_runtime_support/provider_status.py +145 -0
- package/ciel_runtime_support/provider_timeout_policy.py +184 -0
- package/ciel_runtime_support/provider_tool_policy.py +145 -0
- package/ciel_runtime_support/providers/__init__.py +65 -0
- package/ciel_runtime_support/providers/anthropic.py +160 -0
- package/ciel_runtime_support/providers/anthropic_catalog.py +119 -0
- package/ciel_runtime_support/providers/base.py +232 -0
- package/ciel_runtime_support/providers/catalog.py +326 -0
- package/ciel_runtime_support/providers/cloud.py +194 -0
- package/ciel_runtime_support/providers/constants.py +54 -0
- package/ciel_runtime_support/providers/deepseek.py +115 -0
- package/ciel_runtime_support/providers/fireworks.py +158 -0
- package/ciel_runtime_support/providers/github_copilot_oauth.py +199 -0
- package/ciel_runtime_support/providers/kimi.py +297 -0
- package/ciel_runtime_support/providers/lm_studio.py +76 -0
- package/ciel_runtime_support/providers/meta.py +257 -0
- package/ciel_runtime_support/providers/native.py +194 -0
- package/ciel_runtime_support/providers/nim.py +69 -0
- package/ciel_runtime_support/providers/nvidia.py +158 -0
- package/ciel_runtime_support/providers/nvidia_runtime.py +285 -0
- package/ciel_runtime_support/providers/ollama.py +181 -0
- package/ciel_runtime_support/providers/ollama_context.py +195 -0
- package/ciel_runtime_support/providers/ollama_runtime.py +213 -0
- package/ciel_runtime_support/providers/opencode.py +220 -0
- package/ciel_runtime_support/providers/opencode_go.py +37 -0
- package/ciel_runtime_support/providers/openrouter.py +57 -0
- package/ciel_runtime_support/providers/vllm.py +65 -0
- package/ciel_runtime_support/providers/zai.py +119 -0
- package/ciel_runtime_support/pseudo_tool_parser.py +115 -0
- package/ciel_runtime_support/rate_limit_policy.py +117 -0
- package/ciel_runtime_support/rate_limit_repository.py +154 -0
- package/ciel_runtime_support/registry.py +46 -0
- package/ciel_runtime_support/request_shortcuts.py +253 -0
- package/ciel_runtime_support/request_trace.py +323 -0
- package/ciel_runtime_support/response_collection.py +209 -0
- package/ciel_runtime_support/router_access.py +238 -0
- package/ciel_runtime_support/router_client_lifecycle.py +366 -0
- package/ciel_runtime_support/router_health_policy.py +101 -0
- package/ciel_runtime_support/router_http.py +513 -0
- package/ciel_runtime_support/router_process_lifecycle.py +401 -0
- package/ciel_runtime_support/router_rate_limit_service.py +285 -0
- package/ciel_runtime_support/router_server_runtime.py +103 -0
- package/ciel_runtime_support/router_shortcuts.py +201 -0
- package/ciel_runtime_support/routing_fallback.py +73 -0
- package/ciel_runtime_support/runtime_activity_repository.py +143 -0
- package/ciel_runtime_support/runtime_adapters.py +104 -0
- package/ciel_runtime_support/runtime_command_factory.py +73 -0
- package/ciel_runtime_support/runtime_compatibility.py +50 -0
- package/ciel_runtime_support/runtime_constants.py +178 -0
- package/ciel_runtime_support/runtime_launch.py +1602 -0
- package/ciel_runtime_support/runtime_llm_options.py +312 -0
- package/ciel_runtime_support/runtime_logging.py +161 -0
- package/ciel_runtime_support/runtime_paths.py +157 -0
- package/ciel_runtime_support/runtime_restart.py +84 -0
- package/ciel_runtime_support/runtime_upgrade.py +149 -0
- package/ciel_runtime_support/secure_json_repository.py +55 -0
- package/ciel_runtime_support/session_import.py +356 -0
- package/ciel_runtime_support/settings_repository.py +8 -0
- package/ciel_runtime_support/slash_command_assets.py +211 -0
- package/ciel_runtime_support/sse_stream.py +57 -0
- package/ciel_runtime_support/sse_trace.py +225 -0
- package/ciel_runtime_support/statusline_script.py +593 -0
- package/ciel_runtime_support/statusline_settings.py +53 -0
- package/ciel_runtime_support/stream_chunk_policy.py +18 -0
- package/ciel_runtime_support/streaming_anthropic.py +1955 -0
- package/ciel_runtime_support/synthetic_tool_policy.py +105 -0
- package/ciel_runtime_support/terminal_platform_io.py +127 -0
- package/ciel_runtime_support/timeout_profile.py +196 -0
- package/ciel_runtime_support/tool_dialects.py +85 -0
- package/ciel_runtime_support/tool_exposure_policy.py +63 -0
- package/ciel_runtime_support/tool_guard_hooks.py +218 -0
- package/ciel_runtime_support/tool_request_projection.py +96 -0
- package/ciel_runtime_support/tool_schema.py +483 -0
- package/ciel_runtime_support/tool_side_effect_dedupe.py +95 -0
- package/ciel_runtime_support/ui_text.py +266 -0
- package/ciel_runtime_support/upstream_error_policy.py +104 -0
- package/ciel_runtime_support/upstream_retry.py +419 -0
- package/ciel_runtime_support/upstream_stream_io.py +106 -0
- package/ciel_runtime_support/usage_events.py +96 -0
- package/ciel_runtime_support/visible_stream_filters.py +130 -0
- package/ciel_runtime_support/web_endpoints.py +447 -0
- package/ciel_runtime_support/web_ui.py +915 -0
- package/ciel_runtime_support/web_ui_controller.py +189 -0
- package/ciel_runtime_support/windows_console_input.py +137 -0
- package/ciel_runtime_support/windows_console_mode.py +112 -0
- package/docs/Architecture.md +54 -0
- package/docs/Configuration.md +17 -0
- package/docs/Module-Map.md +1093 -26
- package/docs/Providers.md +46 -1
- package/docs/adr/0001-runtime-bounded-contexts.md +295 -0
- package/npm-bin/run-ciel-runtime.js +22 -2
- package/package.json +9 -2
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Provider launch-readiness application service."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, Callable
|
|
7
|
+
|
|
8
|
+
from .architecture import ProviderAdapter, ProviderConfig, ProviderStatusPolicy
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True, slots=True)
|
|
12
|
+
class ProviderReadinessMode:
|
|
13
|
+
direct_native_anthropic: Callable[..., bool]
|
|
14
|
+
native_agy: Callable[..., bool]
|
|
15
|
+
native_codex: Callable[..., bool]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True, slots=True)
|
|
19
|
+
class ProviderReadinessCapabilities:
|
|
20
|
+
ultracode_enabled: Callable[..., bool]
|
|
21
|
+
supported_capabilities: Callable[..., list[str] | tuple[str, ...] | set[str]]
|
|
22
|
+
current_model: Callable[..., str]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True, slots=True)
|
|
26
|
+
class ProviderReadinessLmStudio:
|
|
27
|
+
ensure_model_loaded: Callable[..., Any]
|
|
28
|
+
save_config: Callable[..., Any]
|
|
29
|
+
runtime_info: Callable[..., dict[str, Any]]
|
|
30
|
+
positive_int: Callable[..., int | None]
|
|
31
|
+
minimum_context: int
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True, slots=True)
|
|
35
|
+
class ProviderReadinessServices:
|
|
36
|
+
mode: ProviderReadinessMode
|
|
37
|
+
capabilities: ProviderReadinessCapabilities
|
|
38
|
+
lm_studio: ProviderReadinessLmStudio
|
|
39
|
+
base_url_status: Callable[..., str]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def launch_readiness_errors(
|
|
43
|
+
cfg: dict[str, Any],
|
|
44
|
+
provider: str,
|
|
45
|
+
pcfg: dict[str, Any],
|
|
46
|
+
adapter: ProviderAdapter,
|
|
47
|
+
contract_config: ProviderConfig,
|
|
48
|
+
status_policy: ProviderStatusPolicy,
|
|
49
|
+
*,
|
|
50
|
+
services: ProviderReadinessServices,
|
|
51
|
+
) -> list[str]:
|
|
52
|
+
mode = services.mode
|
|
53
|
+
if (
|
|
54
|
+
mode.direct_native_anthropic(provider, pcfg)
|
|
55
|
+
or mode.native_agy(provider)
|
|
56
|
+
or mode.native_codex(provider)
|
|
57
|
+
):
|
|
58
|
+
return []
|
|
59
|
+
status = services.base_url_status(provider, pcfg)
|
|
60
|
+
errors: list[str] = []
|
|
61
|
+
if any(marker in status.lower() for marker in ("unreachable", "placeholder", "missing")):
|
|
62
|
+
errors.extend((f"Launch blocked: {status}", status_policy.unreachable_hint))
|
|
63
|
+
api_key_error = adapter.launch_api_key_error(contract_config)
|
|
64
|
+
if api_key_error:
|
|
65
|
+
errors.append(api_key_error)
|
|
66
|
+
capabilities = services.capabilities
|
|
67
|
+
if capabilities.ultracode_enabled(provider, pcfg):
|
|
68
|
+
model = capabilities.current_model(provider, pcfg)
|
|
69
|
+
supported = set(capabilities.supported_capabilities(provider, pcfg, model))
|
|
70
|
+
if "xhigh_effort" not in supported:
|
|
71
|
+
errors.append(
|
|
72
|
+
"Launch blocked: ultracode requires a Claude Code model capability set that includes xhigh_effort. "
|
|
73
|
+
"Use a compatible Claude model or set claude_code_supported_capabilities after verifying the provider/model supports xhigh workflow thinking."
|
|
74
|
+
)
|
|
75
|
+
validators = {
|
|
76
|
+
"none": lambda: None,
|
|
77
|
+
"lm_studio": lambda: _validate_lm_studio(cfg, provider, pcfg, errors, services.lm_studio),
|
|
78
|
+
}
|
|
79
|
+
validators[status_policy.readiness_validation]()
|
|
80
|
+
return errors
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _validate_lm_studio(
|
|
84
|
+
cfg: dict[str, Any],
|
|
85
|
+
provider: str,
|
|
86
|
+
pcfg: dict[str, Any],
|
|
87
|
+
errors: list[str],
|
|
88
|
+
services: ProviderReadinessLmStudio,
|
|
89
|
+
) -> None:
|
|
90
|
+
try:
|
|
91
|
+
services.ensure_model_loaded(pcfg, timeout=1.5)
|
|
92
|
+
services.save_config(cfg)
|
|
93
|
+
except Exception as exc:
|
|
94
|
+
errors.append(
|
|
95
|
+
"Launch blocked: Ciel Runtime could not automatically load the selected LM Studio model "
|
|
96
|
+
f"with the recommended context ({type(exc).__name__}: {exc})."
|
|
97
|
+
)
|
|
98
|
+
return
|
|
99
|
+
info = services.runtime_info(provider, pcfg, timeout=1.5)
|
|
100
|
+
loaded = services.positive_int(info.get("loaded_context_len")) if info else None
|
|
101
|
+
state = str(info.get("state") or "") if info else ""
|
|
102
|
+
if loaded and loaded < services.minimum_context:
|
|
103
|
+
errors.append(
|
|
104
|
+
"Launch blocked: LM Studio loaded context is "
|
|
105
|
+
f"{loaded:,} tokens; Claude Code needs at least {services.minimum_context:,}. "
|
|
106
|
+
"Reload the model with a larger context length."
|
|
107
|
+
)
|
|
108
|
+
elif state and state != "loaded":
|
|
109
|
+
errors.append(
|
|
110
|
+
"Launch blocked: selected LM Studio model is not loaded, so the active context length cannot be verified. "
|
|
111
|
+
f"Load it with at least {services.minimum_context:,} context tokens."
|
|
112
|
+
)
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Provider request credentials, headers, model aliases, and routing access."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Mapping
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from ciel_runtime_support.architecture import MessageProtocol, ProviderRequestPolicy
|
|
10
|
+
from ciel_runtime_support.header_forwarding import project_end_to_end_request_headers
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True, slots=True)
|
|
14
|
+
class ProviderRequestAccessPorts:
|
|
15
|
+
request_policy: Callable[
|
|
16
|
+
[str, dict[str, Any]], ProviderRequestPolicy
|
|
17
|
+
]
|
|
18
|
+
select_api_key: Callable[[str, dict[str, Any]], str | None]
|
|
19
|
+
meaningful_key: Callable[[str], bool]
|
|
20
|
+
adapter_headers: Callable[
|
|
21
|
+
[str, dict[str, Any], str | None], Mapping[str, str]
|
|
22
|
+
]
|
|
23
|
+
inbound_credentials: Callable[
|
|
24
|
+
[str, Any | None], Mapping[str, str] | None
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class ProviderRequestAccessEffects:
|
|
30
|
+
user_agent_headers: Callable[[dict[str, str]], dict[str, str]]
|
|
31
|
+
ncp_model_id: Callable[[str], str]
|
|
32
|
+
normalize_provider: Callable[[Any], str]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, slots=True)
|
|
36
|
+
class ProviderRequestAccessService:
|
|
37
|
+
ports: ProviderRequestAccessPorts
|
|
38
|
+
effects: ProviderRequestAccessEffects
|
|
39
|
+
|
|
40
|
+
def upstream_model(
|
|
41
|
+
self, provider: str, config: dict[str, Any], model: str
|
|
42
|
+
) -> str:
|
|
43
|
+
strategy = self.ports.request_policy(
|
|
44
|
+
provider, config
|
|
45
|
+
).model_alias_strategy
|
|
46
|
+
normalizers = {
|
|
47
|
+
"identity": lambda value: value,
|
|
48
|
+
"ncp": self.effects.ncp_model_id,
|
|
49
|
+
}
|
|
50
|
+
return normalizers[strategy](model)
|
|
51
|
+
|
|
52
|
+
def requires_streaming(
|
|
53
|
+
self, provider: str, config: dict[str, Any]
|
|
54
|
+
) -> bool:
|
|
55
|
+
return self.ports.request_policy(provider, config).stream_required
|
|
56
|
+
|
|
57
|
+
@staticmethod
|
|
58
|
+
def key_from_headers(headers: Any) -> str:
|
|
59
|
+
try:
|
|
60
|
+
key = headers.get("x-api-key")
|
|
61
|
+
if key:
|
|
62
|
+
return str(key)
|
|
63
|
+
authorization = str(
|
|
64
|
+
headers.get("authorization")
|
|
65
|
+
or headers.get("Authorization")
|
|
66
|
+
or ""
|
|
67
|
+
)
|
|
68
|
+
except Exception:
|
|
69
|
+
return ""
|
|
70
|
+
if authorization.lower().startswith("bearer "):
|
|
71
|
+
return authorization[7:].strip()
|
|
72
|
+
return authorization.strip()
|
|
73
|
+
|
|
74
|
+
def headers(
|
|
75
|
+
self,
|
|
76
|
+
provider: str,
|
|
77
|
+
config: dict[str, Any],
|
|
78
|
+
inbound_headers: Any | None = None,
|
|
79
|
+
protocol: MessageProtocol | None = None,
|
|
80
|
+
preserve_inbound: bool = False,
|
|
81
|
+
) -> dict[str, str]:
|
|
82
|
+
policy = self.ports.request_policy(provider, config)
|
|
83
|
+
passthrough = (
|
|
84
|
+
(protocol is not None or preserve_inbound)
|
|
85
|
+
and inbound_headers is not None
|
|
86
|
+
)
|
|
87
|
+
if passthrough:
|
|
88
|
+
headers = self.effects.user_agent_headers(
|
|
89
|
+
project_end_to_end_request_headers(
|
|
90
|
+
inbound_headers,
|
|
91
|
+
replace_credentials=True,
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
else:
|
|
95
|
+
headers = self.effects.user_agent_headers(
|
|
96
|
+
{
|
|
97
|
+
"content-type": "application/json",
|
|
98
|
+
"anthropic-version": "2023-06-01",
|
|
99
|
+
}
|
|
100
|
+
)
|
|
101
|
+
key = (
|
|
102
|
+
self.ports.select_api_key(provider, config)
|
|
103
|
+
or str(config.get("api_key") or "")
|
|
104
|
+
or "not-used"
|
|
105
|
+
)
|
|
106
|
+
meaningful = str(key) if self.ports.meaningful_key(str(key)) else None
|
|
107
|
+
if policy.credential_strategy == "anthropic_inbound":
|
|
108
|
+
credential_headers = self.ports.inbound_credentials(
|
|
109
|
+
meaningful or "", inbound_headers
|
|
110
|
+
)
|
|
111
|
+
if credential_headers is None:
|
|
112
|
+
raise RuntimeError(
|
|
113
|
+
"Anthropic routed mode needs a configured API key "
|
|
114
|
+
"or inbound Claude Code auth headers."
|
|
115
|
+
)
|
|
116
|
+
headers.update(credential_headers)
|
|
117
|
+
else:
|
|
118
|
+
headers.update(
|
|
119
|
+
self.ports.adapter_headers(
|
|
120
|
+
provider, config, meaningful
|
|
121
|
+
)
|
|
122
|
+
)
|
|
123
|
+
return headers
|
|
124
|
+
|
|
125
|
+
def current_provider(
|
|
126
|
+
self, config: dict[str, Any]
|
|
127
|
+
) -> tuple[str, dict[str, Any]]:
|
|
128
|
+
provider = self.effects.normalize_provider(
|
|
129
|
+
config.get("current_provider", "nvidia-hosted")
|
|
130
|
+
)
|
|
131
|
+
return provider, config["providers"][provider]
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
"""Build provider wire requests from normalized Anthropic messages."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, slots=True)
|
|
11
|
+
class ProviderRequestBudget:
|
|
12
|
+
context_limit: Callable[..., int]
|
|
13
|
+
positive_int: Callable[[Any], int]
|
|
14
|
+
configured_output: Callable[..., int]
|
|
15
|
+
cap_output_ratio: Callable[..., int]
|
|
16
|
+
reserve: Callable[..., int]
|
|
17
|
+
compact_anthropic: Callable[..., dict[str, Any]]
|
|
18
|
+
compact_messages: Callable[..., list[dict[str, Any]]]
|
|
19
|
+
compact_requested: Callable[[dict[str, Any]], bool]
|
|
20
|
+
cap_output: Callable[..., int]
|
|
21
|
+
write_usage: Callable[..., None]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class OllamaRequestPorts:
|
|
26
|
+
messages: Callable[[dict[str, Any]], list[dict[str, Any]]]
|
|
27
|
+
tools: Callable[[Any], list[dict[str, Any]]]
|
|
28
|
+
extra_options: Callable[[dict[str, Any]], dict[str, Any]]
|
|
29
|
+
context_limit: Callable[[dict[str, Any]], int]
|
|
30
|
+
num_ctx: Callable[..., int]
|
|
31
|
+
think_enabled: Callable[[str | None, dict[str, Any]], bool]
|
|
32
|
+
# Model-card provenance gate for num_predict: None = omit the parameter
|
|
33
|
+
# entirely so the server default applies (operator 2026-07-29). The
|
|
34
|
+
# default implementation passes the capped value through unchanged.
|
|
35
|
+
num_predict: Callable[[dict[str, Any], int | None], int | None] = lambda _config, capped: capped
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True, slots=True)
|
|
39
|
+
class OpenAIRequestPorts:
|
|
40
|
+
messages: Callable[..., list[dict[str, Any]]]
|
|
41
|
+
tools: Callable[[Any], list[dict[str, Any]]]
|
|
42
|
+
context_limit: Callable[[str, dict[str, Any]], int]
|
|
43
|
+
reasoning_passback: Callable[[str, str, dict[str, Any]], bool]
|
|
44
|
+
repair_tools: Callable[[list[dict[str, Any]]], list[dict[str, Any]]]
|
|
45
|
+
reasoning_effort: Callable[..., str | None]
|
|
46
|
+
sampling_allowed: Callable[..., bool]
|
|
47
|
+
omit_tool_choice: Callable[..., bool]
|
|
48
|
+
tool_choice: Callable[[Any], Any]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True, slots=True)
|
|
52
|
+
class ProviderOptionPorts:
|
|
53
|
+
sampling_providers: frozenset[str]
|
|
54
|
+
sampling_options: tuple[str, ...]
|
|
55
|
+
anthropic_runtime_hints: Callable[[str], dict[str, Any]]
|
|
56
|
+
log: Callable[[str, str], None]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class ProviderRequestBuilder:
|
|
60
|
+
def __init__(
|
|
61
|
+
self,
|
|
62
|
+
budget: ProviderRequestBudget,
|
|
63
|
+
ollama: OllamaRequestPorts,
|
|
64
|
+
openai: OpenAIRequestPorts,
|
|
65
|
+
options: ProviderOptionPorts,
|
|
66
|
+
) -> None:
|
|
67
|
+
self.budget = budget
|
|
68
|
+
self.ollama = ollama
|
|
69
|
+
self.openai = openai
|
|
70
|
+
self.options = options
|
|
71
|
+
|
|
72
|
+
def cap_anthropic_body(
|
|
73
|
+
self, provider: str, config: dict[str, Any], body: dict[str, Any]
|
|
74
|
+
) -> dict[str, Any]:
|
|
75
|
+
capped = dict(body)
|
|
76
|
+
if provider == "anthropic":
|
|
77
|
+
return capped
|
|
78
|
+
context_limit = (
|
|
79
|
+
self.budget.context_limit(provider, config)
|
|
80
|
+
or self.budget.positive_int(config.get("max_model_len"))
|
|
81
|
+
or self.budget.positive_int(config.get("context_window"))
|
|
82
|
+
or (32768 if provider == "vllm" else 0)
|
|
83
|
+
)
|
|
84
|
+
if not context_limit:
|
|
85
|
+
return capped
|
|
86
|
+
configured = self.budget.configured_output(config, capped)
|
|
87
|
+
ratio_capped = self.budget.cap_output_ratio(provider, config, configured)
|
|
88
|
+
if ratio_capped:
|
|
89
|
+
capped["max_tokens"] = ratio_capped
|
|
90
|
+
reserve = self.budget.reserve(config, context_limit)
|
|
91
|
+
output_reserve = self.budget.positive_int(capped.get("max_tokens")) or configured or 4096
|
|
92
|
+
input_budget = max(8192, context_limit - output_reserve - reserve)
|
|
93
|
+
capped = self.budget.compact_anthropic(
|
|
94
|
+
capped,
|
|
95
|
+
input_budget,
|
|
96
|
+
provider=provider,
|
|
97
|
+
pcfg=config,
|
|
98
|
+
model=str(capped.get("model") or config.get("current_model") or ""),
|
|
99
|
+
full_compact_request=self.budget.compact_requested(capped),
|
|
100
|
+
)
|
|
101
|
+
output_tokens = self.budget.cap_output(
|
|
102
|
+
config,
|
|
103
|
+
capped,
|
|
104
|
+
{key: value for key, value in capped.items() if key != "max_tokens"},
|
|
105
|
+
context_limit,
|
|
106
|
+
self.budget.positive_int(capped.get("max_tokens")) or configured,
|
|
107
|
+
)
|
|
108
|
+
if output_tokens:
|
|
109
|
+
capped["max_tokens"] = output_tokens
|
|
110
|
+
return capped
|
|
111
|
+
|
|
112
|
+
def apply_options(
|
|
113
|
+
self, provider: str, config: dict[str, Any], body: dict[str, Any]
|
|
114
|
+
) -> dict[str, Any]:
|
|
115
|
+
if provider not in self.options.sampling_providers:
|
|
116
|
+
return body
|
|
117
|
+
projected = dict(body)
|
|
118
|
+
for key in self.options.sampling_options:
|
|
119
|
+
if config.get(key) is not None:
|
|
120
|
+
projected[key] = config[key]
|
|
121
|
+
return projected
|
|
122
|
+
|
|
123
|
+
def normalize_anthropic_options(
|
|
124
|
+
self,
|
|
125
|
+
provider: str,
|
|
126
|
+
body: dict[str, Any],
|
|
127
|
+
model_id: str,
|
|
128
|
+
) -> dict[str, Any]:
|
|
129
|
+
if provider != "anthropic":
|
|
130
|
+
return body
|
|
131
|
+
unsupported = self.options.anthropic_runtime_hints(model_id).get(
|
|
132
|
+
"unsupported_sampling_parameters"
|
|
133
|
+
)
|
|
134
|
+
if not isinstance(unsupported, list) or not unsupported:
|
|
135
|
+
return body
|
|
136
|
+
projected = dict(body)
|
|
137
|
+
removed = [key for key in unsupported if isinstance(key, str) and key in projected]
|
|
138
|
+
for key in removed:
|
|
139
|
+
projected.pop(key, None)
|
|
140
|
+
if removed:
|
|
141
|
+
self.options.log(
|
|
142
|
+
"INFO",
|
|
143
|
+
f"anthropic_request_options_removed model={model_id} "
|
|
144
|
+
f"keys={','.join(removed)}",
|
|
145
|
+
)
|
|
146
|
+
return projected
|
|
147
|
+
|
|
148
|
+
def ollama_chat(
|
|
149
|
+
self,
|
|
150
|
+
model: str,
|
|
151
|
+
body: dict[str, Any],
|
|
152
|
+
config: dict[str, Any],
|
|
153
|
+
*,
|
|
154
|
+
stream: bool = True,
|
|
155
|
+
provider: str = "ollama",
|
|
156
|
+
) -> dict[str, Any]:
|
|
157
|
+
messages = self.ollama.messages(body)
|
|
158
|
+
tools = self.ollama.tools(body.get("tools"))
|
|
159
|
+
context_limit = self.ollama.context_limit(config)
|
|
160
|
+
configured = self.budget.configured_output(config, body, "num_predict")
|
|
161
|
+
reserve = self.budget.reserve(config, context_limit)
|
|
162
|
+
output_reserve = configured or self.budget.positive_int(body.get("max_tokens")) or 4096
|
|
163
|
+
payload = {"messages": messages, "tools": tools}
|
|
164
|
+
messages = self.budget.compact_messages(
|
|
165
|
+
messages,
|
|
166
|
+
tools,
|
|
167
|
+
max(8192, context_limit - output_reserve - reserve),
|
|
168
|
+
provider=provider,
|
|
169
|
+
model=model,
|
|
170
|
+
pcfg=config,
|
|
171
|
+
full_compact_request=self.budget.compact_requested(body),
|
|
172
|
+
wire="ollama",
|
|
173
|
+
)
|
|
174
|
+
payload["messages"] = messages
|
|
175
|
+
self.budget.write_usage(provider, config, payload, "ollama_upstream")
|
|
176
|
+
request: dict[str, Any] = {
|
|
177
|
+
"model": model,
|
|
178
|
+
"messages": messages,
|
|
179
|
+
"stream": stream,
|
|
180
|
+
"think": self.ollama.think_enabled(model, config),
|
|
181
|
+
}
|
|
182
|
+
if config.get("keep_alive"):
|
|
183
|
+
request["keep_alive"] = str(config["keep_alive"])
|
|
184
|
+
if tools:
|
|
185
|
+
request["tools"] = tools
|
|
186
|
+
options = self.ollama.extra_options(config)
|
|
187
|
+
token_cache: dict[int, int] = {}
|
|
188
|
+
num_ctx = self.ollama.num_ctx(config, payload, _token_cache=token_cache)
|
|
189
|
+
num_predict = self.budget.cap_output(
|
|
190
|
+
config,
|
|
191
|
+
body,
|
|
192
|
+
payload,
|
|
193
|
+
num_ctx,
|
|
194
|
+
configured,
|
|
195
|
+
_token_cache=token_cache,
|
|
196
|
+
)
|
|
197
|
+
num_predict = self.ollama.num_predict(config, num_predict)
|
|
198
|
+
if num_predict:
|
|
199
|
+
options["num_predict"] = num_predict
|
|
200
|
+
if num_ctx:
|
|
201
|
+
options.setdefault("num_ctx", num_ctx)
|
|
202
|
+
if options:
|
|
203
|
+
request["options"] = options
|
|
204
|
+
return request
|
|
205
|
+
|
|
206
|
+
def openai_chat(
|
|
207
|
+
self,
|
|
208
|
+
provider: str,
|
|
209
|
+
model: str,
|
|
210
|
+
body: dict[str, Any],
|
|
211
|
+
config: dict[str, Any],
|
|
212
|
+
*,
|
|
213
|
+
stream: bool = False,
|
|
214
|
+
) -> dict[str, Any]:
|
|
215
|
+
passback = self.openai.reasoning_passback(provider, model, config)
|
|
216
|
+
messages = self.openai.messages(body, reasoning_passback=passback)
|
|
217
|
+
tools = self.openai.tools(body.get("tools"))
|
|
218
|
+
context_limit = self.openai.context_limit(provider, config)
|
|
219
|
+
configured = self.budget.configured_output(config, body)
|
|
220
|
+
reserve = self.budget.reserve(config, context_limit)
|
|
221
|
+
output_reserve = configured or self.budget.positive_int(body.get("max_tokens")) or 4096
|
|
222
|
+
messages = self.budget.compact_messages(
|
|
223
|
+
messages,
|
|
224
|
+
tools,
|
|
225
|
+
max(8192, context_limit - output_reserve - reserve),
|
|
226
|
+
provider=provider,
|
|
227
|
+
model=model,
|
|
228
|
+
pcfg=config,
|
|
229
|
+
full_compact_request=self.budget.compact_requested(body),
|
|
230
|
+
wire="openai",
|
|
231
|
+
)
|
|
232
|
+
messages = self.openai.repair_tools(messages)
|
|
233
|
+
request: dict[str, Any] = {"model": model, "messages": messages, "stream": stream}
|
|
234
|
+
reasoning_effort = self.openai.reasoning_effort(
|
|
235
|
+
provider, model, body, config
|
|
236
|
+
)
|
|
237
|
+
if reasoning_effort:
|
|
238
|
+
request["reasoning_effort"] = reasoning_effort
|
|
239
|
+
if tools:
|
|
240
|
+
request["tools"] = tools
|
|
241
|
+
if body.get("tool_choice") is not None and not self.openai.omit_tool_choice(
|
|
242
|
+
provider, model, body, config
|
|
243
|
+
):
|
|
244
|
+
request["tool_choice"] = self.openai.tool_choice(body.get("tool_choice"))
|
|
245
|
+
if configured:
|
|
246
|
+
request["max_tokens"] = configured
|
|
247
|
+
for key in ("temperature", "top_p"):
|
|
248
|
+
if self.openai.sampling_allowed(provider, config) and config.get(key) is not None:
|
|
249
|
+
request[key] = config[key]
|
|
250
|
+
return request
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Native OpenAI Responses passthrough for compatible model providers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import urllib.request
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Any, Callable, Mapping
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True, slots=True)
|
|
12
|
+
class ProviderResponsesPassthroughPorts:
|
|
13
|
+
project_channel_context: Callable[
|
|
14
|
+
[dict[str, Any]], tuple[dict[str, Any], dict[str, Any]]
|
|
15
|
+
]
|
|
16
|
+
begin_channel_delivery: Callable[[Any, dict[str, Any]], None]
|
|
17
|
+
normalize_model: Callable[[str, dict[str, Any], str], str]
|
|
18
|
+
normalize_request: Callable[
|
|
19
|
+
[str, dict[str, Any], Mapping[str, Any]], Mapping[str, Any]
|
|
20
|
+
]
|
|
21
|
+
upstream_base: Callable[[str, dict[str, Any]], str]
|
|
22
|
+
join_url: Callable[[str, str], str]
|
|
23
|
+
headers: Callable[[str, dict[str, Any], Any], dict[str, str]]
|
|
24
|
+
urlopen: Callable[..., Any]
|
|
25
|
+
timeout_seconds: Callable[[dict[str, Any]], float]
|
|
26
|
+
copy_response_headers: Callable[[Any, Any], None]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ProviderResponsesPassthrough:
|
|
30
|
+
"""Forward Responses without collapsing typed items into another protocol."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, ports: ProviderResponsesPassthroughPorts) -> None:
|
|
33
|
+
self._ports = ports
|
|
34
|
+
|
|
35
|
+
def forward(
|
|
36
|
+
self,
|
|
37
|
+
handler: Any,
|
|
38
|
+
provider: str,
|
|
39
|
+
config: dict[str, Any],
|
|
40
|
+
body: dict[str, Any],
|
|
41
|
+
) -> dict[str, Any]:
|
|
42
|
+
upstream_body = dict(body)
|
|
43
|
+
upstream_body["model"] = self._ports.normalize_model(
|
|
44
|
+
provider, config, str(body.get("model") or "")
|
|
45
|
+
)
|
|
46
|
+
upstream_body = dict(
|
|
47
|
+
self._ports.normalize_request(provider, config, upstream_body)
|
|
48
|
+
)
|
|
49
|
+
upstream_body, delivery_body = self._ports.project_channel_context(
|
|
50
|
+
upstream_body
|
|
51
|
+
)
|
|
52
|
+
self._ports.begin_channel_delivery(handler, delivery_body)
|
|
53
|
+
url = self._ports.join_url(
|
|
54
|
+
self._ports.upstream_base(provider, config),
|
|
55
|
+
"/v1/responses",
|
|
56
|
+
)
|
|
57
|
+
request = urllib.request.Request(
|
|
58
|
+
url,
|
|
59
|
+
data=json.dumps(upstream_body, ensure_ascii=False).encode("utf-8"),
|
|
60
|
+
headers=self._ports.headers(provider, config, handler.headers),
|
|
61
|
+
method="POST",
|
|
62
|
+
)
|
|
63
|
+
with self._ports.urlopen(
|
|
64
|
+
request,
|
|
65
|
+
timeout=self._ports.timeout_seconds(config),
|
|
66
|
+
provider=provider,
|
|
67
|
+
pcfg=config,
|
|
68
|
+
) as response:
|
|
69
|
+
handler.send_response(getattr(response, "status", 200))
|
|
70
|
+
self._ports.copy_response_headers(handler, response.headers)
|
|
71
|
+
handler.end_headers()
|
|
72
|
+
while chunk := response.read(65_536):
|
|
73
|
+
handler.wfile.write(chunk)
|
|
74
|
+
handler.wfile.flush()
|
|
75
|
+
return delivery_body
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
__all__ = [
|
|
79
|
+
"ProviderResponsesPassthrough",
|
|
80
|
+
"ProviderResponsesPassthroughPorts",
|
|
81
|
+
]
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Provider-neutral runtime model metadata discovery service."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Any, Callable
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True, slots=True)
|
|
9
|
+
class ProviderRuntimeInfoPorts:
|
|
10
|
+
strategy: Callable[[str], str]
|
|
11
|
+
lm_studio_info: Callable[..., dict[str, Any] | None]
|
|
12
|
+
request_base: Callable[[str, dict[str, Any]], str]
|
|
13
|
+
current_model: Callable[[str, dict[str, Any]], str]
|
|
14
|
+
http_json: Callable[..., Any]
|
|
15
|
+
join_url: Callable[[str, str], str]
|
|
16
|
+
model_headers: Callable[[str, dict[str, Any]], dict[str, str]]
|
|
17
|
+
positive_int: Callable[[Any], int | None]
|
|
18
|
+
log: Callable[[str, str], None]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, slots=True)
|
|
22
|
+
class ProviderRuntimeInfoService:
|
|
23
|
+
ports: ProviderRuntimeInfoPorts
|
|
24
|
+
|
|
25
|
+
@staticmethod
|
|
26
|
+
def model_context(item: dict[str, Any]) -> int | None:
|
|
27
|
+
keys = (
|
|
28
|
+
"max_model_len",
|
|
29
|
+
"max_context_length",
|
|
30
|
+
"context_length",
|
|
31
|
+
"contextLength",
|
|
32
|
+
"max_context_tokens",
|
|
33
|
+
"max_position_embeddings",
|
|
34
|
+
"trainingContextLength",
|
|
35
|
+
)
|
|
36
|
+
for key in keys:
|
|
37
|
+
value = ProviderRuntimeInfoService._positive_int(item.get(key))
|
|
38
|
+
if value:
|
|
39
|
+
return value
|
|
40
|
+
for key, value in item.items():
|
|
41
|
+
if isinstance(key, str) and key.rsplit(".", 1)[-1] in keys:
|
|
42
|
+
fixed = ProviderRuntimeInfoService._positive_int(value)
|
|
43
|
+
if fixed:
|
|
44
|
+
return fixed
|
|
45
|
+
details = item.get("details")
|
|
46
|
+
if isinstance(details, dict):
|
|
47
|
+
for key in keys:
|
|
48
|
+
value = ProviderRuntimeInfoService._positive_int(details.get(key))
|
|
49
|
+
if value:
|
|
50
|
+
return value
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
@staticmethod
|
|
54
|
+
def _positive_int(value: Any) -> int | None:
|
|
55
|
+
try:
|
|
56
|
+
fixed = int(value)
|
|
57
|
+
except (TypeError, ValueError, OverflowError):
|
|
58
|
+
return None
|
|
59
|
+
return fixed if fixed > 0 else None
|
|
60
|
+
|
|
61
|
+
def discover(
|
|
62
|
+
self,
|
|
63
|
+
provider: str,
|
|
64
|
+
provider_config: dict[str, Any],
|
|
65
|
+
timeout: float = 3.0,
|
|
66
|
+
) -> dict[str, Any] | None:
|
|
67
|
+
strategy = self.ports.strategy(provider)
|
|
68
|
+
if not strategy:
|
|
69
|
+
return None
|
|
70
|
+
if strategy == "lm_studio":
|
|
71
|
+
info = self.ports.lm_studio_info(provider_config, timeout=timeout)
|
|
72
|
+
if info:
|
|
73
|
+
return info
|
|
74
|
+
base = self.ports.request_base(provider, provider_config)
|
|
75
|
+
if not base:
|
|
76
|
+
return None
|
|
77
|
+
current = self.ports.current_model(provider, provider_config)
|
|
78
|
+
models_url = self.ports.join_url(base, "/v1/models")
|
|
79
|
+
try:
|
|
80
|
+
data = self.ports.http_json(
|
|
81
|
+
models_url,
|
|
82
|
+
headers=self.ports.model_headers(provider, provider_config),
|
|
83
|
+
timeout=timeout,
|
|
84
|
+
)
|
|
85
|
+
except Exception as exc:
|
|
86
|
+
self.ports.log(
|
|
87
|
+
"WARN",
|
|
88
|
+
f"provider_runtime_info_failed provider={provider} error={type(exc).__name__}: {exc}",
|
|
89
|
+
)
|
|
90
|
+
return None
|
|
91
|
+
items = data.get("data") if isinstance(data, dict) else None
|
|
92
|
+
if not isinstance(items, list):
|
|
93
|
+
return None
|
|
94
|
+
candidates = [item for item in items if isinstance(item, dict)]
|
|
95
|
+
selected = next((item for item in candidates if str(item.get("id") or "") == current), None)
|
|
96
|
+
selected = selected or (candidates[0] if candidates else None)
|
|
97
|
+
if not selected:
|
|
98
|
+
return None
|
|
99
|
+
return {
|
|
100
|
+
"models_url": models_url,
|
|
101
|
+
"requested_model": current,
|
|
102
|
+
"runtime_model": str(selected.get("id") or ""),
|
|
103
|
+
"max_model_len": self.model_context(selected),
|
|
104
|
+
"owned_by": selected.get("owned_by"),
|
|
105
|
+
"root": selected.get("root"),
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
def context_limit(self, provider: str, provider_config: dict[str, Any], timeout: float = 3.0) -> int | None:
|
|
109
|
+
info = self.discover(provider, provider_config, timeout)
|
|
110
|
+
return self.ports.positive_int(info.get("max_model_len")) if info else None
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
__all__ = ["ProviderRuntimeInfoPorts", "ProviderRuntimeInfoService"]
|