steerable-agent-runtime 0.6.5__tar.gz → 0.6.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/pyproject.toml +1 -1
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/__init__.py +8 -0
- steerable_agent_runtime-0.6.7/src/steerable_agent_runtime/gateway_catalog.py +319 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/openai_compat.py +38 -8
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/openai_responses.py +33 -7
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/model_catalog.py +2046 -1923
- steerable_agent_runtime-0.6.7/src/steerable_agent_runtime/model_info.py +460 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/model_resolve.py +45 -1
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/pool.py +15 -4
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/subagent.py +17 -1
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/todo.py +67 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime.egg-info/SOURCES.txt +3 -0
- steerable_agent_runtime-0.6.7/tests/test_gateway_catalog.py +194 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_llm_responses_wire.py +28 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_llm_wire_helpers.py +18 -5
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_model_catalog.py +2 -2
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_model_equivalence.py +3 -1
- steerable_agent_runtime-0.6.7/tests/test_model_info.py +376 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_subagent.py +126 -0
- steerable_agent_runtime-0.6.7/tests/test_todo_gate.py +94 -0
- steerable_agent_runtime-0.6.5/src/steerable_agent_runtime/model_info.py +0 -247
- steerable_agent_runtime-0.6.5/tests/test_model_info.py +0 -185
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/README.md +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/setup.cfg +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/approval.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/ask_user.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/branch.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/cache_control.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/calibration.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/compaction.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/config.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/default.harness.json +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/default.harness.yaml +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/errors.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/handoff.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/harness.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/harness_spec.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/history.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/hooks.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/__init__.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/compat.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/errors.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/google_genai.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/parts.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/presets.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/loop.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/maintenance.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/mcp.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/orchestration.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/otel.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/plugins.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/pricing.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/pseudo.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/recording.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/reminders.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/replay.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/resume.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/retry.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/skills.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/spill.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/storage/__init__.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/tokens.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/tool_search.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/tools.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/tracing.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/transport/__init__.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime/world_state.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_approval.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_ask_user.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_branch.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_cache_control.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_cache_instrumentation.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_calibration.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_compaction.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_config.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_content_parts.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_error_taxonomy.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_fragment_bounds.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_golden.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_handoff.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_harness.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_harness_spec.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_history.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_history_persistence.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_hooks.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_in_memory_storage.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_llm_gemini_wire.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_llm_stream_e2e.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_long_session.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_loop.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_loop_cancellation.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_loop_replay.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_loop_sandbox_event.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_maintenance.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_mcp.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_model_resolve.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_orchestration.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_otel.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_parallel_tools.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_plugins.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_provider_compat.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_provider_presets.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_pseudo.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_recording.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_reminders.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_replay_crosslang.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_resume.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_retry_hooks.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_safety_gate.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_skills.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_soft_timeout.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_spill.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_sqlite_storage.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_steer.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_storage_contract.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_stream_strip.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_todo.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tokens.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tool_exposure.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tool_hygiene.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tool_router.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_tool_timeout.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_trace_recorder.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_transport_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_transport_sse.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_usage_attribution.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_world_state.py +0 -0
- {steerable_agent_runtime-0.6.5 → steerable_agent_runtime-0.6.7}/tests/test_write_lease.py +0 -0
|
@@ -38,6 +38,7 @@ from .ask_user import (
|
|
|
38
38
|
from .todo import (
|
|
39
39
|
TODO_SCHEMA,
|
|
40
40
|
TODO_TOOL_NAME,
|
|
41
|
+
TodoCompletionGate,
|
|
41
42
|
TodoStore,
|
|
42
43
|
make_todo_write_tool,
|
|
43
44
|
todo_write_tool_descriptor,
|
|
@@ -156,7 +157,10 @@ from .model_info import (
|
|
|
156
157
|
MODEL_INFOS,
|
|
157
158
|
REASONING_EFFORT_ORDER,
|
|
158
159
|
ModelInfo,
|
|
160
|
+
ReasoningEffortUnsupported,
|
|
159
161
|
clamp_reasoning_effort,
|
|
162
|
+
clear_gateway_models,
|
|
163
|
+
register_gateway_models,
|
|
160
164
|
register_model_info,
|
|
161
165
|
resolve_model_info,
|
|
162
166
|
)
|
|
@@ -266,6 +270,7 @@ __all__ = [
|
|
|
266
270
|
"MODEL_TOKEN_FACTORS",
|
|
267
271
|
"REASONING_EFFORT_ORDER",
|
|
268
272
|
"RECORD_FORMAT_VERSION",
|
|
273
|
+
"ReasoningEffortUnsupported",
|
|
269
274
|
"TOOL_SEARCH_NAME",
|
|
270
275
|
"ASK_USER_SCHEMA",
|
|
271
276
|
"ASK_USER_TOOL_NAME",
|
|
@@ -391,6 +396,7 @@ __all__ = [
|
|
|
391
396
|
"ToolExecutor",
|
|
392
397
|
"ToolExposure",
|
|
393
398
|
"ToolRouter",
|
|
399
|
+
"TodoCompletionGate",
|
|
394
400
|
"TodoStore",
|
|
395
401
|
"TraceRecorder",
|
|
396
402
|
"TranscriptAppend",
|
|
@@ -408,6 +414,7 @@ __all__ = [
|
|
|
408
414
|
"branch_label",
|
|
409
415
|
"build_step_decision_event",
|
|
410
416
|
"clamp_reasoning_effort",
|
|
417
|
+
"clear_gateway_models",
|
|
411
418
|
"close_dangling_tool_calls",
|
|
412
419
|
"detect_claimed_execution",
|
|
413
420
|
"detect_deferred_execution",
|
|
@@ -449,6 +456,7 @@ __all__ = [
|
|
|
449
456
|
"qualify_mcp_name",
|
|
450
457
|
"reduce_execution_state",
|
|
451
458
|
"register_mcp_catalog",
|
|
459
|
+
"register_gateway_models",
|
|
452
460
|
"register_model_factor",
|
|
453
461
|
"register_model_info",
|
|
454
462
|
"register_model_price",
|
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
"""Live gateway model listing — the discovery half of the model catalog.
|
|
2
|
+
|
|
3
|
+
The bundled catalog (``model_catalog.py``) is a models.dev snapshot, stale by
|
|
4
|
+
construction; the gateway's own ``GET /models`` is the live list of ids the
|
|
5
|
+
endpoint actually accepts. This module fetches and parses that listing into
|
|
6
|
+
``GatewayModel`` rows, tolerating the two listing shapes in the wild
|
|
7
|
+
(OpenAI-style ``data`` array, models.dev-style ``models`` object) and the
|
|
8
|
+
field variants each uses for window / max-output / pricing — the tolerance
|
|
9
|
+
dsh's ``readListing`` (llm-pi-ai ``discovery.ts``) implements, verified
|
|
10
|
+
against its recorded provider-listing fixtures.
|
|
11
|
+
|
|
12
|
+
Discovery, not routing: a gateway id absent from every catalog still appears
|
|
13
|
+
in the listing with empty capability fields and ``joined_from=None``; the
|
|
14
|
+
listing never gates whether a request may be sent.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import logging
|
|
20
|
+
import time
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from typing import TYPE_CHECKING, Any, Iterable
|
|
23
|
+
|
|
24
|
+
if TYPE_CHECKING:
|
|
25
|
+
from .model_info import ModelInfo
|
|
26
|
+
|
|
27
|
+
_log = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
#: Default freshness window for the in-process listing cache. The gateway
|
|
30
|
+
#: listing changes on deployment boundaries, not per request; 60s keeps a
|
|
31
|
+
#: settings UI snappy without hammering the endpoint.
|
|
32
|
+
DEFAULT_TTL_SEC = 60.0
|
|
33
|
+
|
|
34
|
+
#: Default timeout for the listing request itself.
|
|
35
|
+
DEFAULT_TIMEOUT_SEC = 10.0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class GatewayCatalogError(Exception):
|
|
39
|
+
"""The gateway listing could not be fetched and no cache survives."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, base_url: str, reason: str) -> None:
|
|
42
|
+
self.base_url = base_url
|
|
43
|
+
self.reason = reason
|
|
44
|
+
super().__init__(f"gateway catalog fetch failed for {base_url!r}: {reason}")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True, slots=True)
|
|
48
|
+
class GatewayModel:
|
|
49
|
+
"""One row of the gateway's live listing, normalized across shapes.
|
|
50
|
+
|
|
51
|
+
Capability fields are ``None``/empty when the gateway does not advertise
|
|
52
|
+
them — the catalog join (``merge_with_catalog``) fills what models.dev
|
|
53
|
+
knows, and what neither knows stays unknown.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
id: str
|
|
57
|
+
name: str
|
|
58
|
+
context_window: int | None
|
|
59
|
+
max_output_tokens: int | None
|
|
60
|
+
input_modalities: tuple[str, ...]
|
|
61
|
+
#: USD per million tokens, parsed from OpenRouter-style ``pricing``
|
|
62
|
+
#: strings; ``None`` when the gateway does not advertise pricing.
|
|
63
|
+
prompt_price_per_mtok: float | None
|
|
64
|
+
completion_price_per_mtok: float | None
|
|
65
|
+
#: OpenRouter-style ``supported_parameters`` (``"reasoning"`` present
|
|
66
|
+
#: means the endpoint accepts a reasoning knob, levels unknown).
|
|
67
|
+
supported_parameters: tuple[str, ...]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True, slots=True)
|
|
71
|
+
class GatewayListing:
|
|
72
|
+
"""A fetched listing plus its freshness provenance."""
|
|
73
|
+
|
|
74
|
+
entries: tuple[GatewayModel, ...]
|
|
75
|
+
#: Epoch seconds of the successful fetch (a stale listing keeps the
|
|
76
|
+
#: timestamp of the fetch that produced it, not of the failed refresh).
|
|
77
|
+
fetched_at: float
|
|
78
|
+
#: True when the last refresh failed and this is the previous listing.
|
|
79
|
+
stale: bool
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True, slots=True)
|
|
83
|
+
class GatewayCatalogRow:
|
|
84
|
+
"""A gateway id joined with catalog capabilities, ready for the wire.
|
|
85
|
+
|
|
86
|
+
``joined_from`` names the catalog key that supplied capability fields
|
|
87
|
+
(``None`` when no catalog tier matched — the "capabilities unknown"
|
|
88
|
+
case the UI marks). ``info`` is the merged capability descriptor:
|
|
89
|
+
gateway-advertised fields win where the gateway provides them, catalog
|
|
90
|
+
fields (reasoning levels above all) fill the rest.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
id: str
|
|
94
|
+
name: str
|
|
95
|
+
info: ModelInfo
|
|
96
|
+
joined_from: str | None
|
|
97
|
+
prompt_price_per_mtok: float | None
|
|
98
|
+
completion_price_per_mtok: float | None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ---------------------------------------------------------------------------
|
|
102
|
+
# Parsing (pure)
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _first_int(*values: Any) -> int | None:
|
|
107
|
+
for value in values:
|
|
108
|
+
if isinstance(value, bool):
|
|
109
|
+
continue
|
|
110
|
+
if isinstance(value, (int, float)) and value > 0:
|
|
111
|
+
return int(value)
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _first_str_list(value: Any) -> tuple[str, ...]:
|
|
116
|
+
if isinstance(value, (list, tuple)):
|
|
117
|
+
return tuple(str(v) for v in value if isinstance(v, str))
|
|
118
|
+
return ()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _price_per_mtok(pricing: Any, key: str) -> float | None:
|
|
122
|
+
"""OpenRouter ``pricing`` values are USD-per-token strings."""
|
|
123
|
+
if not isinstance(pricing, dict):
|
|
124
|
+
return None
|
|
125
|
+
raw = pricing.get(key)
|
|
126
|
+
if not isinstance(raw, (str, int, float)) or isinstance(raw, bool):
|
|
127
|
+
return None
|
|
128
|
+
try:
|
|
129
|
+
return float(raw) * 1_000_000
|
|
130
|
+
except (TypeError, ValueError):
|
|
131
|
+
return None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _parse_entry(entry: Any, *, key_hint: str | None) -> GatewayModel | None:
|
|
135
|
+
if not isinstance(entry, dict):
|
|
136
|
+
return None
|
|
137
|
+
model_id = entry.get("id") or key_hint
|
|
138
|
+
if not isinstance(model_id, str) or not model_id:
|
|
139
|
+
return None
|
|
140
|
+
name = entry.get("name") or entry.get("display_name") or entry.get("displayName")
|
|
141
|
+
limit = entry.get("limit") if isinstance(entry.get("limit"), dict) else {}
|
|
142
|
+
top_provider = (
|
|
143
|
+
entry.get("top_provider") if isinstance(entry.get("top_provider"), dict) else {}
|
|
144
|
+
)
|
|
145
|
+
architecture = (
|
|
146
|
+
entry.get("architecture") if isinstance(entry.get("architecture"), dict) else {}
|
|
147
|
+
)
|
|
148
|
+
return GatewayModel(
|
|
149
|
+
id=model_id,
|
|
150
|
+
name=str(name) if isinstance(name, str) and name else model_id,
|
|
151
|
+
context_window=_first_int(
|
|
152
|
+
entry.get("context_length"),
|
|
153
|
+
entry.get("context_window"),
|
|
154
|
+
entry.get("contextWindow"),
|
|
155
|
+
entry.get("max_input_tokens"),
|
|
156
|
+
limit.get("context"),
|
|
157
|
+
),
|
|
158
|
+
max_output_tokens=_first_int(
|
|
159
|
+
top_provider.get("max_completion_tokens"),
|
|
160
|
+
entry.get("max_completion_tokens"),
|
|
161
|
+
entry.get("max_output_tokens"),
|
|
162
|
+
entry.get("maxOutputTokens"),
|
|
163
|
+
entry.get("maxTokens"),
|
|
164
|
+
entry.get("max_tokens"),
|
|
165
|
+
limit.get("output"),
|
|
166
|
+
),
|
|
167
|
+
input_modalities=_first_str_list(architecture.get("input_modalities")),
|
|
168
|
+
prompt_price_per_mtok=_price_per_mtok(entry.get("pricing"), "prompt"),
|
|
169
|
+
completion_price_per_mtok=_price_per_mtok(entry.get("pricing"), "completion"),
|
|
170
|
+
supported_parameters=_first_str_list(entry.get("supported_parameters")),
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def parse_models_listing(payload: Any) -> list[GatewayModel]:
|
|
175
|
+
"""Parse a ``GET /models`` body into normalized rows.
|
|
176
|
+
|
|
177
|
+
Two shapes are in the wild: OpenAI-style ``{"data": [...]}`` (the common
|
|
178
|
+
case, OpenRouter's enriched variant included) and models.dev-style
|
|
179
|
+
``{"models": {...}}`` where the property key is the id. Anything else is
|
|
180
|
+
a parse error the caller surfaces as ``GatewayCatalogError``.
|
|
181
|
+
"""
|
|
182
|
+
if not isinstance(payload, dict):
|
|
183
|
+
raise ValueError("listing payload is not a JSON object")
|
|
184
|
+
data = payload.get("data")
|
|
185
|
+
if isinstance(data, list):
|
|
186
|
+
rows = [_parse_entry(entry, key_hint=None) for entry in data]
|
|
187
|
+
return [row for row in rows if row is not None]
|
|
188
|
+
models = payload.get("models")
|
|
189
|
+
if isinstance(models, dict):
|
|
190
|
+
rows = [_parse_entry(entry, key_hint=key) for key, entry in models.items()]
|
|
191
|
+
return [row for row in rows if row is not None]
|
|
192
|
+
if isinstance(models, list):
|
|
193
|
+
rows = [_parse_entry(entry, key_hint=None) for entry in models]
|
|
194
|
+
return [row for row in rows if row is not None]
|
|
195
|
+
raise ValueError("listing has neither a 'data' array nor a 'models' map")
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
# ---------------------------------------------------------------------------
|
|
199
|
+
# Catalog join (pure)
|
|
200
|
+
# ---------------------------------------------------------------------------
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def merge_with_catalog(models: Iterable[GatewayModel]) -> list[GatewayCatalogRow]:
|
|
204
|
+
"""Join gateway rows with catalog capabilities (cross-provider leaf).
|
|
205
|
+
|
|
206
|
+
The gateway's own fields (window, modalities) win where advertised —
|
|
207
|
+
they describe this deployment. Capability fields come from
|
|
208
|
+
``resolve_model_info`` itself — the very resolution the request path's
|
|
209
|
+
``clamp_reasoning_effort`` uses, legacy-union rule included — so
|
|
210
|
+
``models.list`` and the request path never disagree about a model's
|
|
211
|
+
knob (a picker that hides a level the backend would accept is the same
|
|
212
|
+
class of drift as silently dropping one). ``joined_from`` keeps the
|
|
213
|
+
leaf-join provenance: ``None`` when no catalog tier matched, even if
|
|
214
|
+
the hand-owned legacy table knows the model.
|
|
215
|
+
"""
|
|
216
|
+
from .model_info import (
|
|
217
|
+
TOOL_FORMAT_OPENAI,
|
|
218
|
+
ModelInfo,
|
|
219
|
+
resolve_model_info,
|
|
220
|
+
)
|
|
221
|
+
from .model_resolve import resolve_leaf_cross_provider
|
|
222
|
+
|
|
223
|
+
rows: list[GatewayCatalogRow] = []
|
|
224
|
+
for model in models:
|
|
225
|
+
hit = resolve_leaf_cross_provider(model.id)
|
|
226
|
+
base = resolve_model_info(model.id)
|
|
227
|
+
context_window = (
|
|
228
|
+
model.context_window
|
|
229
|
+
if model.context_window is not None
|
|
230
|
+
else base.context_window
|
|
231
|
+
)
|
|
232
|
+
modalities = (
|
|
233
|
+
frozenset(model.input_modalities)
|
|
234
|
+
if model.input_modalities
|
|
235
|
+
else base.modalities
|
|
236
|
+
)
|
|
237
|
+
rows.append(
|
|
238
|
+
GatewayCatalogRow(
|
|
239
|
+
id=model.id,
|
|
240
|
+
name=model.name,
|
|
241
|
+
info=ModelInfo(
|
|
242
|
+
pattern=model.id.lower(),
|
|
243
|
+
context_window=context_window,
|
|
244
|
+
modalities=modalities,
|
|
245
|
+
tool_format=TOOL_FORMAT_OPENAI,
|
|
246
|
+
reasoning_levels=base.reasoning_levels,
|
|
247
|
+
),
|
|
248
|
+
joined_from=hit.key if hit is not None else None,
|
|
249
|
+
prompt_price_per_mtok=model.prompt_price_per_mtok,
|
|
250
|
+
completion_price_per_mtok=model.completion_price_per_mtok,
|
|
251
|
+
)
|
|
252
|
+
)
|
|
253
|
+
return rows
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
# ---------------------------------------------------------------------------
|
|
257
|
+
# Fetch (network, TTL-cached, stale-on-error)
|
|
258
|
+
# ---------------------------------------------------------------------------
|
|
259
|
+
|
|
260
|
+
#: base_url (normalized) -> (time.monotonic() at fetch, listing). The wall
|
|
261
|
+
#: clock ``listing.fetched_at`` is for display; the monotonic clock drives
|
|
262
|
+
#: TTL so a clock jump cannot expire or resurrect a listing.
|
|
263
|
+
_cache: dict[str, tuple[float, GatewayListing]] = {}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _cache_key(base_url: str) -> str:
|
|
267
|
+
return base_url.rstrip("/").lower()
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def clear_gateway_cache() -> None:
|
|
271
|
+
"""Drop every cached listing (tests, forced refresh)."""
|
|
272
|
+
_cache.clear()
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
async def fetch_gateway_models(
|
|
276
|
+
base_url: str,
|
|
277
|
+
api_key: str | None = None,
|
|
278
|
+
*,
|
|
279
|
+
timeout_sec: float = DEFAULT_TIMEOUT_SEC,
|
|
280
|
+
ttl_sec: float = DEFAULT_TTL_SEC,
|
|
281
|
+
) -> GatewayListing:
|
|
282
|
+
"""Fetch the gateway's live model listing, with a process-local cache.
|
|
283
|
+
|
|
284
|
+
A fresh-enough cached listing is served without network. On refresh
|
|
285
|
+
failure the previous listing is returned marked ``stale`` — the catalog
|
|
286
|
+
degrades instead of disappearing when the gateway flaps. With no cache
|
|
287
|
+
to fall back on, ``GatewayCatalogError`` is raised and the caller
|
|
288
|
+
decides (the sidecar answers ``catalog_status: "offline"``).
|
|
289
|
+
"""
|
|
290
|
+
import httpx # local import — keeps the runtime importable without httpx
|
|
291
|
+
|
|
292
|
+
from .llm.system_proxy import client_env_kwargs
|
|
293
|
+
|
|
294
|
+
key = _cache_key(base_url)
|
|
295
|
+
cached = _cache.get(key)
|
|
296
|
+
if cached is not None and (time.monotonic() - cached[0]) < ttl_sec:
|
|
297
|
+
return cached[1]
|
|
298
|
+
|
|
299
|
+
url = f"{base_url.rstrip('/')}/models"
|
|
300
|
+
headers = {"Authorization": f"Bearer {api_key}"} if api_key else {}
|
|
301
|
+
try:
|
|
302
|
+
async with httpx.AsyncClient(
|
|
303
|
+
timeout=httpx.Timeout(timeout_sec),
|
|
304
|
+
**client_env_kwargs(base_url),
|
|
305
|
+
) as client:
|
|
306
|
+
response = await client.get(url, headers=headers)
|
|
307
|
+
response.raise_for_status()
|
|
308
|
+
entries = tuple(parse_models_listing(response.json()))
|
|
309
|
+
except Exception as exc:
|
|
310
|
+
if cached is not None:
|
|
311
|
+
_log.info("gateway listing refresh failed (%s); serving stale", exc)
|
|
312
|
+
return GatewayListing(
|
|
313
|
+
entries=cached[1].entries, fetched_at=cached[1].fetched_at, stale=True
|
|
314
|
+
)
|
|
315
|
+
raise GatewayCatalogError(base_url, str(exc)) from exc
|
|
316
|
+
|
|
317
|
+
listing = GatewayListing(entries=entries, fetched_at=time.time(), stale=False)
|
|
318
|
+
_cache[key] = (time.monotonic(), listing)
|
|
319
|
+
return listing
|
|
@@ -24,7 +24,10 @@ from typing import Any, Literal
|
|
|
24
24
|
|
|
25
25
|
from steerable_agent_protocol.generated import ToolCall
|
|
26
26
|
|
|
27
|
-
from ..model_info import
|
|
27
|
+
from ..model_info import (
|
|
28
|
+
ReasoningEffortUnsupported,
|
|
29
|
+
clamp_reasoning_effort,
|
|
30
|
+
)
|
|
28
31
|
from . import LLMMessage, LLMStreamChunk, LLMUsage
|
|
29
32
|
from .compat import OpenAICompatFlags
|
|
30
33
|
from .errors import LLMError, classify_http_status, parse_retry_after_ms
|
|
@@ -126,6 +129,12 @@ class OpenAICompatProvider:
|
|
|
126
129
|
#: a ``ProviderPreset`` instance pins that preset regardless of the
|
|
127
130
|
#: registry (host settings UIs send an explicit choice this way).
|
|
128
131
|
preset: ProviderPreset | Literal["auto", "off"] = "auto"
|
|
132
|
+
#: Per-request reasoning effort from the host's model picker; wins over
|
|
133
|
+
#: ``STEERABLE_REASONING_EFFORT`` and the preset default. Explicitly
|
|
134
|
+
#: requested effort is validated strict (``clamp_reasoning_effort``) —
|
|
135
|
+
#: a level the model cannot honor fails the request instead of being
|
|
136
|
+
#: silently dropped (EVALS 2.5.22).
|
|
137
|
+
reasoning_effort: str | None = None
|
|
129
138
|
|
|
130
139
|
def __post_init__(self) -> None:
|
|
131
140
|
if not self.base_url:
|
|
@@ -376,14 +385,35 @@ class OpenAICompatProvider:
|
|
|
376
385
|
# actually supports (structured ModelInfo replaces the raw env
|
|
377
386
|
# passthrough). A model with no reasoning knob gets no parameter at
|
|
378
387
|
# all — sending one would be an unsupported-field error on strict
|
|
379
|
-
# APIs.
|
|
380
|
-
#
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
388
|
+
# APIs. Precedence: the host's per-request pick, then the env var,
|
|
389
|
+
# then the preset's documented default. The compat gate runs FIRST:
|
|
390
|
+
# a vendor that rejects the field outright (Moonshot thinking 400s
|
|
391
|
+
# on reasoning_effort) must drop it silently, not fail. Otherwise an
|
|
392
|
+
# explicit request is strict: an effort the resolved catalog entry
|
|
393
|
+
# cannot honor raises instead of being silently dropped
|
|
394
|
+
# (EVALS 2.5.22).
|
|
395
|
+
requested_effort = (
|
|
396
|
+
self.reasoning_effort
|
|
397
|
+
or os.environ.get("STEERABLE_REASONING_EFFORT", "")
|
|
398
|
+
or (preset.reasoning_effort if preset is not None else "")
|
|
385
399
|
)
|
|
386
|
-
|
|
400
|
+
effort: str | None = None
|
|
401
|
+
if requested_effort and compat.supports_reasoning_effort:
|
|
402
|
+
try:
|
|
403
|
+
effort = clamp_reasoning_effort(
|
|
404
|
+
self.model,
|
|
405
|
+
requested_effort,
|
|
406
|
+
provider=self.name,
|
|
407
|
+
base_url=self.base_url,
|
|
408
|
+
strict=True,
|
|
409
|
+
)
|
|
410
|
+
except ReasoningEffortUnsupported as exc:
|
|
411
|
+
raise LLMError(
|
|
412
|
+
f"{self.name}: {exc}",
|
|
413
|
+
kind="invalid_request",
|
|
414
|
+
provider=self.name,
|
|
415
|
+
) from exc
|
|
416
|
+
if effort:
|
|
387
417
|
# GLM-5.3 default is ``max``; ``high`` is a downgrade. TB uses max.
|
|
388
418
|
if "reasoning_effort" not in body:
|
|
389
419
|
body["reasoning_effort"] = effort
|
|
@@ -23,7 +23,7 @@ from typing import Any, Literal
|
|
|
23
23
|
|
|
24
24
|
from steerable_agent_protocol.generated import ToolCall
|
|
25
25
|
|
|
26
|
-
from ..model_info import clamp_reasoning_effort
|
|
26
|
+
from ..model_info import ReasoningEffortUnsupported, clamp_reasoning_effort
|
|
27
27
|
from . import LLMMessage, LLMStreamChunk, LLMUsage
|
|
28
28
|
from .errors import LLMError, classify_http_status, parse_retry_after_ms
|
|
29
29
|
from .parts import ImagePart, TextPart
|
|
@@ -75,6 +75,12 @@ class OpenAIResponsesProvider:
|
|
|
75
75
|
api_key: str | None = None
|
|
76
76
|
default_temperature: float | None = None
|
|
77
77
|
preset: ProviderPreset | Literal["auto", "off"] = "auto"
|
|
78
|
+
#: Per-request reasoning effort from the host's model picker; wins over
|
|
79
|
+
#: ``STEERABLE_REASONING_EFFORT`` and the preset default. Explicitly
|
|
80
|
+
#: requested effort is validated strict (``clamp_reasoning_effort``) —
|
|
81
|
+
#: a level the model cannot honor fails the request instead of being
|
|
82
|
+
#: silently dropped (EVALS 2.5.22).
|
|
83
|
+
reasoning_effort: str | None = None
|
|
78
84
|
|
|
79
85
|
def __post_init__(self) -> None:
|
|
80
86
|
if not self.base_url:
|
|
@@ -284,13 +290,33 @@ class OpenAIResponsesProvider:
|
|
|
284
290
|
for key, value in preset.extra_body.items():
|
|
285
291
|
if key not in body:
|
|
286
292
|
body[key] = value
|
|
287
|
-
# Reasoning lives under ``reasoning.effort`` on this wire
|
|
288
|
-
#
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
+
# Reasoning lives under ``reasoning.effort`` on this wire.
|
|
294
|
+
# Precedence: the host's per-request pick, then the env var, then
|
|
295
|
+
# the preset's documented default. An explicit request is validated
|
|
296
|
+
# strict against the resolved catalog entry — a level the model
|
|
297
|
+
# cannot honor raises instead of being silently dropped
|
|
298
|
+
# (EVALS 2.5.22), same as the chat-completions path.
|
|
299
|
+
requested_effort = (
|
|
300
|
+
self.reasoning_effort
|
|
301
|
+
or os.environ.get("STEERABLE_REASONING_EFFORT", "")
|
|
302
|
+
or (preset.reasoning_effort if preset is not None else "")
|
|
293
303
|
)
|
|
304
|
+
effort: str | None = None
|
|
305
|
+
if requested_effort:
|
|
306
|
+
try:
|
|
307
|
+
effort = clamp_reasoning_effort(
|
|
308
|
+
self.model,
|
|
309
|
+
requested_effort,
|
|
310
|
+
provider=self.name,
|
|
311
|
+
base_url=self.base_url,
|
|
312
|
+
strict=True,
|
|
313
|
+
)
|
|
314
|
+
except ReasoningEffortUnsupported as exc:
|
|
315
|
+
raise LLMError(
|
|
316
|
+
f"{self.name}: {exc}",
|
|
317
|
+
kind="invalid_request",
|
|
318
|
+
provider=self.name,
|
|
319
|
+
) from exc
|
|
294
320
|
if effort and "reasoning" not in body:
|
|
295
321
|
body["reasoning"] = {"effort": effort}
|
|
296
322
|
if "include" not in body:
|