steerable-agent-runtime 0.6.4__tar.gz → 0.6.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/pyproject.toml +1 -1
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/__init__.py +8 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/ask_user.py +17 -7
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/compaction.py +71 -4
- steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/gateway_catalog.py +319 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/harness.py +7 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/__init__.py +4 -0
- steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/llm/google_genai.py +444 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/openai_compat.py +38 -8
- steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/llm/openai_responses.py +544 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/model_catalog.py +2046 -1923
- steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/model_info.py +460 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/model_resolve.py +45 -1
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pool.py +15 -4
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/reminders.py +33 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/subagent.py +17 -1
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/todo.py +67 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/SOURCES.txt +8 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_ask_user.py +29 -6
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_compaction.py +289 -0
- steerable_agent_runtime-0.6.6/tests/test_gateway_catalog.py +194 -0
- steerable_agent_runtime-0.6.6/tests/test_llm_gemini_wire.py +256 -0
- steerable_agent_runtime-0.6.6/tests/test_llm_responses_wire.py +396 -0
- steerable_agent_runtime-0.6.6/tests/test_llm_stream_e2e.py +233 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_llm_wire_helpers.py +18 -5
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_catalog.py +2 -2
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_equivalence.py +3 -1
- steerable_agent_runtime-0.6.6/tests/test_model_info.py +376 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_subagent.py +126 -0
- steerable_agent_runtime-0.6.6/tests/test_todo_gate.py +94 -0
- steerable_agent_runtime-0.6.4/src/steerable_agent_runtime/model_info.py +0 -247
- steerable_agent_runtime-0.6.4/tests/test_model_info.py +0 -185
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/README.md +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/setup.cfg +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/approval.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/branch.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/cache_control.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/calibration.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/config.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/default.harness.json +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/default.harness.yaml +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/errors.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/handoff.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/harness_spec.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/history.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/hooks.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/compat.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/errors.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/parts.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/presets.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/loop.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/maintenance.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/mcp.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/orchestration.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/otel.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/plugins.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pricing.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pseudo.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/recording.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/replay.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/resume.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/retry.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/skills.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/spill.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/__init__.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tokens.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tool_search.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tools.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tracing.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/__init__.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/world_state.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_approval.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_branch.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_cache_control.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_cache_instrumentation.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_calibration.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_config.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_content_parts.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_error_taxonomy.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_fragment_bounds.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_golden.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_handoff.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_harness.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_harness_spec.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_history.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_history_persistence.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_hooks.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_in_memory_storage.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_long_session.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_cancellation.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_replay.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_sandbox_event.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_maintenance.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_mcp.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_resolve.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_orchestration.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_otel.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_parallel_tools.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_plugins.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_provider_compat.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_provider_presets.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_pseudo.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_recording.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_reminders.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_replay_crosslang.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_resume.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_retry_hooks.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_safety_gate.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_skills.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_soft_timeout.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_spill.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_sqlite_storage.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_steer.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_storage_contract.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_stream_strip.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_todo.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tokens.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_exposure.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_hygiene.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_router.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_timeout.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_trace_recorder.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_transport_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_transport_sse.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_usage_attribution.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_world_state.py +0 -0
- {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_write_lease.py +0 -0
|
@@ -38,6 +38,7 @@ from .ask_user import (
|
|
|
38
38
|
from .todo import (
|
|
39
39
|
TODO_SCHEMA,
|
|
40
40
|
TODO_TOOL_NAME,
|
|
41
|
+
TodoCompletionGate,
|
|
41
42
|
TodoStore,
|
|
42
43
|
make_todo_write_tool,
|
|
43
44
|
todo_write_tool_descriptor,
|
|
@@ -156,7 +157,10 @@ from .model_info import (
|
|
|
156
157
|
MODEL_INFOS,
|
|
157
158
|
REASONING_EFFORT_ORDER,
|
|
158
159
|
ModelInfo,
|
|
160
|
+
ReasoningEffortUnsupported,
|
|
159
161
|
clamp_reasoning_effort,
|
|
162
|
+
clear_gateway_models,
|
|
163
|
+
register_gateway_models,
|
|
160
164
|
register_model_info,
|
|
161
165
|
resolve_model_info,
|
|
162
166
|
)
|
|
@@ -266,6 +270,7 @@ __all__ = [
|
|
|
266
270
|
"MODEL_TOKEN_FACTORS",
|
|
267
271
|
"REASONING_EFFORT_ORDER",
|
|
268
272
|
"RECORD_FORMAT_VERSION",
|
|
273
|
+
"ReasoningEffortUnsupported",
|
|
269
274
|
"TOOL_SEARCH_NAME",
|
|
270
275
|
"ASK_USER_SCHEMA",
|
|
271
276
|
"ASK_USER_TOOL_NAME",
|
|
@@ -391,6 +396,7 @@ __all__ = [
|
|
|
391
396
|
"ToolExecutor",
|
|
392
397
|
"ToolExposure",
|
|
393
398
|
"ToolRouter",
|
|
399
|
+
"TodoCompletionGate",
|
|
394
400
|
"TodoStore",
|
|
395
401
|
"TraceRecorder",
|
|
396
402
|
"TranscriptAppend",
|
|
@@ -408,6 +414,7 @@ __all__ = [
|
|
|
408
414
|
"branch_label",
|
|
409
415
|
"build_step_decision_event",
|
|
410
416
|
"clamp_reasoning_effort",
|
|
417
|
+
"clear_gateway_models",
|
|
411
418
|
"close_dangling_tool_calls",
|
|
412
419
|
"detect_claimed_execution",
|
|
413
420
|
"detect_deferred_execution",
|
|
@@ -449,6 +456,7 @@ __all__ = [
|
|
|
449
456
|
"qualify_mcp_name",
|
|
450
457
|
"reduce_execution_state",
|
|
451
458
|
"register_mcp_catalog",
|
|
459
|
+
"register_gateway_models",
|
|
452
460
|
"register_model_factor",
|
|
453
461
|
"register_model_info",
|
|
454
462
|
"register_model_price",
|
|
@@ -71,9 +71,16 @@ ASK_USER_SCHEMA: dict[str, Any] = {
|
|
|
71
71
|
),
|
|
72
72
|
},
|
|
73
73
|
"placeholder": {"type": "string"},
|
|
74
|
-
"multiSelect": {
|
|
74
|
+
"multiSelect": {
|
|
75
|
+
"type": "boolean",
|
|
76
|
+
"description": (
|
|
77
|
+
"Whether the user may pick several options. "
|
|
78
|
+
"Required (CC parity): commit to single vs multi "
|
|
79
|
+
"explicitly — pass false for a single-select."
|
|
80
|
+
),
|
|
81
|
+
},
|
|
75
82
|
},
|
|
76
|
-
"required": ["id", "text"],
|
|
83
|
+
"required": ["id", "text", "multiSelect"],
|
|
77
84
|
},
|
|
78
85
|
},
|
|
79
86
|
},
|
|
@@ -170,12 +177,15 @@ def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]
|
|
|
170
177
|
f"ask_user: questions[{index}].header is {len(header)} chars; "
|
|
171
178
|
f"the chip label must be <= {_MAX_HEADER_LEN}."
|
|
172
179
|
)
|
|
173
|
-
# multiSelect: CC requires the model to commit to single vs multi
|
|
174
|
-
#
|
|
175
|
-
#
|
|
180
|
+
# multiSelect: CC requires the model to commit to single vs multi —
|
|
181
|
+
# the field is required, so an omission is a model error to fix, not
|
|
182
|
+
# a default to assume.
|
|
176
183
|
if "multiSelect" not in q or q["multiSelect"] is None:
|
|
177
|
-
|
|
178
|
-
|
|
184
|
+
raise ToolDispatchError(
|
|
185
|
+
f"ask_user: questions[{index}].multiSelect is required "
|
|
186
|
+
"(CC parity) — pass false for a single-select question."
|
|
187
|
+
)
|
|
188
|
+
if not isinstance(q["multiSelect"], bool):
|
|
179
189
|
raise ToolDispatchError(
|
|
180
190
|
f"ask_user: questions[{index}].multiSelect must be a boolean, "
|
|
181
191
|
f"got {type(q['multiSelect']).__name__}."
|
|
@@ -57,6 +57,14 @@ Two safety rails share the machinery:
|
|
|
57
57
|
the turn still fails loud (bounded retries, then a named error) instead
|
|
58
58
|
of spinning. A round that lands under threshold — or any successful
|
|
59
59
|
compaction — resets the count.
|
|
60
|
+
- A **rapid-refill breaker** (CC parity): a compaction that lands under
|
|
61
|
+
threshold but refills within ``rapid_refill_window_rounds`` rounds of the
|
|
62
|
+
previous one counts as a refill; ``max_rapid_refills`` consecutive
|
|
63
|
+
refills open the same circuit — the transcript is churning faster than
|
|
64
|
+
compaction can help, so further rewrites only kill the prompt cache.
|
|
65
|
+
The tripping round appends a ``CompactionThrashingReminder`` (model- and
|
|
66
|
+
UI-visible) with an actionable converge notice. ``circuit_reason``
|
|
67
|
+
records which breaker tripped.
|
|
60
68
|
"""
|
|
61
69
|
|
|
62
70
|
from __future__ import annotations
|
|
@@ -64,9 +72,10 @@ from __future__ import annotations
|
|
|
64
72
|
from collections.abc import Sequence
|
|
65
73
|
from typing import Any
|
|
66
74
|
|
|
67
|
-
from .hooks import NoopHooks, PreStepAction, RetryAction, RewriteRequest
|
|
75
|
+
from .hooks import NoopHooks, PreStepAction, RetryAction, RewriteRequest, TranscriptAppend
|
|
68
76
|
from .llm import LLMMessage, LLMProvider
|
|
69
77
|
from .llm.errors import classify_error
|
|
78
|
+
from .reminders import CompactionThrashingReminder
|
|
70
79
|
from .tokens import estimate_tokens
|
|
71
80
|
|
|
72
81
|
_SUMMARY_MARKER = "[context compacted: earlier conversation summarized]"
|
|
@@ -108,6 +117,8 @@ class CompactionHooks(NoopHooks):
|
|
|
108
117
|
recompact_margin_ratio: float = 0.1,
|
|
109
118
|
fold_excerpt_chars: int = _FOLD_EXCERPT_CHARS,
|
|
110
119
|
micro_compact_interval_rounds: int = 0,
|
|
120
|
+
rapid_refill_window_rounds: int = 3,
|
|
121
|
+
max_rapid_refills: int = 3,
|
|
111
122
|
) -> None:
|
|
112
123
|
if not 0 < threshold_ratio <= 1:
|
|
113
124
|
raise ValueError("threshold_ratio must be in (0, 1]")
|
|
@@ -115,6 +126,10 @@ class CompactionHooks(NoopHooks):
|
|
|
115
126
|
raise ValueError("recompact_margin_ratio must be >= 0")
|
|
116
127
|
if not 0 <= micro_compact_interval_rounds:
|
|
117
128
|
raise ValueError("micro_compact_interval_rounds must be >= 0")
|
|
129
|
+
if not 1 <= rapid_refill_window_rounds:
|
|
130
|
+
raise ValueError("rapid_refill_window_rounds must be >= 1")
|
|
131
|
+
if not 1 <= max_rapid_refills:
|
|
132
|
+
raise ValueError("max_rapid_refills must be >= 1")
|
|
118
133
|
self._max_tokens = max_context_tokens
|
|
119
134
|
self._threshold = threshold_ratio
|
|
120
135
|
self._keep_last = keep_last_messages
|
|
@@ -155,9 +170,22 @@ class CompactionHooks(NoopHooks):
|
|
|
155
170
|
#: parity). The overflow path is bounded separately and unaffected.
|
|
156
171
|
self.max_consecutive_failures = 3
|
|
157
172
|
self._consecutive_failures = 0
|
|
173
|
+
#: Rapid-refill breaker (CC parity): a compaction that *worked* but
|
|
174
|
+
#: whose freed space refills within ``rapid_refill_window_rounds``
|
|
175
|
+
#: rounds counts as a refill; ``max_rapid_refills`` consecutive
|
|
176
|
+
#: refills open the circuit — the transcript is churning faster than
|
|
177
|
+
#: compaction can help, and each rewrite only kills the prompt-cache
|
|
178
|
+
#: prefix. Distinct from the failure breaker above, which catches
|
|
179
|
+
#: compactions that never get under threshold at all.
|
|
180
|
+
self.rapid_refill_window_rounds = rapid_refill_window_rounds
|
|
181
|
+
self.max_rapid_refills = max_rapid_refills
|
|
182
|
+
self._rapid_refills = 0
|
|
183
|
+
self._last_compact_round: int | None = None
|
|
158
184
|
# Observability: True once the breaker tripped; the pressure path
|
|
159
185
|
# stops firing for the rest of the session.
|
|
160
186
|
self.circuit_open = False
|
|
187
|
+
#: Which breaker tripped: "consecutive_failures" | "rapid_refill".
|
|
188
|
+
self.circuit_reason: str | None = None
|
|
161
189
|
|
|
162
190
|
def _estimate(self, transcript: Sequence[LLMMessage]) -> int:
|
|
163
191
|
return estimate_tokens(transcript, model=self._model)
|
|
@@ -214,9 +242,10 @@ class CompactionHooks(NoopHooks):
|
|
|
214
242
|
)
|
|
215
243
|
pressure = self._pressure(transcript, ctx)
|
|
216
244
|
if pressure < self._threshold * self._max_tokens:
|
|
217
|
-
# A healthy round resets
|
|
218
|
-
# the transcript that caused it) is gone.
|
|
245
|
+
# A healthy round resets both breaker counts — the pathology
|
|
246
|
+
# (or the transcript that caused it) is gone.
|
|
219
247
|
self._consecutive_failures = 0
|
|
248
|
+
self._rapid_refills = 0
|
|
220
249
|
return PreStepAction(kind="proceed")
|
|
221
250
|
if self.circuit_open:
|
|
222
251
|
# Breaker tripped: further pressure rewrites only invalidate the
|
|
@@ -234,6 +263,7 @@ class CompactionHooks(NoopHooks):
|
|
|
234
263
|
if self._estimate(compacted) < threshold:
|
|
235
264
|
self.compactions += 1
|
|
236
265
|
self._consecutive_failures = 0
|
|
266
|
+
notice = self._note_successful_compaction(round_index)
|
|
237
267
|
self._last_compaction_pressure = pressure
|
|
238
268
|
self._reset_observed(ctx)
|
|
239
269
|
return PreStepAction(
|
|
@@ -245,13 +275,17 @@ class CompactionHooks(NoopHooks):
|
|
|
245
275
|
pre_tokens=pressure,
|
|
246
276
|
post_tokens=self._estimate(compacted),
|
|
247
277
|
),
|
|
278
|
+
appends=[notice] if notice is not None else None,
|
|
279
|
+
append_action="reminder" if notice is not None else None,
|
|
248
280
|
)
|
|
249
281
|
|
|
250
282
|
compacted = await self._summarize_middle(compacted)
|
|
251
283
|
post = self._estimate(compacted)
|
|
252
284
|
self.compactions += 1
|
|
285
|
+
notice: TranscriptAppend | None = None
|
|
253
286
|
if post < threshold:
|
|
254
287
|
self._consecutive_failures = 0
|
|
288
|
+
notice = self._note_successful_compaction(round_index)
|
|
255
289
|
else:
|
|
256
290
|
# Ineffective: even fold+summarize stayed over threshold (e.g. a
|
|
257
291
|
# single kept tool result bigger than the window). Three in a
|
|
@@ -259,6 +293,7 @@ class CompactionHooks(NoopHooks):
|
|
|
259
293
|
self._consecutive_failures += 1
|
|
260
294
|
if self._consecutive_failures >= self.max_consecutive_failures:
|
|
261
295
|
self.circuit_open = True
|
|
296
|
+
self.circuit_reason = "consecutive_failures"
|
|
262
297
|
self._last_compaction_pressure = pressure
|
|
263
298
|
self._reset_observed(ctx)
|
|
264
299
|
return PreStepAction(
|
|
@@ -270,20 +305,52 @@ class CompactionHooks(NoopHooks):
|
|
|
270
305
|
pre_tokens=pressure,
|
|
271
306
|
post_tokens=post,
|
|
272
307
|
),
|
|
308
|
+
appends=[notice] if notice is not None else None,
|
|
309
|
+
append_action="reminder" if notice is not None else None,
|
|
273
310
|
)
|
|
274
311
|
|
|
312
|
+
def _note_successful_compaction(self, round_index: int) -> TranscriptAppend | None:
|
|
313
|
+
"""Rapid-refill bookkeeping for a compaction that landed under
|
|
314
|
+
threshold. A refill is a successful compaction within
|
|
315
|
+
``rapid_refill_window_rounds`` of the previous one; on the
|
|
316
|
+
``max_rapid_refills``-th consecutive refill the circuit opens and
|
|
317
|
+
the round carries the thrashing reminder so both the model and the
|
|
318
|
+
host UI see the actionable notice."""
|
|
319
|
+
if (
|
|
320
|
+
self._last_compact_round is not None
|
|
321
|
+
and round_index - self._last_compact_round <= self.rapid_refill_window_rounds
|
|
322
|
+
):
|
|
323
|
+
self._rapid_refills += 1
|
|
324
|
+
else:
|
|
325
|
+
self._rapid_refills = 0
|
|
326
|
+
self._last_compact_round = round_index
|
|
327
|
+
if self._rapid_refills >= self.max_rapid_refills and not self.circuit_open:
|
|
328
|
+
self.circuit_open = True
|
|
329
|
+
self.circuit_reason = "rapid_refill"
|
|
330
|
+
reminder = CompactionThrashingReminder(
|
|
331
|
+
self._rapid_refills, self.rapid_refill_window_rounds
|
|
332
|
+
)
|
|
333
|
+
return TranscriptAppend(
|
|
334
|
+
message=LLMMessage.text_of(reminder.role, reminder.render()),
|
|
335
|
+
kind=reminder.content_kind,
|
|
336
|
+
fragment=reminder,
|
|
337
|
+
)
|
|
338
|
+
return None
|
|
339
|
+
|
|
275
340
|
async def compact_now(self, transcript: list[LLMMessage], ctx: Any) -> PreStepAction:
|
|
276
341
|
"""Manual compaction (CC ``/compact`` parity): fold old tool results,
|
|
277
342
|
then summarize the middle, regardless of pressure, hysteresis, or the
|
|
278
343
|
circuit breaker — the user asked for it. Returns a ``proceed`` action
|
|
279
344
|
carrying the rewrite; when neither stage changes anything the action
|
|
280
345
|
carries no rewrite (nothing was worth invalidating the cache for).
|
|
281
|
-
A manual pass resets
|
|
346
|
+
A manual pass resets both breaker counts: the user has taken over.
|
|
282
347
|
"""
|
|
283
348
|
pre = self._estimate(transcript)
|
|
284
349
|
compacted = self._fold_old_tool_results(transcript)
|
|
285
350
|
compacted = await self._summarize_middle(compacted)
|
|
286
351
|
self._consecutive_failures = 0
|
|
352
|
+
self._rapid_refills = 0
|
|
353
|
+
self._last_compact_round = None
|
|
287
354
|
if compacted is transcript or compacted == transcript:
|
|
288
355
|
return PreStepAction(kind="proceed", reason="compact: nothing to fold")
|
|
289
356
|
self.compactions += 1
|
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
"""Live gateway model listing — the discovery half of the model catalog.
|
|
2
|
+
|
|
3
|
+
The bundled catalog (``model_catalog.py``) is a models.dev snapshot, stale by
|
|
4
|
+
construction; the gateway's own ``GET /models`` is the live list of ids the
|
|
5
|
+
endpoint actually accepts. This module fetches and parses that listing into
|
|
6
|
+
``GatewayModel`` rows, tolerating the two listing shapes in the wild
|
|
7
|
+
(OpenAI-style ``data`` array, models.dev-style ``models`` object) and the
|
|
8
|
+
field variants each uses for window / max-output / pricing — the tolerance
|
|
9
|
+
dsh's ``readListing`` (llm-pi-ai ``discovery.ts``) implements, verified
|
|
10
|
+
against its recorded provider-listing fixtures.
|
|
11
|
+
|
|
12
|
+
Discovery, not routing: a gateway id absent from every catalog still appears
|
|
13
|
+
in the listing with empty capability fields and ``joined_from=None``; the
|
|
14
|
+
listing never gates whether a request may be sent.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import logging
|
|
20
|
+
import time
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from typing import TYPE_CHECKING, Any, Iterable
|
|
23
|
+
|
|
24
|
+
if TYPE_CHECKING:
|
|
25
|
+
from .model_info import ModelInfo
|
|
26
|
+
|
|
27
|
+
_log = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
#: Default freshness window for the in-process listing cache. The gateway
|
|
30
|
+
#: listing changes on deployment boundaries, not per request; 60s keeps a
|
|
31
|
+
#: settings UI snappy without hammering the endpoint.
|
|
32
|
+
DEFAULT_TTL_SEC = 60.0
|
|
33
|
+
|
|
34
|
+
#: Default timeout for the listing request itself.
|
|
35
|
+
DEFAULT_TIMEOUT_SEC = 10.0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class GatewayCatalogError(Exception):
|
|
39
|
+
"""The gateway listing could not be fetched and no cache survives."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, base_url: str, reason: str) -> None:
|
|
42
|
+
self.base_url = base_url
|
|
43
|
+
self.reason = reason
|
|
44
|
+
super().__init__(f"gateway catalog fetch failed for {base_url!r}: {reason}")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True, slots=True)
|
|
48
|
+
class GatewayModel:
|
|
49
|
+
"""One row of the gateway's live listing, normalized across shapes.
|
|
50
|
+
|
|
51
|
+
Capability fields are ``None``/empty when the gateway does not advertise
|
|
52
|
+
them — the catalog join (``merge_with_catalog``) fills what models.dev
|
|
53
|
+
knows, and what neither knows stays unknown.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
id: str
|
|
57
|
+
name: str
|
|
58
|
+
context_window: int | None
|
|
59
|
+
max_output_tokens: int | None
|
|
60
|
+
input_modalities: tuple[str, ...]
|
|
61
|
+
#: USD per million tokens, parsed from OpenRouter-style ``pricing``
|
|
62
|
+
#: strings; ``None`` when the gateway does not advertise pricing.
|
|
63
|
+
prompt_price_per_mtok: float | None
|
|
64
|
+
completion_price_per_mtok: float | None
|
|
65
|
+
#: OpenRouter-style ``supported_parameters`` (``"reasoning"`` present
|
|
66
|
+
#: means the endpoint accepts a reasoning knob, levels unknown).
|
|
67
|
+
supported_parameters: tuple[str, ...]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True, slots=True)
|
|
71
|
+
class GatewayListing:
|
|
72
|
+
"""A fetched listing plus its freshness provenance."""
|
|
73
|
+
|
|
74
|
+
entries: tuple[GatewayModel, ...]
|
|
75
|
+
#: Epoch seconds of the successful fetch (a stale listing keeps the
|
|
76
|
+
#: timestamp of the fetch that produced it, not of the failed refresh).
|
|
77
|
+
fetched_at: float
|
|
78
|
+
#: True when the last refresh failed and this is the previous listing.
|
|
79
|
+
stale: bool
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True, slots=True)
|
|
83
|
+
class GatewayCatalogRow:
|
|
84
|
+
"""A gateway id joined with catalog capabilities, ready for the wire.
|
|
85
|
+
|
|
86
|
+
``joined_from`` names the catalog key that supplied capability fields
|
|
87
|
+
(``None`` when no catalog tier matched — the "capabilities unknown"
|
|
88
|
+
case the UI marks). ``info`` is the merged capability descriptor:
|
|
89
|
+
gateway-advertised fields win where the gateway provides them, catalog
|
|
90
|
+
fields (reasoning levels above all) fill the rest.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
id: str
|
|
94
|
+
name: str
|
|
95
|
+
info: ModelInfo
|
|
96
|
+
joined_from: str | None
|
|
97
|
+
prompt_price_per_mtok: float | None
|
|
98
|
+
completion_price_per_mtok: float | None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ---------------------------------------------------------------------------
|
|
102
|
+
# Parsing (pure)
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _first_int(*values: Any) -> int | None:
|
|
107
|
+
for value in values:
|
|
108
|
+
if isinstance(value, bool):
|
|
109
|
+
continue
|
|
110
|
+
if isinstance(value, (int, float)) and value > 0:
|
|
111
|
+
return int(value)
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _first_str_list(value: Any) -> tuple[str, ...]:
|
|
116
|
+
if isinstance(value, (list, tuple)):
|
|
117
|
+
return tuple(str(v) for v in value if isinstance(v, str))
|
|
118
|
+
return ()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _price_per_mtok(pricing: Any, key: str) -> float | None:
|
|
122
|
+
"""OpenRouter ``pricing`` values are USD-per-token strings."""
|
|
123
|
+
if not isinstance(pricing, dict):
|
|
124
|
+
return None
|
|
125
|
+
raw = pricing.get(key)
|
|
126
|
+
if not isinstance(raw, (str, int, float)) or isinstance(raw, bool):
|
|
127
|
+
return None
|
|
128
|
+
try:
|
|
129
|
+
return float(raw) * 1_000_000
|
|
130
|
+
except (TypeError, ValueError):
|
|
131
|
+
return None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _parse_entry(entry: Any, *, key_hint: str | None) -> GatewayModel | None:
|
|
135
|
+
if not isinstance(entry, dict):
|
|
136
|
+
return None
|
|
137
|
+
model_id = entry.get("id") or key_hint
|
|
138
|
+
if not isinstance(model_id, str) or not model_id:
|
|
139
|
+
return None
|
|
140
|
+
name = entry.get("name") or entry.get("display_name") or entry.get("displayName")
|
|
141
|
+
limit = entry.get("limit") if isinstance(entry.get("limit"), dict) else {}
|
|
142
|
+
top_provider = (
|
|
143
|
+
entry.get("top_provider") if isinstance(entry.get("top_provider"), dict) else {}
|
|
144
|
+
)
|
|
145
|
+
architecture = (
|
|
146
|
+
entry.get("architecture") if isinstance(entry.get("architecture"), dict) else {}
|
|
147
|
+
)
|
|
148
|
+
return GatewayModel(
|
|
149
|
+
id=model_id,
|
|
150
|
+
name=str(name) if isinstance(name, str) and name else model_id,
|
|
151
|
+
context_window=_first_int(
|
|
152
|
+
entry.get("context_length"),
|
|
153
|
+
entry.get("context_window"),
|
|
154
|
+
entry.get("contextWindow"),
|
|
155
|
+
entry.get("max_input_tokens"),
|
|
156
|
+
limit.get("context"),
|
|
157
|
+
),
|
|
158
|
+
max_output_tokens=_first_int(
|
|
159
|
+
top_provider.get("max_completion_tokens"),
|
|
160
|
+
entry.get("max_completion_tokens"),
|
|
161
|
+
entry.get("max_output_tokens"),
|
|
162
|
+
entry.get("maxOutputTokens"),
|
|
163
|
+
entry.get("maxTokens"),
|
|
164
|
+
entry.get("max_tokens"),
|
|
165
|
+
limit.get("output"),
|
|
166
|
+
),
|
|
167
|
+
input_modalities=_first_str_list(architecture.get("input_modalities")),
|
|
168
|
+
prompt_price_per_mtok=_price_per_mtok(entry.get("pricing"), "prompt"),
|
|
169
|
+
completion_price_per_mtok=_price_per_mtok(entry.get("pricing"), "completion"),
|
|
170
|
+
supported_parameters=_first_str_list(entry.get("supported_parameters")),
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def parse_models_listing(payload: Any) -> list[GatewayModel]:
|
|
175
|
+
"""Parse a ``GET /models`` body into normalized rows.
|
|
176
|
+
|
|
177
|
+
Two shapes are in the wild: OpenAI-style ``{"data": [...]}`` (the common
|
|
178
|
+
case, OpenRouter's enriched variant included) and models.dev-style
|
|
179
|
+
``{"models": {...}}`` where the property key is the id. Anything else is
|
|
180
|
+
a parse error the caller surfaces as ``GatewayCatalogError``.
|
|
181
|
+
"""
|
|
182
|
+
if not isinstance(payload, dict):
|
|
183
|
+
raise ValueError("listing payload is not a JSON object")
|
|
184
|
+
data = payload.get("data")
|
|
185
|
+
if isinstance(data, list):
|
|
186
|
+
rows = [_parse_entry(entry, key_hint=None) for entry in data]
|
|
187
|
+
return [row for row in rows if row is not None]
|
|
188
|
+
models = payload.get("models")
|
|
189
|
+
if isinstance(models, dict):
|
|
190
|
+
rows = [_parse_entry(entry, key_hint=key) for key, entry in models.items()]
|
|
191
|
+
return [row for row in rows if row is not None]
|
|
192
|
+
if isinstance(models, list):
|
|
193
|
+
rows = [_parse_entry(entry, key_hint=None) for entry in models]
|
|
194
|
+
return [row for row in rows if row is not None]
|
|
195
|
+
raise ValueError("listing has neither a 'data' array nor a 'models' map")
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
# ---------------------------------------------------------------------------
|
|
199
|
+
# Catalog join (pure)
|
|
200
|
+
# ---------------------------------------------------------------------------
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def merge_with_catalog(models: Iterable[GatewayModel]) -> list[GatewayCatalogRow]:
|
|
204
|
+
"""Join gateway rows with catalog capabilities (cross-provider leaf).
|
|
205
|
+
|
|
206
|
+
The gateway's own fields (window, modalities) win where advertised —
|
|
207
|
+
they describe this deployment. Capability fields come from
|
|
208
|
+
``resolve_model_info`` itself — the very resolution the request path's
|
|
209
|
+
``clamp_reasoning_effort`` uses, legacy-union rule included — so
|
|
210
|
+
``models.list`` and the request path never disagree about a model's
|
|
211
|
+
knob (a picker that hides a level the backend would accept is the same
|
|
212
|
+
class of drift as silently dropping one). ``joined_from`` keeps the
|
|
213
|
+
leaf-join provenance: ``None`` when no catalog tier matched, even if
|
|
214
|
+
the hand-owned legacy table knows the model.
|
|
215
|
+
"""
|
|
216
|
+
from .model_info import (
|
|
217
|
+
TOOL_FORMAT_OPENAI,
|
|
218
|
+
ModelInfo,
|
|
219
|
+
resolve_model_info,
|
|
220
|
+
)
|
|
221
|
+
from .model_resolve import resolve_leaf_cross_provider
|
|
222
|
+
|
|
223
|
+
rows: list[GatewayCatalogRow] = []
|
|
224
|
+
for model in models:
|
|
225
|
+
hit = resolve_leaf_cross_provider(model.id)
|
|
226
|
+
base = resolve_model_info(model.id)
|
|
227
|
+
context_window = (
|
|
228
|
+
model.context_window
|
|
229
|
+
if model.context_window is not None
|
|
230
|
+
else base.context_window
|
|
231
|
+
)
|
|
232
|
+
modalities = (
|
|
233
|
+
frozenset(model.input_modalities)
|
|
234
|
+
if model.input_modalities
|
|
235
|
+
else base.modalities
|
|
236
|
+
)
|
|
237
|
+
rows.append(
|
|
238
|
+
GatewayCatalogRow(
|
|
239
|
+
id=model.id,
|
|
240
|
+
name=model.name,
|
|
241
|
+
info=ModelInfo(
|
|
242
|
+
pattern=model.id.lower(),
|
|
243
|
+
context_window=context_window,
|
|
244
|
+
modalities=modalities,
|
|
245
|
+
tool_format=TOOL_FORMAT_OPENAI,
|
|
246
|
+
reasoning_levels=base.reasoning_levels,
|
|
247
|
+
),
|
|
248
|
+
joined_from=hit.key if hit is not None else None,
|
|
249
|
+
prompt_price_per_mtok=model.prompt_price_per_mtok,
|
|
250
|
+
completion_price_per_mtok=model.completion_price_per_mtok,
|
|
251
|
+
)
|
|
252
|
+
)
|
|
253
|
+
return rows
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
# ---------------------------------------------------------------------------
|
|
257
|
+
# Fetch (network, TTL-cached, stale-on-error)
|
|
258
|
+
# ---------------------------------------------------------------------------
|
|
259
|
+
|
|
260
|
+
#: base_url (normalized) -> (time.monotonic() at fetch, listing). The wall
|
|
261
|
+
#: clock ``listing.fetched_at`` is for display; the monotonic clock drives
|
|
262
|
+
#: TTL so a clock jump cannot expire or resurrect a listing.
|
|
263
|
+
_cache: dict[str, tuple[float, GatewayListing]] = {}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _cache_key(base_url: str) -> str:
|
|
267
|
+
return base_url.rstrip("/").lower()
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def clear_gateway_cache() -> None:
|
|
271
|
+
"""Drop every cached listing (tests, forced refresh)."""
|
|
272
|
+
_cache.clear()
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
async def fetch_gateway_models(
|
|
276
|
+
base_url: str,
|
|
277
|
+
api_key: str | None = None,
|
|
278
|
+
*,
|
|
279
|
+
timeout_sec: float = DEFAULT_TIMEOUT_SEC,
|
|
280
|
+
ttl_sec: float = DEFAULT_TTL_SEC,
|
|
281
|
+
) -> GatewayListing:
|
|
282
|
+
"""Fetch the gateway's live model listing, with a process-local cache.
|
|
283
|
+
|
|
284
|
+
A fresh-enough cached listing is served without network. On refresh
|
|
285
|
+
failure the previous listing is returned marked ``stale`` — the catalog
|
|
286
|
+
degrades instead of disappearing when the gateway flaps. With no cache
|
|
287
|
+
to fall back on, ``GatewayCatalogError`` is raised and the caller
|
|
288
|
+
decides (the sidecar answers ``catalog_status: "offline"``).
|
|
289
|
+
"""
|
|
290
|
+
import httpx # local import — keeps the runtime importable without httpx
|
|
291
|
+
|
|
292
|
+
from .llm.system_proxy import client_env_kwargs
|
|
293
|
+
|
|
294
|
+
key = _cache_key(base_url)
|
|
295
|
+
cached = _cache.get(key)
|
|
296
|
+
if cached is not None and (time.monotonic() - cached[0]) < ttl_sec:
|
|
297
|
+
return cached[1]
|
|
298
|
+
|
|
299
|
+
url = f"{base_url.rstrip('/')}/models"
|
|
300
|
+
headers = {"Authorization": f"Bearer {api_key}"} if api_key else {}
|
|
301
|
+
try:
|
|
302
|
+
async with httpx.AsyncClient(
|
|
303
|
+
timeout=httpx.Timeout(timeout_sec),
|
|
304
|
+
**client_env_kwargs(base_url),
|
|
305
|
+
) as client:
|
|
306
|
+
response = await client.get(url, headers=headers)
|
|
307
|
+
response.raise_for_status()
|
|
308
|
+
entries = tuple(parse_models_listing(response.json()))
|
|
309
|
+
except Exception as exc:
|
|
310
|
+
if cached is not None:
|
|
311
|
+
_log.info("gateway listing refresh failed (%s); serving stale", exc)
|
|
312
|
+
return GatewayListing(
|
|
313
|
+
entries=cached[1].entries, fetched_at=cached[1].fetched_at, stale=True
|
|
314
|
+
)
|
|
315
|
+
raise GatewayCatalogError(base_url, str(exc)) from exc
|
|
316
|
+
|
|
317
|
+
listing = GatewayListing(entries=entries, fetched_at=time.time(), stale=False)
|
|
318
|
+
_cache[key] = (time.monotonic(), listing)
|
|
319
|
+
return listing
|
|
@@ -209,6 +209,11 @@ class PressureCompaction:
|
|
|
209
209
|
# trades cache hits for a bounded transcript. R16 found compaction does not
|
|
210
210
|
# move the eval score, so this stays opt-in capability, not a default.
|
|
211
211
|
micro_compact_interval_rounds: int = 0
|
|
212
|
+
# Rapid-refill breaker (CC parity): a successful compaction whose freed
|
|
213
|
+
# space refills within this many rounds of the previous one counts as a
|
|
214
|
+
# refill; max_rapid_refills consecutive refills open the circuit.
|
|
215
|
+
rapid_refill_window_rounds: int = 3
|
|
216
|
+
max_rapid_refills: int = 3
|
|
212
217
|
name: str = "pressure_compaction"
|
|
213
218
|
assumes: str = (
|
|
214
219
|
"long trajectories exceed the window; older detail is expendable "
|
|
@@ -231,6 +236,8 @@ class PressureCompaction:
|
|
|
231
236
|
summarizer=provider,
|
|
232
237
|
model=self.model,
|
|
233
238
|
micro_compact_interval_rounds=self.micro_compact_interval_rounds,
|
|
239
|
+
rapid_refill_window_rounds=self.rapid_refill_window_rounds,
|
|
240
|
+
max_rapid_refills=self.max_rapid_refills,
|
|
234
241
|
**extra,
|
|
235
242
|
),
|
|
236
243
|
forward=("compact_now",),
|
|
@@ -158,7 +158,9 @@ from .errors import (
|
|
|
158
158
|
classify_http_status,
|
|
159
159
|
is_retryable,
|
|
160
160
|
)
|
|
161
|
+
from .google_genai import GoogleGenAIProvider
|
|
161
162
|
from .openai_compat import OpenAICompatProvider
|
|
163
|
+
from .openai_responses import OpenAIResponsesProvider
|
|
162
164
|
from .presets import (
|
|
163
165
|
PROVIDER_PRESETS,
|
|
164
166
|
PresetEntry,
|
|
@@ -174,6 +176,7 @@ __all__ = [
|
|
|
174
176
|
"RETRYABLE_KINDS",
|
|
175
177
|
"AnthropicProvider",
|
|
176
178
|
"ContentPart",
|
|
179
|
+
"GoogleGenAIProvider",
|
|
177
180
|
"ImagePart",
|
|
178
181
|
"LLMError",
|
|
179
182
|
"LLMErrorKind",
|
|
@@ -188,6 +191,7 @@ __all__ = [
|
|
|
188
191
|
"describe_compat_flags",
|
|
189
192
|
"describe_provider_presets",
|
|
190
193
|
"OpenAICompatProvider",
|
|
194
|
+
"OpenAIResponsesProvider",
|
|
191
195
|
"TextPart",
|
|
192
196
|
"classify_error",
|
|
193
197
|
"classify_http_status",
|