steerable-agent-runtime 0.6.4__tar.gz → 0.6.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/PKG-INFO +1 -1
  2. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/pyproject.toml +1 -1
  3. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/__init__.py +8 -0
  4. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/ask_user.py +17 -7
  5. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/compaction.py +71 -4
  6. steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/gateway_catalog.py +319 -0
  7. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/harness.py +7 -0
  8. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/__init__.py +4 -0
  9. steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/llm/google_genai.py +444 -0
  10. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/openai_compat.py +38 -8
  11. steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/llm/openai_responses.py +544 -0
  12. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/model_catalog.py +2046 -1923
  13. steerable_agent_runtime-0.6.6/src/steerable_agent_runtime/model_info.py +460 -0
  14. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/model_resolve.py +45 -1
  15. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pool.py +15 -4
  16. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/reminders.py +33 -0
  17. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/subagent.py +17 -1
  18. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/todo.py +67 -0
  19. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
  20. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/SOURCES.txt +8 -0
  21. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_ask_user.py +29 -6
  22. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_compaction.py +289 -0
  23. steerable_agent_runtime-0.6.6/tests/test_gateway_catalog.py +194 -0
  24. steerable_agent_runtime-0.6.6/tests/test_llm_gemini_wire.py +256 -0
  25. steerable_agent_runtime-0.6.6/tests/test_llm_responses_wire.py +396 -0
  26. steerable_agent_runtime-0.6.6/tests/test_llm_stream_e2e.py +233 -0
  27. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_llm_wire_helpers.py +18 -5
  28. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_catalog.py +2 -2
  29. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_equivalence.py +3 -1
  30. steerable_agent_runtime-0.6.6/tests/test_model_info.py +376 -0
  31. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_subagent.py +126 -0
  32. steerable_agent_runtime-0.6.6/tests/test_todo_gate.py +94 -0
  33. steerable_agent_runtime-0.6.4/src/steerable_agent_runtime/model_info.py +0 -247
  34. steerable_agent_runtime-0.6.4/tests/test_model_info.py +0 -185
  35. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/README.md +0 -0
  36. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/setup.cfg +0 -0
  37. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/antihallucination.py +0 -0
  38. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/approval.py +0 -0
  39. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/approval_policy.py +0 -0
  40. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/branch.py +0 -0
  41. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/cache_control.py +0 -0
  42. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/calibration.py +0 -0
  43. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/config.py +0 -0
  44. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/default.harness.json +0 -0
  45. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/default.harness.yaml +0 -0
  46. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/errors.py +0 -0
  47. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/handoff.py +0 -0
  48. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/harness_spec.py +0 -0
  49. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/history.py +0 -0
  50. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/hooks.py +0 -0
  51. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
  52. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/compat.py +0 -0
  53. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/errors.py +0 -0
  54. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/parts.py +0 -0
  55. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/presets.py +0 -0
  56. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
  57. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/loop.py +0 -0
  58. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/maintenance.py +0 -0
  59. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/mcp.py +0 -0
  60. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/mcp_server.py +0 -0
  61. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/observation_aging.py +0 -0
  62. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/orchestration.py +0 -0
  63. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/otel.py +0 -0
  64. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/plugins.py +0 -0
  65. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pricing.py +0 -0
  66. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/pseudo.py +0 -0
  67. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/recording.py +0 -0
  68. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/replay.py +0 -0
  69. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/resume.py +0 -0
  70. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/retry.py +0 -0
  71. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/sandboxed.py +0 -0
  72. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/skills.py +0 -0
  73. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/spill.py +0 -0
  74. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/__init__.py +0 -0
  75. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
  76. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
  77. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
  78. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
  79. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tokens.py +0 -0
  80. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tool_schema.py +0 -0
  81. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tool_search.py +0 -0
  82. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tools.py +0 -0
  83. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/tracing.py +0 -0
  84. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/__init__.py +0 -0
  85. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
  86. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
  87. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime/world_state.py +0 -0
  88. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
  89. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
  90. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
  91. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_antihallucination.py +0 -0
  92. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_approval.py +0 -0
  93. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_approval_policy.py +0 -0
  94. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_branch.py +0 -0
  95. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_cache_control.py +0 -0
  96. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_cache_instrumentation.py +0 -0
  97. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_calibration.py +0 -0
  98. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_config.py +0 -0
  99. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_content_parts.py +0 -0
  100. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_error_taxonomy.py +0 -0
  101. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_fragment_bounds.py +0 -0
  102. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_golden.py +0 -0
  103. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_handoff.py +0 -0
  104. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_harness.py +0 -0
  105. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_harness_spec.py +0 -0
  106. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_history.py +0 -0
  107. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_history_persistence.py +0 -0
  108. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_hooks.py +0 -0
  109. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_in_memory_storage.py +0 -0
  110. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_long_session.py +0 -0
  111. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop.py +0 -0
  112. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_cancellation.py +0 -0
  113. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_replay.py +0 -0
  114. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_loop_sandbox_event.py +0 -0
  115. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_maintenance.py +0 -0
  116. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_mcp.py +0 -0
  117. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_mcp_server.py +0 -0
  118. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_model_resolve.py +0 -0
  119. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_observation_aging.py +0 -0
  120. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_orchestration.py +0 -0
  121. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_otel.py +0 -0
  122. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_parallel_tools.py +0 -0
  123. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_plugins.py +0 -0
  124. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_provider_compat.py +0 -0
  125. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_provider_presets.py +0 -0
  126. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_pseudo.py +0 -0
  127. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_recording.py +0 -0
  128. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_reminders.py +0 -0
  129. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_replay_crosslang.py +0 -0
  130. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_resume.py +0 -0
  131. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_retry_hooks.py +0 -0
  132. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_safety_gate.py +0 -0
  133. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_sandboxed.py +0 -0
  134. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_skills.py +0 -0
  135. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_soft_timeout.py +0 -0
  136. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_spill.py +0 -0
  137. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_sqlite_storage.py +0 -0
  138. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_steer.py +0 -0
  139. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_storage_contract.py +0 -0
  140. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_stream_strip.py +0 -0
  141. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_system_proxy.py +0 -0
  142. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_todo.py +0 -0
  143. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tokens.py +0 -0
  144. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_exposure.py +0 -0
  145. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_hygiene.py +0 -0
  146. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_router.py +0 -0
  147. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_schema.py +0 -0
  148. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_tool_timeout.py +0 -0
  149. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_trace_recorder.py +0 -0
  150. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_transport_jsonrpc.py +0 -0
  151. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_transport_sse.py +0 -0
  152. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_usage_attribution.py +0 -0
  153. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_world_state.py +0 -0
  154. {steerable_agent_runtime-0.6.4 → steerable_agent_runtime-0.6.6}/tests/test_write_lease.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steerable-agent-runtime
3
- Version: 0.6.4
3
+ Version: 0.6.6
4
4
  Summary: Steerable agent runtime: LLM, tool, storage, and transport adapters.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "steerable-agent-runtime"
3
- version = "0.6.4"
3
+ version = "0.6.6"
4
4
  description = "Steerable agent runtime: LLM, tool, storage, and transport adapters."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -38,6 +38,7 @@ from .ask_user import (
38
38
  from .todo import (
39
39
  TODO_SCHEMA,
40
40
  TODO_TOOL_NAME,
41
+ TodoCompletionGate,
41
42
  TodoStore,
42
43
  make_todo_write_tool,
43
44
  todo_write_tool_descriptor,
@@ -156,7 +157,10 @@ from .model_info import (
156
157
  MODEL_INFOS,
157
158
  REASONING_EFFORT_ORDER,
158
159
  ModelInfo,
160
+ ReasoningEffortUnsupported,
159
161
  clamp_reasoning_effort,
162
+ clear_gateway_models,
163
+ register_gateway_models,
160
164
  register_model_info,
161
165
  resolve_model_info,
162
166
  )
@@ -266,6 +270,7 @@ __all__ = [
266
270
  "MODEL_TOKEN_FACTORS",
267
271
  "REASONING_EFFORT_ORDER",
268
272
  "RECORD_FORMAT_VERSION",
273
+ "ReasoningEffortUnsupported",
269
274
  "TOOL_SEARCH_NAME",
270
275
  "ASK_USER_SCHEMA",
271
276
  "ASK_USER_TOOL_NAME",
@@ -391,6 +396,7 @@ __all__ = [
391
396
  "ToolExecutor",
392
397
  "ToolExposure",
393
398
  "ToolRouter",
399
+ "TodoCompletionGate",
394
400
  "TodoStore",
395
401
  "TraceRecorder",
396
402
  "TranscriptAppend",
@@ -408,6 +414,7 @@ __all__ = [
408
414
  "branch_label",
409
415
  "build_step_decision_event",
410
416
  "clamp_reasoning_effort",
417
+ "clear_gateway_models",
411
418
  "close_dangling_tool_calls",
412
419
  "detect_claimed_execution",
413
420
  "detect_deferred_execution",
@@ -449,6 +456,7 @@ __all__ = [
449
456
  "qualify_mcp_name",
450
457
  "reduce_execution_state",
451
458
  "register_mcp_catalog",
459
+ "register_gateway_models",
452
460
  "register_model_factor",
453
461
  "register_model_info",
454
462
  "register_model_price",
@@ -71,9 +71,16 @@ ASK_USER_SCHEMA: dict[str, Any] = {
71
71
  ),
72
72
  },
73
73
  "placeholder": {"type": "string"},
74
- "multiSelect": {"type": "boolean"},
74
+ "multiSelect": {
75
+ "type": "boolean",
76
+ "description": (
77
+ "Whether the user may pick several options. "
78
+ "Required (CC parity): commit to single vs multi "
79
+ "explicitly — pass false for a single-select."
80
+ ),
81
+ },
75
82
  },
76
- "required": ["id", "text"],
83
+ "required": ["id", "text", "multiSelect"],
77
84
  },
78
85
  },
79
86
  },
@@ -170,12 +177,15 @@ def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]
170
177
  f"ask_user: questions[{index}].header is {len(header)} chars; "
171
178
  f"the chip label must be <= {_MAX_HEADER_LEN}."
172
179
  )
173
- # multiSelect: CC requires the model to commit to single vs multi.
174
- # Default to False when absent so existing single-select calls keep
175
- # working, but stamp it so hosts always read an explicit boolean.
180
+ # multiSelect: CC requires the model to commit to single vs multi —
181
+ # the field is required, so an omission is a model error to fix, not
182
+ # a default to assume.
176
183
  if "multiSelect" not in q or q["multiSelect"] is None:
177
- q["multiSelect"] = False
178
- elif not isinstance(q["multiSelect"], bool):
184
+ raise ToolDispatchError(
185
+ f"ask_user: questions[{index}].multiSelect is required "
186
+ "(CC parity) — pass false for a single-select question."
187
+ )
188
+ if not isinstance(q["multiSelect"], bool):
179
189
  raise ToolDispatchError(
180
190
  f"ask_user: questions[{index}].multiSelect must be a boolean, "
181
191
  f"got {type(q['multiSelect']).__name__}."
@@ -57,6 +57,14 @@ Two safety rails share the machinery:
57
57
  the turn still fails loud (bounded retries, then a named error) instead
58
58
  of spinning. A round that lands under threshold — or any successful
59
59
  compaction — resets the count.
60
+ - A **rapid-refill breaker** (CC parity): a compaction that lands under
61
+ threshold but refills within ``rapid_refill_window_rounds`` rounds of the
62
+ previous one counts as a refill; ``max_rapid_refills`` consecutive
63
+ refills open the same circuit — the transcript is churning faster than
64
+ compaction can help, so further rewrites only kill the prompt cache.
65
+ The tripping round appends a ``CompactionThrashingReminder`` (model- and
66
+ UI-visible) with an actionable converge notice. ``circuit_reason``
67
+ records which breaker tripped.
60
68
  """
61
69
 
62
70
  from __future__ import annotations
@@ -64,9 +72,10 @@ from __future__ import annotations
64
72
  from collections.abc import Sequence
65
73
  from typing import Any
66
74
 
67
- from .hooks import NoopHooks, PreStepAction, RetryAction, RewriteRequest
75
+ from .hooks import NoopHooks, PreStepAction, RetryAction, RewriteRequest, TranscriptAppend
68
76
  from .llm import LLMMessage, LLMProvider
69
77
  from .llm.errors import classify_error
78
+ from .reminders import CompactionThrashingReminder
70
79
  from .tokens import estimate_tokens
71
80
 
72
81
  _SUMMARY_MARKER = "[context compacted: earlier conversation summarized]"
@@ -108,6 +117,8 @@ class CompactionHooks(NoopHooks):
108
117
  recompact_margin_ratio: float = 0.1,
109
118
  fold_excerpt_chars: int = _FOLD_EXCERPT_CHARS,
110
119
  micro_compact_interval_rounds: int = 0,
120
+ rapid_refill_window_rounds: int = 3,
121
+ max_rapid_refills: int = 3,
111
122
  ) -> None:
112
123
  if not 0 < threshold_ratio <= 1:
113
124
  raise ValueError("threshold_ratio must be in (0, 1]")
@@ -115,6 +126,10 @@ class CompactionHooks(NoopHooks):
115
126
  raise ValueError("recompact_margin_ratio must be >= 0")
116
127
  if not 0 <= micro_compact_interval_rounds:
117
128
  raise ValueError("micro_compact_interval_rounds must be >= 0")
129
+ if not 1 <= rapid_refill_window_rounds:
130
+ raise ValueError("rapid_refill_window_rounds must be >= 1")
131
+ if not 1 <= max_rapid_refills:
132
+ raise ValueError("max_rapid_refills must be >= 1")
118
133
  self._max_tokens = max_context_tokens
119
134
  self._threshold = threshold_ratio
120
135
  self._keep_last = keep_last_messages
@@ -155,9 +170,22 @@ class CompactionHooks(NoopHooks):
155
170
  #: parity). The overflow path is bounded separately and unaffected.
156
171
  self.max_consecutive_failures = 3
157
172
  self._consecutive_failures = 0
173
+ #: Rapid-refill breaker (CC parity): a compaction that *worked* but
174
+ #: whose freed space refills within ``rapid_refill_window_rounds``
175
+ #: rounds counts as a refill; ``max_rapid_refills`` consecutive
176
+ #: refills open the circuit — the transcript is churning faster than
177
+ #: compaction can help, and each rewrite only kills the prompt-cache
178
+ #: prefix. Distinct from the failure breaker above, which catches
179
+ #: compactions that never get under threshold at all.
180
+ self.rapid_refill_window_rounds = rapid_refill_window_rounds
181
+ self.max_rapid_refills = max_rapid_refills
182
+ self._rapid_refills = 0
183
+ self._last_compact_round: int | None = None
158
184
  # Observability: True once the breaker tripped; the pressure path
159
185
  # stops firing for the rest of the session.
160
186
  self.circuit_open = False
187
+ #: Which breaker tripped: "consecutive_failures" | "rapid_refill".
188
+ self.circuit_reason: str | None = None
161
189
 
162
190
  def _estimate(self, transcript: Sequence[LLMMessage]) -> int:
163
191
  return estimate_tokens(transcript, model=self._model)
@@ -214,9 +242,10 @@ class CompactionHooks(NoopHooks):
214
242
  )
215
243
  pressure = self._pressure(transcript, ctx)
216
244
  if pressure < self._threshold * self._max_tokens:
217
- # A healthy round resets the breaker count — the pathology (or
218
- # the transcript that caused it) is gone.
245
+ # A healthy round resets both breaker counts — the pathology
246
+ # (or the transcript that caused it) is gone.
219
247
  self._consecutive_failures = 0
248
+ self._rapid_refills = 0
220
249
  return PreStepAction(kind="proceed")
221
250
  if self.circuit_open:
222
251
  # Breaker tripped: further pressure rewrites only invalidate the
@@ -234,6 +263,7 @@ class CompactionHooks(NoopHooks):
234
263
  if self._estimate(compacted) < threshold:
235
264
  self.compactions += 1
236
265
  self._consecutive_failures = 0
266
+ notice = self._note_successful_compaction(round_index)
237
267
  self._last_compaction_pressure = pressure
238
268
  self._reset_observed(ctx)
239
269
  return PreStepAction(
@@ -245,13 +275,17 @@ class CompactionHooks(NoopHooks):
245
275
  pre_tokens=pressure,
246
276
  post_tokens=self._estimate(compacted),
247
277
  ),
278
+ appends=[notice] if notice is not None else None,
279
+ append_action="reminder" if notice is not None else None,
248
280
  )
249
281
 
250
282
  compacted = await self._summarize_middle(compacted)
251
283
  post = self._estimate(compacted)
252
284
  self.compactions += 1
285
+ notice: TranscriptAppend | None = None
253
286
  if post < threshold:
254
287
  self._consecutive_failures = 0
288
+ notice = self._note_successful_compaction(round_index)
255
289
  else:
256
290
  # Ineffective: even fold+summarize stayed over threshold (e.g. a
257
291
  # single kept tool result bigger than the window). Three in a
@@ -259,6 +293,7 @@ class CompactionHooks(NoopHooks):
259
293
  self._consecutive_failures += 1
260
294
  if self._consecutive_failures >= self.max_consecutive_failures:
261
295
  self.circuit_open = True
296
+ self.circuit_reason = "consecutive_failures"
262
297
  self._last_compaction_pressure = pressure
263
298
  self._reset_observed(ctx)
264
299
  return PreStepAction(
@@ -270,20 +305,52 @@ class CompactionHooks(NoopHooks):
270
305
  pre_tokens=pressure,
271
306
  post_tokens=post,
272
307
  ),
308
+ appends=[notice] if notice is not None else None,
309
+ append_action="reminder" if notice is not None else None,
273
310
  )
274
311
 
312
+ def _note_successful_compaction(self, round_index: int) -> TranscriptAppend | None:
313
+ """Rapid-refill bookkeeping for a compaction that landed under
314
+ threshold. A refill is a successful compaction within
315
+ ``rapid_refill_window_rounds`` of the previous one; on the
316
+ ``max_rapid_refills``-th consecutive refill the circuit opens and
317
+ the round carries the thrashing reminder so both the model and the
318
+ host UI see the actionable notice."""
319
+ if (
320
+ self._last_compact_round is not None
321
+ and round_index - self._last_compact_round <= self.rapid_refill_window_rounds
322
+ ):
323
+ self._rapid_refills += 1
324
+ else:
325
+ self._rapid_refills = 0
326
+ self._last_compact_round = round_index
327
+ if self._rapid_refills >= self.max_rapid_refills and not self.circuit_open:
328
+ self.circuit_open = True
329
+ self.circuit_reason = "rapid_refill"
330
+ reminder = CompactionThrashingReminder(
331
+ self._rapid_refills, self.rapid_refill_window_rounds
332
+ )
333
+ return TranscriptAppend(
334
+ message=LLMMessage.text_of(reminder.role, reminder.render()),
335
+ kind=reminder.content_kind,
336
+ fragment=reminder,
337
+ )
338
+ return None
339
+
275
340
  async def compact_now(self, transcript: list[LLMMessage], ctx: Any) -> PreStepAction:
276
341
  """Manual compaction (CC ``/compact`` parity): fold old tool results,
277
342
  then summarize the middle, regardless of pressure, hysteresis, or the
278
343
  circuit breaker — the user asked for it. Returns a ``proceed`` action
279
344
  carrying the rewrite; when neither stage changes anything the action
280
345
  carries no rewrite (nothing was worth invalidating the cache for).
281
- A manual pass resets the breaker count: the user has taken over.
346
+ A manual pass resets both breaker counts: the user has taken over.
282
347
  """
283
348
  pre = self._estimate(transcript)
284
349
  compacted = self._fold_old_tool_results(transcript)
285
350
  compacted = await self._summarize_middle(compacted)
286
351
  self._consecutive_failures = 0
352
+ self._rapid_refills = 0
353
+ self._last_compact_round = None
287
354
  if compacted is transcript or compacted == transcript:
288
355
  return PreStepAction(kind="proceed", reason="compact: nothing to fold")
289
356
  self.compactions += 1
@@ -0,0 +1,319 @@
1
+ """Live gateway model listing — the discovery half of the model catalog.
2
+
3
+ The bundled catalog (``model_catalog.py``) is a models.dev snapshot, stale by
4
+ construction; the gateway's own ``GET /models`` is the live list of ids the
5
+ endpoint actually accepts. This module fetches and parses that listing into
6
+ ``GatewayModel`` rows, tolerating the two listing shapes in the wild
7
+ (OpenAI-style ``data`` array, models.dev-style ``models`` object) and the
8
+ field variants each uses for window / max-output / pricing — the tolerance
9
+ dsh's ``readListing`` (llm-pi-ai ``discovery.ts``) implements, verified
10
+ against its recorded provider-listing fixtures.
11
+
12
+ Discovery, not routing: a gateway id absent from every catalog still appears
13
+ in the listing with empty capability fields and ``joined_from=None``; the
14
+ listing never gates whether a request may be sent.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import logging
20
+ import time
21
+ from dataclasses import dataclass
22
+ from typing import TYPE_CHECKING, Any, Iterable
23
+
24
+ if TYPE_CHECKING:
25
+ from .model_info import ModelInfo
26
+
27
+ _log = logging.getLogger(__name__)
28
+
29
+ #: Default freshness window for the in-process listing cache. The gateway
30
+ #: listing changes on deployment boundaries, not per request; 60s keeps a
31
+ #: settings UI snappy without hammering the endpoint.
32
+ DEFAULT_TTL_SEC = 60.0
33
+
34
+ #: Default timeout for the listing request itself.
35
+ DEFAULT_TIMEOUT_SEC = 10.0
36
+
37
+
38
+ class GatewayCatalogError(Exception):
39
+ """The gateway listing could not be fetched and no cache survives."""
40
+
41
+ def __init__(self, base_url: str, reason: str) -> None:
42
+ self.base_url = base_url
43
+ self.reason = reason
44
+ super().__init__(f"gateway catalog fetch failed for {base_url!r}: {reason}")
45
+
46
+
47
+ @dataclass(frozen=True, slots=True)
48
+ class GatewayModel:
49
+ """One row of the gateway's live listing, normalized across shapes.
50
+
51
+ Capability fields are ``None``/empty when the gateway does not advertise
52
+ them — the catalog join (``merge_with_catalog``) fills what models.dev
53
+ knows, and what neither knows stays unknown.
54
+ """
55
+
56
+ id: str
57
+ name: str
58
+ context_window: int | None
59
+ max_output_tokens: int | None
60
+ input_modalities: tuple[str, ...]
61
+ #: USD per million tokens, parsed from OpenRouter-style ``pricing``
62
+ #: strings; ``None`` when the gateway does not advertise pricing.
63
+ prompt_price_per_mtok: float | None
64
+ completion_price_per_mtok: float | None
65
+ #: OpenRouter-style ``supported_parameters`` (``"reasoning"`` present
66
+ #: means the endpoint accepts a reasoning knob, levels unknown).
67
+ supported_parameters: tuple[str, ...]
68
+
69
+
70
+ @dataclass(frozen=True, slots=True)
71
+ class GatewayListing:
72
+ """A fetched listing plus its freshness provenance."""
73
+
74
+ entries: tuple[GatewayModel, ...]
75
+ #: Epoch seconds of the successful fetch (a stale listing keeps the
76
+ #: timestamp of the fetch that produced it, not of the failed refresh).
77
+ fetched_at: float
78
+ #: True when the last refresh failed and this is the previous listing.
79
+ stale: bool
80
+
81
+
82
+ @dataclass(frozen=True, slots=True)
83
+ class GatewayCatalogRow:
84
+ """A gateway id joined with catalog capabilities, ready for the wire.
85
+
86
+ ``joined_from`` names the catalog key that supplied capability fields
87
+ (``None`` when no catalog tier matched — the "capabilities unknown"
88
+ case the UI marks). ``info`` is the merged capability descriptor:
89
+ gateway-advertised fields win where the gateway provides them, catalog
90
+ fields (reasoning levels above all) fill the rest.
91
+ """
92
+
93
+ id: str
94
+ name: str
95
+ info: ModelInfo
96
+ joined_from: str | None
97
+ prompt_price_per_mtok: float | None
98
+ completion_price_per_mtok: float | None
99
+
100
+
101
+ # ---------------------------------------------------------------------------
102
+ # Parsing (pure)
103
+ # ---------------------------------------------------------------------------
104
+
105
+
106
+ def _first_int(*values: Any) -> int | None:
107
+ for value in values:
108
+ if isinstance(value, bool):
109
+ continue
110
+ if isinstance(value, (int, float)) and value > 0:
111
+ return int(value)
112
+ return None
113
+
114
+
115
+ def _first_str_list(value: Any) -> tuple[str, ...]:
116
+ if isinstance(value, (list, tuple)):
117
+ return tuple(str(v) for v in value if isinstance(v, str))
118
+ return ()
119
+
120
+
121
+ def _price_per_mtok(pricing: Any, key: str) -> float | None:
122
+ """OpenRouter ``pricing`` values are USD-per-token strings."""
123
+ if not isinstance(pricing, dict):
124
+ return None
125
+ raw = pricing.get(key)
126
+ if not isinstance(raw, (str, int, float)) or isinstance(raw, bool):
127
+ return None
128
+ try:
129
+ return float(raw) * 1_000_000
130
+ except (TypeError, ValueError):
131
+ return None
132
+
133
+
134
+ def _parse_entry(entry: Any, *, key_hint: str | None) -> GatewayModel | None:
135
+ if not isinstance(entry, dict):
136
+ return None
137
+ model_id = entry.get("id") or key_hint
138
+ if not isinstance(model_id, str) or not model_id:
139
+ return None
140
+ name = entry.get("name") or entry.get("display_name") or entry.get("displayName")
141
+ limit = entry.get("limit") if isinstance(entry.get("limit"), dict) else {}
142
+ top_provider = (
143
+ entry.get("top_provider") if isinstance(entry.get("top_provider"), dict) else {}
144
+ )
145
+ architecture = (
146
+ entry.get("architecture") if isinstance(entry.get("architecture"), dict) else {}
147
+ )
148
+ return GatewayModel(
149
+ id=model_id,
150
+ name=str(name) if isinstance(name, str) and name else model_id,
151
+ context_window=_first_int(
152
+ entry.get("context_length"),
153
+ entry.get("context_window"),
154
+ entry.get("contextWindow"),
155
+ entry.get("max_input_tokens"),
156
+ limit.get("context"),
157
+ ),
158
+ max_output_tokens=_first_int(
159
+ top_provider.get("max_completion_tokens"),
160
+ entry.get("max_completion_tokens"),
161
+ entry.get("max_output_tokens"),
162
+ entry.get("maxOutputTokens"),
163
+ entry.get("maxTokens"),
164
+ entry.get("max_tokens"),
165
+ limit.get("output"),
166
+ ),
167
+ input_modalities=_first_str_list(architecture.get("input_modalities")),
168
+ prompt_price_per_mtok=_price_per_mtok(entry.get("pricing"), "prompt"),
169
+ completion_price_per_mtok=_price_per_mtok(entry.get("pricing"), "completion"),
170
+ supported_parameters=_first_str_list(entry.get("supported_parameters")),
171
+ )
172
+
173
+
174
+ def parse_models_listing(payload: Any) -> list[GatewayModel]:
175
+ """Parse a ``GET /models`` body into normalized rows.
176
+
177
+ Two shapes are in the wild: OpenAI-style ``{"data": [...]}`` (the common
178
+ case, OpenRouter's enriched variant included) and models.dev-style
179
+ ``{"models": {...}}`` where the property key is the id. Anything else is
180
+ a parse error the caller surfaces as ``GatewayCatalogError``.
181
+ """
182
+ if not isinstance(payload, dict):
183
+ raise ValueError("listing payload is not a JSON object")
184
+ data = payload.get("data")
185
+ if isinstance(data, list):
186
+ rows = [_parse_entry(entry, key_hint=None) for entry in data]
187
+ return [row for row in rows if row is not None]
188
+ models = payload.get("models")
189
+ if isinstance(models, dict):
190
+ rows = [_parse_entry(entry, key_hint=key) for key, entry in models.items()]
191
+ return [row for row in rows if row is not None]
192
+ if isinstance(models, list):
193
+ rows = [_parse_entry(entry, key_hint=None) for entry in models]
194
+ return [row for row in rows if row is not None]
195
+ raise ValueError("listing has neither a 'data' array nor a 'models' map")
196
+
197
+
198
+ # ---------------------------------------------------------------------------
199
+ # Catalog join (pure)
200
+ # ---------------------------------------------------------------------------
201
+
202
+
203
+ def merge_with_catalog(models: Iterable[GatewayModel]) -> list[GatewayCatalogRow]:
204
+ """Join gateway rows with catalog capabilities (cross-provider leaf).
205
+
206
+ The gateway's own fields (window, modalities) win where advertised —
207
+ they describe this deployment. Capability fields come from
208
+ ``resolve_model_info`` itself — the very resolution the request path's
209
+ ``clamp_reasoning_effort`` uses, legacy-union rule included — so
210
+ ``models.list`` and the request path never disagree about a model's
211
+ knob (a picker that hides a level the backend would accept is the same
212
+ class of drift as silently dropping one). ``joined_from`` keeps the
213
+ leaf-join provenance: ``None`` when no catalog tier matched, even if
214
+ the hand-owned legacy table knows the model.
215
+ """
216
+ from .model_info import (
217
+ TOOL_FORMAT_OPENAI,
218
+ ModelInfo,
219
+ resolve_model_info,
220
+ )
221
+ from .model_resolve import resolve_leaf_cross_provider
222
+
223
+ rows: list[GatewayCatalogRow] = []
224
+ for model in models:
225
+ hit = resolve_leaf_cross_provider(model.id)
226
+ base = resolve_model_info(model.id)
227
+ context_window = (
228
+ model.context_window
229
+ if model.context_window is not None
230
+ else base.context_window
231
+ )
232
+ modalities = (
233
+ frozenset(model.input_modalities)
234
+ if model.input_modalities
235
+ else base.modalities
236
+ )
237
+ rows.append(
238
+ GatewayCatalogRow(
239
+ id=model.id,
240
+ name=model.name,
241
+ info=ModelInfo(
242
+ pattern=model.id.lower(),
243
+ context_window=context_window,
244
+ modalities=modalities,
245
+ tool_format=TOOL_FORMAT_OPENAI,
246
+ reasoning_levels=base.reasoning_levels,
247
+ ),
248
+ joined_from=hit.key if hit is not None else None,
249
+ prompt_price_per_mtok=model.prompt_price_per_mtok,
250
+ completion_price_per_mtok=model.completion_price_per_mtok,
251
+ )
252
+ )
253
+ return rows
254
+
255
+
256
+ # ---------------------------------------------------------------------------
257
+ # Fetch (network, TTL-cached, stale-on-error)
258
+ # ---------------------------------------------------------------------------
259
+
260
+ #: base_url (normalized) -> (time.monotonic() at fetch, listing). The wall
261
+ #: clock ``listing.fetched_at`` is for display; the monotonic clock drives
262
+ #: TTL so a clock jump cannot expire or resurrect a listing.
263
+ _cache: dict[str, tuple[float, GatewayListing]] = {}
264
+
265
+
266
+ def _cache_key(base_url: str) -> str:
267
+ return base_url.rstrip("/").lower()
268
+
269
+
270
+ def clear_gateway_cache() -> None:
271
+ """Drop every cached listing (tests, forced refresh)."""
272
+ _cache.clear()
273
+
274
+
275
+ async def fetch_gateway_models(
276
+ base_url: str,
277
+ api_key: str | None = None,
278
+ *,
279
+ timeout_sec: float = DEFAULT_TIMEOUT_SEC,
280
+ ttl_sec: float = DEFAULT_TTL_SEC,
281
+ ) -> GatewayListing:
282
+ """Fetch the gateway's live model listing, with a process-local cache.
283
+
284
+ A fresh-enough cached listing is served without network. On refresh
285
+ failure the previous listing is returned marked ``stale`` — the catalog
286
+ degrades instead of disappearing when the gateway flaps. With no cache
287
+ to fall back on, ``GatewayCatalogError`` is raised and the caller
288
+ decides (the sidecar answers ``catalog_status: "offline"``).
289
+ """
290
+ import httpx # local import — keeps the runtime importable without httpx
291
+
292
+ from .llm.system_proxy import client_env_kwargs
293
+
294
+ key = _cache_key(base_url)
295
+ cached = _cache.get(key)
296
+ if cached is not None and (time.monotonic() - cached[0]) < ttl_sec:
297
+ return cached[1]
298
+
299
+ url = f"{base_url.rstrip('/')}/models"
300
+ headers = {"Authorization": f"Bearer {api_key}"} if api_key else {}
301
+ try:
302
+ async with httpx.AsyncClient(
303
+ timeout=httpx.Timeout(timeout_sec),
304
+ **client_env_kwargs(base_url),
305
+ ) as client:
306
+ response = await client.get(url, headers=headers)
307
+ response.raise_for_status()
308
+ entries = tuple(parse_models_listing(response.json()))
309
+ except Exception as exc:
310
+ if cached is not None:
311
+ _log.info("gateway listing refresh failed (%s); serving stale", exc)
312
+ return GatewayListing(
313
+ entries=cached[1].entries, fetched_at=cached[1].fetched_at, stale=True
314
+ )
315
+ raise GatewayCatalogError(base_url, str(exc)) from exc
316
+
317
+ listing = GatewayListing(entries=entries, fetched_at=time.time(), stale=False)
318
+ _cache[key] = (time.monotonic(), listing)
319
+ return listing
@@ -209,6 +209,11 @@ class PressureCompaction:
209
209
  # trades cache hits for a bounded transcript. R16 found compaction does not
210
210
  # move the eval score, so this stays opt-in capability, not a default.
211
211
  micro_compact_interval_rounds: int = 0
212
+ # Rapid-refill breaker (CC parity): a successful compaction whose freed
213
+ # space refills within this many rounds of the previous one counts as a
214
+ # refill; max_rapid_refills consecutive refills open the circuit.
215
+ rapid_refill_window_rounds: int = 3
216
+ max_rapid_refills: int = 3
212
217
  name: str = "pressure_compaction"
213
218
  assumes: str = (
214
219
  "long trajectories exceed the window; older detail is expendable "
@@ -231,6 +236,8 @@ class PressureCompaction:
231
236
  summarizer=provider,
232
237
  model=self.model,
233
238
  micro_compact_interval_rounds=self.micro_compact_interval_rounds,
239
+ rapid_refill_window_rounds=self.rapid_refill_window_rounds,
240
+ max_rapid_refills=self.max_rapid_refills,
234
241
  **extra,
235
242
  ),
236
243
  forward=("compact_now",),
@@ -158,7 +158,9 @@ from .errors import (
158
158
  classify_http_status,
159
159
  is_retryable,
160
160
  )
161
+ from .google_genai import GoogleGenAIProvider
161
162
  from .openai_compat import OpenAICompatProvider
163
+ from .openai_responses import OpenAIResponsesProvider
162
164
  from .presets import (
163
165
  PROVIDER_PRESETS,
164
166
  PresetEntry,
@@ -174,6 +176,7 @@ __all__ = [
174
176
  "RETRYABLE_KINDS",
175
177
  "AnthropicProvider",
176
178
  "ContentPart",
179
+ "GoogleGenAIProvider",
177
180
  "ImagePart",
178
181
  "LLMError",
179
182
  "LLMErrorKind",
@@ -188,6 +191,7 @@ __all__ = [
188
191
  "describe_compat_flags",
189
192
  "describe_provider_presets",
190
193
  "OpenAICompatProvider",
194
+ "OpenAIResponsesProvider",
191
195
  "TextPart",
192
196
  "classify_error",
193
197
  "classify_http_status",