steerable-agent-runtime 0.4.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/PKG-INFO +1 -1
  2. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/pyproject.toml +1 -1
  3. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/__init__.py +34 -0
  4. steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/ask_user.py +154 -0
  5. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/cache_control.py +102 -2
  6. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/compaction.py +136 -2
  7. steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/config.py +225 -0
  8. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/harness.py +185 -3
  9. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/history.py +21 -0
  10. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/hooks.py +27 -4
  11. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/anthropic_native.py +15 -8
  12. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/compat.py +2 -2
  13. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/loop.py +103 -20
  14. steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/plugins.py +67 -0
  15. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/reminders.py +19 -20
  16. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/skills.py +5 -2
  17. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/subagent.py +130 -22
  18. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
  19. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/SOURCES.txt +6 -0
  20. steerable_agent_runtime-0.6.0/tests/test_ask_user.py +254 -0
  21. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_cache_control.py +103 -2
  22. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_compaction.py +244 -0
  23. steerable_agent_runtime-0.6.0/tests/test_config.py +217 -0
  24. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_harness.py +114 -0
  25. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_hooks.py +56 -0
  26. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_long_session.py +3 -3
  27. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop.py +220 -0
  28. steerable_agent_runtime-0.6.0/tests/test_plugins.py +84 -0
  29. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_reminders.py +25 -1
  30. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_subagent.py +171 -0
  31. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/README.md +0 -0
  32. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/setup.cfg +0 -0
  33. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/antihallucination.py +0 -0
  34. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/approval.py +0 -0
  35. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/approval_policy.py +0 -0
  36. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/branch.py +0 -0
  37. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/calibration.py +0 -0
  38. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/default.harness.json +0 -0
  39. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/default.harness.yaml +0 -0
  40. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/errors.py +0 -0
  41. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/handoff.py +0 -0
  42. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/harness_spec.py +0 -0
  43. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/__init__.py +0 -0
  44. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/errors.py +0 -0
  45. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/openai_compat.py +0 -0
  46. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/parts.py +0 -0
  47. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
  48. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/maintenance.py +0 -0
  49. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/mcp.py +0 -0
  50. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/mcp_server.py +0 -0
  51. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_catalog.py +0 -0
  52. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_info.py +0 -0
  53. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_resolve.py +0 -0
  54. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/observation_aging.py +0 -0
  55. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/orchestration.py +0 -0
  56. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/otel.py +0 -0
  57. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/pricing.py +0 -0
  58. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/pseudo.py +0 -0
  59. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/recording.py +0 -0
  60. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/replay.py +0 -0
  61. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/resume.py +0 -0
  62. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/retry.py +0 -0
  63. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/sandboxed.py +0 -0
  64. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/spill.py +0 -0
  65. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/__init__.py +0 -0
  66. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
  67. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
  68. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
  69. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
  70. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tokens.py +0 -0
  71. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tool_schema.py +0 -0
  72. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tool_search.py +0 -0
  73. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tools.py +0 -0
  74. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tracing.py +0 -0
  75. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/__init__.py +0 -0
  76. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
  77. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
  78. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/world_state.py +0 -0
  79. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
  80. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
  81. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
  82. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_antihallucination.py +0 -0
  83. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_approval.py +0 -0
  84. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_approval_policy.py +0 -0
  85. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_branch.py +0 -0
  86. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_cache_instrumentation.py +0 -0
  87. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_calibration.py +0 -0
  88. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_content_parts.py +0 -0
  89. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_error_taxonomy.py +0 -0
  90. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_fragment_bounds.py +0 -0
  91. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_golden.py +0 -0
  92. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_handoff.py +0 -0
  93. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_harness_spec.py +0 -0
  94. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_history.py +0 -0
  95. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_history_persistence.py +0 -0
  96. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_in_memory_storage.py +0 -0
  97. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_llm_wire_helpers.py +0 -0
  98. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_cancellation.py +0 -0
  99. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_replay.py +0 -0
  100. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_sandbox_event.py +0 -0
  101. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_maintenance.py +0 -0
  102. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_mcp.py +0 -0
  103. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_mcp_server.py +0 -0
  104. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_catalog.py +0 -0
  105. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_equivalence.py +0 -0
  106. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_info.py +0 -0
  107. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_resolve.py +0 -0
  108. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_observation_aging.py +0 -0
  109. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_orchestration.py +0 -0
  110. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_otel.py +0 -0
  111. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_parallel_tools.py +0 -0
  112. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_provider_compat.py +0 -0
  113. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_pseudo.py +0 -0
  114. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_recording.py +0 -0
  115. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_replay_crosslang.py +0 -0
  116. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_resume.py +0 -0
  117. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_retry_hooks.py +0 -0
  118. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_safety_gate.py +0 -0
  119. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_sandboxed.py +0 -0
  120. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_skills.py +0 -0
  121. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_soft_timeout.py +0 -0
  122. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_spill.py +0 -0
  123. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_sqlite_storage.py +0 -0
  124. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_steer.py +0 -0
  125. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_storage_contract.py +0 -0
  126. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_stream_strip.py +0 -0
  127. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_system_proxy.py +0 -0
  128. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tokens.py +0 -0
  129. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_exposure.py +0 -0
  130. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_hygiene.py +0 -0
  131. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_router.py +0 -0
  132. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_schema.py +0 -0
  133. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_timeout.py +0 -0
  134. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_trace_recorder.py +0 -0
  135. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_transport_jsonrpc.py +0 -0
  136. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_transport_sse.py +0 -0
  137. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_usage_attribution.py +0 -0
  138. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_world_state.py +0 -0
  139. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_write_lease.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steerable-agent-runtime
3
- Version: 0.4.0
3
+ Version: 0.6.0
4
4
  Summary: Steerable agent runtime: LLM, tool, storage, and transport adapters.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "steerable-agent-runtime"
3
- version = "0.4.0"
3
+ version = "0.6.0"
4
4
  description = "Steerable agent runtime: LLM, tool, storage, and transport adapters."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -29,6 +29,24 @@ from .approval_policy import (
29
29
  PolicyApprover,
30
30
  rule_from_amendment,
31
31
  )
32
+ from .ask_user import (
33
+ ASK_USER_SCHEMA,
34
+ ASK_USER_TOOL_NAME,
35
+ AskUserHandler,
36
+ make_ask_user_tool,
37
+ )
38
+ from .config import (
39
+ DEFAULT_CONFIG_PATH,
40
+ ConfigError,
41
+ ResolvedConfig,
42
+ load_user_config,
43
+ resolve_config,
44
+ )
45
+ from .plugins import (
46
+ STEERABLE_TOOLS_ENTRY_POINT_GROUP,
47
+ PluginLoadError,
48
+ load_tool_entry_points,
49
+ )
32
50
  from .branch import (
33
51
  BranchPoint,
34
52
  ForkResult,
@@ -39,6 +57,7 @@ from .branch import (
39
57
  )
40
58
  from .cache_control import (
41
59
  CacheControlProvider,
60
+ CacheDriftMonitor,
42
61
  CacheRetention,
43
62
  place_cache_breakpoints,
44
63
  system_blocks_with_cache,
@@ -195,6 +214,7 @@ from .subagent import (
195
214
  FilteredToolsExecutor,
196
215
  SubagentConfig,
197
216
  SubagentExecutor,
217
+ SubagentRegistry,
198
218
  subagent_tool_descriptor,
199
219
  )
200
220
  from .tokens import (
@@ -230,8 +250,13 @@ __all__ = [
230
250
  "REASONING_EFFORT_ORDER",
231
251
  "RECORD_FORMAT_VERSION",
232
252
  "TOOL_SEARCH_NAME",
253
+ "ASK_USER_SCHEMA",
254
+ "ASK_USER_TOOL_NAME",
255
+ "STEERABLE_TOOLS_ENTRY_POINT_GROUP",
233
256
  "AgentPool",
234
257
  "AntiHallucinationConfig",
258
+ "AskUserHandler",
259
+ "PluginLoadError",
235
260
  "AntiHallucinationHooks",
236
261
  "ApprovalAborted",
237
262
  "ApprovalDecision",
@@ -242,9 +267,13 @@ __all__ = [
242
267
  "ApprovalStore",
243
268
  "Approver",
244
269
  "AutoApprover",
270
+ "ConfigError",
271
+ "DEFAULT_CONFIG_PATH",
272
+ "ResolvedConfig",
245
273
  "BranchPoint",
246
274
  "BudgetExhaustedError",
247
275
  "CacheControlProvider",
276
+ "CacheDriftMonitor",
248
277
  "CacheRetention",
249
278
  "CalibratingProvider",
250
279
  "ChainHooks",
@@ -328,6 +357,7 @@ __all__ = [
328
357
  "StorageError",
329
358
  "SubagentConfig",
330
359
  "SubagentExecutor",
360
+ "SubagentRegistry",
331
361
  "TextPart",
332
362
  "ToolDispatchError",
333
363
  "ToolExecutor",
@@ -367,8 +397,12 @@ __all__ = [
367
397
  "last_world_state_snapshot",
368
398
  "lineage",
369
399
  "load_history_transcript",
400
+ "load_tool_entry_points",
401
+ "load_user_config",
402
+ "resolve_config",
370
403
  "load_recorded_requests",
371
404
  "load_transcript",
405
+ "make_ask_user_tool",
372
406
  "matches_conditions",
373
407
  "mcp_invoker",
374
408
  "merge_patch",
@@ -0,0 +1,154 @@
1
+ """Structured user questions — the ``ask_user`` tool.
2
+
3
+ The agent pauses mid-run and asks the user a structured set of questions
4
+ (select / text / password) before continuing. The wire payload mirrors the
5
+ protocol's ``AskUserQuestionsPayload`` (TS) so the desktop card renders it
6
+ unchanged; the handler is the seam a product injects to actually collect the
7
+ answers (a UI card over the reverse channel, an ACP elicitation, a CLI
8
+ prompt).
9
+
10
+ The tool is blocking by construction: ``dispatch`` awaits the handler, and
11
+ the answers are returned as the tool result so they land in the durable
12
+ record and the model's next context. This mirrors dsh's ``ask_user_question``
13
+ — the loop does not continue until the user (or a timeout) answers.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from collections.abc import Awaitable, Callable
19
+ from typing import Any, Protocol
20
+
21
+ from .errors import ToolDispatchError
22
+
23
+ #: The tool name the model calls.
24
+ ASK_USER_TOOL_NAME = "ask_user"
25
+
26
+ #: JSON Schema for the tool's arguments — one intro plus a list of questions.
27
+ #: Field names match ``AskUserQuestionsPayload`` so the host can render the
28
+ #: arguments directly as the card payload.
29
+ ASK_USER_SCHEMA: dict[str, Any] = {
30
+ "type": "object",
31
+ "properties": {
32
+ "intro": {
33
+ "type": "string",
34
+ "description": "One-line framing shown above the questions.",
35
+ },
36
+ "outro": {
37
+ "type": "string",
38
+ "description": "Optional line shown below the questions.",
39
+ },
40
+ "questions": {
41
+ "type": "array",
42
+ "items": {
43
+ "type": "object",
44
+ "properties": {
45
+ "id": {"type": "string"},
46
+ "text": {"type": "string"},
47
+ "type": {
48
+ "type": "string",
49
+ "enum": ["select", "text", "password"],
50
+ },
51
+ "options": {"type": "array", "items": {"type": "string"}},
52
+ "placeholder": {"type": "string"},
53
+ "multiSelect": {"type": "boolean"},
54
+ },
55
+ "required": ["id", "text"],
56
+ },
57
+ },
58
+ },
59
+ "required": ["intro", "questions"],
60
+ }
61
+
62
+ #: Alias keys real models emit for the canonical ``AskUserQuestionsPayload``
63
+ #: fields (observed: gpt-oss via Ollama answers the schema with Inquirer-style
64
+ #: ``name``/``message``/``choices``). Normalized at this model-JSON boundary so
65
+ #: hosts only ever render the canonical payload; the canonical key wins when
66
+ #: both are present.
67
+ _QUESTION_ALIASES = {"id": "name", "text": "message", "options": "choices"}
68
+
69
+
70
+ def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]]:
71
+ """Map known alias keys onto the canonical payload fields and require the
72
+ two fields the host card cannot render without (``id`` to key the answer,
73
+ ``text`` to label the control). A violation raises ``ToolDispatchError`` —
74
+ the router wraps it as the tool result, so the model sees exactly what to
75
+ fix and retries within the same turn."""
76
+ if not isinstance(questions, list):
77
+ raise ToolDispatchError(
78
+ f"ask_user: questions must be an array, got {type(questions).__name__}"
79
+ )
80
+ normalized: list[dict[str, Any]] = []
81
+ for index, question in enumerate(questions):
82
+ if not isinstance(question, dict):
83
+ raise ToolDispatchError(
84
+ f"ask_user: questions[{index}] must be an object, "
85
+ f"got {type(question).__name__}"
86
+ )
87
+ q = dict(question)
88
+ for canonical, alias in _QUESTION_ALIASES.items():
89
+ if canonical not in q and alias in q:
90
+ q[canonical] = q.pop(alias)
91
+ if not isinstance(q.get("id"), str) or not q["id"]:
92
+ raise ToolDispatchError(
93
+ f"ask_user: questions[{index}] is missing a non-empty string "
94
+ '"id" (the answer-map key).'
95
+ )
96
+ if not isinstance(q.get("text"), str) or not q["text"]:
97
+ raise ToolDispatchError(
98
+ f"ask_user: questions[{index}] is missing a non-empty string "
99
+ '"text" (the question label shown to the user).'
100
+ )
101
+ normalized.append(q)
102
+ return normalized
103
+
104
+
105
+ class AskUserHandler(Protocol):
106
+ """Product-injected seam that collects answers for one question set.
107
+
108
+ Receives the validated ``intro`` and ``questions`` list and returns a
109
+ mapping of question id → answer (a string, or a list of strings for a
110
+ multi-select). Implementations must not raise on a user cancel — return
111
+ an empty mapping so the loop records "no answer" and moves on.
112
+ """
113
+
114
+ def __call__(
115
+ self, intro: str, questions: list[dict[str, Any]]
116
+ ) -> Awaitable[dict[str, Any]]: ...
117
+
118
+
119
+ def make_ask_user_tool(handler: AskUserHandler) -> Callable[..., Awaitable[dict[str, Any]]]:
120
+ """Build the ``ask_user`` tool handler bound to ``handler``.
121
+
122
+ The returned coroutine function carries the registration metadata the
123
+ router reads (name, description, schema) so a host registers it with one
124
+ call. The tool is marked ``require_consent=False`` and
125
+ ``concurrency_safe=False``: asking the user is itself the interaction, and
126
+ two concurrent question cards would race for the same UI surface.
127
+ """
128
+
129
+ async def ask_user(
130
+ intro: str, questions: list[dict[str, Any]], outro: str | None = None
131
+ ) -> dict[str, Any]:
132
+ normalized = _normalize_questions(questions)
133
+ answers = await handler(intro, normalized)
134
+ return {
135
+ "intro": intro,
136
+ "outro": outro,
137
+ "questions": normalized,
138
+ "answers": answers,
139
+ }
140
+
141
+ ask_user.__steerable_tool_meta__ = { # type: ignore[attr-defined]
142
+ "name": ASK_USER_TOOL_NAME,
143
+ "mode": "read", # no side effects on the workspace; it gathers input
144
+ "description": (
145
+ "Pause and ask the user a structured set of questions (select / "
146
+ "text / password) when you need information only they have. "
147
+ "Blocks until they answer; the answers come back as the result."
148
+ ),
149
+ "schema": ASK_USER_SCHEMA,
150
+ "require_consent": False,
151
+ "concurrency_safe": False,
152
+ "exposure": "direct",
153
+ }
154
+ return ask_user
@@ -23,12 +23,16 @@ comes from the prefix stability the rest of the stack already keeps.
23
23
 
24
24
  from __future__ import annotations
25
25
 
26
+ import logging
26
27
  from collections.abc import AsyncIterator, Sequence
27
28
  from dataclasses import dataclass
28
29
  from typing import Any, Iterable, Literal
29
30
 
31
+ from .hooks import NoopHooks, PreStepAction
30
32
  from .llm import LLMMessage, LLMProvider, LLMStreamChunk, LLMUsage
31
33
 
34
+ logger = logging.getLogger(__name__)
35
+
32
36
  CacheRetention = Literal["none", "short", "long"]
33
37
 
34
38
  #: Anthropic's explicit breakpoint API. Other providers have no breakpoint
@@ -36,6 +40,19 @@ CacheRetention = Literal["none", "short", "long"]
36
40
  _EXPLICIT_CACHE_PROVIDERS = {"anthropic", "claude"}
37
41
 
38
42
 
43
+ def _marker_for(retention: CacheRetention) -> dict[str, Any]:
44
+ """Map a retention class onto the Anthropic breakpoint marker.
45
+
46
+ ``short`` is the 5-minute default; ``long`` opts into the 1-hour TTL
47
+ (CC ``CLAUDE_CODE_PROMPT_CACHE_TTL`` parity — a deliberate trade: fewer
48
+ cache writes on slow-moving prefixes, at Anthropic's higher 1h write
49
+ price). ``none`` never reaches here (the request skips anchoring).
50
+ """
51
+ if retention == "long":
52
+ return {"type": "ephemeral", "ttl": "1h"}
53
+ return {"type": "ephemeral"}
54
+
55
+
39
56
  @dataclass(slots=True)
40
57
  class CacheControlProvider:
41
58
  """LLMProvider decorator: emits prompt-cache breakpoints per request.
@@ -105,10 +122,15 @@ class CacheControlProvider:
105
122
  kwargs.pop("cache_retention", None)
106
123
  return tools, kwargs
107
124
  out = dict(kwargs)
125
+ marker = _marker_for(retention)
108
126
  shaped_tools = (
109
- place_cache_breakpoints(list(tools)) if tools is not None else None
127
+ place_cache_breakpoints(list(tools), cache_control=marker)
128
+ if tools is not None
129
+ else None
110
130
  )
111
- out["_cache_tail_anchor"] = True
131
+ # The provider-owned anchors (system block, transcript tail) read the
132
+ # marker off this key so all three breakpoints share one TTL.
133
+ out["_cache_tail_anchor"] = marker
112
134
  return shaped_tools, out
113
135
 
114
136
  def _effective_retention(self, kwargs: dict[str, Any]) -> CacheRetention:
@@ -152,3 +174,81 @@ def system_blocks_with_cache(
152
174
  """
153
175
  marker = cache_control or {"type": "ephemeral"}
154
176
  return [{"type": "text", "text": system_text, "cache_control": marker}]
177
+
178
+
179
+ class CacheDriftMonitor(NoopHooks):
180
+ """``pre_step`` observer: flag prompt-cache drift from provider usage.
181
+
182
+ Each round it reads the PREVIOUS request's accounting off the loop ctx
183
+ (``last_prompt_tokens`` / ``last_cached_prompt_tokens`` — the read side
184
+ surfaced on ``stage_complete``). The monitor arms once a round shows the
185
+ cache actually serving tokens; from then on, ``consecutive_rounds``
186
+ rounds whose hit rate stays below ``min_hit_rate`` (while the prompt is
187
+ at least ``min_prompt_tokens`` — below that, caching barely matters) set
188
+ ``drift_detected`` and log a warning with the numbers. A round at or
189
+ above the rate resets the count and clears the flag.
190
+
191
+ This is the deliberate loop logic behind the raw ``stage_complete``
192
+ numbers (CC ``globalCacheStrategy`` / ``cacheControlHash`` parity): a
193
+ collapse means the cached prefix broke — a transcript rewrite, a changed
194
+ tool list, a TTL expiry — and one low round is normal right after a
195
+ compaction (the prefix re-warms on the next request), which is why the
196
+ verdict requires consecutive low rounds. Providers without cache
197
+ accounting never arm the monitor, so implicit-cache deployments that
198
+ report nothing see no false drift.
199
+ """
200
+
201
+ def __init__(
202
+ self,
203
+ *,
204
+ min_prompt_tokens: int = 1024,
205
+ min_hit_rate: float = 0.5,
206
+ consecutive_rounds: int = 3,
207
+ ) -> None:
208
+ if not 0 < min_hit_rate <= 1:
209
+ raise ValueError("min_hit_rate must be in (0, 1]")
210
+ if not 1 <= consecutive_rounds:
211
+ raise ValueError("consecutive_rounds must be >= 1")
212
+ self._min_prompt = min_prompt_tokens
213
+ self._min_hit_rate = min_hit_rate
214
+ self._consecutive_rounds = consecutive_rounds
215
+ # Observability for callers/tests (CompactionHooks-counter pattern).
216
+ self.armed = False
217
+ self.drift_detected = False
218
+ self.consecutive_low_hit_rounds = 0
219
+ self.last_hit_rate: float | None = None
220
+
221
+ async def pre_step(
222
+ self, transcript: list[LLMMessage], ctx: Any
223
+ ) -> PreStepAction:
224
+ prompt = getattr(ctx, "last_prompt_tokens", None)
225
+ cached = getattr(ctx, "last_cached_prompt_tokens", None)
226
+ if not prompt or cached is None or prompt < self._min_prompt:
227
+ return PreStepAction(kind="proceed")
228
+ hit_rate = cached / prompt
229
+ self.last_hit_rate = hit_rate
230
+ if cached > 0:
231
+ self.armed = True
232
+ if not self.armed:
233
+ return PreStepAction(kind="proceed")
234
+ if hit_rate >= self._min_hit_rate:
235
+ self.consecutive_low_hit_rounds = 0
236
+ self.drift_detected = False
237
+ return PreStepAction(kind="proceed")
238
+ self.consecutive_low_hit_rounds += 1
239
+ if (
240
+ self.consecutive_low_hit_rounds >= self._consecutive_rounds
241
+ and not self.drift_detected
242
+ ):
243
+ self.drift_detected = True
244
+ logger.warning(
245
+ "prompt-cache drift: hit rate %.2f below %.2f for %d "
246
+ "consecutive rounds (prompt=%d cached=%d) — the cached "
247
+ "prefix broke (rewrite, tool-list change, or TTL expiry)",
248
+ hit_rate,
249
+ self._min_hit_rate,
250
+ self.consecutive_low_hit_rounds,
251
+ prompt,
252
+ cached,
253
+ )
254
+ return PreStepAction(kind="proceed")
@@ -25,6 +25,36 @@ System messages and the first user message (the goal) are always kept; the
25
25
  most recent ``keep_last_messages`` are never touched. Between compactions
26
26
  the transcript is append-only, so provider prompt caches keep hitting; a
27
27
  rewrite invalidates the cache once, then the prefix is stable again.
28
+
29
+ Three trigger paths share the fold/summarize machinery:
30
+
31
+ - **pressure** (``pre_step``, above) — the reactive default;
32
+ - **overflow recovery** (``on_request_error``) — the heuristic threshold can
33
+ still miss the real window, so a context-overflow error forces one
34
+ compaction pass and retries, bounded per round;
35
+ - **micro-compaction** (``pre_step``, opt-in ``micro_compact_interval_rounds``)
36
+ — every N rounds, fold old tool results regardless of pressure (CC
37
+ time-based microcompact parity). Off by default: each fold invalidates the
38
+ prompt-cache prefix, so the interval trades cache hits for a bounded
39
+ transcript;
40
+ - **manual** (``compact_now``) — the host-command path (CC ``/compact``
41
+ parity): fold + summarize on demand, bypassing threshold, hysteresis, and
42
+ the circuit breaker.
43
+
44
+ Two safety rails share the machinery:
45
+
46
+ - Every rewrite carries its ``pre_tokens`` / ``post_tokens`` estimate onto
47
+ the recorded ``CompactionBoundary`` (CC ``compact_boundary`` parity), so
48
+ traces chart compaction effectiveness without re-estimating from bodies.
49
+ - A **circuit breaker** (CC auto-compact breaker parity): a pressure
50
+ compaction whose post-estimate is still over threshold counts as
51
+ ineffective; three consecutive ineffective compactions open the circuit
52
+ and the pressure path stops firing — each further rewrite would only
53
+ invalidate the prompt-cache prefix without shrinking the transcript. The
54
+ overflow-recovery path keeps its own per-round bound and stays live, so
55
+ the turn still fails loud (bounded retries, then a named error) instead
56
+ of spinning. A round that lands under threshold — or any successful
57
+ compaction — resets the count.
28
58
  """
29
59
 
30
60
  from __future__ import annotations
@@ -75,17 +105,26 @@ class CompactionHooks(NoopHooks):
75
105
  model: str | None = None,
76
106
  recompact_margin_ratio: float = 0.1,
77
107
  fold_excerpt_chars: int = _FOLD_EXCERPT_CHARS,
108
+ micro_compact_interval_rounds: int = 0,
78
109
  ) -> None:
79
110
  if not 0 < threshold_ratio <= 1:
80
111
  raise ValueError("threshold_ratio must be in (0, 1]")
81
112
  if not 0 <= recompact_margin_ratio:
82
113
  raise ValueError("recompact_margin_ratio must be >= 0")
114
+ if not 0 <= micro_compact_interval_rounds:
115
+ raise ValueError("micro_compact_interval_rounds must be >= 0")
83
116
  self._max_tokens = max_context_tokens
84
117
  self._threshold = threshold_ratio
85
118
  self._keep_last = keep_last_messages
86
119
  self._keep_last_tools = keep_last_tool_results
87
120
  self._fold_excerpt_chars = fold_excerpt_chars
88
121
  self._summarizer = summarizer
122
+ #: Proactive micro-compaction (CC time-based microcompact parity):
123
+ #: every N rounds, fold old tool results regardless of pressure, so a
124
+ #: long chatty run never drifts toward the window. 0 (default) is off
125
+ #: — each fold invalidates the provider prompt-cache prefix, so the
126
+ #: interval trades cache hits for a bounded transcript.
127
+ self._micro_compact_interval = micro_compact_interval_rounds
89
128
  #: Model name used for calibrated token estimates (see tokens.py).
90
129
  self._model = model
91
130
  #: Hysteresis: after a compaction, pressure must grow by
@@ -98,6 +137,8 @@ class CompactionHooks(NoopHooks):
98
137
  self._last_compaction_pressure: int | None = None
99
138
  # Observability for callers/tests: how many compactions happened.
100
139
  self.compactions = 0
140
+ # Observability: how many of them were periodic micro-compactions.
141
+ self.micro_compactions = 0
101
142
  # Overflow-recovery state: consecutive context-overflow retries within
102
143
  # one round. Capped so a pathological transcript cannot spin a
103
144
  # compact→overflow→compact loop forever.
@@ -107,6 +148,14 @@ class CompactionHooks(NoopHooks):
107
148
  self.max_overflow_retries = 2
108
149
  # Observability: how many overflow-driven compactions happened.
109
150
  self.overflow_recoveries = 0
151
+ #: Consecutive ineffective pressure compactions (post-estimate still
152
+ #: over threshold) before the circuit opens (CC auto-compact breaker
153
+ #: parity). The overflow path is bounded separately and unaffected.
154
+ self.max_consecutive_failures = 3
155
+ self._consecutive_failures = 0
156
+ # Observability: True once the breaker tripped; the pressure path
157
+ # stops firing for the rest of the session.
158
+ self.circuit_open = False
110
159
 
111
160
  def _estimate(self, transcript: Sequence[LLMMessage]) -> int:
112
161
  return estimate_tokens(transcript, model=self._model)
@@ -137,8 +186,40 @@ class CompactionHooks(NoopHooks):
137
186
  async def pre_step(
138
187
  self, transcript: list[LLMMessage], ctx: Any
139
188
  ) -> PreStepAction:
189
+ round_index = getattr(ctx, "round_index", 0)
190
+ if (
191
+ self._micro_compact_interval > 0
192
+ and round_index > 0
193
+ and round_index % self._micro_compact_interval == 0
194
+ ):
195
+ pruned = self._fold_old_tool_results(transcript)
196
+ if pruned is not transcript:
197
+ self.compactions += 1
198
+ self.micro_compactions += 1
199
+ self._reset_observed(ctx)
200
+ return PreStepAction(
201
+ kind="proceed",
202
+ rewrite=RewriteRequest(
203
+ messages=pruned,
204
+ reason=(
205
+ "micro-compact: periodic tool-result prune "
206
+ f"(every {self._micro_compact_interval} rounds)"
207
+ ),
208
+ action="micro_compact",
209
+ pre_tokens=self._estimate(transcript),
210
+ post_tokens=self._estimate(pruned),
211
+ ),
212
+ )
140
213
  pressure = self._pressure(transcript, ctx)
141
214
  if pressure < self._threshold * self._max_tokens:
215
+ # A healthy round resets the breaker count — the pathology (or
216
+ # the transcript that caused it) is gone.
217
+ self._consecutive_failures = 0
218
+ return PreStepAction(kind="proceed")
219
+ if self.circuit_open:
220
+ # Breaker tripped: further pressure rewrites only invalidate the
221
+ # prompt-cache prefix without shrinking the transcript. The
222
+ # overflow path stays live and fails loud if the window is hit.
142
223
  return PreStepAction(kind="proceed")
143
224
  if (
144
225
  self._last_compaction_pressure is not None
@@ -146,9 +227,11 @@ class CompactionHooks(NoopHooks):
146
227
  ):
147
228
  return PreStepAction(kind="proceed")
148
229
 
230
+ threshold = self._threshold * self._max_tokens
149
231
  compacted = self._fold_old_tool_results(transcript)
150
- if self._estimate(compacted) < self._threshold * self._max_tokens:
232
+ if self._estimate(compacted) < threshold:
151
233
  self.compactions += 1
234
+ self._consecutive_failures = 0
152
235
  self._last_compaction_pressure = pressure
153
236
  self._reset_observed(ctx)
154
237
  return PreStepAction(
@@ -157,11 +240,23 @@ class CompactionHooks(NoopHooks):
157
240
  messages=compacted,
158
241
  reason="context pressure: folded old tool results",
159
242
  action="compact",
243
+ pre_tokens=pressure,
244
+ post_tokens=self._estimate(compacted),
160
245
  ),
161
246
  )
162
247
 
163
248
  compacted = await self._summarize_middle(compacted)
249
+ post = self._estimate(compacted)
164
250
  self.compactions += 1
251
+ if post < threshold:
252
+ self._consecutive_failures = 0
253
+ else:
254
+ # Ineffective: even fold+summarize stayed over threshold (e.g. a
255
+ # single kept tool result bigger than the window). Three in a
256
+ # row open the circuit.
257
+ self._consecutive_failures += 1
258
+ if self._consecutive_failures >= self.max_consecutive_failures:
259
+ self.circuit_open = True
165
260
  self._last_compaction_pressure = pressure
166
261
  self._reset_observed(ctx)
167
262
  return PreStepAction(
@@ -170,6 +265,36 @@ class CompactionHooks(NoopHooks):
170
265
  messages=compacted,
171
266
  reason="context pressure: summarized middle",
172
267
  action="compact",
268
+ pre_tokens=pressure,
269
+ post_tokens=post,
270
+ ),
271
+ )
272
+
273
+ async def compact_now(self, transcript: list[LLMMessage], ctx: Any) -> PreStepAction:
274
+ """Manual compaction (CC ``/compact`` parity): fold old tool results,
275
+ then summarize the middle, regardless of pressure, hysteresis, or the
276
+ circuit breaker — the user asked for it. Returns a ``proceed`` action
277
+ carrying the rewrite; when neither stage changes anything the action
278
+ carries no rewrite (nothing was worth invalidating the cache for).
279
+ A manual pass resets the breaker count: the user has taken over.
280
+ """
281
+ pre = self._estimate(transcript)
282
+ compacted = self._fold_old_tool_results(transcript)
283
+ compacted = await self._summarize_middle(compacted)
284
+ self._consecutive_failures = 0
285
+ if compacted is transcript or compacted == transcript:
286
+ return PreStepAction(kind="proceed", reason="compact: nothing to fold")
287
+ self.compactions += 1
288
+ self._last_compaction_pressure = pre
289
+ self._reset_observed(ctx)
290
+ return PreStepAction(
291
+ kind="proceed",
292
+ rewrite=RewriteRequest(
293
+ messages=compacted,
294
+ reason="manual compact",
295
+ action="compact",
296
+ pre_tokens=pre,
297
+ post_tokens=self._estimate(compacted),
173
298
  ),
174
299
  )
175
300
 
@@ -214,13 +339,22 @@ class CompactionHooks(NoopHooks):
214
339
  messages=compacted,
215
340
  reason="context overflow: compacted transcript before retry",
216
341
  action="overflow_recovery",
342
+ pre_tokens=self._estimate(transcript),
343
+ post_tokens=self._estimate(compacted),
217
344
  ),
218
345
  )
219
346
 
220
347
  # ------------------------------------------------------------------
221
348
 
222
349
  def _fold_old_tool_results(self, transcript: list[LLMMessage]) -> list[LLMMessage]:
223
- tool_idx = [i for i, m in enumerate(transcript) if m.role == "tool"]
350
+ # Already-folded results are excluded: re-folding them would stack
351
+ # markers and invalidate the prompt-cache prefix for zero gain, and
352
+ # ``keep_last_tool_results`` should count *readable* results.
353
+ tool_idx = [
354
+ i
355
+ for i, m in enumerate(transcript)
356
+ if m.role == "tool" and not m.content_text.startswith(_FOLDED_TOOL_MARKER)
357
+ ]
224
358
  fold_before = len(tool_idx) - self._keep_last_tools
225
359
  if fold_before <= 0:
226
360
  return transcript