steerable-agent-runtime 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/PKG-INFO +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/pyproject.toml +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/__init__.py +34 -0
- steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/ask_user.py +154 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/cache_control.py +102 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/compaction.py +136 -2
- steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/config.py +225 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/harness.py +185 -3
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/history.py +21 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/hooks.py +27 -4
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/anthropic_native.py +15 -8
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/compat.py +2 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/loop.py +103 -20
- steerable_agent_runtime-0.6.0/src/steerable_agent_runtime/plugins.py +67 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/reminders.py +19 -20
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/skills.py +5 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/subagent.py +130 -22
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/SOURCES.txt +6 -0
- steerable_agent_runtime-0.6.0/tests/test_ask_user.py +254 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_cache_control.py +103 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_compaction.py +244 -0
- steerable_agent_runtime-0.6.0/tests/test_config.py +217 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_harness.py +114 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_hooks.py +56 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_long_session.py +3 -3
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop.py +220 -0
- steerable_agent_runtime-0.6.0/tests/test_plugins.py +84 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_reminders.py +25 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_subagent.py +171 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/README.md +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/setup.cfg +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/antihallucination.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/approval.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/approval_policy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/branch.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/calibration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/default.harness.json +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/default.harness.yaml +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/errors.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/handoff.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/harness_spec.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/errors.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/openai_compat.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/parts.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/maintenance.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/mcp.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/mcp_server.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_catalog.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_info.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/model_resolve.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/observation_aging.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/orchestration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/otel.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/pricing.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/pseudo.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/recording.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/replay.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/resume.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/retry.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/sandboxed.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/spill.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tokens.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tool_schema.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tool_search.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tools.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/tracing.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime/world_state.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_antihallucination.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_approval.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_approval_policy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_branch.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_cache_instrumentation.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_calibration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_content_parts.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_error_taxonomy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_fragment_bounds.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_golden.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_handoff.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_harness_spec.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_history.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_history_persistence.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_in_memory_storage.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_llm_wire_helpers.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_cancellation.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_replay.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_loop_sandbox_event.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_maintenance.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_mcp.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_mcp_server.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_catalog.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_equivalence.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_info.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_model_resolve.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_observation_aging.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_orchestration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_otel.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_parallel_tools.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_provider_compat.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_pseudo.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_recording.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_replay_crosslang.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_resume.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_retry_hooks.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_safety_gate.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_sandboxed.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_skills.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_soft_timeout.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_spill.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_sqlite_storage.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_steer.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_storage_contract.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_stream_strip.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_system_proxy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tokens.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_exposure.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_hygiene.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_router.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_schema.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_tool_timeout.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_trace_recorder.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_transport_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_transport_sse.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_usage_attribution.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_world_state.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.6.0}/tests/test_write_lease.py +0 -0
|
@@ -29,6 +29,24 @@ from .approval_policy import (
|
|
|
29
29
|
PolicyApprover,
|
|
30
30
|
rule_from_amendment,
|
|
31
31
|
)
|
|
32
|
+
from .ask_user import (
|
|
33
|
+
ASK_USER_SCHEMA,
|
|
34
|
+
ASK_USER_TOOL_NAME,
|
|
35
|
+
AskUserHandler,
|
|
36
|
+
make_ask_user_tool,
|
|
37
|
+
)
|
|
38
|
+
from .config import (
|
|
39
|
+
DEFAULT_CONFIG_PATH,
|
|
40
|
+
ConfigError,
|
|
41
|
+
ResolvedConfig,
|
|
42
|
+
load_user_config,
|
|
43
|
+
resolve_config,
|
|
44
|
+
)
|
|
45
|
+
from .plugins import (
|
|
46
|
+
STEERABLE_TOOLS_ENTRY_POINT_GROUP,
|
|
47
|
+
PluginLoadError,
|
|
48
|
+
load_tool_entry_points,
|
|
49
|
+
)
|
|
32
50
|
from .branch import (
|
|
33
51
|
BranchPoint,
|
|
34
52
|
ForkResult,
|
|
@@ -39,6 +57,7 @@ from .branch import (
|
|
|
39
57
|
)
|
|
40
58
|
from .cache_control import (
|
|
41
59
|
CacheControlProvider,
|
|
60
|
+
CacheDriftMonitor,
|
|
42
61
|
CacheRetention,
|
|
43
62
|
place_cache_breakpoints,
|
|
44
63
|
system_blocks_with_cache,
|
|
@@ -195,6 +214,7 @@ from .subagent import (
|
|
|
195
214
|
FilteredToolsExecutor,
|
|
196
215
|
SubagentConfig,
|
|
197
216
|
SubagentExecutor,
|
|
217
|
+
SubagentRegistry,
|
|
198
218
|
subagent_tool_descriptor,
|
|
199
219
|
)
|
|
200
220
|
from .tokens import (
|
|
@@ -230,8 +250,13 @@ __all__ = [
|
|
|
230
250
|
"REASONING_EFFORT_ORDER",
|
|
231
251
|
"RECORD_FORMAT_VERSION",
|
|
232
252
|
"TOOL_SEARCH_NAME",
|
|
253
|
+
"ASK_USER_SCHEMA",
|
|
254
|
+
"ASK_USER_TOOL_NAME",
|
|
255
|
+
"STEERABLE_TOOLS_ENTRY_POINT_GROUP",
|
|
233
256
|
"AgentPool",
|
|
234
257
|
"AntiHallucinationConfig",
|
|
258
|
+
"AskUserHandler",
|
|
259
|
+
"PluginLoadError",
|
|
235
260
|
"AntiHallucinationHooks",
|
|
236
261
|
"ApprovalAborted",
|
|
237
262
|
"ApprovalDecision",
|
|
@@ -242,9 +267,13 @@ __all__ = [
|
|
|
242
267
|
"ApprovalStore",
|
|
243
268
|
"Approver",
|
|
244
269
|
"AutoApprover",
|
|
270
|
+
"ConfigError",
|
|
271
|
+
"DEFAULT_CONFIG_PATH",
|
|
272
|
+
"ResolvedConfig",
|
|
245
273
|
"BranchPoint",
|
|
246
274
|
"BudgetExhaustedError",
|
|
247
275
|
"CacheControlProvider",
|
|
276
|
+
"CacheDriftMonitor",
|
|
248
277
|
"CacheRetention",
|
|
249
278
|
"CalibratingProvider",
|
|
250
279
|
"ChainHooks",
|
|
@@ -328,6 +357,7 @@ __all__ = [
|
|
|
328
357
|
"StorageError",
|
|
329
358
|
"SubagentConfig",
|
|
330
359
|
"SubagentExecutor",
|
|
360
|
+
"SubagentRegistry",
|
|
331
361
|
"TextPart",
|
|
332
362
|
"ToolDispatchError",
|
|
333
363
|
"ToolExecutor",
|
|
@@ -367,8 +397,12 @@ __all__ = [
|
|
|
367
397
|
"last_world_state_snapshot",
|
|
368
398
|
"lineage",
|
|
369
399
|
"load_history_transcript",
|
|
400
|
+
"load_tool_entry_points",
|
|
401
|
+
"load_user_config",
|
|
402
|
+
"resolve_config",
|
|
370
403
|
"load_recorded_requests",
|
|
371
404
|
"load_transcript",
|
|
405
|
+
"make_ask_user_tool",
|
|
372
406
|
"matches_conditions",
|
|
373
407
|
"mcp_invoker",
|
|
374
408
|
"merge_patch",
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Structured user questions — the ``ask_user`` tool.
|
|
2
|
+
|
|
3
|
+
The agent pauses mid-run and asks the user a structured set of questions
|
|
4
|
+
(select / text / password) before continuing. The wire payload mirrors the
|
|
5
|
+
protocol's ``AskUserQuestionsPayload`` (TS) so the desktop card renders it
|
|
6
|
+
unchanged; the handler is the seam a product injects to actually collect the
|
|
7
|
+
answers (a UI card over the reverse channel, an ACP elicitation, a CLI
|
|
8
|
+
prompt).
|
|
9
|
+
|
|
10
|
+
The tool is blocking by construction: ``dispatch`` awaits the handler, and
|
|
11
|
+
the answers are returned as the tool result so they land in the durable
|
|
12
|
+
record and the model's next context. This mirrors dsh's ``ask_user_question``
|
|
13
|
+
— the loop does not continue until the user (or a timeout) answers.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from collections.abc import Awaitable, Callable
|
|
19
|
+
from typing import Any, Protocol
|
|
20
|
+
|
|
21
|
+
from .errors import ToolDispatchError
|
|
22
|
+
|
|
23
|
+
#: The tool name the model calls.
|
|
24
|
+
ASK_USER_TOOL_NAME = "ask_user"
|
|
25
|
+
|
|
26
|
+
#: JSON Schema for the tool's arguments — one intro plus a list of questions.
|
|
27
|
+
#: Field names match ``AskUserQuestionsPayload`` so the host can render the
|
|
28
|
+
#: arguments directly as the card payload.
|
|
29
|
+
ASK_USER_SCHEMA: dict[str, Any] = {
|
|
30
|
+
"type": "object",
|
|
31
|
+
"properties": {
|
|
32
|
+
"intro": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"description": "One-line framing shown above the questions.",
|
|
35
|
+
},
|
|
36
|
+
"outro": {
|
|
37
|
+
"type": "string",
|
|
38
|
+
"description": "Optional line shown below the questions.",
|
|
39
|
+
},
|
|
40
|
+
"questions": {
|
|
41
|
+
"type": "array",
|
|
42
|
+
"items": {
|
|
43
|
+
"type": "object",
|
|
44
|
+
"properties": {
|
|
45
|
+
"id": {"type": "string"},
|
|
46
|
+
"text": {"type": "string"},
|
|
47
|
+
"type": {
|
|
48
|
+
"type": "string",
|
|
49
|
+
"enum": ["select", "text", "password"],
|
|
50
|
+
},
|
|
51
|
+
"options": {"type": "array", "items": {"type": "string"}},
|
|
52
|
+
"placeholder": {"type": "string"},
|
|
53
|
+
"multiSelect": {"type": "boolean"},
|
|
54
|
+
},
|
|
55
|
+
"required": ["id", "text"],
|
|
56
|
+
},
|
|
57
|
+
},
|
|
58
|
+
},
|
|
59
|
+
"required": ["intro", "questions"],
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
#: Alias keys real models emit for the canonical ``AskUserQuestionsPayload``
|
|
63
|
+
#: fields (observed: gpt-oss via Ollama answers the schema with Inquirer-style
|
|
64
|
+
#: ``name``/``message``/``choices``). Normalized at this model-JSON boundary so
|
|
65
|
+
#: hosts only ever render the canonical payload; the canonical key wins when
|
|
66
|
+
#: both are present.
|
|
67
|
+
_QUESTION_ALIASES = {"id": "name", "text": "message", "options": "choices"}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
71
|
+
"""Map known alias keys onto the canonical payload fields and require the
|
|
72
|
+
two fields the host card cannot render without (``id`` to key the answer,
|
|
73
|
+
``text`` to label the control). A violation raises ``ToolDispatchError`` —
|
|
74
|
+
the router wraps it as the tool result, so the model sees exactly what to
|
|
75
|
+
fix and retries within the same turn."""
|
|
76
|
+
if not isinstance(questions, list):
|
|
77
|
+
raise ToolDispatchError(
|
|
78
|
+
f"ask_user: questions must be an array, got {type(questions).__name__}"
|
|
79
|
+
)
|
|
80
|
+
normalized: list[dict[str, Any]] = []
|
|
81
|
+
for index, question in enumerate(questions):
|
|
82
|
+
if not isinstance(question, dict):
|
|
83
|
+
raise ToolDispatchError(
|
|
84
|
+
f"ask_user: questions[{index}] must be an object, "
|
|
85
|
+
f"got {type(question).__name__}"
|
|
86
|
+
)
|
|
87
|
+
q = dict(question)
|
|
88
|
+
for canonical, alias in _QUESTION_ALIASES.items():
|
|
89
|
+
if canonical not in q and alias in q:
|
|
90
|
+
q[canonical] = q.pop(alias)
|
|
91
|
+
if not isinstance(q.get("id"), str) or not q["id"]:
|
|
92
|
+
raise ToolDispatchError(
|
|
93
|
+
f"ask_user: questions[{index}] is missing a non-empty string "
|
|
94
|
+
'"id" (the answer-map key).'
|
|
95
|
+
)
|
|
96
|
+
if not isinstance(q.get("text"), str) or not q["text"]:
|
|
97
|
+
raise ToolDispatchError(
|
|
98
|
+
f"ask_user: questions[{index}] is missing a non-empty string "
|
|
99
|
+
'"text" (the question label shown to the user).'
|
|
100
|
+
)
|
|
101
|
+
normalized.append(q)
|
|
102
|
+
return normalized
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class AskUserHandler(Protocol):
|
|
106
|
+
"""Product-injected seam that collects answers for one question set.
|
|
107
|
+
|
|
108
|
+
Receives the validated ``intro`` and ``questions`` list and returns a
|
|
109
|
+
mapping of question id → answer (a string, or a list of strings for a
|
|
110
|
+
multi-select). Implementations must not raise on a user cancel — return
|
|
111
|
+
an empty mapping so the loop records "no answer" and moves on.
|
|
112
|
+
"""
|
|
113
|
+
|
|
114
|
+
def __call__(
|
|
115
|
+
self, intro: str, questions: list[dict[str, Any]]
|
|
116
|
+
) -> Awaitable[dict[str, Any]]: ...
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def make_ask_user_tool(handler: AskUserHandler) -> Callable[..., Awaitable[dict[str, Any]]]:
|
|
120
|
+
"""Build the ``ask_user`` tool handler bound to ``handler``.
|
|
121
|
+
|
|
122
|
+
The returned coroutine function carries the registration metadata the
|
|
123
|
+
router reads (name, description, schema) so a host registers it with one
|
|
124
|
+
call. The tool is marked ``require_consent=False`` and
|
|
125
|
+
``concurrency_safe=False``: asking the user is itself the interaction, and
|
|
126
|
+
two concurrent question cards would race for the same UI surface.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
async def ask_user(
|
|
130
|
+
intro: str, questions: list[dict[str, Any]], outro: str | None = None
|
|
131
|
+
) -> dict[str, Any]:
|
|
132
|
+
normalized = _normalize_questions(questions)
|
|
133
|
+
answers = await handler(intro, normalized)
|
|
134
|
+
return {
|
|
135
|
+
"intro": intro,
|
|
136
|
+
"outro": outro,
|
|
137
|
+
"questions": normalized,
|
|
138
|
+
"answers": answers,
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
ask_user.__steerable_tool_meta__ = { # type: ignore[attr-defined]
|
|
142
|
+
"name": ASK_USER_TOOL_NAME,
|
|
143
|
+
"mode": "read", # no side effects on the workspace; it gathers input
|
|
144
|
+
"description": (
|
|
145
|
+
"Pause and ask the user a structured set of questions (select / "
|
|
146
|
+
"text / password) when you need information only they have. "
|
|
147
|
+
"Blocks until they answer; the answers come back as the result."
|
|
148
|
+
),
|
|
149
|
+
"schema": ASK_USER_SCHEMA,
|
|
150
|
+
"require_consent": False,
|
|
151
|
+
"concurrency_safe": False,
|
|
152
|
+
"exposure": "direct",
|
|
153
|
+
}
|
|
154
|
+
return ask_user
|
|
@@ -23,12 +23,16 @@ comes from the prefix stability the rest of the stack already keeps.
|
|
|
23
23
|
|
|
24
24
|
from __future__ import annotations
|
|
25
25
|
|
|
26
|
+
import logging
|
|
26
27
|
from collections.abc import AsyncIterator, Sequence
|
|
27
28
|
from dataclasses import dataclass
|
|
28
29
|
from typing import Any, Iterable, Literal
|
|
29
30
|
|
|
31
|
+
from .hooks import NoopHooks, PreStepAction
|
|
30
32
|
from .llm import LLMMessage, LLMProvider, LLMStreamChunk, LLMUsage
|
|
31
33
|
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
32
36
|
CacheRetention = Literal["none", "short", "long"]
|
|
33
37
|
|
|
34
38
|
#: Anthropic's explicit breakpoint API. Other providers have no breakpoint
|
|
@@ -36,6 +40,19 @@ CacheRetention = Literal["none", "short", "long"]
|
|
|
36
40
|
_EXPLICIT_CACHE_PROVIDERS = {"anthropic", "claude"}
|
|
37
41
|
|
|
38
42
|
|
|
43
|
+
def _marker_for(retention: CacheRetention) -> dict[str, Any]:
|
|
44
|
+
"""Map a retention class onto the Anthropic breakpoint marker.
|
|
45
|
+
|
|
46
|
+
``short`` is the 5-minute default; ``long`` opts into the 1-hour TTL
|
|
47
|
+
(CC ``CLAUDE_CODE_PROMPT_CACHE_TTL`` parity — a deliberate trade: fewer
|
|
48
|
+
cache writes on slow-moving prefixes, at Anthropic's higher 1h write
|
|
49
|
+
price). ``none`` never reaches here (the request skips anchoring).
|
|
50
|
+
"""
|
|
51
|
+
if retention == "long":
|
|
52
|
+
return {"type": "ephemeral", "ttl": "1h"}
|
|
53
|
+
return {"type": "ephemeral"}
|
|
54
|
+
|
|
55
|
+
|
|
39
56
|
@dataclass(slots=True)
|
|
40
57
|
class CacheControlProvider:
|
|
41
58
|
"""LLMProvider decorator: emits prompt-cache breakpoints per request.
|
|
@@ -105,10 +122,15 @@ class CacheControlProvider:
|
|
|
105
122
|
kwargs.pop("cache_retention", None)
|
|
106
123
|
return tools, kwargs
|
|
107
124
|
out = dict(kwargs)
|
|
125
|
+
marker = _marker_for(retention)
|
|
108
126
|
shaped_tools = (
|
|
109
|
-
place_cache_breakpoints(list(tools))
|
|
127
|
+
place_cache_breakpoints(list(tools), cache_control=marker)
|
|
128
|
+
if tools is not None
|
|
129
|
+
else None
|
|
110
130
|
)
|
|
111
|
-
|
|
131
|
+
# The provider-owned anchors (system block, transcript tail) read the
|
|
132
|
+
# marker off this key so all three breakpoints share one TTL.
|
|
133
|
+
out["_cache_tail_anchor"] = marker
|
|
112
134
|
return shaped_tools, out
|
|
113
135
|
|
|
114
136
|
def _effective_retention(self, kwargs: dict[str, Any]) -> CacheRetention:
|
|
@@ -152,3 +174,81 @@ def system_blocks_with_cache(
|
|
|
152
174
|
"""
|
|
153
175
|
marker = cache_control or {"type": "ephemeral"}
|
|
154
176
|
return [{"type": "text", "text": system_text, "cache_control": marker}]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class CacheDriftMonitor(NoopHooks):
|
|
180
|
+
"""``pre_step`` observer: flag prompt-cache drift from provider usage.
|
|
181
|
+
|
|
182
|
+
Each round it reads the PREVIOUS request's accounting off the loop ctx
|
|
183
|
+
(``last_prompt_tokens`` / ``last_cached_prompt_tokens`` — the read side
|
|
184
|
+
surfaced on ``stage_complete``). The monitor arms once a round shows the
|
|
185
|
+
cache actually serving tokens; from then on, ``consecutive_rounds``
|
|
186
|
+
rounds whose hit rate stays below ``min_hit_rate`` (while the prompt is
|
|
187
|
+
at least ``min_prompt_tokens`` — below that, caching barely matters) set
|
|
188
|
+
``drift_detected`` and log a warning with the numbers. A round at or
|
|
189
|
+
above the rate resets the count and clears the flag.
|
|
190
|
+
|
|
191
|
+
This is the deliberate loop logic behind the raw ``stage_complete``
|
|
192
|
+
numbers (CC ``globalCacheStrategy`` / ``cacheControlHash`` parity): a
|
|
193
|
+
collapse means the cached prefix broke — a transcript rewrite, a changed
|
|
194
|
+
tool list, a TTL expiry — and one low round is normal right after a
|
|
195
|
+
compaction (the prefix re-warms on the next request), which is why the
|
|
196
|
+
verdict requires consecutive low rounds. Providers without cache
|
|
197
|
+
accounting never arm the monitor, so implicit-cache deployments that
|
|
198
|
+
report nothing see no false drift.
|
|
199
|
+
"""
|
|
200
|
+
|
|
201
|
+
def __init__(
|
|
202
|
+
self,
|
|
203
|
+
*,
|
|
204
|
+
min_prompt_tokens: int = 1024,
|
|
205
|
+
min_hit_rate: float = 0.5,
|
|
206
|
+
consecutive_rounds: int = 3,
|
|
207
|
+
) -> None:
|
|
208
|
+
if not 0 < min_hit_rate <= 1:
|
|
209
|
+
raise ValueError("min_hit_rate must be in (0, 1]")
|
|
210
|
+
if not 1 <= consecutive_rounds:
|
|
211
|
+
raise ValueError("consecutive_rounds must be >= 1")
|
|
212
|
+
self._min_prompt = min_prompt_tokens
|
|
213
|
+
self._min_hit_rate = min_hit_rate
|
|
214
|
+
self._consecutive_rounds = consecutive_rounds
|
|
215
|
+
# Observability for callers/tests (CompactionHooks-counter pattern).
|
|
216
|
+
self.armed = False
|
|
217
|
+
self.drift_detected = False
|
|
218
|
+
self.consecutive_low_hit_rounds = 0
|
|
219
|
+
self.last_hit_rate: float | None = None
|
|
220
|
+
|
|
221
|
+
async def pre_step(
|
|
222
|
+
self, transcript: list[LLMMessage], ctx: Any
|
|
223
|
+
) -> PreStepAction:
|
|
224
|
+
prompt = getattr(ctx, "last_prompt_tokens", None)
|
|
225
|
+
cached = getattr(ctx, "last_cached_prompt_tokens", None)
|
|
226
|
+
if not prompt or cached is None or prompt < self._min_prompt:
|
|
227
|
+
return PreStepAction(kind="proceed")
|
|
228
|
+
hit_rate = cached / prompt
|
|
229
|
+
self.last_hit_rate = hit_rate
|
|
230
|
+
if cached > 0:
|
|
231
|
+
self.armed = True
|
|
232
|
+
if not self.armed:
|
|
233
|
+
return PreStepAction(kind="proceed")
|
|
234
|
+
if hit_rate >= self._min_hit_rate:
|
|
235
|
+
self.consecutive_low_hit_rounds = 0
|
|
236
|
+
self.drift_detected = False
|
|
237
|
+
return PreStepAction(kind="proceed")
|
|
238
|
+
self.consecutive_low_hit_rounds += 1
|
|
239
|
+
if (
|
|
240
|
+
self.consecutive_low_hit_rounds >= self._consecutive_rounds
|
|
241
|
+
and not self.drift_detected
|
|
242
|
+
):
|
|
243
|
+
self.drift_detected = True
|
|
244
|
+
logger.warning(
|
|
245
|
+
"prompt-cache drift: hit rate %.2f below %.2f for %d "
|
|
246
|
+
"consecutive rounds (prompt=%d cached=%d) — the cached "
|
|
247
|
+
"prefix broke (rewrite, tool-list change, or TTL expiry)",
|
|
248
|
+
hit_rate,
|
|
249
|
+
self._min_hit_rate,
|
|
250
|
+
self.consecutive_low_hit_rounds,
|
|
251
|
+
prompt,
|
|
252
|
+
cached,
|
|
253
|
+
)
|
|
254
|
+
return PreStepAction(kind="proceed")
|
|
@@ -25,6 +25,36 @@ System messages and the first user message (the goal) are always kept; the
|
|
|
25
25
|
most recent ``keep_last_messages`` are never touched. Between compactions
|
|
26
26
|
the transcript is append-only, so provider prompt caches keep hitting; a
|
|
27
27
|
rewrite invalidates the cache once, then the prefix is stable again.
|
|
28
|
+
|
|
29
|
+
Three trigger paths share the fold/summarize machinery:
|
|
30
|
+
|
|
31
|
+
- **pressure** (``pre_step``, above) — the reactive default;
|
|
32
|
+
- **overflow recovery** (``on_request_error``) — the heuristic threshold can
|
|
33
|
+
still miss the real window, so a context-overflow error forces one
|
|
34
|
+
compaction pass and retries, bounded per round;
|
|
35
|
+
- **micro-compaction** (``pre_step``, opt-in ``micro_compact_interval_rounds``)
|
|
36
|
+
— every N rounds, fold old tool results regardless of pressure (CC
|
|
37
|
+
time-based microcompact parity). Off by default: each fold invalidates the
|
|
38
|
+
prompt-cache prefix, so the interval trades cache hits for a bounded
|
|
39
|
+
transcript;
|
|
40
|
+
- **manual** (``compact_now``) — the host-command path (CC ``/compact``
|
|
41
|
+
parity): fold + summarize on demand, bypassing threshold, hysteresis, and
|
|
42
|
+
the circuit breaker.
|
|
43
|
+
|
|
44
|
+
Two safety rails share the machinery:
|
|
45
|
+
|
|
46
|
+
- Every rewrite carries its ``pre_tokens`` / ``post_tokens`` estimate onto
|
|
47
|
+
the recorded ``CompactionBoundary`` (CC ``compact_boundary`` parity), so
|
|
48
|
+
traces chart compaction effectiveness without re-estimating from bodies.
|
|
49
|
+
- A **circuit breaker** (CC auto-compact breaker parity): a pressure
|
|
50
|
+
compaction whose post-estimate is still over threshold counts as
|
|
51
|
+
ineffective; three consecutive ineffective compactions open the circuit
|
|
52
|
+
and the pressure path stops firing — each further rewrite would only
|
|
53
|
+
invalidate the prompt-cache prefix without shrinking the transcript. The
|
|
54
|
+
overflow-recovery path keeps its own per-round bound and stays live, so
|
|
55
|
+
the turn still fails loud (bounded retries, then a named error) instead
|
|
56
|
+
of spinning. A round that lands under threshold — or any successful
|
|
57
|
+
compaction — resets the count.
|
|
28
58
|
"""
|
|
29
59
|
|
|
30
60
|
from __future__ import annotations
|
|
@@ -75,17 +105,26 @@ class CompactionHooks(NoopHooks):
|
|
|
75
105
|
model: str | None = None,
|
|
76
106
|
recompact_margin_ratio: float = 0.1,
|
|
77
107
|
fold_excerpt_chars: int = _FOLD_EXCERPT_CHARS,
|
|
108
|
+
micro_compact_interval_rounds: int = 0,
|
|
78
109
|
) -> None:
|
|
79
110
|
if not 0 < threshold_ratio <= 1:
|
|
80
111
|
raise ValueError("threshold_ratio must be in (0, 1]")
|
|
81
112
|
if not 0 <= recompact_margin_ratio:
|
|
82
113
|
raise ValueError("recompact_margin_ratio must be >= 0")
|
|
114
|
+
if not 0 <= micro_compact_interval_rounds:
|
|
115
|
+
raise ValueError("micro_compact_interval_rounds must be >= 0")
|
|
83
116
|
self._max_tokens = max_context_tokens
|
|
84
117
|
self._threshold = threshold_ratio
|
|
85
118
|
self._keep_last = keep_last_messages
|
|
86
119
|
self._keep_last_tools = keep_last_tool_results
|
|
87
120
|
self._fold_excerpt_chars = fold_excerpt_chars
|
|
88
121
|
self._summarizer = summarizer
|
|
122
|
+
#: Proactive micro-compaction (CC time-based microcompact parity):
|
|
123
|
+
#: every N rounds, fold old tool results regardless of pressure, so a
|
|
124
|
+
#: long chatty run never drifts toward the window. 0 (default) is off
|
|
125
|
+
#: — each fold invalidates the provider prompt-cache prefix, so the
|
|
126
|
+
#: interval trades cache hits for a bounded transcript.
|
|
127
|
+
self._micro_compact_interval = micro_compact_interval_rounds
|
|
89
128
|
#: Model name used for calibrated token estimates (see tokens.py).
|
|
90
129
|
self._model = model
|
|
91
130
|
#: Hysteresis: after a compaction, pressure must grow by
|
|
@@ -98,6 +137,8 @@ class CompactionHooks(NoopHooks):
|
|
|
98
137
|
self._last_compaction_pressure: int | None = None
|
|
99
138
|
# Observability for callers/tests: how many compactions happened.
|
|
100
139
|
self.compactions = 0
|
|
140
|
+
# Observability: how many of them were periodic micro-compactions.
|
|
141
|
+
self.micro_compactions = 0
|
|
101
142
|
# Overflow-recovery state: consecutive context-overflow retries within
|
|
102
143
|
# one round. Capped so a pathological transcript cannot spin a
|
|
103
144
|
# compact→overflow→compact loop forever.
|
|
@@ -107,6 +148,14 @@ class CompactionHooks(NoopHooks):
|
|
|
107
148
|
self.max_overflow_retries = 2
|
|
108
149
|
# Observability: how many overflow-driven compactions happened.
|
|
109
150
|
self.overflow_recoveries = 0
|
|
151
|
+
#: Consecutive ineffective pressure compactions (post-estimate still
|
|
152
|
+
#: over threshold) before the circuit opens (CC auto-compact breaker
|
|
153
|
+
#: parity). The overflow path is bounded separately and unaffected.
|
|
154
|
+
self.max_consecutive_failures = 3
|
|
155
|
+
self._consecutive_failures = 0
|
|
156
|
+
# Observability: True once the breaker tripped; the pressure path
|
|
157
|
+
# stops firing for the rest of the session.
|
|
158
|
+
self.circuit_open = False
|
|
110
159
|
|
|
111
160
|
def _estimate(self, transcript: Sequence[LLMMessage]) -> int:
|
|
112
161
|
return estimate_tokens(transcript, model=self._model)
|
|
@@ -137,8 +186,40 @@ class CompactionHooks(NoopHooks):
|
|
|
137
186
|
async def pre_step(
|
|
138
187
|
self, transcript: list[LLMMessage], ctx: Any
|
|
139
188
|
) -> PreStepAction:
|
|
189
|
+
round_index = getattr(ctx, "round_index", 0)
|
|
190
|
+
if (
|
|
191
|
+
self._micro_compact_interval > 0
|
|
192
|
+
and round_index > 0
|
|
193
|
+
and round_index % self._micro_compact_interval == 0
|
|
194
|
+
):
|
|
195
|
+
pruned = self._fold_old_tool_results(transcript)
|
|
196
|
+
if pruned is not transcript:
|
|
197
|
+
self.compactions += 1
|
|
198
|
+
self.micro_compactions += 1
|
|
199
|
+
self._reset_observed(ctx)
|
|
200
|
+
return PreStepAction(
|
|
201
|
+
kind="proceed",
|
|
202
|
+
rewrite=RewriteRequest(
|
|
203
|
+
messages=pruned,
|
|
204
|
+
reason=(
|
|
205
|
+
"micro-compact: periodic tool-result prune "
|
|
206
|
+
f"(every {self._micro_compact_interval} rounds)"
|
|
207
|
+
),
|
|
208
|
+
action="micro_compact",
|
|
209
|
+
pre_tokens=self._estimate(transcript),
|
|
210
|
+
post_tokens=self._estimate(pruned),
|
|
211
|
+
),
|
|
212
|
+
)
|
|
140
213
|
pressure = self._pressure(transcript, ctx)
|
|
141
214
|
if pressure < self._threshold * self._max_tokens:
|
|
215
|
+
# A healthy round resets the breaker count — the pathology (or
|
|
216
|
+
# the transcript that caused it) is gone.
|
|
217
|
+
self._consecutive_failures = 0
|
|
218
|
+
return PreStepAction(kind="proceed")
|
|
219
|
+
if self.circuit_open:
|
|
220
|
+
# Breaker tripped: further pressure rewrites only invalidate the
|
|
221
|
+
# prompt-cache prefix without shrinking the transcript. The
|
|
222
|
+
# overflow path stays live and fails loud if the window is hit.
|
|
142
223
|
return PreStepAction(kind="proceed")
|
|
143
224
|
if (
|
|
144
225
|
self._last_compaction_pressure is not None
|
|
@@ -146,9 +227,11 @@ class CompactionHooks(NoopHooks):
|
|
|
146
227
|
):
|
|
147
228
|
return PreStepAction(kind="proceed")
|
|
148
229
|
|
|
230
|
+
threshold = self._threshold * self._max_tokens
|
|
149
231
|
compacted = self._fold_old_tool_results(transcript)
|
|
150
|
-
if self._estimate(compacted) <
|
|
232
|
+
if self._estimate(compacted) < threshold:
|
|
151
233
|
self.compactions += 1
|
|
234
|
+
self._consecutive_failures = 0
|
|
152
235
|
self._last_compaction_pressure = pressure
|
|
153
236
|
self._reset_observed(ctx)
|
|
154
237
|
return PreStepAction(
|
|
@@ -157,11 +240,23 @@ class CompactionHooks(NoopHooks):
|
|
|
157
240
|
messages=compacted,
|
|
158
241
|
reason="context pressure: folded old tool results",
|
|
159
242
|
action="compact",
|
|
243
|
+
pre_tokens=pressure,
|
|
244
|
+
post_tokens=self._estimate(compacted),
|
|
160
245
|
),
|
|
161
246
|
)
|
|
162
247
|
|
|
163
248
|
compacted = await self._summarize_middle(compacted)
|
|
249
|
+
post = self._estimate(compacted)
|
|
164
250
|
self.compactions += 1
|
|
251
|
+
if post < threshold:
|
|
252
|
+
self._consecutive_failures = 0
|
|
253
|
+
else:
|
|
254
|
+
# Ineffective: even fold+summarize stayed over threshold (e.g. a
|
|
255
|
+
# single kept tool result bigger than the window). Three in a
|
|
256
|
+
# row open the circuit.
|
|
257
|
+
self._consecutive_failures += 1
|
|
258
|
+
if self._consecutive_failures >= self.max_consecutive_failures:
|
|
259
|
+
self.circuit_open = True
|
|
165
260
|
self._last_compaction_pressure = pressure
|
|
166
261
|
self._reset_observed(ctx)
|
|
167
262
|
return PreStepAction(
|
|
@@ -170,6 +265,36 @@ class CompactionHooks(NoopHooks):
|
|
|
170
265
|
messages=compacted,
|
|
171
266
|
reason="context pressure: summarized middle",
|
|
172
267
|
action="compact",
|
|
268
|
+
pre_tokens=pressure,
|
|
269
|
+
post_tokens=post,
|
|
270
|
+
),
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
async def compact_now(self, transcript: list[LLMMessage], ctx: Any) -> PreStepAction:
|
|
274
|
+
"""Manual compaction (CC ``/compact`` parity): fold old tool results,
|
|
275
|
+
then summarize the middle, regardless of pressure, hysteresis, or the
|
|
276
|
+
circuit breaker — the user asked for it. Returns a ``proceed`` action
|
|
277
|
+
carrying the rewrite; when neither stage changes anything the action
|
|
278
|
+
carries no rewrite (nothing was worth invalidating the cache for).
|
|
279
|
+
A manual pass resets the breaker count: the user has taken over.
|
|
280
|
+
"""
|
|
281
|
+
pre = self._estimate(transcript)
|
|
282
|
+
compacted = self._fold_old_tool_results(transcript)
|
|
283
|
+
compacted = await self._summarize_middle(compacted)
|
|
284
|
+
self._consecutive_failures = 0
|
|
285
|
+
if compacted is transcript or compacted == transcript:
|
|
286
|
+
return PreStepAction(kind="proceed", reason="compact: nothing to fold")
|
|
287
|
+
self.compactions += 1
|
|
288
|
+
self._last_compaction_pressure = pre
|
|
289
|
+
self._reset_observed(ctx)
|
|
290
|
+
return PreStepAction(
|
|
291
|
+
kind="proceed",
|
|
292
|
+
rewrite=RewriteRequest(
|
|
293
|
+
messages=compacted,
|
|
294
|
+
reason="manual compact",
|
|
295
|
+
action="compact",
|
|
296
|
+
pre_tokens=pre,
|
|
297
|
+
post_tokens=self._estimate(compacted),
|
|
173
298
|
),
|
|
174
299
|
)
|
|
175
300
|
|
|
@@ -214,13 +339,22 @@ class CompactionHooks(NoopHooks):
|
|
|
214
339
|
messages=compacted,
|
|
215
340
|
reason="context overflow: compacted transcript before retry",
|
|
216
341
|
action="overflow_recovery",
|
|
342
|
+
pre_tokens=self._estimate(transcript),
|
|
343
|
+
post_tokens=self._estimate(compacted),
|
|
217
344
|
),
|
|
218
345
|
)
|
|
219
346
|
|
|
220
347
|
# ------------------------------------------------------------------
|
|
221
348
|
|
|
222
349
|
def _fold_old_tool_results(self, transcript: list[LLMMessage]) -> list[LLMMessage]:
|
|
223
|
-
|
|
350
|
+
# Already-folded results are excluded: re-folding them would stack
|
|
351
|
+
# markers and invalidate the prompt-cache prefix for zero gain, and
|
|
352
|
+
# ``keep_last_tool_results`` should count *readable* results.
|
|
353
|
+
tool_idx = [
|
|
354
|
+
i
|
|
355
|
+
for i, m in enumerate(transcript)
|
|
356
|
+
if m.role == "tool" and not m.content_text.startswith(_FOLDED_TOOL_MARKER)
|
|
357
|
+
]
|
|
224
358
|
fold_before = len(tool_idx) - self._keep_last_tools
|
|
225
359
|
if fold_before <= 0:
|
|
226
360
|
return transcript
|