steerable-agent-runtime 0.6.2__tar.gz → 0.6.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/pyproject.toml +1 -1
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/__init__.py +20 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/ask_user.py +101 -2
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/branch.py +125 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/compaction.py +4 -2
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/harness.py +39 -5
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/hooks.py +26 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/__init__.py +14 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/openai_compat.py +39 -6
- steerable_agent_runtime-0.6.3/src/steerable_agent_runtime/llm/presets.py +259 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/loop.py +61 -1
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/orchestration.py +26 -309
- steerable_agent_runtime-0.6.3/src/steerable_agent_runtime/plugins.py +487 -0
- steerable_agent_runtime-0.6.3/src/steerable_agent_runtime/pool.py +384 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/subagent.py +130 -36
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime.egg-info/SOURCES.txt +3 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_ask_user.py +144 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_branch.py +100 -2
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_compaction.py +72 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_harness.py +19 -2
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_harness_spec.py +2 -2
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_hooks.py +44 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_loop.py +31 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_orchestration.py +52 -0
- steerable_agent_runtime-0.6.3/tests/test_plugins.py +400 -0
- steerable_agent_runtime-0.6.3/tests/test_provider_presets.py +254 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_subagent.py +326 -1
- steerable_agent_runtime-0.6.2/src/steerable_agent_runtime/plugins.py +0 -67
- steerable_agent_runtime-0.6.2/tests/test_plugins.py +0 -84
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/README.md +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/setup.cfg +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/approval.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/cache_control.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/calibration.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/config.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/default.harness.json +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/default.harness.yaml +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/errors.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/handoff.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/harness_spec.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/history.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/compat.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/errors.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/parts.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/maintenance.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/mcp.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/model_catalog.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/model_info.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/model_resolve.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/otel.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/pricing.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/pseudo.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/recording.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/reminders.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/replay.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/resume.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/retry.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/skills.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/spill.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/storage/__init__.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/tokens.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/tool_search.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/tools.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/tracing.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/transport/__init__.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/world_state.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_antihallucination.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_approval.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_approval_policy.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_cache_control.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_cache_instrumentation.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_calibration.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_config.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_content_parts.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_error_taxonomy.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_fragment_bounds.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_golden.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_handoff.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_history.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_history_persistence.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_in_memory_storage.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_llm_wire_helpers.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_long_session.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_loop_cancellation.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_loop_replay.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_loop_sandbox_event.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_maintenance.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_mcp.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_mcp_server.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_model_catalog.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_model_equivalence.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_model_info.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_model_resolve.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_observation_aging.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_otel.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_parallel_tools.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_provider_compat.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_pseudo.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_recording.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_reminders.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_replay_crosslang.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_resume.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_retry_hooks.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_safety_gate.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_sandboxed.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_skills.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_soft_timeout.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_spill.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_sqlite_storage.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_steer.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_storage_contract.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_stream_strip.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_system_proxy.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tokens.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tool_exposure.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tool_hygiene.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tool_router.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tool_schema.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_tool_timeout.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_trace_recorder.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_transport_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_transport_sse.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_usage_attribution.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_world_state.py +0 -0
- {steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/tests/test_write_lease.py +0 -0
|
@@ -44,13 +44,23 @@ from .config import (
|
|
|
44
44
|
)
|
|
45
45
|
from .plugins import (
|
|
46
46
|
STEERABLE_TOOLS_ENTRY_POINT_GROUP,
|
|
47
|
+
DirectorySource,
|
|
48
|
+
EntryPointSource,
|
|
47
49
|
PluginLoadError,
|
|
50
|
+
PluginRecord,
|
|
51
|
+
PluginRegistry,
|
|
52
|
+
PluginSource,
|
|
53
|
+
PluginSpec,
|
|
54
|
+
PluginStateError,
|
|
48
55
|
load_tool_entry_points,
|
|
49
56
|
)
|
|
50
57
|
from .branch import (
|
|
51
58
|
BranchPoint,
|
|
59
|
+
FamilyTree,
|
|
60
|
+
FamilyTreeNode,
|
|
52
61
|
ForkResult,
|
|
53
62
|
branch_label,
|
|
63
|
+
family_tree,
|
|
54
64
|
fork_record,
|
|
55
65
|
lineage,
|
|
56
66
|
resolve_fork_seq,
|
|
@@ -256,7 +266,14 @@ __all__ = [
|
|
|
256
266
|
"AgentPool",
|
|
257
267
|
"AntiHallucinationConfig",
|
|
258
268
|
"AskUserHandler",
|
|
269
|
+
"DirectorySource",
|
|
270
|
+
"EntryPointSource",
|
|
259
271
|
"PluginLoadError",
|
|
272
|
+
"PluginRecord",
|
|
273
|
+
"PluginRegistry",
|
|
274
|
+
"PluginSource",
|
|
275
|
+
"PluginSpec",
|
|
276
|
+
"PluginStateError",
|
|
260
277
|
"AntiHallucinationHooks",
|
|
261
278
|
"ApprovalAborted",
|
|
262
279
|
"ApprovalDecision",
|
|
@@ -288,6 +305,8 @@ __all__ = [
|
|
|
288
305
|
"ContextManager",
|
|
289
306
|
"CoreLoop",
|
|
290
307
|
"ExecutionBudget",
|
|
308
|
+
"FamilyTree",
|
|
309
|
+
"FamilyTreeNode",
|
|
291
310
|
"FilesystemSkillProvider",
|
|
292
311
|
"FilesystemSpillStore",
|
|
293
312
|
"FilteredToolsExecutor",
|
|
@@ -393,6 +412,7 @@ __all__ = [
|
|
|
393
412
|
"export_trace",
|
|
394
413
|
"extract_inline_tool_calls",
|
|
395
414
|
"factor_for_model",
|
|
415
|
+
"family_tree",
|
|
396
416
|
"fork_record",
|
|
397
417
|
"last_world_state_snapshot",
|
|
398
418
|
"lineage",
|
|
@@ -39,16 +39,37 @@ ASK_USER_SCHEMA: dict[str, Any] = {
|
|
|
39
39
|
},
|
|
40
40
|
"questions": {
|
|
41
41
|
"type": "array",
|
|
42
|
+
"minItems": 1,
|
|
43
|
+
"maxItems": 4,
|
|
42
44
|
"items": {
|
|
43
45
|
"type": "object",
|
|
44
46
|
"properties": {
|
|
45
47
|
"id": {"type": "string"},
|
|
46
48
|
"text": {"type": "string"},
|
|
49
|
+
"header": {
|
|
50
|
+
"type": "string",
|
|
51
|
+
"maxLength": 12,
|
|
52
|
+
"description": (
|
|
53
|
+
"Short chip label shown above the question "
|
|
54
|
+
"(<=12 chars). Optional; the host derives one "
|
|
55
|
+
"from `text` when absent."
|
|
56
|
+
),
|
|
57
|
+
},
|
|
47
58
|
"type": {
|
|
48
59
|
"type": "string",
|
|
49
60
|
"enum": ["select", "text", "password"],
|
|
50
61
|
},
|
|
51
|
-
"options": {
|
|
62
|
+
"options": {
|
|
63
|
+
"type": "array",
|
|
64
|
+
"items": {"type": "string"},
|
|
65
|
+
"minItems": 2,
|
|
66
|
+
"maxItems": 4,
|
|
67
|
+
"description": (
|
|
68
|
+
"Choices for a select question (2-4). The host "
|
|
69
|
+
"auto-appends an 'Other' free-text escape; do "
|
|
70
|
+
"not list it yourself."
|
|
71
|
+
),
|
|
72
|
+
},
|
|
52
73
|
"placeholder": {"type": "string"},
|
|
53
74
|
"multiSelect": {"type": "boolean"},
|
|
54
75
|
},
|
|
@@ -67,8 +88,40 @@ ASK_USER_SCHEMA: dict[str, Any] = {
|
|
|
67
88
|
_QUESTION_ALIASES = {"id": "name", "text": "message", "options": "choices"}
|
|
68
89
|
|
|
69
90
|
|
|
91
|
+
#: Claude Code ``AskUserQuestion`` parity bounds. The card renders a batched
|
|
92
|
+
#: decision, not an interview — cap the question count so the model cannot
|
|
93
|
+
#: bombard the user, and cap select options so it pre-categorizes instead of
|
|
94
|
+
#: dumping a long list. The host auto-appends an "Other" free-text escape, so
|
|
95
|
+
#: the model never lists it (the option space the model thought of is not the
|
|
96
|
+
#: complete space).
|
|
97
|
+
_MIN_QUESTIONS = 1
|
|
98
|
+
_MAX_QUESTIONS = 4
|
|
99
|
+
_MIN_OPTIONS = 2
|
|
100
|
+
_MAX_OPTIONS = 4
|
|
101
|
+
#: ``header`` is the short chip label shown above the question (CC parity).
|
|
102
|
+
#: Optional on the wire; derived from ``text`` when absent.
|
|
103
|
+
_MAX_HEADER_LEN = 12
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _derive_header(text: str) -> str:
|
|
107
|
+
"""Derive a <=12-char chip label from the question text when the model did
|
|
108
|
+
not supply ``header``. Strips trailing punctuation and truncates on a word
|
|
109
|
+
boundary so the chip stays readable."""
|
|
110
|
+
cleaned = text.strip().rstrip("?.!。")
|
|
111
|
+
if len(cleaned) <= _MAX_HEADER_LEN:
|
|
112
|
+
return cleaned
|
|
113
|
+
truncated = cleaned[:_MAX_HEADER_LEN]
|
|
114
|
+
# Prefer cutting on the last space so the chip does not end mid-word.
|
|
115
|
+
space = truncated.rfind(" ")
|
|
116
|
+
if space > 0:
|
|
117
|
+
truncated = truncated[:space]
|
|
118
|
+
return truncated
|
|
119
|
+
|
|
120
|
+
|
|
70
121
|
def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
71
|
-
"""Map known alias keys onto the canonical payload fields
|
|
122
|
+
"""Map known alias keys onto the canonical payload fields, enforce the
|
|
123
|
+
Claude Code ``AskUserQuestion`` bounds (1-4 questions, 2-4 options per
|
|
124
|
+
select, ``header`` <=12 chars, ``multiSelect`` explicit), and require the
|
|
72
125
|
two fields the host card cannot render without (``id`` to key the answer,
|
|
73
126
|
``text`` to label the control). A violation raises ``ToolDispatchError`` —
|
|
74
127
|
the router wraps it as the tool result, so the model sees exactly what to
|
|
@@ -77,6 +130,11 @@ def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]
|
|
|
77
130
|
raise ToolDispatchError(
|
|
78
131
|
f"ask_user: questions must be an array, got {type(questions).__name__}"
|
|
79
132
|
)
|
|
133
|
+
if not (_MIN_QUESTIONS <= len(questions) <= _MAX_QUESTIONS):
|
|
134
|
+
raise ToolDispatchError(
|
|
135
|
+
f"ask_user: questions must contain {_MIN_QUESTIONS}-{_MAX_QUESTIONS} "
|
|
136
|
+
f"items, got {len(questions)}. Batch a small decision, not an interview."
|
|
137
|
+
)
|
|
80
138
|
normalized: list[dict[str, Any]] = []
|
|
81
139
|
for index, question in enumerate(questions):
|
|
82
140
|
if not isinstance(question, dict):
|
|
@@ -98,6 +156,47 @@ def _normalize_questions(questions: list[dict[str, Any]]) -> list[dict[str, Any]
|
|
|
98
156
|
f"ask_user: questions[{index}] is missing a non-empty string "
|
|
99
157
|
'"text" (the question label shown to the user).'
|
|
100
158
|
)
|
|
159
|
+
# header: optional on the wire; validate length when present, derive
|
|
160
|
+
# from text when absent so the card always has a chip label.
|
|
161
|
+
header = q.get("header")
|
|
162
|
+
if header is None:
|
|
163
|
+
q["header"] = _derive_header(q["text"])
|
|
164
|
+
elif not isinstance(header, str) or not header:
|
|
165
|
+
raise ToolDispatchError(
|
|
166
|
+
f"ask_user: questions[{index}].header must be a non-empty string."
|
|
167
|
+
)
|
|
168
|
+
elif len(header) > _MAX_HEADER_LEN:
|
|
169
|
+
raise ToolDispatchError(
|
|
170
|
+
f"ask_user: questions[{index}].header is {len(header)} chars; "
|
|
171
|
+
f"the chip label must be <= {_MAX_HEADER_LEN}."
|
|
172
|
+
)
|
|
173
|
+
# multiSelect: CC requires the model to commit to single vs multi.
|
|
174
|
+
# Default to False when absent so existing single-select calls keep
|
|
175
|
+
# working, but stamp it so hosts always read an explicit boolean.
|
|
176
|
+
if "multiSelect" not in q or q["multiSelect"] is None:
|
|
177
|
+
q["multiSelect"] = False
|
|
178
|
+
elif not isinstance(q["multiSelect"], bool):
|
|
179
|
+
raise ToolDispatchError(
|
|
180
|
+
f"ask_user: questions[{index}].multiSelect must be a boolean, "
|
|
181
|
+
f"got {type(q['multiSelect']).__name__}."
|
|
182
|
+
)
|
|
183
|
+
# options: only a select question carries choices; enforce the 2-4
|
|
184
|
+
# bound so the model pre-categorizes instead of dumping a long list.
|
|
185
|
+
qtype = q.get("type", "select")
|
|
186
|
+
options = q.get("options")
|
|
187
|
+
if qtype == "select":
|
|
188
|
+
if options is not None:
|
|
189
|
+
if not isinstance(options, list):
|
|
190
|
+
raise ToolDispatchError(
|
|
191
|
+
f"ask_user: questions[{index}].options must be an array, "
|
|
192
|
+
f"got {type(options).__name__}."
|
|
193
|
+
)
|
|
194
|
+
if not (_MIN_OPTIONS <= len(options) <= _MAX_OPTIONS):
|
|
195
|
+
raise ToolDispatchError(
|
|
196
|
+
f"ask_user: questions[{index}] has {len(options)} options; "
|
|
197
|
+
f"a select question needs {_MIN_OPTIONS}-{_MAX_OPTIONS}. "
|
|
198
|
+
"The host auto-appends an 'Other' escape — do not list it."
|
|
199
|
+
)
|
|
101
200
|
normalized.append(q)
|
|
102
201
|
return normalized
|
|
103
202
|
|
|
@@ -19,6 +19,8 @@ This module ties the existing primitives (``load_history_items`` +
|
|
|
19
19
|
messages, not record seqs (regenerate = fork keeping the last user
|
|
20
20
|
turn, dropping the assistant reply after it).
|
|
21
21
|
- ``lineage`` — walk the seed-provenance chain upwards, cycle-guarded.
|
|
22
|
+
- ``family_tree`` — the full family containing a record: lineage to the
|
|
23
|
+
root, then every descendant expanded from one storage-wide scan.
|
|
22
24
|
|
|
23
25
|
Children discovery is deliberately host-side: ``StorageAdapter`` has no
|
|
24
26
|
record enumeration, so stores that can list records implement
|
|
@@ -49,8 +51,11 @@ if TYPE_CHECKING:
|
|
|
49
51
|
|
|
50
52
|
__all__ = [
|
|
51
53
|
"BranchPoint",
|
|
54
|
+
"FamilyTree",
|
|
55
|
+
"FamilyTreeNode",
|
|
52
56
|
"ForkResult",
|
|
53
57
|
"branch_label",
|
|
58
|
+
"family_tree",
|
|
54
59
|
"fork_record",
|
|
55
60
|
"lineage",
|
|
56
61
|
"resolve_fork_seq",
|
|
@@ -60,6 +65,11 @@ __all__ = [
|
|
|
60
65
|
#: corruption, not a deep tree; fail loud past this depth.
|
|
61
66
|
_MAX_LINEAGE_DEPTH = 32
|
|
62
67
|
|
|
68
|
+
#: Bound on ``family_tree`` expansion — record enumeration is
|
|
69
|
+
#: storage-wide, so a runaway family is cut off and reported
|
|
70
|
+
#: (``FamilyTree.truncated``) rather than streamed unbounded to the host.
|
|
71
|
+
_MAX_TREE_NODES = 500
|
|
72
|
+
|
|
63
73
|
#: Reverse-scan page for ``resolve_fork_seq`` — the regen fork point is
|
|
64
74
|
#: almost always in the tail page.
|
|
65
75
|
_FORK_SCAN_PAGE = 128
|
|
@@ -95,6 +105,32 @@ class ForkResult:
|
|
|
95
105
|
messages: list["LLMMessage"]
|
|
96
106
|
|
|
97
107
|
|
|
108
|
+
@dataclass(frozen=True, slots=True)
|
|
109
|
+
class FamilyTreeNode:
|
|
110
|
+
"""One node in a full branch-family tree — a ``BranchPoint`` plus its
|
|
111
|
+
children, recursively. ``depth`` is 0 on the family root, +1 per fork
|
|
112
|
+
hop, matching the depths ``lineage`` assigns along the chain."""
|
|
113
|
+
|
|
114
|
+
record_id: str
|
|
115
|
+
source_record_id: str | None
|
|
116
|
+
source_until_seq: int | None
|
|
117
|
+
label: str
|
|
118
|
+
depth: int
|
|
119
|
+
children: tuple["FamilyTreeNode", ...] = ()
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass(frozen=True, slots=True)
|
|
123
|
+
class FamilyTree:
|
|
124
|
+
"""The full branch family containing one record, expanded from the
|
|
125
|
+
family root. ``truncated`` reports that a safety bound (depth or node
|
|
126
|
+
count) cut the expansion — the returned tree is still valid, just
|
|
127
|
+
incomplete below the cut."""
|
|
128
|
+
|
|
129
|
+
root: FamilyTreeNode
|
|
130
|
+
node_count: int
|
|
131
|
+
truncated: bool
|
|
132
|
+
|
|
133
|
+
|
|
98
134
|
def branch_label(messages: list["LLMMessage"], *, max_chars: int = 60) -> str:
|
|
99
135
|
"""Derive a branch summary from the forked prefix — deterministic, no
|
|
100
136
|
LLM call: the last user message, whitespace-collapsed and truncated.
|
|
@@ -292,3 +328,92 @@ async def lineage(
|
|
|
292
328
|
)
|
|
293
329
|
for index, point in enumerate(chain)
|
|
294
330
|
]
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
async def family_tree(
|
|
334
|
+
storage: "StorageAdapter",
|
|
335
|
+
record_id: str,
|
|
336
|
+
*,
|
|
337
|
+
max_depth: int = _MAX_LINEAGE_DEPTH,
|
|
338
|
+
max_nodes: int = _MAX_TREE_NODES,
|
|
339
|
+
) -> FamilyTree:
|
|
340
|
+
"""Full branch family containing ``record_id``, expanded from the root.
|
|
341
|
+
|
|
342
|
+
Walks seed provenance to the family root (``lineage``), then expands
|
|
343
|
+
every descendant from ONE storage-wide scan: parent edges live in each
|
|
344
|
+
record's first-entry seed, so reading that entry per enumerated record
|
|
345
|
+
(``list_history_records``) yields the children of every family member
|
|
346
|
+
at once — the same discovery ``agent.session.branches`` does for one
|
|
347
|
+
level, applied recursively without re-scanning per node.
|
|
348
|
+
|
|
349
|
+
Raises ``KeyError`` when the record has no entries (an empty record is
|
|
350
|
+
indistinguishable from a missing one — records are created by append)
|
|
351
|
+
and ``ValueError`` on lineage corruption (cycle / over-deep chain),
|
|
352
|
+
matching ``fork_record`` / ``lineage``. Expansion is bounded twice:
|
|
353
|
+
``max_depth`` cuts descendants below that depth from the root (the
|
|
354
|
+
default matches the lineage corruption bound, so by default only
|
|
355
|
+
corrupt families are cut) and ``max_nodes`` caps the total — either
|
|
356
|
+
cut is reported via ``FamilyTree.truncated`` instead of streaming
|
|
357
|
+
unbounded data. ``max_depth`` does NOT relax the lineage walk itself:
|
|
358
|
+
a queried record whose own chain exceeds the corruption bound still
|
|
359
|
+
raises ``ValueError``.
|
|
360
|
+
"""
|
|
361
|
+
|
|
362
|
+
if not await storage.list_history(record_id, limit=1):
|
|
363
|
+
raise KeyError(f"record not found: {record_id}")
|
|
364
|
+
chain = await lineage(storage, record_id)
|
|
365
|
+
root_point = chain[0]
|
|
366
|
+
|
|
367
|
+
children_of: dict[str, list[BranchPoint]] = {}
|
|
368
|
+
for candidate in await storage.list_history_records():
|
|
369
|
+
if candidate == root_point.record_id:
|
|
370
|
+
continue
|
|
371
|
+
first = await storage.list_history(candidate, limit=1)
|
|
372
|
+
if not first:
|
|
373
|
+
continue
|
|
374
|
+
entry = entry_from_dict(first[0])
|
|
375
|
+
if isinstance(entry, HistorySeed) and entry.source_record_id is not None:
|
|
376
|
+
children_of.setdefault(entry.source_record_id, []).append(
|
|
377
|
+
BranchPoint(
|
|
378
|
+
record_id=candidate,
|
|
379
|
+
source_record_id=entry.source_record_id,
|
|
380
|
+
source_until_seq=entry.source_until_seq,
|
|
381
|
+
label=branch_label(list(entry.messages)),
|
|
382
|
+
)
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
total = 0
|
|
386
|
+
truncated = False
|
|
387
|
+
|
|
388
|
+
def build(point: BranchPoint, depth: int) -> FamilyTreeNode:
|
|
389
|
+
nonlocal total, truncated
|
|
390
|
+
total += 1
|
|
391
|
+
children: list[FamilyTreeNode] = []
|
|
392
|
+
if depth >= max_depth:
|
|
393
|
+
if children_of.get(point.record_id):
|
|
394
|
+
truncated = True
|
|
395
|
+
else:
|
|
396
|
+
for child in children_of.get(point.record_id, []):
|
|
397
|
+
if total >= max_nodes:
|
|
398
|
+
truncated = True
|
|
399
|
+
break
|
|
400
|
+
children.append(build(child, depth + 1))
|
|
401
|
+
return FamilyTreeNode(
|
|
402
|
+
record_id=point.record_id,
|
|
403
|
+
source_record_id=point.source_record_id,
|
|
404
|
+
source_until_seq=point.source_until_seq,
|
|
405
|
+
label=point.label,
|
|
406
|
+
depth=depth,
|
|
407
|
+
children=tuple(children),
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
root = build(
|
|
411
|
+
BranchPoint(
|
|
412
|
+
record_id=root_point.record_id,
|
|
413
|
+
source_record_id=root_point.source_record_id,
|
|
414
|
+
source_until_seq=root_point.source_until_seq,
|
|
415
|
+
label=root_point.label,
|
|
416
|
+
),
|
|
417
|
+
0,
|
|
418
|
+
)
|
|
419
|
+
return FamilyTree(root=root, node_count=total, truncated=truncated)
|
|
@@ -38,8 +38,10 @@ Three trigger paths share the fold/summarize machinery:
|
|
|
38
38
|
prompt-cache prefix, so the interval trades cache hits for a bounded
|
|
39
39
|
transcript;
|
|
40
40
|
- **manual** (``compact_now``) — the host-command path (CC ``/compact``
|
|
41
|
-
parity):
|
|
42
|
-
the
|
|
41
|
+
parity): the host calls the sidecar's ``agent.chat.compact`` RPC, which
|
|
42
|
+
sets ``CoreLoop.request_compact()``; the loop runs ``compact_now`` at the
|
|
43
|
+
next pre_step boundary, folding + summarizing on demand, bypassing
|
|
44
|
+
threshold, hysteresis, and the circuit breaker.
|
|
43
45
|
|
|
44
46
|
Two safety rails share the machinery:
|
|
45
47
|
|
|
@@ -128,12 +128,27 @@ class OrchestrationStrategy(Protocol):
|
|
|
128
128
|
|
|
129
129
|
|
|
130
130
|
class _PreStepOnly(NoopHooks):
|
|
131
|
-
|
|
131
|
+
"""Project a hooks impl onto the pre_step slice.
|
|
132
|
+
|
|
133
|
+
``forward`` names optional off-protocol capabilities (probed with
|
|
134
|
+
``getattr``, e.g. ``compact_now``) that must stay reachable through the
|
|
135
|
+
projection — without it the wrapper would silently sever the sidecar's
|
|
136
|
+
``agent.chat.compact`` path, which probes the assembled chain for the
|
|
137
|
+
one hook implementing manual compaction.
|
|
138
|
+
"""
|
|
139
|
+
|
|
140
|
+
def __init__(self, inner: LoopHooks, *, forward: tuple[str, ...] = ()) -> None:
|
|
132
141
|
self._inner = inner
|
|
142
|
+
self._forward = forward
|
|
133
143
|
|
|
134
144
|
async def pre_step(self, transcript: Any, ctx: Any) -> Any:
|
|
135
145
|
return await self._inner.pre_step(transcript, ctx)
|
|
136
146
|
|
|
147
|
+
def __getattr__(self, name: str) -> Any:
|
|
148
|
+
if name in self._forward:
|
|
149
|
+
return getattr(self._inner, name)
|
|
150
|
+
raise AttributeError(name)
|
|
151
|
+
|
|
137
152
|
|
|
138
153
|
class _OnRequestErrorOnly(NoopHooks):
|
|
139
154
|
def __init__(self, inner: LoopHooks) -> None:
|
|
@@ -188,6 +203,12 @@ class PressureCompaction:
|
|
|
188
203
|
keep_last_tool_results: int = 2
|
|
189
204
|
fold_excerpt_chars: int | None = None
|
|
190
205
|
model: str | None = None
|
|
206
|
+
# Proactive micro-compaction (CC time-based microcompact parity): fold old
|
|
207
|
+
# tool results every N rounds regardless of pressure. 0 (default) is off —
|
|
208
|
+
# each fold invalidates the provider prompt-cache prefix, so the interval
|
|
209
|
+
# trades cache hits for a bounded transcript. R16 found compaction does not
|
|
210
|
+
# move the eval score, so this stays opt-in capability, not a default.
|
|
211
|
+
micro_compact_interval_rounds: int = 0
|
|
191
212
|
name: str = "pressure_compaction"
|
|
192
213
|
assumes: str = (
|
|
193
214
|
"long trajectories exceed the window; older detail is expendable "
|
|
@@ -209,8 +230,10 @@ class PressureCompaction:
|
|
|
209
230
|
keep_last_tool_results=self.keep_last_tool_results,
|
|
210
231
|
summarizer=provider,
|
|
211
232
|
model=self.model,
|
|
233
|
+
micro_compact_interval_rounds=self.micro_compact_interval_rounds,
|
|
212
234
|
**extra,
|
|
213
|
-
)
|
|
235
|
+
),
|
|
236
|
+
forward=("compact_now",),
|
|
214
237
|
)
|
|
215
238
|
|
|
216
239
|
|
|
@@ -805,8 +828,16 @@ class SingleAgent:
|
|
|
805
828
|
|
|
806
829
|
|
|
807
830
|
@dataclass(frozen=True, slots=True)
|
|
808
|
-
class
|
|
809
|
-
"""The
|
|
831
|
+
class PoolOrchestration:
|
|
832
|
+
"""The six-tool orchestration family over AgentPool (OrchestrationExecutor).
|
|
833
|
+
|
|
834
|
+
This is the advanced, model-driven orchestration arm: the parent gets
|
|
835
|
+
``agent_spawn``/``agent_send``/``agent_wait``/``agent_close``/
|
|
836
|
+
``agent_list``/``agent_interrupt`` and drives parallel children
|
|
837
|
+
explicitly. (The single-tool ``delegate_subagent`` seam —
|
|
838
|
+
``SubagentExecutor`` — is the default multi-agent surface on the
|
|
839
|
+
sidecar chat path; both run on the same AgentPool engine.)
|
|
840
|
+
"""
|
|
810
841
|
|
|
811
842
|
name: str = "subagent"
|
|
812
843
|
assumes: str = (
|
|
@@ -852,5 +883,8 @@ STRATEGY_REGISTRY: dict[str, dict[str, type]] = {
|
|
|
852
883
|
"progressive": ProgressiveDisclosure,
|
|
853
884
|
},
|
|
854
885
|
"memory": {"stateless": Stateless, "filesystem": FilesystemState},
|
|
855
|
-
|
|
886
|
+
# The "subagent" key names the eval arm (harness specs predate the
|
|
887
|
+
# delegate-on-pool unification); the implementation is the six-tool
|
|
888
|
+
# pool orchestration family.
|
|
889
|
+
"orchestration": {"single": SingleAgent, "subagent": PoolOrchestration},
|
|
856
890
|
}
|
{steerable_agent_runtime-0.6.2 → steerable_agent_runtime-0.6.3}/src/steerable_agent_runtime/hooks.py
RENAMED
|
@@ -309,6 +309,10 @@ class ChainHooks:
|
|
|
309
309
|
so a trailing delivery gate still forces writes when an earlier
|
|
310
310
|
validator would only ask for a no-tools summary.
|
|
311
311
|
- ``wrap_up_may_drop_tools``: False if any hook forbids dropping tools.
|
|
312
|
+
- ``compact_now``: optional capability (not part of the ``LoopHooks``
|
|
313
|
+
protocol), probed with ``getattr`` like ``on_stream_chunk`` —
|
|
314
|
+
delegated to the first hook that implements it (CompactionHooks);
|
|
315
|
+
a chain without one returns a no-op proceed.
|
|
312
316
|
|
|
313
317
|
This is how a product stacks e.g. compaction + spill + retry without the
|
|
314
318
|
loop knowing about any of them.
|
|
@@ -363,6 +367,28 @@ class ChainHooks:
|
|
|
363
367
|
append_action=append_action,
|
|
364
368
|
)
|
|
365
369
|
|
|
370
|
+
async def compact_now(
|
|
371
|
+
self, transcript: list[LLMMessage], ctx: LoopContext
|
|
372
|
+
) -> PreStepAction:
|
|
373
|
+
"""Manual compaction (CC ``/compact`` parity): delegate to the first
|
|
374
|
+
hook in the chain that implements ``compact_now`` (CompactionHooks).
|
|
375
|
+
|
|
376
|
+
``compact_now`` is an optional capability, not a ``LoopHooks``
|
|
377
|
+
protocol method, so membership is probed with ``getattr`` — the same
|
|
378
|
+
pattern the loop uses for ``on_stream_chunk``. ``NoopHooks`` defines
|
|
379
|
+
no ``compact_now``, so the probe skips it (and any other hook that
|
|
380
|
+
only has the protocol surface) without a special case. A chain with
|
|
381
|
+
no compaction hook returns a no-op proceed: a manual compact request
|
|
382
|
+
must never fail the turn.
|
|
383
|
+
"""
|
|
384
|
+
for hook in self._hooks:
|
|
385
|
+
callback = getattr(hook, "compact_now", None)
|
|
386
|
+
if callable(callback):
|
|
387
|
+
return await callback(transcript, ctx)
|
|
388
|
+
return PreStepAction(
|
|
389
|
+
kind="proceed", reason="compact: no compaction hook in chain"
|
|
390
|
+
)
|
|
391
|
+
|
|
366
392
|
async def post_tool_result(
|
|
367
393
|
self, result: ToolResult, call: ToolCall, ctx: LoopContext
|
|
368
394
|
) -> ToolResult:
|
|
@@ -159,9 +159,18 @@ from .errors import (
|
|
|
159
159
|
is_retryable,
|
|
160
160
|
)
|
|
161
161
|
from .openai_compat import OpenAICompatProvider
|
|
162
|
+
from .presets import (
|
|
163
|
+
PROVIDER_PRESETS,
|
|
164
|
+
PresetEntry,
|
|
165
|
+
ProviderPreset,
|
|
166
|
+
describe_provider_presets,
|
|
167
|
+
preset_for,
|
|
168
|
+
register_provider_preset,
|
|
169
|
+
)
|
|
162
170
|
|
|
163
171
|
__all__ = [
|
|
164
172
|
"PROVIDER_COMPAT_HOSTS",
|
|
173
|
+
"PROVIDER_PRESETS",
|
|
165
174
|
"RETRYABLE_KINDS",
|
|
166
175
|
"AnthropicProvider",
|
|
167
176
|
"ContentPart",
|
|
@@ -174,7 +183,10 @@ __all__ = [
|
|
|
174
183
|
"LLMStreamChunk",
|
|
175
184
|
"LLMUsage",
|
|
176
185
|
"OpenAICompatFlags",
|
|
186
|
+
"PresetEntry",
|
|
187
|
+
"ProviderPreset",
|
|
177
188
|
"describe_compat_flags",
|
|
189
|
+
"describe_provider_presets",
|
|
178
190
|
"OpenAICompatProvider",
|
|
179
191
|
"TextPart",
|
|
180
192
|
"classify_error",
|
|
@@ -182,5 +194,7 @@ __all__ = [
|
|
|
182
194
|
"compat_for_base_url",
|
|
183
195
|
"content_text",
|
|
184
196
|
"is_retryable",
|
|
197
|
+
"preset_for",
|
|
198
|
+
"register_provider_preset",
|
|
185
199
|
"text_parts",
|
|
186
200
|
]
|
|
@@ -20,7 +20,7 @@ import logging
|
|
|
20
20
|
import os
|
|
21
21
|
from collections.abc import AsyncIterator, Iterable, Sequence
|
|
22
22
|
from dataclasses import dataclass
|
|
23
|
-
from typing import Any
|
|
23
|
+
from typing import Any, Literal
|
|
24
24
|
|
|
25
25
|
from steerable_agent_protocol.generated import ToolCall
|
|
26
26
|
|
|
@@ -29,6 +29,7 @@ from . import LLMMessage, LLMStreamChunk, LLMUsage
|
|
|
29
29
|
from .compat import OpenAICompatFlags
|
|
30
30
|
from .errors import LLMError, classify_http_status, parse_retry_after_ms
|
|
31
31
|
from .parts import ImagePart, TextPart
|
|
32
|
+
from .presets import ProviderPreset, preset_for
|
|
32
33
|
from .system_proxy import client_env_kwargs
|
|
33
34
|
|
|
34
35
|
logger = logging.getLogger(__name__)
|
|
@@ -120,6 +121,11 @@ class OpenAICompatProvider:
|
|
|
120
121
|
api_key: str | None = None
|
|
121
122
|
default_temperature: float | None = None
|
|
122
123
|
compat: OpenAICompatFlags | None = None
|
|
124
|
+
#: Vendor-preset selection: ``"auto"`` matches ``llm.presets.preset_for``
|
|
125
|
+
#: on (base_url, model); ``"off"`` disables the layer for this provider;
|
|
126
|
+
#: a ``ProviderPreset`` instance pins that preset regardless of the
|
|
127
|
+
#: registry (host settings UIs send an explicit choice this way).
|
|
128
|
+
preset: ProviderPreset | Literal["auto", "off"] = "auto"
|
|
123
129
|
|
|
124
130
|
def __post_init__(self) -> None:
|
|
125
131
|
if not self.base_url:
|
|
@@ -310,6 +316,17 @@ class OpenAICompatProvider:
|
|
|
310
316
|
extra: dict[str, Any],
|
|
311
317
|
) -> dict[str, Any]:
|
|
312
318
|
compat = self.compat or OpenAICompatFlags()
|
|
319
|
+
# Vendor-documented optimal parameters (llm.presets): defaults filled
|
|
320
|
+
# only where the caller left the field unset — explicit per-request
|
|
321
|
+
# fields, host extra kwargs, and default_temperature always win, and
|
|
322
|
+
# compat flags still gate what may be sent at all. The provider's
|
|
323
|
+
# `preset` field selects the source: auto-match, off, or pinned.
|
|
324
|
+
if self.preset == "off":
|
|
325
|
+
preset = None
|
|
326
|
+
elif self.preset == "auto":
|
|
327
|
+
preset = preset_for(self.base_url, self.model)
|
|
328
|
+
else:
|
|
329
|
+
preset = self.preset
|
|
313
330
|
body: dict[str, Any] = {
|
|
314
331
|
"model": self.model,
|
|
315
332
|
"messages": [_encode_message(m, compat=compat) for m in messages],
|
|
@@ -322,15 +339,27 @@ class OpenAICompatProvider:
|
|
|
322
339
|
# vLLM, DeepSeek; strict vendors flag out via compat.
|
|
323
340
|
body["stream_options"] = {"include_usage": True}
|
|
324
341
|
eff_temperature = temperature if temperature is not None else self.default_temperature
|
|
342
|
+
if eff_temperature is None and preset is not None:
|
|
343
|
+
eff_temperature = preset.temperature
|
|
325
344
|
if eff_temperature is not None and compat.supports_temperature:
|
|
326
345
|
body["temperature"] = eff_temperature
|
|
327
|
-
|
|
328
|
-
|
|
346
|
+
eff_max_tokens = max_tokens
|
|
347
|
+
if eff_max_tokens is None and preset is not None:
|
|
348
|
+
eff_max_tokens = preset.max_tokens
|
|
349
|
+
if eff_max_tokens is not None:
|
|
350
|
+
body[compat.max_tokens_field] = eff_max_tokens
|
|
329
351
|
if tools is not None:
|
|
330
352
|
tools_list = list(tools)
|
|
331
353
|
if tools_list:
|
|
332
354
|
body["tools"] = tools_list
|
|
333
355
|
body.update(extra)
|
|
356
|
+
if preset is not None:
|
|
357
|
+
# Fill-only-when-absent: anything in ``extra`` already won above.
|
|
358
|
+
if preset.top_p is not None and "top_p" not in body:
|
|
359
|
+
body["top_p"] = preset.top_p
|
|
360
|
+
for key, value in preset.extra_body.items():
|
|
361
|
+
if key not in body:
|
|
362
|
+
body[key] = value
|
|
334
363
|
# ``tool_choice=required`` downgrade paths:
|
|
335
364
|
# - compat flag: vendors whose thinking mode 400s the forced value
|
|
336
365
|
# (DeepSeek: "Thinking mode does not support this tool_choice");
|
|
@@ -343,12 +372,16 @@ class OpenAICompatProvider:
|
|
|
343
372
|
or _z_ai_tool_choice_auto_only(self.model, self.base_url)
|
|
344
373
|
):
|
|
345
374
|
body["tool_choice"] = "auto"
|
|
346
|
-
# W6-8: clamp the
|
|
375
|
+
# W6-8: clamp the requested reasoning effort to a level the model
|
|
347
376
|
# actually supports (structured ModelInfo replaces the raw env
|
|
348
377
|
# passthrough). A model with no reasoning knob gets no parameter at
|
|
349
|
-
# all — sending one would be an unsupported-field error on strict
|
|
378
|
+
# all — sending one would be an unsupported-field error on strict
|
|
379
|
+
# APIs. The env var is the explicit request; the preset's documented
|
|
380
|
+
# default applies only when the env is unset.
|
|
350
381
|
effort = clamp_reasoning_effort(
|
|
351
|
-
self.model,
|
|
382
|
+
self.model,
|
|
383
|
+
os.environ.get("STEERABLE_REASONING_EFFORT", "")
|
|
384
|
+
or (preset.reasoning_effort if preset is not None else ""),
|
|
352
385
|
)
|
|
353
386
|
if effort and compat.supports_reasoning_effort:
|
|
354
387
|
# GLM-5.3 default is ``max``; ``high`` is a downgrade. TB uses max.
|