steerable-agent-runtime 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/PKG-INFO +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/pyproject.toml +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/harness.py +24 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/hooks.py +20 -4
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/loop.py +72 -10
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/reminders.py +19 -20
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/skills.py +5 -2
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_harness.py +25 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_hooks.py +56 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_long_session.py +3 -3
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop.py +189 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_reminders.py +25 -1
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/README.md +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/setup.cfg +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/antihallucination.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/approval.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/approval_policy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/branch.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/cache_control.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/calibration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/compaction.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/default.harness.json +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/default.harness.yaml +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/errors.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/handoff.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/harness_spec.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/history.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/compat.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/errors.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/openai_compat.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/parts.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/maintenance.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/mcp.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/mcp_server.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_catalog.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_info.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_resolve.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/observation_aging.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/orchestration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/otel.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/pricing.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/pseudo.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/recording.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/replay.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/resume.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/retry.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/sandboxed.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/spill.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/subagent.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tokens.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tool_schema.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tool_search.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tools.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tracing.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/__init__.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/world_state.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/SOURCES.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_antihallucination.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_approval.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_approval_policy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_branch.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_cache_control.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_cache_instrumentation.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_calibration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_compaction.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_content_parts.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_error_taxonomy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_fragment_bounds.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_golden.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_handoff.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_harness_spec.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_history.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_history_persistence.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_in_memory_storage.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_llm_wire_helpers.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_cancellation.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_replay.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_sandbox_event.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_maintenance.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_mcp.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_mcp_server.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_catalog.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_equivalence.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_info.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_resolve.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_observation_aging.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_orchestration.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_otel.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_parallel_tools.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_provider_compat.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_pseudo.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_recording.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_replay_crosslang.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_resume.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_retry_hooks.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_safety_gate.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_sandboxed.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_skills.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_soft_timeout.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_spill.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_sqlite_storage.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_steer.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_storage_contract.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_stream_strip.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_subagent.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_system_proxy.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tokens.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_exposure.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_hygiene.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_router.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_schema.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_timeout.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_trace_recorder.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_transport_jsonrpc.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_transport_sse.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_usage_attribution.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_world_state.py +0 -0
- {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_write_lease.py +0 -0
|
@@ -371,6 +371,10 @@ class NullValidator:
|
|
|
371
371
|
class SelfCritique:
|
|
372
372
|
"""The existing discipline-retry + grounding stack (AntiHallucinationHooks)."""
|
|
373
373
|
|
|
374
|
+
#: Headless fills this from the TB instruction. Empty leaves claimed /
|
|
375
|
+
#: eager-deferred blind (``detect_execution_intent_in_user_message``
|
|
376
|
+
#: is false on "") and the grounding judge sees no user question.
|
|
377
|
+
user_question: str = ""
|
|
374
378
|
name: str = "self_critique"
|
|
375
379
|
assumes: str = (
|
|
376
380
|
"the model sometimes claims executions it did not perform; a "
|
|
@@ -380,9 +384,27 @@ class SelfCritique:
|
|
|
380
384
|
def hooks(self, *, provider: LLMProvider | None = None) -> LoopHooks:
|
|
381
385
|
if provider is None:
|
|
382
386
|
raise ValueError("SelfCritique requires the turn's LLM provider")
|
|
383
|
-
from .antihallucination import
|
|
387
|
+
from .antihallucination import (
|
|
388
|
+
AntiHallucinationConfig,
|
|
389
|
+
AntiHallucinationHooks,
|
|
390
|
+
detect_execution_intent_in_user_message,
|
|
391
|
+
)
|
|
384
392
|
|
|
385
|
-
|
|
393
|
+
question = self.user_question
|
|
394
|
+
# Desktop exec-intent is chat ("跑一下" / "execute"). TB instructions
|
|
395
|
+
# are unattended execution even when they never say "run".
|
|
396
|
+
if question and not detect_execution_intent_in_user_message(question):
|
|
397
|
+
question = f"Execute the following.\n{question}"
|
|
398
|
+
return _BeforeCompletionOnly(
|
|
399
|
+
AntiHallucinationHooks(
|
|
400
|
+
provider,
|
|
401
|
+
AntiHallucinationConfig(
|
|
402
|
+
user_question=question,
|
|
403
|
+
tools_available=True,
|
|
404
|
+
enable_routing=False,
|
|
405
|
+
),
|
|
406
|
+
)
|
|
407
|
+
)
|
|
386
408
|
|
|
387
409
|
|
|
388
410
|
# ---------------------------------------------------------------------------
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/hooks.py
RENAMED
|
@@ -150,6 +150,18 @@ class CompletionAction:
|
|
|
150
150
|
reason: str | None = None
|
|
151
151
|
|
|
152
152
|
|
|
153
|
+
_COMPLETION_KIND_RANK = {"accept": 0, "narrate": 1, "retry": 2}
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _completion_outranks(new: CompletionAction, current: CompletionAction) -> bool:
|
|
157
|
+
"""Prefer forcing another working round over a no-tools summary."""
|
|
158
|
+
new_rank = _COMPLETION_KIND_RANK[new.kind]
|
|
159
|
+
old_rank = _COMPLETION_KIND_RANK[current.kind]
|
|
160
|
+
if new_rank != old_rank:
|
|
161
|
+
return new_rank > old_rank
|
|
162
|
+
return new.kind != "accept"
|
|
163
|
+
|
|
164
|
+
|
|
153
165
|
@dataclass(slots=True)
|
|
154
166
|
class RetryAction:
|
|
155
167
|
"""Outcome of an ``on_request_error`` hook.
|
|
@@ -285,7 +297,10 @@ class ChainHooks:
|
|
|
285
297
|
- ``post_tool_result``: the result threads through each hook in order.
|
|
286
298
|
- ``on_request_error``: the first ``retry`` decision wins; if every hook
|
|
287
299
|
says ``fail``, the first failure reason is surfaced.
|
|
288
|
-
- ``before_completion``:
|
|
300
|
+
- ``before_completion``: every hook runs. ``retry`` outranks
|
|
301
|
+
``narrate``, which outranks ``accept``. Same-rank later hooks win,
|
|
302
|
+
so a trailing delivery gate still forces writes when an earlier
|
|
303
|
+
validator would only ask for a no-tools summary.
|
|
289
304
|
- ``wrap_up_may_drop_tools``: False if any hook forbids dropping tools.
|
|
290
305
|
|
|
291
306
|
This is how a product stacks e.g. compaction + spill + retry without the
|
|
@@ -364,11 +379,12 @@ class ChainHooks:
|
|
|
364
379
|
async def before_completion(
|
|
365
380
|
self, draft: CompletionDraft, ctx: LoopContext
|
|
366
381
|
) -> CompletionAction:
|
|
382
|
+
picked = CompletionAction(kind="accept")
|
|
367
383
|
for hook in self._hooks:
|
|
368
384
|
action = await hook.before_completion(draft, ctx)
|
|
369
|
-
if action
|
|
370
|
-
|
|
371
|
-
return
|
|
385
|
+
if _completion_outranks(action, picked):
|
|
386
|
+
picked = action
|
|
387
|
+
return picked
|
|
372
388
|
|
|
373
389
|
def on_stream_chunk(self, chunk: Any, ctx: LoopContext) -> None:
|
|
374
390
|
"""Fan the chunk out to every hook; one bad hook must not break the rest."""
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/loop.py
RENAMED
|
@@ -76,7 +76,7 @@ from .history import (
|
|
|
76
76
|
entry_to_dict,
|
|
77
77
|
)
|
|
78
78
|
from .hooks import CompletionDraft, LoopHooks, NoopHooks
|
|
79
|
-
from .llm import LLMMessage, LLMProvider, LLMUsage
|
|
79
|
+
from .llm import ImagePart, LLMMessage, LLMProvider, LLMUsage, TextPart
|
|
80
80
|
from .pseudo import (
|
|
81
81
|
PseudoStreamStripper,
|
|
82
82
|
extract_inline_tool_calls,
|
|
@@ -1641,6 +1641,7 @@ class CoreLoop:
|
|
|
1641
1641
|
|
|
1642
1642
|
breaker_tripped = False
|
|
1643
1643
|
steer_interrupted = False
|
|
1644
|
+
round_images: list[tuple[str, ImagePart]] = []
|
|
1644
1645
|
for batch_idx, (batch_safe, batch) in enumerate(batches):
|
|
1645
1646
|
if breaker_tripped or steer_interrupted:
|
|
1646
1647
|
break
|
|
@@ -1845,6 +1846,19 @@ class CoreLoop:
|
|
|
1845
1846
|
else:
|
|
1846
1847
|
ctx.consecutive_tool_errors += 1
|
|
1847
1848
|
|
|
1849
|
+
# Pop pixels before spill: a base64 PNG in ``data`` is
|
|
1850
|
+
# always larger than the 16 KB inline budget, and spilling
|
|
1851
|
+
# it would hide the image from the next request. OpenAI
|
|
1852
|
+
# tool messages are text-only, so the pixels ride on a
|
|
1853
|
+
# user message after this batch (``_append_read_images``).
|
|
1854
|
+
path = ""
|
|
1855
|
+
if isinstance(result.data, dict):
|
|
1856
|
+
path = str(result.data.get("path") or "")
|
|
1857
|
+
image = _pop_result_image(result)
|
|
1858
|
+
if image is not None:
|
|
1859
|
+
if isinstance(result.data, dict):
|
|
1860
|
+
result.data["pixels"] = "attached"
|
|
1861
|
+
round_images.append((path or call.name, image))
|
|
1848
1862
|
# ── hook: post_tool_result (spill / truncation) ──────
|
|
1849
1863
|
result = await self._hooks.post_tool_result(result, call, ctx)
|
|
1850
1864
|
|
|
@@ -1897,14 +1911,7 @@ class CoreLoop:
|
|
|
1897
1911
|
**({"error": result.error} if result.error else {}),
|
|
1898
1912
|
},
|
|
1899
1913
|
)
|
|
1900
|
-
manager.append(
|
|
1901
|
-
LLMMessage.text_of(
|
|
1902
|
-
"tool",
|
|
1903
|
-
_result_content(result),
|
|
1904
|
-
name=call.name,
|
|
1905
|
-
tool_call_id=call.id,
|
|
1906
|
-
)
|
|
1907
|
-
)
|
|
1914
|
+
manager.append(_tool_result_message(call, result))
|
|
1908
1915
|
|
|
1909
1916
|
# Approval abort ends the turn: record the batch like the
|
|
1910
1917
|
# breaker does (real results for executed calls, synthetic
|
|
@@ -2055,6 +2062,7 @@ class CoreLoop:
|
|
|
2055
2062
|
# above — skip the "executing" bookkeeping and run the wrap-up
|
|
2056
2063
|
# round directly. Artifact wrap-up still executes tools, so it
|
|
2057
2064
|
# falls through and increments like a normal round.
|
|
2065
|
+
_append_read_images(manager, round_images)
|
|
2058
2066
|
if wrap_up and withholding_tools():
|
|
2059
2067
|
continue
|
|
2060
2068
|
if wrap_up:
|
|
@@ -2459,6 +2467,57 @@ def _step_summary(
|
|
|
2459
2467
|
}
|
|
2460
2468
|
|
|
2461
2469
|
|
|
2470
|
+
_MAX_READ_IMAGES_PER_ROUND = 4
|
|
2471
|
+
|
|
2472
|
+
|
|
2473
|
+
def _pop_result_image(result: ToolResult) -> ImagePart | None:
|
|
2474
|
+
"""Take ``data._image`` off the payload so spill and JSON never carry it."""
|
|
2475
|
+
data = result.data
|
|
2476
|
+
if not isinstance(data, dict):
|
|
2477
|
+
return None
|
|
2478
|
+
blob = data.pop("_image", None)
|
|
2479
|
+
if not isinstance(blob, dict):
|
|
2480
|
+
return None
|
|
2481
|
+
b64 = blob.get("b64")
|
|
2482
|
+
media = blob.get("media_type") or "image/png"
|
|
2483
|
+
if not isinstance(b64, str) or not b64:
|
|
2484
|
+
return None
|
|
2485
|
+
return ImagePart.from_base64(b64, media_type=str(media))
|
|
2486
|
+
|
|
2487
|
+
|
|
2488
|
+
def _tool_result_message(call: ToolCall, result: ToolResult) -> LLMMessage:
|
|
2489
|
+
"""Tool observation as text. OpenAI forbids image_url on role=tool."""
|
|
2490
|
+
return LLMMessage.text_of(
|
|
2491
|
+
"tool", _result_content(result), name=call.name, tool_call_id=call.id
|
|
2492
|
+
)
|
|
2493
|
+
|
|
2494
|
+
|
|
2495
|
+
def _append_read_images(
|
|
2496
|
+
manager: ContextManager, images: list[tuple[str, ImagePart]]
|
|
2497
|
+
) -> None:
|
|
2498
|
+
"""Put PNG/JPEG pixels on a user turn after the tool-result batch.
|
|
2499
|
+
|
|
2500
|
+
OpenAI ``ChatCompletionToolMessageParam.content`` is string or text
|
|
2501
|
+
parts only. Claude Code's Anthropic wire can nest images in
|
|
2502
|
+
``tool_result``; the Harbor GLM cell is OpenAI-compat, so the pixels
|
|
2503
|
+
follow the tool JSON as a user message instead.
|
|
2504
|
+
"""
|
|
2505
|
+
if not images:
|
|
2506
|
+
return
|
|
2507
|
+
kept = images[:_MAX_READ_IMAGES_PER_ROUND]
|
|
2508
|
+
parts: list = [
|
|
2509
|
+
TextPart(
|
|
2510
|
+
"Pixels from the files just read. The tool JSON is an ASCII "
|
|
2511
|
+
"preview; look at these images for the actual contents."
|
|
2512
|
+
)
|
|
2513
|
+
]
|
|
2514
|
+
for path, image in kept:
|
|
2515
|
+
if path:
|
|
2516
|
+
parts.append(TextPart(f"\n{path}:"))
|
|
2517
|
+
parts.append(image)
|
|
2518
|
+
manager.append(LLMMessage(role="user", content=parts), kind="tool.image")
|
|
2519
|
+
|
|
2520
|
+
|
|
2462
2521
|
def _result_content(result: ToolResult) -> str:
|
|
2463
2522
|
"""Serialize a ToolResult into the tool-message content for the transcript."""
|
|
2464
2523
|
|
|
@@ -2468,7 +2527,10 @@ def _result_content(result: ToolResult) -> str:
|
|
|
2468
2527
|
if result.error:
|
|
2469
2528
|
payload["error"] = result.error
|
|
2470
2529
|
if result.data is not None:
|
|
2471
|
-
|
|
2530
|
+
data = result.data
|
|
2531
|
+
if isinstance(data, dict) and "_image" in data:
|
|
2532
|
+
data = {k: v for k, v in data.items() if k != "_image"}
|
|
2533
|
+
payload["data"] = data
|
|
2472
2534
|
if result.message:
|
|
2473
2535
|
payload["message"] = result.message
|
|
2474
2536
|
return json.dumps(payload, ensure_ascii=False)
|
|
@@ -71,7 +71,7 @@ class AbandonedRecoveryReminder(ContextFragment):
|
|
|
71
71
|
|
|
72
72
|
|
|
73
73
|
class RunawayExplorationReminder(ContextFragment):
|
|
74
|
-
"""Many tool calls without a
|
|
74
|
+
"""Many tool calls without a write since the last one."""
|
|
75
75
|
|
|
76
76
|
content_kind = "reminder.runaway_exploration"
|
|
77
77
|
max_tokens = 200
|
|
@@ -81,15 +81,14 @@ class RunawayExplorationReminder(ContextFragment):
|
|
|
81
81
|
|
|
82
82
|
def body(self) -> str:
|
|
83
83
|
return (
|
|
84
|
-
f"[system notice]
|
|
85
|
-
"
|
|
86
|
-
"
|
|
87
|
-
"need, produce the artifact now."
|
|
84
|
+
f"[system notice] {self._calls} tool calls since the last write. "
|
|
85
|
+
"If the required files already exist, run them; if they do "
|
|
86
|
+
"not, write them now. Do not keep inspecting source."
|
|
88
87
|
)
|
|
89
88
|
|
|
90
89
|
@classmethod
|
|
91
90
|
def type_markers(cls) -> tuple[str, str]:
|
|
92
|
-
return ("[system notice]
|
|
91
|
+
return ("[system notice]", "Do not keep inspecting source.")
|
|
93
92
|
|
|
94
93
|
|
|
95
94
|
# ---------------------------------------------------------------------------
|
|
@@ -157,7 +156,7 @@ REMINDER_CATALOG: tuple[ReminderEntry, ...] = (
|
|
|
157
156
|
),
|
|
158
157
|
ReminderEntry(
|
|
159
158
|
id="exploration.runaway",
|
|
160
|
-
failure_mode="
|
|
159
|
+
failure_mode="探索失控:连续非写入调用,含写过之后继续只读",
|
|
161
160
|
fragment=RunawayExplorationReminder,
|
|
162
161
|
),
|
|
163
162
|
)
|
|
@@ -190,7 +189,8 @@ class ReminderRules:
|
|
|
190
189
|
#: Fire ``recovery.error_streak`` when consecutive errors reach this
|
|
191
190
|
#: fraction of the loop's circuit breaker.
|
|
192
191
|
error_streak_ratio: float = 0.5
|
|
193
|
-
#: Fire ``exploration.runaway`` after this many tool calls without a
|
|
192
|
+
#: Fire ``exploration.runaway`` after this many tool calls without a
|
|
193
|
+
#: write since the last one (or since the start).
|
|
194
194
|
runaway_calls: int = 12
|
|
195
195
|
#: Re-fire a still-true rule after this many rounds (fade-out is the
|
|
196
196
|
#: point, but every-round spam is noise).
|
|
@@ -200,11 +200,11 @@ class ReminderRules:
|
|
|
200
200
|
class ReminderHooks(NoopHooks):
|
|
201
201
|
"""Fires catalog reminders from observed loop state.
|
|
202
202
|
|
|
203
|
-
``post_tool_result`` tracks
|
|
204
|
-
write
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
only after ``refire_rounds`` while its condition still holds.
|
|
203
|
+
``post_tool_result`` tracks error streaks and consecutive non-write
|
|
204
|
+
calls (a write resets that streak; a prior write does not silence
|
|
205
|
+
later inspect-only runs). ``pre_step`` appends a due reminder as the
|
|
206
|
+
last transcript item before the request. Each rule fires once, then
|
|
207
|
+
re-fires only after ``refire_rounds`` while its condition still holds.
|
|
208
208
|
"""
|
|
209
209
|
|
|
210
210
|
def __init__(
|
|
@@ -217,14 +217,14 @@ class ReminderHooks(NoopHooks):
|
|
|
217
217
|
self._rules = rules or ReminderRules()
|
|
218
218
|
self._max_tool_errors = max_tool_errors
|
|
219
219
|
self._write_tools = write_tools
|
|
220
|
-
self.
|
|
221
|
-
self._calls = 0
|
|
220
|
+
self._since_write = 0
|
|
222
221
|
self._fired_at: dict[str, int] = {}
|
|
223
222
|
|
|
224
223
|
async def post_tool_result(self, result: Any, call: Any, ctx: Any) -> Any:
|
|
225
|
-
self._calls += 1
|
|
226
224
|
if call.name in self._write_tools and getattr(result, "success", False):
|
|
227
|
-
self.
|
|
225
|
+
self._since_write = 0
|
|
226
|
+
else:
|
|
227
|
+
self._since_write += 1
|
|
228
228
|
return result
|
|
229
229
|
|
|
230
230
|
async def pre_step(self, transcript: Any, ctx: Any) -> PreStepAction:
|
|
@@ -261,8 +261,7 @@ class ReminderHooks(NoopHooks):
|
|
|
261
261
|
self._fired_at["recovery.error_streak"] = round_index
|
|
262
262
|
return "recovery.error_streak"
|
|
263
263
|
if (
|
|
264
|
-
self.
|
|
265
|
-
and not self._saw_write
|
|
264
|
+
self._since_write >= self._rules.runaway_calls
|
|
266
265
|
and ready("exploration.runaway")
|
|
267
266
|
):
|
|
268
267
|
self._fired_at["exploration.runaway"] = round_index
|
|
@@ -275,5 +274,5 @@ class ReminderHooks(NoopHooks):
|
|
|
275
274
|
getattr(ctx, "consecutive_tool_errors", 0), self._max_tool_errors
|
|
276
275
|
)
|
|
277
276
|
if entry.id == "exploration.runaway":
|
|
278
|
-
return RunawayExplorationReminder(self.
|
|
277
|
+
return RunawayExplorationReminder(self._since_write)
|
|
279
278
|
return entry.fragment()
|
|
@@ -397,9 +397,12 @@ def _parse_skill_dir(skill_dir: Path) -> SkillDefinition | None:
|
|
|
397
397
|
model_invocable = fm.get("disable-model-invocation") is not True
|
|
398
398
|
|
|
399
399
|
# Same `{scripts}` resolution as the TS prompt assembly: point at the
|
|
400
|
-
# skill's own scripts/ directory with uniform separators.
|
|
400
|
+
# skill's own scripts/ directory with uniform separators. The re.sub
|
|
401
|
+
# replacement MUST be a lambda: on Windows scripts_path carries
|
|
402
|
+
# backslashes, and a plain-string replacement would have its `\U`, `\n`
|
|
403
|
+
# etc. interpreted as regex escapes (re.error: bad escape \U).
|
|
401
404
|
scripts_path = str(skill_dir / "scripts")
|
|
402
|
-
content = re.sub(r"\{scripts\}[/\\]", scripts_path + "/", content)
|
|
405
|
+
content = re.sub(r"\{scripts\}[/\\]", lambda _m: scripts_path + "/", content)
|
|
403
406
|
content = content.replace("{scripts}", scripts_path)
|
|
404
407
|
|
|
405
408
|
return SkillDefinition(
|
|
@@ -164,6 +164,31 @@ async def test_null_validator_accepts() -> None:
|
|
|
164
164
|
assert action.kind == "accept"
|
|
165
165
|
|
|
166
166
|
|
|
167
|
+
def test_self_critique_passes_the_instruction_into_the_judge() -> None:
|
|
168
|
+
"""Harbor's self_critique arm is a silent no-op if user_question stays
|
|
169
|
+
empty: claimed/eager-deferred need exec intent, and the grounding judge
|
|
170
|
+
is prompted with an empty 用户提问."""
|
|
171
|
+
from steerable_agent_runtime.harness import SelfCritique
|
|
172
|
+
|
|
173
|
+
class _Provider:
|
|
174
|
+
name = "fake"
|
|
175
|
+
model = "fake-model"
|
|
176
|
+
|
|
177
|
+
hooks = SelfCritique(user_question="Write /app/out.txt").hooks(
|
|
178
|
+
provider=_Provider()
|
|
179
|
+
)
|
|
180
|
+
config = hooks._inner._config
|
|
181
|
+
assert config.user_question.startswith("Execute the following.")
|
|
182
|
+
assert "Write /app/out.txt" in config.user_question
|
|
183
|
+
assert config.tools_available is True
|
|
184
|
+
assert config.enable_routing is False
|
|
185
|
+
|
|
186
|
+
already = SelfCritique(user_question="execute the hidden tests").hooks(
|
|
187
|
+
provider=_Provider()
|
|
188
|
+
)
|
|
189
|
+
assert already._inner._config.user_question == "execute the hidden tests"
|
|
190
|
+
|
|
191
|
+
|
|
167
192
|
# -- tools -------------------------------------------------------------------
|
|
168
193
|
|
|
169
194
|
|
|
@@ -15,6 +15,8 @@ from steerable_agent_protocol.generated import ToolCall, ToolResult
|
|
|
15
15
|
|
|
16
16
|
from steerable_agent_runtime import (
|
|
17
17
|
ChainHooks,
|
|
18
|
+
CompletionAction,
|
|
19
|
+
CompletionDraft,
|
|
18
20
|
CoreLoop,
|
|
19
21
|
LoopEvent,
|
|
20
22
|
NoopHooks,
|
|
@@ -254,6 +256,60 @@ async def test_chain_hooks_appends_only_compose_in_order() -> None:
|
|
|
254
256
|
assert [a.message.content_text for a in action.appends] == ["A", "B"]
|
|
255
257
|
|
|
256
258
|
|
|
259
|
+
# ---------------------------------------------------------------------------
|
|
260
|
+
# before_completion
|
|
261
|
+
# ---------------------------------------------------------------------------
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _draft(**overrides: Any) -> CompletionDraft:
|
|
265
|
+
values = dict(
|
|
266
|
+
status="failed",
|
|
267
|
+
reason="empty",
|
|
268
|
+
content="",
|
|
269
|
+
round_index=3,
|
|
270
|
+
had_tool_calls=False,
|
|
271
|
+
tool_calls_used=4,
|
|
272
|
+
tool_successes=2,
|
|
273
|
+
)
|
|
274
|
+
values.update(overrides)
|
|
275
|
+
return CompletionDraft(**values)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
class _Narrate(NoopHooks):
|
|
279
|
+
async def before_completion(self, draft, ctx):
|
|
280
|
+
return CompletionAction(kind="narrate", reason="narration")
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
class _RetryWrite(NoopHooks):
|
|
284
|
+
async def before_completion(self, draft, ctx):
|
|
285
|
+
return CompletionAction(kind="retry", reason="empty_round")
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
@pytest.mark.asyncio
|
|
289
|
+
async def test_chain_before_completion_retry_outranks_earlier_narrate() -> None:
|
|
290
|
+
hooks = ChainHooks(_Narrate(), _RetryWrite())
|
|
291
|
+
action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
|
|
292
|
+
assert action == CompletionAction(kind="retry", reason="empty_round")
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@pytest.mark.asyncio
|
|
296
|
+
async def test_chain_before_completion_retry_outranks_later_narrate() -> None:
|
|
297
|
+
hooks = ChainHooks(_RetryWrite(), _Narrate())
|
|
298
|
+
action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
|
|
299
|
+
assert action == CompletionAction(kind="retry", reason="empty_round")
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
@pytest.mark.asyncio
|
|
303
|
+
async def test_chain_before_completion_later_retry_wins_same_kind() -> None:
|
|
304
|
+
class _FirstRetry(NoopHooks):
|
|
305
|
+
async def before_completion(self, draft, ctx):
|
|
306
|
+
return CompletionAction(kind="retry", reason="deferred_execution")
|
|
307
|
+
|
|
308
|
+
hooks = ChainHooks(_FirstRetry(), _RetryWrite())
|
|
309
|
+
action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
|
|
310
|
+
assert action == CompletionAction(kind="retry", reason="empty_round")
|
|
311
|
+
|
|
312
|
+
|
|
257
313
|
# ---------------------------------------------------------------------------
|
|
258
314
|
# post_tool_result
|
|
259
315
|
# ---------------------------------------------------------------------------
|
|
@@ -147,7 +147,7 @@ async def test_long_session_old_results_folded_in_late_requests() -> None:
|
|
|
147
147
|
|
|
148
148
|
@pytest.mark.asyncio
|
|
149
149
|
async def test_long_session_runaway_reminder_fires() -> None:
|
|
150
|
-
"""35 reads without a
|
|
150
|
+
"""35 reads without a write is the runaway-exploration failure
|
|
151
151
|
mode; the reminder must land at the highest-recency position."""
|
|
152
152
|
provider, _ = await _run_session(
|
|
153
153
|
ChainHooks(ReminderHooks(max_tool_errors=16, rules=None))
|
|
@@ -155,9 +155,9 @@ async def test_long_session_runaway_reminder_fires() -> None:
|
|
|
155
155
|
hits = [
|
|
156
156
|
call
|
|
157
157
|
for call in provider.calls
|
|
158
|
-
if any("
|
|
158
|
+
if any("tool calls since the last write" in m.content_text for m in call)
|
|
159
159
|
]
|
|
160
160
|
assert hits, "runaway reminder never fired"
|
|
161
161
|
# Recency: in the firing request the reminder is the LAST message.
|
|
162
162
|
firing = hits[0]
|
|
163
|
-
assert "
|
|
163
|
+
assert "tool calls since the last write" in firing[-1].content_text
|
|
@@ -123,6 +123,195 @@ async def test_tool_round_then_completion() -> None:
|
|
|
123
123
|
assert '"success": true' in tool_msgs[0].content_text
|
|
124
124
|
|
|
125
125
|
|
|
126
|
+
@pytest.mark.asyncio
|
|
127
|
+
async def test_tool_result_image_reaches_the_next_request() -> None:
|
|
128
|
+
"""OpenAI tool messages are text-only; pixels follow as a user turn."""
|
|
129
|
+
from steerable_agent_runtime.llm.parts import ImagePart
|
|
130
|
+
from steerable_agent_runtime.llm.openai_compat import _encode_message
|
|
131
|
+
|
|
132
|
+
provider = make_provider(
|
|
133
|
+
[
|
|
134
|
+
{"content": "", "tool_calls": [tc("peek")]},
|
|
135
|
+
{"content": "saw it"},
|
|
136
|
+
]
|
|
137
|
+
)
|
|
138
|
+
router = ToolRouter()
|
|
139
|
+
|
|
140
|
+
async def peek() -> ToolResult:
|
|
141
|
+
return ToolResult(
|
|
142
|
+
success=True,
|
|
143
|
+
data={
|
|
144
|
+
"path": "/app/code.png",
|
|
145
|
+
"content": "PNG 4x2 ASCII preview",
|
|
146
|
+
"kind": "png_ascii",
|
|
147
|
+
"_image": {"b64": "QUJD", "media_type": "image/png"},
|
|
148
|
+
},
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
router.register(peek)
|
|
152
|
+
loop = CoreLoop(provider, RouterToolExecutor(router))
|
|
153
|
+
await collect(loop.run([LLMMessage.text_of("user", "look")]))
|
|
154
|
+
second = provider.calls[1]
|
|
155
|
+
tool_msgs = [m for m in second if m.role == "tool"]
|
|
156
|
+
assert len(tool_msgs) == 1
|
|
157
|
+
assert "_image" not in tool_msgs[0].content_text
|
|
158
|
+
assert "pixels" in tool_msgs[0].content_text
|
|
159
|
+
assert isinstance(_encode_message(tool_msgs[0])["content"], str)
|
|
160
|
+
image_msgs = [
|
|
161
|
+
m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
|
|
162
|
+
]
|
|
163
|
+
assert len(image_msgs) == 1
|
|
164
|
+
tool_i = next(i for i, m in enumerate(second) if m.role == "tool")
|
|
165
|
+
img_i = next(i for i, m in enumerate(second) if m is image_msgs[0])
|
|
166
|
+
assert tool_i < img_i
|
|
167
|
+
encoded = _encode_message(image_msgs[0])
|
|
168
|
+
assert encoded["role"] == "user"
|
|
169
|
+
assert encoded["content"][-1] == {
|
|
170
|
+
"type": "image_url",
|
|
171
|
+
"image_url": {"url": "data:image/png;base64,QUJD"},
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@pytest.mark.asyncio
|
|
176
|
+
async def test_tool_result_image_survives_spill() -> None:
|
|
177
|
+
"""Pop ``_image`` before spill; a base64 PNG would always exceed 16 KB."""
|
|
178
|
+
from steerable_agent_runtime.llm.parts import ImagePart
|
|
179
|
+
from steerable_agent_runtime.spill import InMemorySpillStore, SpillHooks
|
|
180
|
+
|
|
181
|
+
b64 = "A" * 20_000
|
|
182
|
+
provider = make_provider(
|
|
183
|
+
[
|
|
184
|
+
{"content": "", "tool_calls": [tc("peek")]},
|
|
185
|
+
{"content": "saw it"},
|
|
186
|
+
]
|
|
187
|
+
)
|
|
188
|
+
router = ToolRouter()
|
|
189
|
+
|
|
190
|
+
async def peek() -> ToolResult:
|
|
191
|
+
return ToolResult(
|
|
192
|
+
success=True,
|
|
193
|
+
data={
|
|
194
|
+
"path": "/app/code.png",
|
|
195
|
+
"content": "PNG ASCII preview",
|
|
196
|
+
"kind": "png_ascii",
|
|
197
|
+
"_image": {"b64": b64, "media_type": "image/png"},
|
|
198
|
+
},
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
router.register(peek)
|
|
202
|
+
loop = CoreLoop(
|
|
203
|
+
provider,
|
|
204
|
+
RouterToolExecutor(router),
|
|
205
|
+
hooks=SpillHooks(InMemorySpillStore(), max_inline_bytes=16_000),
|
|
206
|
+
)
|
|
207
|
+
await collect(loop.run([LLMMessage.text_of("user", "look")]))
|
|
208
|
+
second = provider.calls[1]
|
|
209
|
+
tool_msgs = [m for m in second if m.role == "tool"]
|
|
210
|
+
assert "_image" not in tool_msgs[0].content_text
|
|
211
|
+
assert '"spilled": true' not in tool_msgs[0].content_text
|
|
212
|
+
image_msgs = [
|
|
213
|
+
m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
|
|
214
|
+
]
|
|
215
|
+
assert len(image_msgs) == 1
|
|
216
|
+
image = next(p for p in image_msgs[0].content if isinstance(p, ImagePart))
|
|
217
|
+
assert image.source == b64
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@pytest.mark.asyncio
|
|
221
|
+
async def test_two_read_images_share_one_user_message_after_tools() -> None:
|
|
222
|
+
"""OpenAI pairing: all tool messages, then one user turn with both images."""
|
|
223
|
+
from steerable_agent_runtime.llm.parts import ImagePart
|
|
224
|
+
|
|
225
|
+
provider = make_provider(
|
|
226
|
+
[
|
|
227
|
+
{
|
|
228
|
+
"content": "",
|
|
229
|
+
"tool_calls": [
|
|
230
|
+
tc("peek_a", call_id="c_a"),
|
|
231
|
+
tc("peek_b", call_id="c_b"),
|
|
232
|
+
],
|
|
233
|
+
},
|
|
234
|
+
{"content": "saw both"},
|
|
235
|
+
]
|
|
236
|
+
)
|
|
237
|
+
router = ToolRouter()
|
|
238
|
+
|
|
239
|
+
async def peek_a() -> ToolResult:
|
|
240
|
+
return ToolResult(
|
|
241
|
+
success=True,
|
|
242
|
+
data={
|
|
243
|
+
"path": "/app/a.png",
|
|
244
|
+
"content": "A",
|
|
245
|
+
"_image": {"b64": "QQ==", "media_type": "image/png"},
|
|
246
|
+
},
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
async def peek_b() -> ToolResult:
|
|
250
|
+
return ToolResult(
|
|
251
|
+
success=True,
|
|
252
|
+
data={
|
|
253
|
+
"path": "/app/b.png",
|
|
254
|
+
"content": "B",
|
|
255
|
+
"_image": {"b64": "Qg==", "media_type": "image/png"},
|
|
256
|
+
},
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
router.register(peek_a)
|
|
260
|
+
router.register(peek_b)
|
|
261
|
+
loop = CoreLoop(provider, RouterToolExecutor(router))
|
|
262
|
+
await collect(loop.run([LLMMessage.text_of("user", "look")]))
|
|
263
|
+
second = provider.calls[1]
|
|
264
|
+
roles = [m.role for m in second]
|
|
265
|
+
tool_idxs = [i for i, role in enumerate(roles) if role == "tool"]
|
|
266
|
+
image_idxs = [
|
|
267
|
+
i
|
|
268
|
+
for i, m in enumerate(second)
|
|
269
|
+
if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
|
|
270
|
+
]
|
|
271
|
+
assert len(tool_idxs) == 2
|
|
272
|
+
assert len(image_idxs) == 1
|
|
273
|
+
assert tool_idxs[-1] < image_idxs[0]
|
|
274
|
+
images = [p for p in second[image_idxs[0]].content if isinstance(p, ImagePart)]
|
|
275
|
+
assert [p.source for p in images] == ["QQ==", "Qg=="]
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
@pytest.mark.asyncio
|
|
279
|
+
async def test_read_images_cap_at_four_per_round() -> None:
|
|
280
|
+
from steerable_agent_runtime.llm.parts import ImagePart
|
|
281
|
+
|
|
282
|
+
names = [f"peek{i}" for i in range(5)]
|
|
283
|
+
provider = make_provider(
|
|
284
|
+
[
|
|
285
|
+
{"content": "", "tool_calls": [tc(n) for n in names]},
|
|
286
|
+
{"content": "saw it"},
|
|
287
|
+
]
|
|
288
|
+
)
|
|
289
|
+
router = ToolRouter()
|
|
290
|
+
for i, name in enumerate(names):
|
|
291
|
+
async def peek(
|
|
292
|
+
path: str = f"/app/{i}.png", b64: str = f"IMG{i}"
|
|
293
|
+
) -> ToolResult:
|
|
294
|
+
return ToolResult(
|
|
295
|
+
success=True,
|
|
296
|
+
data={
|
|
297
|
+
"path": path,
|
|
298
|
+
"content": "preview",
|
|
299
|
+
"_image": {"b64": b64, "media_type": "image/png"},
|
|
300
|
+
},
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
router.register(peek, name=name)
|
|
304
|
+
loop = CoreLoop(provider, RouterToolExecutor(router))
|
|
305
|
+
await collect(loop.run([LLMMessage.text_of("user", "look")]))
|
|
306
|
+
second = provider.calls[1]
|
|
307
|
+
image_msgs = [
|
|
308
|
+
m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
|
|
309
|
+
]
|
|
310
|
+
assert len(image_msgs) == 1
|
|
311
|
+
images = [p for p in image_msgs[0].content if isinstance(p, ImagePart)]
|
|
312
|
+
assert [p.source for p in images] == ["IMG0", "IMG1", "IMG2", "IMG3"]
|
|
313
|
+
|
|
314
|
+
|
|
126
315
|
@pytest.mark.asyncio
|
|
127
316
|
async def test_loop_echoes_reasoning_details_after_tools() -> None:
|
|
128
317
|
"""OpenRouter GLM continues thinking only if the prior details come back."""
|
|
@@ -91,7 +91,7 @@ async def test_runaway_exploration_fires_without_writes() -> None:
|
|
|
91
91
|
|
|
92
92
|
|
|
93
93
|
@pytest.mark.asyncio
|
|
94
|
-
async def
|
|
94
|
+
async def test_runaway_exploration_silent_right_after_a_write() -> None:
|
|
95
95
|
hooks = ReminderHooks(max_tool_errors=4, rules=ReminderRules(runaway_calls=3))
|
|
96
96
|
ctx = _Ctx(round_index=1)
|
|
97
97
|
for i in range(3):
|
|
@@ -109,6 +109,30 @@ async def test_runaway_exploration_silent_after_a_write() -> None:
|
|
|
109
109
|
assert not action.appends
|
|
110
110
|
|
|
111
111
|
|
|
112
|
+
@pytest.mark.asyncio
|
|
113
|
+
async def test_runaway_exploration_fires_after_write_then_inspect() -> None:
|
|
114
|
+
"""make-mips 33547943349: vm.js existed, then 200 bash greps. A lifetime
|
|
115
|
+
write flag would have silenced the reminder for the rest of the run."""
|
|
116
|
+
hooks = ReminderHooks(max_tool_errors=4, rules=ReminderRules(runaway_calls=3))
|
|
117
|
+
ctx = _Ctx(round_index=1)
|
|
118
|
+
await hooks.post_tool_result(
|
|
119
|
+
ToolResult(success=True, data={}),
|
|
120
|
+
ToolCall(id="w", name="write_file", arguments={}),
|
|
121
|
+
ctx,
|
|
122
|
+
)
|
|
123
|
+
for i in range(3):
|
|
124
|
+
await hooks.post_tool_result(
|
|
125
|
+
ToolResult(success=True, data={}),
|
|
126
|
+
ToolCall(id=str(i), name="read_file", arguments={}),
|
|
127
|
+
ctx,
|
|
128
|
+
)
|
|
129
|
+
action = await hooks.pre_step([], ctx)
|
|
130
|
+
assert action.appends
|
|
131
|
+
fragment = action.appends[0].fragment
|
|
132
|
+
assert isinstance(fragment, RunawayExplorationReminder)
|
|
133
|
+
assert "3 tool calls since the last write" in fragment.render()
|
|
134
|
+
|
|
135
|
+
|
|
112
136
|
@pytest.mark.asyncio
|
|
113
137
|
async def test_refire_waits_for_the_configured_gap() -> None:
|
|
114
138
|
hooks = ReminderHooks(
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/mcp.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/otel.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/retry.py
RENAMED
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/spill.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tools.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_antihallucination.py
RENAMED
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_approval_policy.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_cache_instrumentation.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_error_taxonomy.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_fragment_bounds.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_history_persistence.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_in_memory_storage.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_llm_wire_helpers.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_cancellation.py
RENAMED
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_sandbox_event.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_equivalence.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_observation_aging.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_parallel_tools.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_provider_compat.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_replay_crosslang.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_sqlite_storage.py
RENAMED
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_storage_contract.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_trace_recorder.py
RENAMED
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_transport_jsonrpc.py
RENAMED
|
File without changes
|
|
File without changes
|
{steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_usage_attribution.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|