steerable-agent-runtime 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/PKG-INFO +1 -1
  2. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/pyproject.toml +1 -1
  3. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/harness.py +24 -2
  4. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/hooks.py +20 -4
  5. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/loop.py +72 -10
  6. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/reminders.py +19 -20
  7. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/skills.py +5 -2
  8. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/PKG-INFO +1 -1
  9. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_harness.py +25 -0
  10. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_hooks.py +56 -0
  11. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_long_session.py +3 -3
  12. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop.py +189 -0
  13. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_reminders.py +25 -1
  14. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/README.md +0 -0
  15. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/setup.cfg +0 -0
  16. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/__init__.py +0 -0
  17. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/antihallucination.py +0 -0
  18. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/approval.py +0 -0
  19. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/approval_policy.py +0 -0
  20. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/branch.py +0 -0
  21. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/cache_control.py +0 -0
  22. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/calibration.py +0 -0
  23. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/compaction.py +0 -0
  24. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/default.harness.json +0 -0
  25. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/default.harness.yaml +0 -0
  26. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/errors.py +0 -0
  27. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/handoff.py +0 -0
  28. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/harness_spec.py +0 -0
  29. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/history.py +0 -0
  30. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/__init__.py +0 -0
  31. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/anthropic_native.py +0 -0
  32. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/compat.py +0 -0
  33. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/errors.py +0 -0
  34. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/openai_compat.py +0 -0
  35. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/parts.py +0 -0
  36. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/llm/system_proxy.py +0 -0
  37. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/maintenance.py +0 -0
  38. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/mcp.py +0 -0
  39. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/mcp_server.py +0 -0
  40. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_catalog.py +0 -0
  41. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_info.py +0 -0
  42. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/model_resolve.py +0 -0
  43. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/observation_aging.py +0 -0
  44. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/orchestration.py +0 -0
  45. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/otel.py +0 -0
  46. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/pricing.py +0 -0
  47. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/pseudo.py +0 -0
  48. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/recording.py +0 -0
  49. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/replay.py +0 -0
  50. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/resume.py +0 -0
  51. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/retry.py +0 -0
  52. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/sandboxed.py +0 -0
  53. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/spill.py +0 -0
  54. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/__init__.py +0 -0
  55. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/in_memory.py +0 -0
  56. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/sqlalchemy_store.py +0 -0
  57. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/sqlite_store.py +0 -0
  58. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/storage/write_lease.py +0 -0
  59. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/subagent.py +0 -0
  60. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tokens.py +0 -0
  61. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tool_schema.py +0 -0
  62. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tool_search.py +0 -0
  63. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tools.py +0 -0
  64. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/tracing.py +0 -0
  65. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/__init__.py +0 -0
  66. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/fastapi_sse.py +0 -0
  67. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/transport/stdio_jsonrpc.py +0 -0
  68. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime/world_state.py +0 -0
  69. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/SOURCES.txt +0 -0
  70. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/dependency_links.txt +0 -0
  71. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/requires.txt +0 -0
  72. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/src/steerable_agent_runtime.egg-info/top_level.txt +0 -0
  73. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_antihallucination.py +0 -0
  74. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_approval.py +0 -0
  75. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_approval_policy.py +0 -0
  76. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_branch.py +0 -0
  77. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_cache_control.py +0 -0
  78. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_cache_instrumentation.py +0 -0
  79. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_calibration.py +0 -0
  80. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_compaction.py +0 -0
  81. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_content_parts.py +0 -0
  82. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_error_taxonomy.py +0 -0
  83. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_fragment_bounds.py +0 -0
  84. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_golden.py +0 -0
  85. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_handoff.py +0 -0
  86. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_harness_spec.py +0 -0
  87. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_history.py +0 -0
  88. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_history_persistence.py +0 -0
  89. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_in_memory_storage.py +0 -0
  90. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_llm_wire_helpers.py +0 -0
  91. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_cancellation.py +0 -0
  92. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_replay.py +0 -0
  93. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_loop_sandbox_event.py +0 -0
  94. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_maintenance.py +0 -0
  95. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_mcp.py +0 -0
  96. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_mcp_server.py +0 -0
  97. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_catalog.py +0 -0
  98. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_equivalence.py +0 -0
  99. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_info.py +0 -0
  100. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_model_resolve.py +0 -0
  101. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_observation_aging.py +0 -0
  102. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_orchestration.py +0 -0
  103. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_otel.py +0 -0
  104. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_parallel_tools.py +0 -0
  105. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_provider_compat.py +0 -0
  106. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_pseudo.py +0 -0
  107. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_recording.py +0 -0
  108. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_replay_crosslang.py +0 -0
  109. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_resume.py +0 -0
  110. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_retry_hooks.py +0 -0
  111. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_safety_gate.py +0 -0
  112. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_sandboxed.py +0 -0
  113. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_skills.py +0 -0
  114. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_soft_timeout.py +0 -0
  115. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_spill.py +0 -0
  116. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_sqlite_storage.py +0 -0
  117. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_steer.py +0 -0
  118. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_storage_contract.py +0 -0
  119. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_stream_strip.py +0 -0
  120. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_subagent.py +0 -0
  121. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_system_proxy.py +0 -0
  122. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tokens.py +0 -0
  123. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_exposure.py +0 -0
  124. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_hygiene.py +0 -0
  125. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_router.py +0 -0
  126. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_schema.py +0 -0
  127. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_tool_timeout.py +0 -0
  128. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_trace_recorder.py +0 -0
  129. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_transport_jsonrpc.py +0 -0
  130. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_transport_sse.py +0 -0
  131. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_usage_attribution.py +0 -0
  132. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_world_state.py +0 -0
  133. {steerable_agent_runtime-0.4.0 → steerable_agent_runtime-0.5.0}/tests/test_write_lease.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steerable-agent-runtime
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Steerable agent runtime: LLM, tool, storage, and transport adapters.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "steerable-agent-runtime"
3
- version = "0.4.0"
3
+ version = "0.5.0"
4
4
  description = "Steerable agent runtime: LLM, tool, storage, and transport adapters."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -371,6 +371,10 @@ class NullValidator:
371
371
  class SelfCritique:
372
372
  """The existing discipline-retry + grounding stack (AntiHallucinationHooks)."""
373
373
 
374
+ #: Headless fills this from the TB instruction. Empty leaves claimed /
375
+ #: eager-deferred blind (``detect_execution_intent_in_user_message``
376
+ #: is false on "") and the grounding judge sees no user question.
377
+ user_question: str = ""
374
378
  name: str = "self_critique"
375
379
  assumes: str = (
376
380
  "the model sometimes claims executions it did not perform; a "
@@ -380,9 +384,27 @@ class SelfCritique:
380
384
  def hooks(self, *, provider: LLMProvider | None = None) -> LoopHooks:
381
385
  if provider is None:
382
386
  raise ValueError("SelfCritique requires the turn's LLM provider")
383
- from .antihallucination import AntiHallucinationHooks
387
+ from .antihallucination import (
388
+ AntiHallucinationConfig,
389
+ AntiHallucinationHooks,
390
+ detect_execution_intent_in_user_message,
391
+ )
384
392
 
385
- return _BeforeCompletionOnly(AntiHallucinationHooks(provider))
393
+ question = self.user_question
394
+ # Desktop exec-intent is chat ("跑一下" / "execute"). TB instructions
395
+ # are unattended execution even when they never say "run".
396
+ if question and not detect_execution_intent_in_user_message(question):
397
+ question = f"Execute the following.\n{question}"
398
+ return _BeforeCompletionOnly(
399
+ AntiHallucinationHooks(
400
+ provider,
401
+ AntiHallucinationConfig(
402
+ user_question=question,
403
+ tools_available=True,
404
+ enable_routing=False,
405
+ ),
406
+ )
407
+ )
386
408
 
387
409
 
388
410
  # ---------------------------------------------------------------------------
@@ -150,6 +150,18 @@ class CompletionAction:
150
150
  reason: str | None = None
151
151
 
152
152
 
153
+ _COMPLETION_KIND_RANK = {"accept": 0, "narrate": 1, "retry": 2}
154
+
155
+
156
+ def _completion_outranks(new: CompletionAction, current: CompletionAction) -> bool:
157
+ """Prefer forcing another working round over a no-tools summary."""
158
+ new_rank = _COMPLETION_KIND_RANK[new.kind]
159
+ old_rank = _COMPLETION_KIND_RANK[current.kind]
160
+ if new_rank != old_rank:
161
+ return new_rank > old_rank
162
+ return new.kind != "accept"
163
+
164
+
153
165
  @dataclass(slots=True)
154
166
  class RetryAction:
155
167
  """Outcome of an ``on_request_error`` hook.
@@ -285,7 +297,10 @@ class ChainHooks:
285
297
  - ``post_tool_result``: the result threads through each hook in order.
286
298
  - ``on_request_error``: the first ``retry`` decision wins; if every hook
287
299
  says ``fail``, the first failure reason is surfaced.
288
- - ``before_completion``: the first non-``accept`` action wins.
300
+ - ``before_completion``: every hook runs. ``retry`` outranks
301
+ ``narrate``, which outranks ``accept``. Same-rank later hooks win,
302
+ so a trailing delivery gate still forces writes when an earlier
303
+ validator would only ask for a no-tools summary.
289
304
  - ``wrap_up_may_drop_tools``: False if any hook forbids dropping tools.
290
305
 
291
306
  This is how a product stacks e.g. compaction + spill + retry without the
@@ -364,11 +379,12 @@ class ChainHooks:
364
379
  async def before_completion(
365
380
  self, draft: CompletionDraft, ctx: LoopContext
366
381
  ) -> CompletionAction:
382
+ picked = CompletionAction(kind="accept")
367
383
  for hook in self._hooks:
368
384
  action = await hook.before_completion(draft, ctx)
369
- if action.kind != "accept":
370
- return action
371
- return CompletionAction(kind="accept")
385
+ if _completion_outranks(action, picked):
386
+ picked = action
387
+ return picked
372
388
 
373
389
  def on_stream_chunk(self, chunk: Any, ctx: LoopContext) -> None:
374
390
  """Fan the chunk out to every hook; one bad hook must not break the rest."""
@@ -76,7 +76,7 @@ from .history import (
76
76
  entry_to_dict,
77
77
  )
78
78
  from .hooks import CompletionDraft, LoopHooks, NoopHooks
79
- from .llm import LLMMessage, LLMProvider, LLMUsage
79
+ from .llm import ImagePart, LLMMessage, LLMProvider, LLMUsage, TextPart
80
80
  from .pseudo import (
81
81
  PseudoStreamStripper,
82
82
  extract_inline_tool_calls,
@@ -1641,6 +1641,7 @@ class CoreLoop:
1641
1641
 
1642
1642
  breaker_tripped = False
1643
1643
  steer_interrupted = False
1644
+ round_images: list[tuple[str, ImagePart]] = []
1644
1645
  for batch_idx, (batch_safe, batch) in enumerate(batches):
1645
1646
  if breaker_tripped or steer_interrupted:
1646
1647
  break
@@ -1845,6 +1846,19 @@ class CoreLoop:
1845
1846
  else:
1846
1847
  ctx.consecutive_tool_errors += 1
1847
1848
 
1849
+ # Pop pixels before spill: a base64 PNG in ``data`` is
1850
+ # always larger than the 16 KB inline budget, and spilling
1851
+ # it would hide the image from the next request. OpenAI
1852
+ # tool messages are text-only, so the pixels ride on a
1853
+ # user message after this batch (``_append_read_images``).
1854
+ path = ""
1855
+ if isinstance(result.data, dict):
1856
+ path = str(result.data.get("path") or "")
1857
+ image = _pop_result_image(result)
1858
+ if image is not None:
1859
+ if isinstance(result.data, dict):
1860
+ result.data["pixels"] = "attached"
1861
+ round_images.append((path or call.name, image))
1848
1862
  # ── hook: post_tool_result (spill / truncation) ──────
1849
1863
  result = await self._hooks.post_tool_result(result, call, ctx)
1850
1864
 
@@ -1897,14 +1911,7 @@ class CoreLoop:
1897
1911
  **({"error": result.error} if result.error else {}),
1898
1912
  },
1899
1913
  )
1900
- manager.append(
1901
- LLMMessage.text_of(
1902
- "tool",
1903
- _result_content(result),
1904
- name=call.name,
1905
- tool_call_id=call.id,
1906
- )
1907
- )
1914
+ manager.append(_tool_result_message(call, result))
1908
1915
 
1909
1916
  # Approval abort ends the turn: record the batch like the
1910
1917
  # breaker does (real results for executed calls, synthetic
@@ -2055,6 +2062,7 @@ class CoreLoop:
2055
2062
  # above — skip the "executing" bookkeeping and run the wrap-up
2056
2063
  # round directly. Artifact wrap-up still executes tools, so it
2057
2064
  # falls through and increments like a normal round.
2065
+ _append_read_images(manager, round_images)
2058
2066
  if wrap_up and withholding_tools():
2059
2067
  continue
2060
2068
  if wrap_up:
@@ -2459,6 +2467,57 @@ def _step_summary(
2459
2467
  }
2460
2468
 
2461
2469
 
2470
+ _MAX_READ_IMAGES_PER_ROUND = 4
2471
+
2472
+
2473
+ def _pop_result_image(result: ToolResult) -> ImagePart | None:
2474
+ """Take ``data._image`` off the payload so spill and JSON never carry it."""
2475
+ data = result.data
2476
+ if not isinstance(data, dict):
2477
+ return None
2478
+ blob = data.pop("_image", None)
2479
+ if not isinstance(blob, dict):
2480
+ return None
2481
+ b64 = blob.get("b64")
2482
+ media = blob.get("media_type") or "image/png"
2483
+ if not isinstance(b64, str) or not b64:
2484
+ return None
2485
+ return ImagePart.from_base64(b64, media_type=str(media))
2486
+
2487
+
2488
+ def _tool_result_message(call: ToolCall, result: ToolResult) -> LLMMessage:
2489
+ """Tool observation as text. OpenAI forbids image_url on role=tool."""
2490
+ return LLMMessage.text_of(
2491
+ "tool", _result_content(result), name=call.name, tool_call_id=call.id
2492
+ )
2493
+
2494
+
2495
+ def _append_read_images(
2496
+ manager: ContextManager, images: list[tuple[str, ImagePart]]
2497
+ ) -> None:
2498
+ """Put PNG/JPEG pixels on a user turn after the tool-result batch.
2499
+
2500
+ OpenAI ``ChatCompletionToolMessageParam.content`` is string or text
2501
+ parts only. Claude Code's Anthropic wire can nest images in
2502
+ ``tool_result``; the Harbor GLM cell is OpenAI-compat, so the pixels
2503
+ follow the tool JSON as a user message instead.
2504
+ """
2505
+ if not images:
2506
+ return
2507
+ kept = images[:_MAX_READ_IMAGES_PER_ROUND]
2508
+ parts: list = [
2509
+ TextPart(
2510
+ "Pixels from the files just read. The tool JSON is an ASCII "
2511
+ "preview; look at these images for the actual contents."
2512
+ )
2513
+ ]
2514
+ for path, image in kept:
2515
+ if path:
2516
+ parts.append(TextPart(f"\n{path}:"))
2517
+ parts.append(image)
2518
+ manager.append(LLMMessage(role="user", content=parts), kind="tool.image")
2519
+
2520
+
2462
2521
  def _result_content(result: ToolResult) -> str:
2463
2522
  """Serialize a ToolResult into the tool-message content for the transcript."""
2464
2523
 
@@ -2468,7 +2527,10 @@ def _result_content(result: ToolResult) -> str:
2468
2527
  if result.error:
2469
2528
  payload["error"] = result.error
2470
2529
  if result.data is not None:
2471
- payload["data"] = result.data
2530
+ data = result.data
2531
+ if isinstance(data, dict) and "_image" in data:
2532
+ data = {k: v for k, v in data.items() if k != "_image"}
2533
+ payload["data"] = data
2472
2534
  if result.message:
2473
2535
  payload["message"] = result.message
2474
2536
  return json.dumps(payload, ensure_ascii=False)
@@ -71,7 +71,7 @@ class AbandonedRecoveryReminder(ContextFragment):
71
71
 
72
72
 
73
73
  class RunawayExplorationReminder(ContextFragment):
74
- """Many tool calls without a single write — exploration without output."""
74
+ """Many tool calls without a write since the last one."""
75
75
 
76
76
  content_kind = "reminder.runaway_exploration"
77
77
  max_tokens = 200
@@ -81,15 +81,14 @@ class RunawayExplorationReminder(ContextFragment):
81
81
 
82
82
  def body(self) -> str:
83
83
  return (
84
- f"[system notice] Exploration without output: {self._calls} tool "
85
- "calls so far, none of them a write or edit. If you are still "
86
- "exploring, say what you are looking for; if you have what you "
87
- "need, produce the artifact now."
84
+ f"[system notice] {self._calls} tool calls since the last write. "
85
+ "If the required files already exist, run them; if they do "
86
+ "not, write them now. Do not keep inspecting source."
88
87
  )
89
88
 
90
89
  @classmethod
91
90
  def type_markers(cls) -> tuple[str, str]:
92
- return ("[system notice] Exploration without output:", "")
91
+ return ("[system notice]", "Do not keep inspecting source.")
93
92
 
94
93
 
95
94
  # ---------------------------------------------------------------------------
@@ -157,7 +156,7 @@ REMINDER_CATALOG: tuple[ReminderEntry, ...] = (
157
156
  ),
158
157
  ReminderEntry(
159
158
  id="exploration.runaway",
160
- failure_mode="探索失控:大量读取类调用而零产出",
159
+ failure_mode="探索失控:连续非写入调用,含写过之后继续只读",
161
160
  fragment=RunawayExplorationReminder,
162
161
  ),
163
162
  )
@@ -190,7 +189,8 @@ class ReminderRules:
190
189
  #: Fire ``recovery.error_streak`` when consecutive errors reach this
191
190
  #: fraction of the loop's circuit breaker.
192
191
  error_streak_ratio: float = 0.5
193
- #: Fire ``exploration.runaway`` after this many tool calls without a write.
192
+ #: Fire ``exploration.runaway`` after this many tool calls without a
193
+ #: write since the last one (or since the start).
194
194
  runaway_calls: int = 12
195
195
  #: Re-fire a still-true rule after this many rounds (fade-out is the
196
196
  #: point, but every-round spam is noise).
@@ -200,11 +200,11 @@ class ReminderRules:
200
200
  class ReminderHooks(NoopHooks):
201
201
  """Fires catalog reminders from observed loop state.
202
202
 
203
- ``post_tool_result`` tracks the signals (error streaks, whether any
204
- write happened); ``pre_step`` evaluates the rules and appends a due
205
- reminder as the last transcript item before the request — the
206
- highest-recency position (W2.2.3). Each rule fires once, then re-fires
207
- only after ``refire_rounds`` while its condition still holds.
203
+ ``post_tool_result`` tracks error streaks and consecutive non-write
204
+ calls (a write resets that streak; a prior write does not silence
205
+ later inspect-only runs). ``pre_step`` appends a due reminder as the
206
+ last transcript item before the request. Each rule fires once, then
207
+ re-fires only after ``refire_rounds`` while its condition still holds.
208
208
  """
209
209
 
210
210
  def __init__(
@@ -217,14 +217,14 @@ class ReminderHooks(NoopHooks):
217
217
  self._rules = rules or ReminderRules()
218
218
  self._max_tool_errors = max_tool_errors
219
219
  self._write_tools = write_tools
220
- self._saw_write = False
221
- self._calls = 0
220
+ self._since_write = 0
222
221
  self._fired_at: dict[str, int] = {}
223
222
 
224
223
  async def post_tool_result(self, result: Any, call: Any, ctx: Any) -> Any:
225
- self._calls += 1
226
224
  if call.name in self._write_tools and getattr(result, "success", False):
227
- self._saw_write = True
225
+ self._since_write = 0
226
+ else:
227
+ self._since_write += 1
228
228
  return result
229
229
 
230
230
  async def pre_step(self, transcript: Any, ctx: Any) -> PreStepAction:
@@ -261,8 +261,7 @@ class ReminderHooks(NoopHooks):
261
261
  self._fired_at["recovery.error_streak"] = round_index
262
262
  return "recovery.error_streak"
263
263
  if (
264
- self._calls >= self._rules.runaway_calls
265
- and not self._saw_write
264
+ self._since_write >= self._rules.runaway_calls
266
265
  and ready("exploration.runaway")
267
266
  ):
268
267
  self._fired_at["exploration.runaway"] = round_index
@@ -275,5 +274,5 @@ class ReminderHooks(NoopHooks):
275
274
  getattr(ctx, "consecutive_tool_errors", 0), self._max_tool_errors
276
275
  )
277
276
  if entry.id == "exploration.runaway":
278
- return RunawayExplorationReminder(self._calls)
277
+ return RunawayExplorationReminder(self._since_write)
279
278
  return entry.fragment()
@@ -397,9 +397,12 @@ def _parse_skill_dir(skill_dir: Path) -> SkillDefinition | None:
397
397
  model_invocable = fm.get("disable-model-invocation") is not True
398
398
 
399
399
  # Same `{scripts}` resolution as the TS prompt assembly: point at the
400
- # skill's own scripts/ directory with uniform separators.
400
+ # skill's own scripts/ directory with uniform separators. The re.sub
401
+ # replacement MUST be a lambda: on Windows scripts_path carries
402
+ # backslashes, and a plain-string replacement would have its `\U`, `\n`
403
+ # etc. interpreted as regex escapes (re.error: bad escape \U).
401
404
  scripts_path = str(skill_dir / "scripts")
402
- content = re.sub(r"\{scripts\}[/\\]", scripts_path + "/", content)
405
+ content = re.sub(r"\{scripts\}[/\\]", lambda _m: scripts_path + "/", content)
403
406
  content = content.replace("{scripts}", scripts_path)
404
407
 
405
408
  return SkillDefinition(
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steerable-agent-runtime
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Steerable agent runtime: LLM, tool, storage, and transport adapters.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -164,6 +164,31 @@ async def test_null_validator_accepts() -> None:
164
164
  assert action.kind == "accept"
165
165
 
166
166
 
167
+ def test_self_critique_passes_the_instruction_into_the_judge() -> None:
168
+ """Harbor's self_critique arm is a silent no-op if user_question stays
169
+ empty: claimed/eager-deferred need exec intent, and the grounding judge
170
+ is prompted with an empty 用户提问."""
171
+ from steerable_agent_runtime.harness import SelfCritique
172
+
173
+ class _Provider:
174
+ name = "fake"
175
+ model = "fake-model"
176
+
177
+ hooks = SelfCritique(user_question="Write /app/out.txt").hooks(
178
+ provider=_Provider()
179
+ )
180
+ config = hooks._inner._config
181
+ assert config.user_question.startswith("Execute the following.")
182
+ assert "Write /app/out.txt" in config.user_question
183
+ assert config.tools_available is True
184
+ assert config.enable_routing is False
185
+
186
+ already = SelfCritique(user_question="execute the hidden tests").hooks(
187
+ provider=_Provider()
188
+ )
189
+ assert already._inner._config.user_question == "execute the hidden tests"
190
+
191
+
167
192
  # -- tools -------------------------------------------------------------------
168
193
 
169
194
 
@@ -15,6 +15,8 @@ from steerable_agent_protocol.generated import ToolCall, ToolResult
15
15
 
16
16
  from steerable_agent_runtime import (
17
17
  ChainHooks,
18
+ CompletionAction,
19
+ CompletionDraft,
18
20
  CoreLoop,
19
21
  LoopEvent,
20
22
  NoopHooks,
@@ -254,6 +256,60 @@ async def test_chain_hooks_appends_only_compose_in_order() -> None:
254
256
  assert [a.message.content_text for a in action.appends] == ["A", "B"]
255
257
 
256
258
 
259
+ # ---------------------------------------------------------------------------
260
+ # before_completion
261
+ # ---------------------------------------------------------------------------
262
+
263
+
264
+ def _draft(**overrides: Any) -> CompletionDraft:
265
+ values = dict(
266
+ status="failed",
267
+ reason="empty",
268
+ content="",
269
+ round_index=3,
270
+ had_tool_calls=False,
271
+ tool_calls_used=4,
272
+ tool_successes=2,
273
+ )
274
+ values.update(overrides)
275
+ return CompletionDraft(**values)
276
+
277
+
278
+ class _Narrate(NoopHooks):
279
+ async def before_completion(self, draft, ctx):
280
+ return CompletionAction(kind="narrate", reason="narration")
281
+
282
+
283
+ class _RetryWrite(NoopHooks):
284
+ async def before_completion(self, draft, ctx):
285
+ return CompletionAction(kind="retry", reason="empty_round")
286
+
287
+
288
+ @pytest.mark.asyncio
289
+ async def test_chain_before_completion_retry_outranks_earlier_narrate() -> None:
290
+ hooks = ChainHooks(_Narrate(), _RetryWrite())
291
+ action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
292
+ assert action == CompletionAction(kind="retry", reason="empty_round")
293
+
294
+
295
+ @pytest.mark.asyncio
296
+ async def test_chain_before_completion_retry_outranks_later_narrate() -> None:
297
+ hooks = ChainHooks(_RetryWrite(), _Narrate())
298
+ action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
299
+ assert action == CompletionAction(kind="retry", reason="empty_round")
300
+
301
+
302
+ @pytest.mark.asyncio
303
+ async def test_chain_before_completion_later_retry_wins_same_kind() -> None:
304
+ class _FirstRetry(NoopHooks):
305
+ async def before_completion(self, draft, ctx):
306
+ return CompletionAction(kind="retry", reason="deferred_execution")
307
+
308
+ hooks = ChainHooks(_FirstRetry(), _RetryWrite())
309
+ action = await hooks.before_completion(_draft(), ctx=None) # type: ignore[arg-type]
310
+ assert action == CompletionAction(kind="retry", reason="empty_round")
311
+
312
+
257
313
  # ---------------------------------------------------------------------------
258
314
  # post_tool_result
259
315
  # ---------------------------------------------------------------------------
@@ -147,7 +147,7 @@ async def test_long_session_old_results_folded_in_late_requests() -> None:
147
147
 
148
148
  @pytest.mark.asyncio
149
149
  async def test_long_session_runaway_reminder_fires() -> None:
150
- """35 reads without a single write is the runaway-exploration failure
150
+ """35 reads without a write is the runaway-exploration failure
151
151
  mode; the reminder must land at the highest-recency position."""
152
152
  provider, _ = await _run_session(
153
153
  ChainHooks(ReminderHooks(max_tool_errors=16, rules=None))
@@ -155,9 +155,9 @@ async def test_long_session_runaway_reminder_fires() -> None:
155
155
  hits = [
156
156
  call
157
157
  for call in provider.calls
158
- if any("Exploration without output" in m.content_text for m in call)
158
+ if any("tool calls since the last write" in m.content_text for m in call)
159
159
  ]
160
160
  assert hits, "runaway reminder never fired"
161
161
  # Recency: in the firing request the reminder is the LAST message.
162
162
  firing = hits[0]
163
- assert "Exploration without output" in firing[-1].content_text
163
+ assert "tool calls since the last write" in firing[-1].content_text
@@ -123,6 +123,195 @@ async def test_tool_round_then_completion() -> None:
123
123
  assert '"success": true' in tool_msgs[0].content_text
124
124
 
125
125
 
126
+ @pytest.mark.asyncio
127
+ async def test_tool_result_image_reaches_the_next_request() -> None:
128
+ """OpenAI tool messages are text-only; pixels follow as a user turn."""
129
+ from steerable_agent_runtime.llm.parts import ImagePart
130
+ from steerable_agent_runtime.llm.openai_compat import _encode_message
131
+
132
+ provider = make_provider(
133
+ [
134
+ {"content": "", "tool_calls": [tc("peek")]},
135
+ {"content": "saw it"},
136
+ ]
137
+ )
138
+ router = ToolRouter()
139
+
140
+ async def peek() -> ToolResult:
141
+ return ToolResult(
142
+ success=True,
143
+ data={
144
+ "path": "/app/code.png",
145
+ "content": "PNG 4x2 ASCII preview",
146
+ "kind": "png_ascii",
147
+ "_image": {"b64": "QUJD", "media_type": "image/png"},
148
+ },
149
+ )
150
+
151
+ router.register(peek)
152
+ loop = CoreLoop(provider, RouterToolExecutor(router))
153
+ await collect(loop.run([LLMMessage.text_of("user", "look")]))
154
+ second = provider.calls[1]
155
+ tool_msgs = [m for m in second if m.role == "tool"]
156
+ assert len(tool_msgs) == 1
157
+ assert "_image" not in tool_msgs[0].content_text
158
+ assert "pixels" in tool_msgs[0].content_text
159
+ assert isinstance(_encode_message(tool_msgs[0])["content"], str)
160
+ image_msgs = [
161
+ m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
162
+ ]
163
+ assert len(image_msgs) == 1
164
+ tool_i = next(i for i, m in enumerate(second) if m.role == "tool")
165
+ img_i = next(i for i, m in enumerate(second) if m is image_msgs[0])
166
+ assert tool_i < img_i
167
+ encoded = _encode_message(image_msgs[0])
168
+ assert encoded["role"] == "user"
169
+ assert encoded["content"][-1] == {
170
+ "type": "image_url",
171
+ "image_url": {"url": "data:image/png;base64,QUJD"},
172
+ }
173
+
174
+
175
+ @pytest.mark.asyncio
176
+ async def test_tool_result_image_survives_spill() -> None:
177
+ """Pop ``_image`` before spill; a base64 PNG would always exceed 16 KB."""
178
+ from steerable_agent_runtime.llm.parts import ImagePart
179
+ from steerable_agent_runtime.spill import InMemorySpillStore, SpillHooks
180
+
181
+ b64 = "A" * 20_000
182
+ provider = make_provider(
183
+ [
184
+ {"content": "", "tool_calls": [tc("peek")]},
185
+ {"content": "saw it"},
186
+ ]
187
+ )
188
+ router = ToolRouter()
189
+
190
+ async def peek() -> ToolResult:
191
+ return ToolResult(
192
+ success=True,
193
+ data={
194
+ "path": "/app/code.png",
195
+ "content": "PNG ASCII preview",
196
+ "kind": "png_ascii",
197
+ "_image": {"b64": b64, "media_type": "image/png"},
198
+ },
199
+ )
200
+
201
+ router.register(peek)
202
+ loop = CoreLoop(
203
+ provider,
204
+ RouterToolExecutor(router),
205
+ hooks=SpillHooks(InMemorySpillStore(), max_inline_bytes=16_000),
206
+ )
207
+ await collect(loop.run([LLMMessage.text_of("user", "look")]))
208
+ second = provider.calls[1]
209
+ tool_msgs = [m for m in second if m.role == "tool"]
210
+ assert "_image" not in tool_msgs[0].content_text
211
+ assert '"spilled": true' not in tool_msgs[0].content_text
212
+ image_msgs = [
213
+ m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
214
+ ]
215
+ assert len(image_msgs) == 1
216
+ image = next(p for p in image_msgs[0].content if isinstance(p, ImagePart))
217
+ assert image.source == b64
218
+
219
+
220
+ @pytest.mark.asyncio
221
+ async def test_two_read_images_share_one_user_message_after_tools() -> None:
222
+ """OpenAI pairing: all tool messages, then one user turn with both images."""
223
+ from steerable_agent_runtime.llm.parts import ImagePart
224
+
225
+ provider = make_provider(
226
+ [
227
+ {
228
+ "content": "",
229
+ "tool_calls": [
230
+ tc("peek_a", call_id="c_a"),
231
+ tc("peek_b", call_id="c_b"),
232
+ ],
233
+ },
234
+ {"content": "saw both"},
235
+ ]
236
+ )
237
+ router = ToolRouter()
238
+
239
+ async def peek_a() -> ToolResult:
240
+ return ToolResult(
241
+ success=True,
242
+ data={
243
+ "path": "/app/a.png",
244
+ "content": "A",
245
+ "_image": {"b64": "QQ==", "media_type": "image/png"},
246
+ },
247
+ )
248
+
249
+ async def peek_b() -> ToolResult:
250
+ return ToolResult(
251
+ success=True,
252
+ data={
253
+ "path": "/app/b.png",
254
+ "content": "B",
255
+ "_image": {"b64": "Qg==", "media_type": "image/png"},
256
+ },
257
+ )
258
+
259
+ router.register(peek_a)
260
+ router.register(peek_b)
261
+ loop = CoreLoop(provider, RouterToolExecutor(router))
262
+ await collect(loop.run([LLMMessage.text_of("user", "look")]))
263
+ second = provider.calls[1]
264
+ roles = [m.role for m in second]
265
+ tool_idxs = [i for i, role in enumerate(roles) if role == "tool"]
266
+ image_idxs = [
267
+ i
268
+ for i, m in enumerate(second)
269
+ if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
270
+ ]
271
+ assert len(tool_idxs) == 2
272
+ assert len(image_idxs) == 1
273
+ assert tool_idxs[-1] < image_idxs[0]
274
+ images = [p for p in second[image_idxs[0]].content if isinstance(p, ImagePart)]
275
+ assert [p.source for p in images] == ["QQ==", "Qg=="]
276
+
277
+
278
+ @pytest.mark.asyncio
279
+ async def test_read_images_cap_at_four_per_round() -> None:
280
+ from steerable_agent_runtime.llm.parts import ImagePart
281
+
282
+ names = [f"peek{i}" for i in range(5)]
283
+ provider = make_provider(
284
+ [
285
+ {"content": "", "tool_calls": [tc(n) for n in names]},
286
+ {"content": "saw it"},
287
+ ]
288
+ )
289
+ router = ToolRouter()
290
+ for i, name in enumerate(names):
291
+ async def peek(
292
+ path: str = f"/app/{i}.png", b64: str = f"IMG{i}"
293
+ ) -> ToolResult:
294
+ return ToolResult(
295
+ success=True,
296
+ data={
297
+ "path": path,
298
+ "content": "preview",
299
+ "_image": {"b64": b64, "media_type": "image/png"},
300
+ },
301
+ )
302
+
303
+ router.register(peek, name=name)
304
+ loop = CoreLoop(provider, RouterToolExecutor(router))
305
+ await collect(loop.run([LLMMessage.text_of("user", "look")]))
306
+ second = provider.calls[1]
307
+ image_msgs = [
308
+ m for m in second if m.role == "user" and any(isinstance(p, ImagePart) for p in m.content)
309
+ ]
310
+ assert len(image_msgs) == 1
311
+ images = [p for p in image_msgs[0].content if isinstance(p, ImagePart)]
312
+ assert [p.source for p in images] == ["IMG0", "IMG1", "IMG2", "IMG3"]
313
+
314
+
126
315
  @pytest.mark.asyncio
127
316
  async def test_loop_echoes_reasoning_details_after_tools() -> None:
128
317
  """OpenRouter GLM continues thinking only if the prior details come back."""
@@ -91,7 +91,7 @@ async def test_runaway_exploration_fires_without_writes() -> None:
91
91
 
92
92
 
93
93
  @pytest.mark.asyncio
94
- async def test_runaway_exploration_silent_after_a_write() -> None:
94
+ async def test_runaway_exploration_silent_right_after_a_write() -> None:
95
95
  hooks = ReminderHooks(max_tool_errors=4, rules=ReminderRules(runaway_calls=3))
96
96
  ctx = _Ctx(round_index=1)
97
97
  for i in range(3):
@@ -109,6 +109,30 @@ async def test_runaway_exploration_silent_after_a_write() -> None:
109
109
  assert not action.appends
110
110
 
111
111
 
112
+ @pytest.mark.asyncio
113
+ async def test_runaway_exploration_fires_after_write_then_inspect() -> None:
114
+ """make-mips 33547943349: vm.js existed, then 200 bash greps. A lifetime
115
+ write flag would have silenced the reminder for the rest of the run."""
116
+ hooks = ReminderHooks(max_tool_errors=4, rules=ReminderRules(runaway_calls=3))
117
+ ctx = _Ctx(round_index=1)
118
+ await hooks.post_tool_result(
119
+ ToolResult(success=True, data={}),
120
+ ToolCall(id="w", name="write_file", arguments={}),
121
+ ctx,
122
+ )
123
+ for i in range(3):
124
+ await hooks.post_tool_result(
125
+ ToolResult(success=True, data={}),
126
+ ToolCall(id=str(i), name="read_file", arguments={}),
127
+ ctx,
128
+ )
129
+ action = await hooks.pre_step([], ctx)
130
+ assert action.appends
131
+ fragment = action.appends[0].fragment
132
+ assert isinstance(fragment, RunawayExplorationReminder)
133
+ assert "3 tool calls since the last write" in fragment.render()
134
+
135
+
112
136
  @pytest.mark.asyncio
113
137
  async def test_refire_waits_for_the_configured_gap() -> None:
114
138
  hooks = ReminderHooks(