openrouter-agent-cli 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/PKG-INFO +13 -10
  2. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/README.md +12 -9
  3. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/cli.py +63 -4
  4. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/runner.py +12 -2
  5. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/PKG-INFO +13 -10
  6. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/pyproject.toml +1 -1
  7. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_runner.py +40 -0
  8. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_long_running.py +191 -0
  9. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/LICENSE +0 -0
  10. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/__init__.py +0 -0
  11. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/__main__.py +0 -0
  12. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/cache.py +0 -0
  13. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/completion.py +0 -0
  14. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/concurrent.py +0 -0
  15. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/discovery.py +0 -0
  16. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/__init__.py +0 -0
  17. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/audit.py +0 -0
  18. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/cli.py +0 -0
  19. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/compare.py +0 -0
  20. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/policy.py +0 -0
  21. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/records.py +0 -0
  22. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/report.py +0 -0
  23. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/sandbox.py +0 -0
  24. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/suite.py +0 -0
  25. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/transport.py +0 -0
  26. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/uncertainty.py +0 -0
  27. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/verify.py +0 -0
  28. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/selftest.py +0 -0
  29. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/utils.py +0 -0
  30. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/SOURCES.txt +0 -0
  31. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/dependency_links.txt +0 -0
  32. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/entry_points.txt +0 -0
  33. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/requires.txt +0 -0
  34. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/top_level.txt +0 -0
  35. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/setup.cfg +0 -0
  36. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_cli.py +0 -0
  37. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_discovery.py +0 -0
  38. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_audit.py +0 -0
  39. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_blockers.py +0 -0
  40. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_pins.py +0 -0
  41. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_policy.py +0 -0
  42. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_sandbox.py +0 -0
  43. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_uncertainty.py +0 -0
  44. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_output.py +0 -0
  45. {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_tool_safety.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: openrouter-agent-cli
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Standalone terminal agent for OpenRouter with tool actions and context management.
5
5
  License-Expression: MIT
6
6
  Project-URL: Homepage, https://github.com/protostatis/openrouter-agent-cli
@@ -94,11 +94,12 @@ openrouter-agent --workdir ./my-repo \
94
94
  --verify-command "pytest tests/test_auth.py"
95
95
  ```
96
96
 
97
- The command runs once at the completion boundary. A failing command receives at
98
- most one generic repair cycle; a timeout or execution error is reported as
99
- `not_verified` rather than treated as success. The workflow can be exercised
100
- without credentials or network access with `openrouter-agent-self-test` (or
101
- `openrouter-agent --self-test`).
97
+ The command runs at the completion boundary (once initially, and once more
98
+ after the single permitted repair cycle when the first check fails). A failing
99
+ command receives at most one generic repair cycle; a timeout or execution error
100
+ is reported as `not_verified` rather than treated as success. The workflow can
101
+ be exercised without credentials or network access with
102
+ `openrouter-agent-self-test` (or `openrouter-agent --self-test`).
102
103
 
103
104
  ## Useful flags
104
105
 
@@ -457,10 +458,12 @@ Evaluator artifacts:
457
458
  ## Offline evaluation workflow
458
459
 
459
460
  The repository includes a small evaluation workflow that runs the real CLI
460
- engine, real tool layer, and real task verifiers. The mock profile is fully
461
- offline: it makes no provider calls. Suite manifests and mock scripts are
462
- trusted operator-provided input: their commands execute on the host inside a
463
- disposable working directory, so review them like test code before running.
461
+ engine, real tool layer, and real task verifiers. The mock profile is
462
+ provider-offline: it makes no provider calls (mock-generated commands are host
463
+ shell commands, so they are not a network guarantee). Suite manifests and mock
464
+ scripts are trusted operator-provided input: their commands execute on the host
465
+ inside a disposable working directory, so review them like test code before
466
+ running.
464
467
 
465
468
  After installing the source checkout with `pip install -e .`:
466
469
 
@@ -67,11 +67,12 @@ openrouter-agent --workdir ./my-repo \
67
67
  --verify-command "pytest tests/test_auth.py"
68
68
  ```
69
69
 
70
- The command runs once at the completion boundary. A failing command receives at
71
- most one generic repair cycle; a timeout or execution error is reported as
72
- `not_verified` rather than treated as success. The workflow can be exercised
73
- without credentials or network access with `openrouter-agent-self-test` (or
74
- `openrouter-agent --self-test`).
70
+ The command runs at the completion boundary (once initially, and once more
71
+ after the single permitted repair cycle when the first check fails). A failing
72
+ command receives at most one generic repair cycle; a timeout or execution error
73
+ is reported as `not_verified` rather than treated as success. The workflow can
74
+ be exercised without credentials or network access with
75
+ `openrouter-agent-self-test` (or `openrouter-agent --self-test`).
75
76
 
76
77
  ## Useful flags
77
78
 
@@ -430,10 +431,12 @@ Evaluator artifacts:
430
431
  ## Offline evaluation workflow
431
432
 
432
433
  The repository includes a small evaluation workflow that runs the real CLI
433
- engine, real tool layer, and real task verifiers. The mock profile is fully
434
- offline: it makes no provider calls. Suite manifests and mock scripts are
435
- trusted operator-provided input: their commands execute on the host inside a
436
- disposable working directory, so review them like test code before running.
434
+ engine, real tool layer, and real task verifiers. The mock profile is
435
+ provider-offline: it makes no provider calls (mock-generated commands are host
436
+ shell commands, so they are not a network guarantee). Suite manifests and mock
437
+ scripts are trusted operator-provided input: their commands execute on the host
438
+ inside a disposable working directory, so review them like test code before
439
+ running.
437
440
 
438
441
  After installing the source checkout with `pip install -e .`:
439
442
 
@@ -773,9 +773,11 @@ class OpenRouterAgentCLI:
773
773
  self._batch_allow.clear()
774
774
  self._batch_deny.clear()
775
775
  stored_workdir = data.get("workdir", "")
776
+ workdir_mismatch = False
776
777
  if stored_workdir:
777
778
  current_workdir = str(Path(self.workdir).resolve())
778
779
  if stored_workdir != current_workdir:
780
+ workdir_mismatch = True
779
781
  print(
780
782
  _strip_control_chars(
781
783
  f"[session] Working directory changed since last session (was {stored_workdir}, now {current_workdir}). "
@@ -800,13 +802,21 @@ class OpenRouterAgentCLI:
800
802
  "last_check": stored_work_order.get("last_check"),
801
803
  "updated_at": stored_work_order.get("updated_at"),
802
804
  }
805
+ if workdir_mismatch:
806
+ # An acceptance result from another directory does not
807
+ # apply here; never display stale verified evidence.
808
+ self.work_order["status"] = "not_verified"
809
+ self.work_order["last_check"] = None
803
810
  stored_cache = data.get("cache_context")
804
811
  if self.cache_mode != "off" and isinstance(stored_cache, dict):
812
+ # Restore only cumulative observations. The pairwise
813
+ # stable-prefix fields are transient (there are no stored
814
+ # request hashes to back them), so they start fresh and are
815
+ # recomputed on the next request instead of displaying a
816
+ # stale prefix with no fingerprint.
805
817
  for name in (
806
818
  "requests",
807
819
  "compactions",
808
- "stable_prefix_tokens",
809
- "stable_prefix_messages",
810
820
  "observed_cached_tokens",
811
821
  "provider_cache_observations",
812
822
  ):
@@ -896,6 +906,7 @@ class OpenRouterAgentCLI:
896
906
  "exit_code": result.get("exit_code"),
897
907
  "timed_out": bool(result.get("timed_out")),
898
908
  "duration_ms": result.get("duration_ms"),
909
+ "checked_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
899
910
  "stdout": result.get("stdout", ""),
900
911
  "stderr": result.get("stderr", ""),
901
912
  "error": result.get("error", ""),
@@ -1563,6 +1574,7 @@ class OpenRouterAgentCLI:
1563
1574
  "%Y-%m-%dT%H:%M:%SZ", time.gmtime()
1564
1575
  )
1565
1576
  self._install_completion_policy()
1577
+ self._save_session()
1566
1578
  print("Acceptance check re-scoped to the new working directory (status reset to not_verified).")
1567
1579
  print(_strip_control_chars(f"Working directory set to: {self.workdir}"))
1568
1580
  print("Session-scoped permission grants cleared after cwd change.")
@@ -2980,6 +2992,34 @@ class OpenRouterAgentCLI:
2980
2992
  self._display_check_result(result)
2981
2993
  self._last_displayed_check = result
2982
2994
 
2995
+ def _emit_completion_summary(self) -> None:
2996
+ """Emit a deterministic completion notice when a repair response ended
2997
+ with tools and there is no model final text to show.
2998
+
2999
+ One-shot mode promises the assistant reply on stdout; without this the
3000
+ turn would end silently with an empty reply even after verified work.
3001
+ The notice is CLI-generated and clearly labeled, never fabricated
3002
+ model text, and applies only to the user acceptance policy.
3003
+ """
3004
+ if not isinstance(self.checkpoint_hook, UserCompletionPolicy):
3005
+ return
3006
+ if self.completion_policy is None:
3007
+ return
3008
+ result = self.completion_policy.last_result
3009
+ if not result:
3010
+ return
3011
+ status = str(result.get("status") or "not_verified").upper()
3012
+ command = str(result.get("command") or "")
3013
+ exit_code = result.get("exit_code")
3014
+ changed = list(result.get("changed_files") or [])
3015
+ code = f" (exit_code={exit_code})" if exit_code is not None else ""
3016
+ changed_text = f" Changed files: {', '.join(changed)}." if changed else ""
3017
+ summary = (
3018
+ f"[cli] Turn ended after the repair check: {status}: "
3019
+ f"{command}{code}{changed_text}"
3020
+ )
3021
+ self._output_response(summary)
3022
+
2983
3023
  async def _run_user_turn(self, client: httpx.AsyncClient, user_text: str) -> str:
2984
3024
  self._turn_allow.clear()
2985
3025
  self._turn_deny.clear()
@@ -3102,8 +3142,14 @@ class OpenRouterAgentCLI:
3102
3142
  or ""
3103
3143
  )
3104
3144
  self.messages.append(forced_message)
3105
- except Exception:
3106
- text = ""
3145
+ except Exception as e:
3146
+ # A loop-breaker failure is still a provider failure. Do not
3147
+ # fabricate a normal answer: record it like the primary
3148
+ # request handlers and terminate without grading the turn.
3149
+ self._log(f"[openrouter] Loop-breaker request failed: {e}")
3150
+ self.terminal_status = "provider_error"
3151
+ self._save_session()
3152
+ return ""
3107
3153
  if not text:
3108
3154
  text = "I got stuck in a tool loop and could not make progress."
3109
3155
  should_continue, result = await self._handle_final_answer(
@@ -3287,6 +3333,7 @@ class OpenRouterAgentCLI:
3287
3333
  continue
3288
3334
  self._display_policy_check()
3289
3335
  if action == "stop" or self._checkpoint_repair_pending:
3336
+ self._emit_completion_summary()
3290
3337
  self._save_session()
3291
3338
  return ""
3292
3339
 
@@ -3300,6 +3347,7 @@ class OpenRouterAgentCLI:
3300
3347
  )
3301
3348
  self._display_policy_check()
3302
3349
  self._apply_checkpoint_decision(decision)
3350
+ self._emit_completion_summary()
3303
3351
  self._save_session()
3304
3352
  return ""
3305
3353
 
@@ -3593,6 +3641,17 @@ def main() -> int:
3593
3641
  except KeyboardInterrupt:
3594
3642
  pass
3595
3643
 
3644
+ # One-shot mode must signal provider failures to callers instead of
3645
+ # exiting 0 after an empty turn.
3646
+ if args.prompt is not None and cli.terminal_status != "ok":
3647
+ print(
3648
+ f"[openrouter] request failed; turn ended with terminal_status="
3649
+ f"{cli.terminal_status}",
3650
+ file=sys.stderr,
3651
+ )
3652
+ return 1
3653
+ return 0
3654
+
3596
3655
 
3597
3656
  if __name__ == "__main__":
3598
3657
  raise SystemExit(main())
@@ -197,11 +197,18 @@ class SuiteRunner:
197
197
  engine.policy = ToolPermissionPolicy(allow={"*"})
198
198
  engine.non_interactive_mode = True
199
199
  engine.one_shot_prompt = task.prompt
200
- if getattr(self, "_sandboxed", False) and not profile.uses_mock:
200
+ if profile.uses_mock:
201
+ record["engine"]["execution_mode"] = "mock"
202
+ elif getattr(self, "_sandboxed", False):
201
203
  from .sandbox import BubblewrapBashRunner
202
204
 
203
205
  engine.bash_runner = BubblewrapBashRunner(str(workspace))
204
206
  record["engine"]["sandbox"] = "bubblewrap"
207
+ record["engine"]["execution_mode"] = "bubblewrap"
208
+ else:
209
+ # Explicitly acknowledged host-mode execution (operator dev
210
+ # checks only; never valid experiment evidence).
211
+ record["engine"]["execution_mode"] = "host_acknowledged"
205
212
  if policy is not None:
206
213
  engine.checkpoint_hook = policy
207
214
  if profile.uses_mock:
@@ -324,7 +331,10 @@ class SuiteRunner:
324
331
  expected_task_ids=[task.id for task in self.suite.tasks],
325
332
  expected_profile_names=[profile.name for profile in self.profiles],
326
333
  expected_repeats=self.repeats,
327
- require_containment=any(not profile.uses_mock for profile in self.profiles),
334
+ # Containment is required only for the mode that was actually
335
+ # executed: an explicitly acknowledged host-mode run is
336
+ # auditable as host execution, not silently rejected.
337
+ require_containment=self._sandboxed,
328
338
  )
329
339
  finally:
330
340
  self.cleanup_workspaces(records)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: openrouter-agent-cli
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Standalone terminal agent for OpenRouter with tool actions and context management.
5
5
  License-Expression: MIT
6
6
  Project-URL: Homepage, https://github.com/protostatis/openrouter-agent-cli
@@ -94,11 +94,12 @@ openrouter-agent --workdir ./my-repo \
94
94
  --verify-command "pytest tests/test_auth.py"
95
95
  ```
96
96
 
97
- The command runs once at the completion boundary. A failing command receives at
98
- most one generic repair cycle; a timeout or execution error is reported as
99
- `not_verified` rather than treated as success. The workflow can be exercised
100
- without credentials or network access with `openrouter-agent-self-test` (or
101
- `openrouter-agent --self-test`).
97
+ The command runs at the completion boundary (once initially, and once more
98
+ after the single permitted repair cycle when the first check fails). A failing
99
+ command receives at most one generic repair cycle; a timeout or execution error
100
+ is reported as `not_verified` rather than treated as success. The workflow can
101
+ be exercised without credentials or network access with
102
+ `openrouter-agent-self-test` (or `openrouter-agent --self-test`).
102
103
 
103
104
  ## Useful flags
104
105
 
@@ -457,10 +458,12 @@ Evaluator artifacts:
457
458
  ## Offline evaluation workflow
458
459
 
459
460
  The repository includes a small evaluation workflow that runs the real CLI
460
- engine, real tool layer, and real task verifiers. The mock profile is fully
461
- offline: it makes no provider calls. Suite manifests and mock scripts are
462
- trusted operator-provided input: their commands execute on the host inside a
463
- disposable working directory, so review them like test code before running.
461
+ engine, real tool layer, and real task verifiers. The mock profile is
462
+ provider-offline: it makes no provider calls (mock-generated commands are host
463
+ shell commands, so they are not a network guarantee). Suite manifests and mock
464
+ scripts are trusted operator-provided input: their commands execute on the host
465
+ inside a disposable working directory, so review them like test code before
466
+ running.
464
467
 
465
468
  After installing the source checkout with `pip install -e .`:
466
469
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "openrouter-agent-cli"
7
- version = "0.2.0"
7
+ version = "0.2.1"
8
8
  description = "Standalone terminal agent for OpenRouter with tool actions and context management."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -298,3 +298,43 @@ def test_unknown_assisted_profile_is_rejected(tmp_path: Path, suite: Path) -> No
298
298
  None,
299
299
  assisted_profiles={"worker", "ghost"},
300
300
  )
301
+
302
+
303
+ def test_host_mode_audit_uses_actual_containment(
304
+ tmp_path: Path, suite: Path, monkeypatch: pytest.MonkeyPatch
305
+ ) -> None:
306
+ """An explicitly acknowledged host-mode real run must be audited as host
307
+ execution (require_containment follows _sandboxed), not rejected."""
308
+ from openrouter_agent_cli.eval.records import make_record, new_run_id
309
+
310
+ monkeypatch.setenv("OPENROUTER_API_KEY", "test-key")
311
+ monkeypatch.setenv("AGENT_EVAL_ALLOW_HOST_EXECUTION", "1")
312
+ monkeypatch.setenv("AGENT_EVAL_SANDBOX", "0")
313
+ loaded = load_suite(suite)
314
+ runner = SuiteRunner(
315
+ loaded,
316
+ [Profile(name="real", prompt="P", model="m")],
317
+ eval_dir=tmp_path / "eval",
318
+ )
319
+ assert runner._sandboxed is False
320
+
321
+ captured: dict = {}
322
+ monkeypatch.setattr(
323
+ "openrouter_agent_cli.eval.runner.assert_audited",
324
+ lambda records, **kwargs: captured.update(kwargs),
325
+ )
326
+
327
+ async def fake_run():
328
+ r = make_record(
329
+ run_id=new_run_id(), suite_id=loaded.suite_id,
330
+ task_id=loaded.tasks[0].id, cluster_id=loaded.tasks[0].cluster_id,
331
+ profile_name="real", profile_prompt="P", model="m",
332
+ transport="openrouter", workdir=str(tmp_path), scheduled_index=0,
333
+ )
334
+ r["engine"] = {"execution_mode": "host_acknowledged", "error": None}
335
+ r["verdict"] = "infrastructure_error"
336
+ return [r]
337
+
338
+ monkeypatch.setattr(runner, "run", fake_run)
339
+ asyncio.run(runner.run_and_verify())
340
+ assert captured.get("require_containment") is False
@@ -2,6 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import json
5
6
  import os
6
7
  import shlex
7
8
  import sys
@@ -303,3 +304,193 @@ async def test_provider_failure_sets_terminal_status(tmp_path, monkeypatch):
303
304
  async with httpx.AsyncClient() as client:
304
305
  await cli._run_user_turn(client, "hi")
305
306
  assert cli.terminal_status == "provider_error"
307
+
308
+
309
+ @pytest.mark.asyncio
310
+ async def test_loop_breaker_failure_is_provider_error(tmp_path, monkeypatch):
311
+ """A failed forced loop-breaker call must be recorded as a provider error
312
+ and terminate, not fabricate a normal answer."""
313
+ session_dir = tmp_path / "sessions"
314
+ monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
315
+ cli = OpenRouterAgentCLI(
316
+ api_key="test-key",
317
+ model="test-model",
318
+ session_id="loop-break-test",
319
+ workdir=str(tmp_path),
320
+ max_turns=10,
321
+ max_history_messages=60,
322
+ command_timeout=5,
323
+ tools_enabled=True,
324
+ system_prompt=DEFAULT_SYSTEM_PROMPT,
325
+ discovery_mode="off",
326
+ )
327
+ cli.non_interactive_mode = True
328
+ cli.policy = ToolPermissionPolicy(allow={"*"})
329
+ repeated = {"name": "run_bash", "arguments": {"command": "true"}}
330
+
331
+ class FailingLoopBreaker:
332
+ def __init__(self):
333
+ self.requests = 0
334
+
335
+ async def __call__(self, client, **kwargs):
336
+ self.requests += 1
337
+ if kwargs.get("tool_choice") == "none":
338
+ raise RuntimeError("loop-breaker provider failure")
339
+ return {
340
+ "choices": [
341
+ {
342
+ "message": {
343
+ "role": "assistant",
344
+ "content": None,
345
+ "tool_calls": [
346
+ {
347
+ "id": f"tc-{self.requests}",
348
+ "type": "function",
349
+ "function": {
350
+ "name": "run_bash",
351
+ "arguments": '{"command": "true"}',
352
+ },
353
+ }
354
+ ],
355
+ },
356
+ "finish_reason": "tool_calls",
357
+ }
358
+ ],
359
+ "usage": {
360
+ "prompt_tokens": 1,
361
+ "completion_tokens": 1,
362
+ "total_tokens": 2,
363
+ },
364
+ }
365
+
366
+ transport = FailingLoopBreaker()
367
+ cli.model_transport = transport
368
+ async with httpx.AsyncClient() as client:
369
+ result = await cli._run_user_turn(client, "do it")
370
+ assert result == ""
371
+ assert cli.terminal_status == "provider_error"
372
+ # The repeated tool call was nudged; the forced request failed once.
373
+ assert transport.requests == 3
374
+
375
+
376
+ @pytest.mark.asyncio
377
+ async def test_tool_repair_emits_completion_notice(
378
+ tmp_path, monkeypatch, capsys
379
+ ):
380
+ """The tool-using repair path must print a deterministic completion notice
381
+ so one-shot mode is not left with an empty stdout."""
382
+ engine = _engine_with_task(
383
+ tmp_path,
384
+ monkeypatch,
385
+ task="Create marker.txt",
386
+ verify_command="test -f marker.txt",
387
+ responses=[
388
+ {"tool_calls": [
389
+ {"name": "write_file",
390
+ "arguments": {"path": "tmp.txt", "content": "x\n"}}
391
+ ]},
392
+ {"text": "I wrote tmp.txt, marker is next"},
393
+ {"tool_calls": [
394
+ {"name": "write_file",
395
+ "arguments": {"path": "marker.txt", "content": "ok\n"}}
396
+ ]},
397
+ {"text": "must not be requested"},
398
+ ],
399
+ )
400
+ async with httpx.AsyncClient() as client:
401
+ await engine._run_user_turn(client, "Do the work.")
402
+ out = capsys.readouterr().out
403
+ assert "Turn ended after the repair check: VERIFIED" in out
404
+
405
+
406
+ def test_cached_prefix_is_not_restored_on_resume(tmp_path, monkeypatch):
407
+ """Session loading restores cumulative counters but never transient
408
+ stable-prefix state (there are no stored request hashes to back it)."""
409
+ session_dir = tmp_path / "sessions"
410
+ monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
411
+ kwargs = {
412
+ "api_key": "test-key",
413
+ "model": "test-model",
414
+ "workdir": str(tmp_path),
415
+ "max_turns": 2,
416
+ "max_history_messages": 60,
417
+ "command_timeout": 5,
418
+ "tools_enabled": True,
419
+ "system_prompt": DEFAULT_SYSTEM_PROMPT,
420
+ "discovery_mode": "off",
421
+ }
422
+ first = OpenRouterAgentCLI(**kwargs, session_id="cache-resume")
423
+ prefix = [
424
+ {"role": "system", "content": "stable-system"},
425
+ {"role": "user", "content": "hello"},
426
+ ]
427
+ first.cache_context.observe_request(prefix)
428
+ first.cache_context.observe_request(prefix + [{"role": "assistant", "content": "hi"}])
429
+ assert first.cache_context.stable_prefix_tokens > 0
430
+ first._save_session()
431
+
432
+ resumed = OpenRouterAgentCLI(**kwargs, session_id="cache-resume")
433
+ assert resumed.cache_context.stable_prefix_tokens == 0
434
+ assert resumed.cache_context.stable_prefix_messages == 0
435
+ # Cumulative observations survive.
436
+ assert resumed.cache_context.requests == 2
437
+
438
+
439
+ def test_workdir_mismatch_invalidates_acceptance(tmp_path, monkeypatch):
440
+ """An acceptance result from another directory must never be shown as
441
+ verified after resuming in a different workdir."""
442
+ session_dir = tmp_path / "sessions"
443
+ other = tmp_path / "other"
444
+ other.mkdir()
445
+ monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
446
+ kwargs = {
447
+ "api_key": "test-key",
448
+ "model": "test-model",
449
+ "max_turns": 2,
450
+ "max_history_messages": 60,
451
+ "command_timeout": 5,
452
+ "tools_enabled": True,
453
+ "system_prompt": DEFAULT_SYSTEM_PROMPT,
454
+ "discovery_mode": "off",
455
+ }
456
+ first = OpenRouterAgentCLI(
457
+ **kwargs, session_id="workdir-switch", workdir=str(tmp_path),
458
+ task="Task X", verify_command="pytest -q",
459
+ )
460
+ first.work_order["status"] = "verified"
461
+ first._save_session()
462
+
463
+ resumed = OpenRouterAgentCLI(
464
+ **kwargs, session_id="workdir-switch", workdir=str(other)
465
+ )
466
+ assert resumed.work_order["objective"] == "Task X"
467
+ assert resumed.work_order["status"] == "not_verified"
468
+ assert resumed.work_order["last_check"] is None
469
+
470
+
471
+ @pytest.mark.asyncio
472
+ async def test_cwd_change_persists_contract_reset(tmp_path, monkeypatch):
473
+ session_dir = tmp_path / "sessions"
474
+ other = tmp_path / "other-project"
475
+ other.mkdir()
476
+ monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
477
+ cli = OpenRouterAgentCLI(
478
+ api_key="test-key",
479
+ model="test-model",
480
+ session_id="cwd-persist",
481
+ workdir=str(tmp_path),
482
+ max_turns=2,
483
+ max_history_messages=60,
484
+ command_timeout=5,
485
+ tools_enabled=True,
486
+ system_prompt=DEFAULT_SYSTEM_PROMPT,
487
+ discovery_mode="off",
488
+ task="Task X",
489
+ verify_command="pytest -q",
490
+ )
491
+ cli.work_order["status"] = "verified"
492
+ await cli._handle_command(None, f"/cwd {other}")
493
+
494
+ persisted = json.loads(cli._session_path.read_text())
495
+ assert persisted["work_order"]["status"] == "not_verified"
496
+ assert persisted["work_order"]["verify_command"] == "pytest -q"