openrouter-agent-cli 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/PKG-INFO +13 -10
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/README.md +12 -9
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/cli.py +63 -4
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/runner.py +12 -2
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/PKG-INFO +13 -10
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/pyproject.toml +1 -1
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_runner.py +40 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_long_running.py +191 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/LICENSE +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/__init__.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/__main__.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/cache.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/completion.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/concurrent.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/discovery.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/__init__.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/audit.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/cli.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/compare.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/policy.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/records.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/report.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/sandbox.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/suite.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/transport.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/uncertainty.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/verify.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/selftest.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/utils.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/SOURCES.txt +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/dependency_links.txt +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/entry_points.txt +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/requires.txt +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/top_level.txt +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/setup.cfg +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_cli.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_discovery.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_audit.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_blockers.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_pins.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_policy.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_sandbox.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_eval_uncertainty.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_output.py +0 -0
- {openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/tests/test_tool_safety.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: openrouter-agent-cli
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Standalone terminal agent for OpenRouter with tool actions and context management.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/protostatis/openrouter-agent-cli
|
|
@@ -94,11 +94,12 @@ openrouter-agent --workdir ./my-repo \
|
|
|
94
94
|
--verify-command "pytest tests/test_auth.py"
|
|
95
95
|
```
|
|
96
96
|
|
|
97
|
-
The command runs
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
97
|
+
The command runs at the completion boundary (once initially, and once more
|
|
98
|
+
after the single permitted repair cycle when the first check fails). A failing
|
|
99
|
+
command receives at most one generic repair cycle; a timeout or execution error
|
|
100
|
+
is reported as `not_verified` rather than treated as success. The workflow can
|
|
101
|
+
be exercised without credentials or network access with
|
|
102
|
+
`openrouter-agent-self-test` (or `openrouter-agent --self-test`).
|
|
102
103
|
|
|
103
104
|
## Useful flags
|
|
104
105
|
|
|
@@ -457,10 +458,12 @@ Evaluator artifacts:
|
|
|
457
458
|
## Offline evaluation workflow
|
|
458
459
|
|
|
459
460
|
The repository includes a small evaluation workflow that runs the real CLI
|
|
460
|
-
engine, real tool layer, and real task verifiers. The mock profile is
|
|
461
|
-
offline: it makes no provider calls
|
|
462
|
-
|
|
463
|
-
|
|
461
|
+
engine, real tool layer, and real task verifiers. The mock profile is
|
|
462
|
+
provider-offline: it makes no provider calls (mock-generated commands are host
|
|
463
|
+
shell commands, so they are not a network guarantee). Suite manifests and mock
|
|
464
|
+
scripts are trusted operator-provided input: their commands execute on the host
|
|
465
|
+
inside a disposable working directory, so review them like test code before
|
|
466
|
+
running.
|
|
464
467
|
|
|
465
468
|
After installing the source checkout with `pip install -e .`:
|
|
466
469
|
|
|
@@ -67,11 +67,12 @@ openrouter-agent --workdir ./my-repo \
|
|
|
67
67
|
--verify-command "pytest tests/test_auth.py"
|
|
68
68
|
```
|
|
69
69
|
|
|
70
|
-
The command runs
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
70
|
+
The command runs at the completion boundary (once initially, and once more
|
|
71
|
+
after the single permitted repair cycle when the first check fails). A failing
|
|
72
|
+
command receives at most one generic repair cycle; a timeout or execution error
|
|
73
|
+
is reported as `not_verified` rather than treated as success. The workflow can
|
|
74
|
+
be exercised without credentials or network access with
|
|
75
|
+
`openrouter-agent-self-test` (or `openrouter-agent --self-test`).
|
|
75
76
|
|
|
76
77
|
## Useful flags
|
|
77
78
|
|
|
@@ -430,10 +431,12 @@ Evaluator artifacts:
|
|
|
430
431
|
## Offline evaluation workflow
|
|
431
432
|
|
|
432
433
|
The repository includes a small evaluation workflow that runs the real CLI
|
|
433
|
-
engine, real tool layer, and real task verifiers. The mock profile is
|
|
434
|
-
offline: it makes no provider calls
|
|
435
|
-
|
|
436
|
-
|
|
434
|
+
engine, real tool layer, and real task verifiers. The mock profile is
|
|
435
|
+
provider-offline: it makes no provider calls (mock-generated commands are host
|
|
436
|
+
shell commands, so they are not a network guarantee). Suite manifests and mock
|
|
437
|
+
scripts are trusted operator-provided input: their commands execute on the host
|
|
438
|
+
inside a disposable working directory, so review them like test code before
|
|
439
|
+
running.
|
|
437
440
|
|
|
438
441
|
After installing the source checkout with `pip install -e .`:
|
|
439
442
|
|
|
@@ -773,9 +773,11 @@ class OpenRouterAgentCLI:
|
|
|
773
773
|
self._batch_allow.clear()
|
|
774
774
|
self._batch_deny.clear()
|
|
775
775
|
stored_workdir = data.get("workdir", "")
|
|
776
|
+
workdir_mismatch = False
|
|
776
777
|
if stored_workdir:
|
|
777
778
|
current_workdir = str(Path(self.workdir).resolve())
|
|
778
779
|
if stored_workdir != current_workdir:
|
|
780
|
+
workdir_mismatch = True
|
|
779
781
|
print(
|
|
780
782
|
_strip_control_chars(
|
|
781
783
|
f"[session] Working directory changed since last session (was {stored_workdir}, now {current_workdir}). "
|
|
@@ -800,13 +802,21 @@ class OpenRouterAgentCLI:
|
|
|
800
802
|
"last_check": stored_work_order.get("last_check"),
|
|
801
803
|
"updated_at": stored_work_order.get("updated_at"),
|
|
802
804
|
}
|
|
805
|
+
if workdir_mismatch:
|
|
806
|
+
# An acceptance result from another directory does not
|
|
807
|
+
# apply here; never display stale verified evidence.
|
|
808
|
+
self.work_order["status"] = "not_verified"
|
|
809
|
+
self.work_order["last_check"] = None
|
|
803
810
|
stored_cache = data.get("cache_context")
|
|
804
811
|
if self.cache_mode != "off" and isinstance(stored_cache, dict):
|
|
812
|
+
# Restore only cumulative observations. The pairwise
|
|
813
|
+
# stable-prefix fields are transient (there are no stored
|
|
814
|
+
# request hashes to back them), so they start fresh and are
|
|
815
|
+
# recomputed on the next request instead of displaying a
|
|
816
|
+
# stale prefix with no fingerprint.
|
|
805
817
|
for name in (
|
|
806
818
|
"requests",
|
|
807
819
|
"compactions",
|
|
808
|
-
"stable_prefix_tokens",
|
|
809
|
-
"stable_prefix_messages",
|
|
810
820
|
"observed_cached_tokens",
|
|
811
821
|
"provider_cache_observations",
|
|
812
822
|
):
|
|
@@ -896,6 +906,7 @@ class OpenRouterAgentCLI:
|
|
|
896
906
|
"exit_code": result.get("exit_code"),
|
|
897
907
|
"timed_out": bool(result.get("timed_out")),
|
|
898
908
|
"duration_ms": result.get("duration_ms"),
|
|
909
|
+
"checked_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
899
910
|
"stdout": result.get("stdout", ""),
|
|
900
911
|
"stderr": result.get("stderr", ""),
|
|
901
912
|
"error": result.get("error", ""),
|
|
@@ -1563,6 +1574,7 @@ class OpenRouterAgentCLI:
|
|
|
1563
1574
|
"%Y-%m-%dT%H:%M:%SZ", time.gmtime()
|
|
1564
1575
|
)
|
|
1565
1576
|
self._install_completion_policy()
|
|
1577
|
+
self._save_session()
|
|
1566
1578
|
print("Acceptance check re-scoped to the new working directory (status reset to not_verified).")
|
|
1567
1579
|
print(_strip_control_chars(f"Working directory set to: {self.workdir}"))
|
|
1568
1580
|
print("Session-scoped permission grants cleared after cwd change.")
|
|
@@ -2980,6 +2992,34 @@ class OpenRouterAgentCLI:
|
|
|
2980
2992
|
self._display_check_result(result)
|
|
2981
2993
|
self._last_displayed_check = result
|
|
2982
2994
|
|
|
2995
|
+
def _emit_completion_summary(self) -> None:
|
|
2996
|
+
"""Emit a deterministic completion notice when a repair response ended
|
|
2997
|
+
with tools and there is no model final text to show.
|
|
2998
|
+
|
|
2999
|
+
One-shot mode promises the assistant reply on stdout; without this the
|
|
3000
|
+
turn would end silently with an empty reply even after verified work.
|
|
3001
|
+
The notice is CLI-generated and clearly labeled, never fabricated
|
|
3002
|
+
model text, and applies only to the user acceptance policy.
|
|
3003
|
+
"""
|
|
3004
|
+
if not isinstance(self.checkpoint_hook, UserCompletionPolicy):
|
|
3005
|
+
return
|
|
3006
|
+
if self.completion_policy is None:
|
|
3007
|
+
return
|
|
3008
|
+
result = self.completion_policy.last_result
|
|
3009
|
+
if not result:
|
|
3010
|
+
return
|
|
3011
|
+
status = str(result.get("status") or "not_verified").upper()
|
|
3012
|
+
command = str(result.get("command") or "")
|
|
3013
|
+
exit_code = result.get("exit_code")
|
|
3014
|
+
changed = list(result.get("changed_files") or [])
|
|
3015
|
+
code = f" (exit_code={exit_code})" if exit_code is not None else ""
|
|
3016
|
+
changed_text = f" Changed files: {', '.join(changed)}." if changed else ""
|
|
3017
|
+
summary = (
|
|
3018
|
+
f"[cli] Turn ended after the repair check: {status}: "
|
|
3019
|
+
f"{command}{code}{changed_text}"
|
|
3020
|
+
)
|
|
3021
|
+
self._output_response(summary)
|
|
3022
|
+
|
|
2983
3023
|
async def _run_user_turn(self, client: httpx.AsyncClient, user_text: str) -> str:
|
|
2984
3024
|
self._turn_allow.clear()
|
|
2985
3025
|
self._turn_deny.clear()
|
|
@@ -3102,8 +3142,14 @@ class OpenRouterAgentCLI:
|
|
|
3102
3142
|
or ""
|
|
3103
3143
|
)
|
|
3104
3144
|
self.messages.append(forced_message)
|
|
3105
|
-
except Exception:
|
|
3106
|
-
|
|
3145
|
+
except Exception as e:
|
|
3146
|
+
# A loop-breaker failure is still a provider failure. Do not
|
|
3147
|
+
# fabricate a normal answer: record it like the primary
|
|
3148
|
+
# request handlers and terminate without grading the turn.
|
|
3149
|
+
self._log(f"[openrouter] Loop-breaker request failed: {e}")
|
|
3150
|
+
self.terminal_status = "provider_error"
|
|
3151
|
+
self._save_session()
|
|
3152
|
+
return ""
|
|
3107
3153
|
if not text:
|
|
3108
3154
|
text = "I got stuck in a tool loop and could not make progress."
|
|
3109
3155
|
should_continue, result = await self._handle_final_answer(
|
|
@@ -3287,6 +3333,7 @@ class OpenRouterAgentCLI:
|
|
|
3287
3333
|
continue
|
|
3288
3334
|
self._display_policy_check()
|
|
3289
3335
|
if action == "stop" or self._checkpoint_repair_pending:
|
|
3336
|
+
self._emit_completion_summary()
|
|
3290
3337
|
self._save_session()
|
|
3291
3338
|
return ""
|
|
3292
3339
|
|
|
@@ -3300,6 +3347,7 @@ class OpenRouterAgentCLI:
|
|
|
3300
3347
|
)
|
|
3301
3348
|
self._display_policy_check()
|
|
3302
3349
|
self._apply_checkpoint_decision(decision)
|
|
3350
|
+
self._emit_completion_summary()
|
|
3303
3351
|
self._save_session()
|
|
3304
3352
|
return ""
|
|
3305
3353
|
|
|
@@ -3593,6 +3641,17 @@ def main() -> int:
|
|
|
3593
3641
|
except KeyboardInterrupt:
|
|
3594
3642
|
pass
|
|
3595
3643
|
|
|
3644
|
+
# One-shot mode must signal provider failures to callers instead of
|
|
3645
|
+
# exiting 0 after an empty turn.
|
|
3646
|
+
if args.prompt is not None and cli.terminal_status != "ok":
|
|
3647
|
+
print(
|
|
3648
|
+
f"[openrouter] request failed; turn ended with terminal_status="
|
|
3649
|
+
f"{cli.terminal_status}",
|
|
3650
|
+
file=sys.stderr,
|
|
3651
|
+
)
|
|
3652
|
+
return 1
|
|
3653
|
+
return 0
|
|
3654
|
+
|
|
3596
3655
|
|
|
3597
3656
|
if __name__ == "__main__":
|
|
3598
3657
|
raise SystemExit(main())
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/runner.py
RENAMED
|
@@ -197,11 +197,18 @@ class SuiteRunner:
|
|
|
197
197
|
engine.policy = ToolPermissionPolicy(allow={"*"})
|
|
198
198
|
engine.non_interactive_mode = True
|
|
199
199
|
engine.one_shot_prompt = task.prompt
|
|
200
|
-
if
|
|
200
|
+
if profile.uses_mock:
|
|
201
|
+
record["engine"]["execution_mode"] = "mock"
|
|
202
|
+
elif getattr(self, "_sandboxed", False):
|
|
201
203
|
from .sandbox import BubblewrapBashRunner
|
|
202
204
|
|
|
203
205
|
engine.bash_runner = BubblewrapBashRunner(str(workspace))
|
|
204
206
|
record["engine"]["sandbox"] = "bubblewrap"
|
|
207
|
+
record["engine"]["execution_mode"] = "bubblewrap"
|
|
208
|
+
else:
|
|
209
|
+
# Explicitly acknowledged host-mode execution (operator dev
|
|
210
|
+
# checks only; never valid experiment evidence).
|
|
211
|
+
record["engine"]["execution_mode"] = "host_acknowledged"
|
|
205
212
|
if policy is not None:
|
|
206
213
|
engine.checkpoint_hook = policy
|
|
207
214
|
if profile.uses_mock:
|
|
@@ -324,7 +331,10 @@ class SuiteRunner:
|
|
|
324
331
|
expected_task_ids=[task.id for task in self.suite.tasks],
|
|
325
332
|
expected_profile_names=[profile.name for profile in self.profiles],
|
|
326
333
|
expected_repeats=self.repeats,
|
|
327
|
-
|
|
334
|
+
# Containment is required only for the mode that was actually
|
|
335
|
+
# executed: an explicitly acknowledged host-mode run is
|
|
336
|
+
# auditable as host execution, not silently rejected.
|
|
337
|
+
require_containment=self._sandboxed,
|
|
328
338
|
)
|
|
329
339
|
finally:
|
|
330
340
|
self.cleanup_workspaces(records)
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: openrouter-agent-cli
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Standalone terminal agent for OpenRouter with tool actions and context management.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/protostatis/openrouter-agent-cli
|
|
@@ -94,11 +94,12 @@ openrouter-agent --workdir ./my-repo \
|
|
|
94
94
|
--verify-command "pytest tests/test_auth.py"
|
|
95
95
|
```
|
|
96
96
|
|
|
97
|
-
The command runs
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
97
|
+
The command runs at the completion boundary (once initially, and once more
|
|
98
|
+
after the single permitted repair cycle when the first check fails). A failing
|
|
99
|
+
command receives at most one generic repair cycle; a timeout or execution error
|
|
100
|
+
is reported as `not_verified` rather than treated as success. The workflow can
|
|
101
|
+
be exercised without credentials or network access with
|
|
102
|
+
`openrouter-agent-self-test` (or `openrouter-agent --self-test`).
|
|
102
103
|
|
|
103
104
|
## Useful flags
|
|
104
105
|
|
|
@@ -457,10 +458,12 @@ Evaluator artifacts:
|
|
|
457
458
|
## Offline evaluation workflow
|
|
458
459
|
|
|
459
460
|
The repository includes a small evaluation workflow that runs the real CLI
|
|
460
|
-
engine, real tool layer, and real task verifiers. The mock profile is
|
|
461
|
-
offline: it makes no provider calls
|
|
462
|
-
|
|
463
|
-
|
|
461
|
+
engine, real tool layer, and real task verifiers. The mock profile is
|
|
462
|
+
provider-offline: it makes no provider calls (mock-generated commands are host
|
|
463
|
+
shell commands, so they are not a network guarantee). Suite manifests and mock
|
|
464
|
+
scripts are trusted operator-provided input: their commands execute on the host
|
|
465
|
+
inside a disposable working directory, so review them like test code before
|
|
466
|
+
running.
|
|
464
467
|
|
|
465
468
|
After installing the source checkout with `pip install -e .`:
|
|
466
469
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "openrouter-agent-cli"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "Standalone terminal agent for OpenRouter with tool actions and context management."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -298,3 +298,43 @@ def test_unknown_assisted_profile_is_rejected(tmp_path: Path, suite: Path) -> No
|
|
|
298
298
|
None,
|
|
299
299
|
assisted_profiles={"worker", "ghost"},
|
|
300
300
|
)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def test_host_mode_audit_uses_actual_containment(
|
|
304
|
+
tmp_path: Path, suite: Path, monkeypatch: pytest.MonkeyPatch
|
|
305
|
+
) -> None:
|
|
306
|
+
"""An explicitly acknowledged host-mode real run must be audited as host
|
|
307
|
+
execution (require_containment follows _sandboxed), not rejected."""
|
|
308
|
+
from openrouter_agent_cli.eval.records import make_record, new_run_id
|
|
309
|
+
|
|
310
|
+
monkeypatch.setenv("OPENROUTER_API_KEY", "test-key")
|
|
311
|
+
monkeypatch.setenv("AGENT_EVAL_ALLOW_HOST_EXECUTION", "1")
|
|
312
|
+
monkeypatch.setenv("AGENT_EVAL_SANDBOX", "0")
|
|
313
|
+
loaded = load_suite(suite)
|
|
314
|
+
runner = SuiteRunner(
|
|
315
|
+
loaded,
|
|
316
|
+
[Profile(name="real", prompt="P", model="m")],
|
|
317
|
+
eval_dir=tmp_path / "eval",
|
|
318
|
+
)
|
|
319
|
+
assert runner._sandboxed is False
|
|
320
|
+
|
|
321
|
+
captured: dict = {}
|
|
322
|
+
monkeypatch.setattr(
|
|
323
|
+
"openrouter_agent_cli.eval.runner.assert_audited",
|
|
324
|
+
lambda records, **kwargs: captured.update(kwargs),
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
async def fake_run():
|
|
328
|
+
r = make_record(
|
|
329
|
+
run_id=new_run_id(), suite_id=loaded.suite_id,
|
|
330
|
+
task_id=loaded.tasks[0].id, cluster_id=loaded.tasks[0].cluster_id,
|
|
331
|
+
profile_name="real", profile_prompt="P", model="m",
|
|
332
|
+
transport="openrouter", workdir=str(tmp_path), scheduled_index=0,
|
|
333
|
+
)
|
|
334
|
+
r["engine"] = {"execution_mode": "host_acknowledged", "error": None}
|
|
335
|
+
r["verdict"] = "infrastructure_error"
|
|
336
|
+
return [r]
|
|
337
|
+
|
|
338
|
+
monkeypatch.setattr(runner, "run", fake_run)
|
|
339
|
+
asyncio.run(runner.run_and_verify())
|
|
340
|
+
assert captured.get("require_containment") is False
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import json
|
|
5
6
|
import os
|
|
6
7
|
import shlex
|
|
7
8
|
import sys
|
|
@@ -303,3 +304,193 @@ async def test_provider_failure_sets_terminal_status(tmp_path, monkeypatch):
|
|
|
303
304
|
async with httpx.AsyncClient() as client:
|
|
304
305
|
await cli._run_user_turn(client, "hi")
|
|
305
306
|
assert cli.terminal_status == "provider_error"
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
@pytest.mark.asyncio
|
|
310
|
+
async def test_loop_breaker_failure_is_provider_error(tmp_path, monkeypatch):
|
|
311
|
+
"""A failed forced loop-breaker call must be recorded as a provider error
|
|
312
|
+
and terminate, not fabricate a normal answer."""
|
|
313
|
+
session_dir = tmp_path / "sessions"
|
|
314
|
+
monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
|
|
315
|
+
cli = OpenRouterAgentCLI(
|
|
316
|
+
api_key="test-key",
|
|
317
|
+
model="test-model",
|
|
318
|
+
session_id="loop-break-test",
|
|
319
|
+
workdir=str(tmp_path),
|
|
320
|
+
max_turns=10,
|
|
321
|
+
max_history_messages=60,
|
|
322
|
+
command_timeout=5,
|
|
323
|
+
tools_enabled=True,
|
|
324
|
+
system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
325
|
+
discovery_mode="off",
|
|
326
|
+
)
|
|
327
|
+
cli.non_interactive_mode = True
|
|
328
|
+
cli.policy = ToolPermissionPolicy(allow={"*"})
|
|
329
|
+
repeated = {"name": "run_bash", "arguments": {"command": "true"}}
|
|
330
|
+
|
|
331
|
+
class FailingLoopBreaker:
|
|
332
|
+
def __init__(self):
|
|
333
|
+
self.requests = 0
|
|
334
|
+
|
|
335
|
+
async def __call__(self, client, **kwargs):
|
|
336
|
+
self.requests += 1
|
|
337
|
+
if kwargs.get("tool_choice") == "none":
|
|
338
|
+
raise RuntimeError("loop-breaker provider failure")
|
|
339
|
+
return {
|
|
340
|
+
"choices": [
|
|
341
|
+
{
|
|
342
|
+
"message": {
|
|
343
|
+
"role": "assistant",
|
|
344
|
+
"content": None,
|
|
345
|
+
"tool_calls": [
|
|
346
|
+
{
|
|
347
|
+
"id": f"tc-{self.requests}",
|
|
348
|
+
"type": "function",
|
|
349
|
+
"function": {
|
|
350
|
+
"name": "run_bash",
|
|
351
|
+
"arguments": '{"command": "true"}',
|
|
352
|
+
},
|
|
353
|
+
}
|
|
354
|
+
],
|
|
355
|
+
},
|
|
356
|
+
"finish_reason": "tool_calls",
|
|
357
|
+
}
|
|
358
|
+
],
|
|
359
|
+
"usage": {
|
|
360
|
+
"prompt_tokens": 1,
|
|
361
|
+
"completion_tokens": 1,
|
|
362
|
+
"total_tokens": 2,
|
|
363
|
+
},
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
transport = FailingLoopBreaker()
|
|
367
|
+
cli.model_transport = transport
|
|
368
|
+
async with httpx.AsyncClient() as client:
|
|
369
|
+
result = await cli._run_user_turn(client, "do it")
|
|
370
|
+
assert result == ""
|
|
371
|
+
assert cli.terminal_status == "provider_error"
|
|
372
|
+
# The repeated tool call was nudged; the forced request failed once.
|
|
373
|
+
assert transport.requests == 3
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
@pytest.mark.asyncio
|
|
377
|
+
async def test_tool_repair_emits_completion_notice(
|
|
378
|
+
tmp_path, monkeypatch, capsys
|
|
379
|
+
):
|
|
380
|
+
"""The tool-using repair path must print a deterministic completion notice
|
|
381
|
+
so one-shot mode is not left with an empty stdout."""
|
|
382
|
+
engine = _engine_with_task(
|
|
383
|
+
tmp_path,
|
|
384
|
+
monkeypatch,
|
|
385
|
+
task="Create marker.txt",
|
|
386
|
+
verify_command="test -f marker.txt",
|
|
387
|
+
responses=[
|
|
388
|
+
{"tool_calls": [
|
|
389
|
+
{"name": "write_file",
|
|
390
|
+
"arguments": {"path": "tmp.txt", "content": "x\n"}}
|
|
391
|
+
]},
|
|
392
|
+
{"text": "I wrote tmp.txt, marker is next"},
|
|
393
|
+
{"tool_calls": [
|
|
394
|
+
{"name": "write_file",
|
|
395
|
+
"arguments": {"path": "marker.txt", "content": "ok\n"}}
|
|
396
|
+
]},
|
|
397
|
+
{"text": "must not be requested"},
|
|
398
|
+
],
|
|
399
|
+
)
|
|
400
|
+
async with httpx.AsyncClient() as client:
|
|
401
|
+
await engine._run_user_turn(client, "Do the work.")
|
|
402
|
+
out = capsys.readouterr().out
|
|
403
|
+
assert "Turn ended after the repair check: VERIFIED" in out
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def test_cached_prefix_is_not_restored_on_resume(tmp_path, monkeypatch):
|
|
407
|
+
"""Session loading restores cumulative counters but never transient
|
|
408
|
+
stable-prefix state (there are no stored request hashes to back it)."""
|
|
409
|
+
session_dir = tmp_path / "sessions"
|
|
410
|
+
monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
|
|
411
|
+
kwargs = {
|
|
412
|
+
"api_key": "test-key",
|
|
413
|
+
"model": "test-model",
|
|
414
|
+
"workdir": str(tmp_path),
|
|
415
|
+
"max_turns": 2,
|
|
416
|
+
"max_history_messages": 60,
|
|
417
|
+
"command_timeout": 5,
|
|
418
|
+
"tools_enabled": True,
|
|
419
|
+
"system_prompt": DEFAULT_SYSTEM_PROMPT,
|
|
420
|
+
"discovery_mode": "off",
|
|
421
|
+
}
|
|
422
|
+
first = OpenRouterAgentCLI(**kwargs, session_id="cache-resume")
|
|
423
|
+
prefix = [
|
|
424
|
+
{"role": "system", "content": "stable-system"},
|
|
425
|
+
{"role": "user", "content": "hello"},
|
|
426
|
+
]
|
|
427
|
+
first.cache_context.observe_request(prefix)
|
|
428
|
+
first.cache_context.observe_request(prefix + [{"role": "assistant", "content": "hi"}])
|
|
429
|
+
assert first.cache_context.stable_prefix_tokens > 0
|
|
430
|
+
first._save_session()
|
|
431
|
+
|
|
432
|
+
resumed = OpenRouterAgentCLI(**kwargs, session_id="cache-resume")
|
|
433
|
+
assert resumed.cache_context.stable_prefix_tokens == 0
|
|
434
|
+
assert resumed.cache_context.stable_prefix_messages == 0
|
|
435
|
+
# Cumulative observations survive.
|
|
436
|
+
assert resumed.cache_context.requests == 2
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def test_workdir_mismatch_invalidates_acceptance(tmp_path, monkeypatch):
|
|
440
|
+
"""An acceptance result from another directory must never be shown as
|
|
441
|
+
verified after resuming in a different workdir."""
|
|
442
|
+
session_dir = tmp_path / "sessions"
|
|
443
|
+
other = tmp_path / "other"
|
|
444
|
+
other.mkdir()
|
|
445
|
+
monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
|
|
446
|
+
kwargs = {
|
|
447
|
+
"api_key": "test-key",
|
|
448
|
+
"model": "test-model",
|
|
449
|
+
"max_turns": 2,
|
|
450
|
+
"max_history_messages": 60,
|
|
451
|
+
"command_timeout": 5,
|
|
452
|
+
"tools_enabled": True,
|
|
453
|
+
"system_prompt": DEFAULT_SYSTEM_PROMPT,
|
|
454
|
+
"discovery_mode": "off",
|
|
455
|
+
}
|
|
456
|
+
first = OpenRouterAgentCLI(
|
|
457
|
+
**kwargs, session_id="workdir-switch", workdir=str(tmp_path),
|
|
458
|
+
task="Task X", verify_command="pytest -q",
|
|
459
|
+
)
|
|
460
|
+
first.work_order["status"] = "verified"
|
|
461
|
+
first._save_session()
|
|
462
|
+
|
|
463
|
+
resumed = OpenRouterAgentCLI(
|
|
464
|
+
**kwargs, session_id="workdir-switch", workdir=str(other)
|
|
465
|
+
)
|
|
466
|
+
assert resumed.work_order["objective"] == "Task X"
|
|
467
|
+
assert resumed.work_order["status"] == "not_verified"
|
|
468
|
+
assert resumed.work_order["last_check"] is None
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
@pytest.mark.asyncio
|
|
472
|
+
async def test_cwd_change_persists_contract_reset(tmp_path, monkeypatch):
|
|
473
|
+
session_dir = tmp_path / "sessions"
|
|
474
|
+
other = tmp_path / "other-project"
|
|
475
|
+
other.mkdir()
|
|
476
|
+
monkeypatch.setenv("OPENROUTER_AGENT_SESSION_DIR", str(session_dir))
|
|
477
|
+
cli = OpenRouterAgentCLI(
|
|
478
|
+
api_key="test-key",
|
|
479
|
+
model="test-model",
|
|
480
|
+
session_id="cwd-persist",
|
|
481
|
+
workdir=str(tmp_path),
|
|
482
|
+
max_turns=2,
|
|
483
|
+
max_history_messages=60,
|
|
484
|
+
command_timeout=5,
|
|
485
|
+
tools_enabled=True,
|
|
486
|
+
system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
487
|
+
discovery_mode="off",
|
|
488
|
+
task="Task X",
|
|
489
|
+
verify_command="pytest -q",
|
|
490
|
+
)
|
|
491
|
+
cli.work_order["status"] = "verified"
|
|
492
|
+
await cli._handle_command(None, f"/cwd {other}")
|
|
493
|
+
|
|
494
|
+
persisted = json.loads(cli._session_path.read_text())
|
|
495
|
+
assert persisted["work_order"]["status"] == "not_verified"
|
|
496
|
+
assert persisted["work_order"]["verify_command"] == "pytest -q"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/completion.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/concurrent.py
RENAMED
|
File without changes
|
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/__init__.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/audit.py
RENAMED
|
File without changes
|
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/compare.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/policy.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/records.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/report.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/sandbox.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/suite.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/transport.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/uncertainty.py
RENAMED
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli/eval/verify.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{openrouter_agent_cli-0.2.0 → openrouter_agent_cli-0.2.1}/openrouter_agent_cli.egg-info/requires.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|