agentprobe-testing 0.5.0__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/PKG-INFO +1 -1
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/README.md +37 -2
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/__init__.py +3 -2
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/quickstart.py +81 -1
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/PKG-INFO +1 -1
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/pyproject.toml +1 -1
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_quickstart.py +99 -1
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/LICENSE +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/__init__.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/base.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/rule_based.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/scripted.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/target_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agreement.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/classifier.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/cli.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/diff.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domain.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/__init__.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/__init__.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/clean.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/complex_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/decoy.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/domain.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/entities.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/injector_prompt.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/rule_based_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/scenarios.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/split.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/tools.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/trap.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/feedback.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/generic_world.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/injection.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/injector.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/llm.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/playbook.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/reachability.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/registry.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/report.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/runner.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenario.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/__init__.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/clean.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/decoy.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/registry.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/split.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/trap.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/termui.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/tools.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/trajectory.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/triage.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/validate_scenarios.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/world.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/SOURCES.txt +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/dependency_links.txt +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/entry_points.txt +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/requires.txt +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/top_level.txt +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/setup.cfg +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_domain.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_rule_based_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_scenarios.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_tools.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_agreement.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_classifier.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_cli.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_complex_access_control_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_diff.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_domain.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_feedback.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_generic_world.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_injection.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_injector.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_llm.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_package_api.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_playbook.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_reachability.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_registry.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_report.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_rule_based_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_runner.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_scenarios.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_target_agent.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_termui.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_tools.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_trajectory.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_triage.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_validate_scenarios.py +0 -0
- {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_world.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agentprobe-testing
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.2
|
|
4
4
|
Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
|
|
5
5
|
License: Business Source License 1.1
|
|
6
6
|
|
|
@@ -18,6 +18,19 @@ adversarial content disguised as ordinary data (`PROMPT_INJECTION`).
|
|
|
18
18
|
|
|
19
19
|
## Quickstart
|
|
20
20
|
|
|
21
|
+
Using this in your own project (no need to clone this repo):
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install agentprobe-testing
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
`import agentprobe as ap` afterward, same as everywhere else in this
|
|
28
|
+
README — the PyPI *distribution* name is `agentprobe-testing`, but the
|
|
29
|
+
Python package you import is still plain `agentprobe`.
|
|
30
|
+
|
|
31
|
+
To run the examples in this repo (or work on the library itself), clone
|
|
32
|
+
it and install from source instead:
|
|
33
|
+
|
|
21
34
|
```bash
|
|
22
35
|
pip install -e .
|
|
23
36
|
python examples/free_zero_cost_demo.py # no API key needed, no network calls at all
|
|
@@ -105,6 +118,12 @@ surface. The `access_control` domain lives at
|
|
|
105
118
|
top level, to keep this surface to the one domain most people reach for
|
|
106
119
|
first.
|
|
107
120
|
|
|
121
|
+
Testing your own agent instead of the built-in `TargetAgent`? Pass any
|
|
122
|
+
plain `(task, history) -> AgentAction` function as `agent` — no subclass
|
|
123
|
+
needed, `quick_test`/`wrap_agent` handle the wiring, and if your agent
|
|
124
|
+
already runs as a service you can POST to, `ap.http_agent(url)` builds
|
|
125
|
+
that function for you (see [USAGE.md](USAGE.md)).
|
|
126
|
+
|
|
108
127
|
Beyond one-scenario `quick_test()`, see [USAGE.md](USAGE.md) for: running
|
|
109
128
|
a whole domain at once (`quick_test_all`), a one-line pytest integration
|
|
110
129
|
(`assert_passes`), catching regressions between uploaded runs
|
|
@@ -166,7 +185,10 @@ Writing `agentprobe/domain.py`'s `Domain` bundle by hand (entities, tools,
|
|
|
166
185
|
business rules) is real, unavoidable work — see "Domains" above. If you'd
|
|
167
186
|
rather not write it yourself, [agentprobe-api.agentprobe.workers.dev](https://agentprobe-api.agentprobe.workers.dev)
|
|
168
187
|
can generate one from your existing tool schemas plus a plain-English
|
|
169
|
-
description of your business rules, and hand you back a key
|
|
188
|
+
description of your business rules, and hand you back a key — either via
|
|
189
|
+
[/generate](https://agentprobe-api.agentprobe.workers.dev/generate) on the
|
|
190
|
+
site (sign in, fill in a form, get the key and the generated source back
|
|
191
|
+
right on the page — no `curl` needed) or the raw API:
|
|
170
192
|
|
|
171
193
|
```python
|
|
172
194
|
import agentprobe as ap
|
|
@@ -204,6 +226,19 @@ is the real site, not just the API — sign in with GitHub and you get:
|
|
|
204
226
|
- **API access** (on the Dashboard) — create a personal key
|
|
205
227
|
(`ap_live_...`, shown once at creation) to authenticate the two things
|
|
206
228
|
above from your own scripts.
|
|
229
|
+
- **`/generate`** — the same domain generation as `POST /v1/generate`,
|
|
230
|
+
as a web form instead of `curl` — fill it in, get back the key and the
|
|
231
|
+
generated source to review right on the page.
|
|
232
|
+
- **`/run`** — execute a real chaos test from the website itself, no
|
|
233
|
+
terminal needed: paste a domain key, pick a scenario from a real
|
|
234
|
+
dropdown (or "All scenarios" to run every one, combined into a single
|
|
235
|
+
report), point at your agent's HTTP endpoint and your own Anthropic API
|
|
236
|
+
key, and it runs in the background — the page comes back immediately so
|
|
237
|
+
you can go do something else on the site while it finishes, and it
|
|
238
|
+
shows up under Your Runs once it does. Free to use — you're billed by
|
|
239
|
+
Anthropic for your own API usage, the same as running it yourself. See
|
|
240
|
+
[`runner/README.md`](runner/README.md) for how this actually executes server-side (the real `agentprobe-testing`
|
|
241
|
+
package, not a reimplementation).
|
|
207
242
|
|
|
208
243
|
Attribute a generated domain to your account by passing the key as a
|
|
209
244
|
bearer token:
|
|
@@ -269,7 +304,7 @@ The website/API server (`server/`) has its own separate suite, same rule:
|
|
|
269
304
|
cd server && node --test test/
|
|
270
305
|
```
|
|
271
306
|
|
|
272
|
-
|
|
307
|
+
160+ tests against fake D1/KV/fetch — no real Cloudflare account, network
|
|
273
308
|
access, or secrets needed for either suite. Both run in CI on every push.
|
|
274
309
|
|
|
275
310
|
See [CHANGELOG.md](CHANGELOG.md) for what's new in each version.
|
|
@@ -44,7 +44,7 @@ from agentprobe.injector import (
|
|
|
44
44
|
RecordingInjector,
|
|
45
45
|
ReplayInjector,
|
|
46
46
|
)
|
|
47
|
-
from agentprobe.quickstart import assert_passes, quick_test, quick_test_all, wrap_agent
|
|
47
|
+
from agentprobe.quickstart import assert_passes, http_agent, quick_test, quick_test_all, wrap_agent
|
|
48
48
|
from agentprobe.registry import (
|
|
49
49
|
FetchedDomain,
|
|
50
50
|
RegressionResult,
|
|
@@ -60,7 +60,7 @@ from agentprobe.scenario import CommitPattern, FactPattern, GoalSpec, Scenario
|
|
|
60
60
|
from agentprobe.scenarios.registry import ALL_SCENARIOS as TICKET_SCENARIOS
|
|
61
61
|
from agentprobe.scenarios.registry import BY_ID as TICKET_SCENARIOS_BY_ID
|
|
62
62
|
|
|
63
|
-
__version__ = "0.5.
|
|
63
|
+
__version__ = "0.5.2"
|
|
64
64
|
|
|
65
65
|
__all__ = [
|
|
66
66
|
"Agent",
|
|
@@ -93,6 +93,7 @@ __all__ = [
|
|
|
93
93
|
"check_regression",
|
|
94
94
|
"fetch_domain",
|
|
95
95
|
"get_latest_run",
|
|
96
|
+
"http_agent",
|
|
96
97
|
"injection_was_triggered",
|
|
97
98
|
"quick_test",
|
|
98
99
|
"quick_test_all",
|
|
@@ -9,6 +9,9 @@ what to do next, test it against one scenario."
|
|
|
9
9
|
|
|
10
10
|
from __future__ import annotations
|
|
11
11
|
|
|
12
|
+
import json
|
|
13
|
+
import time
|
|
14
|
+
import urllib.request
|
|
12
15
|
from typing import Any, Callable, Optional
|
|
13
16
|
|
|
14
17
|
from agentprobe.agents.base import Agent, AgentAction
|
|
@@ -57,6 +60,64 @@ def wrap_agent(decide_fn: DecideFn) -> Callable[[], Agent]:
|
|
|
57
60
|
return lambda: _FunctionAgent(decide_fn)
|
|
58
61
|
|
|
59
62
|
|
|
63
|
+
HttpRequester = Callable[[str, dict, dict], dict] # (url, body, headers) -> response dict
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _default_http_requester(url: str, body: dict, headers: dict) -> dict:
|
|
67
|
+
data = json.dumps(body).encode("utf-8")
|
|
68
|
+
request = urllib.request.Request(
|
|
69
|
+
url,
|
|
70
|
+
data=data,
|
|
71
|
+
method="POST",
|
|
72
|
+
headers={"Content-Type": "application/json", "User-Agent": "agentprobe-client/0.5.0", **headers},
|
|
73
|
+
)
|
|
74
|
+
with urllib.request.urlopen(request, timeout=30) as resp: # noqa: S310 -- caller-supplied url, same trust model as any target agent's own API
|
|
75
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def http_agent(
|
|
79
|
+
url: str,
|
|
80
|
+
headers: Optional[dict] = None,
|
|
81
|
+
requester: Optional[HttpRequester] = None,
|
|
82
|
+
) -> Callable[[], Agent]:
|
|
83
|
+
"""Adapt a hosted HTTP agent into a target_factory for quick_test()/
|
|
84
|
+
run_robustness_pair() -- for when your real agent already runs as a
|
|
85
|
+
service you can POST to, instead of a Python function/object living in
|
|
86
|
+
this same process. No custom decide_fn to write: this builds one for
|
|
87
|
+
you and wraps it exactly like wrap_agent() does.
|
|
88
|
+
|
|
89
|
+
On every step, POSTs `{"task": <str>, "history": [...]}` as JSON to
|
|
90
|
+
`url` -- `history` is the same shape wrap_agent's decide_fn receives, a
|
|
91
|
+
list of `{"tool_name", "tool_args", "result", "ok"}` dicts covering
|
|
92
|
+
everything observed so far this run. Expects back JSON shaped like
|
|
93
|
+
either:
|
|
94
|
+
|
|
95
|
+
{"tool_name": "...", "tool_args": {...}} -- call this tool next
|
|
96
|
+
{"final_answer": "..."} -- done, this is the reply
|
|
97
|
+
|
|
98
|
+
Raises ValueError if a response matches neither shape, rather than
|
|
99
|
+
guessing what was meant. Pass `headers` (e.g.
|
|
100
|
+
`{"Authorization": "Bearer ..."}`) for an endpoint that needs auth, and
|
|
101
|
+
`requester` in tests instead of hitting the real network -- see
|
|
102
|
+
tests/test_quickstart.py.
|
|
103
|
+
"""
|
|
104
|
+
requester = requester if requester is not None else _default_http_requester
|
|
105
|
+
resolved_headers = headers or {}
|
|
106
|
+
|
|
107
|
+
def decide_fn(task: str, history: list[dict[str, Any]]) -> AgentAction:
|
|
108
|
+
response = requester(url, {"task": task, "history": history}, resolved_headers)
|
|
109
|
+
if "final_answer" in response:
|
|
110
|
+
return AgentAction(kind="final_answer", text=response["final_answer"])
|
|
111
|
+
if "tool_name" in response and "tool_args" in response:
|
|
112
|
+
return AgentAction(kind="tool_call", tool_name=response["tool_name"], tool_args=response["tool_args"])
|
|
113
|
+
raise ValueError(
|
|
114
|
+
f"http_agent: response from {url} must include either 'final_answer' or "
|
|
115
|
+
f"both 'tool_name' and 'tool_args', got keys {sorted(response.keys())}"
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
return wrap_agent(decide_fn)
|
|
119
|
+
|
|
120
|
+
|
|
60
121
|
def quick_test(
|
|
61
122
|
scenario: Scenario,
|
|
62
123
|
agent: Callable[[], Agent] | DecideFn,
|
|
@@ -149,6 +210,7 @@ def quick_test_all(
|
|
|
149
210
|
upload_api_key: Optional[str] = None,
|
|
150
211
|
domain_key: Optional[str] = None,
|
|
151
212
|
record_injections: bool = False,
|
|
213
|
+
max_duration_seconds: Optional[float] = None,
|
|
152
214
|
) -> Report:
|
|
153
215
|
"""Run every scenario in `scenarios` (a dict[str, Scenario] -- e.g.
|
|
154
216
|
TICKET_SCENARIOS or a FetchedDomain's .scenarios -- or a plain list of
|
|
@@ -168,6 +230,16 @@ def quick_test_all(
|
|
|
168
230
|
run (so Your Runs shows them individually, the same granularity as
|
|
169
231
|
calling quick_test() per scenario and uploading each) -- the Report
|
|
170
232
|
this function returns is still the combined summary across all of them.
|
|
233
|
+
|
|
234
|
+
`max_duration_seconds`, if given, stops starting new scenarios once
|
|
235
|
+
that much wall-clock time has passed since the call began -- for a
|
|
236
|
+
caller with its own hard execution ceiling (e.g. a serverless platform
|
|
237
|
+
that kills a job outright past some limit, which can't be caught or
|
|
238
|
+
cleaned up after), a report covering fewer scenarios than asked for is
|
|
239
|
+
far better than getting killed mid-scenario with nothing to show for
|
|
240
|
+
it. The returned Report's `_scenarios_run`/`_scenarios_total`
|
|
241
|
+
attributes say whether this happened, so a caller can tell the
|
|
242
|
+
difference between "ran everything" and "stopped early."
|
|
171
243
|
"""
|
|
172
244
|
items = list(scenarios.items()) if isinstance(scenarios, dict) else [(s.id, s) for s in scenarios]
|
|
173
245
|
if not items:
|
|
@@ -186,8 +258,13 @@ def quick_test_all(
|
|
|
186
258
|
all_records: list[InjectionRecord] = []
|
|
187
259
|
total_injector_cost = 0.0
|
|
188
260
|
combined_injector_model: Optional[str] = None
|
|
261
|
+
start = time.monotonic()
|
|
262
|
+
scenarios_run = 0
|
|
189
263
|
|
|
190
264
|
for _scenario_id, scenario in items:
|
|
265
|
+
if max_duration_seconds is not None and scenarios_run > 0 and (time.monotonic() - start) >= max_duration_seconds:
|
|
266
|
+
break # already have at least one scenario's worth of real results -- stop here, not mid-scenario
|
|
267
|
+
scenarios_run += 1
|
|
191
268
|
injector = make_injector()
|
|
192
269
|
resolved_injector_model = injector_model if injector_model is not None else type(injector).__name__
|
|
193
270
|
if combined_injector_model is None:
|
|
@@ -229,7 +306,7 @@ def quick_test_all(
|
|
|
229
306
|
recorded = [serialize_armed_injection(a) for a in recording.recorded] if recording is not None else None
|
|
230
307
|
upload_run(scenario_report, api_key=upload_api_key, domain_key=domain_key, recorded_injections=recorded)
|
|
231
308
|
|
|
232
|
-
|
|
309
|
+
report = Report(
|
|
233
310
|
mode=mode,
|
|
234
311
|
injector_model=combined_injector_model or "none",
|
|
235
312
|
target_model=target_model,
|
|
@@ -238,6 +315,9 @@ def quick_test_all(
|
|
|
238
315
|
injection_records=all_records,
|
|
239
316
|
injector_cost_usd=total_injector_cost,
|
|
240
317
|
)
|
|
318
|
+
report._scenarios_run = len(all_chaos)
|
|
319
|
+
report._scenarios_total = len(items)
|
|
320
|
+
return report
|
|
241
321
|
|
|
242
322
|
|
|
243
323
|
def assert_passes(report: Report, save_html_on_failure: Optional[str] = None) -> None:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agentprobe-testing
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.2
|
|
4
4
|
Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
|
|
5
5
|
License: Business Source License 1.1
|
|
6
6
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "agentprobe-testing"
|
|
3
|
-
version = "0.5.
|
|
3
|
+
version = "0.5.2"
|
|
4
4
|
description = "Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance."
|
|
5
5
|
requires-python = ">=3.11"
|
|
6
6
|
dependencies = ["anthropic>=1.0.0", "python-dotenv>=1.0.0"]
|
|
@@ -3,7 +3,7 @@ import pytest
|
|
|
3
3
|
from agentprobe.agents.base import AgentAction
|
|
4
4
|
from agentprobe.agents.rule_based import RuleBasedAgent
|
|
5
5
|
from agentprobe.injector import HardcodedToolErrorInjector, Injector, NullInjector
|
|
6
|
-
from agentprobe.quickstart import assert_passes, quick_test, quick_test_all, wrap_agent
|
|
6
|
+
from agentprobe.quickstart import assert_passes, http_agent, quick_test, quick_test_all, wrap_agent
|
|
7
7
|
from agentprobe.report import Report
|
|
8
8
|
from agentprobe.scenarios.registry import BY_ID
|
|
9
9
|
|
|
@@ -52,6 +52,81 @@ def test_wrap_agent_gives_each_call_its_own_history_not_shared_across_instances(
|
|
|
52
52
|
assert action.tool_name == "get_ticket" # fresh history, not polluted by `first`
|
|
53
53
|
|
|
54
54
|
|
|
55
|
+
def make_requester(responses):
|
|
56
|
+
"""Returns each queued response in order, on every call -- the same
|
|
57
|
+
scripted-sequence pattern ScriptedAgent uses, since http_agent's whole
|
|
58
|
+
point is standing in for a real remote agent step by step."""
|
|
59
|
+
calls = []
|
|
60
|
+
responses = list(responses)
|
|
61
|
+
|
|
62
|
+
def requester(url, body, headers):
|
|
63
|
+
calls.append({"url": url, "body": body, "headers": headers})
|
|
64
|
+
return responses.pop(0)
|
|
65
|
+
|
|
66
|
+
requester.calls = calls
|
|
67
|
+
return requester
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def test_http_agent_drives_a_full_run_from_scripted_json_responses():
|
|
71
|
+
# mode="recovery" -- a robustness-mode run calls the target_factory
|
|
72
|
+
# twice (clean then chaos), which would drain this same scripted
|
|
73
|
+
# response queue twice over; recovery mode's single trajectory keeps
|
|
74
|
+
# this test's response list simple and matched 1:1 to real steps.
|
|
75
|
+
scenario = BY_ID["clean-1"]
|
|
76
|
+
requester = make_requester(
|
|
77
|
+
[
|
|
78
|
+
{"tool_name": "get_ticket", "tool_args": {"id": "T-1001"}},
|
|
79
|
+
{"tool_name": "issue_refund", "tool_args": {"ticket_id": "T-1001", "amount": 49.99}},
|
|
80
|
+
{"tool_name": "close_ticket", "tool_args": {"ticket_id": "T-1001", "resolution": "refunded"}},
|
|
81
|
+
{"final_answer": "done"},
|
|
82
|
+
]
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
report = quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
|
|
86
|
+
|
|
87
|
+
assert report._compute()["chaos_passed"] == 1
|
|
88
|
+
assert requester.calls[0]["url"] == "https://example.com/agent"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def test_http_agent_sends_task_and_history_as_the_request_body():
|
|
92
|
+
scenario = BY_ID["clean-1"]
|
|
93
|
+
requester = make_requester(
|
|
94
|
+
[
|
|
95
|
+
{"tool_name": "get_ticket", "tool_args": {"id": "T-1001"}},
|
|
96
|
+
{"final_answer": "done"},
|
|
97
|
+
]
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
|
|
101
|
+
|
|
102
|
+
first_call, second_call = requester.calls
|
|
103
|
+
assert first_call["body"]["task"] == "Handle ticket T-1001."
|
|
104
|
+
assert first_call["body"]["history"] == []
|
|
105
|
+
assert second_call["body"]["history"][0]["tool_name"] == "get_ticket"
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_http_agent_passes_through_custom_headers():
|
|
109
|
+
scenario = BY_ID["clean-1"]
|
|
110
|
+
requester = make_requester([{"final_answer": "done"}])
|
|
111
|
+
|
|
112
|
+
quick_test(
|
|
113
|
+
scenario,
|
|
114
|
+
http_agent("https://example.com/agent", headers={"Authorization": "Bearer secret"}, requester=requester),
|
|
115
|
+
injector=NullInjector(),
|
|
116
|
+
mode="recovery",
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
assert requester.calls[0]["headers"] == {"Authorization": "Bearer secret"}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_http_agent_raises_a_clear_error_on_a_response_matching_neither_shape():
|
|
123
|
+
scenario = BY_ID["clean-1"]
|
|
124
|
+
requester = make_requester([{"something_else": "not a valid response"}])
|
|
125
|
+
|
|
126
|
+
with pytest.raises(ValueError, match="must include either 'final_answer' or"):
|
|
127
|
+
quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
|
|
128
|
+
|
|
129
|
+
|
|
55
130
|
def test_quick_test_runs_a_plain_function_end_to_end_and_reports_pass():
|
|
56
131
|
scenario = BY_ID["clean-1"]
|
|
57
132
|
report = quick_test(scenario, _clean1_decide, injector=NullInjector())
|
|
@@ -198,6 +273,29 @@ def test_quick_test_all_recovery_mode_has_no_clean_trajectories():
|
|
|
198
273
|
assert len(report.chaos_trajectories) == 2
|
|
199
274
|
|
|
200
275
|
|
|
276
|
+
def test_quick_test_all_reports_scenarios_run_and_total_when_nothing_stops_it_early():
|
|
277
|
+
scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"]}
|
|
278
|
+
report = quick_test_all(scenarios, lambda: RuleBasedAgent())
|
|
279
|
+
assert report._scenarios_run == 2
|
|
280
|
+
assert report._scenarios_total == 2
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def test_quick_test_all_stops_early_past_max_duration_seconds_but_always_runs_at_least_one():
|
|
284
|
+
# A caller with its own hard execution ceiling (e.g. a serverless
|
|
285
|
+
# platform that kills the whole job outright past some limit, with no
|
|
286
|
+
# chance to clean up) needs a report covering fewer scenarios instead
|
|
287
|
+
# of getting killed mid-run with nothing to show. max_duration_seconds=0
|
|
288
|
+
# means "already over budget" as soon as any real time has passed --
|
|
289
|
+
# deterministic regardless of how fast RuleBasedAgent actually runs,
|
|
290
|
+
# since real wall-clock time (time.monotonic()) always advances by a
|
|
291
|
+
# nonzero amount between the loop's start and its first elapsed check.
|
|
292
|
+
scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"], "clean-1-again": BY_ID["clean-1"]}
|
|
293
|
+
report = quick_test_all(scenarios, lambda: RuleBasedAgent(), max_duration_seconds=0)
|
|
294
|
+
assert report._scenarios_run == 1 # never zero -- see the "always run at least one" guard
|
|
295
|
+
assert report._scenarios_total == 3
|
|
296
|
+
assert report._compute()["n"] == 1 # the returned Report itself only covers what actually ran
|
|
297
|
+
|
|
298
|
+
|
|
201
299
|
def test_quick_test_all_uploads_each_scenario_as_its_own_run():
|
|
202
300
|
import agentprobe.registry as registry_module
|
|
203
301
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/__init__.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/agent.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/clean.py
RENAMED
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/decoy.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/domain.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/entities.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/scenarios.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/split.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/tools.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/trap.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/requires.txt
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_rule_based_agent.py
RENAMED
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_scenarios.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_complex_access_control_agent.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|