agentprobe-testing 0.5.0__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/PKG-INFO +1 -1
  2. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/README.md +37 -2
  3. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/__init__.py +3 -2
  4. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/quickstart.py +81 -1
  5. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/PKG-INFO +1 -1
  6. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/pyproject.toml +1 -1
  7. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_quickstart.py +99 -1
  8. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/LICENSE +0 -0
  9. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/__init__.py +0 -0
  10. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/base.py +0 -0
  11. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/rule_based.py +0 -0
  12. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/scripted.py +0 -0
  13. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agents/target_agent.py +0 -0
  14. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/agreement.py +0 -0
  15. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/classifier.py +0 -0
  16. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/cli.py +0 -0
  17. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/diff.py +0 -0
  18. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domain.py +0 -0
  19. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/__init__.py +0 -0
  20. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/__init__.py +0 -0
  21. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/agent.py +0 -0
  22. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/clean.py +0 -0
  23. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/complex_agent.py +0 -0
  24. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/decoy.py +0 -0
  25. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/domain.py +0 -0
  26. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/entities.py +0 -0
  27. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/injector_prompt.py +0 -0
  28. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/rule_based_agent.py +0 -0
  29. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/scenarios.py +0 -0
  30. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/split.py +0 -0
  31. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/tools.py +0 -0
  32. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/trap.py +0 -0
  33. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/feedback.py +0 -0
  34. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/generic_world.py +0 -0
  35. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/injection.py +0 -0
  36. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/injector.py +0 -0
  37. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/llm.py +0 -0
  38. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/playbook.py +0 -0
  39. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/reachability.py +0 -0
  40. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/registry.py +0 -0
  41. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/report.py +0 -0
  42. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/runner.py +0 -0
  43. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenario.py +0 -0
  44. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/__init__.py +0 -0
  45. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/clean.py +0 -0
  46. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/decoy.py +0 -0
  47. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/registry.py +0 -0
  48. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/split.py +0 -0
  49. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/scenarios/trap.py +0 -0
  50. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/termui.py +0 -0
  51. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/tools.py +0 -0
  52. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/trajectory.py +0 -0
  53. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/triage.py +0 -0
  54. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/validate_scenarios.py +0 -0
  55. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe/world.py +0 -0
  56. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/SOURCES.txt +0 -0
  57. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/dependency_links.txt +0 -0
  58. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/entry_points.txt +0 -0
  59. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/requires.txt +0 -0
  60. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/top_level.txt +0 -0
  61. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/setup.cfg +0 -0
  62. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_agent.py +0 -0
  63. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_domain.py +0 -0
  64. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_rule_based_agent.py +0 -0
  65. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_scenarios.py +0 -0
  66. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_access_control_tools.py +0 -0
  67. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_agreement.py +0 -0
  68. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_classifier.py +0 -0
  69. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_cli.py +0 -0
  70. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_complex_access_control_agent.py +0 -0
  71. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_diff.py +0 -0
  72. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_domain.py +0 -0
  73. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_feedback.py +0 -0
  74. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_generic_world.py +0 -0
  75. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_injection.py +0 -0
  76. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_injector.py +0 -0
  77. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_llm.py +0 -0
  78. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_package_api.py +0 -0
  79. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_playbook.py +0 -0
  80. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_reachability.py +0 -0
  81. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_registry.py +0 -0
  82. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_report.py +0 -0
  83. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_rule_based_agent.py +0 -0
  84. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_runner.py +0 -0
  85. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_scenarios.py +0 -0
  86. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_target_agent.py +0 -0
  87. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_termui.py +0 -0
  88. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_tools.py +0 -0
  89. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_trajectory.py +0 -0
  90. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_triage.py +0 -0
  91. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_validate_scenarios.py +0 -0
  92. {agentprobe_testing-0.5.0 → agentprobe_testing-0.5.2}/tests/test_world.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.0
3
+ Version: 0.5.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -18,6 +18,19 @@ adversarial content disguised as ordinary data (`PROMPT_INJECTION`).
18
18
 
19
19
  ## Quickstart
20
20
 
21
+ Using this in your own project (no need to clone this repo):
22
+
23
+ ```bash
24
+ pip install agentprobe-testing
25
+ ```
26
+
27
+ `import agentprobe as ap` afterward, same as everywhere else in this
28
+ README — the PyPI *distribution* name is `agentprobe-testing`, but the
29
+ Python package you import is still plain `agentprobe`.
30
+
31
+ To run the examples in this repo (or work on the library itself), clone
32
+ it and install from source instead:
33
+
21
34
  ```bash
22
35
  pip install -e .
23
36
  python examples/free_zero_cost_demo.py # no API key needed, no network calls at all
@@ -105,6 +118,12 @@ surface. The `access_control` domain lives at
105
118
  top level, to keep this surface to the one domain most people reach for
106
119
  first.
107
120
 
121
+ Testing your own agent instead of the built-in `TargetAgent`? Pass any
122
+ plain `(task, history) -> AgentAction` function as `agent` — no subclass
123
+ needed, `quick_test`/`wrap_agent` handle the wiring, and if your agent
124
+ already runs as a service you can POST to, `ap.http_agent(url)` builds
125
+ that function for you (see [USAGE.md](USAGE.md)).
126
+
108
127
  Beyond one-scenario `quick_test()`, see [USAGE.md](USAGE.md) for: running
109
128
  a whole domain at once (`quick_test_all`), a one-line pytest integration
110
129
  (`assert_passes`), catching regressions between uploaded runs
@@ -166,7 +185,10 @@ Writing `agentprobe/domain.py`'s `Domain` bundle by hand (entities, tools,
166
185
  business rules) is real, unavoidable work — see "Domains" above. If you'd
167
186
  rather not write it yourself, [agentprobe-api.agentprobe.workers.dev](https://agentprobe-api.agentprobe.workers.dev)
168
187
  can generate one from your existing tool schemas plus a plain-English
169
- description of your business rules, and hand you back a key:
188
+ description of your business rules, and hand you back a key — either via
189
+ [/generate](https://agentprobe-api.agentprobe.workers.dev/generate) on the
190
+ site (sign in, fill in a form, get the key and the generated source back
191
+ right on the page — no `curl` needed) or the raw API:
170
192
 
171
193
  ```python
172
194
  import agentprobe as ap
@@ -204,6 +226,19 @@ is the real site, not just the API — sign in with GitHub and you get:
204
226
  - **API access** (on the Dashboard) — create a personal key
205
227
  (`ap_live_...`, shown once at creation) to authenticate the two things
206
228
  above from your own scripts.
229
+ - **`/generate`** — the same domain generation as `POST /v1/generate`,
230
+ as a web form instead of `curl` — fill it in, get back the key and the
231
+ generated source to review right on the page.
232
+ - **`/run`** — execute a real chaos test from the website itself, no
233
+ terminal needed: paste a domain key, pick a scenario from a real
234
+ dropdown (or "All scenarios" to run every one, combined into a single
235
+ report), point at your agent's HTTP endpoint and your own Anthropic API
236
+ key, and it runs in the background — the page comes back immediately so
237
+ you can go do something else on the site while it finishes, and it
238
+ shows up under Your Runs once it does. Free to use — you're billed by
239
+ Anthropic for your own API usage, the same as running it yourself. See
240
+ [`runner/README.md`](runner/README.md) for how this actually executes server-side (the real `agentprobe-testing`
241
+ package, not a reimplementation).
207
242
 
208
243
  Attribute a generated domain to your account by passing the key as a
209
244
  bearer token:
@@ -269,7 +304,7 @@ The website/API server (`server/`) has its own separate suite, same rule:
269
304
  cd server && node --test test/
270
305
  ```
271
306
 
272
- 90+ tests against fake D1/KV/fetch — no real Cloudflare account, network
307
+ 160+ tests against fake D1/KV/fetch — no real Cloudflare account, network
273
308
  access, or secrets needed for either suite. Both run in CI on every push.
274
309
 
275
310
  See [CHANGELOG.md](CHANGELOG.md) for what's new in each version.
@@ -44,7 +44,7 @@ from agentprobe.injector import (
44
44
  RecordingInjector,
45
45
  ReplayInjector,
46
46
  )
47
- from agentprobe.quickstart import assert_passes, quick_test, quick_test_all, wrap_agent
47
+ from agentprobe.quickstart import assert_passes, http_agent, quick_test, quick_test_all, wrap_agent
48
48
  from agentprobe.registry import (
49
49
  FetchedDomain,
50
50
  RegressionResult,
@@ -60,7 +60,7 @@ from agentprobe.scenario import CommitPattern, FactPattern, GoalSpec, Scenario
60
60
  from agentprobe.scenarios.registry import ALL_SCENARIOS as TICKET_SCENARIOS
61
61
  from agentprobe.scenarios.registry import BY_ID as TICKET_SCENARIOS_BY_ID
62
62
 
63
- __version__ = "0.5.0"
63
+ __version__ = "0.5.2"
64
64
 
65
65
  __all__ = [
66
66
  "Agent",
@@ -93,6 +93,7 @@ __all__ = [
93
93
  "check_regression",
94
94
  "fetch_domain",
95
95
  "get_latest_run",
96
+ "http_agent",
96
97
  "injection_was_triggered",
97
98
  "quick_test",
98
99
  "quick_test_all",
@@ -9,6 +9,9 @@ what to do next, test it against one scenario."
9
9
 
10
10
  from __future__ import annotations
11
11
 
12
+ import json
13
+ import time
14
+ import urllib.request
12
15
  from typing import Any, Callable, Optional
13
16
 
14
17
  from agentprobe.agents.base import Agent, AgentAction
@@ -57,6 +60,64 @@ def wrap_agent(decide_fn: DecideFn) -> Callable[[], Agent]:
57
60
  return lambda: _FunctionAgent(decide_fn)
58
61
 
59
62
 
63
+ HttpRequester = Callable[[str, dict, dict], dict] # (url, body, headers) -> response dict
64
+
65
+
66
+ def _default_http_requester(url: str, body: dict, headers: dict) -> dict:
67
+ data = json.dumps(body).encode("utf-8")
68
+ request = urllib.request.Request(
69
+ url,
70
+ data=data,
71
+ method="POST",
72
+ headers={"Content-Type": "application/json", "User-Agent": "agentprobe-client/0.5.0", **headers},
73
+ )
74
+ with urllib.request.urlopen(request, timeout=30) as resp: # noqa: S310 -- caller-supplied url, same trust model as any target agent's own API
75
+ return json.loads(resp.read().decode("utf-8"))
76
+
77
+
78
+ def http_agent(
79
+ url: str,
80
+ headers: Optional[dict] = None,
81
+ requester: Optional[HttpRequester] = None,
82
+ ) -> Callable[[], Agent]:
83
+ """Adapt a hosted HTTP agent into a target_factory for quick_test()/
84
+ run_robustness_pair() -- for when your real agent already runs as a
85
+ service you can POST to, instead of a Python function/object living in
86
+ this same process. No custom decide_fn to write: this builds one for
87
+ you and wraps it exactly like wrap_agent() does.
88
+
89
+ On every step, POSTs `{"task": <str>, "history": [...]}` as JSON to
90
+ `url` -- `history` is the same shape wrap_agent's decide_fn receives, a
91
+ list of `{"tool_name", "tool_args", "result", "ok"}` dicts covering
92
+ everything observed so far this run. Expects back JSON shaped like
93
+ either:
94
+
95
+ {"tool_name": "...", "tool_args": {...}} -- call this tool next
96
+ {"final_answer": "..."} -- done, this is the reply
97
+
98
+ Raises ValueError if a response matches neither shape, rather than
99
+ guessing what was meant. Pass `headers` (e.g.
100
+ `{"Authorization": "Bearer ..."}`) for an endpoint that needs auth, and
101
+ `requester` in tests instead of hitting the real network -- see
102
+ tests/test_quickstart.py.
103
+ """
104
+ requester = requester if requester is not None else _default_http_requester
105
+ resolved_headers = headers or {}
106
+
107
+ def decide_fn(task: str, history: list[dict[str, Any]]) -> AgentAction:
108
+ response = requester(url, {"task": task, "history": history}, resolved_headers)
109
+ if "final_answer" in response:
110
+ return AgentAction(kind="final_answer", text=response["final_answer"])
111
+ if "tool_name" in response and "tool_args" in response:
112
+ return AgentAction(kind="tool_call", tool_name=response["tool_name"], tool_args=response["tool_args"])
113
+ raise ValueError(
114
+ f"http_agent: response from {url} must include either 'final_answer' or "
115
+ f"both 'tool_name' and 'tool_args', got keys {sorted(response.keys())}"
116
+ )
117
+
118
+ return wrap_agent(decide_fn)
119
+
120
+
60
121
  def quick_test(
61
122
  scenario: Scenario,
62
123
  agent: Callable[[], Agent] | DecideFn,
@@ -149,6 +210,7 @@ def quick_test_all(
149
210
  upload_api_key: Optional[str] = None,
150
211
  domain_key: Optional[str] = None,
151
212
  record_injections: bool = False,
213
+ max_duration_seconds: Optional[float] = None,
152
214
  ) -> Report:
153
215
  """Run every scenario in `scenarios` (a dict[str, Scenario] -- e.g.
154
216
  TICKET_SCENARIOS or a FetchedDomain's .scenarios -- or a plain list of
@@ -168,6 +230,16 @@ def quick_test_all(
168
230
  run (so Your Runs shows them individually, the same granularity as
169
231
  calling quick_test() per scenario and uploading each) -- the Report
170
232
  this function returns is still the combined summary across all of them.
233
+
234
+ `max_duration_seconds`, if given, stops starting new scenarios once
235
+ that much wall-clock time has passed since the call began -- for a
236
+ caller with its own hard execution ceiling (e.g. a serverless platform
237
+ that kills a job outright past some limit, which can't be caught or
238
+ cleaned up after), a report covering fewer scenarios than asked for is
239
+ far better than getting killed mid-scenario with nothing to show for
240
+ it. The returned Report's `_scenarios_run`/`_scenarios_total`
241
+ attributes say whether this happened, so a caller can tell the
242
+ difference between "ran everything" and "stopped early."
171
243
  """
172
244
  items = list(scenarios.items()) if isinstance(scenarios, dict) else [(s.id, s) for s in scenarios]
173
245
  if not items:
@@ -186,8 +258,13 @@ def quick_test_all(
186
258
  all_records: list[InjectionRecord] = []
187
259
  total_injector_cost = 0.0
188
260
  combined_injector_model: Optional[str] = None
261
+ start = time.monotonic()
262
+ scenarios_run = 0
189
263
 
190
264
  for _scenario_id, scenario in items:
265
+ if max_duration_seconds is not None and scenarios_run > 0 and (time.monotonic() - start) >= max_duration_seconds:
266
+ break # already have at least one scenario's worth of real results -- stop here, not mid-scenario
267
+ scenarios_run += 1
191
268
  injector = make_injector()
192
269
  resolved_injector_model = injector_model if injector_model is not None else type(injector).__name__
193
270
  if combined_injector_model is None:
@@ -229,7 +306,7 @@ def quick_test_all(
229
306
  recorded = [serialize_armed_injection(a) for a in recording.recorded] if recording is not None else None
230
307
  upload_run(scenario_report, api_key=upload_api_key, domain_key=domain_key, recorded_injections=recorded)
231
308
 
232
- return Report(
309
+ report = Report(
233
310
  mode=mode,
234
311
  injector_model=combined_injector_model or "none",
235
312
  target_model=target_model,
@@ -238,6 +315,9 @@ def quick_test_all(
238
315
  injection_records=all_records,
239
316
  injector_cost_usd=total_injector_cost,
240
317
  )
318
+ report._scenarios_run = len(all_chaos)
319
+ report._scenarios_total = len(items)
320
+ return report
241
321
 
242
322
 
243
323
  def assert_passes(report: Report, save_html_on_failure: Optional[str] = None) -> None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.0
3
+ Version: 0.5.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agentprobe-testing"
3
- version = "0.5.0"
3
+ version = "0.5.2"
4
4
  description = "Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance."
5
5
  requires-python = ">=3.11"
6
6
  dependencies = ["anthropic>=1.0.0", "python-dotenv>=1.0.0"]
@@ -3,7 +3,7 @@ import pytest
3
3
  from agentprobe.agents.base import AgentAction
4
4
  from agentprobe.agents.rule_based import RuleBasedAgent
5
5
  from agentprobe.injector import HardcodedToolErrorInjector, Injector, NullInjector
6
- from agentprobe.quickstart import assert_passes, quick_test, quick_test_all, wrap_agent
6
+ from agentprobe.quickstart import assert_passes, http_agent, quick_test, quick_test_all, wrap_agent
7
7
  from agentprobe.report import Report
8
8
  from agentprobe.scenarios.registry import BY_ID
9
9
 
@@ -52,6 +52,81 @@ def test_wrap_agent_gives_each_call_its_own_history_not_shared_across_instances(
52
52
  assert action.tool_name == "get_ticket" # fresh history, not polluted by `first`
53
53
 
54
54
 
55
+ def make_requester(responses):
56
+ """Returns each queued response in order, on every call -- the same
57
+ scripted-sequence pattern ScriptedAgent uses, since http_agent's whole
58
+ point is standing in for a real remote agent step by step."""
59
+ calls = []
60
+ responses = list(responses)
61
+
62
+ def requester(url, body, headers):
63
+ calls.append({"url": url, "body": body, "headers": headers})
64
+ return responses.pop(0)
65
+
66
+ requester.calls = calls
67
+ return requester
68
+
69
+
70
+ def test_http_agent_drives_a_full_run_from_scripted_json_responses():
71
+ # mode="recovery" -- a robustness-mode run calls the target_factory
72
+ # twice (clean then chaos), which would drain this same scripted
73
+ # response queue twice over; recovery mode's single trajectory keeps
74
+ # this test's response list simple and matched 1:1 to real steps.
75
+ scenario = BY_ID["clean-1"]
76
+ requester = make_requester(
77
+ [
78
+ {"tool_name": "get_ticket", "tool_args": {"id": "T-1001"}},
79
+ {"tool_name": "issue_refund", "tool_args": {"ticket_id": "T-1001", "amount": 49.99}},
80
+ {"tool_name": "close_ticket", "tool_args": {"ticket_id": "T-1001", "resolution": "refunded"}},
81
+ {"final_answer": "done"},
82
+ ]
83
+ )
84
+
85
+ report = quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
86
+
87
+ assert report._compute()["chaos_passed"] == 1
88
+ assert requester.calls[0]["url"] == "https://example.com/agent"
89
+
90
+
91
+ def test_http_agent_sends_task_and_history_as_the_request_body():
92
+ scenario = BY_ID["clean-1"]
93
+ requester = make_requester(
94
+ [
95
+ {"tool_name": "get_ticket", "tool_args": {"id": "T-1001"}},
96
+ {"final_answer": "done"},
97
+ ]
98
+ )
99
+
100
+ quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
101
+
102
+ first_call, second_call = requester.calls
103
+ assert first_call["body"]["task"] == "Handle ticket T-1001."
104
+ assert first_call["body"]["history"] == []
105
+ assert second_call["body"]["history"][0]["tool_name"] == "get_ticket"
106
+
107
+
108
+ def test_http_agent_passes_through_custom_headers():
109
+ scenario = BY_ID["clean-1"]
110
+ requester = make_requester([{"final_answer": "done"}])
111
+
112
+ quick_test(
113
+ scenario,
114
+ http_agent("https://example.com/agent", headers={"Authorization": "Bearer secret"}, requester=requester),
115
+ injector=NullInjector(),
116
+ mode="recovery",
117
+ )
118
+
119
+ assert requester.calls[0]["headers"] == {"Authorization": "Bearer secret"}
120
+
121
+
122
+ def test_http_agent_raises_a_clear_error_on_a_response_matching_neither_shape():
123
+ scenario = BY_ID["clean-1"]
124
+ requester = make_requester([{"something_else": "not a valid response"}])
125
+
126
+ with pytest.raises(ValueError, match="must include either 'final_answer' or"):
127
+ quick_test(scenario, http_agent("https://example.com/agent", requester=requester), injector=NullInjector(), mode="recovery")
128
+
129
+
55
130
  def test_quick_test_runs_a_plain_function_end_to_end_and_reports_pass():
56
131
  scenario = BY_ID["clean-1"]
57
132
  report = quick_test(scenario, _clean1_decide, injector=NullInjector())
@@ -198,6 +273,29 @@ def test_quick_test_all_recovery_mode_has_no_clean_trajectories():
198
273
  assert len(report.chaos_trajectories) == 2
199
274
 
200
275
 
276
+ def test_quick_test_all_reports_scenarios_run_and_total_when_nothing_stops_it_early():
277
+ scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"]}
278
+ report = quick_test_all(scenarios, lambda: RuleBasedAgent())
279
+ assert report._scenarios_run == 2
280
+ assert report._scenarios_total == 2
281
+
282
+
283
+ def test_quick_test_all_stops_early_past_max_duration_seconds_but_always_runs_at_least_one():
284
+ # A caller with its own hard execution ceiling (e.g. a serverless
285
+ # platform that kills the whole job outright past some limit, with no
286
+ # chance to clean up) needs a report covering fewer scenarios instead
287
+ # of getting killed mid-run with nothing to show. max_duration_seconds=0
288
+ # means "already over budget" as soon as any real time has passed --
289
+ # deterministic regardless of how fast RuleBasedAgent actually runs,
290
+ # since real wall-clock time (time.monotonic()) always advances by a
291
+ # nonzero amount between the loop's start and its first elapsed check.
292
+ scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"], "clean-1-again": BY_ID["clean-1"]}
293
+ report = quick_test_all(scenarios, lambda: RuleBasedAgent(), max_duration_seconds=0)
294
+ assert report._scenarios_run == 1 # never zero -- see the "always run at least one" guard
295
+ assert report._scenarios_total == 3
296
+ assert report._compute()["n"] == 1 # the returned Report itself only covers what actually ran
297
+
298
+
201
299
  def test_quick_test_all_uploads_each_scenario_as_its_own_run():
202
300
  import agentprobe.registry as registry_module
203
301