agentprobe-testing 0.5.1__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/PKG-INFO +1 -1
  2. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/README.md +18 -2
  3. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/__init__.py +1 -1
  4. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/quickstart.py +21 -1
  5. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/PKG-INFO +1 -1
  6. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/pyproject.toml +1 -1
  7. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_quickstart.py +23 -0
  8. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/LICENSE +0 -0
  9. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agents/__init__.py +0 -0
  10. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agents/base.py +0 -0
  11. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agents/rule_based.py +0 -0
  12. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agents/scripted.py +0 -0
  13. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agents/target_agent.py +0 -0
  14. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/agreement.py +0 -0
  15. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/classifier.py +0 -0
  16. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/cli.py +0 -0
  17. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/diff.py +0 -0
  18. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domain.py +0 -0
  19. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/__init__.py +0 -0
  20. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/__init__.py +0 -0
  21. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/agent.py +0 -0
  22. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/clean.py +0 -0
  23. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/complex_agent.py +0 -0
  24. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/decoy.py +0 -0
  25. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/domain.py +0 -0
  26. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/entities.py +0 -0
  27. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/injector_prompt.py +0 -0
  28. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/rule_based_agent.py +0 -0
  29. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/scenarios.py +0 -0
  30. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/split.py +0 -0
  31. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/tools.py +0 -0
  32. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/domains/access_control/trap.py +0 -0
  33. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/feedback.py +0 -0
  34. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/generic_world.py +0 -0
  35. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/injection.py +0 -0
  36. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/injector.py +0 -0
  37. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/llm.py +0 -0
  38. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/playbook.py +0 -0
  39. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/reachability.py +0 -0
  40. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/registry.py +0 -0
  41. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/report.py +0 -0
  42. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/runner.py +0 -0
  43. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenario.py +0 -0
  44. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/__init__.py +0 -0
  45. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/clean.py +0 -0
  46. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/decoy.py +0 -0
  47. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/registry.py +0 -0
  48. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/split.py +0 -0
  49. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/scenarios/trap.py +0 -0
  50. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/termui.py +0 -0
  51. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/tools.py +0 -0
  52. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/trajectory.py +0 -0
  53. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/triage.py +0 -0
  54. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/validate_scenarios.py +0 -0
  55. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe/world.py +0 -0
  56. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/SOURCES.txt +0 -0
  57. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/dependency_links.txt +0 -0
  58. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/entry_points.txt +0 -0
  59. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/requires.txt +0 -0
  60. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/agentprobe_testing.egg-info/top_level.txt +0 -0
  61. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/setup.cfg +0 -0
  62. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_access_control_agent.py +0 -0
  63. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_access_control_domain.py +0 -0
  64. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_access_control_rule_based_agent.py +0 -0
  65. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_access_control_scenarios.py +0 -0
  66. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_access_control_tools.py +0 -0
  67. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_agreement.py +0 -0
  68. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_classifier.py +0 -0
  69. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_cli.py +0 -0
  70. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_complex_access_control_agent.py +0 -0
  71. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_diff.py +0 -0
  72. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_domain.py +0 -0
  73. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_feedback.py +0 -0
  74. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_generic_world.py +0 -0
  75. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_injection.py +0 -0
  76. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_injector.py +0 -0
  77. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_llm.py +0 -0
  78. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_package_api.py +0 -0
  79. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_playbook.py +0 -0
  80. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_reachability.py +0 -0
  81. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_registry.py +0 -0
  82. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_report.py +0 -0
  83. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_rule_based_agent.py +0 -0
  84. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_runner.py +0 -0
  85. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_scenarios.py +0 -0
  86. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_target_agent.py +0 -0
  87. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_termui.py +0 -0
  88. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_tools.py +0 -0
  89. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_trajectory.py +0 -0
  90. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_triage.py +0 -0
  91. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_validate_scenarios.py +0 -0
  92. {agentprobe_testing-0.5.1 → agentprobe_testing-0.5.2}/tests/test_world.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -185,7 +185,10 @@ Writing `agentprobe/domain.py`'s `Domain` bundle by hand (entities, tools,
185
185
  business rules) is real, unavoidable work — see "Domains" above. If you'd
186
186
  rather not write it yourself, [agentprobe-api.agentprobe.workers.dev](https://agentprobe-api.agentprobe.workers.dev)
187
187
  can generate one from your existing tool schemas plus a plain-English
188
- description of your business rules, and hand you back a key:
188
+ description of your business rules, and hand you back a key — either via
189
+ [/generate](https://agentprobe-api.agentprobe.workers.dev/generate) on the
190
+ site (sign in, fill in a form, get the key and the generated source back
191
+ right on the page — no `curl` needed) or the raw API:
189
192
 
190
193
  ```python
191
194
  import agentprobe as ap
@@ -223,6 +226,19 @@ is the real site, not just the API — sign in with GitHub and you get:
223
226
  - **API access** (on the Dashboard) — create a personal key
224
227
  (`ap_live_...`, shown once at creation) to authenticate the two things
225
228
  above from your own scripts.
229
+ - **`/generate`** — the same domain generation as `POST /v1/generate`,
230
+ as a web form instead of `curl` — fill it in, get back the key and the
231
+ generated source to review right on the page.
232
+ - **`/run`** — execute a real chaos test from the website itself, no
233
+ terminal needed: paste a domain key, pick a scenario from a real
234
+ dropdown (or "All scenarios" to run every one, combined into a single
235
+ report), point at your agent's HTTP endpoint and your own Anthropic API
236
+ key, and it runs in the background — the page comes back immediately so
237
+ you can go do something else on the site while it finishes, and it
238
+ shows up under Your Runs once it does. Free to use — you're billed by
239
+ Anthropic for your own API usage, the same as running it yourself. See
240
+ [`runner/README.md`](runner/README.md) for how this actually executes server-side (the real `agentprobe-testing`
241
+ package, not a reimplementation).
226
242
 
227
243
  Attribute a generated domain to your account by passing the key as a
228
244
  bearer token:
@@ -288,7 +304,7 @@ The website/API server (`server/`) has its own separate suite, same rule:
288
304
  cd server && node --test test/
289
305
  ```
290
306
 
291
- 90+ tests against fake D1/KV/fetch — no real Cloudflare account, network
307
+ 160+ tests against fake D1/KV/fetch — no real Cloudflare account, network
292
308
  access, or secrets needed for either suite. Both run in CI on every push.
293
309
 
294
310
  See [CHANGELOG.md](CHANGELOG.md) for what's new in each version.
@@ -60,7 +60,7 @@ from agentprobe.scenario import CommitPattern, FactPattern, GoalSpec, Scenario
60
60
  from agentprobe.scenarios.registry import ALL_SCENARIOS as TICKET_SCENARIOS
61
61
  from agentprobe.scenarios.registry import BY_ID as TICKET_SCENARIOS_BY_ID
62
62
 
63
- __version__ = "0.5.1"
63
+ __version__ = "0.5.2"
64
64
 
65
65
  __all__ = [
66
66
  "Agent",
@@ -10,6 +10,7 @@ what to do next, test it against one scenario."
10
10
  from __future__ import annotations
11
11
 
12
12
  import json
13
+ import time
13
14
  import urllib.request
14
15
  from typing import Any, Callable, Optional
15
16
 
@@ -209,6 +210,7 @@ def quick_test_all(
209
210
  upload_api_key: Optional[str] = None,
210
211
  domain_key: Optional[str] = None,
211
212
  record_injections: bool = False,
213
+ max_duration_seconds: Optional[float] = None,
212
214
  ) -> Report:
213
215
  """Run every scenario in `scenarios` (a dict[str, Scenario] -- e.g.
214
216
  TICKET_SCENARIOS or a FetchedDomain's .scenarios -- or a plain list of
@@ -228,6 +230,16 @@ def quick_test_all(
228
230
  run (so Your Runs shows them individually, the same granularity as
229
231
  calling quick_test() per scenario and uploading each) -- the Report
230
232
  this function returns is still the combined summary across all of them.
233
+
234
+ `max_duration_seconds`, if given, stops starting new scenarios once
235
+ that much wall-clock time has passed since the call began -- for a
236
+ caller with its own hard execution ceiling (e.g. a serverless platform
237
+ that kills a job outright past some limit, which can't be caught or
238
+ cleaned up after), a report covering fewer scenarios than asked for is
239
+ far better than getting killed mid-scenario with nothing to show for
240
+ it. The returned Report's `_scenarios_run`/`_scenarios_total`
241
+ attributes say whether this happened, so a caller can tell the
242
+ difference between "ran everything" and "stopped early."
231
243
  """
232
244
  items = list(scenarios.items()) if isinstance(scenarios, dict) else [(s.id, s) for s in scenarios]
233
245
  if not items:
@@ -246,8 +258,13 @@ def quick_test_all(
246
258
  all_records: list[InjectionRecord] = []
247
259
  total_injector_cost = 0.0
248
260
  combined_injector_model: Optional[str] = None
261
+ start = time.monotonic()
262
+ scenarios_run = 0
249
263
 
250
264
  for _scenario_id, scenario in items:
265
+ if max_duration_seconds is not None and scenarios_run > 0 and (time.monotonic() - start) >= max_duration_seconds:
266
+ break # already have at least one scenario's worth of real results -- stop here, not mid-scenario
267
+ scenarios_run += 1
251
268
  injector = make_injector()
252
269
  resolved_injector_model = injector_model if injector_model is not None else type(injector).__name__
253
270
  if combined_injector_model is None:
@@ -289,7 +306,7 @@ def quick_test_all(
289
306
  recorded = [serialize_armed_injection(a) for a in recording.recorded] if recording is not None else None
290
307
  upload_run(scenario_report, api_key=upload_api_key, domain_key=domain_key, recorded_injections=recorded)
291
308
 
292
- return Report(
309
+ report = Report(
293
310
  mode=mode,
294
311
  injector_model=combined_injector_model or "none",
295
312
  target_model=target_model,
@@ -298,6 +315,9 @@ def quick_test_all(
298
315
  injection_records=all_records,
299
316
  injector_cost_usd=total_injector_cost,
300
317
  )
318
+ report._scenarios_run = len(all_chaos)
319
+ report._scenarios_total = len(items)
320
+ return report
301
321
 
302
322
 
303
323
  def assert_passes(report: Report, save_html_on_failure: Optional[str] = None) -> None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agentprobe-testing"
3
- version = "0.5.1"
3
+ version = "0.5.2"
4
4
  description = "Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance."
5
5
  requires-python = ">=3.11"
6
6
  dependencies = ["anthropic>=1.0.0", "python-dotenv>=1.0.0"]
@@ -273,6 +273,29 @@ def test_quick_test_all_recovery_mode_has_no_clean_trajectories():
273
273
  assert len(report.chaos_trajectories) == 2
274
274
 
275
275
 
276
+ def test_quick_test_all_reports_scenarios_run_and_total_when_nothing_stops_it_early():
277
+ scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"]}
278
+ report = quick_test_all(scenarios, lambda: RuleBasedAgent())
279
+ assert report._scenarios_run == 2
280
+ assert report._scenarios_total == 2
281
+
282
+
283
+ def test_quick_test_all_stops_early_past_max_duration_seconds_but_always_runs_at_least_one():
284
+ # A caller with its own hard execution ceiling (e.g. a serverless
285
+ # platform that kills the whole job outright past some limit, with no
286
+ # chance to clean up) needs a report covering fewer scenarios instead
287
+ # of getting killed mid-run with nothing to show. max_duration_seconds=0
288
+ # means "already over budget" as soon as any real time has passed --
289
+ # deterministic regardless of how fast RuleBasedAgent actually runs,
290
+ # since real wall-clock time (time.monotonic()) always advances by a
291
+ # nonzero amount between the loop's start and its first elapsed check.
292
+ scenarios = {"clean-1": BY_ID["clean-1"], "clean-2": BY_ID["clean-2"], "clean-1-again": BY_ID["clean-1"]}
293
+ report = quick_test_all(scenarios, lambda: RuleBasedAgent(), max_duration_seconds=0)
294
+ assert report._scenarios_run == 1 # never zero -- see the "always run at least one" guard
295
+ assert report._scenarios_total == 3
296
+ assert report._compute()["n"] == 1 # the returned Report itself only covers what actually ran
297
+
298
+
276
299
  def test_quick_test_all_uploads_each_scenario_as_its_own_run():
277
300
  import agentprobe.registry as registry_module
278
301