agentprobe-testing 0.5.2__tar.gz → 0.5.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/PKG-INFO +1 -1
  2. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/README.md +22 -20
  3. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/__init__.py +1 -1
  4. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/quickstart.py +1 -1
  5. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/registry.py +6 -6
  6. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/report.py +4 -4
  7. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/PKG-INFO +1 -1
  8. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/pyproject.toml +1 -1
  9. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_registry.py +2 -2
  10. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/LICENSE +0 -0
  11. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agents/__init__.py +0 -0
  12. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agents/base.py +0 -0
  13. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agents/rule_based.py +0 -0
  14. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agents/scripted.py +0 -0
  15. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agents/target_agent.py +0 -0
  16. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/agreement.py +0 -0
  17. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/classifier.py +0 -0
  18. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/cli.py +0 -0
  19. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/diff.py +0 -0
  20. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domain.py +0 -0
  21. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/__init__.py +0 -0
  22. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/__init__.py +0 -0
  23. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/agent.py +0 -0
  24. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/clean.py +0 -0
  25. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/complex_agent.py +0 -0
  26. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/decoy.py +0 -0
  27. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/domain.py +0 -0
  28. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/entities.py +0 -0
  29. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/injector_prompt.py +0 -0
  30. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/rule_based_agent.py +0 -0
  31. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/scenarios.py +0 -0
  32. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/split.py +0 -0
  33. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/tools.py +0 -0
  34. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/domains/access_control/trap.py +0 -0
  35. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/feedback.py +0 -0
  36. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/generic_world.py +0 -0
  37. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/injection.py +0 -0
  38. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/injector.py +0 -0
  39. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/llm.py +0 -0
  40. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/playbook.py +0 -0
  41. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/reachability.py +0 -0
  42. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/runner.py +0 -0
  43. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenario.py +0 -0
  44. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/__init__.py +0 -0
  45. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/clean.py +0 -0
  46. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/decoy.py +0 -0
  47. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/registry.py +0 -0
  48. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/split.py +0 -0
  49. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/scenarios/trap.py +0 -0
  50. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/termui.py +0 -0
  51. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/tools.py +0 -0
  52. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/trajectory.py +0 -0
  53. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/triage.py +0 -0
  54. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/validate_scenarios.py +0 -0
  55. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe/world.py +0 -0
  56. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/SOURCES.txt +0 -0
  57. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/dependency_links.txt +0 -0
  58. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/entry_points.txt +0 -0
  59. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/requires.txt +0 -0
  60. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/agentprobe_testing.egg-info/top_level.txt +0 -0
  61. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/setup.cfg +0 -0
  62. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_access_control_agent.py +0 -0
  63. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_access_control_domain.py +0 -0
  64. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_access_control_rule_based_agent.py +0 -0
  65. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_access_control_scenarios.py +0 -0
  66. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_access_control_tools.py +0 -0
  67. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_agreement.py +0 -0
  68. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_classifier.py +0 -0
  69. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_cli.py +0 -0
  70. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_complex_access_control_agent.py +0 -0
  71. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_diff.py +0 -0
  72. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_domain.py +0 -0
  73. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_feedback.py +0 -0
  74. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_generic_world.py +0 -0
  75. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_injection.py +0 -0
  76. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_injector.py +0 -0
  77. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_llm.py +0 -0
  78. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_package_api.py +0 -0
  79. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_playbook.py +0 -0
  80. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_quickstart.py +0 -0
  81. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_reachability.py +0 -0
  82. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_report.py +0 -0
  83. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_rule_based_agent.py +0 -0
  84. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_runner.py +0 -0
  85. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_scenarios.py +0 -0
  86. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_target_agent.py +0 -0
  87. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_termui.py +0 -0
  88. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_tools.py +0 -0
  89. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_trajectory.py +0 -0
  90. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_triage.py +0 -0
  91. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_validate_scenarios.py +0 -0
  92. {agentprobe_testing-0.5.2 → agentprobe_testing-0.5.3}/tests/test_world.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.2
3
+ Version: 0.5.3
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -193,10 +193,10 @@ right on the page — no `curl` needed) or the raw API:
193
193
  ```python
194
194
  import agentprobe as ap
195
195
 
196
- fetched = ap.fetch_domain("joeblow-ai-agents_bd0725d86767c7481581fc41")
196
+ fetched = ap.fetch_domain("meridian-ai-agents_bd0725d86767c7481581fc41")
197
197
  report = ap.quick_test(
198
198
  fetched.scenarios["wrong-order-among-several"],
199
- lambda: ap.TargetAgent(system_prompt="You are JoeBlow's order support agent."),
199
+ lambda: ap.TargetAgent(system_prompt="You are Meridian's order support agent."),
200
200
  domain=fetched.domain,
201
201
  )
202
202
  print(report.render())
@@ -213,32 +213,34 @@ it just makes your test pass without proving anything real about your
213
213
  agent. Only fetch a key you generated for your own business, the same
214
214
  rule as never installing a package from someone you don't trust.
215
215
 
216
- ## The website: accounts, personal API keys, and a dashboard
216
+ ## The website: sign in and run a real test in your browser
217
217
 
218
218
  [agentprobe-api.agentprobe.workers.dev](https://agentprobe-api.agentprobe.workers.dev)
219
219
  is the real site, not just the API — sign in with GitHub and you get:
220
220
 
221
- - **Your Keys** — every domain you've generated (via `/v1/generate` with
222
- your API key attached), click-to-copy, so you don't have to keep your
223
- own notes of which key was which business.
224
- - **Your Runs** — a history of test runs you've uploaded, each with its
225
- full report viewable in the browser.
226
- - **API access** (on the Dashboard) — create a personal key
227
- (`ap_live_...`, shown once at creation) to authenticate the two things
228
- above from your own scripts.
229
- - **`/generate`** — the same domain generation as `POST /v1/generate`,
230
- as a web form instead of `curl` — fill it in, get back the key and the
231
- generated source to review right on the page.
232
- - **`/run`** — execute a real chaos test from the website itself, no
233
- terminal needed: paste a domain key, pick a scenario from a real
234
- dropdown (or "All scenarios" to run every one, combined into a single
235
- report), point at your agent's HTTP endpoint and your own Anthropic API
236
- key, and it runs in the background — the page comes back immediately so
221
+ - **`/run`** — one flow, describing to running: describe your business in
222
+ plain English (same generation as `POST /v1/generate`, as a web form
223
+ instead of `curl`), and it writes a real test domain for it, then falls
224
+ straight into picking a scenario from a real dropdown (or "All
225
+ scenarios" to run every one, combined into a single report), naming the
226
+ run, and pointing at your agent's HTTP endpoint plus your own Anthropic
227
+ API key. It runs in the background — the page comes back immediately so
237
228
  you can go do something else on the site while it finishes, and it
238
229
  shows up under Your Runs once it does. Free to use — you're billed by
239
230
  Anthropic for your own API usage, the same as running it yourself. See
240
231
  [`runner/README.md`](runner/README.md) for how this actually executes server-side (the real `agentprobe-testing`
241
232
  package, not a reimplementation).
233
+ - **Your Runs** — a history of test runs you've uploaded or executed,
234
+ each with its full report viewable in the browser, and a Rerun link
235
+ that jumps straight back into `/run`'s scenario/agent step for any run
236
+ that was started from the site.
237
+
238
+ There's no separate domain-key page on the site anymore — a domain
239
+ generated through `/run` is only ever referenced internally, not
240
+ something you need to copy around. A personal API key
241
+ (`ap_live_...`) for scripting the CLI/library against your account, if
242
+ your account has one, is issued separately from the website (there's no
243
+ self-serve "create a key" page).
242
244
 
243
245
  Attribute a generated domain to your account by passing the key as a
244
246
  bearer token:
@@ -256,7 +258,7 @@ same key to `quick_test()`:
256
258
  ```python
257
259
  report = ap.quick_test(
258
260
  scenario, my_agent, domain=fetched.domain,
259
- upload_api_key="ap_live_...", # from Your Dashboard
261
+ upload_api_key="ap_live_...", # a personal key issued for your account
260
262
  domain_key=fetched_key, # optional -- links the run back to its domain
261
263
  )
262
264
  ```
@@ -60,7 +60,7 @@ from agentprobe.scenario import CommitPattern, FactPattern, GoalSpec, Scenario
60
60
  from agentprobe.scenarios.registry import ALL_SCENARIOS as TICKET_SCENARIOS
61
61
  from agentprobe.scenarios.registry import BY_ID as TICKET_SCENARIOS_BY_ID
62
62
 
63
- __version__ = "0.5.2"
63
+ __version__ = "0.5.3"
64
64
 
65
65
  __all__ = [
66
66
  "Agent",
@@ -136,7 +136,7 @@ def quick_test(
136
136
  or a plain decide_fn(task, history) -> AgentAction, which gets wrap_agent()
137
137
  applied automatically -- either way, nothing else here changes.
138
138
 
139
- Pass `upload_api_key` (a personal key from /dashboard) to also upload the
139
+ Pass `upload_api_key` (a personal key issued for your account) to also upload the
140
140
  finished report to your AgentProbe account, so it shows up under Your
141
141
  Runs -- see registry.upload_run for what this does and its failure
142
142
  behavior. Pass `domain_key` too if `domain` came from fetch_domain(), so
@@ -125,11 +125,11 @@ def upload_run(
125
125
  recorded_injections: Optional[list[dict]] = None,
126
126
  ) -> None:
127
127
  """Upload a finished quick_test() report to your AgentProbe account (POST
128
- /v1/runs) so it shows up under Your Runs on the dashboard. `api_key` is a
129
- personal key created on /dashboard -- create one there, it's never shown
130
- again after creation so save it somewhere real (an env var, not a
131
- committed file). `report` must cover exactly one scenario, same as every
132
- report quick_test() itself produces.
128
+ /v1/runs) so it shows up under Your Runs on the website. `api_key` is a
129
+ personal key issued for your account -- it's never shown again after
130
+ issuance so save it somewhere real (an env var, not a committed file).
131
+ `report` must cover exactly one scenario, same as every report
132
+ quick_test() itself produces.
133
133
 
134
134
  Pass `recorded_injections` (a list of agentprobe.injector.
135
135
  serialize_armed_injection() dicts) to make the uploaded run replayable
@@ -274,7 +274,7 @@ def replay_run(
274
274
  target_model: str = "custom",
275
275
  ) -> "Report":
276
276
  """Replay the exact chaos sequence recorded for `run_id` (an id from
277
- your dashboard's Your Runs table, or a RegressionResult.previous_run_id)
277
+ the website's Your Runs table, or a RegressionResult.previous_run_id)
278
278
  against a new agent or model version, instead of paying for a fresh
279
279
  ModelInjector decision sequence -- only the Target's own calls repeat.
280
280
 
@@ -564,14 +564,14 @@ class Report:
564
564
  {classification_html}
565
565
  </section>
566
566
  <section>
567
- <h2>By injection kind &mdash; handled rate</h2>
567
+ <h2>By injection kind: handled rate</h2>
568
568
  {kind_table}
569
569
  </section>
570
570
  <section>
571
- <h2>Scenario detail &mdash; step by step</h2>
571
+ <h2>Scenario detail: step by step</h2>
572
572
  {scenario_detail_html}
573
573
  </section>
574
- <footer>Generated by AgentProbe &mdash; adaptive chaos-testing for LLM agents.</footer>
574
+ <footer>Generated by AgentProbe. Adaptive chaos-testing for LLM agents.</footer>
575
575
  </div>
576
576
  </body></html>"""
577
577
 
@@ -621,7 +621,7 @@ class Report:
621
621
  notes = ""
622
622
  for a in traj.injections:
623
623
  note_class = "injection-note" if a.valid else "injection-note injection-note--invalid"
624
- validity = "" if a.valid else " &mdash; <strong>INVALID</strong>"
624
+ validity = "" if a.valid else " (<strong>INVALID</strong>)"
625
625
  notes += (
626
626
  f"<div class='{note_class}'>&#9889; injection fired: <strong>{esc(a.injection.kind.value)}</strong> "
627
627
  f"@ step {a.fired_at_step} (trigger: <code>{esc(a.trigger_kind)}</code>){validity}"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentprobe-testing
3
- Version: 0.5.2
3
+ Version: 0.5.3
4
4
  Summary: Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance.
5
5
  License: Business Source License 1.1
6
6
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agentprobe-testing"
3
- version = "0.5.2"
3
+ version = "0.5.3"
4
4
  description = "Adaptive chaos-testing for LLM agents: a live model-driven Injector that reads an agent's real trajectory and decides where to break something, instead of scripting perturbations in advance."
5
5
  requires-python = ">=3.11"
6
6
  dependencies = ["anthropic>=1.0.0", "python-dotenv>=1.0.0"]
@@ -175,9 +175,9 @@ def test_upload_run_includes_domain_key_when_given():
175
175
  report = quick_test(scenario, _clean1_decide)
176
176
  poster = make_poster()
177
177
 
178
- upload_run(report, api_key="ap_live_test123", domain_key="joeblow_1234", poster=poster)
178
+ upload_run(report, api_key="ap_live_test123", domain_key="meridian_1234", poster=poster)
179
179
 
180
- assert poster.calls[0]["body"]["domain_key"] == "joeblow_1234"
180
+ assert poster.calls[0]["body"]["domain_key"] == "meridian_1234"
181
181
 
182
182
 
183
183
  def test_upload_run_rejects_a_report_covering_more_or_fewer_than_one_scenario():