agentprobe-testing 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. agentprobe/__init__.py +104 -0
  2. agentprobe/agents/__init__.py +0 -0
  3. agentprobe/agents/base.py +32 -0
  4. agentprobe/agents/rule_based.py +336 -0
  5. agentprobe/agents/scripted.py +30 -0
  6. agentprobe/agents/target_agent.py +106 -0
  7. agentprobe/agreement.py +80 -0
  8. agentprobe/classifier.py +159 -0
  9. agentprobe/cli.py +684 -0
  10. agentprobe/diff.py +150 -0
  11. agentprobe/domain.py +121 -0
  12. agentprobe/domains/__init__.py +0 -0
  13. agentprobe/domains/access_control/__init__.py +0 -0
  14. agentprobe/domains/access_control/agent.py +90 -0
  15. agentprobe/domains/access_control/clean.py +154 -0
  16. agentprobe/domains/access_control/complex_agent.py +123 -0
  17. agentprobe/domains/access_control/decoy.py +124 -0
  18. agentprobe/domains/access_control/domain.py +35 -0
  19. agentprobe/domains/access_control/entities.py +43 -0
  20. agentprobe/domains/access_control/injector_prompt.py +196 -0
  21. agentprobe/domains/access_control/rule_based_agent.py +263 -0
  22. agentprobe/domains/access_control/scenarios.py +17 -0
  23. agentprobe/domains/access_control/split.py +96 -0
  24. agentprobe/domains/access_control/tools.py +235 -0
  25. agentprobe/domains/access_control/trap.py +100 -0
  26. agentprobe/feedback.py +121 -0
  27. agentprobe/generic_world.py +99 -0
  28. agentprobe/injection.py +475 -0
  29. agentprobe/injector.py +810 -0
  30. agentprobe/llm.py +123 -0
  31. agentprobe/playbook.py +211 -0
  32. agentprobe/quickstart.py +295 -0
  33. agentprobe/reachability.py +196 -0
  34. agentprobe/registry.py +313 -0
  35. agentprobe/report.py +666 -0
  36. agentprobe/runner.py +317 -0
  37. agentprobe/scenario.py +75 -0
  38. agentprobe/scenarios/__init__.py +0 -0
  39. agentprobe/scenarios/clean.py +194 -0
  40. agentprobe/scenarios/decoy.py +272 -0
  41. agentprobe/scenarios/registry.py +16 -0
  42. agentprobe/scenarios/split.py +203 -0
  43. agentprobe/scenarios/trap.py +215 -0
  44. agentprobe/termui.py +154 -0
  45. agentprobe/tools.py +275 -0
  46. agentprobe/trajectory.py +107 -0
  47. agentprobe/triage.py +153 -0
  48. agentprobe/validate_scenarios.py +489 -0
  49. agentprobe/world.py +189 -0
  50. agentprobe_testing-0.5.0.dist-info/METADATA +127 -0
  51. agentprobe_testing-0.5.0.dist-info/RECORD +55 -0
  52. agentprobe_testing-0.5.0.dist-info/WHEEL +5 -0
  53. agentprobe_testing-0.5.0.dist-info/entry_points.txt +4 -0
  54. agentprobe_testing-0.5.0.dist-info/licenses/LICENSE +109 -0
  55. agentprobe_testing-0.5.0.dist-info/top_level.txt +1 -0
agentprobe/cli.py ADDED
@@ -0,0 +1,684 @@
1
+ """CLI entrypoint: python -m agentprobe.cli run --mode robustness|recovery"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+
9
+ from dotenv import load_dotenv
10
+
11
+ from agentprobe.report import InjectionRecord, Report
12
+ from agentprobe.runner import run_recovery, run_robustness_pair
13
+ from agentprobe.scenarios.registry import ALL_SCENARIOS
14
+ from agentprobe.termui import Spinner, green, pass_fail, red, yellow
15
+
16
+
17
+ def _make_injector(
18
+ name: str,
19
+ model: str,
20
+ hardcoded_offsets: list[int],
21
+ hardcoded_tool: str,
22
+ kind_policy: str,
23
+ scenario_id: str | None = None,
24
+ replay_data: dict | None = None,
25
+ playbook=None,
26
+ scenario_shape=None,
27
+ domain=None,
28
+ ):
29
+ if name == "none":
30
+ from agentprobe.injector import NullInjector
31
+
32
+ return NullInjector()
33
+ if name == "hardcoded":
34
+ from agentprobe.injector import HardcodedToolErrorInjector
35
+
36
+ return HardcodedToolErrorInjector(offsets=hardcoded_offsets, tool_name=hardcoded_tool)
37
+ if name == "model":
38
+ from agentprobe.domain import TICKET_DOMAIN
39
+
40
+ domain = domain or TICKET_DOMAIN
41
+ if domain.injector_system_prompt is None:
42
+ # This Domain hasn't been given its own Injector system prompt
43
+ # (see domain.py's injector_system_prompt field) -- pointing
44
+ # ModelInjector at it anyway would send a prompt written for a
45
+ # different domain's vocabulary. hardcoded/none work fine under
46
+ # any domain in the meantime (see _make_injector's other
47
+ # branches -- neither references domain vocabulary at all).
48
+ raise ValueError(
49
+ "--injector model isn't wired up for this --domain yet (no injector_system_prompt "
50
+ "set on its Domain). Use --injector hardcoded or none instead."
51
+ )
52
+ from agentprobe.injector import ModelInjector
53
+
54
+ return ModelInjector(
55
+ model=model,
56
+ policy=kind_policy,
57
+ playbook=playbook,
58
+ scenario_shape=scenario_shape,
59
+ system_prompt=domain.injector_system_prompt,
60
+ all_tools=domain.all_tools,
61
+ commit_tools=domain.commit_tools,
62
+ entity_types=domain.injector_entity_types,
63
+ )
64
+ if name == "replay":
65
+ from agentprobe.injector import NullInjector, ReplayInjector, deserialize_armed_injection
66
+
67
+ recorded = (replay_data or {}).get(scenario_id)
68
+ if recorded is None:
69
+ print(
70
+ f"no recorded injections for scenario {scenario_id!r} in replay file -- using NullInjector",
71
+ file=sys.stderr,
72
+ )
73
+ return NullInjector()
74
+ return ReplayInjector([deserialize_armed_injection(d) for d in recorded])
75
+ raise ValueError(f"unknown injector {name!r}")
76
+
77
+
78
+ def _load_json_file(path: str) -> dict | None:
79
+ """Loads a JSON file for a CLI flag/argument, printing a clean
80
+ one-line message and returning None on a bad path or malformed JSON
81
+ instead of letting FileNotFoundError/JSONDecodeError surface as a raw
82
+ traceback -- both are ordinary user typos (wrong path, half-written
83
+ file), not harness bugs."""
84
+ try:
85
+ with open(path) as f:
86
+ return json.load(f)
87
+ except FileNotFoundError:
88
+ print(f"no such file: {path}", file=sys.stderr)
89
+ return None
90
+ except json.JSONDecodeError as e:
91
+ print(f"{path} is not valid JSON: {e}", file=sys.stderr)
92
+ return None
93
+
94
+
95
+ def _injector_cost_usd(injector) -> float:
96
+ """ModelInjector is the only injector type with API cost (none/
97
+ hardcoded/replay never call a model). Unwraps RecordingInjector's
98
+ wrapping too, since --record-injections swaps in a wrapper that
99
+ doesn't have its own total_cost_usd."""
100
+ inner = getattr(injector, "_inner", injector)
101
+ return getattr(inner, "total_cost_usd", 0.0)
102
+
103
+
104
+ def _target_factory(target: str, model: str, domain_name: str = "ticket"):
105
+ """Target is an Agent (agentprobe/agents/base.py) -- the runner only
106
+ ever calls start()/next_action()/observe(), so any implementation is
107
+ swappable here. "claude" wraps the real Claude tool-calling loop for
108
+ whichever domain is selected; "scripted" is the dummy replay agent
109
+ used to exercise the world/tools/runner without live API calls (see
110
+ agents/scripted.py) -- it always immediately gives a final answer with
111
+ no script, since a useful script is scenario-specific and has no
112
+ general CLI form, and it's domain-agnostic (no tool-name knowledge
113
+ baked in), so it works under any --domain. "rule_based" is a real (if
114
+ arbitrary) deterministic decision tree -- zero API cost like scripted,
115
+ but with actual branching/retry/recovery logic for the harness to
116
+ chaos-test against -- each domain has its own (see agents/
117
+ rule_based.py for ticket, domains/access_control/rule_based_agent.py
118
+ for access_control).
119
+ """
120
+ if target == "scripted":
121
+ from agentprobe.agents.scripted import ScriptedAgent
122
+
123
+ return lambda: ScriptedAgent(script=[])
124
+ if domain_name == "access_control":
125
+ if target == "claude":
126
+ from agentprobe.domains.access_control.agent import AccessControlAgent
127
+
128
+ return lambda: AccessControlAgent(model=model)
129
+ if target == "rule_based":
130
+ from agentprobe.domains.access_control.rule_based_agent import AccessControlRuleBasedAgent
131
+
132
+ return lambda: AccessControlRuleBasedAgent()
133
+ raise ValueError(f"target {target!r} is not available under --domain access_control (try 'claude', 'scripted', or 'rule_based')")
134
+ if target == "claude":
135
+ from agentprobe.agents.target_agent import TargetAgent
136
+
137
+ return lambda: TargetAgent(model=model)
138
+ if target == "rule_based":
139
+ from agentprobe.agents.rule_based import RuleBasedAgent
140
+
141
+ return lambda: RuleBasedAgent()
142
+ raise ValueError(f"unknown target {target!r}")
143
+
144
+
145
+ def _serialize_trajectory(t) -> dict:
146
+ return {
147
+ "scenario_id": t.scenario_id,
148
+ "variant": t.variant,
149
+ "passed": t.passed,
150
+ "final_status": t.final_status.value,
151
+ "breaking_step": t.breaking_step,
152
+ "wasted_steps": t.wasted_steps,
153
+ "halted_reason": t.halted_reason,
154
+ "final_answer": t.final_answer,
155
+ "discarded": t.discarded,
156
+ "injector_errors": t.injector_errors,
157
+ "steps": [
158
+ {
159
+ "index": s.index,
160
+ "tool_name": s.tool_name,
161
+ "tool_args": s.tool_args,
162
+ "ok": s.ok,
163
+ "result": s.result,
164
+ "reachability": s.reachability.status.value,
165
+ "reason": s.reachability.reason,
166
+ "injected_error": s.injected_error,
167
+ }
168
+ for s in t.steps
169
+ ],
170
+ "injections": [
171
+ {
172
+ "armed_at_step": a.armed_at_step,
173
+ "fired_at_step": a.fired_at_step,
174
+ "kind": a.injection.kind.value,
175
+ "payload": a.injection.payload,
176
+ "intent": a.injection.intent,
177
+ "expected_signature": a.injection.expected_signature.description,
178
+ "rationale": a.injection.rationale,
179
+ "effect": a.effect,
180
+ "valid": a.valid,
181
+ }
182
+ for a in t.injections
183
+ ],
184
+ "expired_injections": [
185
+ {
186
+ "armed_at_step": e.armed_at_step,
187
+ "kind": e.injection.kind.value,
188
+ "payload": e.injection.payload,
189
+ "trigger": e.trigger.describe(),
190
+ "rationale": e.injection.rationale,
191
+ }
192
+ for e in t.expired_injections
193
+ ],
194
+ }
195
+
196
+
197
+ def main(argv: list[str] | None = None) -> int:
198
+ from agentprobe import __version__
199
+
200
+ parser = argparse.ArgumentParser(prog="agentprobe")
201
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
202
+ sub = parser.add_subparsers(dest="command", required=True)
203
+
204
+ run_p = sub.add_parser("run", help="Run scenarios in robustness or recovery mode.")
205
+ run_p.add_argument("--mode", default="robustness", choices=["robustness", "recovery"])
206
+ run_p.add_argument("--injector", default="hardcoded", choices=["none", "hardcoded", "model", "replay"])
207
+ run_p.add_argument("--injector-model", default="claude-sonnet-5")
208
+ run_p.add_argument(
209
+ "--replay-file",
210
+ default=None,
211
+ help="Required with --injector replay: a JSON file (from --record-injections) of "
212
+ "recorded armed injections keyed by scenario id. Lets you re-run the identical chaos "
213
+ "sequence against a different --target/--target-model with zero Injector-side API cost.",
214
+ )
215
+ run_p.add_argument(
216
+ "--record-injections",
217
+ default=None,
218
+ help="Write every armed injection decided during this run to this JSON file, keyed by "
219
+ "scenario id, for later replay via --injector replay --replay-file. Most useful with "
220
+ "--injector model: pay for the Injector's decisions once, replay them against as many "
221
+ "Target models as you want for free.",
222
+ )
223
+ run_p.add_argument(
224
+ "--kind-policy",
225
+ default="round_robin",
226
+ choices=["free", "round_robin", "forced_coverage"],
227
+ help="Only applies to --injector model. free: no restriction, matches whichever kind is "
228
+ "easiest to justify (starves TOOL_ERROR/AMBIGUITY). round_robin (default): offers only "
229
+ "the next uncovered kind, forces once budget is tight. forced_coverage: forces an "
230
+ "injection every step while any kind remains uncovered, no wait allowed.",
231
+ )
232
+ run_p.add_argument(
233
+ "--domain",
234
+ default="ticket",
235
+ choices=["ticket", "access_control"],
236
+ help="Which business domain's world/tools/scenarios to run against. ticket (default): the "
237
+ "customer-support domain this harness shipped with. access_control: employees requesting "
238
+ "access to internal systems -- a structurally different domain built on the same generic "
239
+ "chaos-injection engine, with its own Injector prompt too (see domain.py).",
240
+ )
241
+ run_p.add_argument(
242
+ "--target",
243
+ default="claude",
244
+ choices=["claude", "scripted", "rule_based"],
245
+ help="claude: real Claude tool-calling loop (--target-model selects which model). "
246
+ "scripted: dummy replay Agent, no live calls -- for exercising the harness itself. "
247
+ "rule_based: deterministic decision-tree Agent, no live calls -- a real (if arbitrary) "
248
+ "policy to chaos-test for free -- each --domain has its own (see agents/rule_based.py).",
249
+ )
250
+ run_p.add_argument("--target-model", default="claude-haiku-4-5-20251001")
251
+ run_p.add_argument(
252
+ "--hardcoded-offsets",
253
+ default="0",
254
+ help="Comma-separated offsets (from baseline end, not absolute step index) at which the hardcoded injector fires, e.g. '0,2'.",
255
+ )
256
+ run_p.add_argument(
257
+ "--hardcoded-tool",
258
+ default=None,
259
+ help="Defaults to the chosen --domain's own default_hardcoded_tool (e.g. issue_refund "
260
+ "for ticket, grant_access for access_control) -- a tool name from the wrong domain can "
261
+ "never match a real call, so the injector would silently arm every step and never fire.",
262
+ )
263
+ run_p.add_argument("--scenario", default=None, help="Run a single scenario id.")
264
+ run_p.add_argument(
265
+ "--list-scenarios",
266
+ action="store_true",
267
+ help="Print every scenario in --domain as JSON (id, scenario_class, baseline_until, "
268
+ "max_steps, required_commits_count) and exit without running anything -- for building a "
269
+ "CI test matrix or picking scenario ids programmatically.",
270
+ )
271
+ run_p.add_argument(
272
+ "--classify", action="store_true", help="Run the response classifier on every valid injection."
273
+ )
274
+ run_p.add_argument(
275
+ "--max-total-cost-usd",
276
+ type=float,
277
+ default=None,
278
+ help="Stop starting new scenarios once combined Target + Injector spend so far reaches this "
279
+ "amount -- a whole-invocation safety net, separate from each scenario's own (much smaller) "
280
+ "max_cost_usd. Only applies to --target claude and/or --injector model; checked between "
281
+ "scenarios, not mid-scenario, so a single scenario can still slightly overshoot it. Does not "
282
+ "cover --classify's cost, which runs after every scenario in this invocation has finished.",
283
+ )
284
+ run_p.add_argument("--classifier-model", default="claude-sonnet-5")
285
+ run_p.add_argument("--json-out", default=None, help="Write full trajectory JSON here (mandatory per spec section 13 in spirit -- never analyze from memory only).")
286
+ run_p.add_argument("--html-out", default=None, help="Write a self-contained HTML report here -- for handing to a non-engineer stakeholder, not just reading in a terminal. Always includes the scenario-by-scenario step detail (collapsed by default).")
287
+ run_p.add_argument(
288
+ "--detailed",
289
+ action="store_true",
290
+ help="Also print a scenario-by-scenario, step-by-step breakdown of every tool call in both "
291
+ "the clean and chaos runs -- what render() summarizes into pass/fail counts, in full. Off by "
292
+ "default since it's a lot of output; the --html-out report always includes it either way "
293
+ "(collapsed, one click per scenario).",
294
+ )
295
+ run_p.add_argument(
296
+ "--export-for-labeling",
297
+ default=None,
298
+ help="Requires --classify. Write a sample of classified injections to this JSON file for "
299
+ "hand-labeling (fill in human_classification per entry), then run "
300
+ "`agentprobe score-agreement` on it -- spec section 7's mandate: classifier agreement "
301
+ "below ~80%% means downstream results mean nothing.",
302
+ )
303
+ run_p.add_argument("--labeling-sample-size", type=int, default=30, help="Sample size for --export-for-labeling.")
304
+ run_p.add_argument(
305
+ "--use-playbook",
306
+ action="store_true",
307
+ help="Only applies to --injector model. Surface an advisory hint in the Injector's prompt "
308
+ "naming whichever trigger has historically fired most often for scenarios shaped like this "
309
+ "one, and record this run's outcomes back into the playbook for future runs. Off by "
310
+ "default -- opt in once you have runs worth learning from.",
311
+ )
312
+ run_p.add_argument("--playbook-file", default="~/.agentprobe/playbook.json", help="Path for --use-playbook.")
313
+ run_p.add_argument(
314
+ "--export-for-feedback",
315
+ default=None,
316
+ help="Requires --use-playbook. Write a summary of this run's fired injections plus one "
317
+ "blank free-text 'feedback' field for the whole run (what was good, what should change, "
318
+ "what was missing) to this JSON file, then run `agentprobe apply-feedback` on it -- feeds "
319
+ "human judgment of injection quality into the playbook, a different axis from the "
320
+ "automated fire-rate/handled tracking --use-playbook already does on its own.",
321
+ )
322
+
323
+ score_p = sub.add_parser("score-agreement", help="Compute classifier agreement from a hand-labeled --export-for-labeling file.")
324
+ score_p.add_argument("labeled_file", help="A file from --export-for-labeling with human_classification filled in.")
325
+
326
+ feedback_p = sub.add_parser("apply-feedback", help="Fold a hand-written --export-for-feedback file back into the playbook.")
327
+ feedback_p.add_argument("feedback_file", help="A file from --export-for-feedback with the run-level 'feedback' field filled in.")
328
+ feedback_p.add_argument("--playbook-file", default="~/.agentprobe/playbook.json", help="Playbook file to update.")
329
+
330
+ diff_p = sub.add_parser("diff", help="Compare two --json-out dumps and report pass/fail regressions.")
331
+ diff_p.add_argument("before", help="Path to an earlier --json-out dump.")
332
+ diff_p.add_argument("after", help="Path to a later --json-out dump to compare against it.")
333
+ diff_p.add_argument(
334
+ "--fail-on-regression",
335
+ action="store_true",
336
+ help="Exit 1 if any scenario that passed in `before` fails in `after` -- for a CI gate.",
337
+ )
338
+
339
+ args = parser.parse_args(argv)
340
+ load_dotenv()
341
+
342
+ if args.command == "diff":
343
+ from agentprobe.diff import diff_runs, render_diff
344
+
345
+ before_payload = _load_json_file(args.before)
346
+ if before_payload is None:
347
+ return 1
348
+ after_payload = _load_json_file(args.after)
349
+ if after_payload is None:
350
+ return 1
351
+ result = diff_runs(before_payload, after_payload)
352
+ print(render_diff(result))
353
+ return 1 if (args.fail_on_regression and result.has_regressions) else 0
354
+
355
+ if args.command == "score-agreement":
356
+ from agentprobe.agreement import score_agreement
357
+
358
+ labeled = _load_json_file(args.labeled_file)
359
+ if labeled is None:
360
+ return 1
361
+ try:
362
+ agreed, total = score_agreement(labeled)
363
+ except ValueError as e:
364
+ print(str(e), file=sys.stderr)
365
+ return 1
366
+ if total == 0:
367
+ print("no labeled entries found (human_classification is empty for all of them)")
368
+ return 1
369
+ pct = 100 * agreed / total
370
+ pct_color = green if pct >= 80 else red
371
+ print(f"classifier agreement with human labels: {agreed}/{total} ({pct_color(f'{pct:.0f}%')})")
372
+ if pct < 80:
373
+ print(red("below the ~80% bar spec section 7 sets -- downstream results from this classifier configuration should not be trusted yet"))
374
+ return 0
375
+
376
+ if args.command == "apply-feedback":
377
+ from agentprobe.feedback import apply_feedback
378
+ from agentprobe.playbook import Playbook
379
+
380
+ payload = _load_json_file(args.feedback_file)
381
+ if payload is None:
382
+ return 1
383
+ playbook = Playbook(path=args.playbook_file)
384
+ applied = apply_feedback(payload, playbook)
385
+ if applied == 0:
386
+ print("no feedback found (the 'feedback' field was left blank)")
387
+ return 1
388
+ playbook.save()
389
+ print(f"applied feedback to {applied} kind/trigger/shape combination{'s' if applied != 1 else ''} -> {args.playbook_file}")
390
+ return 0
391
+
392
+ if args.command != "run":
393
+ return 1
394
+
395
+ if args.injector == "replay" and not args.replay_file:
396
+ print("--injector replay requires --replay-file", file=sys.stderr)
397
+ return 1
398
+
399
+ if args.export_for_labeling and not args.classify:
400
+ print("--export-for-labeling requires --classify", file=sys.stderr)
401
+ return 1
402
+
403
+ if args.export_for_feedback and not args.use_playbook:
404
+ print("--export-for-feedback requires --use-playbook", file=sys.stderr)
405
+ return 1
406
+
407
+ try:
408
+ hardcoded_offsets = [int(x) for x in args.hardcoded_offsets.split(",")]
409
+ except ValueError:
410
+ print(
411
+ f"--hardcoded-offsets must be a comma-separated list of integers, got {args.hardcoded_offsets!r}",
412
+ file=sys.stderr,
413
+ )
414
+ return 1
415
+
416
+ replay_data = None
417
+ if args.replay_file:
418
+ replay_data = _load_json_file(args.replay_file)
419
+ if replay_data is None:
420
+ return 1
421
+
422
+ if args.domain == "access_control":
423
+ from agentprobe.domains.access_control.domain import ACCESS_CONTROL_DOMAIN
424
+ from agentprobe.domains.access_control.scenarios import SCENARIOS as ALL_DOMAIN_SCENARIOS
425
+
426
+ domain_obj = ACCESS_CONTROL_DOMAIN
427
+ else:
428
+ from agentprobe.domain import TICKET_DOMAIN
429
+
430
+ ALL_DOMAIN_SCENARIOS = ALL_SCENARIOS
431
+ domain_obj = TICKET_DOMAIN
432
+
433
+ if args.hardcoded_tool is None:
434
+ args.hardcoded_tool = domain_obj.default_hardcoded_tool
435
+
436
+ if args.list_scenarios:
437
+ print(
438
+ json.dumps(
439
+ [
440
+ {
441
+ "id": s.id,
442
+ "domain": domain_obj.name,
443
+ "scenario_class": s.scenario_class,
444
+ "baseline_until": s.baseline_until,
445
+ "max_steps": s.max_steps,
446
+ "required_commits_count": len(s.goal.required_commits),
447
+ }
448
+ for s in ALL_DOMAIN_SCENARIOS
449
+ ],
450
+ indent=2,
451
+ )
452
+ )
453
+ return 0
454
+
455
+ scenarios = ALL_DOMAIN_SCENARIOS
456
+ if args.scenario:
457
+ scenarios = [s for s in scenarios if s.id == args.scenario]
458
+ if not scenarios:
459
+ print(f"no such scenario: {args.scenario}", file=sys.stderr)
460
+ return 1
461
+
462
+ try:
463
+ target_factory = _target_factory(args.target, args.target_model, domain_name=args.domain)
464
+ except ValueError as e:
465
+ print(str(e), file=sys.stderr)
466
+ return 1
467
+
468
+ playbook = None
469
+ if args.use_playbook:
470
+ from agentprobe.playbook import Playbook
471
+
472
+ playbook = Playbook(path=args.playbook_file)
473
+
474
+ recorded_by_scenario: dict[str, list[dict]] = {}
475
+ shapes_by_scenario: dict[str, object] = {}
476
+ injector_cost_usd = 0.0
477
+ clean_trajectories = []
478
+ chaos_trajectories = []
479
+ total_scenarios = len(scenarios)
480
+ for scenario_index, s in enumerate(scenarios, start=1):
481
+ progress = f"[{scenario_index}/{total_scenarios}] " if total_scenarios > 1 else ""
482
+ with Spinner(f"{progress}running {s.id}") as spin:
483
+ current_shape = None
484
+ if playbook is not None:
485
+ from agentprobe.playbook import scenario_shape
486
+
487
+ current_shape = scenario_shape(s, domain_name=domain_obj.name)
488
+ shapes_by_scenario[s.id] = current_shape
489
+ # a fresh injector instance per scenario -- hardcoded/model injectors
490
+ # carry per-run state (e.g. "have I fired yet").
491
+ try:
492
+ injector = _make_injector(
493
+ args.injector,
494
+ args.injector_model,
495
+ hardcoded_offsets,
496
+ args.hardcoded_tool,
497
+ args.kind_policy,
498
+ scenario_id=s.id,
499
+ replay_data=replay_data,
500
+ playbook=playbook,
501
+ scenario_shape=current_shape,
502
+ domain=domain_obj,
503
+ )
504
+ except ValueError as e:
505
+ print(str(e), file=sys.stderr)
506
+ return 1
507
+ recorder = None
508
+ if args.record_injections:
509
+ from agentprobe.injector import RecordingInjector
510
+
511
+ recorder = RecordingInjector(injector)
512
+ injector = recorder
513
+ if args.mode == "robustness":
514
+ clean, chaos = run_robustness_pair(s, target_factory, injector, domain=domain_obj)
515
+ clean_trajectories.append(clean)
516
+ chaos_trajectories.append(chaos)
517
+ suffix = f"clean: {pass_fail(clean.passed)} chaos: {pass_fail(chaos.passed)}"
518
+ if chaos.discarded:
519
+ suffix += f" {yellow('[DISCARDED: invalid injection]')}"
520
+ spin.finish(ok=clean.passed and chaos.passed, suffix=suffix)
521
+ else:
522
+ chaos = run_recovery(s, target_factory, injector, domain=domain_obj)
523
+ chaos_trajectories.append(chaos)
524
+ suffix = pass_fail(chaos.passed)
525
+ if chaos.discarded:
526
+ suffix += f" {yellow('[DISCARDED: invalid injection]')}"
527
+ spin.finish(ok=chaos.passed, suffix=suffix)
528
+ if recorder is not None:
529
+ from agentprobe.injector import serialize_armed_injection
530
+
531
+ recorded_by_scenario[s.id] = [serialize_armed_injection(a) for a in recorder.recorded]
532
+ injector_cost_usd += _injector_cost_usd(injector)
533
+
534
+ if args.max_total_cost_usd is not None:
535
+ target_cost_so_far = sum(t.total_cost_usd for t in clean_trajectories) + sum(
536
+ t.total_cost_usd for t in chaos_trajectories
537
+ )
538
+ spent_so_far = target_cost_so_far + injector_cost_usd
539
+ if spent_so_far >= args.max_total_cost_usd:
540
+ remaining = total_scenarios - scenario_index
541
+ print(
542
+ f"\n--max-total-cost-usd reached (${spent_so_far:.4f} >= ${args.max_total_cost_usd:.2f}) "
543
+ f"after {s.id} -- stopping before {remaining} remaining scenario(s). "
544
+ "Report below covers only what actually ran.",
545
+ file=sys.stderr,
546
+ )
547
+ break
548
+
549
+ if args.record_injections:
550
+ with open(args.record_injections, "w") as f:
551
+ json.dump(recorded_by_scenario, f, indent=2)
552
+ print(f"recorded injections for {len(recorded_by_scenario)} scenario(s) -> {args.record_injections}", file=sys.stderr)
553
+
554
+ from agentprobe.injection import injection_was_triggered
555
+
556
+ injection_records = [
557
+ InjectionRecord(
558
+ scenario_id=t.scenario_id, applied=a, triggered=injection_was_triggered(a, t.steps)
559
+ )
560
+ for t in chaos_trajectories
561
+ for a in t.injections
562
+ ]
563
+
564
+ if args.classify:
565
+ from agentprobe.classifier import classify_response
566
+
567
+ for i, record in enumerate(injection_records):
568
+ if not record.applied.valid or not record.triggered:
569
+ continue
570
+ traj = next(t for t in chaos_trajectories if t.scenario_id == record.scenario_id)
571
+ print(f"classifying {record.scenario_id} step {record.applied.fired_at_step} ...", file=sys.stderr)
572
+ classification = classify_response(
573
+ record.applied, traj.steps, traj.final_answer, model=args.classifier_model
574
+ )
575
+ injection_records[i] = InjectionRecord(
576
+ scenario_id=record.scenario_id,
577
+ applied=record.applied,
578
+ classification=classification,
579
+ triggered=record.triggered,
580
+ )
581
+
582
+ if playbook is not None:
583
+ from agentprobe.playbook import OutcomeRecord
584
+
585
+ for record in injection_records:
586
+ shape = shapes_by_scenario.get(record.scenario_id)
587
+ if shape is None:
588
+ continue
589
+ handled = None
590
+ if record.classification is not None:
591
+ handled = record.classification.classification.value == "HANDLED"
592
+ playbook.record(
593
+ OutcomeRecord(
594
+ kind=record.applied.injection.kind.value,
595
+ trigger_kind=record.applied.trigger_kind,
596
+ scenario_shape=shape,
597
+ fired=True,
598
+ valid=record.applied.valid,
599
+ handled=handled,
600
+ )
601
+ )
602
+ for t in chaos_trajectories:
603
+ shape = shapes_by_scenario.get(t.scenario_id)
604
+ if shape is None:
605
+ continue
606
+ for e in t.expired_injections:
607
+ playbook.record(
608
+ OutcomeRecord(
609
+ kind=e.injection.kind.value,
610
+ trigger_kind=e.trigger.kind.value,
611
+ scenario_shape=shape,
612
+ fired=False,
613
+ )
614
+ )
615
+ playbook.save()
616
+ print(f"playbook updated -> {args.playbook_file} ({len(playbook.records)} total records)", file=sys.stderr)
617
+
618
+ if args.export_for_labeling:
619
+ # already validated earlier that --export-for-labeling requires --classify
620
+ from agentprobe.agreement import export_for_labeling
621
+
622
+ sample = export_for_labeling(injection_records, n=args.labeling_sample_size)
623
+ with open(args.export_for_labeling, "w") as f:
624
+ json.dump(sample, f, indent=2)
625
+ print(
626
+ f"exported {len(sample)} classification(s) for hand-labeling -> {args.export_for_labeling} "
627
+ "(fill in human_classification per entry, then run `agentprobe score-agreement`)",
628
+ file=sys.stderr,
629
+ )
630
+
631
+ if args.export_for_feedback:
632
+ # already validated earlier that --export-for-feedback requires --use-playbook
633
+ from agentprobe.feedback import export_for_feedback
634
+
635
+ summary = export_for_feedback(injection_records, shapes_by_scenario)
636
+ with open(args.export_for_feedback, "w") as f:
637
+ json.dump(summary, f, indent=2)
638
+ print(
639
+ f"exported {len(summary['summary'])} injection(s) for feedback -> {args.export_for_feedback} "
640
+ "(fill in the 'feedback' field, then run `agentprobe apply-feedback`)",
641
+ file=sys.stderr,
642
+ )
643
+
644
+ injector_labels = {
645
+ "none": "none",
646
+ "hardcoded": f"hardcoded({args.hardcoded_tool}@{args.hardcoded_offsets})",
647
+ "model": args.injector_model,
648
+ "replay": f"replay({args.replay_file})",
649
+ }
650
+ report = Report(
651
+ mode=args.mode,
652
+ injector_model=injector_labels[args.injector],
653
+ target_model=args.target_model if args.target == "claude" else args.target,
654
+ clean_trajectories=clean_trajectories,
655
+ chaos_trajectories=chaos_trajectories,
656
+ injection_records=injection_records,
657
+ injector_cost_usd=injector_cost_usd,
658
+ )
659
+ print()
660
+ print(report.render())
661
+
662
+ if args.detailed:
663
+ print()
664
+ print(report.render_scenario_detail())
665
+
666
+ if args.json_out:
667
+ payload = {
668
+ "mode": args.mode,
669
+ "clean_trajectories": [_serialize_trajectory(t) for t in clean_trajectories],
670
+ "chaos_trajectories": [_serialize_trajectory(t) for t in chaos_trajectories],
671
+ }
672
+ with open(args.json_out, "w") as f:
673
+ json.dump(payload, f, indent=2)
674
+
675
+ if args.html_out:
676
+ with open(args.html_out, "w") as f:
677
+ f.write(report.render_html())
678
+ print(f"HTML report -> {args.html_out}", file=sys.stderr)
679
+
680
+ return 0
681
+
682
+
683
+ if __name__ == "__main__":
684
+ raise SystemExit(main())