agentprobe-testing 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprobe/__init__.py +104 -0
- agentprobe/agents/__init__.py +0 -0
- agentprobe/agents/base.py +32 -0
- agentprobe/agents/rule_based.py +336 -0
- agentprobe/agents/scripted.py +30 -0
- agentprobe/agents/target_agent.py +106 -0
- agentprobe/agreement.py +80 -0
- agentprobe/classifier.py +159 -0
- agentprobe/cli.py +684 -0
- agentprobe/diff.py +150 -0
- agentprobe/domain.py +121 -0
- agentprobe/domains/__init__.py +0 -0
- agentprobe/domains/access_control/__init__.py +0 -0
- agentprobe/domains/access_control/agent.py +90 -0
- agentprobe/domains/access_control/clean.py +154 -0
- agentprobe/domains/access_control/complex_agent.py +123 -0
- agentprobe/domains/access_control/decoy.py +124 -0
- agentprobe/domains/access_control/domain.py +35 -0
- agentprobe/domains/access_control/entities.py +43 -0
- agentprobe/domains/access_control/injector_prompt.py +196 -0
- agentprobe/domains/access_control/rule_based_agent.py +263 -0
- agentprobe/domains/access_control/scenarios.py +17 -0
- agentprobe/domains/access_control/split.py +96 -0
- agentprobe/domains/access_control/tools.py +235 -0
- agentprobe/domains/access_control/trap.py +100 -0
- agentprobe/feedback.py +121 -0
- agentprobe/generic_world.py +99 -0
- agentprobe/injection.py +475 -0
- agentprobe/injector.py +810 -0
- agentprobe/llm.py +123 -0
- agentprobe/playbook.py +211 -0
- agentprobe/quickstart.py +295 -0
- agentprobe/reachability.py +196 -0
- agentprobe/registry.py +313 -0
- agentprobe/report.py +666 -0
- agentprobe/runner.py +317 -0
- agentprobe/scenario.py +75 -0
- agentprobe/scenarios/__init__.py +0 -0
- agentprobe/scenarios/clean.py +194 -0
- agentprobe/scenarios/decoy.py +272 -0
- agentprobe/scenarios/registry.py +16 -0
- agentprobe/scenarios/split.py +203 -0
- agentprobe/scenarios/trap.py +215 -0
- agentprobe/termui.py +154 -0
- agentprobe/tools.py +275 -0
- agentprobe/trajectory.py +107 -0
- agentprobe/triage.py +153 -0
- agentprobe/validate_scenarios.py +489 -0
- agentprobe/world.py +189 -0
- agentprobe_testing-0.5.0.dist-info/METADATA +127 -0
- agentprobe_testing-0.5.0.dist-info/RECORD +55 -0
- agentprobe_testing-0.5.0.dist-info/WHEEL +5 -0
- agentprobe_testing-0.5.0.dist-info/entry_points.txt +4 -0
- agentprobe_testing-0.5.0.dist-info/licenses/LICENSE +109 -0
- agentprobe_testing-0.5.0.dist-info/top_level.txt +1 -0
agentprobe/cli.py
ADDED
|
@@ -0,0 +1,684 @@
|
|
|
1
|
+
"""CLI entrypoint: python -m agentprobe.cli run --mode robustness|recovery"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from dotenv import load_dotenv
|
|
10
|
+
|
|
11
|
+
from agentprobe.report import InjectionRecord, Report
|
|
12
|
+
from agentprobe.runner import run_recovery, run_robustness_pair
|
|
13
|
+
from agentprobe.scenarios.registry import ALL_SCENARIOS
|
|
14
|
+
from agentprobe.termui import Spinner, green, pass_fail, red, yellow
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _make_injector(
|
|
18
|
+
name: str,
|
|
19
|
+
model: str,
|
|
20
|
+
hardcoded_offsets: list[int],
|
|
21
|
+
hardcoded_tool: str,
|
|
22
|
+
kind_policy: str,
|
|
23
|
+
scenario_id: str | None = None,
|
|
24
|
+
replay_data: dict | None = None,
|
|
25
|
+
playbook=None,
|
|
26
|
+
scenario_shape=None,
|
|
27
|
+
domain=None,
|
|
28
|
+
):
|
|
29
|
+
if name == "none":
|
|
30
|
+
from agentprobe.injector import NullInjector
|
|
31
|
+
|
|
32
|
+
return NullInjector()
|
|
33
|
+
if name == "hardcoded":
|
|
34
|
+
from agentprobe.injector import HardcodedToolErrorInjector
|
|
35
|
+
|
|
36
|
+
return HardcodedToolErrorInjector(offsets=hardcoded_offsets, tool_name=hardcoded_tool)
|
|
37
|
+
if name == "model":
|
|
38
|
+
from agentprobe.domain import TICKET_DOMAIN
|
|
39
|
+
|
|
40
|
+
domain = domain or TICKET_DOMAIN
|
|
41
|
+
if domain.injector_system_prompt is None:
|
|
42
|
+
# This Domain hasn't been given its own Injector system prompt
|
|
43
|
+
# (see domain.py's injector_system_prompt field) -- pointing
|
|
44
|
+
# ModelInjector at it anyway would send a prompt written for a
|
|
45
|
+
# different domain's vocabulary. hardcoded/none work fine under
|
|
46
|
+
# any domain in the meantime (see _make_injector's other
|
|
47
|
+
# branches -- neither references domain vocabulary at all).
|
|
48
|
+
raise ValueError(
|
|
49
|
+
"--injector model isn't wired up for this --domain yet (no injector_system_prompt "
|
|
50
|
+
"set on its Domain). Use --injector hardcoded or none instead."
|
|
51
|
+
)
|
|
52
|
+
from agentprobe.injector import ModelInjector
|
|
53
|
+
|
|
54
|
+
return ModelInjector(
|
|
55
|
+
model=model,
|
|
56
|
+
policy=kind_policy,
|
|
57
|
+
playbook=playbook,
|
|
58
|
+
scenario_shape=scenario_shape,
|
|
59
|
+
system_prompt=domain.injector_system_prompt,
|
|
60
|
+
all_tools=domain.all_tools,
|
|
61
|
+
commit_tools=domain.commit_tools,
|
|
62
|
+
entity_types=domain.injector_entity_types,
|
|
63
|
+
)
|
|
64
|
+
if name == "replay":
|
|
65
|
+
from agentprobe.injector import NullInjector, ReplayInjector, deserialize_armed_injection
|
|
66
|
+
|
|
67
|
+
recorded = (replay_data or {}).get(scenario_id)
|
|
68
|
+
if recorded is None:
|
|
69
|
+
print(
|
|
70
|
+
f"no recorded injections for scenario {scenario_id!r} in replay file -- using NullInjector",
|
|
71
|
+
file=sys.stderr,
|
|
72
|
+
)
|
|
73
|
+
return NullInjector()
|
|
74
|
+
return ReplayInjector([deserialize_armed_injection(d) for d in recorded])
|
|
75
|
+
raise ValueError(f"unknown injector {name!r}")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _load_json_file(path: str) -> dict | None:
|
|
79
|
+
"""Loads a JSON file for a CLI flag/argument, printing a clean
|
|
80
|
+
one-line message and returning None on a bad path or malformed JSON
|
|
81
|
+
instead of letting FileNotFoundError/JSONDecodeError surface as a raw
|
|
82
|
+
traceback -- both are ordinary user typos (wrong path, half-written
|
|
83
|
+
file), not harness bugs."""
|
|
84
|
+
try:
|
|
85
|
+
with open(path) as f:
|
|
86
|
+
return json.load(f)
|
|
87
|
+
except FileNotFoundError:
|
|
88
|
+
print(f"no such file: {path}", file=sys.stderr)
|
|
89
|
+
return None
|
|
90
|
+
except json.JSONDecodeError as e:
|
|
91
|
+
print(f"{path} is not valid JSON: {e}", file=sys.stderr)
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _injector_cost_usd(injector) -> float:
|
|
96
|
+
"""ModelInjector is the only injector type with API cost (none/
|
|
97
|
+
hardcoded/replay never call a model). Unwraps RecordingInjector's
|
|
98
|
+
wrapping too, since --record-injections swaps in a wrapper that
|
|
99
|
+
doesn't have its own total_cost_usd."""
|
|
100
|
+
inner = getattr(injector, "_inner", injector)
|
|
101
|
+
return getattr(inner, "total_cost_usd", 0.0)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _target_factory(target: str, model: str, domain_name: str = "ticket"):
|
|
105
|
+
"""Target is an Agent (agentprobe/agents/base.py) -- the runner only
|
|
106
|
+
ever calls start()/next_action()/observe(), so any implementation is
|
|
107
|
+
swappable here. "claude" wraps the real Claude tool-calling loop for
|
|
108
|
+
whichever domain is selected; "scripted" is the dummy replay agent
|
|
109
|
+
used to exercise the world/tools/runner without live API calls (see
|
|
110
|
+
agents/scripted.py) -- it always immediately gives a final answer with
|
|
111
|
+
no script, since a useful script is scenario-specific and has no
|
|
112
|
+
general CLI form, and it's domain-agnostic (no tool-name knowledge
|
|
113
|
+
baked in), so it works under any --domain. "rule_based" is a real (if
|
|
114
|
+
arbitrary) deterministic decision tree -- zero API cost like scripted,
|
|
115
|
+
but with actual branching/retry/recovery logic for the harness to
|
|
116
|
+
chaos-test against -- each domain has its own (see agents/
|
|
117
|
+
rule_based.py for ticket, domains/access_control/rule_based_agent.py
|
|
118
|
+
for access_control).
|
|
119
|
+
"""
|
|
120
|
+
if target == "scripted":
|
|
121
|
+
from agentprobe.agents.scripted import ScriptedAgent
|
|
122
|
+
|
|
123
|
+
return lambda: ScriptedAgent(script=[])
|
|
124
|
+
if domain_name == "access_control":
|
|
125
|
+
if target == "claude":
|
|
126
|
+
from agentprobe.domains.access_control.agent import AccessControlAgent
|
|
127
|
+
|
|
128
|
+
return lambda: AccessControlAgent(model=model)
|
|
129
|
+
if target == "rule_based":
|
|
130
|
+
from agentprobe.domains.access_control.rule_based_agent import AccessControlRuleBasedAgent
|
|
131
|
+
|
|
132
|
+
return lambda: AccessControlRuleBasedAgent()
|
|
133
|
+
raise ValueError(f"target {target!r} is not available under --domain access_control (try 'claude', 'scripted', or 'rule_based')")
|
|
134
|
+
if target == "claude":
|
|
135
|
+
from agentprobe.agents.target_agent import TargetAgent
|
|
136
|
+
|
|
137
|
+
return lambda: TargetAgent(model=model)
|
|
138
|
+
if target == "rule_based":
|
|
139
|
+
from agentprobe.agents.rule_based import RuleBasedAgent
|
|
140
|
+
|
|
141
|
+
return lambda: RuleBasedAgent()
|
|
142
|
+
raise ValueError(f"unknown target {target!r}")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _serialize_trajectory(t) -> dict:
|
|
146
|
+
return {
|
|
147
|
+
"scenario_id": t.scenario_id,
|
|
148
|
+
"variant": t.variant,
|
|
149
|
+
"passed": t.passed,
|
|
150
|
+
"final_status": t.final_status.value,
|
|
151
|
+
"breaking_step": t.breaking_step,
|
|
152
|
+
"wasted_steps": t.wasted_steps,
|
|
153
|
+
"halted_reason": t.halted_reason,
|
|
154
|
+
"final_answer": t.final_answer,
|
|
155
|
+
"discarded": t.discarded,
|
|
156
|
+
"injector_errors": t.injector_errors,
|
|
157
|
+
"steps": [
|
|
158
|
+
{
|
|
159
|
+
"index": s.index,
|
|
160
|
+
"tool_name": s.tool_name,
|
|
161
|
+
"tool_args": s.tool_args,
|
|
162
|
+
"ok": s.ok,
|
|
163
|
+
"result": s.result,
|
|
164
|
+
"reachability": s.reachability.status.value,
|
|
165
|
+
"reason": s.reachability.reason,
|
|
166
|
+
"injected_error": s.injected_error,
|
|
167
|
+
}
|
|
168
|
+
for s in t.steps
|
|
169
|
+
],
|
|
170
|
+
"injections": [
|
|
171
|
+
{
|
|
172
|
+
"armed_at_step": a.armed_at_step,
|
|
173
|
+
"fired_at_step": a.fired_at_step,
|
|
174
|
+
"kind": a.injection.kind.value,
|
|
175
|
+
"payload": a.injection.payload,
|
|
176
|
+
"intent": a.injection.intent,
|
|
177
|
+
"expected_signature": a.injection.expected_signature.description,
|
|
178
|
+
"rationale": a.injection.rationale,
|
|
179
|
+
"effect": a.effect,
|
|
180
|
+
"valid": a.valid,
|
|
181
|
+
}
|
|
182
|
+
for a in t.injections
|
|
183
|
+
],
|
|
184
|
+
"expired_injections": [
|
|
185
|
+
{
|
|
186
|
+
"armed_at_step": e.armed_at_step,
|
|
187
|
+
"kind": e.injection.kind.value,
|
|
188
|
+
"payload": e.injection.payload,
|
|
189
|
+
"trigger": e.trigger.describe(),
|
|
190
|
+
"rationale": e.injection.rationale,
|
|
191
|
+
}
|
|
192
|
+
for e in t.expired_injections
|
|
193
|
+
],
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def main(argv: list[str] | None = None) -> int:
|
|
198
|
+
from agentprobe import __version__
|
|
199
|
+
|
|
200
|
+
parser = argparse.ArgumentParser(prog="agentprobe")
|
|
201
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
202
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
203
|
+
|
|
204
|
+
run_p = sub.add_parser("run", help="Run scenarios in robustness or recovery mode.")
|
|
205
|
+
run_p.add_argument("--mode", default="robustness", choices=["robustness", "recovery"])
|
|
206
|
+
run_p.add_argument("--injector", default="hardcoded", choices=["none", "hardcoded", "model", "replay"])
|
|
207
|
+
run_p.add_argument("--injector-model", default="claude-sonnet-5")
|
|
208
|
+
run_p.add_argument(
|
|
209
|
+
"--replay-file",
|
|
210
|
+
default=None,
|
|
211
|
+
help="Required with --injector replay: a JSON file (from --record-injections) of "
|
|
212
|
+
"recorded armed injections keyed by scenario id. Lets you re-run the identical chaos "
|
|
213
|
+
"sequence against a different --target/--target-model with zero Injector-side API cost.",
|
|
214
|
+
)
|
|
215
|
+
run_p.add_argument(
|
|
216
|
+
"--record-injections",
|
|
217
|
+
default=None,
|
|
218
|
+
help="Write every armed injection decided during this run to this JSON file, keyed by "
|
|
219
|
+
"scenario id, for later replay via --injector replay --replay-file. Most useful with "
|
|
220
|
+
"--injector model: pay for the Injector's decisions once, replay them against as many "
|
|
221
|
+
"Target models as you want for free.",
|
|
222
|
+
)
|
|
223
|
+
run_p.add_argument(
|
|
224
|
+
"--kind-policy",
|
|
225
|
+
default="round_robin",
|
|
226
|
+
choices=["free", "round_robin", "forced_coverage"],
|
|
227
|
+
help="Only applies to --injector model. free: no restriction, matches whichever kind is "
|
|
228
|
+
"easiest to justify (starves TOOL_ERROR/AMBIGUITY). round_robin (default): offers only "
|
|
229
|
+
"the next uncovered kind, forces once budget is tight. forced_coverage: forces an "
|
|
230
|
+
"injection every step while any kind remains uncovered, no wait allowed.",
|
|
231
|
+
)
|
|
232
|
+
run_p.add_argument(
|
|
233
|
+
"--domain",
|
|
234
|
+
default="ticket",
|
|
235
|
+
choices=["ticket", "access_control"],
|
|
236
|
+
help="Which business domain's world/tools/scenarios to run against. ticket (default): the "
|
|
237
|
+
"customer-support domain this harness shipped with. access_control: employees requesting "
|
|
238
|
+
"access to internal systems -- a structurally different domain built on the same generic "
|
|
239
|
+
"chaos-injection engine, with its own Injector prompt too (see domain.py).",
|
|
240
|
+
)
|
|
241
|
+
run_p.add_argument(
|
|
242
|
+
"--target",
|
|
243
|
+
default="claude",
|
|
244
|
+
choices=["claude", "scripted", "rule_based"],
|
|
245
|
+
help="claude: real Claude tool-calling loop (--target-model selects which model). "
|
|
246
|
+
"scripted: dummy replay Agent, no live calls -- for exercising the harness itself. "
|
|
247
|
+
"rule_based: deterministic decision-tree Agent, no live calls -- a real (if arbitrary) "
|
|
248
|
+
"policy to chaos-test for free -- each --domain has its own (see agents/rule_based.py).",
|
|
249
|
+
)
|
|
250
|
+
run_p.add_argument("--target-model", default="claude-haiku-4-5-20251001")
|
|
251
|
+
run_p.add_argument(
|
|
252
|
+
"--hardcoded-offsets",
|
|
253
|
+
default="0",
|
|
254
|
+
help="Comma-separated offsets (from baseline end, not absolute step index) at which the hardcoded injector fires, e.g. '0,2'.",
|
|
255
|
+
)
|
|
256
|
+
run_p.add_argument(
|
|
257
|
+
"--hardcoded-tool",
|
|
258
|
+
default=None,
|
|
259
|
+
help="Defaults to the chosen --domain's own default_hardcoded_tool (e.g. issue_refund "
|
|
260
|
+
"for ticket, grant_access for access_control) -- a tool name from the wrong domain can "
|
|
261
|
+
"never match a real call, so the injector would silently arm every step and never fire.",
|
|
262
|
+
)
|
|
263
|
+
run_p.add_argument("--scenario", default=None, help="Run a single scenario id.")
|
|
264
|
+
run_p.add_argument(
|
|
265
|
+
"--list-scenarios",
|
|
266
|
+
action="store_true",
|
|
267
|
+
help="Print every scenario in --domain as JSON (id, scenario_class, baseline_until, "
|
|
268
|
+
"max_steps, required_commits_count) and exit without running anything -- for building a "
|
|
269
|
+
"CI test matrix or picking scenario ids programmatically.",
|
|
270
|
+
)
|
|
271
|
+
run_p.add_argument(
|
|
272
|
+
"--classify", action="store_true", help="Run the response classifier on every valid injection."
|
|
273
|
+
)
|
|
274
|
+
run_p.add_argument(
|
|
275
|
+
"--max-total-cost-usd",
|
|
276
|
+
type=float,
|
|
277
|
+
default=None,
|
|
278
|
+
help="Stop starting new scenarios once combined Target + Injector spend so far reaches this "
|
|
279
|
+
"amount -- a whole-invocation safety net, separate from each scenario's own (much smaller) "
|
|
280
|
+
"max_cost_usd. Only applies to --target claude and/or --injector model; checked between "
|
|
281
|
+
"scenarios, not mid-scenario, so a single scenario can still slightly overshoot it. Does not "
|
|
282
|
+
"cover --classify's cost, which runs after every scenario in this invocation has finished.",
|
|
283
|
+
)
|
|
284
|
+
run_p.add_argument("--classifier-model", default="claude-sonnet-5")
|
|
285
|
+
run_p.add_argument("--json-out", default=None, help="Write full trajectory JSON here (mandatory per spec section 13 in spirit -- never analyze from memory only).")
|
|
286
|
+
run_p.add_argument("--html-out", default=None, help="Write a self-contained HTML report here -- for handing to a non-engineer stakeholder, not just reading in a terminal. Always includes the scenario-by-scenario step detail (collapsed by default).")
|
|
287
|
+
run_p.add_argument(
|
|
288
|
+
"--detailed",
|
|
289
|
+
action="store_true",
|
|
290
|
+
help="Also print a scenario-by-scenario, step-by-step breakdown of every tool call in both "
|
|
291
|
+
"the clean and chaos runs -- what render() summarizes into pass/fail counts, in full. Off by "
|
|
292
|
+
"default since it's a lot of output; the --html-out report always includes it either way "
|
|
293
|
+
"(collapsed, one click per scenario).",
|
|
294
|
+
)
|
|
295
|
+
run_p.add_argument(
|
|
296
|
+
"--export-for-labeling",
|
|
297
|
+
default=None,
|
|
298
|
+
help="Requires --classify. Write a sample of classified injections to this JSON file for "
|
|
299
|
+
"hand-labeling (fill in human_classification per entry), then run "
|
|
300
|
+
"`agentprobe score-agreement` on it -- spec section 7's mandate: classifier agreement "
|
|
301
|
+
"below ~80%% means downstream results mean nothing.",
|
|
302
|
+
)
|
|
303
|
+
run_p.add_argument("--labeling-sample-size", type=int, default=30, help="Sample size for --export-for-labeling.")
|
|
304
|
+
run_p.add_argument(
|
|
305
|
+
"--use-playbook",
|
|
306
|
+
action="store_true",
|
|
307
|
+
help="Only applies to --injector model. Surface an advisory hint in the Injector's prompt "
|
|
308
|
+
"naming whichever trigger has historically fired most often for scenarios shaped like this "
|
|
309
|
+
"one, and record this run's outcomes back into the playbook for future runs. Off by "
|
|
310
|
+
"default -- opt in once you have runs worth learning from.",
|
|
311
|
+
)
|
|
312
|
+
run_p.add_argument("--playbook-file", default="~/.agentprobe/playbook.json", help="Path for --use-playbook.")
|
|
313
|
+
run_p.add_argument(
|
|
314
|
+
"--export-for-feedback",
|
|
315
|
+
default=None,
|
|
316
|
+
help="Requires --use-playbook. Write a summary of this run's fired injections plus one "
|
|
317
|
+
"blank free-text 'feedback' field for the whole run (what was good, what should change, "
|
|
318
|
+
"what was missing) to this JSON file, then run `agentprobe apply-feedback` on it -- feeds "
|
|
319
|
+
"human judgment of injection quality into the playbook, a different axis from the "
|
|
320
|
+
"automated fire-rate/handled tracking --use-playbook already does on its own.",
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
score_p = sub.add_parser("score-agreement", help="Compute classifier agreement from a hand-labeled --export-for-labeling file.")
|
|
324
|
+
score_p.add_argument("labeled_file", help="A file from --export-for-labeling with human_classification filled in.")
|
|
325
|
+
|
|
326
|
+
feedback_p = sub.add_parser("apply-feedback", help="Fold a hand-written --export-for-feedback file back into the playbook.")
|
|
327
|
+
feedback_p.add_argument("feedback_file", help="A file from --export-for-feedback with the run-level 'feedback' field filled in.")
|
|
328
|
+
feedback_p.add_argument("--playbook-file", default="~/.agentprobe/playbook.json", help="Playbook file to update.")
|
|
329
|
+
|
|
330
|
+
diff_p = sub.add_parser("diff", help="Compare two --json-out dumps and report pass/fail regressions.")
|
|
331
|
+
diff_p.add_argument("before", help="Path to an earlier --json-out dump.")
|
|
332
|
+
diff_p.add_argument("after", help="Path to a later --json-out dump to compare against it.")
|
|
333
|
+
diff_p.add_argument(
|
|
334
|
+
"--fail-on-regression",
|
|
335
|
+
action="store_true",
|
|
336
|
+
help="Exit 1 if any scenario that passed in `before` fails in `after` -- for a CI gate.",
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
args = parser.parse_args(argv)
|
|
340
|
+
load_dotenv()
|
|
341
|
+
|
|
342
|
+
if args.command == "diff":
|
|
343
|
+
from agentprobe.diff import diff_runs, render_diff
|
|
344
|
+
|
|
345
|
+
before_payload = _load_json_file(args.before)
|
|
346
|
+
if before_payload is None:
|
|
347
|
+
return 1
|
|
348
|
+
after_payload = _load_json_file(args.after)
|
|
349
|
+
if after_payload is None:
|
|
350
|
+
return 1
|
|
351
|
+
result = diff_runs(before_payload, after_payload)
|
|
352
|
+
print(render_diff(result))
|
|
353
|
+
return 1 if (args.fail_on_regression and result.has_regressions) else 0
|
|
354
|
+
|
|
355
|
+
if args.command == "score-agreement":
|
|
356
|
+
from agentprobe.agreement import score_agreement
|
|
357
|
+
|
|
358
|
+
labeled = _load_json_file(args.labeled_file)
|
|
359
|
+
if labeled is None:
|
|
360
|
+
return 1
|
|
361
|
+
try:
|
|
362
|
+
agreed, total = score_agreement(labeled)
|
|
363
|
+
except ValueError as e:
|
|
364
|
+
print(str(e), file=sys.stderr)
|
|
365
|
+
return 1
|
|
366
|
+
if total == 0:
|
|
367
|
+
print("no labeled entries found (human_classification is empty for all of them)")
|
|
368
|
+
return 1
|
|
369
|
+
pct = 100 * agreed / total
|
|
370
|
+
pct_color = green if pct >= 80 else red
|
|
371
|
+
print(f"classifier agreement with human labels: {agreed}/{total} ({pct_color(f'{pct:.0f}%')})")
|
|
372
|
+
if pct < 80:
|
|
373
|
+
print(red("below the ~80% bar spec section 7 sets -- downstream results from this classifier configuration should not be trusted yet"))
|
|
374
|
+
return 0
|
|
375
|
+
|
|
376
|
+
if args.command == "apply-feedback":
|
|
377
|
+
from agentprobe.feedback import apply_feedback
|
|
378
|
+
from agentprobe.playbook import Playbook
|
|
379
|
+
|
|
380
|
+
payload = _load_json_file(args.feedback_file)
|
|
381
|
+
if payload is None:
|
|
382
|
+
return 1
|
|
383
|
+
playbook = Playbook(path=args.playbook_file)
|
|
384
|
+
applied = apply_feedback(payload, playbook)
|
|
385
|
+
if applied == 0:
|
|
386
|
+
print("no feedback found (the 'feedback' field was left blank)")
|
|
387
|
+
return 1
|
|
388
|
+
playbook.save()
|
|
389
|
+
print(f"applied feedback to {applied} kind/trigger/shape combination{'s' if applied != 1 else ''} -> {args.playbook_file}")
|
|
390
|
+
return 0
|
|
391
|
+
|
|
392
|
+
if args.command != "run":
|
|
393
|
+
return 1
|
|
394
|
+
|
|
395
|
+
if args.injector == "replay" and not args.replay_file:
|
|
396
|
+
print("--injector replay requires --replay-file", file=sys.stderr)
|
|
397
|
+
return 1
|
|
398
|
+
|
|
399
|
+
if args.export_for_labeling and not args.classify:
|
|
400
|
+
print("--export-for-labeling requires --classify", file=sys.stderr)
|
|
401
|
+
return 1
|
|
402
|
+
|
|
403
|
+
if args.export_for_feedback and not args.use_playbook:
|
|
404
|
+
print("--export-for-feedback requires --use-playbook", file=sys.stderr)
|
|
405
|
+
return 1
|
|
406
|
+
|
|
407
|
+
try:
|
|
408
|
+
hardcoded_offsets = [int(x) for x in args.hardcoded_offsets.split(",")]
|
|
409
|
+
except ValueError:
|
|
410
|
+
print(
|
|
411
|
+
f"--hardcoded-offsets must be a comma-separated list of integers, got {args.hardcoded_offsets!r}",
|
|
412
|
+
file=sys.stderr,
|
|
413
|
+
)
|
|
414
|
+
return 1
|
|
415
|
+
|
|
416
|
+
replay_data = None
|
|
417
|
+
if args.replay_file:
|
|
418
|
+
replay_data = _load_json_file(args.replay_file)
|
|
419
|
+
if replay_data is None:
|
|
420
|
+
return 1
|
|
421
|
+
|
|
422
|
+
if args.domain == "access_control":
|
|
423
|
+
from agentprobe.domains.access_control.domain import ACCESS_CONTROL_DOMAIN
|
|
424
|
+
from agentprobe.domains.access_control.scenarios import SCENARIOS as ALL_DOMAIN_SCENARIOS
|
|
425
|
+
|
|
426
|
+
domain_obj = ACCESS_CONTROL_DOMAIN
|
|
427
|
+
else:
|
|
428
|
+
from agentprobe.domain import TICKET_DOMAIN
|
|
429
|
+
|
|
430
|
+
ALL_DOMAIN_SCENARIOS = ALL_SCENARIOS
|
|
431
|
+
domain_obj = TICKET_DOMAIN
|
|
432
|
+
|
|
433
|
+
if args.hardcoded_tool is None:
|
|
434
|
+
args.hardcoded_tool = domain_obj.default_hardcoded_tool
|
|
435
|
+
|
|
436
|
+
if args.list_scenarios:
|
|
437
|
+
print(
|
|
438
|
+
json.dumps(
|
|
439
|
+
[
|
|
440
|
+
{
|
|
441
|
+
"id": s.id,
|
|
442
|
+
"domain": domain_obj.name,
|
|
443
|
+
"scenario_class": s.scenario_class,
|
|
444
|
+
"baseline_until": s.baseline_until,
|
|
445
|
+
"max_steps": s.max_steps,
|
|
446
|
+
"required_commits_count": len(s.goal.required_commits),
|
|
447
|
+
}
|
|
448
|
+
for s in ALL_DOMAIN_SCENARIOS
|
|
449
|
+
],
|
|
450
|
+
indent=2,
|
|
451
|
+
)
|
|
452
|
+
)
|
|
453
|
+
return 0
|
|
454
|
+
|
|
455
|
+
scenarios = ALL_DOMAIN_SCENARIOS
|
|
456
|
+
if args.scenario:
|
|
457
|
+
scenarios = [s for s in scenarios if s.id == args.scenario]
|
|
458
|
+
if not scenarios:
|
|
459
|
+
print(f"no such scenario: {args.scenario}", file=sys.stderr)
|
|
460
|
+
return 1
|
|
461
|
+
|
|
462
|
+
try:
|
|
463
|
+
target_factory = _target_factory(args.target, args.target_model, domain_name=args.domain)
|
|
464
|
+
except ValueError as e:
|
|
465
|
+
print(str(e), file=sys.stderr)
|
|
466
|
+
return 1
|
|
467
|
+
|
|
468
|
+
playbook = None
|
|
469
|
+
if args.use_playbook:
|
|
470
|
+
from agentprobe.playbook import Playbook
|
|
471
|
+
|
|
472
|
+
playbook = Playbook(path=args.playbook_file)
|
|
473
|
+
|
|
474
|
+
recorded_by_scenario: dict[str, list[dict]] = {}
|
|
475
|
+
shapes_by_scenario: dict[str, object] = {}
|
|
476
|
+
injector_cost_usd = 0.0
|
|
477
|
+
clean_trajectories = []
|
|
478
|
+
chaos_trajectories = []
|
|
479
|
+
total_scenarios = len(scenarios)
|
|
480
|
+
for scenario_index, s in enumerate(scenarios, start=1):
|
|
481
|
+
progress = f"[{scenario_index}/{total_scenarios}] " if total_scenarios > 1 else ""
|
|
482
|
+
with Spinner(f"{progress}running {s.id}") as spin:
|
|
483
|
+
current_shape = None
|
|
484
|
+
if playbook is not None:
|
|
485
|
+
from agentprobe.playbook import scenario_shape
|
|
486
|
+
|
|
487
|
+
current_shape = scenario_shape(s, domain_name=domain_obj.name)
|
|
488
|
+
shapes_by_scenario[s.id] = current_shape
|
|
489
|
+
# a fresh injector instance per scenario -- hardcoded/model injectors
|
|
490
|
+
# carry per-run state (e.g. "have I fired yet").
|
|
491
|
+
try:
|
|
492
|
+
injector = _make_injector(
|
|
493
|
+
args.injector,
|
|
494
|
+
args.injector_model,
|
|
495
|
+
hardcoded_offsets,
|
|
496
|
+
args.hardcoded_tool,
|
|
497
|
+
args.kind_policy,
|
|
498
|
+
scenario_id=s.id,
|
|
499
|
+
replay_data=replay_data,
|
|
500
|
+
playbook=playbook,
|
|
501
|
+
scenario_shape=current_shape,
|
|
502
|
+
domain=domain_obj,
|
|
503
|
+
)
|
|
504
|
+
except ValueError as e:
|
|
505
|
+
print(str(e), file=sys.stderr)
|
|
506
|
+
return 1
|
|
507
|
+
recorder = None
|
|
508
|
+
if args.record_injections:
|
|
509
|
+
from agentprobe.injector import RecordingInjector
|
|
510
|
+
|
|
511
|
+
recorder = RecordingInjector(injector)
|
|
512
|
+
injector = recorder
|
|
513
|
+
if args.mode == "robustness":
|
|
514
|
+
clean, chaos = run_robustness_pair(s, target_factory, injector, domain=domain_obj)
|
|
515
|
+
clean_trajectories.append(clean)
|
|
516
|
+
chaos_trajectories.append(chaos)
|
|
517
|
+
suffix = f"clean: {pass_fail(clean.passed)} chaos: {pass_fail(chaos.passed)}"
|
|
518
|
+
if chaos.discarded:
|
|
519
|
+
suffix += f" {yellow('[DISCARDED: invalid injection]')}"
|
|
520
|
+
spin.finish(ok=clean.passed and chaos.passed, suffix=suffix)
|
|
521
|
+
else:
|
|
522
|
+
chaos = run_recovery(s, target_factory, injector, domain=domain_obj)
|
|
523
|
+
chaos_trajectories.append(chaos)
|
|
524
|
+
suffix = pass_fail(chaos.passed)
|
|
525
|
+
if chaos.discarded:
|
|
526
|
+
suffix += f" {yellow('[DISCARDED: invalid injection]')}"
|
|
527
|
+
spin.finish(ok=chaos.passed, suffix=suffix)
|
|
528
|
+
if recorder is not None:
|
|
529
|
+
from agentprobe.injector import serialize_armed_injection
|
|
530
|
+
|
|
531
|
+
recorded_by_scenario[s.id] = [serialize_armed_injection(a) for a in recorder.recorded]
|
|
532
|
+
injector_cost_usd += _injector_cost_usd(injector)
|
|
533
|
+
|
|
534
|
+
if args.max_total_cost_usd is not None:
|
|
535
|
+
target_cost_so_far = sum(t.total_cost_usd for t in clean_trajectories) + sum(
|
|
536
|
+
t.total_cost_usd for t in chaos_trajectories
|
|
537
|
+
)
|
|
538
|
+
spent_so_far = target_cost_so_far + injector_cost_usd
|
|
539
|
+
if spent_so_far >= args.max_total_cost_usd:
|
|
540
|
+
remaining = total_scenarios - scenario_index
|
|
541
|
+
print(
|
|
542
|
+
f"\n--max-total-cost-usd reached (${spent_so_far:.4f} >= ${args.max_total_cost_usd:.2f}) "
|
|
543
|
+
f"after {s.id} -- stopping before {remaining} remaining scenario(s). "
|
|
544
|
+
"Report below covers only what actually ran.",
|
|
545
|
+
file=sys.stderr,
|
|
546
|
+
)
|
|
547
|
+
break
|
|
548
|
+
|
|
549
|
+
if args.record_injections:
|
|
550
|
+
with open(args.record_injections, "w") as f:
|
|
551
|
+
json.dump(recorded_by_scenario, f, indent=2)
|
|
552
|
+
print(f"recorded injections for {len(recorded_by_scenario)} scenario(s) -> {args.record_injections}", file=sys.stderr)
|
|
553
|
+
|
|
554
|
+
from agentprobe.injection import injection_was_triggered
|
|
555
|
+
|
|
556
|
+
injection_records = [
|
|
557
|
+
InjectionRecord(
|
|
558
|
+
scenario_id=t.scenario_id, applied=a, triggered=injection_was_triggered(a, t.steps)
|
|
559
|
+
)
|
|
560
|
+
for t in chaos_trajectories
|
|
561
|
+
for a in t.injections
|
|
562
|
+
]
|
|
563
|
+
|
|
564
|
+
if args.classify:
|
|
565
|
+
from agentprobe.classifier import classify_response
|
|
566
|
+
|
|
567
|
+
for i, record in enumerate(injection_records):
|
|
568
|
+
if not record.applied.valid or not record.triggered:
|
|
569
|
+
continue
|
|
570
|
+
traj = next(t for t in chaos_trajectories if t.scenario_id == record.scenario_id)
|
|
571
|
+
print(f"classifying {record.scenario_id} step {record.applied.fired_at_step} ...", file=sys.stderr)
|
|
572
|
+
classification = classify_response(
|
|
573
|
+
record.applied, traj.steps, traj.final_answer, model=args.classifier_model
|
|
574
|
+
)
|
|
575
|
+
injection_records[i] = InjectionRecord(
|
|
576
|
+
scenario_id=record.scenario_id,
|
|
577
|
+
applied=record.applied,
|
|
578
|
+
classification=classification,
|
|
579
|
+
triggered=record.triggered,
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
if playbook is not None:
|
|
583
|
+
from agentprobe.playbook import OutcomeRecord
|
|
584
|
+
|
|
585
|
+
for record in injection_records:
|
|
586
|
+
shape = shapes_by_scenario.get(record.scenario_id)
|
|
587
|
+
if shape is None:
|
|
588
|
+
continue
|
|
589
|
+
handled = None
|
|
590
|
+
if record.classification is not None:
|
|
591
|
+
handled = record.classification.classification.value == "HANDLED"
|
|
592
|
+
playbook.record(
|
|
593
|
+
OutcomeRecord(
|
|
594
|
+
kind=record.applied.injection.kind.value,
|
|
595
|
+
trigger_kind=record.applied.trigger_kind,
|
|
596
|
+
scenario_shape=shape,
|
|
597
|
+
fired=True,
|
|
598
|
+
valid=record.applied.valid,
|
|
599
|
+
handled=handled,
|
|
600
|
+
)
|
|
601
|
+
)
|
|
602
|
+
for t in chaos_trajectories:
|
|
603
|
+
shape = shapes_by_scenario.get(t.scenario_id)
|
|
604
|
+
if shape is None:
|
|
605
|
+
continue
|
|
606
|
+
for e in t.expired_injections:
|
|
607
|
+
playbook.record(
|
|
608
|
+
OutcomeRecord(
|
|
609
|
+
kind=e.injection.kind.value,
|
|
610
|
+
trigger_kind=e.trigger.kind.value,
|
|
611
|
+
scenario_shape=shape,
|
|
612
|
+
fired=False,
|
|
613
|
+
)
|
|
614
|
+
)
|
|
615
|
+
playbook.save()
|
|
616
|
+
print(f"playbook updated -> {args.playbook_file} ({len(playbook.records)} total records)", file=sys.stderr)
|
|
617
|
+
|
|
618
|
+
if args.export_for_labeling:
|
|
619
|
+
# already validated earlier that --export-for-labeling requires --classify
|
|
620
|
+
from agentprobe.agreement import export_for_labeling
|
|
621
|
+
|
|
622
|
+
sample = export_for_labeling(injection_records, n=args.labeling_sample_size)
|
|
623
|
+
with open(args.export_for_labeling, "w") as f:
|
|
624
|
+
json.dump(sample, f, indent=2)
|
|
625
|
+
print(
|
|
626
|
+
f"exported {len(sample)} classification(s) for hand-labeling -> {args.export_for_labeling} "
|
|
627
|
+
"(fill in human_classification per entry, then run `agentprobe score-agreement`)",
|
|
628
|
+
file=sys.stderr,
|
|
629
|
+
)
|
|
630
|
+
|
|
631
|
+
if args.export_for_feedback:
|
|
632
|
+
# already validated earlier that --export-for-feedback requires --use-playbook
|
|
633
|
+
from agentprobe.feedback import export_for_feedback
|
|
634
|
+
|
|
635
|
+
summary = export_for_feedback(injection_records, shapes_by_scenario)
|
|
636
|
+
with open(args.export_for_feedback, "w") as f:
|
|
637
|
+
json.dump(summary, f, indent=2)
|
|
638
|
+
print(
|
|
639
|
+
f"exported {len(summary['summary'])} injection(s) for feedback -> {args.export_for_feedback} "
|
|
640
|
+
"(fill in the 'feedback' field, then run `agentprobe apply-feedback`)",
|
|
641
|
+
file=sys.stderr,
|
|
642
|
+
)
|
|
643
|
+
|
|
644
|
+
injector_labels = {
|
|
645
|
+
"none": "none",
|
|
646
|
+
"hardcoded": f"hardcoded({args.hardcoded_tool}@{args.hardcoded_offsets})",
|
|
647
|
+
"model": args.injector_model,
|
|
648
|
+
"replay": f"replay({args.replay_file})",
|
|
649
|
+
}
|
|
650
|
+
report = Report(
|
|
651
|
+
mode=args.mode,
|
|
652
|
+
injector_model=injector_labels[args.injector],
|
|
653
|
+
target_model=args.target_model if args.target == "claude" else args.target,
|
|
654
|
+
clean_trajectories=clean_trajectories,
|
|
655
|
+
chaos_trajectories=chaos_trajectories,
|
|
656
|
+
injection_records=injection_records,
|
|
657
|
+
injector_cost_usd=injector_cost_usd,
|
|
658
|
+
)
|
|
659
|
+
print()
|
|
660
|
+
print(report.render())
|
|
661
|
+
|
|
662
|
+
if args.detailed:
|
|
663
|
+
print()
|
|
664
|
+
print(report.render_scenario_detail())
|
|
665
|
+
|
|
666
|
+
if args.json_out:
|
|
667
|
+
payload = {
|
|
668
|
+
"mode": args.mode,
|
|
669
|
+
"clean_trajectories": [_serialize_trajectory(t) for t in clean_trajectories],
|
|
670
|
+
"chaos_trajectories": [_serialize_trajectory(t) for t in chaos_trajectories],
|
|
671
|
+
}
|
|
672
|
+
with open(args.json_out, "w") as f:
|
|
673
|
+
json.dump(payload, f, indent=2)
|
|
674
|
+
|
|
675
|
+
if args.html_out:
|
|
676
|
+
with open(args.html_out, "w") as f:
|
|
677
|
+
f.write(report.render_html())
|
|
678
|
+
print(f"HTML report -> {args.html_out}", file=sys.stderr)
|
|
679
|
+
|
|
680
|
+
return 0
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
if __name__ == "__main__":
|
|
684
|
+
raise SystemExit(main())
|