agentstress 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
File without changes
agentstress/cli.py ADDED
@@ -0,0 +1,313 @@
1
+ """Command line for agentstress.
2
+
3
+ agentstress list what the suite provokes
4
+ agentstress capture --framework crewai --scenario FAQ-12
5
+ agentstress run --framework langgraph --model ollama:qwen2.5:7b-instruct
6
+ agentstress grade --runs runs/ (step repetition: free)
7
+ agentstress grade --runs runs/ --judge (all modes: billed)
8
+ agentstress report --runs runs/
9
+
10
+ `capture` is the one to try first: it prints the prompt your framework actually
11
+ sends the model, which is what the study found drives failure rates.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import argparse
16
+ import json
17
+ import sys
18
+ import time
19
+ from pathlib import Path
20
+
21
+ import agentstress.run_config as rc
22
+ from agentstress.scenarios_phase1 import ALL_PHASE1, BY_ID_PHASE1
23
+
24
+ FRAMEWORKS = {
25
+ "langgraph": "agentstress.harnesses.langgraph_runner",
26
+ "crewai": "agentstress.harnesses.crewai_runner",
27
+ "openai_agents": "agentstress.harnesses.openai_runner",
28
+ }
29
+ MODE_NAMES = {
30
+ "SR": "step repetition", "RAM": "reasoning-action mismatch",
31
+ "UT": "unaware of termination", "FAQ": "fail to ask for clarification",
32
+ "INV": "incorrect / no verification",
33
+ }
34
+
35
+
36
+ def _select(args) -> list:
37
+ picked = [s for s in ALL_PHASE1 if not s.retired]
38
+ if getattr(args, "mode", None):
39
+ picked = [s for s in picked if s.target_mode in args.mode]
40
+ if getattr(args, "scenario", None):
41
+ want = set(args.scenario)
42
+ missing = want - {s.id for s in picked}
43
+ if missing:
44
+ sys.exit(f"unknown scenario id(s): {', '.join(sorted(missing))}")
45
+ picked = [s for s in picked if s.id in want]
46
+ if not picked:
47
+ sys.exit("no scenarios selected")
48
+ return picked
49
+
50
+
51
+ def _apply_model(spec: str) -> str:
52
+ """'ollama:qwen2.5:7b-instruct' or 'openai:gpt-5.4-mini'."""
53
+ provider, _, name = spec.partition(":")
54
+ if provider not in ("ollama", "openai") or not name:
55
+ sys.exit("--model must be ollama:<name> or openai:<name>")
56
+ rc.PROVIDER = provider
57
+ if provider == "ollama":
58
+ rc.OLLAMA_MODEL = name
59
+ else:
60
+ rc.OPENAI_MODEL = name
61
+ import os
62
+
63
+ if not os.environ.get("OPENAI_API_KEY"):
64
+ from dotenv import dotenv_values
65
+
66
+ key = dotenv_values(".env").get("OPENAI_API_KEY")
67
+ if not key:
68
+ sys.exit("set OPENAI_API_KEY (environment or .env) to use an openai model")
69
+ os.environ["OPENAI_API_KEY"] = key.strip()
70
+ return name
71
+
72
+
73
+ def cmd_list(args) -> int:
74
+ rows = _select(args)
75
+ by_mode: dict[str, list] = {}
76
+ for s in rows:
77
+ by_mode.setdefault(s.target_mode, []).append(s)
78
+ for mode, items in by_mode.items():
79
+ print(f"\n{mode} — {MODE_NAMES[mode]} ({len(items)} scenarios)")
80
+ for s in items:
81
+ tags = []
82
+ if s.is_control:
83
+ tags.append("control")
84
+ if s.architecture == "handoff":
85
+ tags.append("handoff")
86
+ if s.is_exploratory:
87
+ tags.append("exploratory")
88
+ tag = f" [{', '.join(tags)}]" if tags else ""
89
+ prompt = (s.prompt or s.researcher_prompt or "").strip().replace("\n", " ")
90
+ print(f" {s.id:8s} {s.name:34s}{tag}")
91
+ if args.verbose:
92
+ print(f" {prompt[:100]}")
93
+ print(f"\n{len(rows)} scenarios. Controls are scenarios where the failure is not "
94
+ f"available; a grader that flags them is over-flagging.")
95
+ return 0
96
+
97
+
98
+ def cmd_capture(args) -> int:
99
+ """Print the requests a framework actually sends for one scenario."""
100
+ from agentstress import proxy as cp
101
+
102
+ sc = BY_ID_PHASE1[args.scenario[0]] if args.scenario else ALL_PHASE1[0]
103
+ _apply_model(args.model)
104
+ if rc.PROVIDER != "ollama":
105
+ sys.exit("capture currently proxies the local Ollama endpoint; use --model ollama:<name>")
106
+ upstream = rc.OLLAMA_BASE_URL
107
+ srv = cp.start()
108
+ cp.upstream = lambda: upstream
109
+ rc.OLLAMA_BASE_URL = f"http://127.0.0.1:{cp.PORT}"
110
+ rc.OLLAMA_OPENAI_BASE_URL = f"http://127.0.0.1:{cp.PORT}/v1"
111
+ cp.CURRENT["fw"] = args.framework
112
+
113
+ import importlib
114
+
115
+ from agentstress.tools import default_world
116
+ from agentstress.trace import Trace, set_run_context
117
+
118
+ tr = Trace(scenario_id=sc.id, framework=args.framework, architecture=sc.architecture)
119
+ set_run_context(tr, default_world())
120
+ importlib.import_module(FRAMEWORKS[args.framework]).run_scenario(sc, tr)
121
+ srv.shutdown()
122
+
123
+ if not cp.LOG:
124
+ print("no requests captured")
125
+ return 1
126
+ body = cp.LOG[0]["body"]
127
+ print(f"=== {args.framework} · {sc.id} · first request to the model ===")
128
+ for m in body.get("messages", [{"role": "user", "content": sc.prompt}]):
129
+ print(f"\n--- {m.get('role')} ---\n{m.get('content')}")
130
+ print(f"\n--- tools offered: {[t.get('function', {}).get('name') for t in body.get('tools', [])]}")
131
+ print(f"--- your task text was: {(sc.prompt or sc.researcher_prompt or '').strip()!r}")
132
+ print("\nEverything above that you did not write is scaffolding the framework added.")
133
+ return 0
134
+
135
+
136
+ def cmd_run(args) -> int:
137
+ import importlib
138
+
139
+ from agentstress.tools import default_world
140
+ from agentstress.trace import Trace, set_run_context
141
+
142
+ model = _apply_model(args.model)
143
+ rc.TEMPERATURE = args.temperature
144
+ scenarios = _select(args)
145
+ out = Path(args.out)
146
+ out.mkdir(parents=True, exist_ok=True)
147
+ runner = importlib.import_module(FRAMEWORKS[args.framework]).run_scenario
148
+
149
+ jobs = [(s, t) for t in range(1, args.trials + 1) for s in scenarios]
150
+ todo = [j for j in jobs if not (out / f"{j[0].id}__{args.framework}__t{j[1]}.json").exists()]
151
+ print(f"{len(scenarios)} scenarios x {args.trials} trials on {args.framework} / {model} "
152
+ f"= {len(jobs)} runs | to run {len(todo)} | out: {out}", flush=True)
153
+ started, errors = time.time(), 0
154
+ for n, (sc, trial) in enumerate(todo, 1):
155
+ tr = Trace(scenario_id=sc.id, framework=args.framework, architecture=sc.architecture)
156
+ set_run_context(tr, default_world())
157
+ try:
158
+ runner(sc, tr)
159
+ except Exception as e:
160
+ tr.error = f"{type(e).__name__}: {e}"
161
+ (out / f"{sc.id}__{args.framework}__t{trial}.json").write_text(json.dumps({
162
+ "scenario_id": sc.id, "scenario_name": sc.name, "target_mode": sc.target_mode,
163
+ "architecture": sc.architecture, "is_control": sc.is_control,
164
+ "is_exploratory": sc.is_exploratory, "framework": args.framework, "trial": trial,
165
+ "model": model, "temperature": rc.TEMPERATURE,
166
+ "expected_clean_calls": sc.expected_clean_calls,
167
+ "agent_output": tr.final_answer, "agent_error": tr.error,
168
+ "wall_seconds": round(tr.wall_seconds, 2), "trace": tr.to_json(),
169
+ }, indent=2))
170
+ errors += bool(tr.error)
171
+ eta = (time.time() - started) / n * (len(todo) - n)
172
+ print(f"[{n:4d}/{len(todo)}] {'ERR' if tr.error else 'ok '} {sc.id:8s} t{trial} "
173
+ f"{len(tr.calls):2d} calls {tr.wall_seconds:6.1f}s eta {eta/60:5.1f}m", flush=True)
174
+ print(f"done — {len(todo)} runs, {errors} harness error(s). Next: agentstress grade --runs {out}")
175
+ return 0
176
+
177
+
178
+ def _grade_dir(runs: Path, use_judge: bool):
179
+ """-> rows [{scenario_id, framework, trial, mode, outcome, detail}]"""
180
+ from agentstress.correctness import check
181
+ from agentstress.grader import grade_trace
182
+
183
+ judge = None
184
+ if use_judge:
185
+ from agentstress.grading.claude_judge import judge_scenario
186
+
187
+ judge = judge_scenario
188
+ rows = []
189
+ for f in sorted(runs.glob("*.json")):
190
+ d = json.loads(f.read_text())
191
+ sc = BY_ID_PHASE1.get(d["scenario_id"])
192
+ if sc is None:
193
+ continue
194
+ if d.get("agent_error"):
195
+ outcome, detail = "ERROR", d["agent_error"][:80]
196
+ elif sc.target_mode == "SR":
197
+ g = grade_trace(d["trace"], sc.repeated_mutations_expected)
198
+ outcome = "FAIL" if g["extended_verdict"] == "FAIL" else "PASS"
199
+ detail = g["detail"] if isinstance(g.get("detail"), str) else ""
200
+ elif judge is None:
201
+ outcome, detail = "UNGRADED", "needs --judge"
202
+ else:
203
+ from agentstress.grading.claude_judge import agent_prompt_shown_to_judge
204
+
205
+ v = judge(sc.id, sc.target_mode, d["framework"], d.get("agent_output") or "",
206
+ agent_prompt_shown_to_judge(sc),
207
+ structural_reason=sc.structural_reason,
208
+ tool_calls=d["trace"]["calls"],
209
+ expected_clean_calls=sc.expected_clean_calls)
210
+ outcome, detail = v["verdict"], f"score {v['score']}"
211
+ corr = check(sc.id, d["trace"], d.get("agent_output") or "")
212
+ rows.append({"scenario_id": sc.id, "framework": d["framework"], "trial": d["trial"],
213
+ "mode": sc.target_mode, "outcome": outcome, "detail": detail,
214
+ "correct": None if corr is None else corr["correct"]})
215
+ return rows
216
+
217
+
218
+ def cmd_grade(args) -> int:
219
+ runs = Path(args.runs)
220
+ if not runs.exists():
221
+ sys.exit(f"no such directory: {runs}")
222
+ modes = [BY_ID_PHASE1[d["scenario_id"]].target_mode
223
+ for d in (json.loads(f.read_text()) for f in runs.glob("*.json"))
224
+ if d["scenario_id"] in BY_ID_PHASE1]
225
+ n_judge = sum(m != "SR" for m in modes)
226
+ if args.judge:
227
+ print(f"judging {n_judge} runs with {rc_judge_model()} — this is billed", flush=True)
228
+ elif n_judge:
229
+ print(f"note: {n_judge} runs need the LLM judge (--judge); grading step repetition only")
230
+ rows = _grade_dir(runs, args.judge)
231
+ (runs / "graded.json").write_text(json.dumps(rows, indent=2))
232
+ print(f"wrote {runs / 'graded.json'} ({len(rows)} rows)")
233
+ return cmd_report(args)
234
+
235
+
236
+ def rc_judge_model() -> str:
237
+ from agentstress.grading.claude_judge import JUDGE_MODEL
238
+
239
+ return JUDGE_MODEL
240
+
241
+
242
+ def cmd_report(args) -> int:
243
+ runs = Path(args.runs)
244
+ path = runs / "graded.json"
245
+ if not path.exists():
246
+ sys.exit(f"no graded.json in {runs} — run: agentstress grade --runs {runs}")
247
+ rows = json.loads(path.read_text())
248
+ by: dict[tuple, list] = {}
249
+ for r in rows:
250
+ by.setdefault((r["mode"], r["framework"]), []).append(r)
251
+ print(f"\n{'mode':6s} {'framework':16s} {'runs':>5} {'failure rate':>13} {'correct':>9}")
252
+ for (mode, fw), items in sorted(by.items()):
253
+ bad = sum(i["outcome"] in ("FAIL", "ERROR") for i in items)
254
+ graded = [i for i in items if i["outcome"] != "UNGRADED"]
255
+ checked = [i for i in items if i["correct"] is not None]
256
+ rate = f"{100 * bad / len(graded):.0f}%" if graded else "ungraded"
257
+ corr = f"{100 * sum(bool(i['correct']) for i in checked) / len(checked):.0f}%" if checked else "-"
258
+ print(f"{mode:6s} {fw:16s} {len(items):5d} {rate:>13} {corr:>9}")
259
+ worst = sorted({r["scenario_id"] for r in rows if r["outcome"] in ("FAIL", "ERROR")})
260
+ if worst:
261
+ print(f"\nscenarios with at least one failure ({len(worst)}): {', '.join(worst[:24])}"
262
+ + (" ..." if len(worst) > 24 else ""))
263
+ print("\nFailure rate and correctness are separate axes: a run can avoid the failure "
264
+ "mode and still answer wrongly.")
265
+ return 0
266
+
267
+
268
+ def main(argv=None) -> int:
269
+ ap = argparse.ArgumentParser(prog="agentstress", description=__doc__,
270
+ formatter_class=argparse.RawDescriptionHelpFormatter)
271
+ sub = ap.add_subparsers(dest="cmd", required=True)
272
+
273
+ def add_select(p):
274
+ p.add_argument("--mode", nargs="+", choices=list(MODE_NAMES), help="limit to these failure modes")
275
+ p.add_argument("--scenario", nargs="+", help="scenario ids, e.g. FAQ-12 SR-02")
276
+
277
+ p = sub.add_parser("list", help="list scenarios")
278
+ add_select(p)
279
+ p.add_argument("--verbose", action="store_true", help="show the task text")
280
+ p.set_defaults(func=cmd_list)
281
+
282
+ p = sub.add_parser("capture", help="print what your framework sends the model")
283
+ p.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
284
+ p.add_argument("--scenario", nargs=1, required=True)
285
+ p.add_argument("--model", default="ollama:qwen2.5:7b-instruct")
286
+ p.set_defaults(func=cmd_capture)
287
+
288
+ p = sub.add_parser("run", help="run scenarios against a framework")
289
+ p.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
290
+ p.add_argument("--model", default="ollama:qwen2.5:7b-instruct",
291
+ help="ollama:<name> or openai:<name>")
292
+ p.add_argument("--trials", type=int, default=2)
293
+ p.add_argument("--temperature", type=float, default=rc.TEMPERATURE)
294
+ p.add_argument("--out", default="runs")
295
+ add_select(p)
296
+ p.set_defaults(func=cmd_run)
297
+
298
+ p = sub.add_parser("grade", help="grade a runs directory")
299
+ p.add_argument("--runs", default="runs")
300
+ p.add_argument("--judge", action="store_true",
301
+ help="use the LLM judge for RAM/UT/FAQ/INV (billed; needs ANTHROPIC_API_KEY)")
302
+ p.set_defaults(func=cmd_grade)
303
+
304
+ p = sub.add_parser("report", help="summarise graded runs")
305
+ p.add_argument("--runs", default="runs")
306
+ p.set_defaults(func=cmd_report)
307
+
308
+ args = ap.parse_args(argv)
309
+ return args.func(args)
310
+
311
+
312
+ if __name__ == "__main__":
313
+ sys.exit(main())