agentstress 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentstress/__init__.py +0 -0
- agentstress/cli.py +313 -0
- agentstress/correctness.py +460 -0
- agentstress/grader.py +300 -0
- agentstress/grading/__init__.py +0 -0
- agentstress/grading/batch_judge.py +211 -0
- agentstress/grading/claude_judge.py +348 -0
- agentstress/grading/faq_rubric.md +175 -0
- agentstress/grading/inv_rubric.md +200 -0
- agentstress/grading/ram_rubric.md +143 -0
- agentstress/grading/ut_rubric.md +172 -0
- agentstress/harnesses/__init__.py +0 -0
- agentstress/harnesses/crewai_runner.py +87 -0
- agentstress/harnesses/crewai_tools_impl.py +77 -0
- agentstress/harnesses/langgraph_runner.py +54 -0
- agentstress/harnesses/langgraph_tools.py +77 -0
- agentstress/harnesses/models.py +52 -0
- agentstress/harnesses/openai_runner.py +73 -0
- agentstress/harnesses/openai_tools_impl.py +82 -0
- agentstress/proxy.py +60 -0
- agentstress/run_config.py +49 -0
- agentstress/scenario_model.py +44 -0
- agentstress/scenarios.py +399 -0
- agentstress/scenarios_expansion.py +1033 -0
- agentstress/scenarios_phase1.py +1082 -0
- agentstress/tools.py +262 -0
- agentstress/trace.py +124 -0
- agentstress-0.1.0.dist-info/METADATA +185 -0
- agentstress-0.1.0.dist-info/RECORD +33 -0
- agentstress-0.1.0.dist-info/WHEEL +5 -0
- agentstress-0.1.0.dist-info/entry_points.txt +2 -0
- agentstress-0.1.0.dist-info/licenses/LICENSE +21 -0
- agentstress-0.1.0.dist-info/top_level.txt +1 -0
agentstress/__init__.py
ADDED
|
File without changes
|
agentstress/cli.py
ADDED
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
"""Command line for agentstress.
|
|
2
|
+
|
|
3
|
+
agentstress list what the suite provokes
|
|
4
|
+
agentstress capture --framework crewai --scenario FAQ-12
|
|
5
|
+
agentstress run --framework langgraph --model ollama:qwen2.5:7b-instruct
|
|
6
|
+
agentstress grade --runs runs/ (step repetition: free)
|
|
7
|
+
agentstress grade --runs runs/ --judge (all modes: billed)
|
|
8
|
+
agentstress report --runs runs/
|
|
9
|
+
|
|
10
|
+
`capture` is the one to try first: it prints the prompt your framework actually
|
|
11
|
+
sends the model, which is what the study found drives failure rates.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import argparse
|
|
16
|
+
import json
|
|
17
|
+
import sys
|
|
18
|
+
import time
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
import agentstress.run_config as rc
|
|
22
|
+
from agentstress.scenarios_phase1 import ALL_PHASE1, BY_ID_PHASE1
|
|
23
|
+
|
|
24
|
+
FRAMEWORKS = {
|
|
25
|
+
"langgraph": "agentstress.harnesses.langgraph_runner",
|
|
26
|
+
"crewai": "agentstress.harnesses.crewai_runner",
|
|
27
|
+
"openai_agents": "agentstress.harnesses.openai_runner",
|
|
28
|
+
}
|
|
29
|
+
MODE_NAMES = {
|
|
30
|
+
"SR": "step repetition", "RAM": "reasoning-action mismatch",
|
|
31
|
+
"UT": "unaware of termination", "FAQ": "fail to ask for clarification",
|
|
32
|
+
"INV": "incorrect / no verification",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _select(args) -> list:
|
|
37
|
+
picked = [s for s in ALL_PHASE1 if not s.retired]
|
|
38
|
+
if getattr(args, "mode", None):
|
|
39
|
+
picked = [s for s in picked if s.target_mode in args.mode]
|
|
40
|
+
if getattr(args, "scenario", None):
|
|
41
|
+
want = set(args.scenario)
|
|
42
|
+
missing = want - {s.id for s in picked}
|
|
43
|
+
if missing:
|
|
44
|
+
sys.exit(f"unknown scenario id(s): {', '.join(sorted(missing))}")
|
|
45
|
+
picked = [s for s in picked if s.id in want]
|
|
46
|
+
if not picked:
|
|
47
|
+
sys.exit("no scenarios selected")
|
|
48
|
+
return picked
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _apply_model(spec: str) -> str:
|
|
52
|
+
"""'ollama:qwen2.5:7b-instruct' or 'openai:gpt-5.4-mini'."""
|
|
53
|
+
provider, _, name = spec.partition(":")
|
|
54
|
+
if provider not in ("ollama", "openai") or not name:
|
|
55
|
+
sys.exit("--model must be ollama:<name> or openai:<name>")
|
|
56
|
+
rc.PROVIDER = provider
|
|
57
|
+
if provider == "ollama":
|
|
58
|
+
rc.OLLAMA_MODEL = name
|
|
59
|
+
else:
|
|
60
|
+
rc.OPENAI_MODEL = name
|
|
61
|
+
import os
|
|
62
|
+
|
|
63
|
+
if not os.environ.get("OPENAI_API_KEY"):
|
|
64
|
+
from dotenv import dotenv_values
|
|
65
|
+
|
|
66
|
+
key = dotenv_values(".env").get("OPENAI_API_KEY")
|
|
67
|
+
if not key:
|
|
68
|
+
sys.exit("set OPENAI_API_KEY (environment or .env) to use an openai model")
|
|
69
|
+
os.environ["OPENAI_API_KEY"] = key.strip()
|
|
70
|
+
return name
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def cmd_list(args) -> int:
|
|
74
|
+
rows = _select(args)
|
|
75
|
+
by_mode: dict[str, list] = {}
|
|
76
|
+
for s in rows:
|
|
77
|
+
by_mode.setdefault(s.target_mode, []).append(s)
|
|
78
|
+
for mode, items in by_mode.items():
|
|
79
|
+
print(f"\n{mode} — {MODE_NAMES[mode]} ({len(items)} scenarios)")
|
|
80
|
+
for s in items:
|
|
81
|
+
tags = []
|
|
82
|
+
if s.is_control:
|
|
83
|
+
tags.append("control")
|
|
84
|
+
if s.architecture == "handoff":
|
|
85
|
+
tags.append("handoff")
|
|
86
|
+
if s.is_exploratory:
|
|
87
|
+
tags.append("exploratory")
|
|
88
|
+
tag = f" [{', '.join(tags)}]" if tags else ""
|
|
89
|
+
prompt = (s.prompt or s.researcher_prompt or "").strip().replace("\n", " ")
|
|
90
|
+
print(f" {s.id:8s} {s.name:34s}{tag}")
|
|
91
|
+
if args.verbose:
|
|
92
|
+
print(f" {prompt[:100]}")
|
|
93
|
+
print(f"\n{len(rows)} scenarios. Controls are scenarios where the failure is not "
|
|
94
|
+
f"available; a grader that flags them is over-flagging.")
|
|
95
|
+
return 0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def cmd_capture(args) -> int:
|
|
99
|
+
"""Print the requests a framework actually sends for one scenario."""
|
|
100
|
+
from agentstress import proxy as cp
|
|
101
|
+
|
|
102
|
+
sc = BY_ID_PHASE1[args.scenario[0]] if args.scenario else ALL_PHASE1[0]
|
|
103
|
+
_apply_model(args.model)
|
|
104
|
+
if rc.PROVIDER != "ollama":
|
|
105
|
+
sys.exit("capture currently proxies the local Ollama endpoint; use --model ollama:<name>")
|
|
106
|
+
upstream = rc.OLLAMA_BASE_URL
|
|
107
|
+
srv = cp.start()
|
|
108
|
+
cp.upstream = lambda: upstream
|
|
109
|
+
rc.OLLAMA_BASE_URL = f"http://127.0.0.1:{cp.PORT}"
|
|
110
|
+
rc.OLLAMA_OPENAI_BASE_URL = f"http://127.0.0.1:{cp.PORT}/v1"
|
|
111
|
+
cp.CURRENT["fw"] = args.framework
|
|
112
|
+
|
|
113
|
+
import importlib
|
|
114
|
+
|
|
115
|
+
from agentstress.tools import default_world
|
|
116
|
+
from agentstress.trace import Trace, set_run_context
|
|
117
|
+
|
|
118
|
+
tr = Trace(scenario_id=sc.id, framework=args.framework, architecture=sc.architecture)
|
|
119
|
+
set_run_context(tr, default_world())
|
|
120
|
+
importlib.import_module(FRAMEWORKS[args.framework]).run_scenario(sc, tr)
|
|
121
|
+
srv.shutdown()
|
|
122
|
+
|
|
123
|
+
if not cp.LOG:
|
|
124
|
+
print("no requests captured")
|
|
125
|
+
return 1
|
|
126
|
+
body = cp.LOG[0]["body"]
|
|
127
|
+
print(f"=== {args.framework} · {sc.id} · first request to the model ===")
|
|
128
|
+
for m in body.get("messages", [{"role": "user", "content": sc.prompt}]):
|
|
129
|
+
print(f"\n--- {m.get('role')} ---\n{m.get('content')}")
|
|
130
|
+
print(f"\n--- tools offered: {[t.get('function', {}).get('name') for t in body.get('tools', [])]}")
|
|
131
|
+
print(f"--- your task text was: {(sc.prompt or sc.researcher_prompt or '').strip()!r}")
|
|
132
|
+
print("\nEverything above that you did not write is scaffolding the framework added.")
|
|
133
|
+
return 0
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def cmd_run(args) -> int:
|
|
137
|
+
import importlib
|
|
138
|
+
|
|
139
|
+
from agentstress.tools import default_world
|
|
140
|
+
from agentstress.trace import Trace, set_run_context
|
|
141
|
+
|
|
142
|
+
model = _apply_model(args.model)
|
|
143
|
+
rc.TEMPERATURE = args.temperature
|
|
144
|
+
scenarios = _select(args)
|
|
145
|
+
out = Path(args.out)
|
|
146
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
147
|
+
runner = importlib.import_module(FRAMEWORKS[args.framework]).run_scenario
|
|
148
|
+
|
|
149
|
+
jobs = [(s, t) for t in range(1, args.trials + 1) for s in scenarios]
|
|
150
|
+
todo = [j for j in jobs if not (out / f"{j[0].id}__{args.framework}__t{j[1]}.json").exists()]
|
|
151
|
+
print(f"{len(scenarios)} scenarios x {args.trials} trials on {args.framework} / {model} "
|
|
152
|
+
f"= {len(jobs)} runs | to run {len(todo)} | out: {out}", flush=True)
|
|
153
|
+
started, errors = time.time(), 0
|
|
154
|
+
for n, (sc, trial) in enumerate(todo, 1):
|
|
155
|
+
tr = Trace(scenario_id=sc.id, framework=args.framework, architecture=sc.architecture)
|
|
156
|
+
set_run_context(tr, default_world())
|
|
157
|
+
try:
|
|
158
|
+
runner(sc, tr)
|
|
159
|
+
except Exception as e:
|
|
160
|
+
tr.error = f"{type(e).__name__}: {e}"
|
|
161
|
+
(out / f"{sc.id}__{args.framework}__t{trial}.json").write_text(json.dumps({
|
|
162
|
+
"scenario_id": sc.id, "scenario_name": sc.name, "target_mode": sc.target_mode,
|
|
163
|
+
"architecture": sc.architecture, "is_control": sc.is_control,
|
|
164
|
+
"is_exploratory": sc.is_exploratory, "framework": args.framework, "trial": trial,
|
|
165
|
+
"model": model, "temperature": rc.TEMPERATURE,
|
|
166
|
+
"expected_clean_calls": sc.expected_clean_calls,
|
|
167
|
+
"agent_output": tr.final_answer, "agent_error": tr.error,
|
|
168
|
+
"wall_seconds": round(tr.wall_seconds, 2), "trace": tr.to_json(),
|
|
169
|
+
}, indent=2))
|
|
170
|
+
errors += bool(tr.error)
|
|
171
|
+
eta = (time.time() - started) / n * (len(todo) - n)
|
|
172
|
+
print(f"[{n:4d}/{len(todo)}] {'ERR' if tr.error else 'ok '} {sc.id:8s} t{trial} "
|
|
173
|
+
f"{len(tr.calls):2d} calls {tr.wall_seconds:6.1f}s eta {eta/60:5.1f}m", flush=True)
|
|
174
|
+
print(f"done — {len(todo)} runs, {errors} harness error(s). Next: agentstress grade --runs {out}")
|
|
175
|
+
return 0
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _grade_dir(runs: Path, use_judge: bool):
|
|
179
|
+
"""-> rows [{scenario_id, framework, trial, mode, outcome, detail}]"""
|
|
180
|
+
from agentstress.correctness import check
|
|
181
|
+
from agentstress.grader import grade_trace
|
|
182
|
+
|
|
183
|
+
judge = None
|
|
184
|
+
if use_judge:
|
|
185
|
+
from agentstress.grading.claude_judge import judge_scenario
|
|
186
|
+
|
|
187
|
+
judge = judge_scenario
|
|
188
|
+
rows = []
|
|
189
|
+
for f in sorted(runs.glob("*.json")):
|
|
190
|
+
d = json.loads(f.read_text())
|
|
191
|
+
sc = BY_ID_PHASE1.get(d["scenario_id"])
|
|
192
|
+
if sc is None:
|
|
193
|
+
continue
|
|
194
|
+
if d.get("agent_error"):
|
|
195
|
+
outcome, detail = "ERROR", d["agent_error"][:80]
|
|
196
|
+
elif sc.target_mode == "SR":
|
|
197
|
+
g = grade_trace(d["trace"], sc.repeated_mutations_expected)
|
|
198
|
+
outcome = "FAIL" if g["extended_verdict"] == "FAIL" else "PASS"
|
|
199
|
+
detail = g["detail"] if isinstance(g.get("detail"), str) else ""
|
|
200
|
+
elif judge is None:
|
|
201
|
+
outcome, detail = "UNGRADED", "needs --judge"
|
|
202
|
+
else:
|
|
203
|
+
from agentstress.grading.claude_judge import agent_prompt_shown_to_judge
|
|
204
|
+
|
|
205
|
+
v = judge(sc.id, sc.target_mode, d["framework"], d.get("agent_output") or "",
|
|
206
|
+
agent_prompt_shown_to_judge(sc),
|
|
207
|
+
structural_reason=sc.structural_reason,
|
|
208
|
+
tool_calls=d["trace"]["calls"],
|
|
209
|
+
expected_clean_calls=sc.expected_clean_calls)
|
|
210
|
+
outcome, detail = v["verdict"], f"score {v['score']}"
|
|
211
|
+
corr = check(sc.id, d["trace"], d.get("agent_output") or "")
|
|
212
|
+
rows.append({"scenario_id": sc.id, "framework": d["framework"], "trial": d["trial"],
|
|
213
|
+
"mode": sc.target_mode, "outcome": outcome, "detail": detail,
|
|
214
|
+
"correct": None if corr is None else corr["correct"]})
|
|
215
|
+
return rows
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def cmd_grade(args) -> int:
|
|
219
|
+
runs = Path(args.runs)
|
|
220
|
+
if not runs.exists():
|
|
221
|
+
sys.exit(f"no such directory: {runs}")
|
|
222
|
+
modes = [BY_ID_PHASE1[d["scenario_id"]].target_mode
|
|
223
|
+
for d in (json.loads(f.read_text()) for f in runs.glob("*.json"))
|
|
224
|
+
if d["scenario_id"] in BY_ID_PHASE1]
|
|
225
|
+
n_judge = sum(m != "SR" for m in modes)
|
|
226
|
+
if args.judge:
|
|
227
|
+
print(f"judging {n_judge} runs with {rc_judge_model()} — this is billed", flush=True)
|
|
228
|
+
elif n_judge:
|
|
229
|
+
print(f"note: {n_judge} runs need the LLM judge (--judge); grading step repetition only")
|
|
230
|
+
rows = _grade_dir(runs, args.judge)
|
|
231
|
+
(runs / "graded.json").write_text(json.dumps(rows, indent=2))
|
|
232
|
+
print(f"wrote {runs / 'graded.json'} ({len(rows)} rows)")
|
|
233
|
+
return cmd_report(args)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def rc_judge_model() -> str:
|
|
237
|
+
from agentstress.grading.claude_judge import JUDGE_MODEL
|
|
238
|
+
|
|
239
|
+
return JUDGE_MODEL
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def cmd_report(args) -> int:
|
|
243
|
+
runs = Path(args.runs)
|
|
244
|
+
path = runs / "graded.json"
|
|
245
|
+
if not path.exists():
|
|
246
|
+
sys.exit(f"no graded.json in {runs} — run: agentstress grade --runs {runs}")
|
|
247
|
+
rows = json.loads(path.read_text())
|
|
248
|
+
by: dict[tuple, list] = {}
|
|
249
|
+
for r in rows:
|
|
250
|
+
by.setdefault((r["mode"], r["framework"]), []).append(r)
|
|
251
|
+
print(f"\n{'mode':6s} {'framework':16s} {'runs':>5} {'failure rate':>13} {'correct':>9}")
|
|
252
|
+
for (mode, fw), items in sorted(by.items()):
|
|
253
|
+
bad = sum(i["outcome"] in ("FAIL", "ERROR") for i in items)
|
|
254
|
+
graded = [i for i in items if i["outcome"] != "UNGRADED"]
|
|
255
|
+
checked = [i for i in items if i["correct"] is not None]
|
|
256
|
+
rate = f"{100 * bad / len(graded):.0f}%" if graded else "ungraded"
|
|
257
|
+
corr = f"{100 * sum(bool(i['correct']) for i in checked) / len(checked):.0f}%" if checked else "-"
|
|
258
|
+
print(f"{mode:6s} {fw:16s} {len(items):5d} {rate:>13} {corr:>9}")
|
|
259
|
+
worst = sorted({r["scenario_id"] for r in rows if r["outcome"] in ("FAIL", "ERROR")})
|
|
260
|
+
if worst:
|
|
261
|
+
print(f"\nscenarios with at least one failure ({len(worst)}): {', '.join(worst[:24])}"
|
|
262
|
+
+ (" ..." if len(worst) > 24 else ""))
|
|
263
|
+
print("\nFailure rate and correctness are separate axes: a run can avoid the failure "
|
|
264
|
+
"mode and still answer wrongly.")
|
|
265
|
+
return 0
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def main(argv=None) -> int:
|
|
269
|
+
ap = argparse.ArgumentParser(prog="agentstress", description=__doc__,
|
|
270
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
271
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
272
|
+
|
|
273
|
+
def add_select(p):
|
|
274
|
+
p.add_argument("--mode", nargs="+", choices=list(MODE_NAMES), help="limit to these failure modes")
|
|
275
|
+
p.add_argument("--scenario", nargs="+", help="scenario ids, e.g. FAQ-12 SR-02")
|
|
276
|
+
|
|
277
|
+
p = sub.add_parser("list", help="list scenarios")
|
|
278
|
+
add_select(p)
|
|
279
|
+
p.add_argument("--verbose", action="store_true", help="show the task text")
|
|
280
|
+
p.set_defaults(func=cmd_list)
|
|
281
|
+
|
|
282
|
+
p = sub.add_parser("capture", help="print what your framework sends the model")
|
|
283
|
+
p.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
|
|
284
|
+
p.add_argument("--scenario", nargs=1, required=True)
|
|
285
|
+
p.add_argument("--model", default="ollama:qwen2.5:7b-instruct")
|
|
286
|
+
p.set_defaults(func=cmd_capture)
|
|
287
|
+
|
|
288
|
+
p = sub.add_parser("run", help="run scenarios against a framework")
|
|
289
|
+
p.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
|
|
290
|
+
p.add_argument("--model", default="ollama:qwen2.5:7b-instruct",
|
|
291
|
+
help="ollama:<name> or openai:<name>")
|
|
292
|
+
p.add_argument("--trials", type=int, default=2)
|
|
293
|
+
p.add_argument("--temperature", type=float, default=rc.TEMPERATURE)
|
|
294
|
+
p.add_argument("--out", default="runs")
|
|
295
|
+
add_select(p)
|
|
296
|
+
p.set_defaults(func=cmd_run)
|
|
297
|
+
|
|
298
|
+
p = sub.add_parser("grade", help="grade a runs directory")
|
|
299
|
+
p.add_argument("--runs", default="runs")
|
|
300
|
+
p.add_argument("--judge", action="store_true",
|
|
301
|
+
help="use the LLM judge for RAM/UT/FAQ/INV (billed; needs ANTHROPIC_API_KEY)")
|
|
302
|
+
p.set_defaults(func=cmd_grade)
|
|
303
|
+
|
|
304
|
+
p = sub.add_parser("report", help="summarise graded runs")
|
|
305
|
+
p.add_argument("--runs", default="runs")
|
|
306
|
+
p.set_defaults(func=cmd_report)
|
|
307
|
+
|
|
308
|
+
args = ap.parse_args(argv)
|
|
309
|
+
return args.func(args)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
if __name__ == "__main__":
|
|
313
|
+
sys.exit(main())
|