autoresearcheval 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoresearcheval-0.1.0/LICENSE +21 -0
- autoresearcheval-0.1.0/PKG-INFO +229 -0
- autoresearcheval-0.1.0/README.md +199 -0
- autoresearcheval-0.1.0/pyproject.toml +55 -0
- autoresearcheval-0.1.0/setup.cfg +4 -0
- autoresearcheval-0.1.0/src/autoresearcheval/__init__.py +36 -0
- autoresearcheval-0.1.0/src/autoresearcheval/aggregate.py +414 -0
- autoresearcheval-0.1.0/src/autoresearcheval/analysis_qa.py +381 -0
- autoresearcheval-0.1.0/src/autoresearcheval/api.py +312 -0
- autoresearcheval-0.1.0/src/autoresearcheval/classify.py +411 -0
- autoresearcheval-0.1.0/src/autoresearcheval/classify_cc.py +371 -0
- autoresearcheval-0.1.0/src/autoresearcheval/config.py +69 -0
- autoresearcheval-0.1.0/src/autoresearcheval/data/ONBOARDING.md +250 -0
- autoresearcheval-0.1.0/src/autoresearcheval/data/analysis_long.md +150 -0
- autoresearcheval-0.1.0/src/autoresearcheval/data/arft_guide.md +274 -0
- autoresearcheval-0.1.0/src/autoresearcheval/generate.py +501 -0
- autoresearcheval-0.1.0/src/autoresearcheval/label_qa.py +200 -0
- autoresearcheval-0.1.0/src/autoresearcheval/patterns.py +251 -0
- autoresearcheval-0.1.0/src/autoresearcheval/status.py +40 -0
- autoresearcheval-0.1.0/src/autoresearcheval/traj_tools.py +956 -0
- autoresearcheval-0.1.0/src/autoresearcheval/verify.py +190 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/PKG-INFO +229 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/SOURCES.txt +26 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/dependency_links.txt +1 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/entry_points.txt +9 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/requires.txt +9 -0
- autoresearcheval-0.1.0/src/autoresearcheval.egg-info/top_level.txt +1 -0
- autoresearcheval-0.1.0/tests/test_package.py +186 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Titan Research Labs
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: autoresearcheval
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Process-level failure diagnosis for autonomous research agents: trajectory -> structured analysis -> ARFT failure labels.
|
|
5
|
+
Author: AutoResearchEval authors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/PrentisAI/AutoResearchEval
|
|
8
|
+
Project-URL: Paper, https://arxiv.org/abs/2608.14905
|
|
9
|
+
Project-URL: Source, https://github.com/PrentisAI/AutoResearchEval/tree/master/agent-as-a-judge
|
|
10
|
+
Keywords: llm,agents,evaluation,failure-taxonomy,agent-as-a-judge,arft
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: httpx>=0.27
|
|
23
|
+
Provides-Extra: stats
|
|
24
|
+
Requires-Dist: pandas>=2.0; extra == "stats"
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
27
|
+
Requires-Dist: build; extra == "dev"
|
|
28
|
+
Requires-Dist: twine; extra == "dev"
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# autoresearcheval
|
|
32
|
+
|
|
33
|
+
[](https://pypi.org/project/autoresearcheval/)
|
|
34
|
+
|
|
35
|
+
A two-stage pipeline for turning raw AI-agent research trajectories into a
|
|
36
|
+
structured, evidence-grounded failure-taxonomy classification.
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
raw trajectory log --[Stage 1]--> analysis.md --[Stage 2]--> ARFT labels
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
from autoresearcheval import generate_analysis, label_arft, pattern_info
|
|
44
|
+
|
|
45
|
+
analysis = generate_analysis("path/to/trajectory_dir") # Stage 1
|
|
46
|
+
result = label_arft(analysis["analysis"], api_key="sk-...") # Stage 2
|
|
47
|
+
|
|
48
|
+
print(result["summary"], result["total_failures"])
|
|
49
|
+
for code in result["failure_modes"]:
|
|
50
|
+
print(code, pattern_info(code)["name"])
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
A trajectory directory (or a single trajectory JSON, or a dict) and an API key are all
|
|
54
|
+
that is required.
|
|
55
|
+
|
|
56
|
+
**Stage 1** spawns one fresh Claude Code session per trajectory to write a deep,
|
|
57
|
+
ONBOARDING-conformant `analysis.md` — a structured, six-stage critique (ideation,
|
|
58
|
+
retrieval & synthesis, execution, analysis, writing, self-review) with a claim-by-claim
|
|
59
|
+
verdict table and independent numerical sanity checks, not a summary.
|
|
60
|
+
|
|
61
|
+
**Stage 2** classifies each `analysis.md` against **ARFT** (the AutoResearch Failure
|
|
62
|
+
Taxonomy: `A.1`–`X.8`, 45 patterns spanning six lifecycle stages plus a cross-cutting
|
|
63
|
+
layer, rolling up to four root-cause pillars — see [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md)) and rolls the
|
|
64
|
+
results into a pattern × model matrix, a root-cause breakdown, co-occurrence stats, and
|
|
65
|
+
cross-model agreement.
|
|
66
|
+
|
|
67
|
+
## Why the judge is artifact-aware
|
|
68
|
+
|
|
69
|
+
Many failures leave no trace in the report — a result the code never produced, a method
|
|
70
|
+
the logs never ran — so catching them means checking the manuscript against the
|
|
71
|
+
artifacts. Stage 1 therefore runs a fresh, zero-history session per trajectory, with
|
|
72
|
+
shell access and no network, handed the full evidence package (task statement, execution
|
|
73
|
+
log, delivered filesystem, the scorer's own source, every scoring call, read-only gold)
|
|
74
|
+
and required to anchor every finding to a line, file, or value.
|
|
75
|
+
|
|
76
|
+
Against three-expert annotation on 50 stratified trajectories this reaches **κ = 0.75**
|
|
77
|
+
(pattern) and **0.83** (root cause), versus 0.53 / 0.62 for a single-call LLM-as-a-judge
|
|
78
|
+
on the transcript alone. Almost all of the gain is recall — which is the point: artifact
|
|
79
|
+
access is what makes transcript-invisible failures detectable.
|
|
80
|
+
|
|
81
|
+
## Install
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install autoresearcheval # the library and the CLIs
|
|
85
|
+
pip install 'autoresearcheval[stats]' # adds pandas, needed for the corpus rollups
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
- Python 3.10+. The only hard dependency is `httpx`.
|
|
89
|
+
- **Stage 2** needs an API key for any OpenAI-compatible chat-completions endpoint. Key
|
|
90
|
+
resolution: the `api_key` argument, then `ARFT_OPENROUTER_KEY`, then
|
|
91
|
+
`~/.openrouter_key`, then `OPENROUTER_API_KEY`. Point it elsewhere with `base_url=` or
|
|
92
|
+
the `AAJ_ENDPOINT` env var.
|
|
93
|
+
- **Stage 1**, and Stage 2's alternate executor, spawn headless
|
|
94
|
+
[Claude Code](https://claude.com/product/claude-code) sessions and need the `claude`
|
|
95
|
+
CLI installed and authenticated.
|
|
96
|
+
|
|
97
|
+
## The two calls
|
|
98
|
+
|
|
99
|
+
| | What it does | Cost | Needs |
|
|
100
|
+
|---|---|---|---|
|
|
101
|
+
| `generate_analysis(trajectory, ...)` | reads the trajectory's artifacts and writes a six-stage critique with a claim-by-claim verdict table | minutes | the `claude` CLI |
|
|
102
|
+
| `label_arft(analysis, ...)` | maps that critique onto the 45 ARFT codes | one completion | an API key |
|
|
103
|
+
|
|
104
|
+
They are separate on purpose. Stage 1 is the expensive half, and it produces the
|
|
105
|
+
artifact that makes the labels auditable; folding the two together would hide both.
|
|
106
|
+
|
|
107
|
+
`generate_analysis` returns `{"task_id", "analysis", "qa", "path", "workspace",
|
|
108
|
+
"duration_s", "returncode"}`. `qa["ok"]` is the depth gate described below — False means
|
|
109
|
+
the analysis came back thinner than the framework's bar, not that the call failed.
|
|
110
|
+
|
|
111
|
+
`label_arft` returns the classification plus `failure_modes` (every established code),
|
|
112
|
+
`total_failures`, and `qa` (the schema-and-polarity gate). Each hit carries its `code`,
|
|
113
|
+
`name`, `stage`, `pillar`, `root_cause`, `confidence`, `evidence` and `why`.
|
|
114
|
+
|
|
115
|
+
### Telling the analyst about your harness
|
|
116
|
+
|
|
117
|
+
Two facts change what counts as a finding: whether the harness's retrieval tools did
|
|
118
|
+
real network I/O or were mocked, and whether gold values are reachable locally. By
|
|
119
|
+
default the analyst is told to **work both out from the trajectory** and report what it
|
|
120
|
+
concluded — so the defaults are correct for any harness and nothing needs editing.
|
|
121
|
+
|
|
122
|
+
Override them with `retrieval_note=` / `gold_note=` (or the `RETRIEVAL_NOTE` /
|
|
123
|
+
`GOLD_NOTE` constants for the batch CLI) **only if you can state the truth**. A declared
|
|
124
|
+
fact beats an inferred one, but a wrong one is worse than neither: an analyst told that
|
|
125
|
+
a real search tool is mocked will report fabricated retrieval that never happened, and
|
|
126
|
+
one told that a shim is real will credit calibration against literature that was never
|
|
127
|
+
fetched.
|
|
128
|
+
|
|
129
|
+
## Batch CLIs
|
|
130
|
+
|
|
131
|
+
The commands that produced the paper's corpus install alongside the library:
|
|
132
|
+
`aaj-generate`, `aaj-classify`, `aaj-classify-cc`, `aaj-aggregate`, `aaj-status`,
|
|
133
|
+
`aaj-verify`, `aaj-analysis-qa`, `aaj-label-qa`.
|
|
134
|
+
|
|
135
|
+
## Quickstart — Stage 1: trajectory → analysis.md
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
aaj-generate --run-dir /path/to/your_model__your_suite \
|
|
139
|
+
--concurrency 4 --resume --model claude-opus-4-8
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Expects `<run-dir>/traj/*.json`, one JSON object per trajectory with at least a
|
|
143
|
+
`task_id` field and a log field `traj_tools.py` can recognize. `traj_tools.py` normalizes
|
|
144
|
+
three log formats out of the box — Claude Code stream-JSON, Gemini CLI NDJSON, Codex CLI
|
|
145
|
+
JSONL — so multi-megabyte logs need no truncation; see `traj_tools.detect_format`.
|
|
146
|
+
Writes `<model>/<task_id>/analysis.md` under `./corpus` by default (override with the
|
|
147
|
+
`AAJ_CORPUS_DIR` env var).
|
|
148
|
+
|
|
149
|
+
### The depth exemplar
|
|
150
|
+
|
|
151
|
+
Each session is handed two references: [`ONBOARDING.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/ONBOARDING.md) (the framework — workflow,
|
|
152
|
+
iron rules, required skeleton) and [`analysis_long.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/analysis_long.md) (a worked example
|
|
153
|
+
of the bar being met). The exemplar is a real analysis of a real trajectory, not a
|
|
154
|
+
template: a microkinetics rollout whose headline finding is refuted by a sweep table the
|
|
155
|
+
agent itself printed. It is what "every issue is a paragraph with a mechanism, a
|
|
156
|
+
fair-credit reading and a numeric anchor, plus a `[stage | root cause]` trailer" looks
|
|
157
|
+
like in practice, and it clears every gate of the quality checker:
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
python -c "from autoresearcheval import config; print(config.exemplar())" # where it lives
|
|
161
|
+
aaj-analysis-qa "$(python -c 'from autoresearcheval import config; print(config.exemplar())')" \
|
|
162
|
+
--reason "soft[current_density]"
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
Point `AAJ_EXEMPLAR` at a different file to calibrate against your own corpus instead.
|
|
166
|
+
If neither exists, the prompt drops the exemplar line and falls back to ONBOARDING §3;
|
|
167
|
+
depth then rests on the QA gate alone, so writing one reference analysis by hand for
|
|
168
|
+
your own domain is worth the effort.
|
|
169
|
+
|
|
170
|
+
Note what the exemplar depends on: several of its sharpest findings turn on knowing that
|
|
171
|
+
this run's `WebSearch` was a shim while `WebFetch` was real. That is exactly the fact
|
|
172
|
+
`RETRIEVAL_NOTE` carries into your own runs — get it wrong and the analyst will confidently
|
|
173
|
+
make the opposite mistake.
|
|
174
|
+
|
|
175
|
+
## Quickstart — Stage 2: analysis.md → ARFT classification
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
export ARFT_OPENROUTER_KEY=... # or drop a key in ~/.openrouter_key
|
|
179
|
+
scripts/run_all_arft_api.sh # self-healing: resumes, retries QA failures
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Reads `./corpus/<model>/<task>/analysis.md` (`$AAJ_CORPUS_DIR` — the same default
|
|
183
|
+
Stage 1 writes to, so the two stages compose with no extra flags), writes
|
|
184
|
+
per-analysis `<model>/<task_id>.json` plus the rolled-up stats to `./results`
|
|
185
|
+
(`$AAJ_OUT_DIR`):
|
|
186
|
+
|
|
187
|
+
| Output | What |
|
|
188
|
+
|---|---|
|
|
189
|
+
| `<model>/<task>.json` | Per-analysis classification, evidence-backed |
|
|
190
|
+
| `agg.json` | Dense `[model, task, {code: score}]` grid |
|
|
191
|
+
| `SUMMARY.md` | Pattern × model HIT/PARTIAL matrix, ranked |
|
|
192
|
+
| `root_cause_stats.md` | Lifecycle stage × root-cause pillar breakdown |
|
|
193
|
+
| `matrix_long.csv` | Tidy long-format table everything else derives from |
|
|
194
|
+
| `cooccurrence.csv` / `agreement.csv` | Pattern co-occurrence; cross-model agreement |
|
|
195
|
+
| `tables.tex` | Paper-ready LaTeX |
|
|
196
|
+
| `UNCOVERED.md` | Findings that fit no existing pattern — taxonomy-gap review |
|
|
197
|
+
|
|
198
|
+
Then check the result is trustworthy before you rely on it:
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
aaj-verify # polarity regression + (optionally) a prior-run comparison
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
`aaj-verify` checks that the classifier isn't mistaking exculpatory language for a
|
|
205
|
+
finding (a real failure mode — some diagnostic vocabulary shows up almost entirely in
|
|
206
|
+
*clearing* statements in this kind of writeup) and, if you pass `--pass2` against a
|
|
207
|
+
second independent run, reports per-pattern Cohen's κ so you know which codes are
|
|
208
|
+
reliably distinguishable and which need their guide entry sharpened.
|
|
209
|
+
|
|
210
|
+
## Taxonomy
|
|
211
|
+
|
|
212
|
+
The 45-pattern label space, its four root-cause pillars, and what the 800-trajectory
|
|
213
|
+
audit found are documented in [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md). The code list lives in
|
|
214
|
+
`autoresearcheval/patterns.py` and the classifier's operational guide ships as package
|
|
215
|
+
data (`config.arft_guide()`).
|
|
216
|
+
|
|
217
|
+
## Reasoning budget
|
|
218
|
+
|
|
219
|
+
The reasoning-token budget is the main quality lever on Stage 2 — don't turn it down.
|
|
220
|
+
Disabling reasoning entirely measured **38% recall** against a hand-verified reference
|
|
221
|
+
labelling, missing several genuinely-present patterns; `3000` reasoning tokens
|
|
222
|
+
(the `reasoning_tokens` default) measured **81% recall**.
|
|
223
|
+
|
|
224
|
+
Stage 1 is the heavier stage per item, being open-ended authoring rather than
|
|
225
|
+
extraction. Use `--dry-run` / `--n` to size a pilot before committing to a full run.
|
|
226
|
+
|
|
227
|
+
## License
|
|
228
|
+
|
|
229
|
+
MIT — see `LICENSE`.
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
# autoresearcheval
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/autoresearcheval/)
|
|
4
|
+
|
|
5
|
+
A two-stage pipeline for turning raw AI-agent research trajectories into a
|
|
6
|
+
structured, evidence-grounded failure-taxonomy classification.
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
raw trajectory log --[Stage 1]--> analysis.md --[Stage 2]--> ARFT labels
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from autoresearcheval import generate_analysis, label_arft, pattern_info
|
|
14
|
+
|
|
15
|
+
analysis = generate_analysis("path/to/trajectory_dir") # Stage 1
|
|
16
|
+
result = label_arft(analysis["analysis"], api_key="sk-...") # Stage 2
|
|
17
|
+
|
|
18
|
+
print(result["summary"], result["total_failures"])
|
|
19
|
+
for code in result["failure_modes"]:
|
|
20
|
+
print(code, pattern_info(code)["name"])
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
A trajectory directory (or a single trajectory JSON, or a dict) and an API key are all
|
|
24
|
+
that is required.
|
|
25
|
+
|
|
26
|
+
**Stage 1** spawns one fresh Claude Code session per trajectory to write a deep,
|
|
27
|
+
ONBOARDING-conformant `analysis.md` — a structured, six-stage critique (ideation,
|
|
28
|
+
retrieval & synthesis, execution, analysis, writing, self-review) with a claim-by-claim
|
|
29
|
+
verdict table and independent numerical sanity checks, not a summary.
|
|
30
|
+
|
|
31
|
+
**Stage 2** classifies each `analysis.md` against **ARFT** (the AutoResearch Failure
|
|
32
|
+
Taxonomy: `A.1`–`X.8`, 45 patterns spanning six lifecycle stages plus a cross-cutting
|
|
33
|
+
layer, rolling up to four root-cause pillars — see [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md)) and rolls the
|
|
34
|
+
results into a pattern × model matrix, a root-cause breakdown, co-occurrence stats, and
|
|
35
|
+
cross-model agreement.
|
|
36
|
+
|
|
37
|
+
## Why the judge is artifact-aware
|
|
38
|
+
|
|
39
|
+
Many failures leave no trace in the report — a result the code never produced, a method
|
|
40
|
+
the logs never ran — so catching them means checking the manuscript against the
|
|
41
|
+
artifacts. Stage 1 therefore runs a fresh, zero-history session per trajectory, with
|
|
42
|
+
shell access and no network, handed the full evidence package (task statement, execution
|
|
43
|
+
log, delivered filesystem, the scorer's own source, every scoring call, read-only gold)
|
|
44
|
+
and required to anchor every finding to a line, file, or value.
|
|
45
|
+
|
|
46
|
+
Against three-expert annotation on 50 stratified trajectories this reaches **κ = 0.75**
|
|
47
|
+
(pattern) and **0.83** (root cause), versus 0.53 / 0.62 for a single-call LLM-as-a-judge
|
|
48
|
+
on the transcript alone. Almost all of the gain is recall — which is the point: artifact
|
|
49
|
+
access is what makes transcript-invisible failures detectable.
|
|
50
|
+
|
|
51
|
+
## Install
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pip install autoresearcheval # the library and the CLIs
|
|
55
|
+
pip install 'autoresearcheval[stats]' # adds pandas, needed for the corpus rollups
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
- Python 3.10+. The only hard dependency is `httpx`.
|
|
59
|
+
- **Stage 2** needs an API key for any OpenAI-compatible chat-completions endpoint. Key
|
|
60
|
+
resolution: the `api_key` argument, then `ARFT_OPENROUTER_KEY`, then
|
|
61
|
+
`~/.openrouter_key`, then `OPENROUTER_API_KEY`. Point it elsewhere with `base_url=` or
|
|
62
|
+
the `AAJ_ENDPOINT` env var.
|
|
63
|
+
- **Stage 1**, and Stage 2's alternate executor, spawn headless
|
|
64
|
+
[Claude Code](https://claude.com/product/claude-code) sessions and need the `claude`
|
|
65
|
+
CLI installed and authenticated.
|
|
66
|
+
|
|
67
|
+
## The two calls
|
|
68
|
+
|
|
69
|
+
| | What it does | Cost | Needs |
|
|
70
|
+
|---|---|---|---|
|
|
71
|
+
| `generate_analysis(trajectory, ...)` | reads the trajectory's artifacts and writes a six-stage critique with a claim-by-claim verdict table | minutes | the `claude` CLI |
|
|
72
|
+
| `label_arft(analysis, ...)` | maps that critique onto the 45 ARFT codes | one completion | an API key |
|
|
73
|
+
|
|
74
|
+
They are separate on purpose. Stage 1 is the expensive half, and it produces the
|
|
75
|
+
artifact that makes the labels auditable; folding the two together would hide both.
|
|
76
|
+
|
|
77
|
+
`generate_analysis` returns `{"task_id", "analysis", "qa", "path", "workspace",
|
|
78
|
+
"duration_s", "returncode"}`. `qa["ok"]` is the depth gate described below — False means
|
|
79
|
+
the analysis came back thinner than the framework's bar, not that the call failed.
|
|
80
|
+
|
|
81
|
+
`label_arft` returns the classification plus `failure_modes` (every established code),
|
|
82
|
+
`total_failures`, and `qa` (the schema-and-polarity gate). Each hit carries its `code`,
|
|
83
|
+
`name`, `stage`, `pillar`, `root_cause`, `confidence`, `evidence` and `why`.
|
|
84
|
+
|
|
85
|
+
### Telling the analyst about your harness
|
|
86
|
+
|
|
87
|
+
Two facts change what counts as a finding: whether the harness's retrieval tools did
|
|
88
|
+
real network I/O or were mocked, and whether gold values are reachable locally. By
|
|
89
|
+
default the analyst is told to **work both out from the trajectory** and report what it
|
|
90
|
+
concluded — so the defaults are correct for any harness and nothing needs editing.
|
|
91
|
+
|
|
92
|
+
Override them with `retrieval_note=` / `gold_note=` (or the `RETRIEVAL_NOTE` /
|
|
93
|
+
`GOLD_NOTE` constants for the batch CLI) **only if you can state the truth**. A declared
|
|
94
|
+
fact beats an inferred one, but a wrong one is worse than neither: an analyst told that
|
|
95
|
+
a real search tool is mocked will report fabricated retrieval that never happened, and
|
|
96
|
+
one told that a shim is real will credit calibration against literature that was never
|
|
97
|
+
fetched.
|
|
98
|
+
|
|
99
|
+
## Batch CLIs
|
|
100
|
+
|
|
101
|
+
The commands that produced the paper's corpus install alongside the library:
|
|
102
|
+
`aaj-generate`, `aaj-classify`, `aaj-classify-cc`, `aaj-aggregate`, `aaj-status`,
|
|
103
|
+
`aaj-verify`, `aaj-analysis-qa`, `aaj-label-qa`.
|
|
104
|
+
|
|
105
|
+
## Quickstart — Stage 1: trajectory → analysis.md
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
aaj-generate --run-dir /path/to/your_model__your_suite \
|
|
109
|
+
--concurrency 4 --resume --model claude-opus-4-8
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Expects `<run-dir>/traj/*.json`, one JSON object per trajectory with at least a
|
|
113
|
+
`task_id` field and a log field `traj_tools.py` can recognize. `traj_tools.py` normalizes
|
|
114
|
+
three log formats out of the box — Claude Code stream-JSON, Gemini CLI NDJSON, Codex CLI
|
|
115
|
+
JSONL — so multi-megabyte logs need no truncation; see `traj_tools.detect_format`.
|
|
116
|
+
Writes `<model>/<task_id>/analysis.md` under `./corpus` by default (override with the
|
|
117
|
+
`AAJ_CORPUS_DIR` env var).
|
|
118
|
+
|
|
119
|
+
### The depth exemplar
|
|
120
|
+
|
|
121
|
+
Each session is handed two references: [`ONBOARDING.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/ONBOARDING.md) (the framework — workflow,
|
|
122
|
+
iron rules, required skeleton) and [`analysis_long.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/analysis_long.md) (a worked example
|
|
123
|
+
of the bar being met). The exemplar is a real analysis of a real trajectory, not a
|
|
124
|
+
template: a microkinetics rollout whose headline finding is refuted by a sweep table the
|
|
125
|
+
agent itself printed. It is what "every issue is a paragraph with a mechanism, a
|
|
126
|
+
fair-credit reading and a numeric anchor, plus a `[stage | root cause]` trailer" looks
|
|
127
|
+
like in practice, and it clears every gate of the quality checker:
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
python -c "from autoresearcheval import config; print(config.exemplar())" # where it lives
|
|
131
|
+
aaj-analysis-qa "$(python -c 'from autoresearcheval import config; print(config.exemplar())')" \
|
|
132
|
+
--reason "soft[current_density]"
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Point `AAJ_EXEMPLAR` at a different file to calibrate against your own corpus instead.
|
|
136
|
+
If neither exists, the prompt drops the exemplar line and falls back to ONBOARDING §3;
|
|
137
|
+
depth then rests on the QA gate alone, so writing one reference analysis by hand for
|
|
138
|
+
your own domain is worth the effort.
|
|
139
|
+
|
|
140
|
+
Note what the exemplar depends on: several of its sharpest findings turn on knowing that
|
|
141
|
+
this run's `WebSearch` was a shim while `WebFetch` was real. That is exactly the fact
|
|
142
|
+
`RETRIEVAL_NOTE` carries into your own runs — get it wrong and the analyst will confidently
|
|
143
|
+
make the opposite mistake.
|
|
144
|
+
|
|
145
|
+
## Quickstart — Stage 2: analysis.md → ARFT classification
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
export ARFT_OPENROUTER_KEY=... # or drop a key in ~/.openrouter_key
|
|
149
|
+
scripts/run_all_arft_api.sh # self-healing: resumes, retries QA failures
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Reads `./corpus/<model>/<task>/analysis.md` (`$AAJ_CORPUS_DIR` — the same default
|
|
153
|
+
Stage 1 writes to, so the two stages compose with no extra flags), writes
|
|
154
|
+
per-analysis `<model>/<task_id>.json` plus the rolled-up stats to `./results`
|
|
155
|
+
(`$AAJ_OUT_DIR`):
|
|
156
|
+
|
|
157
|
+
| Output | What |
|
|
158
|
+
|---|---|
|
|
159
|
+
| `<model>/<task>.json` | Per-analysis classification, evidence-backed |
|
|
160
|
+
| `agg.json` | Dense `[model, task, {code: score}]` grid |
|
|
161
|
+
| `SUMMARY.md` | Pattern × model HIT/PARTIAL matrix, ranked |
|
|
162
|
+
| `root_cause_stats.md` | Lifecycle stage × root-cause pillar breakdown |
|
|
163
|
+
| `matrix_long.csv` | Tidy long-format table everything else derives from |
|
|
164
|
+
| `cooccurrence.csv` / `agreement.csv` | Pattern co-occurrence; cross-model agreement |
|
|
165
|
+
| `tables.tex` | Paper-ready LaTeX |
|
|
166
|
+
| `UNCOVERED.md` | Findings that fit no existing pattern — taxonomy-gap review |
|
|
167
|
+
|
|
168
|
+
Then check the result is trustworthy before you rely on it:
|
|
169
|
+
|
|
170
|
+
```bash
|
|
171
|
+
aaj-verify # polarity regression + (optionally) a prior-run comparison
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
`aaj-verify` checks that the classifier isn't mistaking exculpatory language for a
|
|
175
|
+
finding (a real failure mode — some diagnostic vocabulary shows up almost entirely in
|
|
176
|
+
*clearing* statements in this kind of writeup) and, if you pass `--pass2` against a
|
|
177
|
+
second independent run, reports per-pattern Cohen's κ so you know which codes are
|
|
178
|
+
reliably distinguishable and which need their guide entry sharpened.
|
|
179
|
+
|
|
180
|
+
## Taxonomy
|
|
181
|
+
|
|
182
|
+
The 45-pattern label space, its four root-cause pillars, and what the 800-trajectory
|
|
183
|
+
audit found are documented in [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md). The code list lives in
|
|
184
|
+
`autoresearcheval/patterns.py` and the classifier's operational guide ships as package
|
|
185
|
+
data (`config.arft_guide()`).
|
|
186
|
+
|
|
187
|
+
## Reasoning budget
|
|
188
|
+
|
|
189
|
+
The reasoning-token budget is the main quality lever on Stage 2 — don't turn it down.
|
|
190
|
+
Disabling reasoning entirely measured **38% recall** against a hand-verified reference
|
|
191
|
+
labelling, missing several genuinely-present patterns; `3000` reasoning tokens
|
|
192
|
+
(the `reasoning_tokens` default) measured **81% recall**.
|
|
193
|
+
|
|
194
|
+
Stage 1 is the heavier stage per item, being open-ended authoring rather than
|
|
195
|
+
extraction. Use `--dry-run` / `--n` to size a pilot before committing to a full run.
|
|
196
|
+
|
|
197
|
+
## License
|
|
198
|
+
|
|
199
|
+
MIT — see `LICENSE`.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "autoresearcheval"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Process-level failure diagnosis for autonomous research agents: trajectory -> structured analysis -> ARFT failure labels."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "AutoResearchEval authors" }]
|
|
13
|
+
keywords = ["llm", "agents", "evaluation", "failure-taxonomy", "agent-as-a-judge", "arft"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
23
|
+
]
|
|
24
|
+
dependencies = [
|
|
25
|
+
"httpx>=0.27",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
# Only the corpus-wide rollups need a dataframe library; a single label_arft() call
|
|
30
|
+
# does not, so pandas stays out of the default install.
|
|
31
|
+
stats = ["pandas>=2.0"]
|
|
32
|
+
dev = ["pytest>=7", "build", "twine"]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/PrentisAI/AutoResearchEval"
|
|
36
|
+
Paper = "https://arxiv.org/abs/2608.14905"
|
|
37
|
+
Source = "https://github.com/PrentisAI/AutoResearchEval/tree/master/agent-as-a-judge"
|
|
38
|
+
|
|
39
|
+
[project.scripts]
|
|
40
|
+
aaj-generate = "autoresearcheval.generate:main"
|
|
41
|
+
aaj-classify = "autoresearcheval.classify:main"
|
|
42
|
+
aaj-classify-cc = "autoresearcheval.classify_cc:main"
|
|
43
|
+
aaj-aggregate = "autoresearcheval.aggregate:main"
|
|
44
|
+
aaj-status = "autoresearcheval.status:main"
|
|
45
|
+
aaj-verify = "autoresearcheval.verify:main"
|
|
46
|
+
aaj-analysis-qa = "autoresearcheval.analysis_qa:main"
|
|
47
|
+
aaj-label-qa = "autoresearcheval.label_qa:main"
|
|
48
|
+
|
|
49
|
+
[tool.setuptools.packages.find]
|
|
50
|
+
where = ["src"]
|
|
51
|
+
|
|
52
|
+
[tool.setuptools.package-data]
|
|
53
|
+
# The framework, the depth exemplar and the classifier guide are inputs the code reads
|
|
54
|
+
# at runtime and hands to subprocesses by absolute path, so they must ship in the wheel.
|
|
55
|
+
autoresearcheval = ["data/*.md"]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""AutoResearchEval — process-level failure diagnosis for autonomous research agents.
|
|
2
|
+
|
|
3
|
+
Two stages, two calls:
|
|
4
|
+
|
|
5
|
+
from autoresearcheval import generate_analysis, label_arft
|
|
6
|
+
|
|
7
|
+
analysis = generate_analysis(trajectory, retrieval_note=..., gold_note=...)
|
|
8
|
+
result = label_arft(analysis["analysis"], api_key="...")
|
|
9
|
+
|
|
10
|
+
for code in result["failure_modes"]:
|
|
11
|
+
print(code, pattern_info(code)["name"])
|
|
12
|
+
|
|
13
|
+
Stage 1 reads a trajectory's artifacts — execution log, delivered files, the scorer's
|
|
14
|
+
own source — and writes a structured six-stage critique. Stage 2 maps that critique onto
|
|
15
|
+
**ARFT**, the AutoResearch Failure Taxonomy: 45 patterns across six lifecycle stages plus
|
|
16
|
+
a cross-cutting layer, rolling up to four root-cause pillars.
|
|
17
|
+
|
|
18
|
+
Judging against artifacts rather than the transcript alone is the point: on 50 stratified
|
|
19
|
+
trajectories against three-expert annotation this reaches kappa 0.75 (pattern) and 0.83
|
|
20
|
+
(root cause), versus 0.53 / 0.62 for a single-call judge shown only the transcript.
|
|
21
|
+
Almost all of the difference is recall.
|
|
22
|
+
|
|
23
|
+
The batch CLIs that produced the paper's corpus are installed alongside: ``aaj-generate``
|
|
24
|
+
(Stage 1 over a run directory), ``aaj-classify`` (Stage 2 over a corpus), plus
|
|
25
|
+
``aaj-aggregate``, ``aaj-status`` and ``aaj-verify``.
|
|
26
|
+
"""
|
|
27
|
+
from .api import all_patterns, generate_analysis, label_arft, pattern_info
|
|
28
|
+
|
|
29
|
+
__version__ = "0.1.0"
|
|
30
|
+
__all__ = [
|
|
31
|
+
"generate_analysis",
|
|
32
|
+
"label_arft",
|
|
33
|
+
"pattern_info",
|
|
34
|
+
"all_patterns",
|
|
35
|
+
"__version__",
|
|
36
|
+
]
|