autoresearcheval 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. autoresearcheval-0.1.0/LICENSE +21 -0
  2. autoresearcheval-0.1.0/PKG-INFO +229 -0
  3. autoresearcheval-0.1.0/README.md +199 -0
  4. autoresearcheval-0.1.0/pyproject.toml +55 -0
  5. autoresearcheval-0.1.0/setup.cfg +4 -0
  6. autoresearcheval-0.1.0/src/autoresearcheval/__init__.py +36 -0
  7. autoresearcheval-0.1.0/src/autoresearcheval/aggregate.py +414 -0
  8. autoresearcheval-0.1.0/src/autoresearcheval/analysis_qa.py +381 -0
  9. autoresearcheval-0.1.0/src/autoresearcheval/api.py +312 -0
  10. autoresearcheval-0.1.0/src/autoresearcheval/classify.py +411 -0
  11. autoresearcheval-0.1.0/src/autoresearcheval/classify_cc.py +371 -0
  12. autoresearcheval-0.1.0/src/autoresearcheval/config.py +69 -0
  13. autoresearcheval-0.1.0/src/autoresearcheval/data/ONBOARDING.md +250 -0
  14. autoresearcheval-0.1.0/src/autoresearcheval/data/analysis_long.md +150 -0
  15. autoresearcheval-0.1.0/src/autoresearcheval/data/arft_guide.md +274 -0
  16. autoresearcheval-0.1.0/src/autoresearcheval/generate.py +501 -0
  17. autoresearcheval-0.1.0/src/autoresearcheval/label_qa.py +200 -0
  18. autoresearcheval-0.1.0/src/autoresearcheval/patterns.py +251 -0
  19. autoresearcheval-0.1.0/src/autoresearcheval/status.py +40 -0
  20. autoresearcheval-0.1.0/src/autoresearcheval/traj_tools.py +956 -0
  21. autoresearcheval-0.1.0/src/autoresearcheval/verify.py +190 -0
  22. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/PKG-INFO +229 -0
  23. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/SOURCES.txt +26 -0
  24. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/dependency_links.txt +1 -0
  25. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/entry_points.txt +9 -0
  26. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/requires.txt +9 -0
  27. autoresearcheval-0.1.0/src/autoresearcheval.egg-info/top_level.txt +1 -0
  28. autoresearcheval-0.1.0/tests/test_package.py +186 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Titan Research Labs
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,229 @@
1
+ Metadata-Version: 2.4
2
+ Name: autoresearcheval
3
+ Version: 0.1.0
4
+ Summary: Process-level failure diagnosis for autonomous research agents: trajectory -> structured analysis -> ARFT failure labels.
5
+ Author: AutoResearchEval authors
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/PrentisAI/AutoResearchEval
8
+ Project-URL: Paper, https://arxiv.org/abs/2608.14905
9
+ Project-URL: Source, https://github.com/PrentisAI/AutoResearchEval/tree/master/agent-as-a-judge
10
+ Keywords: llm,agents,evaluation,failure-taxonomy,agent-as-a-judge,arft
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: httpx>=0.27
23
+ Provides-Extra: stats
24
+ Requires-Dist: pandas>=2.0; extra == "stats"
25
+ Provides-Extra: dev
26
+ Requires-Dist: pytest>=7; extra == "dev"
27
+ Requires-Dist: build; extra == "dev"
28
+ Requires-Dist: twine; extra == "dev"
29
+ Dynamic: license-file
30
+
31
+ # autoresearcheval
32
+
33
+ [![PyPI](https://img.shields.io/pypi/v/autoresearcheval)](https://pypi.org/project/autoresearcheval/)
34
+
35
+ A two-stage pipeline for turning raw AI-agent research trajectories into a
36
+ structured, evidence-grounded failure-taxonomy classification.
37
+
38
+ ```
39
+ raw trajectory log --[Stage 1]--> analysis.md --[Stage 2]--> ARFT labels
40
+ ```
41
+
42
+ ```python
43
+ from autoresearcheval import generate_analysis, label_arft, pattern_info
44
+
45
+ analysis = generate_analysis("path/to/trajectory_dir") # Stage 1
46
+ result = label_arft(analysis["analysis"], api_key="sk-...") # Stage 2
47
+
48
+ print(result["summary"], result["total_failures"])
49
+ for code in result["failure_modes"]:
50
+ print(code, pattern_info(code)["name"])
51
+ ```
52
+
53
+ A trajectory directory (or a single trajectory JSON, or a dict) and an API key are all
54
+ that is required.
55
+
56
+ **Stage 1** spawns one fresh Claude Code session per trajectory to write a deep,
57
+ ONBOARDING-conformant `analysis.md` — a structured, six-stage critique (ideation,
58
+ retrieval & synthesis, execution, analysis, writing, self-review) with a claim-by-claim
59
+ verdict table and independent numerical sanity checks, not a summary.
60
+
61
+ **Stage 2** classifies each `analysis.md` against **ARFT** (the AutoResearch Failure
62
+ Taxonomy: `A.1`–`X.8`, 45 patterns spanning six lifecycle stages plus a cross-cutting
63
+ layer, rolling up to four root-cause pillars — see [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md)) and rolls the
64
+ results into a pattern × model matrix, a root-cause breakdown, co-occurrence stats, and
65
+ cross-model agreement.
66
+
67
+ ## Why the judge is artifact-aware
68
+
69
+ Many failures leave no trace in the report — a result the code never produced, a method
70
+ the logs never ran — so catching them means checking the manuscript against the
71
+ artifacts. Stage 1 therefore runs a fresh, zero-history session per trajectory, with
72
+ shell access and no network, handed the full evidence package (task statement, execution
73
+ log, delivered filesystem, the scorer's own source, every scoring call, read-only gold)
74
+ and required to anchor every finding to a line, file, or value.
75
+
76
+ Against three-expert annotation on 50 stratified trajectories this reaches **κ = 0.75**
77
+ (pattern) and **0.83** (root cause), versus 0.53 / 0.62 for a single-call LLM-as-a-judge
78
+ on the transcript alone. Almost all of the gain is recall — which is the point: artifact
79
+ access is what makes transcript-invisible failures detectable.
80
+
81
+ ## Install
82
+
83
+ ```bash
84
+ pip install autoresearcheval # the library and the CLIs
85
+ pip install 'autoresearcheval[stats]' # adds pandas, needed for the corpus rollups
86
+ ```
87
+
88
+ - Python 3.10+. The only hard dependency is `httpx`.
89
+ - **Stage 2** needs an API key for any OpenAI-compatible chat-completions endpoint. Key
90
+ resolution: the `api_key` argument, then `ARFT_OPENROUTER_KEY`, then
91
+ `~/.openrouter_key`, then `OPENROUTER_API_KEY`. Point it elsewhere with `base_url=` or
92
+ the `AAJ_ENDPOINT` env var.
93
+ - **Stage 1**, and Stage 2's alternate executor, spawn headless
94
+ [Claude Code](https://claude.com/product/claude-code) sessions and need the `claude`
95
+ CLI installed and authenticated.
96
+
97
+ ## The two calls
98
+
99
+ | | What it does | Cost | Needs |
100
+ |---|---|---|---|
101
+ | `generate_analysis(trajectory, ...)` | reads the trajectory's artifacts and writes a six-stage critique with a claim-by-claim verdict table | minutes | the `claude` CLI |
102
+ | `label_arft(analysis, ...)` | maps that critique onto the 45 ARFT codes | one completion | an API key |
103
+
104
+ They are separate on purpose. Stage 1 is the expensive half, and it produces the
105
+ artifact that makes the labels auditable; folding the two together would hide both.
106
+
107
+ `generate_analysis` returns `{"task_id", "analysis", "qa", "path", "workspace",
108
+ "duration_s", "returncode"}`. `qa["ok"]` is the depth gate described below — False means
109
+ the analysis came back thinner than the framework's bar, not that the call failed.
110
+
111
+ `label_arft` returns the classification plus `failure_modes` (every established code),
112
+ `total_failures`, and `qa` (the schema-and-polarity gate). Each hit carries its `code`,
113
+ `name`, `stage`, `pillar`, `root_cause`, `confidence`, `evidence` and `why`.
114
+
115
+ ### Telling the analyst about your harness
116
+
117
+ Two facts change what counts as a finding: whether the harness's retrieval tools did
118
+ real network I/O or were mocked, and whether gold values are reachable locally. By
119
+ default the analyst is told to **work both out from the trajectory** and report what it
120
+ concluded — so the defaults are correct for any harness and nothing needs editing.
121
+
122
+ Override them with `retrieval_note=` / `gold_note=` (or the `RETRIEVAL_NOTE` /
123
+ `GOLD_NOTE` constants for the batch CLI) **only if you can state the truth**. A declared
124
+ fact beats an inferred one, but a wrong one is worse than neither: an analyst told that
125
+ a real search tool is mocked will report fabricated retrieval that never happened, and
126
+ one told that a shim is real will credit calibration against literature that was never
127
+ fetched.
128
+
129
+ ## Batch CLIs
130
+
131
+ The commands that produced the paper's corpus install alongside the library:
132
+ `aaj-generate`, `aaj-classify`, `aaj-classify-cc`, `aaj-aggregate`, `aaj-status`,
133
+ `aaj-verify`, `aaj-analysis-qa`, `aaj-label-qa`.
134
+
135
+ ## Quickstart — Stage 1: trajectory → analysis.md
136
+
137
+ ```bash
138
+ aaj-generate --run-dir /path/to/your_model__your_suite \
139
+ --concurrency 4 --resume --model claude-opus-4-8
140
+ ```
141
+
142
+ Expects `<run-dir>/traj/*.json`, one JSON object per trajectory with at least a
143
+ `task_id` field and a log field `traj_tools.py` can recognize. `traj_tools.py` normalizes
144
+ three log formats out of the box — Claude Code stream-JSON, Gemini CLI NDJSON, Codex CLI
145
+ JSONL — so multi-megabyte logs need no truncation; see `traj_tools.detect_format`.
146
+ Writes `<model>/<task_id>/analysis.md` under `./corpus` by default (override with the
147
+ `AAJ_CORPUS_DIR` env var).
148
+
149
+ ### The depth exemplar
150
+
151
+ Each session is handed two references: [`ONBOARDING.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/ONBOARDING.md) (the framework — workflow,
152
+ iron rules, required skeleton) and [`analysis_long.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/analysis_long.md) (a worked example
153
+ of the bar being met). The exemplar is a real analysis of a real trajectory, not a
154
+ template: a microkinetics rollout whose headline finding is refuted by a sweep table the
155
+ agent itself printed. It is what "every issue is a paragraph with a mechanism, a
156
+ fair-credit reading and a numeric anchor, plus a `[stage | root cause]` trailer" looks
157
+ like in practice, and it clears every gate of the quality checker:
158
+
159
+ ```bash
160
+ python -c "from autoresearcheval import config; print(config.exemplar())" # where it lives
161
+ aaj-analysis-qa "$(python -c 'from autoresearcheval import config; print(config.exemplar())')" \
162
+ --reason "soft[current_density]"
163
+ ```
164
+
165
+ Point `AAJ_EXEMPLAR` at a different file to calibrate against your own corpus instead.
166
+ If neither exists, the prompt drops the exemplar line and falls back to ONBOARDING §3;
167
+ depth then rests on the QA gate alone, so writing one reference analysis by hand for
168
+ your own domain is worth the effort.
169
+
170
+ Note what the exemplar depends on: several of its sharpest findings turn on knowing that
171
+ this run's `WebSearch` was a shim while `WebFetch` was real. That is exactly the fact
172
+ `RETRIEVAL_NOTE` carries into your own runs — get it wrong and the analyst will confidently
173
+ make the opposite mistake.
174
+
175
+ ## Quickstart — Stage 2: analysis.md → ARFT classification
176
+
177
+ ```bash
178
+ export ARFT_OPENROUTER_KEY=... # or drop a key in ~/.openrouter_key
179
+ scripts/run_all_arft_api.sh # self-healing: resumes, retries QA failures
180
+ ```
181
+
182
+ Reads `./corpus/<model>/<task>/analysis.md` (`$AAJ_CORPUS_DIR` — the same default
183
+ Stage 1 writes to, so the two stages compose with no extra flags), writes
184
+ per-analysis `<model>/<task_id>.json` plus the rolled-up stats to `./results`
185
+ (`$AAJ_OUT_DIR`):
186
+
187
+ | Output | What |
188
+ |---|---|
189
+ | `<model>/<task>.json` | Per-analysis classification, evidence-backed |
190
+ | `agg.json` | Dense `[model, task, {code: score}]` grid |
191
+ | `SUMMARY.md` | Pattern × model HIT/PARTIAL matrix, ranked |
192
+ | `root_cause_stats.md` | Lifecycle stage × root-cause pillar breakdown |
193
+ | `matrix_long.csv` | Tidy long-format table everything else derives from |
194
+ | `cooccurrence.csv` / `agreement.csv` | Pattern co-occurrence; cross-model agreement |
195
+ | `tables.tex` | Paper-ready LaTeX |
196
+ | `UNCOVERED.md` | Findings that fit no existing pattern — taxonomy-gap review |
197
+
198
+ Then check the result is trustworthy before you rely on it:
199
+
200
+ ```bash
201
+ aaj-verify # polarity regression + (optionally) a prior-run comparison
202
+ ```
203
+
204
+ `aaj-verify` checks that the classifier isn't mistaking exculpatory language for a
205
+ finding (a real failure mode — some diagnostic vocabulary shows up almost entirely in
206
+ *clearing* statements in this kind of writeup) and, if you pass `--pass2` against a
207
+ second independent run, reports per-pattern Cohen's κ so you know which codes are
208
+ reliably distinguishable and which need their guide entry sharpened.
209
+
210
+ ## Taxonomy
211
+
212
+ The 45-pattern label space, its four root-cause pillars, and what the 800-trajectory
213
+ audit found are documented in [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md). The code list lives in
214
+ `autoresearcheval/patterns.py` and the classifier's operational guide ships as package
215
+ data (`config.arft_guide()`).
216
+
217
+ ## Reasoning budget
218
+
219
+ The reasoning-token budget is the main quality lever on Stage 2 — don't turn it down.
220
+ Disabling reasoning entirely measured **38% recall** against a hand-verified reference
221
+ labelling, missing several genuinely-present patterns; `3000` reasoning tokens
222
+ (the `reasoning_tokens` default) measured **81% recall**.
223
+
224
+ Stage 1 is the heavier stage per item, being open-ended authoring rather than
225
+ extraction. Use `--dry-run` / `--n` to size a pilot before committing to a full run.
226
+
227
+ ## License
228
+
229
+ MIT — see `LICENSE`.
@@ -0,0 +1,199 @@
1
+ # autoresearcheval
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/autoresearcheval)](https://pypi.org/project/autoresearcheval/)
4
+
5
+ A two-stage pipeline for turning raw AI-agent research trajectories into a
6
+ structured, evidence-grounded failure-taxonomy classification.
7
+
8
+ ```
9
+ raw trajectory log --[Stage 1]--> analysis.md --[Stage 2]--> ARFT labels
10
+ ```
11
+
12
+ ```python
13
+ from autoresearcheval import generate_analysis, label_arft, pattern_info
14
+
15
+ analysis = generate_analysis("path/to/trajectory_dir") # Stage 1
16
+ result = label_arft(analysis["analysis"], api_key="sk-...") # Stage 2
17
+
18
+ print(result["summary"], result["total_failures"])
19
+ for code in result["failure_modes"]:
20
+ print(code, pattern_info(code)["name"])
21
+ ```
22
+
23
+ A trajectory directory (or a single trajectory JSON, or a dict) and an API key are all
24
+ that is required.
25
+
26
+ **Stage 1** spawns one fresh Claude Code session per trajectory to write a deep,
27
+ ONBOARDING-conformant `analysis.md` — a structured, six-stage critique (ideation,
28
+ retrieval & synthesis, execution, analysis, writing, self-review) with a claim-by-claim
29
+ verdict table and independent numerical sanity checks, not a summary.
30
+
31
+ **Stage 2** classifies each `analysis.md` against **ARFT** (the AutoResearch Failure
32
+ Taxonomy: `A.1`–`X.8`, 45 patterns spanning six lifecycle stages plus a cross-cutting
33
+ layer, rolling up to four root-cause pillars — see [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md)) and rolls the
34
+ results into a pattern × model matrix, a root-cause breakdown, co-occurrence stats, and
35
+ cross-model agreement.
36
+
37
+ ## Why the judge is artifact-aware
38
+
39
+ Many failures leave no trace in the report — a result the code never produced, a method
40
+ the logs never ran — so catching them means checking the manuscript against the
41
+ artifacts. Stage 1 therefore runs a fresh, zero-history session per trajectory, with
42
+ shell access and no network, handed the full evidence package (task statement, execution
43
+ log, delivered filesystem, the scorer's own source, every scoring call, read-only gold)
44
+ and required to anchor every finding to a line, file, or value.
45
+
46
+ Against three-expert annotation on 50 stratified trajectories this reaches **κ = 0.75**
47
+ (pattern) and **0.83** (root cause), versus 0.53 / 0.62 for a single-call LLM-as-a-judge
48
+ on the transcript alone. Almost all of the gain is recall — which is the point: artifact
49
+ access is what makes transcript-invisible failures detectable.
50
+
51
+ ## Install
52
+
53
+ ```bash
54
+ pip install autoresearcheval # the library and the CLIs
55
+ pip install 'autoresearcheval[stats]' # adds pandas, needed for the corpus rollups
56
+ ```
57
+
58
+ - Python 3.10+. The only hard dependency is `httpx`.
59
+ - **Stage 2** needs an API key for any OpenAI-compatible chat-completions endpoint. Key
60
+ resolution: the `api_key` argument, then `ARFT_OPENROUTER_KEY`, then
61
+ `~/.openrouter_key`, then `OPENROUTER_API_KEY`. Point it elsewhere with `base_url=` or
62
+ the `AAJ_ENDPOINT` env var.
63
+ - **Stage 1**, and Stage 2's alternate executor, spawn headless
64
+ [Claude Code](https://claude.com/product/claude-code) sessions and need the `claude`
65
+ CLI installed and authenticated.
66
+
67
+ ## The two calls
68
+
69
+ | | What it does | Cost | Needs |
70
+ |---|---|---|---|
71
+ | `generate_analysis(trajectory, ...)` | reads the trajectory's artifacts and writes a six-stage critique with a claim-by-claim verdict table | minutes | the `claude` CLI |
72
+ | `label_arft(analysis, ...)` | maps that critique onto the 45 ARFT codes | one completion | an API key |
73
+
74
+ They are separate on purpose. Stage 1 is the expensive half, and it produces the
75
+ artifact that makes the labels auditable; folding the two together would hide both.
76
+
77
+ `generate_analysis` returns `{"task_id", "analysis", "qa", "path", "workspace",
78
+ "duration_s", "returncode"}`. `qa["ok"]` is the depth gate described below — False means
79
+ the analysis came back thinner than the framework's bar, not that the call failed.
80
+
81
+ `label_arft` returns the classification plus `failure_modes` (every established code),
82
+ `total_failures`, and `qa` (the schema-and-polarity gate). Each hit carries its `code`,
83
+ `name`, `stage`, `pillar`, `root_cause`, `confidence`, `evidence` and `why`.
84
+
85
+ ### Telling the analyst about your harness
86
+
87
+ Two facts change what counts as a finding: whether the harness's retrieval tools did
88
+ real network I/O or were mocked, and whether gold values are reachable locally. By
89
+ default the analyst is told to **work both out from the trajectory** and report what it
90
+ concluded — so the defaults are correct for any harness and nothing needs editing.
91
+
92
+ Override them with `retrieval_note=` / `gold_note=` (or the `RETRIEVAL_NOTE` /
93
+ `GOLD_NOTE` constants for the batch CLI) **only if you can state the truth**. A declared
94
+ fact beats an inferred one, but a wrong one is worse than neither: an analyst told that
95
+ a real search tool is mocked will report fabricated retrieval that never happened, and
96
+ one told that a shim is real will credit calibration against literature that was never
97
+ fetched.
98
+
99
+ ## Batch CLIs
100
+
101
+ The commands that produced the paper's corpus install alongside the library:
102
+ `aaj-generate`, `aaj-classify`, `aaj-classify-cc`, `aaj-aggregate`, `aaj-status`,
103
+ `aaj-verify`, `aaj-analysis-qa`, `aaj-label-qa`.
104
+
105
+ ## Quickstart — Stage 1: trajectory → analysis.md
106
+
107
+ ```bash
108
+ aaj-generate --run-dir /path/to/your_model__your_suite \
109
+ --concurrency 4 --resume --model claude-opus-4-8
110
+ ```
111
+
112
+ Expects `<run-dir>/traj/*.json`, one JSON object per trajectory with at least a
113
+ `task_id` field and a log field `traj_tools.py` can recognize. `traj_tools.py` normalizes
114
+ three log formats out of the box — Claude Code stream-JSON, Gemini CLI NDJSON, Codex CLI
115
+ JSONL — so multi-megabyte logs need no truncation; see `traj_tools.detect_format`.
116
+ Writes `<model>/<task_id>/analysis.md` under `./corpus` by default (override with the
117
+ `AAJ_CORPUS_DIR` env var).
118
+
119
+ ### The depth exemplar
120
+
121
+ Each session is handed two references: [`ONBOARDING.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/ONBOARDING.md) (the framework — workflow,
122
+ iron rules, required skeleton) and [`analysis_long.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/src/autoresearcheval/data/analysis_long.md) (a worked example
123
+ of the bar being met). The exemplar is a real analysis of a real trajectory, not a
124
+ template: a microkinetics rollout whose headline finding is refuted by a sweep table the
125
+ agent itself printed. It is what "every issue is a paragraph with a mechanism, a
126
+ fair-credit reading and a numeric anchor, plus a `[stage | root cause]` trailer" looks
127
+ like in practice, and it clears every gate of the quality checker:
128
+
129
+ ```bash
130
+ python -c "from autoresearcheval import config; print(config.exemplar())" # where it lives
131
+ aaj-analysis-qa "$(python -c 'from autoresearcheval import config; print(config.exemplar())')" \
132
+ --reason "soft[current_density]"
133
+ ```
134
+
135
+ Point `AAJ_EXEMPLAR` at a different file to calibrate against your own corpus instead.
136
+ If neither exists, the prompt drops the exemplar line and falls back to ONBOARDING §3;
137
+ depth then rests on the QA gate alone, so writing one reference analysis by hand for
138
+ your own domain is worth the effort.
139
+
140
+ Note what the exemplar depends on: several of its sharpest findings turn on knowing that
141
+ this run's `WebSearch` was a shim while `WebFetch` was real. That is exactly the fact
142
+ `RETRIEVAL_NOTE` carries into your own runs — get it wrong and the analyst will confidently
143
+ make the opposite mistake.
144
+
145
+ ## Quickstart — Stage 2: analysis.md → ARFT classification
146
+
147
+ ```bash
148
+ export ARFT_OPENROUTER_KEY=... # or drop a key in ~/.openrouter_key
149
+ scripts/run_all_arft_api.sh # self-healing: resumes, retries QA failures
150
+ ```
151
+
152
+ Reads `./corpus/<model>/<task>/analysis.md` (`$AAJ_CORPUS_DIR` — the same default
153
+ Stage 1 writes to, so the two stages compose with no extra flags), writes
154
+ per-analysis `<model>/<task_id>.json` plus the rolled-up stats to `./results`
155
+ (`$AAJ_OUT_DIR`):
156
+
157
+ | Output | What |
158
+ |---|---|
159
+ | `<model>/<task>.json` | Per-analysis classification, evidence-backed |
160
+ | `agg.json` | Dense `[model, task, {code: score}]` grid |
161
+ | `SUMMARY.md` | Pattern × model HIT/PARTIAL matrix, ranked |
162
+ | `root_cause_stats.md` | Lifecycle stage × root-cause pillar breakdown |
163
+ | `matrix_long.csv` | Tidy long-format table everything else derives from |
164
+ | `cooccurrence.csv` / `agreement.csv` | Pattern co-occurrence; cross-model agreement |
165
+ | `tables.tex` | Paper-ready LaTeX |
166
+ | `UNCOVERED.md` | Findings that fit no existing pattern — taxonomy-gap review |
167
+
168
+ Then check the result is trustworthy before you rely on it:
169
+
170
+ ```bash
171
+ aaj-verify # polarity regression + (optionally) a prior-run comparison
172
+ ```
173
+
174
+ `aaj-verify` checks that the classifier isn't mistaking exculpatory language for a
175
+ finding (a real failure mode — some diagnostic vocabulary shows up almost entirely in
176
+ *clearing* statements in this kind of writeup) and, if you pass `--pass2` against a
177
+ second independent run, reports per-pattern Cohen's κ so you know which codes are
178
+ reliably distinguishable and which need their guide entry sharpened.
179
+
180
+ ## Taxonomy
181
+
182
+ The 45-pattern label space, its four root-cause pillars, and what the 800-trajectory
183
+ audit found are documented in [`ARFT.md`](https://github.com/PrentisAI/AutoResearchEval/blob/master/agent-as-a-judge/ARFT.md). The code list lives in
184
+ `autoresearcheval/patterns.py` and the classifier's operational guide ships as package
185
+ data (`config.arft_guide()`).
186
+
187
+ ## Reasoning budget
188
+
189
+ The reasoning-token budget is the main quality lever on Stage 2 — don't turn it down.
190
+ Disabling reasoning entirely measured **38% recall** against a hand-verified reference
191
+ labelling, missing several genuinely-present patterns; `3000` reasoning tokens
192
+ (the `reasoning_tokens` default) measured **81% recall**.
193
+
194
+ Stage 1 is the heavier stage per item, being open-ended authoring rather than
195
+ extraction. Use `--dry-run` / `--n` to size a pilot before committing to a full run.
196
+
197
+ ## License
198
+
199
+ MIT — see `LICENSE`.
@@ -0,0 +1,55 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "autoresearcheval"
7
+ version = "0.1.0"
8
+ description = "Process-level failure diagnosis for autonomous research agents: trajectory -> structured analysis -> ARFT failure labels."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "AutoResearchEval authors" }]
13
+ keywords = ["llm", "agents", "evaluation", "failure-taxonomy", "agent-as-a-judge", "arft"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Science/Research",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
23
+ ]
24
+ dependencies = [
25
+ "httpx>=0.27",
26
+ ]
27
+
28
+ [project.optional-dependencies]
29
+ # Only the corpus-wide rollups need a dataframe library; a single label_arft() call
30
+ # does not, so pandas stays out of the default install.
31
+ stats = ["pandas>=2.0"]
32
+ dev = ["pytest>=7", "build", "twine"]
33
+
34
+ [project.urls]
35
+ Homepage = "https://github.com/PrentisAI/AutoResearchEval"
36
+ Paper = "https://arxiv.org/abs/2608.14905"
37
+ Source = "https://github.com/PrentisAI/AutoResearchEval/tree/master/agent-as-a-judge"
38
+
39
+ [project.scripts]
40
+ aaj-generate = "autoresearcheval.generate:main"
41
+ aaj-classify = "autoresearcheval.classify:main"
42
+ aaj-classify-cc = "autoresearcheval.classify_cc:main"
43
+ aaj-aggregate = "autoresearcheval.aggregate:main"
44
+ aaj-status = "autoresearcheval.status:main"
45
+ aaj-verify = "autoresearcheval.verify:main"
46
+ aaj-analysis-qa = "autoresearcheval.analysis_qa:main"
47
+ aaj-label-qa = "autoresearcheval.label_qa:main"
48
+
49
+ [tool.setuptools.packages.find]
50
+ where = ["src"]
51
+
52
+ [tool.setuptools.package-data]
53
+ # The framework, the depth exemplar and the classifier guide are inputs the code reads
54
+ # at runtime and hands to subprocesses by absolute path, so they must ship in the wheel.
55
+ autoresearcheval = ["data/*.md"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,36 @@
1
+ """AutoResearchEval — process-level failure diagnosis for autonomous research agents.
2
+
3
+ Two stages, two calls:
4
+
5
+ from autoresearcheval import generate_analysis, label_arft
6
+
7
+ analysis = generate_analysis(trajectory, retrieval_note=..., gold_note=...)
8
+ result = label_arft(analysis["analysis"], api_key="...")
9
+
10
+ for code in result["failure_modes"]:
11
+ print(code, pattern_info(code)["name"])
12
+
13
+ Stage 1 reads a trajectory's artifacts — execution log, delivered files, the scorer's
14
+ own source — and writes a structured six-stage critique. Stage 2 maps that critique onto
15
+ **ARFT**, the AutoResearch Failure Taxonomy: 45 patterns across six lifecycle stages plus
16
+ a cross-cutting layer, rolling up to four root-cause pillars.
17
+
18
+ Judging against artifacts rather than the transcript alone is the point: on 50 stratified
19
+ trajectories against three-expert annotation this reaches kappa 0.75 (pattern) and 0.83
20
+ (root cause), versus 0.53 / 0.62 for a single-call judge shown only the transcript.
21
+ Almost all of the difference is recall.
22
+
23
+ The batch CLIs that produced the paper's corpus are installed alongside: ``aaj-generate``
24
+ (Stage 1 over a run directory), ``aaj-classify`` (Stage 2 over a corpus), plus
25
+ ``aaj-aggregate``, ``aaj-status`` and ``aaj-verify``.
26
+ """
27
+ from .api import all_patterns, generate_analysis, label_arft, pattern_info
28
+
29
+ __version__ = "0.1.0"
30
+ __all__ = [
31
+ "generate_analysis",
32
+ "label_arft",
33
+ "pattern_info",
34
+ "all_patterns",
35
+ "__version__",
36
+ ]