pasr-bench 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pasr_bench-0.2.0.dist-info/METADATA +116 -0
- pasr_bench-0.2.0.dist-info/RECORD +17 -0
- pasr_bench-0.2.0.dist-info/WHEEL +4 -0
- pasr_bench-0.2.0.dist-info/entry_points.txt +2 -0
- pasr_eval/RESULTS.md +151 -0
- pasr_eval/__init__.py +34 -0
- pasr_eval/__main__.py +47 -0
- pasr_eval/agents.py +28 -0
- pasr_eval/arms.py +195 -0
- pasr_eval/bakeoff.py +332 -0
- pasr_eval/llm_agent.py +77 -0
- pasr_eval/metrics.py +103 -0
- pasr_eval/plans/pilot.json +71 -0
- pasr_eval/run.py +202 -0
- pasr_eval/runner.py +96 -0
- pasr_eval/spec.py +125 -0
- pasr_eval/validate.py +77 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pasr-bench
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: PASR-Bench: a pre-registered protocol for measuring a code-context retriever against grep, a repo-map, semantic search, and a whole-repo dump.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Apheironn/pasr
|
|
6
|
+
Project-URL: Source, https://github.com/Apheironn/pasr/tree/main/eval
|
|
7
|
+
Author: Apheironn
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
Keywords: benchmark,coding-agent,context,evaluation,non-inferiority,retrieval
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Software Development :: Testing
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: pasr-mcp>=0.2
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest<9,>=8; extra == 'dev'
|
|
19
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
20
|
+
Provides-Extra: llm
|
|
21
|
+
Requires-Dist: anthropic>=0.40; extra == 'llm'
|
|
22
|
+
Provides-Extra: plots
|
|
23
|
+
Requires-Dist: matplotlib>=3.7; extra == 'plots'
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# PASR-Bench
|
|
27
|
+
|
|
28
|
+
A small, **pre-registered** protocol for measuring a code-context retriever, and the
|
|
29
|
+
harness that runs it. Packaged separately from `pasr-mcp` (`pip install
|
|
30
|
+
./eval`, distribution name `pasr-bench`). Results for PASR itself:
|
|
31
|
+
[`RESULTS.md`](RESULTS.md) and [`../docs/competitors-benchmark.md`](../docs/competitors-benchmark.md).
|
|
32
|
+
|
|
33
|
+
## What it measures
|
|
34
|
+
|
|
35
|
+
**Real-agent run** — four arms per task, each producing the same `ArmResult`:
|
|
36
|
+
|
|
37
|
+
1. **native_search** — deterministic grep + read-files (stand-in for an agent's own tools)
|
|
38
|
+
2. **broad** — the whole repo, source-first, truncated to a cap
|
|
39
|
+
3. **pasr** — `select_context` at a fixed budget
|
|
40
|
+
4. **pasr_fallback** — `pasr`, then one budget widening if confidence is low
|
|
41
|
+
|
|
42
|
+
A model answers each task from **only** that arm's context (`UNKNOWN` if absent); a
|
|
43
|
+
second model judges the answer against the task's expected identifiers;
|
|
44
|
+
`critical_source_hit` is still required. Metrics: task success, model input tokens
|
|
45
|
+
(incl. a flat per-tool-call overhead), round trips, critical-source miss rate, fallback
|
|
46
|
+
rate. Then a **paired non-inferiority** test of `pasr` / `pasr_fallback` vs the
|
|
47
|
+
`baseline_arm` at the plan's `margin_task_success` (point estimate + 10k-resample
|
|
48
|
+
bootstrap CI).
|
|
49
|
+
|
|
50
|
+
**Bake-off** — an offline, no-API retrieval comparison at a shared budget: `grep`,
|
|
51
|
+
`repomap` (aider-style signatures), `embed_lex` (a no-setup semantic floor), `pasr`,
|
|
52
|
+
`pasr_hash`, `pasr_map`. Scored on critical-file hit ∧ keyword coverage, split by task
|
|
53
|
+
kind.
|
|
54
|
+
|
|
55
|
+
**No GPU.** Arms run on CPU; the only model use is one answer + one judge call per
|
|
56
|
+
(task, arm) — ~400 Anthropic calls for the 50-task plan. Full Sonnet ≈ $12–15; a cheap
|
|
57
|
+
`--model` with a strong `--judge-model` ≈ $4–6.
|
|
58
|
+
|
|
59
|
+
## Install & run
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install ./eval # the pasr-bench distribution (deps: pasr-mcp)
|
|
63
|
+
pip install "./eval[llm,plots]" # + anthropic (real agent) + matplotlib (report.png)
|
|
64
|
+
|
|
65
|
+
pasr-bench plans # the packaged plan(s)
|
|
66
|
+
pasr-bench run --agent keyword # offline smoke: full matrix + report + validation
|
|
67
|
+
pasr-bench bakeoff --budget 6000 # offline retrieval bake-off
|
|
68
|
+
|
|
69
|
+
export ANTHROPIC_API_KEY=sk-ant-...
|
|
70
|
+
pasr-bench run --agent claude --max-tasks 4 # cheap trial (~$0.4)
|
|
71
|
+
pasr-bench run --agent claude \
|
|
72
|
+
--model claude-haiku-4-5 --judge-model claude-sonnet-5 \
|
|
73
|
+
--checkout-dir .eval-checkouts # budget-safe full run (~$4–6)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Each finished `(task, arm)` row is appended to `matrix.jsonl` and flushed, so a crash
|
|
77
|
+
keeps every completed row; `--resume <run_dir>` reloads the partial matrix and finishes
|
|
78
|
+
into the same delivery. `--checkout-dir DIR` reuses clones. `PASR_EVAL_BROAD_CAP=30000`
|
|
79
|
+
shrinks the `broad` arm. A run writes `<out>/<plan>_<utc>/` with `matrix.jsonl`,
|
|
80
|
+
`report.{json,md,png}`, `validation.json`, `resolved_commits.json`.
|
|
81
|
+
|
|
82
|
+
`python eval/run_eval.py …` and `python eval/bakeoff.py …` still work as thin shims for
|
|
83
|
+
the two sub-commands.
|
|
84
|
+
|
|
85
|
+
## Bring your own retriever
|
|
86
|
+
|
|
87
|
+
The protocol is retriever-agnostic. To measure a different context tool against the
|
|
88
|
+
same 50 tasks and the same baselines:
|
|
89
|
+
|
|
90
|
+
1. Add an arm in [`pasr_eval/arms.py`](pasr_eval/arms.py): extend `ARMS` and add a
|
|
91
|
+
branch in `_build(...)` that returns `(context, sources, tool_calls, round_trips,
|
|
92
|
+
fallback)` for your retriever. Everything downstream — grading, metrics,
|
|
93
|
+
non-inferiority, the validator — is arm-agnostic.
|
|
94
|
+
2. For a bake-off arm, add a function in [`pasr_eval/bakeoff.py`](pasr_eval/bakeoff.py)
|
|
95
|
+
and list it in that file's `ARMS`.
|
|
96
|
+
3. Keep the plan (`pasr_eval/plans/pilot.json`) fixed so numbers stay comparable, or
|
|
97
|
+
register a new plan and cite it.
|
|
98
|
+
|
|
99
|
+
The validator (`validate_matrix`) rejects synthetic rows, query→answer leaks, and an
|
|
100
|
+
unmatched task×arm matrix, so a submitted result is checkable.
|
|
101
|
+
|
|
102
|
+
## Files
|
|
103
|
+
|
|
104
|
+
| Path | What |
|
|
105
|
+
|---|---|
|
|
106
|
+
| `pasr_eval/spec.py` | `RepoSpec` / `TaskSpec` / `EvalPlan`, `load_plan`; the leak + kind guards |
|
|
107
|
+
| `pasr_eval/arms.py` | the four real-agent arms → `ArmResult` |
|
|
108
|
+
| `pasr_eval/bakeoff.py` | the six offline bake-off arms |
|
|
109
|
+
| `pasr_eval/agents.py` | `AgentRunner` protocol, `KeywordAgent` (offline proxy) |
|
|
110
|
+
| `pasr_eval/llm_agent.py` | `LlmAgent` — answer + judge via the Anthropic API (`[llm]` extra) |
|
|
111
|
+
| `pasr_eval/metrics.py` | grade, aggregate, paired bootstrap CI, non-inferiority, `full_report` |
|
|
112
|
+
| `pasr_eval/runner.py` | `resolve_repos`, `run_plan` (`skip=` / `on_row=`), `write_matrix` |
|
|
113
|
+
| `pasr_eval/validate.py` | `validate_matrix` — no synthetic rows, no leaks, matched matrix |
|
|
114
|
+
| `pasr_eval/run.py` | end-to-end orchestrator: clone → streamed matrix → report → validate |
|
|
115
|
+
| `pasr_eval/plans/pilot.json` | the registered plan — 10 pinned repos, 50 tasks |
|
|
116
|
+
| `RESULTS.md` | pre-registration + n=15 pilot + keyword-50 + the n=50 real-agent headline |
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
pasr_eval/__init__.py,sha256=AdMWfjzTeqOuI9_8t2NMNTTpBMk9UgBM-7R8sP0u-Pk,1156
|
|
2
|
+
pasr_eval/__main__.py,sha256=nkasG0AYtek3_6AGmws07vWKKYX-Uun2-5x5mbAoa-g,1578
|
|
3
|
+
pasr_eval/agents.py,sha256=jyqz-BN4SWbIwseztkZw11v0aIXurMVoNaLd-xDuTec,931
|
|
4
|
+
pasr_eval/arms.py,sha256=Q6q87VzvFbxQINc_rTk-to0iNW0Ms4B94Yzae_UwvQI,6857
|
|
5
|
+
pasr_eval/bakeoff.py,sha256=KqzBPCfz6YUJ-Uw-B3evscMm2TfR-yxsbmmWoks6CwM,13197
|
|
6
|
+
pasr_eval/llm_agent.py,sha256=ksxd2l70YK6z5w0L4Igq9lpoq2jt530otsOw-JoAHAQ,2625
|
|
7
|
+
pasr_eval/metrics.py,sha256=e8uF_qGBl99wrNp1Y8nmIYj9r11pvbjEtIUh6snPJ_g,4119
|
|
8
|
+
pasr_eval/run.py,sha256=LJeDhfSfNr0LXsG-19S7kV4kAyq3p3D-uEp2Ppklsp4,7967
|
|
9
|
+
pasr_eval/runner.py,sha256=j_eMA6_ohIpmvr4cVRMLWzFnHJ2iflZhb3rL1ZfgTSo,3676
|
|
10
|
+
pasr_eval/spec.py,sha256=H9bfYlaTwBFN1NRqWy-6r2_sxjTQCoM3syHA_pm6DUE,4362
|
|
11
|
+
pasr_eval/validate.py,sha256=HaikrR4qAzE-94-GG5k01aFPW2x9wRUMi3nSdkESUU0,2963
|
|
12
|
+
pasr_eval/plans/pilot.json,sha256=zOBS7Hac3xj_g86cgwCciE0K8ogCpo9xpFojcOe9IF8,13463
|
|
13
|
+
pasr_eval/RESULTS.md,sha256=UURZqb52gaf72QTVmZe_UlEHTA1YzEqduUJNNAtkXfQ,9299
|
|
14
|
+
pasr_bench-0.2.0.dist-info/METADATA,sha256=djPvb6EK07FtjgfPl-_rKNI5gPXkBC8z5jwLtypB4lw,5889
|
|
15
|
+
pasr_bench-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
16
|
+
pasr_bench-0.2.0.dist-info/entry_points.txt,sha256=llKv_M2edVivOvKiYPqcl_OTRaaXHVG18_9ObB1cBuU,55
|
|
17
|
+
pasr_bench-0.2.0.dist-info/RECORD,,
|
pasr_eval/RESULTS.md
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# PASR real-agent evaluation — results
|
|
2
|
+
|
|
3
|
+
> **50-task real-agent run done** (`eval/deliveries/realagent50_20260902T221528Z/`,
|
|
4
|
+
> 10 pinned repos, Claude answer + judge). PASR gives the model **5.8k targeted
|
|
5
|
+
> tokens in one tool call** and it answers **48%** of the tasks; the **59k-token
|
|
6
|
+
> source-first repo dump answers 38%** — PASR is **+0.10** on paired task success
|
|
7
|
+
> (`pasr_fallback` +0.12), and the **point estimate clears the -0.05
|
|
8
|
+
> non-inferiority margin** (95% CI still crosses it: `pasr` [-0.08, +0.30]).
|
|
9
|
+
> PASR puts the critical file in context **46/50** times vs the dump's **33/50**.
|
|
10
|
+
>
|
|
11
|
+
> Reading: on localized code questions PASR matches — slightly beats — a 10×-larger
|
|
12
|
+
> whole-repo dump, at **~90% fewer input tokens and one round trip**. `native_search`
|
|
13
|
+
> (grep + read 6 files, 23k tokens, 6 round trips) is the raw-success leader at 0.52
|
|
14
|
+
> but misses the critical file 30% of the time. This is a **bounded efficiency
|
|
15
|
+
> result**, not a superiority claim — one more batch of 50 would settle the interval.
|
|
16
|
+
|
|
17
|
+
## Pre-registration
|
|
18
|
+
|
|
19
|
+
- **Plan:** `pasr_eval/plans/pilot.json` — 10 pinned public Python repos, 50 source-grounded
|
|
20
|
+
`locate` / `trace` / `explain` tasks (5/repo). `pasr-bench run` records each repo's
|
|
21
|
+
resolved commit SHA and the answer/judge model IDs into `matrix.jsonl`'s `_meta` line.
|
|
22
|
+
- **Arms:** `native_search`, `broad`, `pasr`, `pasr_fallback`.
|
|
23
|
+
- **Baseline:** `broad`.
|
|
24
|
+
- **Primary claim (non-inferiority):** on paired tasks, `pasr_fallback` task success
|
|
25
|
+
is non-inferior to `broad` within a **-0.05** margin (both the point estimate and
|
|
26
|
+
the lower bound of a 10k-resample paired bootstrap 95% CI ≥ margin).
|
|
27
|
+
- **Secondary:** `pasr` / `pasr_fallback` reduce mean `tokens_in` and `round_trips`
|
|
28
|
+
vs `broad`; `critical_source_miss_rate` reported per arm; `fallback_rate` reported.
|
|
29
|
+
- **No synthetic rows, no reference leak:** enforced by `validate_matrix` (every row
|
|
30
|
+
traces to a registered task + repo; the query never contains the full answer;
|
|
31
|
+
exactly one row per task×arm).
|
|
32
|
+
|
|
33
|
+
## The two graders
|
|
34
|
+
|
|
35
|
+
- **Keyword proxy** (`--agent keyword`, offline, no API): task success = every expected
|
|
36
|
+
identifier appears verbatim in the arm's context **and** the critical file is present.
|
|
37
|
+
A harsh literal proxy for retrieval quality — machinery + token-story check only.
|
|
38
|
+
- **Real agent** (`--agent claude`): a model answers from **only** the arm's context
|
|
39
|
+
(`UNKNOWN` if not present), then a second model judges the answer against the task's
|
|
40
|
+
expected identifiers; `critical_source_hit` is still required for a pass.
|
|
41
|
+
|
|
42
|
+
The n=15 pilot and the 50-task keyword re-run below are the supporting runs; the
|
|
43
|
+
[50-task real-agent run](#50-task-real-agent-run--claude-answer--judge-10-repos---headline)
|
|
44
|
+
is the headline.
|
|
45
|
+
|
|
46
|
+
## Pilot run — `claude-sonnet-5`, 15 tasks, 5 repos
|
|
47
|
+
|
|
48
|
+
| arm | task success | context tokens | round trips | crit. miss | fallback rate |
|
|
49
|
+
|---|---:|---:|---:|---:|---:|
|
|
50
|
+
| broad (58k source-first repo dump) | 0.467 | 58 146 | 1 | 0.067 | – |
|
|
51
|
+
| native_search (grep + 6 files, 4k each) | 0.400 | 22 771 | 6 | 0.267 | – |
|
|
52
|
+
| **pasr** | **0.533** | **5 764** | **1** | 0.133 | – |
|
|
53
|
+
| pasr_fallback | 0.600 | 5 764 | 1 | 0.133 | 0.000 |
|
|
54
|
+
|
|
55
|
+
- **Token savings vs `broad`: pasr / pasr_fallback +90.0%**; native_search +60.4%.
|
|
56
|
+
- **Non-inferiority (`pasr_fallback` vs `broad`, task success, margin -0.05):**
|
|
57
|
+
delta **+0.133**, 95% CI **[-0.20, +0.47]** — **point estimate PASSES**, the CI
|
|
58
|
+
lower bound (-0.20) does **not** clear the margin. `pasr` vs `broad`: delta +0.067,
|
|
59
|
+
CI [-0.33, +0.47], same picture.
|
|
60
|
+
- `pasr` critical-source miss: **2/15** (`req-02`, `star-03`; `star-03` was also
|
|
61
|
+
missed by `broad` → likely a task-spec problem, not retrieval).
|
|
62
|
+
- The `pasr_fallback` threshold (`confidence < 0.5`) **never triggered** — all 15
|
|
63
|
+
PASR selections were confident. `pasr_fallback` ran on the same contexts as `pasr`;
|
|
64
|
+
its +0.067 over `pasr` is answer/judge sampling noise, not a fallback effect.
|
|
65
|
+
|
|
66
|
+
### Reading it
|
|
67
|
+
|
|
68
|
+
- `broad` puts the critical file in context 14/15 times (58k tokens) yet the model
|
|
69
|
+
answers only 7/15 — a large-haystack utilisation effect. `pasr` gives the model
|
|
70
|
+
5.8k targeted tokens and it answers 8/15. **Same answer quality band, one order of
|
|
71
|
+
magnitude fewer tokens, one tool call instead of the model chewing 58k.**
|
|
72
|
+
- `native_search` is worst: brittle grep (27% critical miss) plus six partial files.
|
|
73
|
+
- **What this does not show:** inferential non-inferiority (n=15 CI is ±0.3). The
|
|
74
|
+
bounded positive claim is a cross-repo efficiency direction, mirroring `researchv2`'s
|
|
75
|
+
LongBench Pro finding.
|
|
76
|
+
|
|
77
|
+
Delivery: `eval/deliveries/pilot_20260902T212108Z/`.
|
|
78
|
+
|
|
79
|
+
## 50-task keyword-proxy re-run (no API — `pasr-bench run --agent keyword`)
|
|
80
|
+
|
|
81
|
+
`eval/deliveries/keyword50_20260902T213744Z/`. 10 repos, 50 source-grounded tasks;
|
|
82
|
+
`_FALLBACK_CONFIDENCE` raised to 0.65.
|
|
83
|
+
|
|
84
|
+
| arm | task success | context tokens | round trips | crit. miss | fallback rate |
|
|
85
|
+
|---|---:|---:|---:|---:|---:|
|
|
86
|
+
| broad (58k source-first) | 0.660 | 59 075 | 1 | **0.340** | – |
|
|
87
|
+
| native_search (6 files, 4k) | 0.640 | 22 901 | 6 | 0.300 | – |
|
|
88
|
+
| **pasr** | **0.700** | **5 777** | 1 | **0.080** | – |
|
|
89
|
+
| pasr_fallback | 0.700 | 5 854 | 1.02 | 0.060 | 0.020 |
|
|
90
|
+
|
|
91
|
+
- **PASR beats both baselines on the literal grader** (0.70 vs 0.66 / 0.64) at **+90%
|
|
92
|
+
tokens**, and its critical-source miss (0.08 = 4/50) is **4× lower than broad's
|
|
93
|
+
0.34** — a 60k source-first dump still fails to include the right file for 17/50
|
|
94
|
+
real-repo tasks; PASR's targeted retrieval gets it in 92% of the time.
|
|
95
|
+
- Non-inferiority (`pasr` vs `broad`, keyword grader): delta **+0.04**, 95% CI
|
|
96
|
+
**[-0.14, +0.22]** — point PASSES, CI is now ±0.18 (was ±0.30 at n=15) and just
|
|
97
|
+
misses clearing the -0.05 margin.
|
|
98
|
+
- Fallback engaged on 1/50 (crit-miss 0.08 → 0.06).
|
|
99
|
+
|
|
100
|
+
## 50-task real-agent run — Claude answer + judge, 10 repos ★ headline
|
|
101
|
+
|
|
102
|
+
`eval/deliveries/realagent50_20260902T221528Z/`. Models recorded in `matrix.jsonl`'s
|
|
103
|
+
`_meta`. `validate_matrix` clean (no synthetic rows, no leak, matched 4×50 matrix).
|
|
104
|
+
|
|
105
|
+
| arm | task success | context tokens | tokens_in | round trips | crit. miss | fallback |
|
|
106
|
+
|---|---:|---:|---:|---:|---:|---:|
|
|
107
|
+
| broad (59k source-first repo dump) | 0.380 | 59 075 | 59 115 | 1 | **0.340** | – |
|
|
108
|
+
| native_search (grep + 6 files, 4k each) | **0.520** | 22 901 | 23 181 | 6 | 0.300 | – |
|
|
109
|
+
| **pasr** | 0.480 | **5 777** | **5 817** | **1** | **0.080** | – |
|
|
110
|
+
| pasr_fallback | 0.500 | 5 854 | 5 895 | 1.02 | 0.060 | 0.020 |
|
|
111
|
+
|
|
112
|
+
- **Token savings vs `broad`: pasr +90.2%, pasr_fallback +90.0%**; native_search +60.8%.
|
|
113
|
+
- **Non-inferiority (task success, margin -0.05):** `pasr` delta **+0.100**, 95% CI
|
|
114
|
+
**[-0.08, +0.30]**; `pasr_fallback` delta **+0.120**, CI **[-0.06, +0.30]**. Both
|
|
115
|
+
**point estimates PASS**; both CI lower bounds miss the margin by ≈0.03 (half-width
|
|
116
|
+
±0.18–0.19 at n=50, down from ±0.30 at n=15).
|
|
117
|
+
- **Critical-source hit: `pasr` 46/50 (92%), `pasr_fallback` 47/50** vs **`broad`
|
|
118
|
+
33/50 (66%)**. A 59k source-first dump *still* omits the answer's file for 17/50
|
|
119
|
+
real-repo tasks — concentrated in the large repos (`typer` 5/5 missed, `jinja` 4/5,
|
|
120
|
+
`packaging` 4/5, `anyio` 3/5); those are exactly the repos where `broad` scores
|
|
121
|
+
0–1/5. Where the file *does* fit (`requests`, `pluggy`, `httpx`) `broad` reaches
|
|
122
|
+
4/5. **`broad`'s 0.38 is a truncation + large-haystack failure, not a grading one.**
|
|
123
|
+
- Head-to-head: `pasr` wins **15** tasks `broad` loses, loses **10** `broad` wins
|
|
124
|
+
(net +5 of 50 = the +0.10). `pasr_fallback` widened once (`pkg`-family), turning one
|
|
125
|
+
loss into a win.
|
|
126
|
+
- **PASR's weak spot — `typer` (0/5, both PASR arms).** The slice reaches the right
|
|
127
|
+
file 3/5 but the answering lines aren't in the selected window: `typer` leans on
|
|
128
|
+
re-exports and decorator plumbing that the current chunker + 6k budget don't
|
|
129
|
+
resolve. Drop `typer` and `pasr` is 24/45 = **0.53**. Logged for a chunker follow-up.
|
|
130
|
+
- `native_search` is the raw-success leader (0.52) but pays **4× the tokens, 6 round
|
|
131
|
+
trips**, and a **30% critical-source miss** — brittle when the query terms don't
|
|
132
|
+
literally appear near the answer.
|
|
133
|
+
|
|
134
|
+
### Reading it
|
|
135
|
+
|
|
136
|
+
The claim PASR makes is a **profile trade**, and the run supports it: at parity-ish
|
|
137
|
+
answer quality with a whole-repo dump (−0 to +0.12 depending on arm/margin), PASR
|
|
138
|
+
costs **one order of magnitude fewer input tokens, one tool call instead of the model
|
|
139
|
+
chewing 59k, and a 92% critical-file hit rate with per-line provenance**. It is *not*
|
|
140
|
+
an inferential non-inferiority pass (CI lower bound −0.08) and *not* a raw-accuracy
|
|
141
|
+
win over an agent's own grep. It mirrors `researchv2`'s LongBench Pro finding: a
|
|
142
|
+
bounded cross-repo efficiency direction.
|
|
143
|
+
|
|
144
|
+
### Next
|
|
145
|
+
|
|
146
|
+
1. **Second batch of 50 tasks** to close the CI (projected half-width ≈±0.13 at
|
|
147
|
+
n=100 would clear the −0.05 margin if the point estimate holds).
|
|
148
|
+
2. **Chunker follow-up for re-export/decorator-heavy repos** (`typer`): the pass that
|
|
149
|
+
turns a critical-file hit into an answerable window.
|
|
150
|
+
3. Consider a relevance-ranked `broad` truncation so it isn't a strawman past ~40k —
|
|
151
|
+
though the point of this arm is precisely "what a naive big-context dump gets you".
|
pasr_eval/__init__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Real-agent evaluation harness for the PASR MCP (roadmap M11 / research Track 2).
|
|
2
|
+
|
|
3
|
+
Not shipped with ``pasr-mcp``. Compares four arms per task — the agent's native
|
|
4
|
+
search, full/broad context, PASR selected context, PASR + one controlled fallback —
|
|
5
|
+
and reports paired metrics with a pre-registered non-inferiority margin.
|
|
6
|
+
|
|
7
|
+
The dry-run path (``KeywordAgent`` + local-mode plan) exercises the whole matrix
|
|
8
|
+
offline; the real run swaps in ``LlmAgent`` (answer + judge over the Anthropic API).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from pasr_eval.arms import ARMS, ArmResult, run_arm
|
|
12
|
+
from pasr_eval.metrics import aggregate, bootstrap_ci, grade, non_inferiority, paired_delta
|
|
13
|
+
from pasr_eval.runner import run_plan, write_matrix
|
|
14
|
+
from pasr_eval.spec import EvalPlan, RepoSpec, TaskSpec, load_plan, plan_from_dict
|
|
15
|
+
from pasr_eval.validate import validate_matrix
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"ARMS",
|
|
19
|
+
"ArmResult",
|
|
20
|
+
"run_arm",
|
|
21
|
+
"run_plan",
|
|
22
|
+
"write_matrix",
|
|
23
|
+
"EvalPlan",
|
|
24
|
+
"RepoSpec",
|
|
25
|
+
"TaskSpec",
|
|
26
|
+
"load_plan",
|
|
27
|
+
"plan_from_dict",
|
|
28
|
+
"aggregate",
|
|
29
|
+
"paired_delta",
|
|
30
|
+
"bootstrap_ci",
|
|
31
|
+
"non_inferiority",
|
|
32
|
+
"grade",
|
|
33
|
+
"validate_matrix",
|
|
34
|
+
]
|
pasr_eval/__main__.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""``pasr-bench`` console entry: a thin dispatcher over the two orchestrators.
|
|
2
|
+
|
|
3
|
+
pasr-bench run [args...] # the 4-arm real-agent / keyword evaluation
|
|
4
|
+
pasr-bench bakeoff [args...] # the offline retrieval bake-off
|
|
5
|
+
pasr-bench plans # list the packaged plans
|
|
6
|
+
|
|
7
|
+
Everything after the sub-command is forwarded verbatim, so
|
|
8
|
+
``pasr-bench run --agent claude --resume ...`` works exactly like the old
|
|
9
|
+
``python eval/run_eval.py``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
_USAGE = "usage: pasr-bench {run|bakeoff|plans} [args...]"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def main(argv: list[str] | None = None) -> int:
|
|
21
|
+
args = list(sys.argv[1:] if argv is None else argv)
|
|
22
|
+
if not args or args[0] in ("-h", "--help"):
|
|
23
|
+
print(_USAGE)
|
|
24
|
+
print("\n run end-to-end evaluation -> a timestamped delivery")
|
|
25
|
+
print(" bakeoff offline retrieval bake-off (grep / repo-map / semantic / PASR)")
|
|
26
|
+
print(" plans list the plan files bundled with this package")
|
|
27
|
+
return 0 if args else 2
|
|
28
|
+
|
|
29
|
+
cmd, rest = args[0], args[1:]
|
|
30
|
+
if cmd == "run":
|
|
31
|
+
from pasr_eval.run import main as run_main
|
|
32
|
+
|
|
33
|
+
return run_main(rest)
|
|
34
|
+
if cmd == "bakeoff":
|
|
35
|
+
from pasr_eval.bakeoff import main as bakeoff_main
|
|
36
|
+
|
|
37
|
+
return bakeoff_main(rest)
|
|
38
|
+
if cmd == "plans":
|
|
39
|
+
for path in sorted((Path(__file__).parent / "plans").glob("*.json")):
|
|
40
|
+
print(path)
|
|
41
|
+
return 0
|
|
42
|
+
print(f"unknown sub-command {cmd!r}\n{_USAGE}", file=sys.stderr)
|
|
43
|
+
return 2
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
if __name__ == "__main__":
|
|
47
|
+
raise SystemExit(main())
|
pasr_eval/agents.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Agent adapters. The offline dry-run uses a deterministic keyword agent; the real
|
|
2
|
+
run uses :class:`pasr_eval.llm_agent.LlmAgent` (answer + judge over the Anthropic API)."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from typing import Protocol, runtime_checkable
|
|
7
|
+
|
|
8
|
+
from pasr_eval.spec import TaskSpec
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@runtime_checkable
|
|
12
|
+
class AgentRunner(Protocol):
|
|
13
|
+
name: str
|
|
14
|
+
|
|
15
|
+
def answer(self, task: TaskSpec, context: str) -> str:
|
|
16
|
+
"""Answer ``task`` given only ``context`` (the arm's provided slice)."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class KeywordAgent:
|
|
20
|
+
"""Deterministic stand-in: 'answers' with the answer keywords it can ground in
|
|
21
|
+
the context. No LLM — a retrieval-quality proxy for the offline dry-run."""
|
|
22
|
+
|
|
23
|
+
name = "keyword"
|
|
24
|
+
|
|
25
|
+
def answer(self, task: TaskSpec, context: str) -> str:
|
|
26
|
+
low = context.casefold()
|
|
27
|
+
grounded = [keyword for keyword in task.answer_keywords if keyword.casefold() in low]
|
|
28
|
+
return " ".join(grounded)
|
pasr_eval/arms.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""The four evaluation arms.
|
|
2
|
+
|
|
3
|
+
Every arm produces the same :class:`ArmResult`: what context the agent got, how many
|
|
4
|
+
tokens / tool calls that cost, whether the critical source made it in, and whether the
|
|
5
|
+
agent could then answer. Deterministic.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
from dataclasses import asdict, dataclass, fields
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from pasr.evidence import extract_keywords
|
|
15
|
+
from pasr.file_discovery import discover_workspace_files
|
|
16
|
+
from pasr.schema import validate_select_context_request
|
|
17
|
+
from pasr.select import run_select_context
|
|
18
|
+
from pasr.tokenize import Tokenizer, get_tokenizer
|
|
19
|
+
from pasr_eval.agents import AgentRunner
|
|
20
|
+
from pasr_eval.metrics import grade
|
|
21
|
+
from pasr_eval.spec import TaskSpec
|
|
22
|
+
|
|
23
|
+
ARMS = ("native_search", "broad", "pasr", "pasr_fallback")
|
|
24
|
+
|
|
25
|
+
_TOOL_OVERHEAD_TOKENS = 40 # flat per-tool-interaction cost estimate
|
|
26
|
+
_NATIVE_MAX_FILES = 6
|
|
27
|
+
_NATIVE_FILE_CAP_TOKENS = 4000 # an agent reads a big file in ranges, not whole
|
|
28
|
+
# a realistic large-context baseline; PASR_EVAL_BROAD_CAP lets a tight API budget shrink it
|
|
29
|
+
_BROAD_CAP_TOKENS = int(os.environ.get("PASR_EVAL_BROAD_CAP", "60000"))
|
|
30
|
+
_PASR_BUDGET = 6000
|
|
31
|
+
_FALLBACK_EXTRA = 4000
|
|
32
|
+
_FALLBACK_CONFIDENCE = 0.65 # widen once when PASR is not confident in the slice
|
|
33
|
+
|
|
34
|
+
_TEST_HINTS = ("/test", "test_", "_test.", "/tests/", "conftest")
|
|
35
|
+
_DOC_EXT = (".md", ".rst", ".txt")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _priority(relative_path: str) -> tuple[int, str]:
|
|
39
|
+
"""Source first, then non-test, then non-doc — so budget truncation keeps code."""
|
|
40
|
+
low = relative_path.casefold()
|
|
41
|
+
is_test = any(hint in low for hint in _TEST_HINTS)
|
|
42
|
+
is_doc = low.endswith(_DOC_EXT)
|
|
43
|
+
return (int(is_test) * 2 + int(is_doc), relative_path)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class ArmResult:
|
|
48
|
+
arm: str
|
|
49
|
+
task_id: str
|
|
50
|
+
repo: str
|
|
51
|
+
context_tokens: int
|
|
52
|
+
tokens_in: int
|
|
53
|
+
tool_calls: int
|
|
54
|
+
round_trips: int
|
|
55
|
+
sources_included: tuple[str, ...]
|
|
56
|
+
critical_source_hit: bool
|
|
57
|
+
fallback_triggered: bool
|
|
58
|
+
answer: str
|
|
59
|
+
task_success: bool
|
|
60
|
+
|
|
61
|
+
def to_dict(self) -> dict:
|
|
62
|
+
return {**asdict(self), "sources_included": list(self.sources_included)}
|
|
63
|
+
|
|
64
|
+
@classmethod
|
|
65
|
+
def from_dict(cls, data: dict) -> ArmResult:
|
|
66
|
+
names = {f.name for f in fields(cls)}
|
|
67
|
+
payload = {k: v for k, v in data.items() if k in names}
|
|
68
|
+
payload["sources_included"] = tuple(payload.get("sources_included", ()))
|
|
69
|
+
return cls(**payload)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def run_arm(
|
|
73
|
+
arm: str,
|
|
74
|
+
task: TaskSpec,
|
|
75
|
+
repo_root: Path,
|
|
76
|
+
agent: AgentRunner,
|
|
77
|
+
tokenizer: Tokenizer | None = None,
|
|
78
|
+
) -> ArmResult:
|
|
79
|
+
if arm not in ARMS:
|
|
80
|
+
raise ValueError(f"unknown arm: {arm!r}")
|
|
81
|
+
tok = tokenizer or get_tokenizer()
|
|
82
|
+
context, sources, tool_calls, round_trips, fallback = _build(arm, task, repo_root, tok)
|
|
83
|
+
|
|
84
|
+
context_tokens = tok.count(context)
|
|
85
|
+
hit = not task.critical_source or any(task.critical_source in source for source in sources)
|
|
86
|
+
answer = agent.answer(task, context)
|
|
87
|
+
|
|
88
|
+
judge = getattr(agent, "judge", None)
|
|
89
|
+
success = bool(judge(task, answer, hit)) if callable(judge) else grade(task, answer, _Probe(hit))
|
|
90
|
+
|
|
91
|
+
return ArmResult(
|
|
92
|
+
arm=arm,
|
|
93
|
+
task_id=task.id,
|
|
94
|
+
repo=task.repo,
|
|
95
|
+
context_tokens=context_tokens,
|
|
96
|
+
tokens_in=context_tokens + tool_calls * _TOOL_OVERHEAD_TOKENS,
|
|
97
|
+
tool_calls=tool_calls,
|
|
98
|
+
round_trips=round_trips,
|
|
99
|
+
sources_included=tuple(sorted(sources)),
|
|
100
|
+
critical_source_hit=hit,
|
|
101
|
+
fallback_triggered=fallback,
|
|
102
|
+
answer=answer,
|
|
103
|
+
task_success=success,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class _Probe:
|
|
108
|
+
"""Minimal object grade() needs: just ``critical_source_hit``."""
|
|
109
|
+
|
|
110
|
+
def __init__(self, hit: bool) -> None:
|
|
111
|
+
self.critical_source_hit = hit
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _build(arm, task, repo_root, tok):
|
|
115
|
+
if arm == "native_search":
|
|
116
|
+
return _native_search(task, repo_root, tok)
|
|
117
|
+
if arm == "broad":
|
|
118
|
+
return _broad(repo_root, tok)
|
|
119
|
+
if arm == "pasr":
|
|
120
|
+
context, sources = _pasr(task, repo_root, budget=_PASR_BUDGET)
|
|
121
|
+
return context, sources, 1, 1, False
|
|
122
|
+
# pasr_fallback
|
|
123
|
+
context, sources = _pasr(task, repo_root, budget=_PASR_BUDGET)
|
|
124
|
+
result = _pasr_full(task, repo_root, _PASR_BUDGET)
|
|
125
|
+
if float(result.get("confidence") or 0.0) < _FALLBACK_CONFIDENCE:
|
|
126
|
+
context, sources = _pasr(task, repo_root, budget=_PASR_BUDGET + _FALLBACK_EXTRA)
|
|
127
|
+
return context, sources, 2, 2, True
|
|
128
|
+
return context, sources, 1, 1, False
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _discovered(repo_root: Path) -> list:
|
|
132
|
+
return discover_workspace_files(repo_root, include_patterns=["."])
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _native_search(task, repo_root, tok):
|
|
136
|
+
terms = [t.casefold() for t in extract_keywords(task.query)]
|
|
137
|
+
scored = []
|
|
138
|
+
for record in _discovered(repo_root):
|
|
139
|
+
text = record.path.read_text(encoding="utf-8", errors="replace")
|
|
140
|
+
low = text.casefold()
|
|
141
|
+
score = sum(low.count(term) for term in terms)
|
|
142
|
+
if score:
|
|
143
|
+
scored.append((score, record.relative_path, text))
|
|
144
|
+
scored.sort(key=lambda row: (-row[0], row[1]))
|
|
145
|
+
chosen = scored[:_NATIVE_MAX_FILES] or (
|
|
146
|
+
[(0, r.relative_path, r.path.read_text(encoding="utf-8", errors="replace")) for r in _discovered(repo_root)[:1]]
|
|
147
|
+
)
|
|
148
|
+
parts = []
|
|
149
|
+
for _, rel, text in chosen:
|
|
150
|
+
clipped = tok.decode(tok.encode(text)[:_NATIVE_FILE_CAP_TOKENS])
|
|
151
|
+
parts.append(f"# {rel}\n{clipped}")
|
|
152
|
+
context = "\n\n".join(parts)
|
|
153
|
+
sources = [rel for _, rel, _ in chosen]
|
|
154
|
+
opened = len(chosen)
|
|
155
|
+
return context, sources, opened + 1, opened, False # +1 tool call for the grep itself
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _broad(repo_root, tok):
|
|
159
|
+
records = sorted(_discovered(repo_root), key=lambda r: _priority(r.relative_path))
|
|
160
|
+
parts = []
|
|
161
|
+
used = 0
|
|
162
|
+
sources = []
|
|
163
|
+
for record in records:
|
|
164
|
+
text = record.path.read_text(encoding="utf-8", errors="replace")
|
|
165
|
+
chunk = f"# {record.relative_path}\n{text}"
|
|
166
|
+
cost = tok.count(chunk)
|
|
167
|
+
if used + cost > _BROAD_CAP_TOKENS:
|
|
168
|
+
budget = _BROAD_CAP_TOKENS - used
|
|
169
|
+
parts.append(tok.decode(tok.encode(chunk)[: max(budget, 0)]))
|
|
170
|
+
sources.append(record.relative_path)
|
|
171
|
+
break
|
|
172
|
+
parts.append(chunk)
|
|
173
|
+
sources.append(record.relative_path)
|
|
174
|
+
used += cost
|
|
175
|
+
return "\n\n".join(parts), sources, 1, 1, False
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _pasr_full(task, repo_root, budget):
|
|
179
|
+
request = validate_select_context_request(
|
|
180
|
+
{
|
|
181
|
+
"query": task.query,
|
|
182
|
+
"include": ["."],
|
|
183
|
+
"budget_tokens": budget,
|
|
184
|
+
"max_files": 20000, # real repos have far more than the 100 default
|
|
185
|
+
"max_file_bytes": 400_000,
|
|
186
|
+
},
|
|
187
|
+
workspace_root=repo_root,
|
|
188
|
+
)
|
|
189
|
+
return run_select_context(request, write_receipt_file=False)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _pasr(task, repo_root, budget):
|
|
193
|
+
result = _pasr_full(task, repo_root, budget)
|
|
194
|
+
sources = [span["source"] for span in result["spans"]] or result["sources"]
|
|
195
|
+
return result["context"], sources
|