evalwarden 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. evalwarden-0.6.0/LICENSE +21 -0
  2. evalwarden-0.6.0/PKG-INFO +252 -0
  3. evalwarden-0.6.0/README.md +235 -0
  4. evalwarden-0.6.0/pyproject.toml +33 -0
  5. evalwarden-0.6.0/setup.cfg +4 -0
  6. evalwarden-0.6.0/src/evalwarden/__init__.py +3 -0
  7. evalwarden-0.6.0/src/evalwarden/adapters/__init__.py +70 -0
  8. evalwarden-0.6.0/src/evalwarden/adapters/inspect_ai.py +213 -0
  9. evalwarden-0.6.0/src/evalwarden/adapters/promptfoo.py +357 -0
  10. evalwarden-0.6.0/src/evalwarden/checks/__init__.py +27 -0
  11. evalwarden-0.6.0/src/evalwarden/checks/base.py +35 -0
  12. evalwarden-0.6.0/src/evalwarden/checks/cost.py +308 -0
  13. evalwarden-0.6.0/src/evalwarden/checks/env_leakage.py +137 -0
  14. evalwarden-0.6.0/src/evalwarden/checks/grader.py +112 -0
  15. evalwarden-0.6.0/src/evalwarden/checks/judge.py +419 -0
  16. evalwarden-0.6.0/src/evalwarden/cli.py +305 -0
  17. evalwarden-0.6.0/src/evalwarden/demo/cost_clean/README.md +11 -0
  18. evalwarden-0.6.0/src/evalwarden/demo/cost_clean/dataset.json +1 -0
  19. evalwarden-0.6.0/src/evalwarden/demo/cost_clean/environment.json +1 -0
  20. evalwarden-0.6.0/src/evalwarden/demo/cost_clean/grader.json +1 -0
  21. evalwarden-0.6.0/src/evalwarden/demo/cost_clean/run.json +1 -0
  22. evalwarden-0.6.0/src/evalwarden/demo/cost_wasteful/README.md +19 -0
  23. evalwarden-0.6.0/src/evalwarden/demo/cost_wasteful/dataset.json +1 -0
  24. evalwarden-0.6.0/src/evalwarden/demo/cost_wasteful/environment.json +1 -0
  25. evalwarden-0.6.0/src/evalwarden/demo/cost_wasteful/grader.json +1 -0
  26. evalwarden-0.6.0/src/evalwarden/demo/cost_wasteful/run.json +1 -0
  27. evalwarden-0.6.0/src/evalwarden/demo/hardened/README.md +14 -0
  28. evalwarden-0.6.0/src/evalwarden/demo/hardened/dataset.json +1 -0
  29. evalwarden-0.6.0/src/evalwarden/demo/hardened/environment.json +1 -0
  30. evalwarden-0.6.0/src/evalwarden/demo/hardened/grader.json +1 -0
  31. evalwarden-0.6.0/src/evalwarden/demo/hardened/run.json +1 -0
  32. evalwarden-0.6.0/src/evalwarden/demo/hardened/run_cheat.json +1 -0
  33. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/README.md +3 -0
  34. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/dataset.json +1 -0
  35. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/environment.json +5 -0
  36. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/grader.json +16 -0
  37. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/judge_run.json +449 -0
  38. evalwarden-0.6.0/src/evalwarden/demo/judge_bad/run.json +150 -0
  39. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/README.md +3 -0
  40. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/dataset.json +1 -0
  41. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/environment.json +5 -0
  42. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/grader.json +21 -0
  43. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/judge_run.json +448 -0
  44. evalwarden-0.6.0/src/evalwarden/demo/judge_clean/run.json +150 -0
  45. evalwarden-0.6.0/src/evalwarden/demo/leaky/README.md +22 -0
  46. evalwarden-0.6.0/src/evalwarden/demo/leaky/dataset.json +1 -0
  47. evalwarden-0.6.0/src/evalwarden/demo/leaky/environment.json +1 -0
  48. evalwarden-0.6.0/src/evalwarden/demo/leaky/gold/patch-task-001.diff +5 -0
  49. evalwarden-0.6.0/src/evalwarden/demo/leaky/gold/patch-task-002.diff +5 -0
  50. evalwarden-0.6.0/src/evalwarden/demo/leaky/gold/patch-task-003.diff +5 -0
  51. evalwarden-0.6.0/src/evalwarden/demo/leaky/gold_map.json +1 -0
  52. evalwarden-0.6.0/src/evalwarden/demo/leaky/grader.json +1 -0
  53. evalwarden-0.6.0/src/evalwarden/demo/leaky/run.json +1 -0
  54. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_bad/README.md +14 -0
  55. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_bad/promptfooconfig.yaml +30 -0
  56. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_bad/results.json +1 -0
  57. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_clean/README.md +13 -0
  58. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_clean/promptfooconfig.yaml +27 -0
  59. evalwarden-0.6.0/src/evalwarden/demo/promptfoo_clean/results.json +1 -0
  60. evalwarden-0.6.0/src/evalwarden/engine.py +104 -0
  61. evalwarden-0.6.0/src/evalwarden/model.py +200 -0
  62. evalwarden-0.6.0/src/evalwarden/reporters/__init__.py +12 -0
  63. evalwarden-0.6.0/src/evalwarden/reporters/html.py +182 -0
  64. evalwarden-0.6.0/src/evalwarden/reporters/report_card.py +297 -0
  65. evalwarden-0.6.0/src/evalwarden/reporters/terminal.py +90 -0
  66. evalwarden-0.6.0/src/evalwarden.egg-info/PKG-INFO +252 -0
  67. evalwarden-0.6.0/src/evalwarden.egg-info/SOURCES.txt +78 -0
  68. evalwarden-0.6.0/src/evalwarden.egg-info/dependency_links.txt +1 -0
  69. evalwarden-0.6.0/src/evalwarden.egg-info/entry_points.txt +2 -0
  70. evalwarden-0.6.0/src/evalwarden.egg-info/requires.txt +6 -0
  71. evalwarden-0.6.0/src/evalwarden.egg-info/top_level.txt +1 -0
  72. evalwarden-0.6.0/tests/test_adapter.py +95 -0
  73. evalwarden-0.6.0/tests/test_cli.py +144 -0
  74. evalwarden-0.6.0/tests/test_cost.py +218 -0
  75. evalwarden-0.6.0/tests/test_engine.py +52 -0
  76. evalwarden-0.6.0/tests/test_env_leakage.py +72 -0
  77. evalwarden-0.6.0/tests/test_grader.py +67 -0
  78. evalwarden-0.6.0/tests/test_judge.py +268 -0
  79. evalwarden-0.6.0/tests/test_promptfoo.py +225 -0
  80. evalwarden-0.6.0/tests/test_report_card.py +151 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yuriy H
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,252 @@
1
+ Metadata-Version: 2.4
2
+ Name: evalwarden
3
+ Version: 0.6.0
4
+ Summary: A linter for agent evaluations: audit the measurement system around a score.
5
+ Author: Yuriy H
6
+ License: MIT
7
+ Keywords: agents,evaluation,benchmarks,linting,llm
8
+ Requires-Python: >=3.11
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: typer>=0.9
12
+ Requires-Dist: jinja2>=3.1
13
+ Requires-Dist: pyyaml>=6
14
+ Provides-Extra: dev
15
+ Requires-Dist: pytest>=8; extra == "dev"
16
+ Dynamic: license-file
17
+
18
+ # evalwarden
19
+
20
+ **A linter for agent evaluations, not another eval framework.**
21
+
22
+ ## The problem
23
+
24
+ Benchmark scores ship decisions: which model to deploy, which paper to
25
+ accept, which agent to buy. But the tools that run evals never check whether
26
+ the measurement itself is sound. A solver can read the task ID from the
27
+ environment, look up the gold answer, and report 100%. A grader can be
28
+ writable by the agent it grades. A model judge can be uncalibrated, biased,
29
+ and self-contradictory. The score looks fine. The score is meaningless.
30
+
31
+ evalwarden audits the measurement system around a score: the dataset, the
32
+ evidence boundary, the grader, the run records, and the cost. It never runs
33
+ your evals and never changes your harness. It reads your eval artifacts and
34
+ tells you whether the score can be trusted, with file-level evidence for
35
+ every finding.
36
+
37
+ ![evalwarden HTML integrity report](docs/images/report-screenshot.png)
38
+
39
+ ## Quickstart
40
+
41
+ Three steps, about a minute, no model key required:
42
+
43
+ ```bash
44
+ pip install .
45
+ evalwarden demo
46
+ ```
47
+
48
+ That audits a deliberately broken benchmark and writes
49
+ `evalwarden-demo-report.html`. Open it in a browser.
50
+
51
+ ## The flagship demo: caught red-handed
52
+
53
+ A tiny synthetic coding benchmark reports **3/3 PASS**. The solver earned
54
+ none of it: it reads `TASK_ID` from the environment, looks up the answer in
55
+ `gold_map.json`, and submits the gold patch. The auditor flags the exact
56
+ leak channels, each with file-level evidence:
57
+
58
+ - `ENV-001` — `TASK_ID`, `RUN_ID`, `AGENT_TOKEN` visible to the agent
59
+ - `ENV-001` — `gold_map.json` mounted where the agent can read it
60
+ - `GRAD-001` — the verifier is writable by the agent
61
+
62
+ Result: **0/100 BLOCKED**. Then the `hardened` fixture shows the fix:
63
+ opaque IDs, no gold mounted, read-only verifier, a genuine solver. Result:
64
+ **100/100 PASS**. Same benchmark, same auditor, before and after.
65
+
66
+ ```bash
67
+ evalwarden demo --fixture leaky # the cheat: 0/100 BLOCKED
68
+ evalwarden demo --fixture hardened # the fix: 100/100 PASS
69
+ ```
70
+
71
+ ## A tour: finding, explain, fix
72
+
73
+ Every finding carries its evidence, and every check explains itself. Here
74
+ is the full loop on the flagship cheat. First the finding:
75
+
76
+ ```bash
77
+ $ evalwarden demo --fixture leaky
78
+ ...
79
+ E ENV-001 [error|confidence:high] Eval-detection variable visible to agent: TASK_ID
80
+ evidence: environment variable 'TASK_ID' is visible to the agent/solver.
81
+ evidence: Known eval-detection signal: an agent can branch on this value
82
+ (e.g. look up a gold patch by task ID).
83
+ at: environment.json -- env.TASK_ID
84
+ ```
85
+
86
+ Then what it means and how to fix it:
87
+
88
+ ```bash
89
+ $ evalwarden explain ENV-001
90
+ ENV-001: Eval-detection signal or leaked state visible to the agent
91
+
92
+ threat: If the agent can observe run IDs, task IDs, or agent tokens -- or read
93
+ answer-bearing material such as gold patches -- it can condition its behavior
94
+ on the measurement instead of the task. The resulting score measures the leak,
95
+ not the capability.
96
+
97
+ fix: Remove eval-detection variables from the agent's environment (keep them
98
+ harness-side only), mount gold/answer material so the agent cannot read it,
99
+ and re-run.
100
+ ```
101
+
102
+ Apply the fix, which is exactly what the `hardened` fixture does, and the
103
+ same audit goes green:
104
+
105
+ ```bash
106
+ $ evalwarden demo --fixture hardened
107
+ ...
108
+ evalwarden audit: tinycode-hardened-1.0
109
+ Integrity: 100 / 100 PASS (0 errors, 0 high, 0 medium, 0 low)
110
+ ```
111
+
112
+ ## Demo fixtures
113
+
114
+ Every fixture ships inside the package, so the demo works from any
115
+ directory. Run any of them with `evalwarden demo --fixture <name>`:
116
+
117
+ | Fixture | What it shows |
118
+ |---------|---------------|
119
+ | `leaky` | The flagship: a cheating solver caught red-handed. 0/100 BLOCKED. |
120
+ | `hardened` | The fix: opaque IDs, no gold mounted, read-only verifier. A genuine solver passes and the audit is clean. 100/100 PASS. |
121
+ | `judge_bad` | A miscalibrated model judge: unvalidated, AB-only protocol, self-contradicting repeats, 48% reference agreement, position and verbosity bias. 50/100 BLOCKED. |
122
+ | `judge_clean` | The validated judge: counterbalanced, temperature 0, anchored rubric, 92% agreement. 100/100 PASS. |
123
+ | `cost_wasteful` | A wasteful run: 2.67 tries per success, 91% of spend on attempts that never passed, one 22,000-token runaway loop. |
124
+ | `cost_clean` | The same tasks solved first try at modest cost. No cost findings. |
125
+ | `promptfoo_bad` | A Promptfoo eval with `TASK_ID`/`RUN_ID` planted in env (ENV-001) and uncalibrated `llm-rubric` assertions (JUDGE-001). 35/100 BLOCKED. |
126
+ | `promptfoo_clean` | The same Promptfoo eval done right: innocuous env, deterministic assertions, full token/latency reporting. 100/100 PASS. |
127
+
128
+ ```bash
129
+ evalwarden demo --fixture judge_bad
130
+ evalwarden demo --fixture cost_wasteful --budget-per-task 0.05
131
+ ```
132
+
133
+ ## Checks
134
+
135
+ Deterministic, high-precision checks only. A linter that cries contamination
136
+ on a clean eval is worse than no auditor, so every finding carries a
137
+ **confidence** label and clean evals produce zero findings.
138
+
139
+ | ID | Check | Severity |
140
+ |----|-------|----------|
141
+ | ENV-001 | Eval-detection signal or leaked state visible to the agent | Error |
142
+ | GRAD-001 | Verifier writable by the agent | Error |
143
+ | GRAD-002 | Grader grants credit without completion | Error |
144
+ | COST-001 | Cost per success not reported (usage data missing) | Medium |
145
+ | COST-002 | Successes cost multiple attempts each (retry multiplier) | Medium |
146
+ | COST-003 | Most spend burned on attempts that never passed | Medium |
147
+ | COST-004 | Runaway attempt burned far more than a typical one | Medium |
148
+ | JUDGE-001 | Model judge lacks validation (no labels, unanchored rubric, hot single-sample) | High |
149
+ | JUDGE-002 | Pairwise order not counterbalanced | High |
150
+ | JUDGE-003 | Judge contradicts itself on repeated judgments | High |
151
+ | JUDGE-004 | Judge disagrees with reference labels | High |
152
+ | JUDGE-005 | Position bias: presentation order predicts the winner | Medium |
153
+ | JUDGE-006 | Verbosity bias: longer answers win disproportionately | Medium |
154
+
155
+ `evalwarden explain COST-004` prints any check's threat model, evidence, and fix.
156
+
157
+ ## Report cards
158
+
159
+ A report card is the public face of an audit: one self-contained HTML page
160
+ per benchmark with the verdict, per-category scores, key findings, and a
161
+ methodology footer. Generate one per eval, or a whole set plus an index:
162
+
163
+ ```bash
164
+ evalwarden report-card path/to/eval --output card.html
165
+ evalwarden report-cards eval-a/ eval-b/ --output-dir cards/
166
+ ```
167
+
168
+ Example cards generated from the demo fixtures live in
169
+ [`examples/report-cards/`](examples/report-cards/) ([index](examples/report-cards/index.html)):
170
+ a blocked cheat, a clean pass, a bad judge, and a validated judge. A full
171
+ sample audit report is at [`examples/sample-report.html`](examples/sample-report.html).
172
+
173
+ ## Real benchmarks
174
+
175
+ The linter also runs against real public benchmarks, translated
176
+ mechanically from their published definitions: pinned sources, no invented
177
+ traces, provenance committed alongside. Cards live in
178
+ [`examples/report-cards/real/`](examples/report-cards/real/):
179
+
180
+ | Benchmark | Result |
181
+ |-----------|--------|
182
+ | [SWE-bench Verified](examples/report-cards/real/swe-bench-verified-via-inspect-evals.html) (via inspect_evals) | 100/100 PASS: gold patches, test patches, and grading stay harness-side where the agent cannot reach them |
183
+ | [HealthBench](examples/report-cards/real/healthbench-via-inspect-evals.html) (via inspect_evals) | 85/100 BLOCKED: the model judge (`openai/gpt-4o-mini`) ships with no calibration set in the definition |
184
+
185
+ The HealthBench card deserves one sentence of context: the benchmark's
186
+ authors validated their grader in the paper through a separate meta-eval
187
+ task. The linter flags that the eval definition itself carries no
188
+ calibration evidence, which is precisely the gap: anyone auditing the
189
+ artifact alone cannot verify the judge. The
190
+ [reading guide](examples/report-cards/real/README.md) walks through both
191
+ cards, the translation, and the scope limits.
192
+
193
+ ## How it works
194
+
195
+ Adapters translate harness artifacts into a framework-neutral integrity
196
+ model. Checks only ever see that model, never harness internals. That
197
+ boundary is what keeps this a linter instead of another eval framework.
198
+
199
+ ```
200
+ eval artifact/ ──▶ adapter (inspect | promptfoo) ──▶ integrity model ──▶ checks ──▶ report
201
+ read-only, offline data · boundary · grader · runs
202
+ ```
203
+
204
+ v0.5 ships two adapters: `inspect` for Inspect-style eval artifact
205
+ directories, and `promptfoo` for Promptfoo's `promptfooconfig.yaml` plus the
206
+ JSON export from `promptfoo eval --output results.json`. Both are strictly
207
+ read-only and offline; variable names are kept for analysis while secret
208
+ values never enter the normalized model. Harbor and BrowserGym plug into the
209
+ same registry. A new check is one module plus one registration line; a new
210
+ reporter is one module plus one import.
211
+
212
+ ## CLI
213
+
214
+ ```
215
+ evalwarden audit <eval-artifact> [--adapter auto|inspect|promptfoo] [--output report.html]
216
+ [--json findings.json] [--fail-on high]
217
+ [--price-in 3.0] [--price-out 15.0]
218
+ [--budget-per-task USD]
219
+ evalwarden demo [--fixture leaky|hardened|judge_bad|judge_clean|cost_wasteful|cost_clean|promptfoo_bad|promptfoo_clean]
220
+ [--output evalwarden-demo-report.html] [--budget-per-task USD]
221
+ evalwarden report-card <eval-artifact> [--output card.html]
222
+ evalwarden report-cards <eval...> [--fixtures a,b] [--output-dir cards/]
223
+ evalwarden explain <CHECK-ID>
224
+ ```
225
+
226
+ Exit codes: `0` policy passes, `1` findings cross `--fail-on`, `2` the audit
227
+ could not complete. The same policy runs locally and as a CI gate.
228
+
229
+ ## Reports
230
+
231
+ Self-contained HTML (inline CSS, no JavaScript, no remote assets), terminal
232
+ output, JSON findings, and report cards. Secret values are never stored, only
233
+ variable *names* enter the model. Integrity scores are diagnostic, not a
234
+ certification: the report says "no blocking findings observed under this
235
+ policy," never "certified safe."
236
+
237
+ ## Non-goals
238
+
239
+ Running or scheduling evaluations, replacing task/solver/scorer APIs, trace
240
+ observability, generic red-teaming, public leaderboards, declaring any
241
+ benchmark contamination-free.
242
+
243
+ ## Development
244
+
245
+ ```bash
246
+ pip install -e ".[dev]"
247
+ pytest
248
+ ```
249
+
250
+ The test suite is the product's credibility: every rule has positive,
251
+ negative, and precision fixtures (clean evals must *not* be flagged), the
252
+ adapter has a read-only contract test, and the fixtures run end to end.
@@ -0,0 +1,235 @@
1
+ # evalwarden
2
+
3
+ **A linter for agent evaluations, not another eval framework.**
4
+
5
+ ## The problem
6
+
7
+ Benchmark scores ship decisions: which model to deploy, which paper to
8
+ accept, which agent to buy. But the tools that run evals never check whether
9
+ the measurement itself is sound. A solver can read the task ID from the
10
+ environment, look up the gold answer, and report 100%. A grader can be
11
+ writable by the agent it grades. A model judge can be uncalibrated, biased,
12
+ and self-contradictory. The score looks fine. The score is meaningless.
13
+
14
+ evalwarden audits the measurement system around a score: the dataset, the
15
+ evidence boundary, the grader, the run records, and the cost. It never runs
16
+ your evals and never changes your harness. It reads your eval artifacts and
17
+ tells you whether the score can be trusted, with file-level evidence for
18
+ every finding.
19
+
20
+ ![evalwarden HTML integrity report](docs/images/report-screenshot.png)
21
+
22
+ ## Quickstart
23
+
24
+ Three steps, about a minute, no model key required:
25
+
26
+ ```bash
27
+ pip install .
28
+ evalwarden demo
29
+ ```
30
+
31
+ That audits a deliberately broken benchmark and writes
32
+ `evalwarden-demo-report.html`. Open it in a browser.
33
+
34
+ ## The flagship demo: caught red-handed
35
+
36
+ A tiny synthetic coding benchmark reports **3/3 PASS**. The solver earned
37
+ none of it: it reads `TASK_ID` from the environment, looks up the answer in
38
+ `gold_map.json`, and submits the gold patch. The auditor flags the exact
39
+ leak channels, each with file-level evidence:
40
+
41
+ - `ENV-001` — `TASK_ID`, `RUN_ID`, `AGENT_TOKEN` visible to the agent
42
+ - `ENV-001` — `gold_map.json` mounted where the agent can read it
43
+ - `GRAD-001` — the verifier is writable by the agent
44
+
45
+ Result: **0/100 BLOCKED**. Then the `hardened` fixture shows the fix:
46
+ opaque IDs, no gold mounted, read-only verifier, a genuine solver. Result:
47
+ **100/100 PASS**. Same benchmark, same auditor, before and after.
48
+
49
+ ```bash
50
+ evalwarden demo --fixture leaky # the cheat: 0/100 BLOCKED
51
+ evalwarden demo --fixture hardened # the fix: 100/100 PASS
52
+ ```
53
+
54
+ ## A tour: finding, explain, fix
55
+
56
+ Every finding carries its evidence, and every check explains itself. Here
57
+ is the full loop on the flagship cheat. First the finding:
58
+
59
+ ```bash
60
+ $ evalwarden demo --fixture leaky
61
+ ...
62
+ E ENV-001 [error|confidence:high] Eval-detection variable visible to agent: TASK_ID
63
+ evidence: environment variable 'TASK_ID' is visible to the agent/solver.
64
+ evidence: Known eval-detection signal: an agent can branch on this value
65
+ (e.g. look up a gold patch by task ID).
66
+ at: environment.json -- env.TASK_ID
67
+ ```
68
+
69
+ Then what it means and how to fix it:
70
+
71
+ ```bash
72
+ $ evalwarden explain ENV-001
73
+ ENV-001: Eval-detection signal or leaked state visible to the agent
74
+
75
+ threat: If the agent can observe run IDs, task IDs, or agent tokens -- or read
76
+ answer-bearing material such as gold patches -- it can condition its behavior
77
+ on the measurement instead of the task. The resulting score measures the leak,
78
+ not the capability.
79
+
80
+ fix: Remove eval-detection variables from the agent's environment (keep them
81
+ harness-side only), mount gold/answer material so the agent cannot read it,
82
+ and re-run.
83
+ ```
84
+
85
+ Apply the fix, which is exactly what the `hardened` fixture does, and the
86
+ same audit goes green:
87
+
88
+ ```bash
89
+ $ evalwarden demo --fixture hardened
90
+ ...
91
+ evalwarden audit: tinycode-hardened-1.0
92
+ Integrity: 100 / 100 PASS (0 errors, 0 high, 0 medium, 0 low)
93
+ ```
94
+
95
+ ## Demo fixtures
96
+
97
+ Every fixture ships inside the package, so the demo works from any
98
+ directory. Run any of them with `evalwarden demo --fixture <name>`:
99
+
100
+ | Fixture | What it shows |
101
+ |---------|---------------|
102
+ | `leaky` | The flagship: a cheating solver caught red-handed. 0/100 BLOCKED. |
103
+ | `hardened` | The fix: opaque IDs, no gold mounted, read-only verifier. A genuine solver passes and the audit is clean. 100/100 PASS. |
104
+ | `judge_bad` | A miscalibrated model judge: unvalidated, AB-only protocol, self-contradicting repeats, 48% reference agreement, position and verbosity bias. 50/100 BLOCKED. |
105
+ | `judge_clean` | The validated judge: counterbalanced, temperature 0, anchored rubric, 92% agreement. 100/100 PASS. |
106
+ | `cost_wasteful` | A wasteful run: 2.67 tries per success, 91% of spend on attempts that never passed, one 22,000-token runaway loop. |
107
+ | `cost_clean` | The same tasks solved first try at modest cost. No cost findings. |
108
+ | `promptfoo_bad` | A Promptfoo eval with `TASK_ID`/`RUN_ID` planted in env (ENV-001) and uncalibrated `llm-rubric` assertions (JUDGE-001). 35/100 BLOCKED. |
109
+ | `promptfoo_clean` | The same Promptfoo eval done right: innocuous env, deterministic assertions, full token/latency reporting. 100/100 PASS. |
110
+
111
+ ```bash
112
+ evalwarden demo --fixture judge_bad
113
+ evalwarden demo --fixture cost_wasteful --budget-per-task 0.05
114
+ ```
115
+
116
+ ## Checks
117
+
118
+ Deterministic, high-precision checks only. A linter that cries contamination
119
+ on a clean eval is worse than no auditor, so every finding carries a
120
+ **confidence** label and clean evals produce zero findings.
121
+
122
+ | ID | Check | Severity |
123
+ |----|-------|----------|
124
+ | ENV-001 | Eval-detection signal or leaked state visible to the agent | Error |
125
+ | GRAD-001 | Verifier writable by the agent | Error |
126
+ | GRAD-002 | Grader grants credit without completion | Error |
127
+ | COST-001 | Cost per success not reported (usage data missing) | Medium |
128
+ | COST-002 | Successes cost multiple attempts each (retry multiplier) | Medium |
129
+ | COST-003 | Most spend burned on attempts that never passed | Medium |
130
+ | COST-004 | Runaway attempt burned far more than a typical one | Medium |
131
+ | JUDGE-001 | Model judge lacks validation (no labels, unanchored rubric, hot single-sample) | High |
132
+ | JUDGE-002 | Pairwise order not counterbalanced | High |
133
+ | JUDGE-003 | Judge contradicts itself on repeated judgments | High |
134
+ | JUDGE-004 | Judge disagrees with reference labels | High |
135
+ | JUDGE-005 | Position bias: presentation order predicts the winner | Medium |
136
+ | JUDGE-006 | Verbosity bias: longer answers win disproportionately | Medium |
137
+
138
+ `evalwarden explain COST-004` prints any check's threat model, evidence, and fix.
139
+
140
+ ## Report cards
141
+
142
+ A report card is the public face of an audit: one self-contained HTML page
143
+ per benchmark with the verdict, per-category scores, key findings, and a
144
+ methodology footer. Generate one per eval, or a whole set plus an index:
145
+
146
+ ```bash
147
+ evalwarden report-card path/to/eval --output card.html
148
+ evalwarden report-cards eval-a/ eval-b/ --output-dir cards/
149
+ ```
150
+
151
+ Example cards generated from the demo fixtures live in
152
+ [`examples/report-cards/`](examples/report-cards/) ([index](examples/report-cards/index.html)):
153
+ a blocked cheat, a clean pass, a bad judge, and a validated judge. A full
154
+ sample audit report is at [`examples/sample-report.html`](examples/sample-report.html).
155
+
156
+ ## Real benchmarks
157
+
158
+ The linter also runs against real public benchmarks, translated
159
+ mechanically from their published definitions: pinned sources, no invented
160
+ traces, provenance committed alongside. Cards live in
161
+ [`examples/report-cards/real/`](examples/report-cards/real/):
162
+
163
+ | Benchmark | Result |
164
+ |-----------|--------|
165
+ | [SWE-bench Verified](examples/report-cards/real/swe-bench-verified-via-inspect-evals.html) (via inspect_evals) | 100/100 PASS: gold patches, test patches, and grading stay harness-side where the agent cannot reach them |
166
+ | [HealthBench](examples/report-cards/real/healthbench-via-inspect-evals.html) (via inspect_evals) | 85/100 BLOCKED: the model judge (`openai/gpt-4o-mini`) ships with no calibration set in the definition |
167
+
168
+ The HealthBench card deserves one sentence of context: the benchmark's
169
+ authors validated their grader in the paper through a separate meta-eval
170
+ task. The linter flags that the eval definition itself carries no
171
+ calibration evidence, which is precisely the gap: anyone auditing the
172
+ artifact alone cannot verify the judge. The
173
+ [reading guide](examples/report-cards/real/README.md) walks through both
174
+ cards, the translation, and the scope limits.
175
+
176
+ ## How it works
177
+
178
+ Adapters translate harness artifacts into a framework-neutral integrity
179
+ model. Checks only ever see that model, never harness internals. That
180
+ boundary is what keeps this a linter instead of another eval framework.
181
+
182
+ ```
183
+ eval artifact/ ──▶ adapter (inspect | promptfoo) ──▶ integrity model ──▶ checks ──▶ report
184
+ read-only, offline data · boundary · grader · runs
185
+ ```
186
+
187
+ v0.5 ships two adapters: `inspect` for Inspect-style eval artifact
188
+ directories, and `promptfoo` for Promptfoo's `promptfooconfig.yaml` plus the
189
+ JSON export from `promptfoo eval --output results.json`. Both are strictly
190
+ read-only and offline; variable names are kept for analysis while secret
191
+ values never enter the normalized model. Harbor and BrowserGym plug into the
192
+ same registry. A new check is one module plus one registration line; a new
193
+ reporter is one module plus one import.
194
+
195
+ ## CLI
196
+
197
+ ```
198
+ evalwarden audit <eval-artifact> [--adapter auto|inspect|promptfoo] [--output report.html]
199
+ [--json findings.json] [--fail-on high]
200
+ [--price-in 3.0] [--price-out 15.0]
201
+ [--budget-per-task USD]
202
+ evalwarden demo [--fixture leaky|hardened|judge_bad|judge_clean|cost_wasteful|cost_clean|promptfoo_bad|promptfoo_clean]
203
+ [--output evalwarden-demo-report.html] [--budget-per-task USD]
204
+ evalwarden report-card <eval-artifact> [--output card.html]
205
+ evalwarden report-cards <eval...> [--fixtures a,b] [--output-dir cards/]
206
+ evalwarden explain <CHECK-ID>
207
+ ```
208
+
209
+ Exit codes: `0` policy passes, `1` findings cross `--fail-on`, `2` the audit
210
+ could not complete. The same policy runs locally and as a CI gate.
211
+
212
+ ## Reports
213
+
214
+ Self-contained HTML (inline CSS, no JavaScript, no remote assets), terminal
215
+ output, JSON findings, and report cards. Secret values are never stored, only
216
+ variable *names* enter the model. Integrity scores are diagnostic, not a
217
+ certification: the report says "no blocking findings observed under this
218
+ policy," never "certified safe."
219
+
220
+ ## Non-goals
221
+
222
+ Running or scheduling evaluations, replacing task/solver/scorer APIs, trace
223
+ observability, generic red-teaming, public leaderboards, declaring any
224
+ benchmark contamination-free.
225
+
226
+ ## Development
227
+
228
+ ```bash
229
+ pip install -e ".[dev]"
230
+ pytest
231
+ ```
232
+
233
+ The test suite is the product's credibility: every rule has positive,
234
+ negative, and precision fixtures (clean evals must *not* be flagged), the
235
+ adapter has a read-only contract test, and the fixtures run end to end.
@@ -0,0 +1,33 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "evalwarden"
7
+ version = "0.6.0"
8
+ description = "A linter for agent evaluations: audit the measurement system around a score."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Yuriy H" }]
13
+ keywords = ["agents", "evaluation", "benchmarks", "linting", "llm"]
14
+ dependencies = [
15
+ "typer>=0.9",
16
+ "jinja2>=3.1",
17
+ "pyyaml>=6",
18
+ ]
19
+
20
+ [project.optional-dependencies]
21
+ dev = ["pytest>=8"]
22
+
23
+ [project.scripts]
24
+ evalwarden = "evalwarden.cli:app"
25
+
26
+ [tool.setuptools.packages.find]
27
+ where = ["src"]
28
+
29
+ [tool.setuptools.package-data]
30
+ evalwarden = ["demo/**/*"]
31
+
32
+ [tool.pytest.ini_options]
33
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,3 @@
1
+ """evalwarden: a linter for agent evaluations."""
2
+
3
+ __version__ = "0.6.0"
@@ -0,0 +1,70 @@
1
+ """Adapter protocol and registry.
2
+
3
+ Adapters are the ONLY harness-specific code in the project. They discover and
4
+ translate; the core never becomes a harness. Each adapter:
5
+
6
+ - is read-only (never mutates the input files),
7
+ - works offline (no network in default mode),
8
+ - preserves source locations,
9
+ - pins its own version and the schema versions it understands,
10
+ - reports unsupported fields explicitly instead of silently dropping them.
11
+
12
+ v0.1 ships one adapter: Inspect AI style eval artifacts. Promptfoo, Harbor,
13
+ and BrowserGym adapters plug into REGISTRY later with the same contract.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from pathlib import Path
18
+ from typing import Protocol
19
+
20
+ from ..model import Confidence, IntegrityModel
21
+
22
+
23
+ class AuditError(Exception):
24
+ """The audit could not complete (exit code 2)."""
25
+
26
+
27
+ class Adapter(Protocol):
28
+ name: str
29
+ version: str
30
+
31
+ def detect(self, path: Path) -> Confidence:
32
+ """How confident are we that this adapter understands `path`?"""
33
+ ...
34
+
35
+ def collect(self, path: Path) -> dict:
36
+ """Read-only collection of raw evidence from the artifact."""
37
+ ...
38
+
39
+ def normalize(self, bundle: dict) -> IntegrityModel:
40
+ """Translate raw evidence into the framework-neutral integrity model."""
41
+ ...
42
+
43
+
44
+ REGISTRY: list[Adapter] = []
45
+
46
+
47
+ def register(adapter: Adapter) -> Adapter:
48
+ # Accept either an instance or a class (instantiated here) so adapters can
49
+ # use either `@register` on the class or `register(MyAdapter())`.
50
+ REGISTRY.append(adapter() if isinstance(adapter, type) else adapter)
51
+ return adapter
52
+
53
+
54
+ def autodetect(path: Path) -> Adapter:
55
+ """Pick the most confident adapter for `path`, or raise AuditError."""
56
+ if not REGISTRY:
57
+ raise AuditError("no adapters registered")
58
+ ranked = sorted(
59
+ ((adapter.detect(path), adapter) for adapter in REGISTRY),
60
+ key=lambda item: (item[0] == Confidence.HIGH, item[0] == Confidence.MEDIUM),
61
+ reverse=True,
62
+ )
63
+ confidence, adapter = ranked[0]
64
+ if confidence == Confidence.LOW:
65
+ raise AuditError(
66
+ f"no adapter recognizes {path} "
67
+ f"(best guess: {adapter.name}, confidence=low). "
68
+ "Expected an eval artifact directory (see demo/leaky for the layout)."
69
+ )
70
+ return adapter