evalwarden 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. evalwarden/__init__.py +3 -0
  2. evalwarden/adapters/__init__.py +70 -0
  3. evalwarden/adapters/inspect_ai.py +213 -0
  4. evalwarden/adapters/promptfoo.py +357 -0
  5. evalwarden/checks/__init__.py +27 -0
  6. evalwarden/checks/base.py +35 -0
  7. evalwarden/checks/cost.py +308 -0
  8. evalwarden/checks/env_leakage.py +137 -0
  9. evalwarden/checks/grader.py +112 -0
  10. evalwarden/checks/judge.py +419 -0
  11. evalwarden/cli.py +305 -0
  12. evalwarden/demo/cost_clean/README.md +11 -0
  13. evalwarden/demo/cost_clean/dataset.json +1 -0
  14. evalwarden/demo/cost_clean/environment.json +1 -0
  15. evalwarden/demo/cost_clean/grader.json +1 -0
  16. evalwarden/demo/cost_clean/run.json +1 -0
  17. evalwarden/demo/cost_wasteful/README.md +19 -0
  18. evalwarden/demo/cost_wasteful/dataset.json +1 -0
  19. evalwarden/demo/cost_wasteful/environment.json +1 -0
  20. evalwarden/demo/cost_wasteful/grader.json +1 -0
  21. evalwarden/demo/cost_wasteful/run.json +1 -0
  22. evalwarden/demo/hardened/README.md +14 -0
  23. evalwarden/demo/hardened/dataset.json +1 -0
  24. evalwarden/demo/hardened/environment.json +1 -0
  25. evalwarden/demo/hardened/grader.json +1 -0
  26. evalwarden/demo/hardened/run.json +1 -0
  27. evalwarden/demo/hardened/run_cheat.json +1 -0
  28. evalwarden/demo/judge_bad/README.md +3 -0
  29. evalwarden/demo/judge_bad/dataset.json +1 -0
  30. evalwarden/demo/judge_bad/environment.json +5 -0
  31. evalwarden/demo/judge_bad/grader.json +16 -0
  32. evalwarden/demo/judge_bad/judge_run.json +449 -0
  33. evalwarden/demo/judge_bad/run.json +150 -0
  34. evalwarden/demo/judge_clean/README.md +3 -0
  35. evalwarden/demo/judge_clean/dataset.json +1 -0
  36. evalwarden/demo/judge_clean/environment.json +5 -0
  37. evalwarden/demo/judge_clean/grader.json +21 -0
  38. evalwarden/demo/judge_clean/judge_run.json +448 -0
  39. evalwarden/demo/judge_clean/run.json +150 -0
  40. evalwarden/demo/leaky/README.md +22 -0
  41. evalwarden/demo/leaky/dataset.json +1 -0
  42. evalwarden/demo/leaky/environment.json +1 -0
  43. evalwarden/demo/leaky/gold/patch-task-001.diff +5 -0
  44. evalwarden/demo/leaky/gold/patch-task-002.diff +5 -0
  45. evalwarden/demo/leaky/gold/patch-task-003.diff +5 -0
  46. evalwarden/demo/leaky/gold_map.json +1 -0
  47. evalwarden/demo/leaky/grader.json +1 -0
  48. evalwarden/demo/leaky/run.json +1 -0
  49. evalwarden/demo/promptfoo_bad/README.md +14 -0
  50. evalwarden/demo/promptfoo_bad/promptfooconfig.yaml +30 -0
  51. evalwarden/demo/promptfoo_bad/results.json +1 -0
  52. evalwarden/demo/promptfoo_clean/README.md +13 -0
  53. evalwarden/demo/promptfoo_clean/promptfooconfig.yaml +27 -0
  54. evalwarden/demo/promptfoo_clean/results.json +1 -0
  55. evalwarden/engine.py +104 -0
  56. evalwarden/model.py +200 -0
  57. evalwarden/reporters/__init__.py +12 -0
  58. evalwarden/reporters/html.py +182 -0
  59. evalwarden/reporters/report_card.py +297 -0
  60. evalwarden/reporters/terminal.py +90 -0
  61. evalwarden-0.6.0.dist-info/METADATA +252 -0
  62. evalwarden-0.6.0.dist-info/RECORD +66 -0
  63. evalwarden-0.6.0.dist-info/WHEEL +5 -0
  64. evalwarden-0.6.0.dist-info/entry_points.txt +2 -0
  65. evalwarden-0.6.0.dist-info/licenses/LICENSE +21 -0
  66. evalwarden-0.6.0.dist-info/top_level.txt +1 -0
evalwarden/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """evalwarden: a linter for agent evaluations."""
2
+
3
+ __version__ = "0.6.0"
@@ -0,0 +1,70 @@
1
+ """Adapter protocol and registry.
2
+
3
+ Adapters are the ONLY harness-specific code in the project. They discover and
4
+ translate; the core never becomes a harness. Each adapter:
5
+
6
+ - is read-only (never mutates the input files),
7
+ - works offline (no network in default mode),
8
+ - preserves source locations,
9
+ - pins its own version and the schema versions it understands,
10
+ - reports unsupported fields explicitly instead of silently dropping them.
11
+
12
+ v0.1 ships one adapter: Inspect AI style eval artifacts. Promptfoo, Harbor,
13
+ and BrowserGym adapters plug into REGISTRY later with the same contract.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from pathlib import Path
18
+ from typing import Protocol
19
+
20
+ from ..model import Confidence, IntegrityModel
21
+
22
+
23
+ class AuditError(Exception):
24
+ """The audit could not complete (exit code 2)."""
25
+
26
+
27
+ class Adapter(Protocol):
28
+ name: str
29
+ version: str
30
+
31
+ def detect(self, path: Path) -> Confidence:
32
+ """How confident are we that this adapter understands `path`?"""
33
+ ...
34
+
35
+ def collect(self, path: Path) -> dict:
36
+ """Read-only collection of raw evidence from the artifact."""
37
+ ...
38
+
39
+ def normalize(self, bundle: dict) -> IntegrityModel:
40
+ """Translate raw evidence into the framework-neutral integrity model."""
41
+ ...
42
+
43
+
44
+ REGISTRY: list[Adapter] = []
45
+
46
+
47
+ def register(adapter: Adapter) -> Adapter:
48
+ # Accept either an instance or a class (instantiated here) so adapters can
49
+ # use either `@register` on the class or `register(MyAdapter())`.
50
+ REGISTRY.append(adapter() if isinstance(adapter, type) else adapter)
51
+ return adapter
52
+
53
+
54
+ def autodetect(path: Path) -> Adapter:
55
+ """Pick the most confident adapter for `path`, or raise AuditError."""
56
+ if not REGISTRY:
57
+ raise AuditError("no adapters registered")
58
+ ranked = sorted(
59
+ ((adapter.detect(path), adapter) for adapter in REGISTRY),
60
+ key=lambda item: (item[0] == Confidence.HIGH, item[0] == Confidence.MEDIUM),
61
+ reverse=True,
62
+ )
63
+ confidence, adapter = ranked[0]
64
+ if confidence == Confidence.LOW:
65
+ raise AuditError(
66
+ f"no adapter recognizes {path} "
67
+ f"(best guess: {adapter.name}, confidence=low). "
68
+ "Expected an eval artifact directory (see demo/leaky for the layout)."
69
+ )
70
+ return adapter
@@ -0,0 +1,213 @@
1
+ """Inspect AI adapter (v0.1).
2
+
3
+ Reads an Inspect-style eval artifact directory -- the v0.1 normalized input
4
+ format, modeled on Inspect's Task / dataset / scorer / .eval-log concepts:
5
+
6
+ eval-artifact/
7
+ dataset.json samples: [{id, prompt, metadata}]
8
+ environment.json env vars visible to the solver, mounts
9
+ grader.json verifier config, pass conditions, tests
10
+ run.json per-task attempts with status, usage, actions
11
+ judge_run.json (optional) model-judge config, judgments, reference labels
12
+
13
+ This is a read-only translation layer. Full-fidelity parsing of real Inspect
14
+ `.eval` logs is a later milestone; the adapter pins the schema version it
15
+ understands and fails clearly on anything else.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import hashlib
20
+ import json
21
+ from pathlib import Path
22
+ from typing import Any
23
+
24
+ from ..model import (
25
+ Attempt,
26
+ Confidence,
27
+ Environment,
28
+ Grader,
29
+ IntegrityModel,
30
+ Judgment,
31
+ Mount,
32
+ TaskSample,
33
+ )
34
+ from . import AuditError, register
35
+
36
+ ADAPTER_NAME = "inspect"
37
+ ADAPTER_VERSION = "0.1.0"
38
+ SCHEMA_VERSION = "evalwarden-artifact-v1"
39
+
40
+
41
+ def _read_json(path: Path) -> Any:
42
+ try:
43
+ return json.loads(path.read_text(encoding="utf-8"))
44
+ except FileNotFoundError as exc:
45
+ raise AuditError(f"missing required file: {path}") from exc
46
+ except json.JSONDecodeError as exc:
47
+ raise AuditError(f"invalid JSON in {path}: {exc}") from exc
48
+
49
+
50
+ def _digest(path: Path) -> str:
51
+ return hashlib.sha256(path.read_bytes()).hexdigest()
52
+
53
+
54
+ @register
55
+ class InspectAdapter:
56
+ name = ADAPTER_NAME
57
+ version = ADAPTER_VERSION
58
+
59
+ def detect(self, path: Path) -> Confidence:
60
+ if not path.is_dir():
61
+ return Confidence.LOW
62
+ has_dataset = (path / "dataset.json").is_file()
63
+ has_grader = (path / "grader.json").is_file()
64
+ if has_dataset and has_grader:
65
+ return Confidence.HIGH
66
+ if has_dataset:
67
+ return Confidence.MEDIUM
68
+ return Confidence.LOW
69
+
70
+ def collect(self, path: Path) -> dict:
71
+ """Read-only: files are opened for reading and never modified."""
72
+ bundle: dict[str, Any] = {"root": path}
73
+ digests: dict[str, str] = {}
74
+ for fname in (
75
+ "dataset.json",
76
+ "environment.json",
77
+ "grader.json",
78
+ "run.json",
79
+ "judge_run.json",
80
+ ):
81
+ fpath = path / fname
82
+ if fpath.is_file():
83
+ bundle[fname] = _read_json(fpath)
84
+ digests[fname] = _digest(fpath)
85
+ for fname in ("gold_map.json",):
86
+ fpath = path / fname
87
+ if fpath.is_file():
88
+ # Recorded for the digest manifest only; never parsed for content
89
+ # beyond its existence (it is the leak, not the evidence).
90
+ digests[fname] = _digest(fpath)
91
+ bundle.setdefault("extra_files", []).append(fname)
92
+ bundle["digests"] = digests
93
+ return bundle
94
+
95
+ def normalize(self, bundle: dict) -> IntegrityModel:
96
+ root: Path = bundle["root"]
97
+ dataset = bundle.get("dataset.json", {})
98
+ environment = bundle.get("environment.json", {})
99
+ grader_cfg = bundle.get("grader.json", {})
100
+ run = bundle.get("run.json", {})
101
+
102
+ if dataset.get("schema_version", SCHEMA_VERSION) != SCHEMA_VERSION and "tasks" not in dataset:
103
+ raise AuditError(
104
+ f"unsupported dataset schema (expected {SCHEMA_VERSION!r} with a 'tasks' list)"
105
+ )
106
+
107
+ tasks = [
108
+ TaskSample(
109
+ id=str(t.get("id", f"task-{i}")),
110
+ prompt=str(t.get("prompt", "")),
111
+ metadata=dict(t.get("metadata", {})),
112
+ )
113
+ for i, t in enumerate(dataset.get("tasks", []))
114
+ ]
115
+
116
+ env_vars = {
117
+ str(name): "<redacted>" # values are never stored; names are the signal
118
+ for name in (environment.get("env") or {})
119
+ }
120
+ mounts = [
121
+ Mount(
122
+ path=str(m.get("path", "")),
123
+ mode=str(m.get("mode", "ro")),
124
+ agent_access=str(m.get("agent_access", "read")),
125
+ )
126
+ for m in (environment.get("mounts") or [])
127
+ ]
128
+
129
+ verifier = grader_cfg.get("verifier") or {}
130
+ judge = grader_cfg.get("judge") or {}
131
+ grader = Grader(
132
+ kind=str(grader_cfg.get("kind", "script")),
133
+ verifier_path=verifier.get("path"),
134
+ verifier_writable_by_agent=bool(verifier.get("writable_by_agent", False)),
135
+ accepts_empty_output=bool(grader_cfg.get("accepts_empty_output", False)),
136
+ tests=[str(t) for t in (grader_cfg.get("tests") or [])],
137
+ judge_model=judge.get("model"),
138
+ judge_family=judge.get("family"),
139
+ protocol=judge.get("protocol"),
140
+ counterbalanced=judge.get("counterbalanced"),
141
+ temperature=judge.get("temperature"),
142
+ repeats=int(judge.get("repeats", 1)),
143
+ rubric_criteria=[str(c) for c in (judge.get("rubric_criteria") or [])],
144
+ scale_anchors={str(k): str(v) for k, v in (judge.get("scale_anchors") or {}).items()},
145
+ )
146
+
147
+ attempts = [
148
+ Attempt(
149
+ task_id=str(a.get("task_id", "")),
150
+ status=str(a.get("status", "error")),
151
+ score=a.get("score"),
152
+ tool_calls=int(a.get("tool_calls", 0)),
153
+ actions=[str(x) for x in (a.get("actions") or [])],
154
+ tokens_in=a.get("tokens_in"),
155
+ tokens_out=a.get("tokens_out"),
156
+ latency_s=a.get("latency_s"),
157
+ tries=int(a.get("tries", 1)),
158
+ empty_submission=bool(a.get("empty_submission", False)),
159
+ )
160
+ for a in (run.get("attempts") or [])
161
+ ]
162
+
163
+ unsupported: list[str] = []
164
+
165
+ judge_run = bundle.get("judge_run.json", {})
166
+ judgments: list[Judgment] = []
167
+ for j in judge_run.get("judgments", []):
168
+ candidates = sorted(str(c) for c in (j.get("candidates") or []))
169
+ judgments.append(
170
+ Judgment(
171
+ task_id=str(j.get("task_id", "")),
172
+ candidates=candidates,
173
+ presentation_order=[str(c) for c in (j.get("presentation_order") or candidates)],
174
+ winner=j.get("winner"),
175
+ scores={str(k): float(v) for k, v in (j.get("scores") or {}).items()},
176
+ lengths={str(k): int(v) for k, v in (j.get("lengths") or {}).items()},
177
+ repeat_index=int(j.get("repeat_index", 0)),
178
+ )
179
+ )
180
+ if any("rationale" in j for j in judge_run.get("judgments", [])):
181
+ # Raw judge text is redacted at the boundary by design, not parsed.
182
+ # Reported here so the redaction is explicit, not silent.
183
+ unsupported.append("judge_run.json:judgments[].rationale (redacted: raw judge text not stored)")
184
+ grader.reference_labels = {
185
+ str(k): str(v) for k, v in (judge_run.get("reference_labels") or {}).items()
186
+ }
187
+ for fname, content in bundle.items():
188
+ if fname in ("root", "digests", "extra_files"):
189
+ continue
190
+ if isinstance(content, dict):
191
+ known = {
192
+ "dataset.json": {"schema_version", "eval_id", "tasks"},
193
+ "environment.json": {"env", "mounts", "notes"},
194
+ "grader.json": {"kind", "verifier", "accepts_empty_output", "tests", "judge"},
195
+ "run.json": {"solver", "attempts", "notes"},
196
+ "judge_run.json": {"judge", "judgments", "reference_labels", "notes"},
197
+ }.get(fname, set())
198
+ for key in content:
199
+ if key not in known:
200
+ unsupported.append(f"{fname}:{key}")
201
+
202
+ return IntegrityModel(
203
+ eval_id=str(dataset.get("eval_id", root.name)),
204
+ adapter_name=self.name,
205
+ adapter_version=self.version,
206
+ tasks=tasks,
207
+ environment=Environment(env_vars=env_vars, mounts=mounts),
208
+ grader=grader,
209
+ attempts=attempts,
210
+ judgments=judgments,
211
+ unsupported=unsupported,
212
+ digests=dict(bundle.get("digests", {})),
213
+ )
@@ -0,0 +1,357 @@
1
+ """Promptfoo adapter (v0.5).
2
+
3
+ Reads a Promptfoo eval artifact directory:
4
+
5
+ eval-artifact/
6
+ promptfooconfig.yaml (or .yml) -- prompts, providers, tests, assertions, env
7
+ results.json (optional) -- JSON export from `promptfoo eval -o results.json`
8
+
9
+ Minimum viable input per the spec: promptfooconfig.yaml plus the JSON export.
10
+ A config-only audit is valid but incomplete (no runs to analyze). When only
11
+ results.json is present, the config embedded in its envelope is used and the
12
+ substitution is reported explicitly.
13
+
14
+ Field names follow promptfoo's documented output schema (OutputFile envelope
15
+ with results.version == 3; per-result success/score/error, vars,
16
+ response.output, response.tokenUsage.{prompt,completion,total}, latencyMs,
17
+ gradingResult.{pass,score,reason,componentResults}). Anything else lands in
18
+ `unsupported`, never silently dropped.
19
+
20
+ This is a read-only translation layer: it parses the config and the recorded
21
+ results. It never executes an eval.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import hashlib
26
+ import json
27
+ from pathlib import Path
28
+ from typing import Any
29
+
30
+ from ..model import (
31
+ Attempt,
32
+ Confidence,
33
+ Environment,
34
+ Grader,
35
+ IntegrityModel,
36
+ TaskSample,
37
+ )
38
+ from . import AuditError, register
39
+
40
+ ADAPTER_NAME = "promptfoo"
41
+ ADAPTER_VERSION = "0.6.0"
42
+ # Promptfoo's documented results envelope version (OutputFile.results.version).
43
+ RESULTS_VERSION = 3
44
+
45
+ CONFIG_NAMES = ("promptfooconfig.yaml", "promptfooconfig.yml")
46
+ RESULTS_NAME = "results.json"
47
+
48
+ # Top-level config keys the adapter understands. Everything else is reported
49
+ # in `unsupported` so coverage gaps are explicit, not silent.
50
+ KNOWN_CONFIG_KEYS = {
51
+ "description",
52
+ "env",
53
+ "prompts",
54
+ "providers",
55
+ "defaultTest",
56
+ "scenarios",
57
+ "tests",
58
+ "metadata",
59
+ "outputPath",
60
+ "writeLatestResults",
61
+ "sharing",
62
+ }
63
+
64
+
65
+ def _envelope_has_config(results_path: Path) -> bool:
66
+ """Cheap peek for detect(): does the results envelope carry a config dict?"""
67
+ try:
68
+ envelope = json.loads(results_path.read_text(encoding="utf-8"))
69
+ except (OSError, ValueError):
70
+ return False
71
+ return isinstance(envelope, dict) and isinstance(envelope.get("config"), dict)
72
+
73
+
74
+ def _read_json(path: Path) -> Any:
75
+ try:
76
+ return json.loads(path.read_text(encoding="utf-8"))
77
+ except FileNotFoundError as exc:
78
+ raise AuditError(f"missing required file: {path}") from exc
79
+ except json.JSONDecodeError as exc:
80
+ raise AuditError(f"invalid JSON in {path}: {exc}") from exc
81
+
82
+
83
+ def _read_yaml(path: Path) -> Any:
84
+ try:
85
+ import yaml
86
+ except ImportError as exc:
87
+ raise AuditError(
88
+ "promptfoo adapter needs PyYAML to parse promptfooconfig.yaml "
89
+ "(pip install 'evalwarden' again, or pip install pyyaml)"
90
+ ) from exc
91
+ try:
92
+ return yaml.safe_load(path.read_text(encoding="utf-8"))
93
+ except FileNotFoundError as exc:
94
+ raise AuditError(f"missing required file: {path}") from exc
95
+ except yaml.YAMLError as exc:
96
+ raise AuditError(f"invalid YAML in {path}: {exc}") from exc
97
+
98
+
99
+ def _digest(path: Path) -> str:
100
+ return hashlib.sha256(path.read_bytes()).hexdigest()
101
+
102
+
103
+ def _config_path(path: Path) -> Path | None:
104
+ for name in CONFIG_NAMES:
105
+ candidate = path / name
106
+ if candidate.is_file():
107
+ return candidate
108
+ return None
109
+
110
+
111
+ @register
112
+ class PromptfooAdapter:
113
+ name = ADAPTER_NAME
114
+ version = ADAPTER_VERSION
115
+
116
+ def detect(self, path: Path) -> Confidence:
117
+ if not path.is_dir():
118
+ return Confidence.LOW
119
+ has_config = _config_path(path) is not None
120
+ results_path = path / RESULTS_NAME
121
+ has_results = results_path.is_file()
122
+ if has_config and has_results:
123
+ return Confidence.HIGH
124
+ if has_config:
125
+ # Config-only audit: valid but incomplete (no runs to analyze).
126
+ return Confidence.MEDIUM
127
+ if has_results and _envelope_has_config(results_path):
128
+ # No promptfooconfig.yaml, but the results envelope carries the
129
+ # config it was produced from.
130
+ return Confidence.MEDIUM
131
+ return Confidence.LOW
132
+
133
+ def collect(self, path: Path) -> dict:
134
+ """Read-only: files are opened for reading and never modified."""
135
+ bundle: dict[str, Any] = {"root": path}
136
+ digests: dict[str, str] = {}
137
+ config_path = _config_path(path)
138
+ if config_path is not None:
139
+ bundle["config"] = _read_yaml(config_path)
140
+ bundle["config_file"] = config_path.name
141
+ digests[config_path.name] = _digest(config_path)
142
+ results_path = path / RESULTS_NAME
143
+ if results_path.is_file():
144
+ bundle["results"] = _read_json(results_path)
145
+ digests[RESULTS_NAME] = _digest(results_path)
146
+ bundle["digests"] = digests
147
+ return bundle
148
+
149
+ def normalize(self, bundle: dict) -> IntegrityModel:
150
+ root: Path = bundle["root"]
151
+ config = bundle.get("config") or {}
152
+ config_file = bundle.get("config_file", "promptfooconfig.yaml")
153
+ envelope = bundle.get("results") or {}
154
+ inner = envelope.get("results") or {}
155
+
156
+ if not isinstance(config, dict):
157
+ raise AuditError(f"{config_file}: top-level mapping expected")
158
+
159
+ unsupported: list[str] = []
160
+ if "config" not in bundle and envelope.get("config") is not None:
161
+ # Graceful degradation: no promptfooconfig.yaml, but the results
162
+ # envelope carries the eval config it was produced from.
163
+ config = envelope["config"]
164
+ unsupported.append(
165
+ f"{RESULTS_NAME}: promptfooconfig.yaml absent; "
166
+ "using the config embedded in the results envelope"
167
+ )
168
+
169
+ version = inner.get("version")
170
+ if "results" in bundle and version != RESULTS_VERSION:
171
+ raise AuditError(
172
+ f"{RESULTS_NAME}: unsupported results version {version!r} "
173
+ f"(this adapter understands version {RESULTS_VERSION})"
174
+ )
175
+
176
+ eval_id = (
177
+ config.get("description")
178
+ or envelope.get("evalId")
179
+ or root.name
180
+ )
181
+
182
+ tasks = _normalize_tests(config, unsupported)
183
+ env_vars = {
184
+ str(name): "<redacted>" # values are never stored; names are the signal
185
+ for name in (config.get("env") or {})
186
+ }
187
+ grader = _normalize_grader(config)
188
+ attempts = _normalize_attempts(inner)
189
+
190
+ for key in config:
191
+ if key not in KNOWN_CONFIG_KEYS:
192
+ unsupported.append(f"{config_file}:{key}")
193
+ if config.get("scenarios"):
194
+ unsupported.append(
195
+ f"{config_file}:scenarios (scenario-generated tests not expanded)"
196
+ )
197
+ for test in _inline_tests(config):
198
+ for key in test:
199
+ if key not in {
200
+ "description", "vars", "assert", "options",
201
+ "threshold", "metadata",
202
+ }:
203
+ unsupported.append(f"{config_file}:tests[]:{key}")
204
+
205
+ return IntegrityModel(
206
+ eval_id=str(eval_id),
207
+ adapter_name=self.name,
208
+ adapter_version=self.version,
209
+ tasks=tasks,
210
+ environment=Environment(env_vars=env_vars, mounts=[]),
211
+ grader=grader,
212
+ attempts=attempts,
213
+ unsupported=unsupported,
214
+ digests=dict(bundle.get("digests", {})),
215
+ # Checks cite canonical artifact names; point them at the real files.
216
+ file_aliases={
217
+ "environment.json": config_file,
218
+ "grader.json": config_file,
219
+ "judge_run.json": config_file,
220
+ "run.json": RESULTS_NAME,
221
+ },
222
+ )
223
+
224
+
225
+ def _inline_tests(config: dict) -> list[dict]:
226
+ """Tests declared inline in the config. file:// and other references are
227
+ left for `unsupported`: expanding them would pull arbitrary files into the
228
+ audit, and the declared config is the contract under review."""
229
+ tests = config.get("tests") or []
230
+ if isinstance(tests, dict):
231
+ tests = [tests]
232
+ return [t for t in tests if isinstance(t, dict)]
233
+
234
+
235
+ def _normalize_tests(config: dict, unsupported: list[str]) -> list[TaskSample]:
236
+ tasks: list[TaskSample] = []
237
+ raw_tests = config.get("tests")
238
+ if isinstance(raw_tests, str):
239
+ unsupported.append(f"tests: {raw_tests} (external test file not expanded)")
240
+ return tasks
241
+ if isinstance(raw_tests, list) and any(
242
+ isinstance(t, str) for t in raw_tests
243
+ ):
244
+ unsupported.append("tests[]: file:// references not expanded")
245
+ for i, test in enumerate(_inline_tests(config)):
246
+ vars_ = test.get("vars") or {}
247
+ assertions = [str(a.get("type", "?")) for a in (test.get("assert") or []) if isinstance(a, dict)]
248
+ tasks.append(
249
+ TaskSample(
250
+ id=str(test.get("description") or f"test-{i}"),
251
+ prompt=str(test.get("description") or ""),
252
+ metadata={
253
+ "vars": sorted(str(k) for k in vars_),
254
+ "assertions": assertions,
255
+ },
256
+ )
257
+ )
258
+ return tasks
259
+
260
+
261
+ def _assertions(config: dict) -> list[dict]:
262
+ """All assertions: shared defaultTest ones plus per-test ones."""
263
+ found: list[dict] = []
264
+ default = config.get("defaultTest") or {}
265
+ if isinstance(default, dict):
266
+ found.extend(a for a in (default.get("assert") or []) if isinstance(a, dict))
267
+ for test in _inline_tests(config):
268
+ found.extend(a for a in (test.get("assert") or []) if isinstance(a, dict))
269
+ return found
270
+
271
+
272
+ def _provider_id(ref: Any) -> str | None:
273
+ if isinstance(ref, str):
274
+ return ref
275
+ if isinstance(ref, dict) and ref.get("id"):
276
+ return str(ref["id"])
277
+ return None
278
+
279
+
280
+ def _grader_provider(config: dict) -> tuple[str | None, float | None]:
281
+ """The model grading llm-rubric assertions: per-test options.provider wins,
282
+ then defaultTest.options.provider. Returns (provider_id, temperature)."""
283
+ for test in _inline_tests(config):
284
+ options = test.get("options") or {}
285
+ if isinstance(options, dict) and options.get("provider") is not None:
286
+ return _provider_id(options["provider"]), _provider_temperature(options["provider"])
287
+ default = config.get("defaultTest") or {}
288
+ options = default.get("options") or {} if isinstance(default, dict) else {}
289
+ if isinstance(options, dict) and options.get("provider") is not None:
290
+ return _provider_id(options["provider"]), _provider_temperature(options["provider"])
291
+ return None, None
292
+
293
+
294
+ def _provider_temperature(ref: Any) -> float | None:
295
+ if isinstance(ref, dict):
296
+ cfg = ref.get("config") or {}
297
+ if isinstance(cfg, dict) and isinstance(cfg.get("temperature"), (int, float)):
298
+ return float(cfg["temperature"])
299
+ return None
300
+
301
+
302
+ def _normalize_grader(config: dict) -> Grader:
303
+ assertions = _assertions(config)
304
+ types = sorted({str(a.get("type", "?")) for a in assertions})
305
+ is_judge = "llm-rubric" in types
306
+ grader = Grader(
307
+ kind="judge" if is_judge else "script",
308
+ tests=types,
309
+ )
310
+ if is_judge:
311
+ provider_id, temperature = _grader_provider(config)
312
+ grader.judge_model = provider_id
313
+ grader.temperature = temperature
314
+ # Each llm-rubric value is the rubric; promptfoo has no anchored scale
315
+ # levels, which JUDGE-001 reports as an unanchored rubric.
316
+ grader.rubric_criteria = [
317
+ str(a["value"]) for a in assertions
318
+ if str(a.get("type")) == "llm-rubric" and a.get("value")
319
+ ]
320
+ return grader
321
+
322
+
323
+ def _normalize_attempts(inner: dict) -> list[Attempt]:
324
+ attempts: list[Attempt] = []
325
+ rows = inner.get("results") or []
326
+ for i, row in enumerate(rows):
327
+ if not isinstance(row, dict):
328
+ continue
329
+ error = row.get("error")
330
+ success = row.get("success")
331
+ status = "error" if error else ("pass" if success else "fail")
332
+ response = row.get("response") or {}
333
+ usage = response.get("tokenUsage") or {}
334
+ latency_ms = row.get("latencyMs")
335
+ output = response.get("output")
336
+ attempts.append(
337
+ Attempt(
338
+ task_id=str(row.get("description") or f"test-{row.get('testIdx', i)}"),
339
+ status=status,
340
+ score=_to_float(row.get("score")),
341
+ tokens_in=_to_int(usage.get("prompt")),
342
+ tokens_out=_to_int(usage.get("completion")),
343
+ latency_s=(float(latency_ms) / 1000.0) if isinstance(latency_ms, (int, float)) else None,
344
+ # An explicitly empty model output scored as a pass is the
345
+ # GRAD-002 signal; anything else is just a short answer.
346
+ empty_submission=isinstance(output, str) and output == "" and status == "pass",
347
+ )
348
+ )
349
+ return attempts
350
+
351
+
352
+ def _to_float(value: Any) -> float | None:
353
+ return float(value) if isinstance(value, (int, float)) else None
354
+
355
+
356
+ def _to_int(value: Any) -> int | None:
357
+ return int(value) if isinstance(value, (int, float)) else None
@@ -0,0 +1,27 @@
1
+ """Rule registry. v0.3 ships deterministic, high-precision checks only.
2
+
3
+ Sequencing rule: earn trust with deterministic evidence before adding
4
+ probabilistic signals. Every finding carries a confidence label; a linter that
5
+ cries contamination on a clean eval is worse than no auditor.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from .base import Check
10
+ from .cost import CHECKS as COST_CHECKS
11
+ from .env_leakage import EnvLeakageCheck
12
+ from .grader import EmptyPathCheck, VerifierWritableCheck
13
+ from .judge import CHECKS as JUDGE_CHECKS
14
+
15
+ REGISTRY: list[Check] = [
16
+ EnvLeakageCheck(), # ENV-001
17
+ VerifierWritableCheck(), # GRAD-001
18
+ EmptyPathCheck(), # GRAD-002
19
+ *COST_CHECKS, # COST-001 .. COST-004
20
+ *JUDGE_CHECKS, # JUDGE-001 .. JUDGE-006
21
+ ]
22
+
23
+ BY_ID: dict[str, Check] = {check.meta.id: check for check in REGISTRY}
24
+
25
+
26
+ def get_check(check_id: str) -> Check | None:
27
+ return BY_ID.get(check_id.upper())