evalwarden 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalwarden/__init__.py +3 -0
- evalwarden/adapters/__init__.py +70 -0
- evalwarden/adapters/inspect_ai.py +213 -0
- evalwarden/adapters/promptfoo.py +357 -0
- evalwarden/checks/__init__.py +27 -0
- evalwarden/checks/base.py +35 -0
- evalwarden/checks/cost.py +308 -0
- evalwarden/checks/env_leakage.py +137 -0
- evalwarden/checks/grader.py +112 -0
- evalwarden/checks/judge.py +419 -0
- evalwarden/cli.py +305 -0
- evalwarden/demo/cost_clean/README.md +11 -0
- evalwarden/demo/cost_clean/dataset.json +1 -0
- evalwarden/demo/cost_clean/environment.json +1 -0
- evalwarden/demo/cost_clean/grader.json +1 -0
- evalwarden/demo/cost_clean/run.json +1 -0
- evalwarden/demo/cost_wasteful/README.md +19 -0
- evalwarden/demo/cost_wasteful/dataset.json +1 -0
- evalwarden/demo/cost_wasteful/environment.json +1 -0
- evalwarden/demo/cost_wasteful/grader.json +1 -0
- evalwarden/demo/cost_wasteful/run.json +1 -0
- evalwarden/demo/hardened/README.md +14 -0
- evalwarden/demo/hardened/dataset.json +1 -0
- evalwarden/demo/hardened/environment.json +1 -0
- evalwarden/demo/hardened/grader.json +1 -0
- evalwarden/demo/hardened/run.json +1 -0
- evalwarden/demo/hardened/run_cheat.json +1 -0
- evalwarden/demo/judge_bad/README.md +3 -0
- evalwarden/demo/judge_bad/dataset.json +1 -0
- evalwarden/demo/judge_bad/environment.json +5 -0
- evalwarden/demo/judge_bad/grader.json +16 -0
- evalwarden/demo/judge_bad/judge_run.json +449 -0
- evalwarden/demo/judge_bad/run.json +150 -0
- evalwarden/demo/judge_clean/README.md +3 -0
- evalwarden/demo/judge_clean/dataset.json +1 -0
- evalwarden/demo/judge_clean/environment.json +5 -0
- evalwarden/demo/judge_clean/grader.json +21 -0
- evalwarden/demo/judge_clean/judge_run.json +448 -0
- evalwarden/demo/judge_clean/run.json +150 -0
- evalwarden/demo/leaky/README.md +22 -0
- evalwarden/demo/leaky/dataset.json +1 -0
- evalwarden/demo/leaky/environment.json +1 -0
- evalwarden/demo/leaky/gold/patch-task-001.diff +5 -0
- evalwarden/demo/leaky/gold/patch-task-002.diff +5 -0
- evalwarden/demo/leaky/gold/patch-task-003.diff +5 -0
- evalwarden/demo/leaky/gold_map.json +1 -0
- evalwarden/demo/leaky/grader.json +1 -0
- evalwarden/demo/leaky/run.json +1 -0
- evalwarden/demo/promptfoo_bad/README.md +14 -0
- evalwarden/demo/promptfoo_bad/promptfooconfig.yaml +30 -0
- evalwarden/demo/promptfoo_bad/results.json +1 -0
- evalwarden/demo/promptfoo_clean/README.md +13 -0
- evalwarden/demo/promptfoo_clean/promptfooconfig.yaml +27 -0
- evalwarden/demo/promptfoo_clean/results.json +1 -0
- evalwarden/engine.py +104 -0
- evalwarden/model.py +200 -0
- evalwarden/reporters/__init__.py +12 -0
- evalwarden/reporters/html.py +182 -0
- evalwarden/reporters/report_card.py +297 -0
- evalwarden/reporters/terminal.py +90 -0
- evalwarden-0.6.0.dist-info/METADATA +252 -0
- evalwarden-0.6.0.dist-info/RECORD +66 -0
- evalwarden-0.6.0.dist-info/WHEEL +5 -0
- evalwarden-0.6.0.dist-info/entry_points.txt +2 -0
- evalwarden-0.6.0.dist-info/licenses/LICENSE +21 -0
- evalwarden-0.6.0.dist-info/top_level.txt +1 -0
evalwarden/__init__.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Adapter protocol and registry.
|
|
2
|
+
|
|
3
|
+
Adapters are the ONLY harness-specific code in the project. They discover and
|
|
4
|
+
translate; the core never becomes a harness. Each adapter:
|
|
5
|
+
|
|
6
|
+
- is read-only (never mutates the input files),
|
|
7
|
+
- works offline (no network in default mode),
|
|
8
|
+
- preserves source locations,
|
|
9
|
+
- pins its own version and the schema versions it understands,
|
|
10
|
+
- reports unsupported fields explicitly instead of silently dropping them.
|
|
11
|
+
|
|
12
|
+
v0.1 ships one adapter: Inspect AI style eval artifacts. Promptfoo, Harbor,
|
|
13
|
+
and BrowserGym adapters plug into REGISTRY later with the same contract.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Protocol
|
|
19
|
+
|
|
20
|
+
from ..model import Confidence, IntegrityModel
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class AuditError(Exception):
|
|
24
|
+
"""The audit could not complete (exit code 2)."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Adapter(Protocol):
|
|
28
|
+
name: str
|
|
29
|
+
version: str
|
|
30
|
+
|
|
31
|
+
def detect(self, path: Path) -> Confidence:
|
|
32
|
+
"""How confident are we that this adapter understands `path`?"""
|
|
33
|
+
...
|
|
34
|
+
|
|
35
|
+
def collect(self, path: Path) -> dict:
|
|
36
|
+
"""Read-only collection of raw evidence from the artifact."""
|
|
37
|
+
...
|
|
38
|
+
|
|
39
|
+
def normalize(self, bundle: dict) -> IntegrityModel:
|
|
40
|
+
"""Translate raw evidence into the framework-neutral integrity model."""
|
|
41
|
+
...
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
REGISTRY: list[Adapter] = []
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def register(adapter: Adapter) -> Adapter:
|
|
48
|
+
# Accept either an instance or a class (instantiated here) so adapters can
|
|
49
|
+
# use either `@register` on the class or `register(MyAdapter())`.
|
|
50
|
+
REGISTRY.append(adapter() if isinstance(adapter, type) else adapter)
|
|
51
|
+
return adapter
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def autodetect(path: Path) -> Adapter:
|
|
55
|
+
"""Pick the most confident adapter for `path`, or raise AuditError."""
|
|
56
|
+
if not REGISTRY:
|
|
57
|
+
raise AuditError("no adapters registered")
|
|
58
|
+
ranked = sorted(
|
|
59
|
+
((adapter.detect(path), adapter) for adapter in REGISTRY),
|
|
60
|
+
key=lambda item: (item[0] == Confidence.HIGH, item[0] == Confidence.MEDIUM),
|
|
61
|
+
reverse=True,
|
|
62
|
+
)
|
|
63
|
+
confidence, adapter = ranked[0]
|
|
64
|
+
if confidence == Confidence.LOW:
|
|
65
|
+
raise AuditError(
|
|
66
|
+
f"no adapter recognizes {path} "
|
|
67
|
+
f"(best guess: {adapter.name}, confidence=low). "
|
|
68
|
+
"Expected an eval artifact directory (see demo/leaky for the layout)."
|
|
69
|
+
)
|
|
70
|
+
return adapter
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Inspect AI adapter (v0.1).
|
|
2
|
+
|
|
3
|
+
Reads an Inspect-style eval artifact directory -- the v0.1 normalized input
|
|
4
|
+
format, modeled on Inspect's Task / dataset / scorer / .eval-log concepts:
|
|
5
|
+
|
|
6
|
+
eval-artifact/
|
|
7
|
+
dataset.json samples: [{id, prompt, metadata}]
|
|
8
|
+
environment.json env vars visible to the solver, mounts
|
|
9
|
+
grader.json verifier config, pass conditions, tests
|
|
10
|
+
run.json per-task attempts with status, usage, actions
|
|
11
|
+
judge_run.json (optional) model-judge config, judgments, reference labels
|
|
12
|
+
|
|
13
|
+
This is a read-only translation layer. Full-fidelity parsing of real Inspect
|
|
14
|
+
`.eval` logs is a later milestone; the adapter pins the schema version it
|
|
15
|
+
understands and fails clearly on anything else.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
from ..model import (
|
|
25
|
+
Attempt,
|
|
26
|
+
Confidence,
|
|
27
|
+
Environment,
|
|
28
|
+
Grader,
|
|
29
|
+
IntegrityModel,
|
|
30
|
+
Judgment,
|
|
31
|
+
Mount,
|
|
32
|
+
TaskSample,
|
|
33
|
+
)
|
|
34
|
+
from . import AuditError, register
|
|
35
|
+
|
|
36
|
+
ADAPTER_NAME = "inspect"
|
|
37
|
+
ADAPTER_VERSION = "0.1.0"
|
|
38
|
+
SCHEMA_VERSION = "evalwarden-artifact-v1"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _read_json(path: Path) -> Any:
|
|
42
|
+
try:
|
|
43
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
44
|
+
except FileNotFoundError as exc:
|
|
45
|
+
raise AuditError(f"missing required file: {path}") from exc
|
|
46
|
+
except json.JSONDecodeError as exc:
|
|
47
|
+
raise AuditError(f"invalid JSON in {path}: {exc}") from exc
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _digest(path: Path) -> str:
|
|
51
|
+
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@register
|
|
55
|
+
class InspectAdapter:
|
|
56
|
+
name = ADAPTER_NAME
|
|
57
|
+
version = ADAPTER_VERSION
|
|
58
|
+
|
|
59
|
+
def detect(self, path: Path) -> Confidence:
|
|
60
|
+
if not path.is_dir():
|
|
61
|
+
return Confidence.LOW
|
|
62
|
+
has_dataset = (path / "dataset.json").is_file()
|
|
63
|
+
has_grader = (path / "grader.json").is_file()
|
|
64
|
+
if has_dataset and has_grader:
|
|
65
|
+
return Confidence.HIGH
|
|
66
|
+
if has_dataset:
|
|
67
|
+
return Confidence.MEDIUM
|
|
68
|
+
return Confidence.LOW
|
|
69
|
+
|
|
70
|
+
def collect(self, path: Path) -> dict:
|
|
71
|
+
"""Read-only: files are opened for reading and never modified."""
|
|
72
|
+
bundle: dict[str, Any] = {"root": path}
|
|
73
|
+
digests: dict[str, str] = {}
|
|
74
|
+
for fname in (
|
|
75
|
+
"dataset.json",
|
|
76
|
+
"environment.json",
|
|
77
|
+
"grader.json",
|
|
78
|
+
"run.json",
|
|
79
|
+
"judge_run.json",
|
|
80
|
+
):
|
|
81
|
+
fpath = path / fname
|
|
82
|
+
if fpath.is_file():
|
|
83
|
+
bundle[fname] = _read_json(fpath)
|
|
84
|
+
digests[fname] = _digest(fpath)
|
|
85
|
+
for fname in ("gold_map.json",):
|
|
86
|
+
fpath = path / fname
|
|
87
|
+
if fpath.is_file():
|
|
88
|
+
# Recorded for the digest manifest only; never parsed for content
|
|
89
|
+
# beyond its existence (it is the leak, not the evidence).
|
|
90
|
+
digests[fname] = _digest(fpath)
|
|
91
|
+
bundle.setdefault("extra_files", []).append(fname)
|
|
92
|
+
bundle["digests"] = digests
|
|
93
|
+
return bundle
|
|
94
|
+
|
|
95
|
+
def normalize(self, bundle: dict) -> IntegrityModel:
|
|
96
|
+
root: Path = bundle["root"]
|
|
97
|
+
dataset = bundle.get("dataset.json", {})
|
|
98
|
+
environment = bundle.get("environment.json", {})
|
|
99
|
+
grader_cfg = bundle.get("grader.json", {})
|
|
100
|
+
run = bundle.get("run.json", {})
|
|
101
|
+
|
|
102
|
+
if dataset.get("schema_version", SCHEMA_VERSION) != SCHEMA_VERSION and "tasks" not in dataset:
|
|
103
|
+
raise AuditError(
|
|
104
|
+
f"unsupported dataset schema (expected {SCHEMA_VERSION!r} with a 'tasks' list)"
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
tasks = [
|
|
108
|
+
TaskSample(
|
|
109
|
+
id=str(t.get("id", f"task-{i}")),
|
|
110
|
+
prompt=str(t.get("prompt", "")),
|
|
111
|
+
metadata=dict(t.get("metadata", {})),
|
|
112
|
+
)
|
|
113
|
+
for i, t in enumerate(dataset.get("tasks", []))
|
|
114
|
+
]
|
|
115
|
+
|
|
116
|
+
env_vars = {
|
|
117
|
+
str(name): "<redacted>" # values are never stored; names are the signal
|
|
118
|
+
for name in (environment.get("env") or {})
|
|
119
|
+
}
|
|
120
|
+
mounts = [
|
|
121
|
+
Mount(
|
|
122
|
+
path=str(m.get("path", "")),
|
|
123
|
+
mode=str(m.get("mode", "ro")),
|
|
124
|
+
agent_access=str(m.get("agent_access", "read")),
|
|
125
|
+
)
|
|
126
|
+
for m in (environment.get("mounts") or [])
|
|
127
|
+
]
|
|
128
|
+
|
|
129
|
+
verifier = grader_cfg.get("verifier") or {}
|
|
130
|
+
judge = grader_cfg.get("judge") or {}
|
|
131
|
+
grader = Grader(
|
|
132
|
+
kind=str(grader_cfg.get("kind", "script")),
|
|
133
|
+
verifier_path=verifier.get("path"),
|
|
134
|
+
verifier_writable_by_agent=bool(verifier.get("writable_by_agent", False)),
|
|
135
|
+
accepts_empty_output=bool(grader_cfg.get("accepts_empty_output", False)),
|
|
136
|
+
tests=[str(t) for t in (grader_cfg.get("tests") or [])],
|
|
137
|
+
judge_model=judge.get("model"),
|
|
138
|
+
judge_family=judge.get("family"),
|
|
139
|
+
protocol=judge.get("protocol"),
|
|
140
|
+
counterbalanced=judge.get("counterbalanced"),
|
|
141
|
+
temperature=judge.get("temperature"),
|
|
142
|
+
repeats=int(judge.get("repeats", 1)),
|
|
143
|
+
rubric_criteria=[str(c) for c in (judge.get("rubric_criteria") or [])],
|
|
144
|
+
scale_anchors={str(k): str(v) for k, v in (judge.get("scale_anchors") or {}).items()},
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
attempts = [
|
|
148
|
+
Attempt(
|
|
149
|
+
task_id=str(a.get("task_id", "")),
|
|
150
|
+
status=str(a.get("status", "error")),
|
|
151
|
+
score=a.get("score"),
|
|
152
|
+
tool_calls=int(a.get("tool_calls", 0)),
|
|
153
|
+
actions=[str(x) for x in (a.get("actions") or [])],
|
|
154
|
+
tokens_in=a.get("tokens_in"),
|
|
155
|
+
tokens_out=a.get("tokens_out"),
|
|
156
|
+
latency_s=a.get("latency_s"),
|
|
157
|
+
tries=int(a.get("tries", 1)),
|
|
158
|
+
empty_submission=bool(a.get("empty_submission", False)),
|
|
159
|
+
)
|
|
160
|
+
for a in (run.get("attempts") or [])
|
|
161
|
+
]
|
|
162
|
+
|
|
163
|
+
unsupported: list[str] = []
|
|
164
|
+
|
|
165
|
+
judge_run = bundle.get("judge_run.json", {})
|
|
166
|
+
judgments: list[Judgment] = []
|
|
167
|
+
for j in judge_run.get("judgments", []):
|
|
168
|
+
candidates = sorted(str(c) for c in (j.get("candidates") or []))
|
|
169
|
+
judgments.append(
|
|
170
|
+
Judgment(
|
|
171
|
+
task_id=str(j.get("task_id", "")),
|
|
172
|
+
candidates=candidates,
|
|
173
|
+
presentation_order=[str(c) for c in (j.get("presentation_order") or candidates)],
|
|
174
|
+
winner=j.get("winner"),
|
|
175
|
+
scores={str(k): float(v) for k, v in (j.get("scores") or {}).items()},
|
|
176
|
+
lengths={str(k): int(v) for k, v in (j.get("lengths") or {}).items()},
|
|
177
|
+
repeat_index=int(j.get("repeat_index", 0)),
|
|
178
|
+
)
|
|
179
|
+
)
|
|
180
|
+
if any("rationale" in j for j in judge_run.get("judgments", [])):
|
|
181
|
+
# Raw judge text is redacted at the boundary by design, not parsed.
|
|
182
|
+
# Reported here so the redaction is explicit, not silent.
|
|
183
|
+
unsupported.append("judge_run.json:judgments[].rationale (redacted: raw judge text not stored)")
|
|
184
|
+
grader.reference_labels = {
|
|
185
|
+
str(k): str(v) for k, v in (judge_run.get("reference_labels") or {}).items()
|
|
186
|
+
}
|
|
187
|
+
for fname, content in bundle.items():
|
|
188
|
+
if fname in ("root", "digests", "extra_files"):
|
|
189
|
+
continue
|
|
190
|
+
if isinstance(content, dict):
|
|
191
|
+
known = {
|
|
192
|
+
"dataset.json": {"schema_version", "eval_id", "tasks"},
|
|
193
|
+
"environment.json": {"env", "mounts", "notes"},
|
|
194
|
+
"grader.json": {"kind", "verifier", "accepts_empty_output", "tests", "judge"},
|
|
195
|
+
"run.json": {"solver", "attempts", "notes"},
|
|
196
|
+
"judge_run.json": {"judge", "judgments", "reference_labels", "notes"},
|
|
197
|
+
}.get(fname, set())
|
|
198
|
+
for key in content:
|
|
199
|
+
if key not in known:
|
|
200
|
+
unsupported.append(f"{fname}:{key}")
|
|
201
|
+
|
|
202
|
+
return IntegrityModel(
|
|
203
|
+
eval_id=str(dataset.get("eval_id", root.name)),
|
|
204
|
+
adapter_name=self.name,
|
|
205
|
+
adapter_version=self.version,
|
|
206
|
+
tasks=tasks,
|
|
207
|
+
environment=Environment(env_vars=env_vars, mounts=mounts),
|
|
208
|
+
grader=grader,
|
|
209
|
+
attempts=attempts,
|
|
210
|
+
judgments=judgments,
|
|
211
|
+
unsupported=unsupported,
|
|
212
|
+
digests=dict(bundle.get("digests", {})),
|
|
213
|
+
)
|
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
"""Promptfoo adapter (v0.5).
|
|
2
|
+
|
|
3
|
+
Reads a Promptfoo eval artifact directory:
|
|
4
|
+
|
|
5
|
+
eval-artifact/
|
|
6
|
+
promptfooconfig.yaml (or .yml) -- prompts, providers, tests, assertions, env
|
|
7
|
+
results.json (optional) -- JSON export from `promptfoo eval -o results.json`
|
|
8
|
+
|
|
9
|
+
Minimum viable input per the spec: promptfooconfig.yaml plus the JSON export.
|
|
10
|
+
A config-only audit is valid but incomplete (no runs to analyze). When only
|
|
11
|
+
results.json is present, the config embedded in its envelope is used and the
|
|
12
|
+
substitution is reported explicitly.
|
|
13
|
+
|
|
14
|
+
Field names follow promptfoo's documented output schema (OutputFile envelope
|
|
15
|
+
with results.version == 3; per-result success/score/error, vars,
|
|
16
|
+
response.output, response.tokenUsage.{prompt,completion,total}, latencyMs,
|
|
17
|
+
gradingResult.{pass,score,reason,componentResults}). Anything else lands in
|
|
18
|
+
`unsupported`, never silently dropped.
|
|
19
|
+
|
|
20
|
+
This is a read-only translation layer: it parses the config and the recorded
|
|
21
|
+
results. It never executes an eval.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import hashlib
|
|
26
|
+
import json
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
from ..model import (
|
|
31
|
+
Attempt,
|
|
32
|
+
Confidence,
|
|
33
|
+
Environment,
|
|
34
|
+
Grader,
|
|
35
|
+
IntegrityModel,
|
|
36
|
+
TaskSample,
|
|
37
|
+
)
|
|
38
|
+
from . import AuditError, register
|
|
39
|
+
|
|
40
|
+
ADAPTER_NAME = "promptfoo"
|
|
41
|
+
ADAPTER_VERSION = "0.6.0"
|
|
42
|
+
# Promptfoo's documented results envelope version (OutputFile.results.version).
|
|
43
|
+
RESULTS_VERSION = 3
|
|
44
|
+
|
|
45
|
+
CONFIG_NAMES = ("promptfooconfig.yaml", "promptfooconfig.yml")
|
|
46
|
+
RESULTS_NAME = "results.json"
|
|
47
|
+
|
|
48
|
+
# Top-level config keys the adapter understands. Everything else is reported
|
|
49
|
+
# in `unsupported` so coverage gaps are explicit, not silent.
|
|
50
|
+
KNOWN_CONFIG_KEYS = {
|
|
51
|
+
"description",
|
|
52
|
+
"env",
|
|
53
|
+
"prompts",
|
|
54
|
+
"providers",
|
|
55
|
+
"defaultTest",
|
|
56
|
+
"scenarios",
|
|
57
|
+
"tests",
|
|
58
|
+
"metadata",
|
|
59
|
+
"outputPath",
|
|
60
|
+
"writeLatestResults",
|
|
61
|
+
"sharing",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _envelope_has_config(results_path: Path) -> bool:
|
|
66
|
+
"""Cheap peek for detect(): does the results envelope carry a config dict?"""
|
|
67
|
+
try:
|
|
68
|
+
envelope = json.loads(results_path.read_text(encoding="utf-8"))
|
|
69
|
+
except (OSError, ValueError):
|
|
70
|
+
return False
|
|
71
|
+
return isinstance(envelope, dict) and isinstance(envelope.get("config"), dict)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _read_json(path: Path) -> Any:
|
|
75
|
+
try:
|
|
76
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
77
|
+
except FileNotFoundError as exc:
|
|
78
|
+
raise AuditError(f"missing required file: {path}") from exc
|
|
79
|
+
except json.JSONDecodeError as exc:
|
|
80
|
+
raise AuditError(f"invalid JSON in {path}: {exc}") from exc
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _read_yaml(path: Path) -> Any:
|
|
84
|
+
try:
|
|
85
|
+
import yaml
|
|
86
|
+
except ImportError as exc:
|
|
87
|
+
raise AuditError(
|
|
88
|
+
"promptfoo adapter needs PyYAML to parse promptfooconfig.yaml "
|
|
89
|
+
"(pip install 'evalwarden' again, or pip install pyyaml)"
|
|
90
|
+
) from exc
|
|
91
|
+
try:
|
|
92
|
+
return yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
93
|
+
except FileNotFoundError as exc:
|
|
94
|
+
raise AuditError(f"missing required file: {path}") from exc
|
|
95
|
+
except yaml.YAMLError as exc:
|
|
96
|
+
raise AuditError(f"invalid YAML in {path}: {exc}") from exc
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _digest(path: Path) -> str:
|
|
100
|
+
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _config_path(path: Path) -> Path | None:
|
|
104
|
+
for name in CONFIG_NAMES:
|
|
105
|
+
candidate = path / name
|
|
106
|
+
if candidate.is_file():
|
|
107
|
+
return candidate
|
|
108
|
+
return None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@register
|
|
112
|
+
class PromptfooAdapter:
|
|
113
|
+
name = ADAPTER_NAME
|
|
114
|
+
version = ADAPTER_VERSION
|
|
115
|
+
|
|
116
|
+
def detect(self, path: Path) -> Confidence:
|
|
117
|
+
if not path.is_dir():
|
|
118
|
+
return Confidence.LOW
|
|
119
|
+
has_config = _config_path(path) is not None
|
|
120
|
+
results_path = path / RESULTS_NAME
|
|
121
|
+
has_results = results_path.is_file()
|
|
122
|
+
if has_config and has_results:
|
|
123
|
+
return Confidence.HIGH
|
|
124
|
+
if has_config:
|
|
125
|
+
# Config-only audit: valid but incomplete (no runs to analyze).
|
|
126
|
+
return Confidence.MEDIUM
|
|
127
|
+
if has_results and _envelope_has_config(results_path):
|
|
128
|
+
# No promptfooconfig.yaml, but the results envelope carries the
|
|
129
|
+
# config it was produced from.
|
|
130
|
+
return Confidence.MEDIUM
|
|
131
|
+
return Confidence.LOW
|
|
132
|
+
|
|
133
|
+
def collect(self, path: Path) -> dict:
|
|
134
|
+
"""Read-only: files are opened for reading and never modified."""
|
|
135
|
+
bundle: dict[str, Any] = {"root": path}
|
|
136
|
+
digests: dict[str, str] = {}
|
|
137
|
+
config_path = _config_path(path)
|
|
138
|
+
if config_path is not None:
|
|
139
|
+
bundle["config"] = _read_yaml(config_path)
|
|
140
|
+
bundle["config_file"] = config_path.name
|
|
141
|
+
digests[config_path.name] = _digest(config_path)
|
|
142
|
+
results_path = path / RESULTS_NAME
|
|
143
|
+
if results_path.is_file():
|
|
144
|
+
bundle["results"] = _read_json(results_path)
|
|
145
|
+
digests[RESULTS_NAME] = _digest(results_path)
|
|
146
|
+
bundle["digests"] = digests
|
|
147
|
+
return bundle
|
|
148
|
+
|
|
149
|
+
def normalize(self, bundle: dict) -> IntegrityModel:
|
|
150
|
+
root: Path = bundle["root"]
|
|
151
|
+
config = bundle.get("config") or {}
|
|
152
|
+
config_file = bundle.get("config_file", "promptfooconfig.yaml")
|
|
153
|
+
envelope = bundle.get("results") or {}
|
|
154
|
+
inner = envelope.get("results") or {}
|
|
155
|
+
|
|
156
|
+
if not isinstance(config, dict):
|
|
157
|
+
raise AuditError(f"{config_file}: top-level mapping expected")
|
|
158
|
+
|
|
159
|
+
unsupported: list[str] = []
|
|
160
|
+
if "config" not in bundle and envelope.get("config") is not None:
|
|
161
|
+
# Graceful degradation: no promptfooconfig.yaml, but the results
|
|
162
|
+
# envelope carries the eval config it was produced from.
|
|
163
|
+
config = envelope["config"]
|
|
164
|
+
unsupported.append(
|
|
165
|
+
f"{RESULTS_NAME}: promptfooconfig.yaml absent; "
|
|
166
|
+
"using the config embedded in the results envelope"
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
version = inner.get("version")
|
|
170
|
+
if "results" in bundle and version != RESULTS_VERSION:
|
|
171
|
+
raise AuditError(
|
|
172
|
+
f"{RESULTS_NAME}: unsupported results version {version!r} "
|
|
173
|
+
f"(this adapter understands version {RESULTS_VERSION})"
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
eval_id = (
|
|
177
|
+
config.get("description")
|
|
178
|
+
or envelope.get("evalId")
|
|
179
|
+
or root.name
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
tasks = _normalize_tests(config, unsupported)
|
|
183
|
+
env_vars = {
|
|
184
|
+
str(name): "<redacted>" # values are never stored; names are the signal
|
|
185
|
+
for name in (config.get("env") or {})
|
|
186
|
+
}
|
|
187
|
+
grader = _normalize_grader(config)
|
|
188
|
+
attempts = _normalize_attempts(inner)
|
|
189
|
+
|
|
190
|
+
for key in config:
|
|
191
|
+
if key not in KNOWN_CONFIG_KEYS:
|
|
192
|
+
unsupported.append(f"{config_file}:{key}")
|
|
193
|
+
if config.get("scenarios"):
|
|
194
|
+
unsupported.append(
|
|
195
|
+
f"{config_file}:scenarios (scenario-generated tests not expanded)"
|
|
196
|
+
)
|
|
197
|
+
for test in _inline_tests(config):
|
|
198
|
+
for key in test:
|
|
199
|
+
if key not in {
|
|
200
|
+
"description", "vars", "assert", "options",
|
|
201
|
+
"threshold", "metadata",
|
|
202
|
+
}:
|
|
203
|
+
unsupported.append(f"{config_file}:tests[]:{key}")
|
|
204
|
+
|
|
205
|
+
return IntegrityModel(
|
|
206
|
+
eval_id=str(eval_id),
|
|
207
|
+
adapter_name=self.name,
|
|
208
|
+
adapter_version=self.version,
|
|
209
|
+
tasks=tasks,
|
|
210
|
+
environment=Environment(env_vars=env_vars, mounts=[]),
|
|
211
|
+
grader=grader,
|
|
212
|
+
attempts=attempts,
|
|
213
|
+
unsupported=unsupported,
|
|
214
|
+
digests=dict(bundle.get("digests", {})),
|
|
215
|
+
# Checks cite canonical artifact names; point them at the real files.
|
|
216
|
+
file_aliases={
|
|
217
|
+
"environment.json": config_file,
|
|
218
|
+
"grader.json": config_file,
|
|
219
|
+
"judge_run.json": config_file,
|
|
220
|
+
"run.json": RESULTS_NAME,
|
|
221
|
+
},
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _inline_tests(config: dict) -> list[dict]:
|
|
226
|
+
"""Tests declared inline in the config. file:// and other references are
|
|
227
|
+
left for `unsupported`: expanding them would pull arbitrary files into the
|
|
228
|
+
audit, and the declared config is the contract under review."""
|
|
229
|
+
tests = config.get("tests") or []
|
|
230
|
+
if isinstance(tests, dict):
|
|
231
|
+
tests = [tests]
|
|
232
|
+
return [t for t in tests if isinstance(t, dict)]
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _normalize_tests(config: dict, unsupported: list[str]) -> list[TaskSample]:
|
|
236
|
+
tasks: list[TaskSample] = []
|
|
237
|
+
raw_tests = config.get("tests")
|
|
238
|
+
if isinstance(raw_tests, str):
|
|
239
|
+
unsupported.append(f"tests: {raw_tests} (external test file not expanded)")
|
|
240
|
+
return tasks
|
|
241
|
+
if isinstance(raw_tests, list) and any(
|
|
242
|
+
isinstance(t, str) for t in raw_tests
|
|
243
|
+
):
|
|
244
|
+
unsupported.append("tests[]: file:// references not expanded")
|
|
245
|
+
for i, test in enumerate(_inline_tests(config)):
|
|
246
|
+
vars_ = test.get("vars") or {}
|
|
247
|
+
assertions = [str(a.get("type", "?")) for a in (test.get("assert") or []) if isinstance(a, dict)]
|
|
248
|
+
tasks.append(
|
|
249
|
+
TaskSample(
|
|
250
|
+
id=str(test.get("description") or f"test-{i}"),
|
|
251
|
+
prompt=str(test.get("description") or ""),
|
|
252
|
+
metadata={
|
|
253
|
+
"vars": sorted(str(k) for k in vars_),
|
|
254
|
+
"assertions": assertions,
|
|
255
|
+
},
|
|
256
|
+
)
|
|
257
|
+
)
|
|
258
|
+
return tasks
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _assertions(config: dict) -> list[dict]:
|
|
262
|
+
"""All assertions: shared defaultTest ones plus per-test ones."""
|
|
263
|
+
found: list[dict] = []
|
|
264
|
+
default = config.get("defaultTest") or {}
|
|
265
|
+
if isinstance(default, dict):
|
|
266
|
+
found.extend(a for a in (default.get("assert") or []) if isinstance(a, dict))
|
|
267
|
+
for test in _inline_tests(config):
|
|
268
|
+
found.extend(a for a in (test.get("assert") or []) if isinstance(a, dict))
|
|
269
|
+
return found
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _provider_id(ref: Any) -> str | None:
|
|
273
|
+
if isinstance(ref, str):
|
|
274
|
+
return ref
|
|
275
|
+
if isinstance(ref, dict) and ref.get("id"):
|
|
276
|
+
return str(ref["id"])
|
|
277
|
+
return None
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _grader_provider(config: dict) -> tuple[str | None, float | None]:
|
|
281
|
+
"""The model grading llm-rubric assertions: per-test options.provider wins,
|
|
282
|
+
then defaultTest.options.provider. Returns (provider_id, temperature)."""
|
|
283
|
+
for test in _inline_tests(config):
|
|
284
|
+
options = test.get("options") or {}
|
|
285
|
+
if isinstance(options, dict) and options.get("provider") is not None:
|
|
286
|
+
return _provider_id(options["provider"]), _provider_temperature(options["provider"])
|
|
287
|
+
default = config.get("defaultTest") or {}
|
|
288
|
+
options = default.get("options") or {} if isinstance(default, dict) else {}
|
|
289
|
+
if isinstance(options, dict) and options.get("provider") is not None:
|
|
290
|
+
return _provider_id(options["provider"]), _provider_temperature(options["provider"])
|
|
291
|
+
return None, None
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _provider_temperature(ref: Any) -> float | None:
|
|
295
|
+
if isinstance(ref, dict):
|
|
296
|
+
cfg = ref.get("config") or {}
|
|
297
|
+
if isinstance(cfg, dict) and isinstance(cfg.get("temperature"), (int, float)):
|
|
298
|
+
return float(cfg["temperature"])
|
|
299
|
+
return None
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _normalize_grader(config: dict) -> Grader:
|
|
303
|
+
assertions = _assertions(config)
|
|
304
|
+
types = sorted({str(a.get("type", "?")) for a in assertions})
|
|
305
|
+
is_judge = "llm-rubric" in types
|
|
306
|
+
grader = Grader(
|
|
307
|
+
kind="judge" if is_judge else "script",
|
|
308
|
+
tests=types,
|
|
309
|
+
)
|
|
310
|
+
if is_judge:
|
|
311
|
+
provider_id, temperature = _grader_provider(config)
|
|
312
|
+
grader.judge_model = provider_id
|
|
313
|
+
grader.temperature = temperature
|
|
314
|
+
# Each llm-rubric value is the rubric; promptfoo has no anchored scale
|
|
315
|
+
# levels, which JUDGE-001 reports as an unanchored rubric.
|
|
316
|
+
grader.rubric_criteria = [
|
|
317
|
+
str(a["value"]) for a in assertions
|
|
318
|
+
if str(a.get("type")) == "llm-rubric" and a.get("value")
|
|
319
|
+
]
|
|
320
|
+
return grader
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _normalize_attempts(inner: dict) -> list[Attempt]:
|
|
324
|
+
attempts: list[Attempt] = []
|
|
325
|
+
rows = inner.get("results") or []
|
|
326
|
+
for i, row in enumerate(rows):
|
|
327
|
+
if not isinstance(row, dict):
|
|
328
|
+
continue
|
|
329
|
+
error = row.get("error")
|
|
330
|
+
success = row.get("success")
|
|
331
|
+
status = "error" if error else ("pass" if success else "fail")
|
|
332
|
+
response = row.get("response") or {}
|
|
333
|
+
usage = response.get("tokenUsage") or {}
|
|
334
|
+
latency_ms = row.get("latencyMs")
|
|
335
|
+
output = response.get("output")
|
|
336
|
+
attempts.append(
|
|
337
|
+
Attempt(
|
|
338
|
+
task_id=str(row.get("description") or f"test-{row.get('testIdx', i)}"),
|
|
339
|
+
status=status,
|
|
340
|
+
score=_to_float(row.get("score")),
|
|
341
|
+
tokens_in=_to_int(usage.get("prompt")),
|
|
342
|
+
tokens_out=_to_int(usage.get("completion")),
|
|
343
|
+
latency_s=(float(latency_ms) / 1000.0) if isinstance(latency_ms, (int, float)) else None,
|
|
344
|
+
# An explicitly empty model output scored as a pass is the
|
|
345
|
+
# GRAD-002 signal; anything else is just a short answer.
|
|
346
|
+
empty_submission=isinstance(output, str) and output == "" and status == "pass",
|
|
347
|
+
)
|
|
348
|
+
)
|
|
349
|
+
return attempts
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _to_float(value: Any) -> float | None:
|
|
353
|
+
return float(value) if isinstance(value, (int, float)) else None
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _to_int(value: Any) -> int | None:
|
|
357
|
+
return int(value) if isinstance(value, (int, float)) else None
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Rule registry. v0.3 ships deterministic, high-precision checks only.
|
|
2
|
+
|
|
3
|
+
Sequencing rule: earn trust with deterministic evidence before adding
|
|
4
|
+
probabilistic signals. Every finding carries a confidence label; a linter that
|
|
5
|
+
cries contamination on a clean eval is worse than no auditor.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .base import Check
|
|
10
|
+
from .cost import CHECKS as COST_CHECKS
|
|
11
|
+
from .env_leakage import EnvLeakageCheck
|
|
12
|
+
from .grader import EmptyPathCheck, VerifierWritableCheck
|
|
13
|
+
from .judge import CHECKS as JUDGE_CHECKS
|
|
14
|
+
|
|
15
|
+
REGISTRY: list[Check] = [
|
|
16
|
+
EnvLeakageCheck(), # ENV-001
|
|
17
|
+
VerifierWritableCheck(), # GRAD-001
|
|
18
|
+
EmptyPathCheck(), # GRAD-002
|
|
19
|
+
*COST_CHECKS, # COST-001 .. COST-004
|
|
20
|
+
*JUDGE_CHECKS, # JUDGE-001 .. JUDGE-006
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
BY_ID: dict[str, Check] = {check.meta.id: check for check in REGISTRY}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def get_check(check_id: str) -> Check | None:
|
|
27
|
+
return BY_ID.get(check_id.upper())
|