evalseal 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalseal/__init__.py +3 -0
- evalseal/adapters/__init__.py +0 -0
- evalseal/adapters/dataset.py +36 -0
- evalseal/adapters/recording.py +56 -0
- evalseal/adapters/scorer.py +74 -0
- evalseal/adapters/target.py +97 -0
- evalseal/analyze.py +82 -0
- evalseal/cli.py +152 -0
- evalseal/executor.py +142 -0
- evalseal/ledger.py +60 -0
- evalseal/models.py +78 -0
- evalseal/report.py +48 -0
- evalseal-0.1.0.dist-info/METADATA +116 -0
- evalseal-0.1.0.dist-info/RECORD +17 -0
- evalseal-0.1.0.dist-info/WHEEL +4 -0
- evalseal-0.1.0.dist-info/entry_points.txt +2 -0
- evalseal-0.1.0.dist-info/licenses/LICENSE +21 -0
evalseal/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class Case:
|
|
11
|
+
case_id: str
|
|
12
|
+
prompt: str
|
|
13
|
+
expected: str | None = None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class Dataset:
|
|
18
|
+
cases: list[Case]
|
|
19
|
+
hash: str
|
|
20
|
+
|
|
21
|
+
@classmethod
|
|
22
|
+
def from_jsonl(cls, path: str | Path) -> "Dataset":
|
|
23
|
+
raw = Path(path).read_text()
|
|
24
|
+
cases = []
|
|
25
|
+
seen: set[str] = set()
|
|
26
|
+
for lineno, line in enumerate(raw.splitlines(), start=1):
|
|
27
|
+
line = line.strip()
|
|
28
|
+
if not line:
|
|
29
|
+
continue
|
|
30
|
+
d = json.loads(line)
|
|
31
|
+
if d["case_id"] in seen:
|
|
32
|
+
raise ValueError(f"{path}:{lineno}: duplicate case_id {d['case_id']!r}")
|
|
33
|
+
seen.add(d["case_id"])
|
|
34
|
+
cases.append(Case(d["case_id"], d["prompt"], d.get("expected")))
|
|
35
|
+
h = "sha256:" + hashlib.sha256(raw.encode()).hexdigest()
|
|
36
|
+
return cls(cases, h)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Callable
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def request_key(payload: dict) -> str:
|
|
12
|
+
"""Stable hash of the *effective* request, so identical requests hit the cassette."""
|
|
13
|
+
blob = json.dumps(payload, sort_keys=True, separators=(",", ":"))
|
|
14
|
+
return hashlib.sha256(blob.encode()).hexdigest()
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Cassette:
|
|
18
|
+
"""JSON-backed record/replay store keyed by request hash.
|
|
19
|
+
|
|
20
|
+
Each key maps to an ordered *list* of responses. Repeating an identical request is
|
|
21
|
+
the whole point of an N-run eval, so the k-th call with a given key is served the
|
|
22
|
+
k-th recorded response. (A single response per key would replay repeat #1 N times
|
|
23
|
+
and hide every flip.)
|
|
24
|
+
|
|
25
|
+
Mode is controlled by env var EVALSEAL_RECORD (or the `record` argument):
|
|
26
|
+
unset/0 -> replay only (raises if a request is missing) — this is CI mode
|
|
27
|
+
1 -> record: already-recorded responses are reused, new calls are captured
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, path: str | Path, record: bool | None = None):
|
|
31
|
+
self.path = Path(path)
|
|
32
|
+
if record is None:
|
|
33
|
+
record = os.environ.get("EVALSEAL_RECORD", "0") == "1"
|
|
34
|
+
self.record = record
|
|
35
|
+
self._data: dict[str, list[dict]] = {}
|
|
36
|
+
self._cursor: Counter[str] = Counter()
|
|
37
|
+
if self.path.exists():
|
|
38
|
+
self._data = json.loads(self.path.read_text())
|
|
39
|
+
|
|
40
|
+
def call(self, payload: dict, do_real_call: Callable[[], dict]) -> dict:
|
|
41
|
+
key = request_key(payload)
|
|
42
|
+
index = self._cursor[key]
|
|
43
|
+
self._cursor[key] += 1
|
|
44
|
+
recorded = self._data.get(key, [])
|
|
45
|
+
if index < len(recorded):
|
|
46
|
+
return recorded[index]
|
|
47
|
+
if not self.record:
|
|
48
|
+
raise RuntimeError(
|
|
49
|
+
f"No cassette entry #{index} for request {key[:12]} in {self.path} and "
|
|
50
|
+
f"recording is off. Run with EVALSEAL_RECORD=1 and an API key to capture it."
|
|
51
|
+
)
|
|
52
|
+
resp = do_real_call() # real network call, record mode only
|
|
53
|
+
self._data.setdefault(key, []).append(resp)
|
|
54
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
self.path.write_text(json.dumps(self._data, indent=2, sort_keys=True))
|
|
56
|
+
return resp
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
7
|
+
|
|
8
|
+
from .target import Target, TargetResponse
|
|
9
|
+
|
|
10
|
+
_VERDICT_RE = re.compile(r"\b(PASS|FAIL)\b")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class ScoreResult:
|
|
15
|
+
score: float # 0/1 for binary scorers
|
|
16
|
+
binary: bool
|
|
17
|
+
verdict: Optional[int] # 0/1 or None
|
|
18
|
+
judge_response: Optional[TargetResponse] = None
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@runtime_checkable
|
|
22
|
+
class Scorer(Protocol):
|
|
23
|
+
kind: str
|
|
24
|
+
def score(self, prompt: str, response_text: str, expected: Optional[str]) -> ScoreResult: ...
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class ExactMatchScorer:
|
|
29
|
+
kind: str = "exact"
|
|
30
|
+
|
|
31
|
+
def score(self, prompt, response_text, expected):
|
|
32
|
+
ok = expected is not None and response_text.strip() == expected.strip()
|
|
33
|
+
return ScoreResult(float(ok), True, int(ok))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class RegexScorer:
|
|
38
|
+
pattern: str
|
|
39
|
+
kind: str = "regex"
|
|
40
|
+
|
|
41
|
+
def score(self, prompt, response_text, expected):
|
|
42
|
+
ok = re.search(self.pattern, response_text) is not None
|
|
43
|
+
return ScoreResult(float(ok), True, int(ok))
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def parse_verdict(text: str) -> int:
|
|
47
|
+
"""First standalone PASS/FAIL token wins; anything unparseable counts as FAIL.
|
|
48
|
+
(A plain substring check would read "FAIL — this does not PASS" as a pass.)"""
|
|
49
|
+
m = _VERDICT_RE.search(text.upper())
|
|
50
|
+
return 1 if m and m.group(1) == "PASS" else 0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class LLMJudgeScorer:
|
|
55
|
+
"""Grades a response pass/fail using another model. The judge is a Target, so it
|
|
56
|
+
gets its own recording + provenance. This is where flips are most visible."""
|
|
57
|
+
judge: Target
|
|
58
|
+
rubric: str
|
|
59
|
+
kind: str = "llm_judge"
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def rubric_hash(self) -> str:
|
|
63
|
+
return "sha256:" + hashlib.sha256(self.rubric.encode()).hexdigest()
|
|
64
|
+
|
|
65
|
+
def score(self, prompt, response_text, expected):
|
|
66
|
+
judge_prompt = (
|
|
67
|
+
f"{self.rubric}\n\n"
|
|
68
|
+
f"USER PROMPT:\n{prompt}\n\n"
|
|
69
|
+
f"RESPONSE TO GRADE:\n{response_text}\n\n"
|
|
70
|
+
f"Answer with exactly one word: PASS or FAIL."
|
|
71
|
+
)
|
|
72
|
+
jr = self.judge.generate(judge_prompt)
|
|
73
|
+
verdict = parse_verdict(jr.text)
|
|
74
|
+
return ScoreResult(float(verdict), True, verdict, judge_response=jr)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Any, Callable, Optional, Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
|
|
9
|
+
from .recording import Cassette
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class TargetResponse:
|
|
14
|
+
text: str
|
|
15
|
+
requested_model: str
|
|
16
|
+
served_model: Optional[str]
|
|
17
|
+
system_fingerprint: Optional[str]
|
|
18
|
+
base_url: str
|
|
19
|
+
effective_params: dict[str, Any]
|
|
20
|
+
params_source: str # "explicit" | "provider_default"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@runtime_checkable
|
|
24
|
+
class Target(Protocol):
|
|
25
|
+
def generate(self, prompt: str) -> TargetResponse: ...
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class LocalCallableTarget:
|
|
30
|
+
"""Wrap any Python function as a target. Great for offline tests: the function
|
|
31
|
+
can return scripted-variance outputs so the executor is testable with no network."""
|
|
32
|
+
fn: Callable[[str], str]
|
|
33
|
+
name: str = "local"
|
|
34
|
+
|
|
35
|
+
def generate(self, prompt: str) -> TargetResponse:
|
|
36
|
+
return TargetResponse(
|
|
37
|
+
text=self.fn(prompt),
|
|
38
|
+
requested_model=self.name,
|
|
39
|
+
served_model=self.name,
|
|
40
|
+
system_fingerprint=None,
|
|
41
|
+
base_url="local://",
|
|
42
|
+
effective_params={},
|
|
43
|
+
params_source="explicit",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class OpenAICompatibleTarget:
|
|
49
|
+
"""Calls any OpenAI-compatible /chat/completions endpoint, through the cassette."""
|
|
50
|
+
model: str
|
|
51
|
+
cassette: Cassette
|
|
52
|
+
base_url: str = "https://api.openai.com/v1"
|
|
53
|
+
temperature: Optional[float] = None # None means "we did not set it"
|
|
54
|
+
seed: Optional[int] = None
|
|
55
|
+
api_key_env: str = "EVALSEAL_API_KEY"
|
|
56
|
+
|
|
57
|
+
def generate(self, prompt: str) -> TargetResponse:
|
|
58
|
+
# Record whether we explicitly set temperature or are inheriting the default.
|
|
59
|
+
params_source = "explicit" if self.temperature is not None else "provider_default"
|
|
60
|
+
body: dict[str, Any] = {
|
|
61
|
+
"model": self.model,
|
|
62
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
63
|
+
}
|
|
64
|
+
if self.temperature is not None:
|
|
65
|
+
body["temperature"] = self.temperature
|
|
66
|
+
if self.seed is not None:
|
|
67
|
+
body["seed"] = self.seed
|
|
68
|
+
|
|
69
|
+
def do_real_call() -> dict:
|
|
70
|
+
key = os.environ.get(self.api_key_env)
|
|
71
|
+
if not key:
|
|
72
|
+
raise RuntimeError(f"Recording needs an API key in ${self.api_key_env}.")
|
|
73
|
+
with httpx.Client(timeout=60) as c:
|
|
74
|
+
r = c.post(
|
|
75
|
+
f"{self.base_url.rstrip('/')}/chat/completions",
|
|
76
|
+
headers={"Authorization": f"Bearer {key}"},
|
|
77
|
+
json=body,
|
|
78
|
+
)
|
|
79
|
+
r.raise_for_status()
|
|
80
|
+
return r.json()
|
|
81
|
+
|
|
82
|
+
# Cassette key is the full effective body (never the API key).
|
|
83
|
+
raw = self.cassette.call({"url": self.base_url, "body": body}, do_real_call)
|
|
84
|
+
|
|
85
|
+
return TargetResponse(
|
|
86
|
+
text=raw["choices"][0]["message"]["content"] or "",
|
|
87
|
+
requested_model=self.model,
|
|
88
|
+
served_model=raw.get("model"),
|
|
89
|
+
system_fingerprint=raw.get("system_fingerprint"),
|
|
90
|
+
base_url=self.base_url,
|
|
91
|
+
effective_params={
|
|
92
|
+
"temperature": self.temperature,
|
|
93
|
+
"seed": self.seed,
|
|
94
|
+
"top_p": None,
|
|
95
|
+
},
|
|
96
|
+
params_source=params_source,
|
|
97
|
+
)
|
evalseal/analyze.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Variance analysis for repeated eval runs. Pure functions, no I/O, no network.
|
|
2
|
+
|
|
3
|
+
A "verdict" is a binary pass(1)/fail(0). A "score" may be binary or a float in [0,1].
|
|
4
|
+
We quantify reproducibility; we never claim determinism.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import random
|
|
9
|
+
import statistics
|
|
10
|
+
from collections import Counter
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
# Stability thresholds (constants so they're auditable, not magic numbers).
|
|
14
|
+
BORDERLINE_MAX_FLIP = 0.20 # flip_rate in (0, 0.20] -> BORDERLINE
|
|
15
|
+
# flip_rate == 0 -> STABLE ; flip_rate > BORDERLINE_MAX_FLIP -> UNSTABLE
|
|
16
|
+
|
|
17
|
+
_BOOTSTRAP_ITERS = 2000
|
|
18
|
+
_RNG_SEED = 12345 # fixed so the CI computation itself is reproducible
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class CaseStats:
|
|
23
|
+
mean: float
|
|
24
|
+
ci95_low: float
|
|
25
|
+
ci95_high: float
|
|
26
|
+
flip_rate: float
|
|
27
|
+
stability: str # "STABLE" | "BORDERLINE" | "UNSTABLE"
|
|
28
|
+
majority_verdict: int | None # for binary; None for pure-float scores
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _bootstrap_ci(scores: list[float], iters: int = _BOOTSTRAP_ITERS) -> tuple[float, float]:
|
|
32
|
+
"""Percentile bootstrap 95% CI of the mean. Deterministic via fixed seed."""
|
|
33
|
+
n = len(scores)
|
|
34
|
+
if n == 0:
|
|
35
|
+
return (0.0, 0.0)
|
|
36
|
+
if n == 1:
|
|
37
|
+
return (scores[0], scores[0])
|
|
38
|
+
rng = random.Random(_RNG_SEED)
|
|
39
|
+
means = []
|
|
40
|
+
for _ in range(iters):
|
|
41
|
+
sample = [scores[rng.randrange(n)] for _ in range(n)]
|
|
42
|
+
means.append(sum(sample) / n)
|
|
43
|
+
means.sort()
|
|
44
|
+
lo = means[int(0.025 * iters)]
|
|
45
|
+
hi = means[int(0.975 * iters)]
|
|
46
|
+
return (lo, hi)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def flip_rate(verdicts: list[int]) -> tuple[float, int]:
|
|
50
|
+
"""Fraction of verdicts disagreeing with the majority. Returns (rate, majority)."""
|
|
51
|
+
if not verdicts:
|
|
52
|
+
return (0.0, 0)
|
|
53
|
+
counts = Counter(verdicts)
|
|
54
|
+
majority, majority_count = counts.most_common(1)[0]
|
|
55
|
+
disagree = len(verdicts) - majority_count
|
|
56
|
+
return (disagree / len(verdicts), majority)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def classify_stability(rate: float) -> str:
|
|
60
|
+
if rate == 0.0:
|
|
61
|
+
return "STABLE"
|
|
62
|
+
if rate <= BORDERLINE_MAX_FLIP:
|
|
63
|
+
return "BORDERLINE"
|
|
64
|
+
return "UNSTABLE"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def analyze_case(scores: list[float], *, binary: bool) -> CaseStats:
|
|
68
|
+
"""Turn N per-run scores into a CaseStats. If binary, scores must be 0/1."""
|
|
69
|
+
if not scores:
|
|
70
|
+
raise ValueError("analyze_case requires at least one score")
|
|
71
|
+
mean = sum(scores) / len(scores)
|
|
72
|
+
lo, hi = _bootstrap_ci(scores)
|
|
73
|
+
if binary:
|
|
74
|
+
verdicts = [int(s) for s in scores]
|
|
75
|
+
rate, majority = flip_rate(verdicts)
|
|
76
|
+
return CaseStats(mean, lo, hi, rate, classify_stability(rate), majority)
|
|
77
|
+
# Float scores: define a "flip" as crossing the run-set's own median.
|
|
78
|
+
# This gives a variance signal without a fixed threshold assumption.
|
|
79
|
+
med = statistics.median(scores)
|
|
80
|
+
pseudo = [1 if s >= med else 0 for s in scores]
|
|
81
|
+
rate, _ = flip_rate(pseudo)
|
|
82
|
+
return CaseStats(mean, lo, hi, rate, classify_stability(rate), None)
|
evalseal/cli.py
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
import typer
|
|
9
|
+
from rich.console import Console
|
|
10
|
+
from rich.markdown import Markdown
|
|
11
|
+
|
|
12
|
+
from .adapters.dataset import Dataset
|
|
13
|
+
from .adapters.recording import Cassette
|
|
14
|
+
from .adapters.scorer import ExactMatchScorer, LLMJudgeScorer, RegexScorer
|
|
15
|
+
from .adapters.target import OpenAICompatibleTarget
|
|
16
|
+
from .executor import run_eval
|
|
17
|
+
from .ledger import LEDGER_PATH, last_hash, load_all, seal_and_append, verify_chain
|
|
18
|
+
from .report import to_markdown, write_json, write_markdown
|
|
19
|
+
|
|
20
|
+
# Distinct from 1 (uncaught error) and 2 (usage error) so CI can tell
|
|
21
|
+
# "the eval is unstable" apart from "the tool broke".
|
|
22
|
+
EXIT_UNSTABLE = 3
|
|
23
|
+
|
|
24
|
+
app = typer.Typer(add_completion=False, help="Reproducibility receipts for LLM evals.")
|
|
25
|
+
console = Console()
|
|
26
|
+
|
|
27
|
+
LedgerOpt = typer.Option(LEDGER_PATH, "--ledger", help="Path to the ledger JSONL file.")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def load_dotenv(path: Path = Path(".env")) -> None:
|
|
31
|
+
"""Load KEY=VALUE lines from ./.env. Variables already in the environment win."""
|
|
32
|
+
if not path.is_file():
|
|
33
|
+
return
|
|
34
|
+
for line in path.read_text().splitlines():
|
|
35
|
+
line = line.strip()
|
|
36
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
37
|
+
continue
|
|
38
|
+
key, _, value = line.partition("=")
|
|
39
|
+
key = key.strip().removeprefix("export ").strip()
|
|
40
|
+
value = value.strip().strip("\"'")
|
|
41
|
+
if key and value:
|
|
42
|
+
os.environ.setdefault(key, value)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@app.callback()
|
|
46
|
+
def _main() -> None:
|
|
47
|
+
load_dotenv()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _build_target(cfg: dict, cassette: Cassette) -> OpenAICompatibleTarget:
|
|
51
|
+
return OpenAICompatibleTarget(
|
|
52
|
+
model=cfg["model"],
|
|
53
|
+
cassette=cassette,
|
|
54
|
+
base_url=cfg.get("base_url", "https://api.openai.com/v1"),
|
|
55
|
+
temperature=cfg.get("temperature"), # omit in config to surface the "unset" warning
|
|
56
|
+
seed=cfg.get("seed"),
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _build_scorer(cfg: dict, cassette: Cassette):
|
|
61
|
+
if cfg["type"] == "exact":
|
|
62
|
+
return ExactMatchScorer()
|
|
63
|
+
if cfg["type"] == "regex":
|
|
64
|
+
return RegexScorer(pattern=cfg["pattern"])
|
|
65
|
+
if cfg["type"] == "llm_judge":
|
|
66
|
+
judge = OpenAICompatibleTarget(
|
|
67
|
+
model=cfg["judge_model"],
|
|
68
|
+
cassette=cassette,
|
|
69
|
+
base_url=cfg.get("base_url", "https://api.openai.com/v1"),
|
|
70
|
+
temperature=cfg.get("judge_temperature"), # leave unset to demonstrate flips
|
|
71
|
+
seed=cfg.get("judge_seed"),
|
|
72
|
+
)
|
|
73
|
+
return LLMJudgeScorer(judge=judge, rubric=cfg["rubric"])
|
|
74
|
+
raise typer.BadParameter(f"unknown scorer type {cfg['type']!r}")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@app.command()
|
|
78
|
+
def run(
|
|
79
|
+
dataset: Path = typer.Option(..., exists=True, dir_okay=False),
|
|
80
|
+
target_config: Path = typer.Option(..., exists=True, dir_okay=False),
|
|
81
|
+
scorer_config: Path = typer.Option(..., exists=True, dir_okay=False),
|
|
82
|
+
n: int = typer.Option(5, min=1, help="Repeats per case (flip rate needs N>=5)."),
|
|
83
|
+
cassette: Path = typer.Option(Path("tests/cassettes/run.json")),
|
|
84
|
+
ledger: Path = LedgerOpt,
|
|
85
|
+
):
|
|
86
|
+
"""Run an eval N times, seal the result, emit report.json + report.md."""
|
|
87
|
+
ds = Dataset.from_jsonl(dataset)
|
|
88
|
+
cass = Cassette(cassette)
|
|
89
|
+
target = _build_target(json.loads(target_config.read_text()), cass)
|
|
90
|
+
scorer = _build_scorer(json.loads(scorer_config.read_text()), cass)
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
record = run_eval(ds, target, scorer, n_repeats=n, prev_hash=last_hash(ledger))
|
|
94
|
+
except RuntimeError as e:
|
|
95
|
+
console.print(f"[red]{e}[/red]")
|
|
96
|
+
raise typer.Exit(code=1)
|
|
97
|
+
except httpx.HTTPStatusError as e:
|
|
98
|
+
resp = e.response
|
|
99
|
+
console.print(f"[red]HTTP {resp.status_code} from provider: {resp.text[:300]}[/red]")
|
|
100
|
+
if resp.status_code == 429:
|
|
101
|
+
console.print(
|
|
102
|
+
"[yellow]Rate limited. Responses recorded so far are saved in the cassette; "
|
|
103
|
+
"re-run the same command later to resume.[/yellow]"
|
|
104
|
+
)
|
|
105
|
+
raise typer.Exit(code=1)
|
|
106
|
+
record = seal_and_append(record, ledger)
|
|
107
|
+
write_json(record)
|
|
108
|
+
write_markdown(record)
|
|
109
|
+
console.print(Markdown(to_markdown(record)))
|
|
110
|
+
|
|
111
|
+
# CI gate: dedicated non-zero exit if any case is UNSTABLE.
|
|
112
|
+
if record.aggregate.n_unstable > 0:
|
|
113
|
+
console.print(f"[red]{record.aggregate.n_unstable} unstable case(s) — failing.[/red]")
|
|
114
|
+
raise typer.Exit(code=EXIT_UNSTABLE)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@app.command()
|
|
118
|
+
def verify(ledger: Path = LedgerOpt):
|
|
119
|
+
"""Check the ledger chain integrity (tamper detection)."""
|
|
120
|
+
ok, msg = verify_chain(ledger)
|
|
121
|
+
color = "green" if ok else "red"
|
|
122
|
+
console.print(f"[{color}]{msg}[/{color}]")
|
|
123
|
+
raise typer.Exit(code=0 if ok else 1)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@app.command()
|
|
127
|
+
def diff(
|
|
128
|
+
a: int = typer.Argument(..., help="Ledger index of the baseline run (negative ok)."),
|
|
129
|
+
b: int = typer.Argument(..., help="Ledger index of the candidate run (negative ok)."),
|
|
130
|
+
ledger: Path = LedgerOpt,
|
|
131
|
+
):
|
|
132
|
+
"""Compare two runs; state whether a score change exceeds the noise floor."""
|
|
133
|
+
recs = load_all(ledger)
|
|
134
|
+
for i in (a, b):
|
|
135
|
+
if not -len(recs) <= i < len(recs):
|
|
136
|
+
raise typer.BadParameter(f"index {i} out of range; ledger has {len(recs)} record(s)")
|
|
137
|
+
ra, rb = recs[a], recs[b]
|
|
138
|
+
ma, mb = ra.aggregate.mean_score, rb.aggregate.mean_score
|
|
139
|
+
|
|
140
|
+
# Noise floor: widest per-case CI half-width across both runs. Deliberately
|
|
141
|
+
# conservative — "REAL CHANGE" is only claimed when no single case's noise explains it.
|
|
142
|
+
def halfwidth(r):
|
|
143
|
+
return max(((c.ci95[1] - c.ci95[0]) / 2 for c in r.results), default=0.0)
|
|
144
|
+
|
|
145
|
+
noise = max(halfwidth(ra), halfwidth(rb))
|
|
146
|
+
delta = mb - ma
|
|
147
|
+
verdict = "within noise" if abs(delta) <= noise else "REAL CHANGE"
|
|
148
|
+
console.print(
|
|
149
|
+
f"mean {ma:.3f} -> {mb:.3f} (delta {delta:+.3f}, noise floor ±{noise:.3f}) => {verdict}"
|
|
150
|
+
)
|
|
151
|
+
if ra.manifest.dataset.hash != rb.manifest.dataset.hash:
|
|
152
|
+
console.print("[yellow]Warning: runs used different datasets.[/yellow]")
|
evalseal/executor.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from urllib.parse import urlparse
|
|
5
|
+
|
|
6
|
+
from .adapters.dataset import Dataset
|
|
7
|
+
from .adapters.scorer import Scorer
|
|
8
|
+
from .adapters.target import Target, TargetResponse
|
|
9
|
+
from .analyze import analyze_case
|
|
10
|
+
from .models import (
|
|
11
|
+
Aggregate, CaseResult, DatasetProvenance, EffectiveParams,
|
|
12
|
+
ProvenanceManifest, RunConfig, RunRecord, ScorerProvenance, TargetProvenance,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_CANONICAL_HOSTS = {"api.openai.com", "generativelanguage.googleapis.com"}
|
|
16
|
+
_LOCAL_SCHEMES = {"local"}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _to_provenance(tr: TargetResponse) -> TargetProvenance:
|
|
20
|
+
return TargetProvenance(
|
|
21
|
+
requested_model=tr.requested_model,
|
|
22
|
+
served_model=tr.served_model,
|
|
23
|
+
system_fingerprint=tr.system_fingerprint,
|
|
24
|
+
base_url=tr.base_url,
|
|
25
|
+
effective_params=EffectiveParams(**tr.effective_params),
|
|
26
|
+
params_source=tr.params_source,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _is_same_model(requested: str, served: str) -> bool:
|
|
31
|
+
# Providers resolve aliases to dated snapshots (gpt-4o-mini -> gpt-4o-mini-2024-07-18,
|
|
32
|
+
# gpt-4 -> gpt-4-0613). That is the same model; the snapshot is still recorded in
|
|
33
|
+
# served_model. A bare prefix check is not enough: gpt-4o-mini is not gpt-4o.
|
|
34
|
+
snapshot = re.compile(rf"{re.escape(requested)}-(\d{{4}}-\d{{2}}-\d{{2}}|\d{{4}})")
|
|
35
|
+
return served == requested or snapshot.fullmatch(served) is not None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _provenance_warnings(tp: TargetProvenance, role: str = "TARGET") -> list[str]:
|
|
39
|
+
w = []
|
|
40
|
+
if tp.served_model and tp.requested_model and not _is_same_model(
|
|
41
|
+
tp.requested_model, tp.served_model
|
|
42
|
+
):
|
|
43
|
+
w.append(
|
|
44
|
+
f"{role} SERVED MODEL MISMATCH: requested '{tp.requested_model}' but served "
|
|
45
|
+
f"'{tp.served_model}'. The report's model name may not be what ran."
|
|
46
|
+
)
|
|
47
|
+
parsed = urlparse(tp.base_url)
|
|
48
|
+
if parsed.scheme not in _LOCAL_SCHEMES and parsed.hostname not in _CANONICAL_HOSTS:
|
|
49
|
+
w.append(
|
|
50
|
+
f"{role} NON-CANONICAL ENDPOINT: {tp.base_url} — responses may be proxied/altered."
|
|
51
|
+
)
|
|
52
|
+
if tp.params_source == "provider_default" and tp.effective_params.temperature is None:
|
|
53
|
+
w.append(
|
|
54
|
+
f"{role} TEMPERATURE NOT SET: the provider default (often 1.0) was used silently; "
|
|
55
|
+
"verdicts near the decision boundary may not be reproducible."
|
|
56
|
+
)
|
|
57
|
+
return w
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _drift_warnings(responses: list[TargetResponse], role: str) -> list[str]:
|
|
61
|
+
"""The backend can change under you mid-run. Surface it instead of averaging over it."""
|
|
62
|
+
w = []
|
|
63
|
+
served = sorted({r.served_model for r in responses if r.served_model})
|
|
64
|
+
if len(served) > 1:
|
|
65
|
+
w.append(f"{role} SERVED MODEL CHANGED DURING RUN: {', '.join(served)}")
|
|
66
|
+
fps = sorted({r.system_fingerprint for r in responses if r.system_fingerprint})
|
|
67
|
+
if len(fps) > 1:
|
|
68
|
+
w.append(f"{role} SYSTEM FINGERPRINT CHANGED DURING RUN: {', '.join(fps)}")
|
|
69
|
+
return w
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _dedupe(items: list[str]) -> list[str]:
|
|
73
|
+
return list(dict.fromkeys(items))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def run_eval(
|
|
77
|
+
dataset: Dataset,
|
|
78
|
+
target: Target,
|
|
79
|
+
scorer: Scorer,
|
|
80
|
+
n_repeats: int = 5,
|
|
81
|
+
prev_hash: str = "GENESIS",
|
|
82
|
+
) -> RunRecord:
|
|
83
|
+
if not dataset.cases:
|
|
84
|
+
raise ValueError("dataset has no cases")
|
|
85
|
+
if n_repeats < 1:
|
|
86
|
+
raise ValueError("n_repeats must be >= 1")
|
|
87
|
+
|
|
88
|
+
results: list[CaseResult] = []
|
|
89
|
+
target_resps: list[TargetResponse] = []
|
|
90
|
+
judge_resps: list[TargetResponse] = []
|
|
91
|
+
|
|
92
|
+
for case in dataset.cases:
|
|
93
|
+
scores: list[float] = []
|
|
94
|
+
binary = True
|
|
95
|
+
for _ in range(n_repeats):
|
|
96
|
+
tr = target.generate(case.prompt)
|
|
97
|
+
target_resps.append(tr)
|
|
98
|
+
sr = scorer.score(case.prompt, tr.text, case.expected)
|
|
99
|
+
if sr.judge_response is not None:
|
|
100
|
+
judge_resps.append(sr.judge_response)
|
|
101
|
+
scores.append(sr.score)
|
|
102
|
+
binary = binary and sr.binary
|
|
103
|
+
stats = analyze_case(scores, binary=binary)
|
|
104
|
+
results.append(CaseResult(
|
|
105
|
+
case_id=case.case_id,
|
|
106
|
+
scores=scores,
|
|
107
|
+
mean=stats.mean,
|
|
108
|
+
ci95=(stats.ci95_low, stats.ci95_high),
|
|
109
|
+
flip_rate=stats.flip_rate,
|
|
110
|
+
stability=stats.stability,
|
|
111
|
+
majority_verdict=stats.majority_verdict,
|
|
112
|
+
))
|
|
113
|
+
|
|
114
|
+
# Every response is checked, not just the last one: a mismatch on any call matters.
|
|
115
|
+
warnings: list[str] = []
|
|
116
|
+
for tr in target_resps:
|
|
117
|
+
warnings += _provenance_warnings(_to_provenance(tr), "TARGET")
|
|
118
|
+
warnings += _drift_warnings(target_resps, "TARGET")
|
|
119
|
+
for jr in judge_resps:
|
|
120
|
+
warnings += _provenance_warnings(_to_provenance(jr), "JUDGE")
|
|
121
|
+
warnings += _drift_warnings(judge_resps, "JUDGE")
|
|
122
|
+
|
|
123
|
+
sp = ScorerProvenance(
|
|
124
|
+
type=scorer.kind,
|
|
125
|
+
judge=_to_provenance(judge_resps[0]) if judge_resps else None,
|
|
126
|
+
rubric_hash=getattr(scorer, "rubric_hash", None),
|
|
127
|
+
)
|
|
128
|
+
manifest = ProvenanceManifest(
|
|
129
|
+
target=_to_provenance(target_resps[0]),
|
|
130
|
+
scorer=sp,
|
|
131
|
+
dataset=DatasetProvenance(hash=dataset.hash, n_cases=len(dataset.cases)),
|
|
132
|
+
run_config=RunConfig(n_repeats=n_repeats),
|
|
133
|
+
)
|
|
134
|
+
agg = Aggregate(
|
|
135
|
+
n_cases=len(results),
|
|
136
|
+
mean_score=sum(r.mean for r in results) / len(results),
|
|
137
|
+
n_stable=sum(r.stability == "STABLE" for r in results),
|
|
138
|
+
n_borderline=sum(r.stability == "BORDERLINE" for r in results),
|
|
139
|
+
n_unstable=sum(r.stability == "UNSTABLE" for r in results),
|
|
140
|
+
warnings=_dedupe(warnings),
|
|
141
|
+
)
|
|
142
|
+
return RunRecord(manifest=manifest, results=results, aggregate=agg, prev_hash=prev_hash)
|
evalseal/ledger.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from .models import RunRecord
|
|
8
|
+
|
|
9
|
+
LEDGER_PATH = Path(".evalseal/ledger.jsonl")
|
|
10
|
+
GENESIS = "GENESIS"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _content_hash(record: RunRecord) -> str:
|
|
14
|
+
# Hash everything except the hash field itself.
|
|
15
|
+
payload = record.model_dump(mode="json")
|
|
16
|
+
payload.pop("hash", None)
|
|
17
|
+
blob = json.dumps(payload, sort_keys=True, separators=(",", ":"))
|
|
18
|
+
return "sha256:" + hashlib.sha256(blob.encode()).hexdigest()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load_all(path: Path = LEDGER_PATH) -> list[RunRecord]:
|
|
22
|
+
if not path.exists():
|
|
23
|
+
return []
|
|
24
|
+
return [
|
|
25
|
+
RunRecord.model_validate_json(line)
|
|
26
|
+
for line in path.read_text().splitlines()
|
|
27
|
+
if line.strip()
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def last_hash(path: Path = LEDGER_PATH) -> str:
|
|
32
|
+
recs = load_all(path)
|
|
33
|
+
return recs[-1].hash if recs else GENESIS
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def seal_and_append(record: RunRecord, path: Path = LEDGER_PATH) -> RunRecord:
|
|
37
|
+
expected_prev = last_hash(path)
|
|
38
|
+
if record.prev_hash != expected_prev:
|
|
39
|
+
raise ValueError(
|
|
40
|
+
f"record.prev_hash {record.prev_hash[:20]} does not link to ledger head "
|
|
41
|
+
f"{expected_prev[:20]}; refusing to append a broken chain."
|
|
42
|
+
)
|
|
43
|
+
record.hash = _content_hash(record)
|
|
44
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
with path.open("a") as f:
|
|
46
|
+
f.write(record.model_dump_json() + "\n")
|
|
47
|
+
return record
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def verify_chain(path: Path = LEDGER_PATH) -> tuple[bool, str]:
|
|
51
|
+
"""Returns (ok, message). Detects a tampered score or a broken prev-link."""
|
|
52
|
+
recs = load_all(path)
|
|
53
|
+
prev = GENESIS
|
|
54
|
+
for i, r in enumerate(recs):
|
|
55
|
+
if r.prev_hash != prev:
|
|
56
|
+
return (False, f"Broken chain at record {i}: prev_hash mismatch.")
|
|
57
|
+
if _content_hash(r) != r.hash:
|
|
58
|
+
return (False, f"TAMPER DETECTED at record {i}: content hash does not match.")
|
|
59
|
+
prev = r.hash
|
|
60
|
+
return (True, f"Chain intact: {len(recs)} record(s).")
|
evalseal/models.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from datetime import datetime, timezone
|
|
4
|
+
from typing import Literal, Optional
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, Field
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _now() -> str:
|
|
10
|
+
return datetime.now(timezone.utc).isoformat()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class EffectiveParams(BaseModel):
|
|
14
|
+
temperature: Optional[float] = None
|
|
15
|
+
seed: Optional[int] = None
|
|
16
|
+
top_p: Optional[float] = None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TargetProvenance(BaseModel):
|
|
20
|
+
requested_model: str
|
|
21
|
+
served_model: Optional[str] = None # from response; WARN if != requested
|
|
22
|
+
system_fingerprint: Optional[str] = None
|
|
23
|
+
base_url: str
|
|
24
|
+
effective_params: EffectiveParams
|
|
25
|
+
params_source: Literal["explicit", "provider_default"] = "explicit"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ScorerProvenance(BaseModel):
|
|
29
|
+
type: Literal["exact", "regex", "llm_judge"]
|
|
30
|
+
judge: Optional[TargetProvenance] = None # the judge is a target too
|
|
31
|
+
rubric_hash: Optional[str] = None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class DatasetProvenance(BaseModel):
|
|
35
|
+
hash: str
|
|
36
|
+
n_cases: int
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class RunConfig(BaseModel):
|
|
40
|
+
n_repeats: int = 5
|
|
41
|
+
harness_version: str = "evalseal/0.1.0"
|
|
42
|
+
started_at: str = Field(default_factory=_now)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ProvenanceManifest(BaseModel):
|
|
46
|
+
schema_version: str = "1.0"
|
|
47
|
+
target: TargetProvenance
|
|
48
|
+
scorer: ScorerProvenance
|
|
49
|
+
dataset: DatasetProvenance
|
|
50
|
+
run_config: RunConfig
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class CaseResult(BaseModel):
|
|
54
|
+
case_id: str
|
|
55
|
+
scores: list[float]
|
|
56
|
+
mean: float
|
|
57
|
+
ci95: tuple[float, float]
|
|
58
|
+
flip_rate: float
|
|
59
|
+
stability: str
|
|
60
|
+
majority_verdict: Optional[int] = None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class Aggregate(BaseModel):
|
|
64
|
+
n_cases: int
|
|
65
|
+
mean_score: float
|
|
66
|
+
n_stable: int
|
|
67
|
+
n_borderline: int
|
|
68
|
+
n_unstable: int
|
|
69
|
+
warnings: list[str] = []
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class RunRecord(BaseModel):
|
|
73
|
+
manifest: ProvenanceManifest
|
|
74
|
+
results: list[CaseResult]
|
|
75
|
+
aggregate: Aggregate
|
|
76
|
+
prev_hash: str
|
|
77
|
+
hash: str = "" # filled by the ledger at seal time
|
|
78
|
+
created_at: str = Field(default_factory=_now)
|
evalseal/report.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from .models import RunRecord
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def write_json(record: RunRecord, path: str | Path = "report.json") -> None:
|
|
9
|
+
Path(path).write_text(record.model_dump_json(indent=2))
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def to_markdown(record: RunRecord) -> str:
|
|
13
|
+
a = record.aggregate
|
|
14
|
+
t = record.manifest.target
|
|
15
|
+
judge = record.manifest.scorer.judge
|
|
16
|
+
lines = [
|
|
17
|
+
"# EvalSeal run\n",
|
|
18
|
+
f"- Model requested: `{t.requested_model}`",
|
|
19
|
+
f"- Served: `{t.served_model}` | fingerprint: `{t.system_fingerprint}`",
|
|
20
|
+
f"- Temperature: `{t.effective_params.temperature}` ({t.params_source})",
|
|
21
|
+
f"- Scorer: `{record.manifest.scorer.type}`"
|
|
22
|
+
+ (f" | judge: `{judge.served_model or judge.requested_model}`"
|
|
23
|
+
f", temperature `{judge.effective_params.temperature}` ({judge.params_source})"
|
|
24
|
+
if judge else ""),
|
|
25
|
+
f"- N repeats: {record.manifest.run_config.n_repeats}",
|
|
26
|
+
f"- Dataset: `{record.manifest.dataset.hash[:20]}...` ({record.manifest.dataset.n_cases} cases)",
|
|
27
|
+
f"- Sealed hash: `{record.hash[:20]}...`\n",
|
|
28
|
+
f"**Aggregate:** {a.n_cases} cases · mean {a.mean_score:.2f} · "
|
|
29
|
+
f"{a.n_stable} stable / {a.n_borderline} borderline / {a.n_unstable} unstable\n",
|
|
30
|
+
]
|
|
31
|
+
if a.warnings:
|
|
32
|
+
lines.append("## ⚠ Provenance warnings")
|
|
33
|
+
lines += [f"- {w}" for w in a.warnings]
|
|
34
|
+
lines.append("")
|
|
35
|
+
lines.append("## Per-case reproducibility\n")
|
|
36
|
+
lines.append("| case | verdicts | mean | 95% CI | flip rate | stability |")
|
|
37
|
+
lines.append("|---|---|---|---|---|---|")
|
|
38
|
+
for r in record.results:
|
|
39
|
+
verdicts = "".join("P" if s >= 0.5 else "F" for s in r.scores)
|
|
40
|
+
lines.append(
|
|
41
|
+
f"| {r.case_id} | `{verdicts}` | {r.mean:.2f} | [{r.ci95[0]:.2f}, {r.ci95[1]:.2f}] "
|
|
42
|
+
f"| {r.flip_rate:.0%} | {r.stability} |"
|
|
43
|
+
)
|
|
44
|
+
return "\n".join(lines) + "\n"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def write_markdown(record: RunRecord, path: str | Path = "report.md") -> None:
|
|
48
|
+
Path(path).write_text(to_markdown(record))
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: evalseal
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Reproducibility and provenance receipts for LLM evaluations.
|
|
5
|
+
Project-URL: Homepage, https://github.com/patibandlavenkatamanideep/evalseal
|
|
6
|
+
Project-URL: Issues, https://github.com/patibandlavenkatamanideep/evalseal/issues
|
|
7
|
+
Author: Venkata Manideep Patibandla
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: evaluation,llm,llm-as-judge,provenance,reproducibility
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Requires-Python: >=3.11
|
|
22
|
+
Requires-Dist: httpx>=0.27
|
|
23
|
+
Requires-Dist: pydantic>=2.6
|
|
24
|
+
Requires-Dist: rich>=13.7
|
|
25
|
+
Requires-Dist: typer>=0.12
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest-cov>=5.0; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# EvalSeal
|
|
32
|
+
|
|
33
|
+
**Reproducibility receipts for LLM evals: run it N times, report the score with its noise, and seal what actually ran.**
|
|
34
|
+
|
|
35
|
+
## The problem
|
|
36
|
+
|
|
37
|
+
An eval score from a single run is one sample of a random process. Run the same eval again
|
|
38
|
+
and borderline items quietly flip from PASS to FAIL, especially when an LLM judge grades
|
|
39
|
+
them. Reports also name the model you *asked* for, not the one that *answered*. EvalSeal
|
|
40
|
+
measures the flips, records the real provenance, and seals both into a tamper-evident ledger.
|
|
41
|
+
|
|
42
|
+
## Quickstart
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install evalseal
|
|
46
|
+
|
|
47
|
+
# The recorded demo (examples + cassette) lives in the repo.
|
|
48
|
+
git clone https://github.com/patibandlavenkatamanideep/evalseal && cd evalseal
|
|
49
|
+
|
|
50
|
+
# Replays the committed cassette: no API key, no network.
|
|
51
|
+
evalseal run \
|
|
52
|
+
--dataset examples/borderline_judge/dataset.jsonl \
|
|
53
|
+
--target-config examples/borderline_judge/target.json \
|
|
54
|
+
--scorer-config examples/borderline_judge/scorer.json \
|
|
55
|
+
--n 5
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Real flip rates
|
|
59
|
+
|
|
60
|
+
Recorded 2026-09-15 and committed in `tests/cassettes/run.json`: `gemini-2.5-flash` as both
|
|
61
|
+
the target and the judge, temperature left at the provider default, 20 arguable prompts,
|
|
62
|
+
5 runs each. The quickstart above replays exactly this run.
|
|
63
|
+
|
|
64
|
+
**Mean score: 0.92, but 5 of 20 cases did not get the same verdict every time.**
|
|
65
|
+
|
|
66
|
+
| case | prompt | verdicts | mean | 95% CI | flip rate | stability |
|
|
67
|
+
|---|---|---|---|---|---|---|
|
|
68
|
+
| b01 | Is a hot dog a sandwich? | `FPPFP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
|
|
69
|
+
| b10 | Blockchain for a child in exactly 20 words | `FPFPP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
|
|
70
|
+
| b16 | "Do we only use 10% of our brains?" in a jokey tone | `PPFFP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
|
|
71
|
+
| b05 | A borderline-polite refusal to a coworker | `PFPPP` | 0.80 | [0.40, 1.00] | 20% | BORDERLINE |
|
|
72
|
+
| b19 | A technically accurate haiku about recursion | `PPFPP` | 0.80 | [0.40, 1.00] | 20% | BORDERLINE |
|
|
73
|
+
| 15 others | | `PPPPP` | 1.00 | [1.00, 1.00] | 0% | STABLE |
|
|
74
|
+
|
|
75
|
+
Treat the k-th repeat of every case as one ordinary single-run eval, and the five
|
|
76
|
+
"single runs" of this identical eval scored **0.90, 0.95, 0.85, 0.90 and 1.00**. A single
|
|
77
|
+
run can't tell you which of those numbers you got.
|
|
78
|
+
|
|
79
|
+
The report also flagged `TEMPERATURE NOT SET` for both the target and the judge, which is
|
|
80
|
+
the reason these borderline verdicts can come out differently from run to run.
|
|
81
|
+
|
|
82
|
+
To re-record with your own key, copy `.env.example` to `.env`, add a free
|
|
83
|
+
[Google AI Studio](https://aistudio.google.com/apikey) key, and run the quickstart with
|
|
84
|
+
`EVALSEAL_RECORD=1`. If the free tier rate-limits you, run the same command again later;
|
|
85
|
+
responses already recorded are kept.
|
|
86
|
+
|
|
87
|
+
## Commands
|
|
88
|
+
|
|
89
|
+
| command | what it does | exit code |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| `evalseal run` | Runs each case N times, analyzes variance, seals a record, writes `report.json` + `report.md`. | `0` all stable/borderline · `3` any case UNSTABLE · `1` error |
|
|
92
|
+
| `evalseal verify` | Recomputes every hash in `.evalseal/ledger.jsonl` and checks the chain links. | `0` intact · `1` tampered or broken |
|
|
93
|
+
| `evalseal diff A B` | Compares two ledger runs and says whether the mean moved beyond the noise floor. Use `--` for negative indices: `evalseal diff -- 0 -1`. | `0` |
|
|
94
|
+
|
|
95
|
+
**Stability classes** are based on the flip rate, the share of a case's N verdicts that
|
|
96
|
+
disagree with its majority: `STABLE` (0), `BORDERLINE` (≤ 20%), `UNSTABLE` (> 20%).
|
|
97
|
+
|
|
98
|
+
**Provenance warnings** show up in the report when:
|
|
99
|
+
- the served model differs from the requested one (for the target or the judge),
|
|
100
|
+
- the endpoint isn't a canonical provider host,
|
|
101
|
+
- temperature was left at the provider default,
|
|
102
|
+
- the served model or system fingerprint changed partway through the run.
|
|
103
|
+
|
|
104
|
+
## How it works
|
|
105
|
+
|
|
106
|
+
The executor sends each prompt to the target N times and scores every response. An LLM
|
|
107
|
+
judge is itself a target, so its own randomness is measured instead of assumed away.
|
|
108
|
+
`analyze.py` computes the mean, a seeded bootstrap 95% CI, and the flip rate for each case.
|
|
109
|
+
Every request goes through a cassette. In record mode, real responses are saved in call
|
|
110
|
+
order; in replay mode, which is the default and what CI uses, they are served back, and a
|
|
111
|
+
missing entry fails loudly. Each run is saved as a `RunRecord`: its manifest (requested vs.
|
|
112
|
+
served model, fingerprint, parameters and whether they were set explicitly, rubric hash,
|
|
113
|
+
dataset hash) plus its results. The record is hashed and linked to the previous record's
|
|
114
|
+
hash in an append-only JSONL ledger, so editing any past score breaks `verify`.
|
|
115
|
+
|
|
116
|
+
See [DESIGN.md](https://github.com/patibandlavenkatamanideep/evalseal/blob/main/DESIGN.md) for what this does and does not prove.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
evalseal/__init__.py,sha256=prMy07gYwx9H-vSb9Qcm68fl8fslACzZLFSDiqCVocU,100
|
|
2
|
+
evalseal/analyze.py,sha256=uzb-QL9tvwo9_qsLlvVyz2VY0l1zjY6fId5w7tyWihY,2894
|
|
3
|
+
evalseal/cli.py,sha256=ZhabImU-yGbhCr1_2u-O8fTclynCQRERZiR9N8CfKaw,5882
|
|
4
|
+
evalseal/executor.py,sha256=FdmNDAJVmUoEdeNFjgwTPfxmLoTKflwNqdZLUrA2Ye0,5523
|
|
5
|
+
evalseal/ledger.py,sha256=SRh5Abs4-MkfCzsZ2SbbciN_WTRV-uzfVywQ4R6i3tQ,1946
|
|
6
|
+
evalseal/models.py,sha256=Ct8TYa0VPfy6upI-Yg-9mGs9uYJLNsBcgDMk2dU0ZXI,1895
|
|
7
|
+
evalseal/report.py,sha256=sb1XWsY2cyLhxq5lPC7K7x3WenQsg5YRM9UCJj2CPGE,2039
|
|
8
|
+
evalseal/adapters/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
9
|
+
evalseal/adapters/dataset.py,sha256=JMaP5dMyGmoSTaDW59IKNhCl8n3RJBu5IKCpq7Z72Mg,950
|
|
10
|
+
evalseal/adapters/recording.py,sha256=0xyaSD8cJliFvkEkAK3Wqjf37iplKUTAOjGYJqkHvZY,2278
|
|
11
|
+
evalseal/adapters/scorer.py,sha256=W6ywL_cCdlZVkajDx_5Yzw05B6ROFtb7RBJAi_vsx1U,2220
|
|
12
|
+
evalseal/adapters/target.py,sha256=zgpSk34KLjAqhmPc-STE5alRGrPDCkZO_B5QzXuly4c,3250
|
|
13
|
+
evalseal-0.1.0.dist-info/METADATA,sha256=oo795y8bdeDFbuSpHiYfmTRH0l35UqP-9h9oJbbdKcQ,5844
|
|
14
|
+
evalseal-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
15
|
+
evalseal-0.1.0.dist-info/entry_points.txt,sha256=j-GPjykIrYuVnOWt-2QaEaXQEpedrXNeA3vs-YwXHQY,46
|
|
16
|
+
evalseal-0.1.0.dist-info/licenses/LICENSE,sha256=BUR4NJbROoLQlsxDQaqjIk8ThVqyIjc9AfXORDMlv9A,1084
|
|
17
|
+
evalseal-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Venkata Manideep Patibandla
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|