evalseal 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evalseal/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """EvalSeal: reproducibility and provenance receipts for LLM evaluations."""
2
+
3
+ __version__ = "0.1.0"
File without changes
@@ -0,0 +1,36 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+
8
+
9
+ @dataclass
10
+ class Case:
11
+ case_id: str
12
+ prompt: str
13
+ expected: str | None = None
14
+
15
+
16
+ @dataclass
17
+ class Dataset:
18
+ cases: list[Case]
19
+ hash: str
20
+
21
+ @classmethod
22
+ def from_jsonl(cls, path: str | Path) -> "Dataset":
23
+ raw = Path(path).read_text()
24
+ cases = []
25
+ seen: set[str] = set()
26
+ for lineno, line in enumerate(raw.splitlines(), start=1):
27
+ line = line.strip()
28
+ if not line:
29
+ continue
30
+ d = json.loads(line)
31
+ if d["case_id"] in seen:
32
+ raise ValueError(f"{path}:{lineno}: duplicate case_id {d['case_id']!r}")
33
+ seen.add(d["case_id"])
34
+ cases.append(Case(d["case_id"], d["prompt"], d.get("expected")))
35
+ h = "sha256:" + hashlib.sha256(raw.encode()).hexdigest()
36
+ return cls(cases, h)
@@ -0,0 +1,56 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import os
6
+ from collections import Counter
7
+ from pathlib import Path
8
+ from typing import Callable
9
+
10
+
11
+ def request_key(payload: dict) -> str:
12
+ """Stable hash of the *effective* request, so identical requests hit the cassette."""
13
+ blob = json.dumps(payload, sort_keys=True, separators=(",", ":"))
14
+ return hashlib.sha256(blob.encode()).hexdigest()
15
+
16
+
17
+ class Cassette:
18
+ """JSON-backed record/replay store keyed by request hash.
19
+
20
+ Each key maps to an ordered *list* of responses. Repeating an identical request is
21
+ the whole point of an N-run eval, so the k-th call with a given key is served the
22
+ k-th recorded response. (A single response per key would replay repeat #1 N times
23
+ and hide every flip.)
24
+
25
+ Mode is controlled by env var EVALSEAL_RECORD (or the `record` argument):
26
+ unset/0 -> replay only (raises if a request is missing) — this is CI mode
27
+ 1 -> record: already-recorded responses are reused, new calls are captured
28
+ """
29
+
30
+ def __init__(self, path: str | Path, record: bool | None = None):
31
+ self.path = Path(path)
32
+ if record is None:
33
+ record = os.environ.get("EVALSEAL_RECORD", "0") == "1"
34
+ self.record = record
35
+ self._data: dict[str, list[dict]] = {}
36
+ self._cursor: Counter[str] = Counter()
37
+ if self.path.exists():
38
+ self._data = json.loads(self.path.read_text())
39
+
40
+ def call(self, payload: dict, do_real_call: Callable[[], dict]) -> dict:
41
+ key = request_key(payload)
42
+ index = self._cursor[key]
43
+ self._cursor[key] += 1
44
+ recorded = self._data.get(key, [])
45
+ if index < len(recorded):
46
+ return recorded[index]
47
+ if not self.record:
48
+ raise RuntimeError(
49
+ f"No cassette entry #{index} for request {key[:12]} in {self.path} and "
50
+ f"recording is off. Run with EVALSEAL_RECORD=1 and an API key to capture it."
51
+ )
52
+ resp = do_real_call() # real network call, record mode only
53
+ self._data.setdefault(key, []).append(resp)
54
+ self.path.parent.mkdir(parents=True, exist_ok=True)
55
+ self.path.write_text(json.dumps(self._data, indent=2, sort_keys=True))
56
+ return resp
@@ -0,0 +1,74 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import re
5
+ from dataclasses import dataclass
6
+ from typing import Optional, Protocol, runtime_checkable
7
+
8
+ from .target import Target, TargetResponse
9
+
10
+ _VERDICT_RE = re.compile(r"\b(PASS|FAIL)\b")
11
+
12
+
13
+ @dataclass
14
+ class ScoreResult:
15
+ score: float # 0/1 for binary scorers
16
+ binary: bool
17
+ verdict: Optional[int] # 0/1 or None
18
+ judge_response: Optional[TargetResponse] = None
19
+
20
+
21
+ @runtime_checkable
22
+ class Scorer(Protocol):
23
+ kind: str
24
+ def score(self, prompt: str, response_text: str, expected: Optional[str]) -> ScoreResult: ...
25
+
26
+
27
+ @dataclass
28
+ class ExactMatchScorer:
29
+ kind: str = "exact"
30
+
31
+ def score(self, prompt, response_text, expected):
32
+ ok = expected is not None and response_text.strip() == expected.strip()
33
+ return ScoreResult(float(ok), True, int(ok))
34
+
35
+
36
+ @dataclass
37
+ class RegexScorer:
38
+ pattern: str
39
+ kind: str = "regex"
40
+
41
+ def score(self, prompt, response_text, expected):
42
+ ok = re.search(self.pattern, response_text) is not None
43
+ return ScoreResult(float(ok), True, int(ok))
44
+
45
+
46
+ def parse_verdict(text: str) -> int:
47
+ """First standalone PASS/FAIL token wins; anything unparseable counts as FAIL.
48
+ (A plain substring check would read "FAIL — this does not PASS" as a pass.)"""
49
+ m = _VERDICT_RE.search(text.upper())
50
+ return 1 if m and m.group(1) == "PASS" else 0
51
+
52
+
53
+ @dataclass
54
+ class LLMJudgeScorer:
55
+ """Grades a response pass/fail using another model. The judge is a Target, so it
56
+ gets its own recording + provenance. This is where flips are most visible."""
57
+ judge: Target
58
+ rubric: str
59
+ kind: str = "llm_judge"
60
+
61
+ @property
62
+ def rubric_hash(self) -> str:
63
+ return "sha256:" + hashlib.sha256(self.rubric.encode()).hexdigest()
64
+
65
+ def score(self, prompt, response_text, expected):
66
+ judge_prompt = (
67
+ f"{self.rubric}\n\n"
68
+ f"USER PROMPT:\n{prompt}\n\n"
69
+ f"RESPONSE TO GRADE:\n{response_text}\n\n"
70
+ f"Answer with exactly one word: PASS or FAIL."
71
+ )
72
+ jr = self.judge.generate(judge_prompt)
73
+ verdict = parse_verdict(jr.text)
74
+ return ScoreResult(float(verdict), True, verdict, judge_response=jr)
@@ -0,0 +1,97 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ from dataclasses import dataclass
5
+ from typing import Any, Callable, Optional, Protocol, runtime_checkable
6
+
7
+ import httpx
8
+
9
+ from .recording import Cassette
10
+
11
+
12
+ @dataclass
13
+ class TargetResponse:
14
+ text: str
15
+ requested_model: str
16
+ served_model: Optional[str]
17
+ system_fingerprint: Optional[str]
18
+ base_url: str
19
+ effective_params: dict[str, Any]
20
+ params_source: str # "explicit" | "provider_default"
21
+
22
+
23
+ @runtime_checkable
24
+ class Target(Protocol):
25
+ def generate(self, prompt: str) -> TargetResponse: ...
26
+
27
+
28
+ @dataclass
29
+ class LocalCallableTarget:
30
+ """Wrap any Python function as a target. Great for offline tests: the function
31
+ can return scripted-variance outputs so the executor is testable with no network."""
32
+ fn: Callable[[str], str]
33
+ name: str = "local"
34
+
35
+ def generate(self, prompt: str) -> TargetResponse:
36
+ return TargetResponse(
37
+ text=self.fn(prompt),
38
+ requested_model=self.name,
39
+ served_model=self.name,
40
+ system_fingerprint=None,
41
+ base_url="local://",
42
+ effective_params={},
43
+ params_source="explicit",
44
+ )
45
+
46
+
47
+ @dataclass
48
+ class OpenAICompatibleTarget:
49
+ """Calls any OpenAI-compatible /chat/completions endpoint, through the cassette."""
50
+ model: str
51
+ cassette: Cassette
52
+ base_url: str = "https://api.openai.com/v1"
53
+ temperature: Optional[float] = None # None means "we did not set it"
54
+ seed: Optional[int] = None
55
+ api_key_env: str = "EVALSEAL_API_KEY"
56
+
57
+ def generate(self, prompt: str) -> TargetResponse:
58
+ # Record whether we explicitly set temperature or are inheriting the default.
59
+ params_source = "explicit" if self.temperature is not None else "provider_default"
60
+ body: dict[str, Any] = {
61
+ "model": self.model,
62
+ "messages": [{"role": "user", "content": prompt}],
63
+ }
64
+ if self.temperature is not None:
65
+ body["temperature"] = self.temperature
66
+ if self.seed is not None:
67
+ body["seed"] = self.seed
68
+
69
+ def do_real_call() -> dict:
70
+ key = os.environ.get(self.api_key_env)
71
+ if not key:
72
+ raise RuntimeError(f"Recording needs an API key in ${self.api_key_env}.")
73
+ with httpx.Client(timeout=60) as c:
74
+ r = c.post(
75
+ f"{self.base_url.rstrip('/')}/chat/completions",
76
+ headers={"Authorization": f"Bearer {key}"},
77
+ json=body,
78
+ )
79
+ r.raise_for_status()
80
+ return r.json()
81
+
82
+ # Cassette key is the full effective body (never the API key).
83
+ raw = self.cassette.call({"url": self.base_url, "body": body}, do_real_call)
84
+
85
+ return TargetResponse(
86
+ text=raw["choices"][0]["message"]["content"] or "",
87
+ requested_model=self.model,
88
+ served_model=raw.get("model"),
89
+ system_fingerprint=raw.get("system_fingerprint"),
90
+ base_url=self.base_url,
91
+ effective_params={
92
+ "temperature": self.temperature,
93
+ "seed": self.seed,
94
+ "top_p": None,
95
+ },
96
+ params_source=params_source,
97
+ )
evalseal/analyze.py ADDED
@@ -0,0 +1,82 @@
1
+ """Variance analysis for repeated eval runs. Pure functions, no I/O, no network.
2
+
3
+ A "verdict" is a binary pass(1)/fail(0). A "score" may be binary or a float in [0,1].
4
+ We quantify reproducibility; we never claim determinism.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import random
9
+ import statistics
10
+ from collections import Counter
11
+ from dataclasses import dataclass
12
+
13
+ # Stability thresholds (constants so they're auditable, not magic numbers).
14
+ BORDERLINE_MAX_FLIP = 0.20 # flip_rate in (0, 0.20] -> BORDERLINE
15
+ # flip_rate == 0 -> STABLE ; flip_rate > BORDERLINE_MAX_FLIP -> UNSTABLE
16
+
17
+ _BOOTSTRAP_ITERS = 2000
18
+ _RNG_SEED = 12345 # fixed so the CI computation itself is reproducible
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class CaseStats:
23
+ mean: float
24
+ ci95_low: float
25
+ ci95_high: float
26
+ flip_rate: float
27
+ stability: str # "STABLE" | "BORDERLINE" | "UNSTABLE"
28
+ majority_verdict: int | None # for binary; None for pure-float scores
29
+
30
+
31
+ def _bootstrap_ci(scores: list[float], iters: int = _BOOTSTRAP_ITERS) -> tuple[float, float]:
32
+ """Percentile bootstrap 95% CI of the mean. Deterministic via fixed seed."""
33
+ n = len(scores)
34
+ if n == 0:
35
+ return (0.0, 0.0)
36
+ if n == 1:
37
+ return (scores[0], scores[0])
38
+ rng = random.Random(_RNG_SEED)
39
+ means = []
40
+ for _ in range(iters):
41
+ sample = [scores[rng.randrange(n)] for _ in range(n)]
42
+ means.append(sum(sample) / n)
43
+ means.sort()
44
+ lo = means[int(0.025 * iters)]
45
+ hi = means[int(0.975 * iters)]
46
+ return (lo, hi)
47
+
48
+
49
+ def flip_rate(verdicts: list[int]) -> tuple[float, int]:
50
+ """Fraction of verdicts disagreeing with the majority. Returns (rate, majority)."""
51
+ if not verdicts:
52
+ return (0.0, 0)
53
+ counts = Counter(verdicts)
54
+ majority, majority_count = counts.most_common(1)[0]
55
+ disagree = len(verdicts) - majority_count
56
+ return (disagree / len(verdicts), majority)
57
+
58
+
59
+ def classify_stability(rate: float) -> str:
60
+ if rate == 0.0:
61
+ return "STABLE"
62
+ if rate <= BORDERLINE_MAX_FLIP:
63
+ return "BORDERLINE"
64
+ return "UNSTABLE"
65
+
66
+
67
+ def analyze_case(scores: list[float], *, binary: bool) -> CaseStats:
68
+ """Turn N per-run scores into a CaseStats. If binary, scores must be 0/1."""
69
+ if not scores:
70
+ raise ValueError("analyze_case requires at least one score")
71
+ mean = sum(scores) / len(scores)
72
+ lo, hi = _bootstrap_ci(scores)
73
+ if binary:
74
+ verdicts = [int(s) for s in scores]
75
+ rate, majority = flip_rate(verdicts)
76
+ return CaseStats(mean, lo, hi, rate, classify_stability(rate), majority)
77
+ # Float scores: define a "flip" as crossing the run-set's own median.
78
+ # This gives a variance signal without a fixed threshold assumption.
79
+ med = statistics.median(scores)
80
+ pseudo = [1 if s >= med else 0 for s in scores]
81
+ rate, _ = flip_rate(pseudo)
82
+ return CaseStats(mean, lo, hi, rate, classify_stability(rate), None)
evalseal/cli.py ADDED
@@ -0,0 +1,152 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ from pathlib import Path
6
+
7
+ import httpx
8
+ import typer
9
+ from rich.console import Console
10
+ from rich.markdown import Markdown
11
+
12
+ from .adapters.dataset import Dataset
13
+ from .adapters.recording import Cassette
14
+ from .adapters.scorer import ExactMatchScorer, LLMJudgeScorer, RegexScorer
15
+ from .adapters.target import OpenAICompatibleTarget
16
+ from .executor import run_eval
17
+ from .ledger import LEDGER_PATH, last_hash, load_all, seal_and_append, verify_chain
18
+ from .report import to_markdown, write_json, write_markdown
19
+
20
+ # Distinct from 1 (uncaught error) and 2 (usage error) so CI can tell
21
+ # "the eval is unstable" apart from "the tool broke".
22
+ EXIT_UNSTABLE = 3
23
+
24
+ app = typer.Typer(add_completion=False, help="Reproducibility receipts for LLM evals.")
25
+ console = Console()
26
+
27
+ LedgerOpt = typer.Option(LEDGER_PATH, "--ledger", help="Path to the ledger JSONL file.")
28
+
29
+
30
+ def load_dotenv(path: Path = Path(".env")) -> None:
31
+ """Load KEY=VALUE lines from ./.env. Variables already in the environment win."""
32
+ if not path.is_file():
33
+ return
34
+ for line in path.read_text().splitlines():
35
+ line = line.strip()
36
+ if not line or line.startswith("#") or "=" not in line:
37
+ continue
38
+ key, _, value = line.partition("=")
39
+ key = key.strip().removeprefix("export ").strip()
40
+ value = value.strip().strip("\"'")
41
+ if key and value:
42
+ os.environ.setdefault(key, value)
43
+
44
+
45
+ @app.callback()
46
+ def _main() -> None:
47
+ load_dotenv()
48
+
49
+
50
+ def _build_target(cfg: dict, cassette: Cassette) -> OpenAICompatibleTarget:
51
+ return OpenAICompatibleTarget(
52
+ model=cfg["model"],
53
+ cassette=cassette,
54
+ base_url=cfg.get("base_url", "https://api.openai.com/v1"),
55
+ temperature=cfg.get("temperature"), # omit in config to surface the "unset" warning
56
+ seed=cfg.get("seed"),
57
+ )
58
+
59
+
60
+ def _build_scorer(cfg: dict, cassette: Cassette):
61
+ if cfg["type"] == "exact":
62
+ return ExactMatchScorer()
63
+ if cfg["type"] == "regex":
64
+ return RegexScorer(pattern=cfg["pattern"])
65
+ if cfg["type"] == "llm_judge":
66
+ judge = OpenAICompatibleTarget(
67
+ model=cfg["judge_model"],
68
+ cassette=cassette,
69
+ base_url=cfg.get("base_url", "https://api.openai.com/v1"),
70
+ temperature=cfg.get("judge_temperature"), # leave unset to demonstrate flips
71
+ seed=cfg.get("judge_seed"),
72
+ )
73
+ return LLMJudgeScorer(judge=judge, rubric=cfg["rubric"])
74
+ raise typer.BadParameter(f"unknown scorer type {cfg['type']!r}")
75
+
76
+
77
+ @app.command()
78
+ def run(
79
+ dataset: Path = typer.Option(..., exists=True, dir_okay=False),
80
+ target_config: Path = typer.Option(..., exists=True, dir_okay=False),
81
+ scorer_config: Path = typer.Option(..., exists=True, dir_okay=False),
82
+ n: int = typer.Option(5, min=1, help="Repeats per case (flip rate needs N>=5)."),
83
+ cassette: Path = typer.Option(Path("tests/cassettes/run.json")),
84
+ ledger: Path = LedgerOpt,
85
+ ):
86
+ """Run an eval N times, seal the result, emit report.json + report.md."""
87
+ ds = Dataset.from_jsonl(dataset)
88
+ cass = Cassette(cassette)
89
+ target = _build_target(json.loads(target_config.read_text()), cass)
90
+ scorer = _build_scorer(json.loads(scorer_config.read_text()), cass)
91
+
92
+ try:
93
+ record = run_eval(ds, target, scorer, n_repeats=n, prev_hash=last_hash(ledger))
94
+ except RuntimeError as e:
95
+ console.print(f"[red]{e}[/red]")
96
+ raise typer.Exit(code=1)
97
+ except httpx.HTTPStatusError as e:
98
+ resp = e.response
99
+ console.print(f"[red]HTTP {resp.status_code} from provider: {resp.text[:300]}[/red]")
100
+ if resp.status_code == 429:
101
+ console.print(
102
+ "[yellow]Rate limited. Responses recorded so far are saved in the cassette; "
103
+ "re-run the same command later to resume.[/yellow]"
104
+ )
105
+ raise typer.Exit(code=1)
106
+ record = seal_and_append(record, ledger)
107
+ write_json(record)
108
+ write_markdown(record)
109
+ console.print(Markdown(to_markdown(record)))
110
+
111
+ # CI gate: dedicated non-zero exit if any case is UNSTABLE.
112
+ if record.aggregate.n_unstable > 0:
113
+ console.print(f"[red]{record.aggregate.n_unstable} unstable case(s) — failing.[/red]")
114
+ raise typer.Exit(code=EXIT_UNSTABLE)
115
+
116
+
117
+ @app.command()
118
+ def verify(ledger: Path = LedgerOpt):
119
+ """Check the ledger chain integrity (tamper detection)."""
120
+ ok, msg = verify_chain(ledger)
121
+ color = "green" if ok else "red"
122
+ console.print(f"[{color}]{msg}[/{color}]")
123
+ raise typer.Exit(code=0 if ok else 1)
124
+
125
+
126
+ @app.command()
127
+ def diff(
128
+ a: int = typer.Argument(..., help="Ledger index of the baseline run (negative ok)."),
129
+ b: int = typer.Argument(..., help="Ledger index of the candidate run (negative ok)."),
130
+ ledger: Path = LedgerOpt,
131
+ ):
132
+ """Compare two runs; state whether a score change exceeds the noise floor."""
133
+ recs = load_all(ledger)
134
+ for i in (a, b):
135
+ if not -len(recs) <= i < len(recs):
136
+ raise typer.BadParameter(f"index {i} out of range; ledger has {len(recs)} record(s)")
137
+ ra, rb = recs[a], recs[b]
138
+ ma, mb = ra.aggregate.mean_score, rb.aggregate.mean_score
139
+
140
+ # Noise floor: widest per-case CI half-width across both runs. Deliberately
141
+ # conservative — "REAL CHANGE" is only claimed when no single case's noise explains it.
142
+ def halfwidth(r):
143
+ return max(((c.ci95[1] - c.ci95[0]) / 2 for c in r.results), default=0.0)
144
+
145
+ noise = max(halfwidth(ra), halfwidth(rb))
146
+ delta = mb - ma
147
+ verdict = "within noise" if abs(delta) <= noise else "REAL CHANGE"
148
+ console.print(
149
+ f"mean {ma:.3f} -> {mb:.3f} (delta {delta:+.3f}, noise floor ±{noise:.3f}) => {verdict}"
150
+ )
151
+ if ra.manifest.dataset.hash != rb.manifest.dataset.hash:
152
+ console.print("[yellow]Warning: runs used different datasets.[/yellow]")
evalseal/executor.py ADDED
@@ -0,0 +1,142 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from urllib.parse import urlparse
5
+
6
+ from .adapters.dataset import Dataset
7
+ from .adapters.scorer import Scorer
8
+ from .adapters.target import Target, TargetResponse
9
+ from .analyze import analyze_case
10
+ from .models import (
11
+ Aggregate, CaseResult, DatasetProvenance, EffectiveParams,
12
+ ProvenanceManifest, RunConfig, RunRecord, ScorerProvenance, TargetProvenance,
13
+ )
14
+
15
+ _CANONICAL_HOSTS = {"api.openai.com", "generativelanguage.googleapis.com"}
16
+ _LOCAL_SCHEMES = {"local"}
17
+
18
+
19
+ def _to_provenance(tr: TargetResponse) -> TargetProvenance:
20
+ return TargetProvenance(
21
+ requested_model=tr.requested_model,
22
+ served_model=tr.served_model,
23
+ system_fingerprint=tr.system_fingerprint,
24
+ base_url=tr.base_url,
25
+ effective_params=EffectiveParams(**tr.effective_params),
26
+ params_source=tr.params_source,
27
+ )
28
+
29
+
30
+ def _is_same_model(requested: str, served: str) -> bool:
31
+ # Providers resolve aliases to dated snapshots (gpt-4o-mini -> gpt-4o-mini-2024-07-18,
32
+ # gpt-4 -> gpt-4-0613). That is the same model; the snapshot is still recorded in
33
+ # served_model. A bare prefix check is not enough: gpt-4o-mini is not gpt-4o.
34
+ snapshot = re.compile(rf"{re.escape(requested)}-(\d{{4}}-\d{{2}}-\d{{2}}|\d{{4}})")
35
+ return served == requested or snapshot.fullmatch(served) is not None
36
+
37
+
38
+ def _provenance_warnings(tp: TargetProvenance, role: str = "TARGET") -> list[str]:
39
+ w = []
40
+ if tp.served_model and tp.requested_model and not _is_same_model(
41
+ tp.requested_model, tp.served_model
42
+ ):
43
+ w.append(
44
+ f"{role} SERVED MODEL MISMATCH: requested '{tp.requested_model}' but served "
45
+ f"'{tp.served_model}'. The report's model name may not be what ran."
46
+ )
47
+ parsed = urlparse(tp.base_url)
48
+ if parsed.scheme not in _LOCAL_SCHEMES and parsed.hostname not in _CANONICAL_HOSTS:
49
+ w.append(
50
+ f"{role} NON-CANONICAL ENDPOINT: {tp.base_url} — responses may be proxied/altered."
51
+ )
52
+ if tp.params_source == "provider_default" and tp.effective_params.temperature is None:
53
+ w.append(
54
+ f"{role} TEMPERATURE NOT SET: the provider default (often 1.0) was used silently; "
55
+ "verdicts near the decision boundary may not be reproducible."
56
+ )
57
+ return w
58
+
59
+
60
+ def _drift_warnings(responses: list[TargetResponse], role: str) -> list[str]:
61
+ """The backend can change under you mid-run. Surface it instead of averaging over it."""
62
+ w = []
63
+ served = sorted({r.served_model for r in responses if r.served_model})
64
+ if len(served) > 1:
65
+ w.append(f"{role} SERVED MODEL CHANGED DURING RUN: {', '.join(served)}")
66
+ fps = sorted({r.system_fingerprint for r in responses if r.system_fingerprint})
67
+ if len(fps) > 1:
68
+ w.append(f"{role} SYSTEM FINGERPRINT CHANGED DURING RUN: {', '.join(fps)}")
69
+ return w
70
+
71
+
72
+ def _dedupe(items: list[str]) -> list[str]:
73
+ return list(dict.fromkeys(items))
74
+
75
+
76
+ def run_eval(
77
+ dataset: Dataset,
78
+ target: Target,
79
+ scorer: Scorer,
80
+ n_repeats: int = 5,
81
+ prev_hash: str = "GENESIS",
82
+ ) -> RunRecord:
83
+ if not dataset.cases:
84
+ raise ValueError("dataset has no cases")
85
+ if n_repeats < 1:
86
+ raise ValueError("n_repeats must be >= 1")
87
+
88
+ results: list[CaseResult] = []
89
+ target_resps: list[TargetResponse] = []
90
+ judge_resps: list[TargetResponse] = []
91
+
92
+ for case in dataset.cases:
93
+ scores: list[float] = []
94
+ binary = True
95
+ for _ in range(n_repeats):
96
+ tr = target.generate(case.prompt)
97
+ target_resps.append(tr)
98
+ sr = scorer.score(case.prompt, tr.text, case.expected)
99
+ if sr.judge_response is not None:
100
+ judge_resps.append(sr.judge_response)
101
+ scores.append(sr.score)
102
+ binary = binary and sr.binary
103
+ stats = analyze_case(scores, binary=binary)
104
+ results.append(CaseResult(
105
+ case_id=case.case_id,
106
+ scores=scores,
107
+ mean=stats.mean,
108
+ ci95=(stats.ci95_low, stats.ci95_high),
109
+ flip_rate=stats.flip_rate,
110
+ stability=stats.stability,
111
+ majority_verdict=stats.majority_verdict,
112
+ ))
113
+
114
+ # Every response is checked, not just the last one: a mismatch on any call matters.
115
+ warnings: list[str] = []
116
+ for tr in target_resps:
117
+ warnings += _provenance_warnings(_to_provenance(tr), "TARGET")
118
+ warnings += _drift_warnings(target_resps, "TARGET")
119
+ for jr in judge_resps:
120
+ warnings += _provenance_warnings(_to_provenance(jr), "JUDGE")
121
+ warnings += _drift_warnings(judge_resps, "JUDGE")
122
+
123
+ sp = ScorerProvenance(
124
+ type=scorer.kind,
125
+ judge=_to_provenance(judge_resps[0]) if judge_resps else None,
126
+ rubric_hash=getattr(scorer, "rubric_hash", None),
127
+ )
128
+ manifest = ProvenanceManifest(
129
+ target=_to_provenance(target_resps[0]),
130
+ scorer=sp,
131
+ dataset=DatasetProvenance(hash=dataset.hash, n_cases=len(dataset.cases)),
132
+ run_config=RunConfig(n_repeats=n_repeats),
133
+ )
134
+ agg = Aggregate(
135
+ n_cases=len(results),
136
+ mean_score=sum(r.mean for r in results) / len(results),
137
+ n_stable=sum(r.stability == "STABLE" for r in results),
138
+ n_borderline=sum(r.stability == "BORDERLINE" for r in results),
139
+ n_unstable=sum(r.stability == "UNSTABLE" for r in results),
140
+ warnings=_dedupe(warnings),
141
+ )
142
+ return RunRecord(manifest=manifest, results=results, aggregate=agg, prev_hash=prev_hash)
evalseal/ledger.py ADDED
@@ -0,0 +1,60 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+
7
+ from .models import RunRecord
8
+
9
+ LEDGER_PATH = Path(".evalseal/ledger.jsonl")
10
+ GENESIS = "GENESIS"
11
+
12
+
13
+ def _content_hash(record: RunRecord) -> str:
14
+ # Hash everything except the hash field itself.
15
+ payload = record.model_dump(mode="json")
16
+ payload.pop("hash", None)
17
+ blob = json.dumps(payload, sort_keys=True, separators=(",", ":"))
18
+ return "sha256:" + hashlib.sha256(blob.encode()).hexdigest()
19
+
20
+
21
+ def load_all(path: Path = LEDGER_PATH) -> list[RunRecord]:
22
+ if not path.exists():
23
+ return []
24
+ return [
25
+ RunRecord.model_validate_json(line)
26
+ for line in path.read_text().splitlines()
27
+ if line.strip()
28
+ ]
29
+
30
+
31
+ def last_hash(path: Path = LEDGER_PATH) -> str:
32
+ recs = load_all(path)
33
+ return recs[-1].hash if recs else GENESIS
34
+
35
+
36
+ def seal_and_append(record: RunRecord, path: Path = LEDGER_PATH) -> RunRecord:
37
+ expected_prev = last_hash(path)
38
+ if record.prev_hash != expected_prev:
39
+ raise ValueError(
40
+ f"record.prev_hash {record.prev_hash[:20]} does not link to ledger head "
41
+ f"{expected_prev[:20]}; refusing to append a broken chain."
42
+ )
43
+ record.hash = _content_hash(record)
44
+ path.parent.mkdir(parents=True, exist_ok=True)
45
+ with path.open("a") as f:
46
+ f.write(record.model_dump_json() + "\n")
47
+ return record
48
+
49
+
50
+ def verify_chain(path: Path = LEDGER_PATH) -> tuple[bool, str]:
51
+ """Returns (ok, message). Detects a tampered score or a broken prev-link."""
52
+ recs = load_all(path)
53
+ prev = GENESIS
54
+ for i, r in enumerate(recs):
55
+ if r.prev_hash != prev:
56
+ return (False, f"Broken chain at record {i}: prev_hash mismatch.")
57
+ if _content_hash(r) != r.hash:
58
+ return (False, f"TAMPER DETECTED at record {i}: content hash does not match.")
59
+ prev = r.hash
60
+ return (True, f"Chain intact: {len(recs)} record(s).")
evalseal/models.py ADDED
@@ -0,0 +1,78 @@
1
+ from __future__ import annotations
2
+
3
+ from datetime import datetime, timezone
4
+ from typing import Literal, Optional
5
+
6
+ from pydantic import BaseModel, Field
7
+
8
+
9
+ def _now() -> str:
10
+ return datetime.now(timezone.utc).isoformat()
11
+
12
+
13
+ class EffectiveParams(BaseModel):
14
+ temperature: Optional[float] = None
15
+ seed: Optional[int] = None
16
+ top_p: Optional[float] = None
17
+
18
+
19
+ class TargetProvenance(BaseModel):
20
+ requested_model: str
21
+ served_model: Optional[str] = None # from response; WARN if != requested
22
+ system_fingerprint: Optional[str] = None
23
+ base_url: str
24
+ effective_params: EffectiveParams
25
+ params_source: Literal["explicit", "provider_default"] = "explicit"
26
+
27
+
28
+ class ScorerProvenance(BaseModel):
29
+ type: Literal["exact", "regex", "llm_judge"]
30
+ judge: Optional[TargetProvenance] = None # the judge is a target too
31
+ rubric_hash: Optional[str] = None
32
+
33
+
34
+ class DatasetProvenance(BaseModel):
35
+ hash: str
36
+ n_cases: int
37
+
38
+
39
+ class RunConfig(BaseModel):
40
+ n_repeats: int = 5
41
+ harness_version: str = "evalseal/0.1.0"
42
+ started_at: str = Field(default_factory=_now)
43
+
44
+
45
+ class ProvenanceManifest(BaseModel):
46
+ schema_version: str = "1.0"
47
+ target: TargetProvenance
48
+ scorer: ScorerProvenance
49
+ dataset: DatasetProvenance
50
+ run_config: RunConfig
51
+
52
+
53
+ class CaseResult(BaseModel):
54
+ case_id: str
55
+ scores: list[float]
56
+ mean: float
57
+ ci95: tuple[float, float]
58
+ flip_rate: float
59
+ stability: str
60
+ majority_verdict: Optional[int] = None
61
+
62
+
63
+ class Aggregate(BaseModel):
64
+ n_cases: int
65
+ mean_score: float
66
+ n_stable: int
67
+ n_borderline: int
68
+ n_unstable: int
69
+ warnings: list[str] = []
70
+
71
+
72
+ class RunRecord(BaseModel):
73
+ manifest: ProvenanceManifest
74
+ results: list[CaseResult]
75
+ aggregate: Aggregate
76
+ prev_hash: str
77
+ hash: str = "" # filled by the ledger at seal time
78
+ created_at: str = Field(default_factory=_now)
evalseal/report.py ADDED
@@ -0,0 +1,48 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ from .models import RunRecord
6
+
7
+
8
+ def write_json(record: RunRecord, path: str | Path = "report.json") -> None:
9
+ Path(path).write_text(record.model_dump_json(indent=2))
10
+
11
+
12
+ def to_markdown(record: RunRecord) -> str:
13
+ a = record.aggregate
14
+ t = record.manifest.target
15
+ judge = record.manifest.scorer.judge
16
+ lines = [
17
+ "# EvalSeal run\n",
18
+ f"- Model requested: `{t.requested_model}`",
19
+ f"- Served: `{t.served_model}` | fingerprint: `{t.system_fingerprint}`",
20
+ f"- Temperature: `{t.effective_params.temperature}` ({t.params_source})",
21
+ f"- Scorer: `{record.manifest.scorer.type}`"
22
+ + (f" | judge: `{judge.served_model or judge.requested_model}`"
23
+ f", temperature `{judge.effective_params.temperature}` ({judge.params_source})"
24
+ if judge else ""),
25
+ f"- N repeats: {record.manifest.run_config.n_repeats}",
26
+ f"- Dataset: `{record.manifest.dataset.hash[:20]}...` ({record.manifest.dataset.n_cases} cases)",
27
+ f"- Sealed hash: `{record.hash[:20]}...`\n",
28
+ f"**Aggregate:** {a.n_cases} cases · mean {a.mean_score:.2f} · "
29
+ f"{a.n_stable} stable / {a.n_borderline} borderline / {a.n_unstable} unstable\n",
30
+ ]
31
+ if a.warnings:
32
+ lines.append("## ⚠ Provenance warnings")
33
+ lines += [f"- {w}" for w in a.warnings]
34
+ lines.append("")
35
+ lines.append("## Per-case reproducibility\n")
36
+ lines.append("| case | verdicts | mean | 95% CI | flip rate | stability |")
37
+ lines.append("|---|---|---|---|---|---|")
38
+ for r in record.results:
39
+ verdicts = "".join("P" if s >= 0.5 else "F" for s in r.scores)
40
+ lines.append(
41
+ f"| {r.case_id} | `{verdicts}` | {r.mean:.2f} | [{r.ci95[0]:.2f}, {r.ci95[1]:.2f}] "
42
+ f"| {r.flip_rate:.0%} | {r.stability} |"
43
+ )
44
+ return "\n".join(lines) + "\n"
45
+
46
+
47
+ def write_markdown(record: RunRecord, path: str | Path = "report.md") -> None:
48
+ Path(path).write_text(to_markdown(record))
@@ -0,0 +1,116 @@
1
+ Metadata-Version: 2.5
2
+ Name: evalseal
3
+ Version: 0.1.0
4
+ Summary: Reproducibility and provenance receipts for LLM evaluations.
5
+ Project-URL: Homepage, https://github.com/patibandlavenkatamanideep/evalseal
6
+ Project-URL: Issues, https://github.com/patibandlavenkatamanideep/evalseal/issues
7
+ Author: Venkata Manideep Patibandla
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: evaluation,llm,llm-as-judge,provenance,reproducibility
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Requires-Python: >=3.11
22
+ Requires-Dist: httpx>=0.27
23
+ Requires-Dist: pydantic>=2.6
24
+ Requires-Dist: rich>=13.7
25
+ Requires-Dist: typer>=0.12
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest-cov>=5.0; extra == 'dev'
28
+ Requires-Dist: pytest>=8.0; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # EvalSeal
32
+
33
+ **Reproducibility receipts for LLM evals: run it N times, report the score with its noise, and seal what actually ran.**
34
+
35
+ ## The problem
36
+
37
+ An eval score from a single run is one sample of a random process. Run the same eval again
38
+ and borderline items quietly flip from PASS to FAIL, especially when an LLM judge grades
39
+ them. Reports also name the model you *asked* for, not the one that *answered*. EvalSeal
40
+ measures the flips, records the real provenance, and seals both into a tamper-evident ledger.
41
+
42
+ ## Quickstart
43
+
44
+ ```bash
45
+ pip install evalseal
46
+
47
+ # The recorded demo (examples + cassette) lives in the repo.
48
+ git clone https://github.com/patibandlavenkatamanideep/evalseal && cd evalseal
49
+
50
+ # Replays the committed cassette: no API key, no network.
51
+ evalseal run \
52
+ --dataset examples/borderline_judge/dataset.jsonl \
53
+ --target-config examples/borderline_judge/target.json \
54
+ --scorer-config examples/borderline_judge/scorer.json \
55
+ --n 5
56
+ ```
57
+
58
+ ## Real flip rates
59
+
60
+ Recorded 2026-09-15 and committed in `tests/cassettes/run.json`: `gemini-2.5-flash` as both
61
+ the target and the judge, temperature left at the provider default, 20 arguable prompts,
62
+ 5 runs each. The quickstart above replays exactly this run.
63
+
64
+ **Mean score: 0.92, but 5 of 20 cases did not get the same verdict every time.**
65
+
66
+ | case | prompt | verdicts | mean | 95% CI | flip rate | stability |
67
+ |---|---|---|---|---|---|---|
68
+ | b01 | Is a hot dog a sandwich? | `FPPFP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
69
+ | b10 | Blockchain for a child in exactly 20 words | `FPFPP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
70
+ | b16 | "Do we only use 10% of our brains?" in a jokey tone | `PPFFP` | 0.60 | [0.20, 1.00] | 40% | UNSTABLE |
71
+ | b05 | A borderline-polite refusal to a coworker | `PFPPP` | 0.80 | [0.40, 1.00] | 20% | BORDERLINE |
72
+ | b19 | A technically accurate haiku about recursion | `PPFPP` | 0.80 | [0.40, 1.00] | 20% | BORDERLINE |
73
+ | 15 others | | `PPPPP` | 1.00 | [1.00, 1.00] | 0% | STABLE |
74
+
75
+ Treat the k-th repeat of every case as one ordinary single-run eval, and the five
76
+ "single runs" of this identical eval scored **0.90, 0.95, 0.85, 0.90 and 1.00**. A single
77
+ run can't tell you which of those numbers you got.
78
+
79
+ The report also flagged `TEMPERATURE NOT SET` for both the target and the judge, which is
80
+ the reason these borderline verdicts can come out differently from run to run.
81
+
82
+ To re-record with your own key, copy `.env.example` to `.env`, add a free
83
+ [Google AI Studio](https://aistudio.google.com/apikey) key, and run the quickstart with
84
+ `EVALSEAL_RECORD=1`. If the free tier rate-limits you, run the same command again later;
85
+ responses already recorded are kept.
86
+
87
+ ## Commands
88
+
89
+ | command | what it does | exit code |
90
+ |---|---|---|
91
+ | `evalseal run` | Runs each case N times, analyzes variance, seals a record, writes `report.json` + `report.md`. | `0` all stable/borderline · `3` any case UNSTABLE · `1` error |
92
+ | `evalseal verify` | Recomputes every hash in `.evalseal/ledger.jsonl` and checks the chain links. | `0` intact · `1` tampered or broken |
93
+ | `evalseal diff A B` | Compares two ledger runs and says whether the mean moved beyond the noise floor. Use `--` for negative indices: `evalseal diff -- 0 -1`. | `0` |
94
+
95
+ **Stability classes** are based on the flip rate, the share of a case's N verdicts that
96
+ disagree with its majority: `STABLE` (0), `BORDERLINE` (≤ 20%), `UNSTABLE` (> 20%).
97
+
98
+ **Provenance warnings** show up in the report when:
99
+ - the served model differs from the requested one (for the target or the judge),
100
+ - the endpoint isn't a canonical provider host,
101
+ - temperature was left at the provider default,
102
+ - the served model or system fingerprint changed partway through the run.
103
+
104
+ ## How it works
105
+
106
+ The executor sends each prompt to the target N times and scores every response. An LLM
107
+ judge is itself a target, so its own randomness is measured instead of assumed away.
108
+ `analyze.py` computes the mean, a seeded bootstrap 95% CI, and the flip rate for each case.
109
+ Every request goes through a cassette. In record mode, real responses are saved in call
110
+ order; in replay mode, which is the default and what CI uses, they are served back, and a
111
+ missing entry fails loudly. Each run is saved as a `RunRecord`: its manifest (requested vs.
112
+ served model, fingerprint, parameters and whether they were set explicitly, rubric hash,
113
+ dataset hash) plus its results. The record is hashed and linked to the previous record's
114
+ hash in an append-only JSONL ledger, so editing any past score breaks `verify`.
115
+
116
+ See [DESIGN.md](https://github.com/patibandlavenkatamanideep/evalseal/blob/main/DESIGN.md) for what this does and does not prove.
@@ -0,0 +1,17 @@
1
+ evalseal/__init__.py,sha256=prMy07gYwx9H-vSb9Qcm68fl8fslACzZLFSDiqCVocU,100
2
+ evalseal/analyze.py,sha256=uzb-QL9tvwo9_qsLlvVyz2VY0l1zjY6fId5w7tyWihY,2894
3
+ evalseal/cli.py,sha256=ZhabImU-yGbhCr1_2u-O8fTclynCQRERZiR9N8CfKaw,5882
4
+ evalseal/executor.py,sha256=FdmNDAJVmUoEdeNFjgwTPfxmLoTKflwNqdZLUrA2Ye0,5523
5
+ evalseal/ledger.py,sha256=SRh5Abs4-MkfCzsZ2SbbciN_WTRV-uzfVywQ4R6i3tQ,1946
6
+ evalseal/models.py,sha256=Ct8TYa0VPfy6upI-Yg-9mGs9uYJLNsBcgDMk2dU0ZXI,1895
7
+ evalseal/report.py,sha256=sb1XWsY2cyLhxq5lPC7K7x3WenQsg5YRM9UCJj2CPGE,2039
8
+ evalseal/adapters/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
9
+ evalseal/adapters/dataset.py,sha256=JMaP5dMyGmoSTaDW59IKNhCl8n3RJBu5IKCpq7Z72Mg,950
10
+ evalseal/adapters/recording.py,sha256=0xyaSD8cJliFvkEkAK3Wqjf37iplKUTAOjGYJqkHvZY,2278
11
+ evalseal/adapters/scorer.py,sha256=W6ywL_cCdlZVkajDx_5Yzw05B6ROFtb7RBJAi_vsx1U,2220
12
+ evalseal/adapters/target.py,sha256=zgpSk34KLjAqhmPc-STE5alRGrPDCkZO_B5QzXuly4c,3250
13
+ evalseal-0.1.0.dist-info/METADATA,sha256=oo795y8bdeDFbuSpHiYfmTRH0l35UqP-9h9oJbbdKcQ,5844
14
+ evalseal-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
15
+ evalseal-0.1.0.dist-info/entry_points.txt,sha256=j-GPjykIrYuVnOWt-2QaEaXQEpedrXNeA3vs-YwXHQY,46
16
+ evalseal-0.1.0.dist-info/licenses/LICENSE,sha256=BUR4NJbROoLQlsxDQaqjIk8ThVqyIjc9AfXORDMlv9A,1084
17
+ evalseal-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ evalseal = evalseal.cli:app
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Venkata Manideep Patibandla
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.