evalcore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,91 @@
1
+ """Built-in numeric (per-case) grader.
2
+
3
+ Promotes numeric ``Output`` fields onto the scorecard as metrics, so values an
4
+ adapter merely *extracted* (a cost, a tool-call error rate, a token count)
5
+ become first-class metrics the runner averages and ``compare``/guardrails can
6
+ gate on. Without this, an adapter's ``fields`` never reach a scorecard - only
7
+ ``Score``s do.
8
+
9
+ Each configured field emits one per-case ``Score`` whose ``value`` is the field
10
+ coerced to float. Give a field a ``min``/``max`` bound and the score also
11
+ carries ``passed`` (in-range), turning it into a pass-rate; leave bounds off
12
+ and it's a pure measurement (``passed`` stays ``None``). Non-numeric or absent
13
+ fields degrade to ``value=None`` rather than raising.
14
+ """
15
+
16
+ from evalkit import models, refs
17
+ from evalkit.graders import base
18
+
19
+
20
+ def _context(case: models.Case, output: models.Output) -> dict:
21
+ return {
22
+ 'input': case.input,
23
+ 'expected': case.expected or {},
24
+ 'output': output.fields,
25
+ 'case': case.model_dump(),
26
+ }
27
+
28
+
29
+ class _Field:
30
+ """One resolved field spec: a ``$ref`` selector, a metric name, bounds."""
31
+
32
+ def __init__(self, spec: str | dict):
33
+ if isinstance(spec, str):
34
+ spec = {'ref': spec}
35
+ self.ref = spec['ref']
36
+ self.minimum = spec.get('min')
37
+ self.maximum = spec.get('max')
38
+ # Metric name defaults to the ref's leaf (``output.cost`` -> ``cost``).
39
+ self.metric = spec.get('name') or self.ref.rsplit('.', 1)[-1]
40
+
41
+
42
+ def _as_float(value) -> float | None:
43
+ if isinstance(value, bool) or value is None:
44
+ return None
45
+ try:
46
+ return float(value)
47
+ except TypeError, ValueError:
48
+ return None
49
+
50
+
51
+ @base.register('numeric')
52
+ class Numeric:
53
+ """Surface numeric output fields as metrics, optionally range-checked."""
54
+
55
+ def __init__(self, fields: list[str | dict], name: str = 'numeric'):
56
+ self.name = name
57
+ self.fields = [_Field(spec) for spec in fields]
58
+
59
+ def grade(
60
+ self, case: models.Case, output: models.Output
61
+ ) -> list[models.Score]:
62
+ context = _context(case, output)
63
+ scores: list[models.Score] = []
64
+ for field in self.fields:
65
+ value = _as_float(refs.resolve_ref(context, field.ref))
66
+ passed, detail = _check(value, field)
67
+ scores.append(
68
+ models.Score(
69
+ grader=self.name,
70
+ metric=field.metric,
71
+ value=value,
72
+ passed=passed,
73
+ detail=detail,
74
+ case_id=case.id,
75
+ kind='per_case',
76
+ )
77
+ )
78
+ return scores
79
+
80
+
81
+ def _check(value: float | None, field: _Field) -> tuple[bool | None, str]:
82
+ """Range-check ``value`` against a field's bounds (``None`` if unbound)."""
83
+ bounded = field.minimum is not None or field.maximum is not None
84
+ if value is None:
85
+ return (False if bounded else None), 'absent/non-numeric'
86
+ if not bounded:
87
+ return None, f'{value:g}'
88
+ ok = (field.minimum is None or value >= field.minimum) and (
89
+ field.maximum is None or value <= field.maximum
90
+ )
91
+ return ok, f'{value:g}'
evalkit/loader.py ADDED
@@ -0,0 +1,129 @@
1
+ """Loading of suite configs and datasets.
2
+
3
+ Data files are YAML when PyYAML is available (nice for multi-line content
4
+ payloads) and fall back to JSON otherwise, so the engine has no hard non-stdlib
5
+ data dependency.
6
+ """
7
+
8
+ import hashlib
9
+ import json
10
+ import pathlib
11
+
12
+ import pydantic
13
+
14
+ from evalkit import models
15
+ from evalkit.retry import RetryConfig # re-exported: SuiteConfig.retry type
16
+
17
+
18
+ def content_hash(data) -> str:
19
+ """Stable short digest of parsed content (canonical JSON, sha256).
20
+
21
+ Versions content independently of any VCS: two loads hash equal iff
22
+ the parsed data is equal, regardless of file formatting or location.
23
+ """
24
+ text = json.dumps(data, sort_keys=True, separators=(',', ':'))
25
+ return hashlib.sha256(text.encode('utf-8')).hexdigest()[:12]
26
+
27
+
28
+ def dataset_hash(cases: list[models.Case]) -> str:
29
+ """Content digest of a loaded case list (order-independent)."""
30
+ dumped = sorted(
31
+ (case.model_dump(mode='json') for case in cases),
32
+ key=lambda data: str(data.get('id')),
33
+ )
34
+ return content_hash(dumped)
35
+
36
+
37
+ def load_data_file(path: str | pathlib.Path) -> dict:
38
+ """Load a YAML or JSON mapping from ``path``."""
39
+ text = pathlib.Path(path).read_text(encoding='utf-8')
40
+ try:
41
+ import yaml
42
+
43
+ return yaml.safe_load(text) or {}
44
+ except ImportError:
45
+ return json.loads(text) if text.strip() else {}
46
+
47
+
48
+ class SuiteConfig(pydantic.BaseModel):
49
+ """Parsed ``evals/suites/<name>.yaml`` for one consumer suite."""
50
+
51
+ project: str
52
+ suite: str
53
+ dataset: str
54
+ dataset_version: str = 'v1'
55
+ mode_default: str = 'http'
56
+ replay_fixtures: str | None = None
57
+ adapter: dict
58
+ graders: list[dict] = pydantic.Field(default_factory=list)
59
+ variants: dict[str, dict] = pydantic.Field(default_factory=dict)
60
+ n_samples: int = 1
61
+ # Max concurrent (case, sample) invocations. Above 1, the adapter and
62
+ # per-case graders must tolerate concurrent calls.
63
+ concurrency: int = 1
64
+ # Transient-failure retry policy for adapter calls (default: no retry).
65
+ retry: RetryConfig = pydantic.Field(default_factory=RetryConfig)
66
+ thresholds: dict = pydantic.Field(default_factory=dict)
67
+ # Computed by load_suite (content digest of the raw file); not for
68
+ # authors to set in the suite file.
69
+ suite_hash: str | None = None
70
+
71
+
72
+ def load_suite(path: str | pathlib.Path) -> SuiteConfig:
73
+ """Load and validate a suite config file.
74
+
75
+ ``dataset`` and ``replay_fixtures`` are resolved relative to the suite
76
+ file's own directory (unless already absolute), so a suite is portable and
77
+ runnable from any working directory.
78
+ """
79
+ path = pathlib.Path(path)
80
+ raw = load_data_file(path)
81
+ config = SuiteConfig.model_validate(raw)
82
+ # Hash the raw content BEFORE resolving paths so the digest is stable
83
+ # across checkout locations.
84
+ config.suite_hash = content_hash(raw)
85
+ base = path.parent
86
+
87
+ def _resolve(value: str) -> str:
88
+ candidate = pathlib.Path(value)
89
+ return value if candidate.is_absolute() else str(base / candidate)
90
+
91
+ config.dataset = _resolve(config.dataset)
92
+ if config.replay_fixtures:
93
+ config.replay_fixtures = _resolve(config.replay_fixtures)
94
+ # Grader specs may carry their own suite-relative fixtures (e.g. the LLM
95
+ # judge's recorded judgments under ``replay_path``) - at the top level
96
+ # for a single judge, or per judge in a ``judges`` panel list.
97
+ for spec in config.graders:
98
+ if isinstance(spec.get('replay_path'), str):
99
+ spec['replay_path'] = _resolve(spec['replay_path'])
100
+ for judge in spec.get('judges', []) or []:
101
+ if isinstance(judge.get('replay_path'), str):
102
+ judge['replay_path'] = _resolve(judge['replay_path'])
103
+ # The pairwise judge's recorded judgments are suite-relative too.
104
+ pairwise = config.thresholds.get('pairwise')
105
+ if isinstance(pairwise, dict) and isinstance(
106
+ pairwise.get('replay_path'), str
107
+ ):
108
+ pairwise['replay_path'] = _resolve(pairwise['replay_path'])
109
+ return config
110
+
111
+
112
+ def load_cases(dataset_dir: str | pathlib.Path) -> list[models.Case]:
113
+ """Load every ``cases/*.yaml|json`` file under a dataset directory."""
114
+ cases_dir = pathlib.Path(dataset_dir) / 'cases'
115
+ if not cases_dir.is_dir():
116
+ raise FileNotFoundError(f'no cases directory at {cases_dir}')
117
+ cases: list[models.Case] = []
118
+ paths = sorted(
119
+ p
120
+ for p in cases_dir.iterdir()
121
+ if p.suffix in ('.yaml', '.yml', '.json')
122
+ )
123
+ for path in paths:
124
+ data = load_data_file(path)
125
+ data.setdefault('id', path.stem)
126
+ cases.append(models.Case.model_validate(data))
127
+ if not cases:
128
+ raise FileNotFoundError(f'no case files found in {cases_dir}')
129
+ return cases
evalkit/models.py ADDED
@@ -0,0 +1,390 @@
1
+ """Core data model for the generic eval engine.
2
+
3
+ None of these types reference any particular consumer. ``Case.input``,
4
+ ``Case.expected``, and ``Variant.knobs`` are opaque blobs interpreted only by
5
+ adapters and graders - that opacity is what keeps the engine generic.
6
+ """
7
+
8
+ import typing
9
+
10
+ import pydantic
11
+
12
+ Json = typing.Any
13
+
14
+
15
+ class Case(pydantic.BaseModel):
16
+ """One eval input. ``input``/``expected`` are consumer-defined blobs."""
17
+
18
+ id: str
19
+ input: dict[str, Json] = pydantic.Field(default_factory=dict)
20
+ expected: dict[str, Json] | None = None
21
+ labels: dict[str, Json] = pydantic.Field(default_factory=dict)
22
+
23
+
24
+ class Variant(pydantic.BaseModel):
25
+ """A configuration under test (e.g. baseline vs candidate).
26
+
27
+ ``knobs`` is opaque to the engine; the adapter understands it (commonly
28
+ ``{model, prompt_version, env}``).
29
+ """
30
+
31
+ name: str
32
+ knobs: dict[str, Json] = pydantic.Field(default_factory=dict)
33
+
34
+
35
+ class Output(pydantic.BaseModel):
36
+ """The system-under-test's response for one case, normalized.
37
+
38
+ ``fields`` holds the values the adapter extracted (graders read these via
39
+ ``output.<field>`` references). ``error`` is set when the invocation
40
+ failed; graders/aggregators treat errored outputs explicitly rather than
41
+ silently scoring them.
42
+ """
43
+
44
+ fields: dict[str, Json] = pydantic.Field(default_factory=dict)
45
+ raw: Json = None
46
+ error: str | None = None
47
+ #: Set by the adapter alongside ``error`` when the failure is *transient*
48
+ #: (a 429, a 5xx, a network timeout) and worth retrying; the runner's
49
+ #: retry loop acts on this. A terminal error (bad request, non-JSON body)
50
+ #: leaves it False so the run doesn't waste attempts on it.
51
+ retryable: bool = False
52
+ latency_ms: float | None = None
53
+ tokens: dict[str, int] | None = None
54
+ cost: float | None = None
55
+ #: Named files the adapter saved for this invocation (screenshots,
56
+ #: rendered HTML, transcripts): name -> filesystem path. Persisted
57
+ #: with the run so downstream review/rating tools can load them.
58
+ artifacts: dict[str, str] = pydantic.Field(default_factory=dict)
59
+
60
+
61
+ class JudgeDetail(pydantic.BaseModel):
62
+ """One judge's full verdict on a case, retained for review.
63
+
64
+ The runner folds ``points``/``overall`` into the aggregate means; this
65
+ keeps the per-judge breakdown the aggregate hides - the raw 1..scale
66
+ point each dimension got and the free-text ``rationale`` - so a low
67
+ score can be read back to *why* without re-running the judge.
68
+ """
69
+
70
+ key: str
71
+ version: str | None = None
72
+ rationale: str | None = None
73
+ #: raw per-dimension points as the judge returned them (1..scale), not
74
+ #: normalized; ``None`` for a dimension the judge did not score.
75
+ points: dict[str, float | None] = pydantic.Field(default_factory=dict)
76
+ #: this judge's own normalized (0..1) overall, so a generous judge shows.
77
+ overall: float | None = None
78
+
79
+
80
+ class Score(pydantic.BaseModel):
81
+ """A single metric emitted by a grader for one case (or aggregate).
82
+
83
+ ``kind='per_case'`` scores are averaged across cases by the runner;
84
+ ``kind='aggregate'`` scores are computed once over the whole run (e.g.
85
+ precision/recall) and stored as-is.
86
+ """
87
+
88
+ grader: str
89
+ metric: str
90
+ value: float | None = None
91
+ passed: bool | None = None
92
+ detail: str | None = None
93
+ case_id: str | None = None
94
+ kind: typing.Literal['per_case', 'aggregate'] = 'per_case'
95
+ #: Populated only on an LLM-judge ``<name>.overall`` score: the per-judge
96
+ #: breakdown (rationale + raw points) behind the aggregated value, so the
97
+ #: store and reporters can surface it. Empty for every other grader.
98
+ judges: list[JudgeDetail] = pydantic.Field(default_factory=list)
99
+
100
+
101
+ class CaseResult(pydantic.BaseModel):
102
+ """The output + per-case scores for one (case, sample)."""
103
+
104
+ case: Case
105
+ variant_name: str
106
+ sample_idx: int
107
+ output: Output
108
+ scores: list[Score] = pydantic.Field(default_factory=list)
109
+
110
+
111
+ class MetricValue(pydantic.BaseModel):
112
+ """An aggregated metric on a scorecard.
113
+
114
+ ``stdev`` is populated for ``mean`` metrics with 2+ observations
115
+ (e.g. ``n_samples > 1``) so repeat runs carry their spread, not just
116
+ the average.
117
+ """
118
+
119
+ metric: str
120
+ value: float | None
121
+ kind: typing.Literal['mean', 'aggregate']
122
+ n: int
123
+ stdev: float | None = None
124
+
125
+
126
+ class Scorecard(pydantic.BaseModel):
127
+ """Aggregated result of running one suite x one variant over a dataset.
128
+
129
+ The key tuple (project, suite, variant, dataset_version, model_id,
130
+ prompt_version, judge_version, revision) makes a scorecard reproducible
131
+ and is exactly the multi-tenant key used by the results store.
132
+
133
+ ``revision`` is an opaque consumer-supplied provenance id (a git SHA,
134
+ image digest, package version, release label, ...) - the engine never
135
+ interprets it. ``suite_hash``/``dataset_hash`` are engine-computed
136
+ content digests of the loaded suite config and cases: declared versions
137
+ state intent, the hashes prove the content actually matched.
138
+ """
139
+
140
+ run_id: str | None = None
141
+ project: str
142
+ suite: str
143
+ variant: Variant
144
+ dataset_version: str
145
+ model_id: str | None = None
146
+ prompt_version: str | None = None
147
+ judge_version: str | None = None
148
+ revision: str | None = None
149
+ suite_hash: str | None = None
150
+ dataset_hash: str | None = None
151
+ mode: str = 'http'
152
+ n_samples: int = 1
153
+ n_cases: int = 0
154
+ created_at: str | None = None
155
+ metrics: dict[str, MetricValue] = pydantic.Field(default_factory=dict)
156
+
157
+
158
+ class RunResult(pydantic.BaseModel):
159
+ """Everything one run produced: the scorecard plus per-sample results.
160
+
161
+ ``results`` is the ground truth the scorecard was aggregated from -
162
+ one entry per (case, sample) with the full output, artifacts, and
163
+ per-case scores. Persisting it is what makes transcript review,
164
+ human rating, and judge-agreement analysis possible after the fact.
165
+ """
166
+
167
+ run_id: str
168
+ scorecard: Scorecard
169
+ results: list[CaseResult] = pydantic.Field(default_factory=list)
170
+
171
+
172
+ class Rating(pydantic.BaseModel):
173
+ """One human's blind rating of a single (case, sample) output.
174
+
175
+ The open interchange format: emitted by the ``rate`` web app and
176
+ ingestible from any external tool. ``scores`` maps rubric dimension ->
177
+ an integer on the same 1..scale the judge used, so human and judge are
178
+ directly comparable.
179
+ """
180
+
181
+ run_id: str
182
+ case_id: str
183
+ sample_idx: int = 0
184
+ rater: str
185
+ scores: dict[str, int] = pydantic.Field(default_factory=dict)
186
+ rated_at: str | None = None
187
+
188
+
189
+ class Preference(pydantic.BaseModel):
190
+ """One human's blind side-by-side preference between two variants' outputs
191
+ for a single (case, sample) - the A-vs-B analog of :class:`Rating`.
192
+
193
+ The open interchange format for the ``rank`` web app (one JSON object per
194
+ line). ``winner`` is the overall pick and ``dims`` the per-dimension picks,
195
+ each ``'a'`` / ``'b'`` / ``'tie'`` in *variant* terms: the app
196
+ counterbalances left/right per rater and un-blinds server-side, so a
197
+ stored ``'a'`` always means ``variant_a`` won regardless of which side it
198
+ was shown on.
199
+ """
200
+
201
+ case_id: str
202
+ sample_idx: int = 0
203
+ variant_a: str
204
+ variant_b: str
205
+ rater: str
206
+ winner: typing.Literal['a', 'b', 'tie'] = 'tie'
207
+ dims: dict[str, typing.Literal['a', 'b', 'tie']] = pydantic.Field(
208
+ default_factory=dict
209
+ )
210
+ rated_at: str | None = None
211
+
212
+
213
+ class DimensionAgreement(pydantic.BaseModel):
214
+ """Judge-vs-human agreement on one rubric dimension."""
215
+
216
+ dimension: str
217
+ n: int
218
+ human_mean: float | None = None
219
+ judge_mean: float | None = None
220
+ mae: float | None = None
221
+ correlation: float | None = None
222
+
223
+
224
+ class AgreementResult(pydantic.BaseModel):
225
+ """How well the judge tracks human raters over a run.
226
+
227
+ The calibration gate: a judge whose scores don't agree with human
228
+ ratings isn't trustworthy as a win metric yet. All values compare the
229
+ per-(case,sample) human mean against the judge's normalized 0..1 score.
230
+ """
231
+
232
+ judge_name: str
233
+ scale: int
234
+ n_ratings: int
235
+ n_raters: int
236
+ dimensions: list[DimensionAgreement] = pydantic.Field(default_factory=list)
237
+ overall_mae: float | None = None
238
+ overall_correlation: float | None = None
239
+
240
+
241
+ class SweepEntry(pydantic.BaseModel):
242
+ """One variant's standing in a sweep, ranked by the win metric."""
243
+
244
+ variant: str
245
+ win_value: float | None
246
+ rank: int
247
+
248
+
249
+ class SweepResult(pydantic.BaseModel):
250
+ """N-way comparison of a suite's variants (a model x prompt matrix).
251
+
252
+ ``entries`` ranks the variants by the configured win metric;
253
+ ``matrix`` is metric -> {variant: value} for the full leaderboard.
254
+ """
255
+
256
+ project: str
257
+ suite: str
258
+ win_metric: str | None = None
259
+ win_higher_is_better: bool = True
260
+ entries: list[SweepEntry] = pydantic.Field(default_factory=list)
261
+ matrix: dict[str, dict[str, float | None]] = pydantic.Field(
262
+ default_factory=dict
263
+ )
264
+
265
+
266
+ class PairwiseOutcome(pydantic.BaseModel):
267
+ """The head-to-head result for one case (counterbalanced for order)."""
268
+
269
+ case_id: str
270
+ sample_idx: int = 0
271
+ winner: typing.Literal['a', 'b', 'tie'] = 'tie'
272
+ detail: str | None = None
273
+
274
+
275
+ class PairwiseResult(pydantic.BaseModel):
276
+ """A-vs-B win-rate from a judge comparing two variants head-to-head.
277
+
278
+ The headline subjective-regression signal: for each case the judge is
279
+ shown both variants' outputs and picks a winner (order counterbalanced,
280
+ so a position-biased flip becomes a tie). ``win_rate_a`` counts ties as
281
+ half, the standard convention.
282
+ """
283
+
284
+ project: str
285
+ suite: str
286
+ variant_a: str
287
+ variant_b: str
288
+ judge_name: str = 'pairwise'
289
+ judge_version: str = 'v1'
290
+ n: int = 0
291
+ a_wins: int = 0
292
+ b_wins: int = 0
293
+ ties: int = 0
294
+ win_rate_a: float | None = None
295
+ outcomes: list[PairwiseOutcome] = pydantic.Field(default_factory=list)
296
+
297
+
298
+ class DimensionPreference(pydantic.BaseModel):
299
+ """Human A-vs-B win-rate on one rubric dimension (ties count as half)."""
300
+
301
+ dimension: str
302
+ n: int = 0
303
+ a_wins: int = 0
304
+ b_wins: int = 0
305
+ ties: int = 0
306
+ win_rate_a: float | None = None
307
+
308
+
309
+ class PreferenceResult(pydantic.BaseModel):
310
+ """Human side-by-side win-rate of A vs B - the human analog of
311
+ :class:`PairwiseResult`.
312
+
313
+ Reports an overall win-rate plus a per-dimension breakdown; ``win_rate_a``
314
+ counts ties as half, the standard convention the pairwise judge uses.
315
+ """
316
+
317
+ project: str
318
+ suite: str
319
+ variant_a: str
320
+ variant_b: str
321
+ n_raters: int = 0
322
+ n: int = 0
323
+ a_wins: int = 0
324
+ b_wins: int = 0
325
+ ties: int = 0
326
+ win_rate_a: float | None = None
327
+ dimensions: list[DimensionPreference] = pydantic.Field(
328
+ default_factory=list
329
+ )
330
+
331
+
332
+ class PairwiseAgreementCase(pydantic.BaseModel):
333
+ """Human-panel vs LLM-judge winner for one case."""
334
+
335
+ case_id: str
336
+ sample_idx: int = 0
337
+ human: typing.Literal['a', 'b', 'tie'] = 'tie'
338
+ judge: typing.Literal['a', 'b', 'tie'] = 'tie'
339
+ agree: bool = False
340
+
341
+
342
+ class PairwiseAgreement(pydantic.BaseModel):
343
+ """How often the human panel and the LLM pairwise judge pick the same
344
+ per-case winner - the head-to-head calibration gate (the A-vs-B analog of
345
+ :class:`AgreementResult`).
346
+ """
347
+
348
+ variant_a: str
349
+ variant_b: str
350
+ judge_name: str = 'pairwise'
351
+ n: int = 0
352
+ agree: int = 0
353
+ human_win_rate_a: float | None = None
354
+ judge_win_rate_a: float | None = None
355
+ agreement_rate: float | None = None
356
+ outcomes: list[PairwiseAgreementCase] = pydantic.Field(
357
+ default_factory=list
358
+ )
359
+
360
+
361
+ class MetricDelta(pydantic.BaseModel):
362
+ """Per-metric baseline->candidate comparison."""
363
+
364
+ metric: str
365
+ baseline: float | None
366
+ candidate: float | None
367
+ delta: float | None
368
+
369
+
370
+ class GuardrailResult(pydantic.BaseModel):
371
+ """Outcome of one guardrail check against the candidate."""
372
+
373
+ metric: str
374
+ passed: bool
375
+ detail: str
376
+
377
+
378
+ class Comparison(pydantic.BaseModel):
379
+ """Candidate-vs-baseline comparison and gate verdict."""
380
+
381
+ project: str
382
+ suite: str
383
+ baseline_variant: str
384
+ candidate_variant: str
385
+ win_metric: str | None = None
386
+ win: typing.Literal['improved', 'regressed', 'neutral'] = 'neutral'
387
+ verdict: typing.Literal['pass', 'warn', 'fail'] = 'pass'
388
+ deltas: list[MetricDelta] = pydantic.Field(default_factory=list)
389
+ guardrails: list[GuardrailResult] = pydantic.Field(default_factory=list)
390
+ summary: str = ''