evalcore 2.3.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {evalcore-2.3.0 → evalcore-2.4.0}/CHANGELOG.md +30 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/PKG-INFO +1 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/pyproject.toml +1 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/compare.py +64 -6
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/models.py +33 -8
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/report.py +9 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/reporters/html.py +13 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/runner.py +5 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/store.py +32 -10
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_store.py +73 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/uv.lock +1 -1
- {evalcore-2.3.0 → evalcore-2.4.0}/.github/workflows/ci.yml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/.github/workflows/publish.yml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/.gitignore +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/.pre-commit-config.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/LICENSE +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/README.md +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/docs/design.md +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/README.md +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/adapter.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/datasets/support_reply/v1/cases/angry_complaint.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/datasets/support_reply/v1/cases/billing_question.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/datasets/support_reply/v1/cases/double_charge.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/datasets/support_reply/v1/cases/refund_request.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/fixtures/support_reply_judge.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/fixtures/support_reply_pairwise.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/fixtures/support_reply_replay.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/graders.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/run_eval.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/suite.yaml +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/tests/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/examples/quickstart/tests/test_quickstart.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/justfile +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/pyrightconfig.json +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/adapters/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/adapters/base.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/adapters/env.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/adapters/http.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/adapters/replay.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/cli.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/errors.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/base.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/classification.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/deterministic.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/judge.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/graders/numeric.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/loader.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/pairwise.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/py.typed +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/rating.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/refs.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/reporters/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/reporters/base.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/reporters/markdown.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/retry.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/src/evalcore/sweep.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/__init__.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_adapters.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_cli.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_edge_cases.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_judge.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_judge_extra.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_live_clients.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_pairwise_extra.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_rating.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_rating_server.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_reporters.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_retry.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_runner.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_sweep_pairwise.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/tests/test_unit.py +0 -0
- {evalcore-2.3.0 → evalcore-2.4.0}/uv.toml +0 -0
|
@@ -6,6 +6,34 @@ All notable changes to this project are documented here. The format is based on
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [2.4.0] - 2026-08-09
|
|
10
|
+
|
|
11
|
+
A run without a baseline can now say whether it passed.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
- `RunResult.checks`, a `models.ThresholdCheck` the runner fills in: the
|
|
15
|
+
suite's **absolute** guardrails (`min`/`max`) measured against this run.
|
|
16
|
+
Most of a suite's rules are absolute and answerable from one run; only the
|
|
17
|
+
win metric and the relative rules need a baseline. Nothing new to call -
|
|
18
|
+
`compare` is for comparing, and a single run was never a comparison.
|
|
19
|
+
- `compare.check_thresholds(scorecard, thresholds)`, which the runner uses.
|
|
20
|
+
|
|
21
|
+
### Changed
|
|
22
|
+
- **Breaking:** `GuardrailResult.passed` is now `bool | None`. `None` means
|
|
23
|
+
the rule could not be evaluated - a `must_not_increase` /
|
|
24
|
+
`must_not_decrease` with no baseline. Test `passed is False` for a breach;
|
|
25
|
+
`not passed` now also catches the skipped case. The Markdown and HTML
|
|
26
|
+
reporters render it as `skipped` rather than a breach.
|
|
27
|
+
- `store.score_rows` reports `run.checks` on the gate columns when no
|
|
28
|
+
`Comparison` is passed: `gate_verdict` and the per-metric `guardrail` become
|
|
29
|
+
real, while `gate_win` stays `'none'`. That is deliberate - `gate_win` is
|
|
30
|
+
what marks the three `win_*` fields as never computed rather than measured
|
|
31
|
+
at zero, and an ungated run must not claim a comparison happened. A rule
|
|
32
|
+
that could not run reports `guardrail = 'none'` with the reason in
|
|
33
|
+
`guardrail_gap`, not a false `pass`.
|
|
34
|
+
- A `Comparison` supersedes the run's own checks: it evaluates the same rules
|
|
35
|
+
with a baseline available, so it gets the relative ones too.
|
|
36
|
+
|
|
9
37
|
## [2.3.0] - 2026-08-08
|
|
10
38
|
|
|
11
39
|
A run now describes its own scores.
|
|
@@ -220,7 +248,8 @@ by semantic versioning: a breaking change to either means a 2.0.
|
|
|
220
248
|
rating + ranking with judge agreement, Markdown/HTML reporters, JSON +
|
|
221
249
|
column-store outbox, and content-hash provenance.
|
|
222
250
|
|
|
223
|
-
[Unreleased]: https://github.com/scottpmiller/evalcore/compare/2.
|
|
251
|
+
[Unreleased]: https://github.com/scottpmiller/evalcore/compare/2.4.0...HEAD
|
|
252
|
+
[2.4.0]: https://github.com/scottpmiller/evalcore/compare/2.3.0...2.4.0
|
|
224
253
|
[2.3.0]: https://github.com/scottpmiller/evalcore/compare/2.2.0...2.3.0
|
|
225
254
|
[2.2.0]: https://github.com/scottpmiller/evalcore/compare/2.1.0...2.2.0
|
|
226
255
|
[2.1.0]: https://github.com/scottpmiller/evalcore/compare/2.0.0...2.1.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: evalcore
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: A generic, consumer-agnostic evaluation engine for prompt, model, and API outputs.
|
|
5
5
|
Project-URL: Homepage, https://github.com/scottpmiller/evalcore
|
|
6
6
|
Project-URL: Repository, https://github.com/scottpmiller/evalcore
|
|
@@ -23,11 +23,27 @@ def _metric(card: models.Scorecard, name: str) -> float | None:
|
|
|
23
23
|
|
|
24
24
|
|
|
25
25
|
def _check_guardrail(
|
|
26
|
-
rule: dict, baseline: models.Scorecard, candidate: models.Scorecard
|
|
26
|
+
rule: dict, baseline: models.Scorecard | None, candidate: models.Scorecard
|
|
27
27
|
) -> models.GuardrailResult:
|
|
28
|
+
"""Evaluate one guardrail rule.
|
|
29
|
+
|
|
30
|
+
``baseline`` is ``None`` when only one run exists. The absolute parts of
|
|
31
|
+
a rule (``min``/``max``) still evaluate; the relative parts
|
|
32
|
+
(``must_not_increase``/``must_not_decrease``) cannot, and are reported
|
|
33
|
+
rather than silently passed - a rule with nothing left to check comes
|
|
34
|
+
back ``passed=None``.
|
|
35
|
+
"""
|
|
28
36
|
metric = rule['metric']
|
|
29
37
|
cand = _metric(candidate, metric)
|
|
30
|
-
base = _metric(baseline, metric)
|
|
38
|
+
base = _metric(baseline, metric) if baseline is not None else None
|
|
39
|
+
relative = [
|
|
40
|
+
name
|
|
41
|
+
for name in ('must_not_increase', 'must_not_decrease')
|
|
42
|
+
if rule.get(name)
|
|
43
|
+
]
|
|
44
|
+
# A relative rule with no baseline is unevaluable, not satisfied.
|
|
45
|
+
skipped = relative if baseline is None else []
|
|
46
|
+
absolute = 'max' in rule or 'min' in rule
|
|
31
47
|
if cand is None:
|
|
32
48
|
return models.GuardrailResult(
|
|
33
49
|
metric=metric, passed=False, detail='metric absent on candidate'
|
|
@@ -51,12 +67,54 @@ def _check_guardrail(
|
|
|
51
67
|
):
|
|
52
68
|
problems.append(f'decreased {base:.4f} -> {cand:.4f}')
|
|
53
69
|
|
|
70
|
+
note = (
|
|
71
|
+
f'{", ".join(skipped)} not evaluated: needs a baseline'
|
|
72
|
+
if skipped
|
|
73
|
+
else ''
|
|
74
|
+
)
|
|
54
75
|
if problems:
|
|
76
|
+
detail = '; '.join([*problems, note] if note else problems)
|
|
55
77
|
return models.GuardrailResult(
|
|
56
|
-
metric=metric, passed=False, detail=
|
|
78
|
+
metric=metric, passed=False, detail=detail
|
|
57
79
|
)
|
|
58
|
-
|
|
59
|
-
|
|
80
|
+
if skipped and not absolute:
|
|
81
|
+
# Nothing was checked, so this is neither a pass nor a breach.
|
|
82
|
+
return models.GuardrailResult(metric=metric, passed=None, detail=note)
|
|
83
|
+
detail = f'{cand:.4f} ok' + (f'; {note}' if note else '')
|
|
84
|
+
return models.GuardrailResult(metric=metric, passed=True, detail=detail)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def check_thresholds(
|
|
88
|
+
scorecard: models.Scorecard, thresholds: dict | None = None
|
|
89
|
+
) -> models.ThresholdCheck:
|
|
90
|
+
"""Measure one run against its suite's absolute thresholds.
|
|
91
|
+
|
|
92
|
+
A gate needs two runs, but a suite's guardrails are mostly absolute - a
|
|
93
|
+
ceiling on the error rate, a floor on a format check - and those are
|
|
94
|
+
answerable from a single run. The runner calls this so a run carries its
|
|
95
|
+
own verdict; a caller with one run should not have to reach for
|
|
96
|
+
``compare``, which is for comparing.
|
|
97
|
+
|
|
98
|
+
The win metric is deliberately not evaluated: it is a comparison by
|
|
99
|
+
definition, and reporting it here would claim a baseline that does not
|
|
100
|
+
exist. Relative guardrails come back ``passed=None`` for the same reason.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
scorecard: The run to measure.
|
|
104
|
+
thresholds: The suite's ``thresholds`` block.
|
|
105
|
+
|
|
106
|
+
Returns:
|
|
107
|
+
The verdict and one result per configured guardrail. ``verdict`` is
|
|
108
|
+
``'none'`` when the suite declares no guardrails.
|
|
109
|
+
|
|
110
|
+
"""
|
|
111
|
+
rules = (thresholds or {}).get('guardrails', [])
|
|
112
|
+
if not rules:
|
|
113
|
+
return models.ThresholdCheck()
|
|
114
|
+
guardrails = [_check_guardrail(rule, None, scorecard) for rule in rules]
|
|
115
|
+
breached = [g for g in guardrails if g.passed is False]
|
|
116
|
+
return models.ThresholdCheck(
|
|
117
|
+
verdict='fail' if breached else 'pass', guardrails=guardrails
|
|
60
118
|
)
|
|
61
119
|
|
|
62
120
|
|
|
@@ -105,7 +163,7 @@ def compare(
|
|
|
105
163
|
]
|
|
106
164
|
win_metric, win = _evaluate_win(thresholds, baseline, candidate)
|
|
107
165
|
|
|
108
|
-
breached = [g for g in guardrails if
|
|
166
|
+
breached = [g for g in guardrails if g.passed is False]
|
|
109
167
|
on_regression = thresholds.get('on_regression', 'warn')
|
|
110
168
|
if breached:
|
|
111
169
|
verdict = 'fail'
|
|
@@ -184,6 +184,38 @@ class GraderInfo(pydantic.BaseModel):
|
|
|
184
184
|
scale: int = 0
|
|
185
185
|
|
|
186
186
|
|
|
187
|
+
class GuardrailResult(pydantic.BaseModel):
|
|
188
|
+
"""Outcome of one guardrail check against the candidate.
|
|
189
|
+
|
|
190
|
+
``passed`` is tri-state. ``None`` means the rule could not be evaluated -
|
|
191
|
+
a ``must_not_increase`` / ``must_not_decrease`` rule with no baseline to
|
|
192
|
+
compare against. That is neither a pass nor a breach, and conflating it
|
|
193
|
+
with either claims a check that never ran, so callers test
|
|
194
|
+
``passed is False`` for a breach rather than ``not passed``.
|
|
195
|
+
"""
|
|
196
|
+
|
|
197
|
+
metric: str
|
|
198
|
+
passed: bool | None
|
|
199
|
+
detail: str
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
class ThresholdCheck(pydantic.BaseModel):
|
|
203
|
+
"""A run measured against its suite's absolute thresholds.
|
|
204
|
+
|
|
205
|
+
A gate needs two runs, but most of a suite's guardrails are absolute -
|
|
206
|
+
a ceiling on the error rate, a floor on a format check - and those are
|
|
207
|
+
answerable from one run. This is that answer, computed by the runner so
|
|
208
|
+
a single run can say whether it passed without a baseline and without
|
|
209
|
+
the caller re-deriving anything.
|
|
210
|
+
|
|
211
|
+
``verdict`` is ``'none'`` when the suite declares no thresholds. The win
|
|
212
|
+
metric is never evaluated here: it is a comparison by definition.
|
|
213
|
+
"""
|
|
214
|
+
|
|
215
|
+
verdict: typing.Literal['none', 'pass', 'warn', 'fail'] = 'none'
|
|
216
|
+
guardrails: list[GuardrailResult] = pydantic.Field(default_factory=list)
|
|
217
|
+
|
|
218
|
+
|
|
187
219
|
class RunResult(pydantic.BaseModel):
|
|
188
220
|
"""Everything one run produced: the scorecard plus per-sample results.
|
|
189
221
|
|
|
@@ -208,6 +240,7 @@ class RunResult(pydantic.BaseModel):
|
|
|
208
240
|
results: list[CaseResult] = pydantic.Field(default_factory=list)
|
|
209
241
|
aggregate_scores: list[Score] = pydantic.Field(default_factory=list)
|
|
210
242
|
graders: dict[str, GraderInfo] = pydantic.Field(default_factory=dict)
|
|
243
|
+
checks: ThresholdCheck = pydantic.Field(default_factory=ThresholdCheck)
|
|
211
244
|
|
|
212
245
|
|
|
213
246
|
class Rating(pydantic.BaseModel):
|
|
@@ -421,14 +454,6 @@ class MetricDelta(pydantic.BaseModel):
|
|
|
421
454
|
delta: float | None
|
|
422
455
|
|
|
423
456
|
|
|
424
|
-
class GuardrailResult(pydantic.BaseModel):
|
|
425
|
-
"""Outcome of one guardrail check against the candidate."""
|
|
426
|
-
|
|
427
|
-
metric: str
|
|
428
|
-
passed: bool
|
|
429
|
-
detail: str
|
|
430
|
-
|
|
431
|
-
|
|
432
457
|
class Comparison(pydantic.BaseModel):
|
|
433
458
|
"""Candidate-vs-baseline comparison and gate verdict."""
|
|
434
459
|
|
|
@@ -59,7 +59,15 @@ def render_comparison(comparison: models.Comparison) -> str:
|
|
|
59
59
|
if comparison.guardrails:
|
|
60
60
|
lines += ['', '**Guardrails**', '']
|
|
61
61
|
for guard in comparison.guardrails:
|
|
62
|
-
|
|
62
|
+
# Tri-state: None is a rule that could not run (a relative
|
|
63
|
+
# guardrail with no baseline), which is not a breach.
|
|
64
|
+
mark = (
|
|
65
|
+
'skipped'
|
|
66
|
+
if guard.passed is None
|
|
67
|
+
else 'ok'
|
|
68
|
+
if guard.passed
|
|
69
|
+
else 'BREACH'
|
|
70
|
+
)
|
|
63
71
|
lines.append(f'- [{mark}] `{guard.metric}` - {guard.detail}')
|
|
64
72
|
return '\n'.join(lines)
|
|
65
73
|
|
|
@@ -158,6 +158,7 @@ _CSS = (
|
|
|
158
158
|
'.guards li{padding:4px 0;font-size:14px}'
|
|
159
159
|
'.guards li.ok::before{content:"\\2713 ";color:#1f8f5f}'
|
|
160
160
|
'.guards li.breach::before{content:"\\2717 ";color:#c0392b}'
|
|
161
|
+
'.guards li.skipped::before{content:"\\2013 ";color:#8a8a8a}'
|
|
161
162
|
'td.fail{color:#c0392b;font-weight:600}'
|
|
162
163
|
'details{margin:6px 0}summary{cursor:pointer;color:#68707a;font-size:13px}'
|
|
163
164
|
'pre{background:#f0f2f4;border-radius:6px;padding:10px;overflow:auto;'
|
|
@@ -196,6 +197,17 @@ def _esc(value) -> str:
|
|
|
196
197
|
return _html.escape(str(value))
|
|
197
198
|
|
|
198
199
|
|
|
200
|
+
def _guard_class(guard) -> str:
|
|
201
|
+
"""CSS class for one guardrail result.
|
|
202
|
+
|
|
203
|
+
``passed`` is tri-state: ``None`` is a rule that could not run - a
|
|
204
|
+
relative guardrail with no baseline - which must not render as a breach.
|
|
205
|
+
"""
|
|
206
|
+
if guard.passed is None:
|
|
207
|
+
return 'skipped'
|
|
208
|
+
return 'ok' if guard.passed else 'breach'
|
|
209
|
+
|
|
210
|
+
|
|
199
211
|
@base.register('html')
|
|
200
212
|
class HtmlReporter:
|
|
201
213
|
"""Render scorecards and comparisons as self-contained HTML."""
|
|
@@ -247,7 +259,7 @@ class HtmlReporter:
|
|
|
247
259
|
guards = ''
|
|
248
260
|
if comparison.guardrails:
|
|
249
261
|
items = ''.join(
|
|
250
|
-
f'<li class="{
|
|
262
|
+
f'<li class="{_guard_class(g)}">'
|
|
251
263
|
f'<code>{_esc(g.metric)}</code> — {_esc(g.detail)}</li>'
|
|
252
264
|
for g in comparison.guardrails
|
|
253
265
|
)
|
|
@@ -18,7 +18,7 @@ import statistics
|
|
|
18
18
|
import time
|
|
19
19
|
import uuid
|
|
20
20
|
|
|
21
|
-
from evalcore import loader, models, store
|
|
21
|
+
from evalcore import compare, loader, models, store
|
|
22
22
|
from evalcore import retry as retry_mod
|
|
23
23
|
from evalcore.adapters import base as adapters_base
|
|
24
24
|
from evalcore.graders import base as graders_base
|
|
@@ -351,6 +351,10 @@ async def run_suite(
|
|
|
351
351
|
results=results,
|
|
352
352
|
aggregate_scores=agg_scores,
|
|
353
353
|
graders=_grader_info(per_case_graders, aggregate_graders),
|
|
354
|
+
# So a single run can say whether it passed. Absolute rules only -
|
|
355
|
+
# the win metric and the relative guardrails need a baseline, and a
|
|
356
|
+
# gate's Comparison supersedes this when there is one.
|
|
357
|
+
checks=compare.check_thresholds(scorecard, suite.thresholds),
|
|
354
358
|
)
|
|
355
359
|
|
|
356
360
|
|
|
@@ -287,7 +287,9 @@ def _run_key(scorecard: models.Scorecard) -> dict:
|
|
|
287
287
|
|
|
288
288
|
|
|
289
289
|
def _gate(
|
|
290
|
-
comparison: models.Comparison | None,
|
|
290
|
+
comparison: models.Comparison | None,
|
|
291
|
+
baseline_run_id: str | None,
|
|
292
|
+
checks: models.ThresholdCheck | None = None,
|
|
291
293
|
) -> dict:
|
|
292
294
|
"""The run-grain gate columns, repeated on every row of a gated run.
|
|
293
295
|
|
|
@@ -297,15 +299,26 @@ def _gate(
|
|
|
297
299
|
0 delta genuinely means no change.
|
|
298
300
|
"""
|
|
299
301
|
if comparison is None:
|
|
302
|
+
# Without a Comparison the run may still have been measured against
|
|
303
|
+
# its suite's absolute thresholds. That is a real verdict, so it is
|
|
304
|
+
# reported - but `gate_win` stays 'none', which is what marks the
|
|
305
|
+
# three win_* fields as never computed rather than measured at zero.
|
|
306
|
+
checks = checks or models.ThresholdCheck()
|
|
307
|
+
breached = [g for g in checks.guardrails if g.passed is False]
|
|
300
308
|
return {
|
|
301
|
-
'gate_verdict':
|
|
309
|
+
'gate_verdict': checks.verdict,
|
|
302
310
|
'gate_win': 'none',
|
|
303
311
|
'baseline_run_id': _NO_UUID,
|
|
304
312
|
'baseline_variant': '',
|
|
305
313
|
'win_baseline': None,
|
|
306
314
|
'win_candidate': None,
|
|
307
315
|
'win_delta': None,
|
|
308
|
-
'gate_summary':
|
|
316
|
+
'gate_summary': (
|
|
317
|
+
'guardrail breach: '
|
|
318
|
+
+ '; '.join(f'{g.metric} ({g.detail})' for g in breached)
|
|
319
|
+
if breached
|
|
320
|
+
else ''
|
|
321
|
+
),
|
|
309
322
|
}
|
|
310
323
|
delta = next(
|
|
311
324
|
(d for d in comparison.deltas if d.metric == comparison.win_metric),
|
|
@@ -332,8 +345,14 @@ def _metric_gate(
|
|
|
332
345
|
rail = rails.get(metric)
|
|
333
346
|
return {
|
|
334
347
|
'win': bool(comparison and comparison.win_metric == metric),
|
|
348
|
+
# rail.passed is tri-state: None means the rule needed a baseline
|
|
349
|
+
# and could not run, which reads as 'none' rather than a false pass.
|
|
335
350
|
'guardrail': (
|
|
336
|
-
'none'
|
|
351
|
+
'none'
|
|
352
|
+
if rail is None or rail.passed is None
|
|
353
|
+
else 'pass'
|
|
354
|
+
if rail.passed
|
|
355
|
+
else 'fail'
|
|
337
356
|
),
|
|
338
357
|
'guardrail_gap': rail.detail if rail else '',
|
|
339
358
|
}
|
|
@@ -410,12 +429,15 @@ def score_rows(
|
|
|
410
429
|
:func:`grader_lookups` builds them from the suite.
|
|
411
430
|
"""
|
|
412
431
|
key = _run_key(run.scorecard)
|
|
413
|
-
gate = _gate(comparison, baseline_run_id)
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
432
|
+
gate = _gate(comparison, baseline_run_id, run.checks)
|
|
433
|
+
# A Comparison supersedes the run's own checks: it evaluates the same
|
|
434
|
+
# rules with a baseline available, so it gets the relative ones too.
|
|
435
|
+
rails = {
|
|
436
|
+
rail.metric: rail
|
|
437
|
+
for rail in (
|
|
438
|
+
comparison.guardrails if comparison else run.checks.guardrails
|
|
439
|
+
)
|
|
440
|
+
}
|
|
419
441
|
# The run describes its own graders as of 2.3.0. The lookups remain for
|
|
420
442
|
# runs written before that, which carry an empty map - a `run.json` on
|
|
421
443
|
# disk outlives the release that wrote it. An explicit lookup still wins,
|
|
@@ -3,9 +3,10 @@
|
|
|
3
3
|
import json
|
|
4
4
|
import pathlib
|
|
5
5
|
import tempfile
|
|
6
|
+
import typing
|
|
6
7
|
import unittest
|
|
7
8
|
|
|
8
|
-
from evalcore import errors, graders, models, store
|
|
9
|
+
from evalcore import compare, errors, graders, models, store
|
|
9
10
|
|
|
10
11
|
|
|
11
12
|
def _run(with_failure: bool = False):
|
|
@@ -394,6 +395,77 @@ class SelfDescribingRunTests(unittest.TestCase):
|
|
|
394
395
|
self.assertEqual(types['j'], 'llm_as_judge')
|
|
395
396
|
|
|
396
397
|
|
|
398
|
+
class UngatedVerdictTests(unittest.TestCase):
|
|
399
|
+
"""A run with no baseline still reports its absolute thresholds."""
|
|
400
|
+
|
|
401
|
+
THRESHOLDS: typing.ClassVar = {
|
|
402
|
+
'guardrails': [
|
|
403
|
+
# absolute, breaches: f1 is 0.9
|
|
404
|
+
{'metric': 'f1', 'min': 1.0},
|
|
405
|
+
# absolute, passes
|
|
406
|
+
{'metric': 'passed_check', 'max': 2.0},
|
|
407
|
+
# relative only: unevaluable without a baseline
|
|
408
|
+
{'metric': 'quality.overall', 'must_not_decrease': True},
|
|
409
|
+
]
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
def _rows(self):
|
|
413
|
+
run = _run()
|
|
414
|
+
# A real value, so the relative rule is unevaluable rather than
|
|
415
|
+
# absent - an absent metric is a breach and outranks the skip.
|
|
416
|
+
run.scorecard.metrics['quality.overall'] = models.MetricValue(
|
|
417
|
+
metric='quality.overall', value=0.8, kind='mean', n=2
|
|
418
|
+
)
|
|
419
|
+
run.checks = compare.check_thresholds(run.scorecard, self.THRESHOLDS)
|
|
420
|
+
return run, store.score_rows(run)
|
|
421
|
+
|
|
422
|
+
def test_the_verdict_is_real(self):
|
|
423
|
+
run, rows = self._rows()
|
|
424
|
+
self.assertEqual(run.checks.verdict, 'fail') # f1 0.9 < min 1.0
|
|
425
|
+
self.assertEqual(rows[0]['gate_verdict'], 'fail')
|
|
426
|
+
|
|
427
|
+
def test_the_win_columns_stay_uncomputed(self):
|
|
428
|
+
"""gate_win is what marks the three zeros as never measured."""
|
|
429
|
+
_, rows = self._rows()
|
|
430
|
+
self.assertEqual(rows[0]['gate_win'], 'none')
|
|
431
|
+
self.assertIsNone(rows[0]['win_baseline'])
|
|
432
|
+
self.assertIsNone(rows[0]['win_delta'])
|
|
433
|
+
self.assertEqual(rows[0]['baseline_run_id'], store._NO_UUID)
|
|
434
|
+
|
|
435
|
+
def test_an_absolute_breach_lands_on_its_metric(self):
|
|
436
|
+
_, rows = self._rows()
|
|
437
|
+
rails = {r['metric']: r['guardrail'] for r in rows}
|
|
438
|
+
self.assertEqual(rails['f1'], 'fail')
|
|
439
|
+
|
|
440
|
+
def test_a_relative_rule_is_not_a_pass(self):
|
|
441
|
+
"""It could not run, so it reads 'none' with the reason in the gap."""
|
|
442
|
+
_, rows = self._rows()
|
|
443
|
+
row = next(r for r in rows if r['metric'] == 'quality.overall')
|
|
444
|
+
self.assertEqual(row['guardrail'], 'none')
|
|
445
|
+
self.assertIn('needs a baseline', row['guardrail_gap'])
|
|
446
|
+
|
|
447
|
+
def test_a_comparison_supersedes_the_runs_own_checks(self):
|
|
448
|
+
run, _ = self._rows()
|
|
449
|
+
rows = store.score_rows(run, _comparison(), baseline_run_id='B')
|
|
450
|
+
self.assertEqual(rows[0]['gate_verdict'], 'fail')
|
|
451
|
+
self.assertNotEqual(rows[0]['gate_win'], 'none')
|
|
452
|
+
self.assertIsNotNone(rows[0]['win_delta'])
|
|
453
|
+
|
|
454
|
+
def test_an_absent_metric_outranks_the_skip(self):
|
|
455
|
+
"""Nothing to measure is a breach, not an unevaluated rule."""
|
|
456
|
+
run = _run()
|
|
457
|
+
check = compare.check_thresholds(
|
|
458
|
+
run.scorecard, {'guardrails': [{'metric': 'nope', 'min': 1.0}]}
|
|
459
|
+
)
|
|
460
|
+
self.assertIs(check.guardrails[0].passed, False)
|
|
461
|
+
self.assertIn('absent', check.guardrails[0].detail)
|
|
462
|
+
|
|
463
|
+
def test_no_thresholds_means_no_verdict(self):
|
|
464
|
+
run = _run()
|
|
465
|
+
self.assertEqual(run.checks.verdict, 'none')
|
|
466
|
+
self.assertEqual(store.score_rows(run)[0]['gate_verdict'], 'none')
|
|
467
|
+
|
|
468
|
+
|
|
397
469
|
class ScoreExporterProtocolTests(unittest.TestCase):
|
|
398
470
|
"""The seam another package implements to publish to a real store."""
|
|
399
471
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|