evalcore 1.0.0__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-2.1.0/CHANGELOG.md +183 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/PKG-INFO +15 -14
- {evalcore-1.0.0 → evalcore-2.1.0}/README.md +14 -13
- {evalcore-1.0.0 → evalcore-2.1.0}/docs/design.md +26 -13
- {evalcore-1.0.0 → evalcore-2.1.0}/justfile +3 -3
- {evalcore-1.0.0 → evalcore-2.1.0}/pyproject.toml +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/__init__.py +10 -1
- evalcore-2.1.0/src/evalcore/graders/base.py +145 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/classification.py +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/deterministic.py +4 -4
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/judge.py +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/numeric.py +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/models.py +27 -7
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/pairwise.py +62 -22
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/rating.py +56 -46
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/base.py +5 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/runner.py +47 -23
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/store.py +89 -55
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_cli.py +16 -6
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_edge_cases.py +2 -2
- evalcore-2.1.0/tests/test_pairwise_extra.py +178 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_rating.py +26 -41
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_rating_server.py +2 -2
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_reporters.py +3 -3
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_runner.py +127 -19
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_store.py +87 -12
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_sweep_pairwise.py +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_unit.py +1 -1
- {evalcore-1.0.0 → evalcore-2.1.0}/uv.lock +1 -1
- evalcore-1.0.0/CHANGELOG.md +0 -96
- evalcore-1.0.0/docs/clickhouse-schema.html +0 -1112
- evalcore-1.0.0/docs/clickhouse-schema.sql +0 -412
- evalcore-1.0.0/src/evalcore/graders/base.py +0 -78
- evalcore-1.0.0/tests/test_pairwise_extra.py +0 -68
- {evalcore-1.0.0 → evalcore-2.1.0}/.github/workflows/ci.yml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/.github/workflows/publish.yml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/.gitignore +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/.pre-commit-config.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/LICENSE +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/README.md +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/adapter.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/angry_complaint.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/billing_question.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/double_charge.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/refund_request.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_judge.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_pairwise.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_replay.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/graders.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/run_eval.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/suite.yaml +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/tests/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/tests/test_quickstart.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/pyrightconfig.json +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/base.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/env.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/http.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/replay.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/cli.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/compare.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/errors.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/loader.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/py.typed +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/refs.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/report.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/html.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/markdown.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/retry.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/sweep.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/__init__.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_adapters.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_judge.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_judge_extra.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_live_clients.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_retry.py +0 -0
- {evalcore-1.0.0 → evalcore-2.1.0}/uv.toml +0 -0
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format is based on
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/), and the project follows
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [2.1.0] - 2026-08-07
|
|
10
|
+
|
|
11
|
+
A grader declares what kind of check it is at registration, so a consumer
|
|
12
|
+
plug-in is categorised the same way a built-in is.
|
|
13
|
+
|
|
14
|
+
Shipped as a minor despite the signature change below. `register` is public,
|
|
15
|
+
so the 1.0.0 policy would call this a major; it goes out as 2.1.0 as a
|
|
16
|
+
deliberate exception, because the break is a one-line edit per grader that
|
|
17
|
+
fails loudly at import.
|
|
18
|
+
|
|
19
|
+
### Changed
|
|
20
|
+
- **Breaking:** `graders.base.register` takes a required second argument,
|
|
21
|
+
`category`, a `graders.GraderType`. Every `@base.register('foo')` becomes
|
|
22
|
+
`@base.register('foo', base.GraderType.HEURISTIC)` or whichever member
|
|
23
|
+
applies; omitting it is a `TypeError` at import. Required rather than
|
|
24
|
+
defaulted on purpose - it is the only source of a row's `grader_type`, and
|
|
25
|
+
a default would be the value every grader forgets to override.
|
|
26
|
+
- `store.grader_lookups` reads the category from the registry instead of a
|
|
27
|
+
closed table of built-in type names, so a plug-in that declares
|
|
28
|
+
`HEURISTIC` reports `heuristic` where it used to report `unknown`. Rows for
|
|
29
|
+
consumer graders change value in the `grader_type` column; nothing about
|
|
30
|
+
the row shape changes.
|
|
31
|
+
|
|
32
|
+
### Added
|
|
33
|
+
- `graders.GraderType`, a `StrEnum` over the closed set the results store's
|
|
34
|
+
`grader_type` column accepts: `unknown`, `heuristic`, `statistical`,
|
|
35
|
+
`llm_as_judge`, `trajectory`, `human`. A `StrEnum` so it needs no
|
|
36
|
+
serializer of its own on the way to a row.
|
|
37
|
+
- `graders.category_of`, the registry lookup behind `grader_lookups`.
|
|
38
|
+
|
|
39
|
+
### Removed
|
|
40
|
+
- `store._GRADER_TYPES`, the private table the categories used to live in.
|
|
41
|
+
Keeping it alongside the registration argument would mean two sources for
|
|
42
|
+
one fact and a precedence rule between them.
|
|
43
|
+
|
|
44
|
+
## [2.0.0] - 2026-07-31
|
|
45
|
+
|
|
46
|
+
Breaks both the public API and the outbox row shape, so it's a major per the
|
|
47
|
+
1.0.0 policy.
|
|
48
|
+
|
|
49
|
+
### Changed
|
|
50
|
+
- **Breaking:** a sample is identified by `sample_hash` - a content digest of
|
|
51
|
+
the output it produced - rather than by the ordinal `sample_idx`. A row now
|
|
52
|
+
names the exact response behind it. The field is renamed on `CaseResult`,
|
|
53
|
+
`Rating`, `Preference`, `PairwiseOutcome` and `PairwiseAgreementCase`, and in
|
|
54
|
+
the outbox rows. Two runs of the same case never share a hash, so anything
|
|
55
|
+
comparing runs aligns on `case_id` and sample order; pairwise and the ranking
|
|
56
|
+
app anchor a pair on the `variant_a` side's digest so human and judge picks
|
|
57
|
+
still join. **Ratings and preferences files written by 1.x do not carry a
|
|
58
|
+
hash and will not join to a run** - re-collect them, or backfill the field.
|
|
59
|
+
- **Breaking:** outbox keys track the store's column names: `created_at` is
|
|
60
|
+
emitted as `started_at`, `n_cases` as `case_count`, `n_samples` as
|
|
61
|
+
`sample_count`.
|
|
62
|
+
- **Breaking:** `store.read_checkpoint_results` is now
|
|
63
|
+
`store.read_checkpoint_samples` and returns `(ordinal, result)` pairs.
|
|
64
|
+
Checkpoint lines nest the result under `result` and tag it with `sample`, so
|
|
65
|
+
resume knows which samples are still owed. 1.x checkpoints cannot be resumed.
|
|
66
|
+
|
|
67
|
+
### Fixed
|
|
68
|
+
- Resuming a run with `concurrency > 1` could skip a sample and re-run another.
|
|
69
|
+
A concurrent run checkpoints in completion order, so an interrupt leaves a
|
|
70
|
+
hole rather than a clean prefix; resume now reruns the samples that are
|
|
71
|
+
actually missing. On a deterministic target the re-run collided with a digest
|
|
72
|
+
already recorded, so the store collapsed two samples into one row.
|
|
73
|
+
- The pairwise judge and the side-by-side ranking app filtered samples
|
|
74
|
+
differently - non-empty content vs. no error - so with `n_samples > 1` and an
|
|
75
|
+
error on either side they paired A's n-th sample against different B samples.
|
|
76
|
+
`agreement` joins the two on A's digest, so it scored two different
|
|
77
|
+
comparisons as one. Both now go through
|
|
78
|
+
`pairwise.comparable_samples`, which requires a successful invocation *and*
|
|
79
|
+
resolvable content. Two behaviour changes fall out: an errored output is no
|
|
80
|
+
longer judged even when its content ref still resolves, and a sample with
|
|
81
|
+
empty content is no longer shown to a rater as a blank panel.
|
|
82
|
+
|
|
83
|
+
### Changed (internal)
|
|
84
|
+
- `pairwise._content_map` is now `pairwise.comparable_samples` and is the one
|
|
85
|
+
place that decides whether a sample can take part in a comparison.
|
|
86
|
+
|
|
87
|
+
### Removed
|
|
88
|
+
- `docs/clickhouse-schema.{sql,html}`. The outbox targets a flat row shape, not
|
|
89
|
+
one vendor's DDL, and the file documented a specific deployment - database
|
|
90
|
+
name, ingestion topology, sample data - none of which the engine needs. A
|
|
91
|
+
store with column types the feed does not match maps the rows at its own
|
|
92
|
+
boundary.
|
|
93
|
+
|
|
94
|
+
## [1.0.0] - 2026-07-28
|
|
95
|
+
|
|
96
|
+
First stable release. The public API and the outbox row shape are now covered
|
|
97
|
+
by semantic versioning: a breaking change to either means a 2.0.
|
|
98
|
+
|
|
99
|
+
### Changed
|
|
100
|
+
- **Breaking:** outbox rows no longer carry `input_tokens`, `output_tokens` or
|
|
101
|
+
`cost`. `Output.tokens` is an open dict an adapter may put anything in, and
|
|
102
|
+
usage accounting is captured outside the results store.
|
|
103
|
+
- **Breaking:** the run-grain `revision` key is emitted as
|
|
104
|
+
`application_revision`, which is its column name in the store.
|
|
105
|
+
|
|
106
|
+
### Added
|
|
107
|
+
- `docs/clickhouse-schema.sql`: the ClickHouse schema the outbox rows target -
|
|
108
|
+
the written `evaluation_scores` table, the row grammar as `CONSTRAINT`s, and
|
|
109
|
+
the canonical scorecard query. Also rendered as
|
|
110
|
+
`docs/clickhouse-schema.html`. (Both removed again in 2.0.0.)
|
|
111
|
+
|
|
112
|
+
## [0.3.0] - 2026-07-28
|
|
113
|
+
|
|
114
|
+
### Changed
|
|
115
|
+
- **Breaking:** the results-store outbox is now one feed at
|
|
116
|
+
`(run, case, sample, grader, metric)` grain, targeting a single flat
|
|
117
|
+
`evaluation_scores` table with the run trend and the invocation grain as
|
|
118
|
+
plain views over it. `store.scorecard_rows` and
|
|
119
|
+
`JsonlOutboxExporter.export` are removed; use `store.score_rows` and
|
|
120
|
+
`export_scores`. `--export-scores` is now an alias for `--export`.
|
|
121
|
+
- **Breaking:** a missing measurement is emitted as `null` rather than the
|
|
122
|
+
`(value=0, has_value=false)` sentinel pair, and the `has_value` / `has_stdev`
|
|
123
|
+
companion keys are gone. The store columns are `Nullable`, and a real `0.0` is
|
|
124
|
+
a meaningful score.
|
|
125
|
+
- **Breaking:** outbox row keys are the store column names, so `project` is
|
|
126
|
+
emitted as `application`, `mode` as `adapter_mode`, `created_at` as
|
|
127
|
+
`timestamp`, and `Output.error` as `is_error` plus `error_text`.
|
|
128
|
+
- **Breaking:** outbox rows no longer carry `model_id` or `prompt_version`. Both
|
|
129
|
+
are `variant.knobs.get(...)` projections and travel inside `variant_knobs`.
|
|
130
|
+
- **Breaking:** `run_id` is now a dashed UUIDv7 rather than `uuid4().hex`, so it
|
|
131
|
+
parses as a ClickHouse `UUID` and carries its own creation time. Uses
|
|
132
|
+
`uuid.uuid7()` on 3.14 and an RFC 9562 implementation on 3.11 through 3.13.
|
|
133
|
+
- Outbox rows now carry the gate: `gate_verdict`, `gate_win`,
|
|
134
|
+
`baseline_run_id`, `baseline_variant`, `win_baseline`, `win_candidate`,
|
|
135
|
+
`win_delta` and `gate_summary` at run grain, plus `win`, `guardrail` and
|
|
136
|
+
`guardrail_gap` on the metric each one refers to. Pass the `Comparison` to
|
|
137
|
+
`score_rows`; without it a run reads as ungated.
|
|
138
|
+
|
|
139
|
+
### Added
|
|
140
|
+
- `RunResult.aggregate_scores`, retaining the `kind='aggregate'` scores so a
|
|
141
|
+
store row keeps the grader that emitted them and what it reported. The
|
|
142
|
+
scorecard kept only their values.
|
|
143
|
+
- `store.grader_lookups`, mapping grader names to a category and to a judge
|
|
144
|
+
scale from a suite's grader specs, since a `Score` carries neither.
|
|
145
|
+
- Three outbox row shapes that previously had no representation: an aggregate
|
|
146
|
+
metric (no `case_id`), an invocation that failed before any grader ran (no
|
|
147
|
+
`grader` or `metric`), and a metric a guardrail or the win metric names but
|
|
148
|
+
never scored (null value).
|
|
149
|
+
|
|
150
|
+
## [0.2.0] - 2026-07-16
|
|
151
|
+
|
|
152
|
+
### Changed
|
|
153
|
+
- **Breaking:** the import package and CLI are now `evalcore` (were `evalkit`).
|
|
154
|
+
Update `import evalkit` to `import evalcore` and the `evalkit` command to
|
|
155
|
+
`evalcore`. The distribution name (`evalcore`) is unchanged.
|
|
156
|
+
- Lowered the minimum Python to **3.11** (was 3.14).
|
|
157
|
+
|
|
158
|
+
### Added
|
|
159
|
+
- `evalcore.__version__`.
|
|
160
|
+
- Public `evalcore.adapters.expand_env` for `${VAR}` expansion in custom
|
|
161
|
+
adapters (replaces the private `adapters._env` module).
|
|
162
|
+
- Exception hierarchy: `EvalcoreError` (base) and `ConfigError` (also a
|
|
163
|
+
`ValueError`, so existing handlers keep working).
|
|
164
|
+
- Top-level convenience entry points: `load_suite`, `load_cases`, `run_suite`,
|
|
165
|
+
`run_suite_sync`.
|
|
166
|
+
- HTML rendering for `sweep` and `pairwise` reports, and `--report` /
|
|
167
|
+
`--report-out` on those CLI commands.
|
|
168
|
+
|
|
169
|
+
## [0.1.0] - 2026-07-16
|
|
170
|
+
|
|
171
|
+
- Initial public release: adapters (http/replay), graders (deterministic,
|
|
172
|
+
numeric, classification, LLM judge + panel), runner (N-sampling, concurrency,
|
|
173
|
+
retries, checkpoint/resume), compare/gate, sweep, pairwise, blind human
|
|
174
|
+
rating + ranking with judge agreement, Markdown/HTML reporters, JSON +
|
|
175
|
+
column-store outbox, and content-hash provenance.
|
|
176
|
+
|
|
177
|
+
[Unreleased]: https://github.com/scottpmiller/evalcore/compare/2.1.0...HEAD
|
|
178
|
+
[2.1.0]: https://github.com/scottpmiller/evalcore/compare/2.0.0...2.1.0
|
|
179
|
+
[2.0.0]: https://github.com/scottpmiller/evalcore/compare/1.0.0...2.0.0
|
|
180
|
+
[1.0.0]: https://github.com/scottpmiller/evalcore/compare/0.3.0...1.0.0
|
|
181
|
+
[0.3.0]: https://github.com/scottpmiller/evalcore/compare/0.2.0...0.3.0
|
|
182
|
+
[0.2.0]: https://github.com/scottpmiller/evalcore/compare/0.1.0...0.2.0
|
|
183
|
+
[0.1.0]: https://github.com/scottpmiller/evalcore/releases/tag/0.1.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: evalcore
|
|
3
|
-
Version: 1.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: A generic, consumer-agnostic evaluation engine for prompt, model, and API outputs.
|
|
5
5
|
Project-URL: Homepage, https://github.com/scottpmiller/evalcore
|
|
6
6
|
Project-URL: Repository, https://github.com/scottpmiller/evalcore
|
|
@@ -695,7 +695,7 @@ shorthand; a repeatable `--view label:kind:ref` gives explicit control.
|
|
|
695
695
|
Sessions are **resumable** (a rater only sees items they haven't scored).
|
|
696
696
|
**Blinding is enforced server-side** - the queue payload carries an opaque
|
|
697
697
|
item id and never the run/variant/model; ratings map back to
|
|
698
|
-
`(run_id, case_id,
|
|
698
|
+
`(run_id, case_id, sample_hash)` only on the server. Ratings land in a JSONL
|
|
699
699
|
file (`models.Rating`) that is the **open interchange format**: any external
|
|
700
700
|
tool or spreadsheet export in the same shape feeds `agreement` too.
|
|
701
701
|
|
|
@@ -774,13 +774,15 @@ scored value, the invocation it came from, the gate verdict, and the full
|
|
|
774
774
|
reproducibility key (incl. `run_id`), so a multi-tenant trend table can
|
|
775
775
|
filter/group on any dimension without joins. Nothing derived is written: a
|
|
776
776
|
run's scorecard is a read-time aggregation over the same rows, so it cannot
|
|
777
|
-
disagree with the scores behind it. A
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
777
|
+
disagree with the scores behind it. A measurement that does not exist is
|
|
778
|
+
`null`, not `0` - a real `0.0` is a meaningful score, and `avg`/`sum`/`count`
|
|
779
|
+
skip nulls, so no query has to remember a filter; `passed` is the tri-state
|
|
780
|
+
string `'true'|'false'|'null'`, since a judge has no pass line by design and
|
|
781
|
+
that is a third state rather than a missing value. Row keys are column names
|
|
782
|
+
rather than model attribute names, so a JSONEachRow-style feed lands in a flat
|
|
783
|
+
table without a mapping layer. A store that forbids nullable columns fills
|
|
784
|
+
those nulls in at ingest, on its side of the seam. Swap the exporter for a real
|
|
785
|
+
database client without touching the runner or any consumer.
|
|
784
786
|
|
|
785
787
|
---
|
|
786
788
|
|
|
@@ -827,19 +829,18 @@ src/evalcore/
|
|
|
827
829
|
tests/ engine unit tests
|
|
828
830
|
examples/quickstart a runnable consumer that doubles as an implementation test
|
|
829
831
|
docs/design.md the design overview
|
|
830
|
-
docs/clickhouse-schema.sql the results-store schema the outbox rows target
|
|
831
832
|
```
|
|
832
833
|
|
|
833
834
|
## Status
|
|
834
835
|
|
|
835
|
-
|
|
836
|
-
either means a
|
|
836
|
+
2.0. The public API and the outbox row shape are stable; a breaking change to
|
|
837
|
+
either means a 3.0. Built: deterministic + classification + **LLM-judge**
|
|
837
838
|
(rubric scoring; single judge or a Claude/GPT **panel** with per-dimension
|
|
838
839
|
means, per-judge overalls, inter-judge disagreement flagging, and
|
|
839
840
|
image/screenshot inputs) graders, http/replay/browser adapters, runner
|
|
840
841
|
(N-sampling, optional concurrency, per-sample `RunResult` + `run_id` +
|
|
841
|
-
variance), comparison/gate, JSON + run + outbox store (one score-grain feed,
|
|
842
|
-
|
|
842
|
+
variance), comparison/gate, JSON + run + outbox store (one score-grain feed),
|
|
843
|
+
**N-way sweeps
|
|
843
844
|
+ counterbalanced pairwise A-vs-B win-rate** (`sweep`/`pairwise`), **blind
|
|
844
845
|
human-rating + side-by-side ranking web apps** with judge↔human agreement and
|
|
845
846
|
human-vs-judge pairwise agreement (`rate`/`agreement`, `rank`/`preferences`),
|
|
@@ -663,7 +663,7 @@ shorthand; a repeatable `--view label:kind:ref` gives explicit control.
|
|
|
663
663
|
Sessions are **resumable** (a rater only sees items they haven't scored).
|
|
664
664
|
**Blinding is enforced server-side** - the queue payload carries an opaque
|
|
665
665
|
item id and never the run/variant/model; ratings map back to
|
|
666
|
-
`(run_id, case_id,
|
|
666
|
+
`(run_id, case_id, sample_hash)` only on the server. Ratings land in a JSONL
|
|
667
667
|
file (`models.Rating`) that is the **open interchange format**: any external
|
|
668
668
|
tool or spreadsheet export in the same shape feeds `agreement` too.
|
|
669
669
|
|
|
@@ -742,13 +742,15 @@ scored value, the invocation it came from, the gate verdict, and the full
|
|
|
742
742
|
reproducibility key (incl. `run_id`), so a multi-tenant trend table can
|
|
743
743
|
filter/group on any dimension without joins. Nothing derived is written: a
|
|
744
744
|
run's scorecard is a read-time aggregation over the same rows, so it cannot
|
|
745
|
-
disagree with the scores behind it. A
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
745
|
+
disagree with the scores behind it. A measurement that does not exist is
|
|
746
|
+
`null`, not `0` - a real `0.0` is a meaningful score, and `avg`/`sum`/`count`
|
|
747
|
+
skip nulls, so no query has to remember a filter; `passed` is the tri-state
|
|
748
|
+
string `'true'|'false'|'null'`, since a judge has no pass line by design and
|
|
749
|
+
that is a third state rather than a missing value. Row keys are column names
|
|
750
|
+
rather than model attribute names, so a JSONEachRow-style feed lands in a flat
|
|
751
|
+
table without a mapping layer. A store that forbids nullable columns fills
|
|
752
|
+
those nulls in at ingest, on its side of the seam. Swap the exporter for a real
|
|
753
|
+
database client without touching the runner or any consumer.
|
|
752
754
|
|
|
753
755
|
---
|
|
754
756
|
|
|
@@ -795,19 +797,18 @@ src/evalcore/
|
|
|
795
797
|
tests/ engine unit tests
|
|
796
798
|
examples/quickstart a runnable consumer that doubles as an implementation test
|
|
797
799
|
docs/design.md the design overview
|
|
798
|
-
docs/clickhouse-schema.sql the results-store schema the outbox rows target
|
|
799
800
|
```
|
|
800
801
|
|
|
801
802
|
## Status
|
|
802
803
|
|
|
803
|
-
|
|
804
|
-
either means a
|
|
804
|
+
2.0. The public API and the outbox row shape are stable; a breaking change to
|
|
805
|
+
either means a 3.0. Built: deterministic + classification + **LLM-judge**
|
|
805
806
|
(rubric scoring; single judge or a Claude/GPT **panel** with per-dimension
|
|
806
807
|
means, per-judge overalls, inter-judge disagreement flagging, and
|
|
807
808
|
image/screenshot inputs) graders, http/replay/browser adapters, runner
|
|
808
809
|
(N-sampling, optional concurrency, per-sample `RunResult` + `run_id` +
|
|
809
|
-
variance), comparison/gate, JSON + run + outbox store (one score-grain feed,
|
|
810
|
-
|
|
810
|
+
variance), comparison/gate, JSON + run + outbox store (one score-grain feed),
|
|
811
|
+
**N-way sweeps
|
|
811
812
|
+ counterbalanced pairwise A-vs-B win-rate** (`sweep`/`pairwise`), **blind
|
|
812
813
|
human-rating + side-by-side ranking web apps** with judge↔human agreement and
|
|
813
814
|
human-vs-judge pairwise agreement (`rate`/`agreement`, `rank`/`preferences`),
|
|
@@ -127,19 +127,32 @@ Scorecards and comparisons serialize to JSON for CI artifacts and local files.
|
|
|
127
127
|
For trend tracking, results can land in a column store (e.g. ClickHouse) keyed
|
|
128
128
|
by `project`/`suite`. Rather than couple the engine to any particular database
|
|
129
129
|
driver, `store.JsonlOutboxExporter` flattens a run into a stable, flat row
|
|
130
|
-
shape and writes JSONL to an **outbox** a separate shipper drains:
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
130
|
+
shape and writes JSONL to an **outbox** a separate shipper drains: one feed at
|
|
131
|
+
`(run, case, sample, grader, metric)` grain, one row per score
|
|
132
|
+
(`run_id, case_id, sample_hash, grader, metric, metric_kind, value, passed,
|
|
133
|
+
detail`), carrying the invocation it came from and the gate verdict alongside.
|
|
134
|
+
|
|
135
|
+
Nothing derived is written. A run's scorecard and its trend are read-time
|
|
136
|
+
aggregations over the same rows, so they cannot disagree with the scores behind
|
|
137
|
+
them.
|
|
138
|
+
|
|
139
|
+
Every row repeats the full reproducibility key so a multi-tenant trend table can
|
|
140
|
+
filter/group on any dimension without joins.
|
|
141
|
+
|
|
142
|
+
A measurement that does not exist is `null`. A real `0.0` is a meaningful score -
|
|
143
|
+
what a failing deterministic check earns - so filling an absent one in with 0
|
|
144
|
+
would make the two unreadable apart, and `avg`/`sum`/`count` skip nulls anyway,
|
|
145
|
+
so no query has to remember a filter. `metric_kind` carries the engine's word
|
|
146
|
+
verbatim and says nothing about presence, so a metric some cases could not score
|
|
147
|
+
still reads as one metric; `'none'` is reserved for the two row shapes that hold
|
|
148
|
+
no score at all. `passed` is the tri-state string
|
|
149
|
+
`'true'|'false'|'null'`, since a judge has no pass line by design: a third
|
|
150
|
+
state, not a missing value.
|
|
151
|
+
|
|
152
|
+
A store that forbids nullable columns is free to fill those nulls in at ingest.
|
|
153
|
+
That mapping belongs at its boundary, not in the engine, which keeps absence
|
|
154
|
+
representable all the way out. Swap the exporter for a real database client
|
|
155
|
+
without touching the runner or any consumer.
|
|
143
156
|
|
|
144
157
|
## 7. Provenance
|
|
145
158
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
#
|
|
1
|
+
# evalcore - a generic, consumer-agnostic eval engine. Standard uv project.
|
|
2
2
|
|
|
3
3
|
# Sync the environment (editable install + dev/extras).
|
|
4
4
|
sync:
|
|
@@ -23,7 +23,7 @@ lint:
|
|
|
23
23
|
|
|
24
24
|
# Run the quickstart suite offline against recorded fixtures.
|
|
25
25
|
example:
|
|
26
|
-
uv run
|
|
26
|
+
uv run evalcore --plugins examples.quickstart.graders gate --suite examples/quickstart/suite.yaml --mode replay
|
|
27
27
|
|
|
28
28
|
# Run the quickstart suite through the Python API (no CLI), offline.
|
|
29
29
|
example-api:
|
|
@@ -31,4 +31,4 @@ example-api:
|
|
|
31
31
|
|
|
32
32
|
# Head-to-head A-vs-B win-rate over the quickstart suite (offline).
|
|
33
33
|
example-pairwise:
|
|
34
|
-
uv run
|
|
34
|
+
uv run evalcore --plugins examples.quickstart.graders pairwise --suite examples/quickstart/suite.yaml --a baseline --b candidate --mode replay
|
|
@@ -20,5 +20,14 @@ from evalcore.graders import (
|
|
|
20
20
|
judge,
|
|
21
21
|
numeric,
|
|
22
22
|
)
|
|
23
|
+
from evalcore.graders.base import GraderType, category_of
|
|
23
24
|
|
|
24
|
-
__all__ = [
|
|
25
|
+
__all__ = [
|
|
26
|
+
'GraderType',
|
|
27
|
+
'base',
|
|
28
|
+
'category_of',
|
|
29
|
+
'classification',
|
|
30
|
+
'deterministic',
|
|
31
|
+
'judge',
|
|
32
|
+
'numeric',
|
|
33
|
+
]
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Grader protocols and the type registry.
|
|
2
|
+
|
|
3
|
+
A grader spec is a plain dict from the suite config: ``{type, name, ...}``.
|
|
4
|
+
``build_graders`` turns a list of specs into grader instances, split into the
|
|
5
|
+
per-case and aggregate buckets the runner needs.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import enum
|
|
9
|
+
import typing
|
|
10
|
+
|
|
11
|
+
from evalcore import models
|
|
12
|
+
from evalcore.errors import ConfigError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class GraderType(enum.StrEnum):
|
|
16
|
+
"""What kind of check a grader performs, not which one.
|
|
17
|
+
|
|
18
|
+
A closed set, unlike the registry's ``type`` names, which any consumer
|
|
19
|
+
may extend. Declared once per grader at registration and read back by
|
|
20
|
+
``store.grader_lookups``; a ``StrEnum`` so it needs no serializer of its
|
|
21
|
+
own on the way to a row.
|
|
22
|
+
|
|
23
|
+
The same set is spelled out in three other places, all of which have to
|
|
24
|
+
change together: ``GraderType`` in ``internal-eval-results``, the
|
|
25
|
+
``grader_type`` ``Enum8`` in that repo's ``schema.sql``, and the deployed
|
|
26
|
+
DDL in ``schemata/clickhouse`` on GHE, which is the source of truth.
|
|
27
|
+
|
|
28
|
+
``UNKNOWN`` exists for a producer with no registry behind it. Nothing in
|
|
29
|
+
evalcore emits it: ``register`` requires a category, so a grader that
|
|
30
|
+
reaches a suite has always declared one.
|
|
31
|
+
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
UNKNOWN = 'unknown'
|
|
35
|
+
HEURISTIC = 'heuristic'
|
|
36
|
+
STATISTICAL = 'statistical'
|
|
37
|
+
LLM_AS_JUDGE = 'llm_as_judge'
|
|
38
|
+
TRAJECTORY = 'trajectory'
|
|
39
|
+
HUMAN = 'human'
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@typing.runtime_checkable
|
|
43
|
+
class Grader(typing.Protocol):
|
|
44
|
+
"""Per-case grader. Scores are averaged across cases by the runner."""
|
|
45
|
+
|
|
46
|
+
name: str
|
|
47
|
+
|
|
48
|
+
def grade(
|
|
49
|
+
self, case: models.Case, output: models.Output
|
|
50
|
+
) -> list[models.Score]: ...
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@typing.runtime_checkable
|
|
54
|
+
class AggregateGrader(typing.Protocol):
|
|
55
|
+
"""Whole-run grader for set-level metrics (P/R/F1, win-rate, ...)."""
|
|
56
|
+
|
|
57
|
+
name: str
|
|
58
|
+
|
|
59
|
+
def aggregate(
|
|
60
|
+
self, results: list[models.CaseResult]
|
|
61
|
+
) -> list[models.Score]: ...
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
_REGISTRY: dict[str, type] = {}
|
|
65
|
+
|
|
66
|
+
_CATEGORIES: dict[str, GraderType] = {}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def register(
|
|
70
|
+
type_name: str, category: GraderType
|
|
71
|
+
) -> typing.Callable[[type], type]:
|
|
72
|
+
"""Class decorator registering a grader under a suite-config ``type``.
|
|
73
|
+
|
|
74
|
+
``category`` is required rather than defaulting, because it is the only
|
|
75
|
+
source of the row's ``grader_type`` and a default would be the value
|
|
76
|
+
every grader forgets to override. A grader's category belongs to its
|
|
77
|
+
implementation, not to a suite's use of it, so it is declared here and
|
|
78
|
+
not in the suite config.
|
|
79
|
+
|
|
80
|
+
Args:
|
|
81
|
+
type_name: The ``type`` a suite spec names to select this grader.
|
|
82
|
+
category: What kind of check it performs.
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
The decorator.
|
|
86
|
+
|
|
87
|
+
Raises:
|
|
88
|
+
ConfigError: If ``type_name`` is already registered.
|
|
89
|
+
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def _decorate(cls: type) -> type:
|
|
93
|
+
if type_name in _REGISTRY:
|
|
94
|
+
raise ConfigError(f'grader type {type_name!r} already registered')
|
|
95
|
+
_REGISTRY[type_name] = cls
|
|
96
|
+
_CATEGORIES[type_name] = GraderType(category)
|
|
97
|
+
return cls
|
|
98
|
+
|
|
99
|
+
return _decorate
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def category_of(type_name: str) -> GraderType:
|
|
103
|
+
"""Return the category a grader type registered under.
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
type_name: The suite spec's ``type``.
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
The declared category, or ``UNKNOWN`` for a type no plug-in has
|
|
110
|
+
registered. A suite naming one cannot run - ``build_graders``
|
|
111
|
+
raises - so ``UNKNOWN`` only reaches a row when a caller builds
|
|
112
|
+
rows without loading the plug-ins that produced them.
|
|
113
|
+
|
|
114
|
+
"""
|
|
115
|
+
return _CATEGORIES.get(type_name, GraderType.UNKNOWN)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def build_graders(
|
|
119
|
+
specs: list[dict],
|
|
120
|
+
) -> tuple[list[Grader], list[AggregateGrader]]:
|
|
121
|
+
"""Instantiate grader specs, partitioned into per-case and aggregate.
|
|
122
|
+
|
|
123
|
+
Each spec's ``type`` selects a registered class; remaining keys (minus
|
|
124
|
+
``type``) are passed as keyword arguments to its constructor.
|
|
125
|
+
"""
|
|
126
|
+
per_case: list[Grader] = []
|
|
127
|
+
aggregate: list[AggregateGrader] = []
|
|
128
|
+
for spec in specs:
|
|
129
|
+
spec = dict(spec)
|
|
130
|
+
type_name = spec.pop('type')
|
|
131
|
+
if type_name not in _REGISTRY:
|
|
132
|
+
raise ConfigError(
|
|
133
|
+
f'unknown grader type {type_name!r}; '
|
|
134
|
+
f'known: {sorted(_REGISTRY)}'
|
|
135
|
+
)
|
|
136
|
+
grader = _REGISTRY[type_name](**spec)
|
|
137
|
+
if isinstance(grader, AggregateGrader):
|
|
138
|
+
aggregate.append(grader)
|
|
139
|
+
elif isinstance(grader, Grader):
|
|
140
|
+
per_case.append(grader)
|
|
141
|
+
else: # pragma: no cover - defensive
|
|
142
|
+
raise TypeError(
|
|
143
|
+
f'{type_name!r} is neither Grader nor AggregateGrader'
|
|
144
|
+
)
|
|
145
|
+
return per_case, aggregate
|
|
@@ -19,7 +19,7 @@ def _safe_div(numerator: float, denominator: float) -> float:
|
|
|
19
19
|
return numerator / denominator if denominator else 0.0
|
|
20
20
|
|
|
21
21
|
|
|
22
|
-
@base.register('classification')
|
|
22
|
+
@base.register('classification', base.GraderType.STATISTICAL)
|
|
23
23
|
class Classification:
|
|
24
24
|
"""Binary precision/recall/F1 + FN/FP rates over a labeled dataset."""
|
|
25
25
|
|
|
@@ -32,7 +32,7 @@ def _score(name: str, metric: str, case_id: str, ok: bool, detail: str):
|
|
|
32
32
|
)
|
|
33
33
|
|
|
34
34
|
|
|
35
|
-
@base.register('max_chars')
|
|
35
|
+
@base.register('max_chars', base.GraderType.HEURISTIC)
|
|
36
36
|
class MaxChars:
|
|
37
37
|
"""Assert a text field is at most ``maximum`` characters long."""
|
|
38
38
|
|
|
@@ -58,7 +58,7 @@ class MaxChars:
|
|
|
58
58
|
]
|
|
59
59
|
|
|
60
60
|
|
|
61
|
-
@base.register('regex_absent')
|
|
61
|
+
@base.register('regex_absent', base.GraderType.HEURISTIC)
|
|
62
62
|
class RegexAbsent:
|
|
63
63
|
"""Assert a text field does NOT match ``pattern`` (e.g. no tokens)."""
|
|
64
64
|
|
|
@@ -78,7 +78,7 @@ class RegexAbsent:
|
|
|
78
78
|
return [_score(self.name, self.name, case.id, ok, detail)]
|
|
79
79
|
|
|
80
80
|
|
|
81
|
-
@base.register('regex_present')
|
|
81
|
+
@base.register('regex_present', base.GraderType.HEURISTIC)
|
|
82
82
|
class RegexPresent:
|
|
83
83
|
"""Assert a text field matches EVERY pattern in ``patterns`` (all-of).
|
|
84
84
|
|
|
@@ -111,7 +111,7 @@ class RegexPresent:
|
|
|
111
111
|
return [_score(self.name, self.name, case.id, ok, detail)]
|
|
112
112
|
|
|
113
113
|
|
|
114
|
-
@base.register('non_empty')
|
|
114
|
+
@base.register('non_empty', base.GraderType.HEURISTIC)
|
|
115
115
|
class NonEmpty:
|
|
116
116
|
"""Assert a field resolves to a non-empty value."""
|
|
117
117
|
|
|
@@ -289,7 +289,7 @@ def _load_images(refs_values: list) -> list[dict]:
|
|
|
289
289
|
return images
|
|
290
290
|
|
|
291
291
|
|
|
292
|
-
@base.register('llm_judge')
|
|
292
|
+
@base.register('llm_judge', base.GraderType.LLM_AS_JUDGE)
|
|
293
293
|
class RubricJudge:
|
|
294
294
|
"""Score an output's text on rubric dimensions with an LLM judge/panel.
|
|
295
295
|
|