evalcore 1.0.0__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. evalcore-2.1.0/CHANGELOG.md +183 -0
  2. {evalcore-1.0.0 → evalcore-2.1.0}/PKG-INFO +15 -14
  3. {evalcore-1.0.0 → evalcore-2.1.0}/README.md +14 -13
  4. {evalcore-1.0.0 → evalcore-2.1.0}/docs/design.md +26 -13
  5. {evalcore-1.0.0 → evalcore-2.1.0}/justfile +3 -3
  6. {evalcore-1.0.0 → evalcore-2.1.0}/pyproject.toml +1 -1
  7. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/__init__.py +10 -1
  8. evalcore-2.1.0/src/evalcore/graders/base.py +145 -0
  9. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/classification.py +1 -1
  10. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/deterministic.py +4 -4
  11. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/judge.py +1 -1
  12. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/graders/numeric.py +1 -1
  13. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/models.py +27 -7
  14. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/pairwise.py +62 -22
  15. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/rating.py +56 -46
  16. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/base.py +5 -1
  17. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/runner.py +47 -23
  18. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/store.py +89 -55
  19. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_cli.py +16 -6
  20. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_edge_cases.py +2 -2
  21. evalcore-2.1.0/tests/test_pairwise_extra.py +178 -0
  22. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_rating.py +26 -41
  23. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_rating_server.py +2 -2
  24. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_reporters.py +3 -3
  25. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_runner.py +127 -19
  26. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_store.py +87 -12
  27. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_sweep_pairwise.py +1 -1
  28. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_unit.py +1 -1
  29. {evalcore-1.0.0 → evalcore-2.1.0}/uv.lock +1 -1
  30. evalcore-1.0.0/CHANGELOG.md +0 -96
  31. evalcore-1.0.0/docs/clickhouse-schema.html +0 -1112
  32. evalcore-1.0.0/docs/clickhouse-schema.sql +0 -412
  33. evalcore-1.0.0/src/evalcore/graders/base.py +0 -78
  34. evalcore-1.0.0/tests/test_pairwise_extra.py +0 -68
  35. {evalcore-1.0.0 → evalcore-2.1.0}/.github/workflows/ci.yml +0 -0
  36. {evalcore-1.0.0 → evalcore-2.1.0}/.github/workflows/publish.yml +0 -0
  37. {evalcore-1.0.0 → evalcore-2.1.0}/.gitignore +0 -0
  38. {evalcore-1.0.0 → evalcore-2.1.0}/.pre-commit-config.yaml +0 -0
  39. {evalcore-1.0.0 → evalcore-2.1.0}/LICENSE +0 -0
  40. {evalcore-1.0.0 → evalcore-2.1.0}/examples/__init__.py +0 -0
  41. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/README.md +0 -0
  42. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/__init__.py +0 -0
  43. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/adapter.py +0 -0
  44. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/angry_complaint.yaml +0 -0
  45. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/billing_question.yaml +0 -0
  46. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/double_charge.yaml +0 -0
  47. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/datasets/support_reply/v1/cases/refund_request.yaml +0 -0
  48. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_judge.yaml +0 -0
  49. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_pairwise.yaml +0 -0
  50. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/fixtures/support_reply_replay.yaml +0 -0
  51. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/graders.py +0 -0
  52. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/run_eval.py +0 -0
  53. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/suite.yaml +0 -0
  54. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/tests/__init__.py +0 -0
  55. {evalcore-1.0.0 → evalcore-2.1.0}/examples/quickstart/tests/test_quickstart.py +0 -0
  56. {evalcore-1.0.0 → evalcore-2.1.0}/pyrightconfig.json +0 -0
  57. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/__init__.py +0 -0
  58. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/__init__.py +0 -0
  59. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/base.py +0 -0
  60. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/env.py +0 -0
  61. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/http.py +0 -0
  62. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/adapters/replay.py +0 -0
  63. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/cli.py +0 -0
  64. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/compare.py +0 -0
  65. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/errors.py +0 -0
  66. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/loader.py +0 -0
  67. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/py.typed +0 -0
  68. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/refs.py +0 -0
  69. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/report.py +0 -0
  70. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/__init__.py +0 -0
  71. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/html.py +0 -0
  72. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/reporters/markdown.py +0 -0
  73. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/retry.py +0 -0
  74. {evalcore-1.0.0 → evalcore-2.1.0}/src/evalcore/sweep.py +0 -0
  75. {evalcore-1.0.0 → evalcore-2.1.0}/tests/__init__.py +0 -0
  76. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_adapters.py +0 -0
  77. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_judge.py +0 -0
  78. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_judge_extra.py +0 -0
  79. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_live_clients.py +0 -0
  80. {evalcore-1.0.0 → evalcore-2.1.0}/tests/test_retry.py +0 -0
  81. {evalcore-1.0.0 → evalcore-2.1.0}/uv.toml +0 -0
@@ -0,0 +1,183 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format is based on
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project follows
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [2.1.0] - 2026-08-07
10
+
11
+ A grader declares what kind of check it is at registration, so a consumer
12
+ plug-in is categorised the same way a built-in is.
13
+
14
+ Shipped as a minor despite the signature change below. `register` is public,
15
+ so the 1.0.0 policy would call this a major; it goes out as 2.1.0 as a
16
+ deliberate exception, because the break is a one-line edit per grader that
17
+ fails loudly at import.
18
+
19
+ ### Changed
20
+ - **Breaking:** `graders.base.register` takes a required second argument,
21
+ `category`, a `graders.GraderType`. Every `@base.register('foo')` becomes
22
+ `@base.register('foo', base.GraderType.HEURISTIC)` or whichever member
23
+ applies; omitting it is a `TypeError` at import. Required rather than
24
+ defaulted on purpose - it is the only source of a row's `grader_type`, and
25
+ a default would be the value every grader forgets to override.
26
+ - `store.grader_lookups` reads the category from the registry instead of a
27
+ closed table of built-in type names, so a plug-in that declares
28
+ `HEURISTIC` reports `heuristic` where it used to report `unknown`. Rows for
29
+ consumer graders change value in the `grader_type` column; nothing about
30
+ the row shape changes.
31
+
32
+ ### Added
33
+ - `graders.GraderType`, a `StrEnum` over the closed set the results store's
34
+ `grader_type` column accepts: `unknown`, `heuristic`, `statistical`,
35
+ `llm_as_judge`, `trajectory`, `human`. A `StrEnum` so it needs no
36
+ serializer of its own on the way to a row.
37
+ - `graders.category_of`, the registry lookup behind `grader_lookups`.
38
+
39
+ ### Removed
40
+ - `store._GRADER_TYPES`, the private table the categories used to live in.
41
+ Keeping it alongside the registration argument would mean two sources for
42
+ one fact and a precedence rule between them.
43
+
44
+ ## [2.0.0] - 2026-07-31
45
+
46
+ Breaks both the public API and the outbox row shape, so it's a major per the
47
+ 1.0.0 policy.
48
+
49
+ ### Changed
50
+ - **Breaking:** a sample is identified by `sample_hash` - a content digest of
51
+ the output it produced - rather than by the ordinal `sample_idx`. A row now
52
+ names the exact response behind it. The field is renamed on `CaseResult`,
53
+ `Rating`, `Preference`, `PairwiseOutcome` and `PairwiseAgreementCase`, and in
54
+ the outbox rows. Two runs of the same case never share a hash, so anything
55
+ comparing runs aligns on `case_id` and sample order; pairwise and the ranking
56
+ app anchor a pair on the `variant_a` side's digest so human and judge picks
57
+ still join. **Ratings and preferences files written by 1.x do not carry a
58
+ hash and will not join to a run** - re-collect them, or backfill the field.
59
+ - **Breaking:** outbox keys track the store's column names: `created_at` is
60
+ emitted as `started_at`, `n_cases` as `case_count`, `n_samples` as
61
+ `sample_count`.
62
+ - **Breaking:** `store.read_checkpoint_results` is now
63
+ `store.read_checkpoint_samples` and returns `(ordinal, result)` pairs.
64
+ Checkpoint lines nest the result under `result` and tag it with `sample`, so
65
+ resume knows which samples are still owed. 1.x checkpoints cannot be resumed.
66
+
67
+ ### Fixed
68
+ - Resuming a run with `concurrency > 1` could skip a sample and re-run another.
69
+ A concurrent run checkpoints in completion order, so an interrupt leaves a
70
+ hole rather than a clean prefix; resume now reruns the samples that are
71
+ actually missing. On a deterministic target the re-run collided with a digest
72
+ already recorded, so the store collapsed two samples into one row.
73
+ - The pairwise judge and the side-by-side ranking app filtered samples
74
+ differently - non-empty content vs. no error - so with `n_samples > 1` and an
75
+ error on either side they paired A's n-th sample against different B samples.
76
+ `agreement` joins the two on A's digest, so it scored two different
77
+ comparisons as one. Both now go through
78
+ `pairwise.comparable_samples`, which requires a successful invocation *and*
79
+ resolvable content. Two behaviour changes fall out: an errored output is no
80
+ longer judged even when its content ref still resolves, and a sample with
81
+ empty content is no longer shown to a rater as a blank panel.
82
+
83
+ ### Changed (internal)
84
+ - `pairwise._content_map` is now `pairwise.comparable_samples` and is the one
85
+ place that decides whether a sample can take part in a comparison.
86
+
87
+ ### Removed
88
+ - `docs/clickhouse-schema.{sql,html}`. The outbox targets a flat row shape, not
89
+ one vendor's DDL, and the file documented a specific deployment - database
90
+ name, ingestion topology, sample data - none of which the engine needs. A
91
+ store with column types the feed does not match maps the rows at its own
92
+ boundary.
93
+
94
+ ## [1.0.0] - 2026-07-28
95
+
96
+ First stable release. The public API and the outbox row shape are now covered
97
+ by semantic versioning: a breaking change to either means a 2.0.
98
+
99
+ ### Changed
100
+ - **Breaking:** outbox rows no longer carry `input_tokens`, `output_tokens` or
101
+ `cost`. `Output.tokens` is an open dict an adapter may put anything in, and
102
+ usage accounting is captured outside the results store.
103
+ - **Breaking:** the run-grain `revision` key is emitted as
104
+ `application_revision`, which is its column name in the store.
105
+
106
+ ### Added
107
+ - `docs/clickhouse-schema.sql`: the ClickHouse schema the outbox rows target -
108
+ the written `evaluation_scores` table, the row grammar as `CONSTRAINT`s, and
109
+ the canonical scorecard query. Also rendered as
110
+ `docs/clickhouse-schema.html`. (Both removed again in 2.0.0.)
111
+
112
+ ## [0.3.0] - 2026-07-28
113
+
114
+ ### Changed
115
+ - **Breaking:** the results-store outbox is now one feed at
116
+ `(run, case, sample, grader, metric)` grain, targeting a single flat
117
+ `evaluation_scores` table with the run trend and the invocation grain as
118
+ plain views over it. `store.scorecard_rows` and
119
+ `JsonlOutboxExporter.export` are removed; use `store.score_rows` and
120
+ `export_scores`. `--export-scores` is now an alias for `--export`.
121
+ - **Breaking:** a missing measurement is emitted as `null` rather than the
122
+ `(value=0, has_value=false)` sentinel pair, and the `has_value` / `has_stdev`
123
+ companion keys are gone. The store columns are `Nullable`, and a real `0.0` is
124
+ a meaningful score.
125
+ - **Breaking:** outbox row keys are the store column names, so `project` is
126
+ emitted as `application`, `mode` as `adapter_mode`, `created_at` as
127
+ `timestamp`, and `Output.error` as `is_error` plus `error_text`.
128
+ - **Breaking:** outbox rows no longer carry `model_id` or `prompt_version`. Both
129
+ are `variant.knobs.get(...)` projections and travel inside `variant_knobs`.
130
+ - **Breaking:** `run_id` is now a dashed UUIDv7 rather than `uuid4().hex`, so it
131
+ parses as a ClickHouse `UUID` and carries its own creation time. Uses
132
+ `uuid.uuid7()` on 3.14 and an RFC 9562 implementation on 3.11 through 3.13.
133
+ - Outbox rows now carry the gate: `gate_verdict`, `gate_win`,
134
+ `baseline_run_id`, `baseline_variant`, `win_baseline`, `win_candidate`,
135
+ `win_delta` and `gate_summary` at run grain, plus `win`, `guardrail` and
136
+ `guardrail_gap` on the metric each one refers to. Pass the `Comparison` to
137
+ `score_rows`; without it a run reads as ungated.
138
+
139
+ ### Added
140
+ - `RunResult.aggregate_scores`, retaining the `kind='aggregate'` scores so a
141
+ store row keeps the grader that emitted them and what it reported. The
142
+ scorecard kept only their values.
143
+ - `store.grader_lookups`, mapping grader names to a category and to a judge
144
+ scale from a suite's grader specs, since a `Score` carries neither.
145
+ - Three outbox row shapes that previously had no representation: an aggregate
146
+ metric (no `case_id`), an invocation that failed before any grader ran (no
147
+ `grader` or `metric`), and a metric a guardrail or the win metric names but
148
+ never scored (null value).
149
+
150
+ ## [0.2.0] - 2026-07-16
151
+
152
+ ### Changed
153
+ - **Breaking:** the import package and CLI are now `evalcore` (were `evalkit`).
154
+ Update `import evalkit` to `import evalcore` and the `evalkit` command to
155
+ `evalcore`. The distribution name (`evalcore`) is unchanged.
156
+ - Lowered the minimum Python to **3.11** (was 3.14).
157
+
158
+ ### Added
159
+ - `evalcore.__version__`.
160
+ - Public `evalcore.adapters.expand_env` for `${VAR}` expansion in custom
161
+ adapters (replaces the private `adapters._env` module).
162
+ - Exception hierarchy: `EvalcoreError` (base) and `ConfigError` (also a
163
+ `ValueError`, so existing handlers keep working).
164
+ - Top-level convenience entry points: `load_suite`, `load_cases`, `run_suite`,
165
+ `run_suite_sync`.
166
+ - HTML rendering for `sweep` and `pairwise` reports, and `--report` /
167
+ `--report-out` on those CLI commands.
168
+
169
+ ## [0.1.0] - 2026-07-16
170
+
171
+ - Initial public release: adapters (http/replay), graders (deterministic,
172
+ numeric, classification, LLM judge + panel), runner (N-sampling, concurrency,
173
+ retries, checkpoint/resume), compare/gate, sweep, pairwise, blind human
174
+ rating + ranking with judge agreement, Markdown/HTML reporters, JSON +
175
+ column-store outbox, and content-hash provenance.
176
+
177
+ [Unreleased]: https://github.com/scottpmiller/evalcore/compare/2.1.0...HEAD
178
+ [2.1.0]: https://github.com/scottpmiller/evalcore/compare/2.0.0...2.1.0
179
+ [2.0.0]: https://github.com/scottpmiller/evalcore/compare/1.0.0...2.0.0
180
+ [1.0.0]: https://github.com/scottpmiller/evalcore/compare/0.3.0...1.0.0
181
+ [0.3.0]: https://github.com/scottpmiller/evalcore/compare/0.2.0...0.3.0
182
+ [0.2.0]: https://github.com/scottpmiller/evalcore/compare/0.1.0...0.2.0
183
+ [0.1.0]: https://github.com/scottpmiller/evalcore/releases/tag/0.1.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: evalcore
3
- Version: 1.0.0
3
+ Version: 2.1.0
4
4
  Summary: A generic, consumer-agnostic evaluation engine for prompt, model, and API outputs.
5
5
  Project-URL: Homepage, https://github.com/scottpmiller/evalcore
6
6
  Project-URL: Repository, https://github.com/scottpmiller/evalcore
@@ -695,7 +695,7 @@ shorthand; a repeatable `--view label:kind:ref` gives explicit control.
695
695
  Sessions are **resumable** (a rater only sees items they haven't scored).
696
696
  **Blinding is enforced server-side** - the queue payload carries an opaque
697
697
  item id and never the run/variant/model; ratings map back to
698
- `(run_id, case_id, sample_idx)` only on the server. Ratings land in a JSONL
698
+ `(run_id, case_id, sample_hash)` only on the server. Ratings land in a JSONL
699
699
  file (`models.Rating`) that is the **open interchange format**: any external
700
700
  tool or spreadsheet export in the same shape feeds `agreement` too.
701
701
 
@@ -774,13 +774,15 @@ scored value, the invocation it came from, the gate verdict, and the full
774
774
  reproducibility key (incl. `run_id`), so a multi-tenant trend table can
775
775
  filter/group on any dimension without joins. Nothing derived is written: a
776
776
  run's scorecard is a read-time aggregation over the same rows, so it cannot
777
- disagree with the scores behind it. A missing measurement is `null` (those
778
- columns are `Nullable`, and a real `0.0` is a meaningful score); `passed` is
779
- the tri-state string `'true'|'false'|'null'`, matching its `Enum8`. Row keys
780
- are the store's column names, so a JSONEachRow-style feed maps onto the schema
781
- in
782
- [`docs/clickhouse-schema.sql`](docs/clickhouse-schema.sql). Swap the exporter
783
- for a real database client without touching the runner or any consumer.
777
+ disagree with the scores behind it. A measurement that does not exist is
778
+ `null`, not `0` - a real `0.0` is a meaningful score, and `avg`/`sum`/`count`
779
+ skip nulls, so no query has to remember a filter; `passed` is the tri-state
780
+ string `'true'|'false'|'null'`, since a judge has no pass line by design and
781
+ that is a third state rather than a missing value. Row keys are column names
782
+ rather than model attribute names, so a JSONEachRow-style feed lands in a flat
783
+ table without a mapping layer. A store that forbids nullable columns fills
784
+ those nulls in at ingest, on its side of the seam. Swap the exporter for a real
785
+ database client without touching the runner or any consumer.
784
786
 
785
787
  ---
786
788
 
@@ -827,19 +829,18 @@ src/evalcore/
827
829
  tests/ engine unit tests
828
830
  examples/quickstart a runnable consumer that doubles as an implementation test
829
831
  docs/design.md the design overview
830
- docs/clickhouse-schema.sql the results-store schema the outbox rows target
831
832
  ```
832
833
 
833
834
  ## Status
834
835
 
835
- 1.0. The public API and the outbox row shape are stable; a breaking change to
836
- either means a 2.0. Built: deterministic + classification + **LLM-judge**
836
+ 2.0. The public API and the outbox row shape are stable; a breaking change to
837
+ either means a 3.0. Built: deterministic + classification + **LLM-judge**
837
838
  (rubric scoring; single judge or a Claude/GPT **panel** with per-dimension
838
839
  means, per-judge overalls, inter-judge disagreement flagging, and
839
840
  image/screenshot inputs) graders, http/replay/browser adapters, runner
840
841
  (N-sampling, optional concurrency, per-sample `RunResult` + `run_id` +
841
- variance), comparison/gate, JSON + run + outbox store (one score-grain feed,
842
- `docs/clickhouse-schema.sql`), **N-way sweeps
842
+ variance), comparison/gate, JSON + run + outbox store (one score-grain feed),
843
+ **N-way sweeps
843
844
  + counterbalanced pairwise A-vs-B win-rate** (`sweep`/`pairwise`), **blind
844
845
  human-rating + side-by-side ranking web apps** with judge↔human agreement and
845
846
  human-vs-judge pairwise agreement (`rate`/`agreement`, `rank`/`preferences`),
@@ -663,7 +663,7 @@ shorthand; a repeatable `--view label:kind:ref` gives explicit control.
663
663
  Sessions are **resumable** (a rater only sees items they haven't scored).
664
664
  **Blinding is enforced server-side** - the queue payload carries an opaque
665
665
  item id and never the run/variant/model; ratings map back to
666
- `(run_id, case_id, sample_idx)` only on the server. Ratings land in a JSONL
666
+ `(run_id, case_id, sample_hash)` only on the server. Ratings land in a JSONL
667
667
  file (`models.Rating`) that is the **open interchange format**: any external
668
668
  tool or spreadsheet export in the same shape feeds `agreement` too.
669
669
 
@@ -742,13 +742,15 @@ scored value, the invocation it came from, the gate verdict, and the full
742
742
  reproducibility key (incl. `run_id`), so a multi-tenant trend table can
743
743
  filter/group on any dimension without joins. Nothing derived is written: a
744
744
  run's scorecard is a read-time aggregation over the same rows, so it cannot
745
- disagree with the scores behind it. A missing measurement is `null` (those
746
- columns are `Nullable`, and a real `0.0` is a meaningful score); `passed` is
747
- the tri-state string `'true'|'false'|'null'`, matching its `Enum8`. Row keys
748
- are the store's column names, so a JSONEachRow-style feed maps onto the schema
749
- in
750
- [`docs/clickhouse-schema.sql`](docs/clickhouse-schema.sql). Swap the exporter
751
- for a real database client without touching the runner or any consumer.
745
+ disagree with the scores behind it. A measurement that does not exist is
746
+ `null`, not `0` - a real `0.0` is a meaningful score, and `avg`/`sum`/`count`
747
+ skip nulls, so no query has to remember a filter; `passed` is the tri-state
748
+ string `'true'|'false'|'null'`, since a judge has no pass line by design and
749
+ that is a third state rather than a missing value. Row keys are column names
750
+ rather than model attribute names, so a JSONEachRow-style feed lands in a flat
751
+ table without a mapping layer. A store that forbids nullable columns fills
752
+ those nulls in at ingest, on its side of the seam. Swap the exporter for a real
753
+ database client without touching the runner or any consumer.
752
754
 
753
755
  ---
754
756
 
@@ -795,19 +797,18 @@ src/evalcore/
795
797
  tests/ engine unit tests
796
798
  examples/quickstart a runnable consumer that doubles as an implementation test
797
799
  docs/design.md the design overview
798
- docs/clickhouse-schema.sql the results-store schema the outbox rows target
799
800
  ```
800
801
 
801
802
  ## Status
802
803
 
803
- 1.0. The public API and the outbox row shape are stable; a breaking change to
804
- either means a 2.0. Built: deterministic + classification + **LLM-judge**
804
+ 2.0. The public API and the outbox row shape are stable; a breaking change to
805
+ either means a 3.0. Built: deterministic + classification + **LLM-judge**
805
806
  (rubric scoring; single judge or a Claude/GPT **panel** with per-dimension
806
807
  means, per-judge overalls, inter-judge disagreement flagging, and
807
808
  image/screenshot inputs) graders, http/replay/browser adapters, runner
808
809
  (N-sampling, optional concurrency, per-sample `RunResult` + `run_id` +
809
- variance), comparison/gate, JSON + run + outbox store (one score-grain feed,
810
- `docs/clickhouse-schema.sql`), **N-way sweeps
810
+ variance), comparison/gate, JSON + run + outbox store (one score-grain feed),
811
+ **N-way sweeps
811
812
  + counterbalanced pairwise A-vs-B win-rate** (`sweep`/`pairwise`), **blind
812
813
  human-rating + side-by-side ranking web apps** with judge↔human agreement and
813
814
  human-vs-judge pairwise agreement (`rate`/`agreement`, `rank`/`preferences`),
@@ -127,19 +127,32 @@ Scorecards and comparisons serialize to JSON for CI artifacts and local files.
127
127
  For trend tracking, results can land in a column store (e.g. ClickHouse) keyed
128
128
  by `project`/`suite`. Rather than couple the engine to any particular database
129
129
  driver, `store.JsonlOutboxExporter` flattens a run into a stable, flat row
130
- shape and writes JSONL to an **outbox** a separate shipper drains:
131
-
132
- - one **metric** row per scorecard metric (`metric, value, stdev, metric_kind,
133
- n`), and
134
- - one **per-case score** row (`run_id, case_id, sample_idx, grader, metric,
135
- value, passed, detail`).
136
-
137
- Both repeat the full reproducibility key so a multi-tenant trend table can
138
- filter/group on any dimension without joins. The rows use a no-`Nullable`
139
- convention that maps cleanly onto a column store (a missing value is the
140
- sentinel pair `(value=0, has_value=false)`; `passed` is the tri-state string
141
- `'true'|'false'|'null'`). Swap the exporter for a real database client without
142
- touching the runner or any consumer.
130
+ shape and writes JSONL to an **outbox** a separate shipper drains: one feed at
131
+ `(run, case, sample, grader, metric)` grain, one row per score
132
+ (`run_id, case_id, sample_hash, grader, metric, metric_kind, value, passed,
133
+ detail`), carrying the invocation it came from and the gate verdict alongside.
134
+
135
+ Nothing derived is written. A run's scorecard and its trend are read-time
136
+ aggregations over the same rows, so they cannot disagree with the scores behind
137
+ them.
138
+
139
+ Every row repeats the full reproducibility key so a multi-tenant trend table can
140
+ filter/group on any dimension without joins.
141
+
142
+ A measurement that does not exist is `null`. A real `0.0` is a meaningful score -
143
+ what a failing deterministic check earns - so filling an absent one in with 0
144
+ would make the two unreadable apart, and `avg`/`sum`/`count` skip nulls anyway,
145
+ so no query has to remember a filter. `metric_kind` carries the engine's word
146
+ verbatim and says nothing about presence, so a metric some cases could not score
147
+ still reads as one metric; `'none'` is reserved for the two row shapes that hold
148
+ no score at all. `passed` is the tri-state string
149
+ `'true'|'false'|'null'`, since a judge has no pass line by design: a third
150
+ state, not a missing value.
151
+
152
+ A store that forbids nullable columns is free to fill those nulls in at ingest.
153
+ That mapping belongs at its boundary, not in the engine, which keeps absence
154
+ representable all the way out. Swap the exporter for a real database client
155
+ without touching the runner or any consumer.
143
156
 
144
157
  ## 7. Provenance
145
158
 
@@ -1,4 +1,4 @@
1
- # evalkit - a generic, consumer-agnostic eval engine. Standard uv project.
1
+ # evalcore - a generic, consumer-agnostic eval engine. Standard uv project.
2
2
 
3
3
  # Sync the environment (editable install + dev/extras).
4
4
  sync:
@@ -23,7 +23,7 @@ lint:
23
23
 
24
24
  # Run the quickstart suite offline against recorded fixtures.
25
25
  example:
26
- uv run evalkit --plugins examples.quickstart.graders gate --suite examples/quickstart/suite.yaml --mode replay
26
+ uv run evalcore --plugins examples.quickstart.graders gate --suite examples/quickstart/suite.yaml --mode replay
27
27
 
28
28
  # Run the quickstart suite through the Python API (no CLI), offline.
29
29
  example-api:
@@ -31,4 +31,4 @@ example-api:
31
31
 
32
32
  # Head-to-head A-vs-B win-rate over the quickstart suite (offline).
33
33
  example-pairwise:
34
- uv run evalkit --plugins examples.quickstart.graders pairwise --suite examples/quickstart/suite.yaml --a baseline --b candidate --mode replay
34
+ uv run evalcore --plugins examples.quickstart.graders pairwise --suite examples/quickstart/suite.yaml --a baseline --b candidate --mode replay
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "evalcore"
3
- version = "1.0.0"
3
+ version = "2.1.0"
4
4
  description = "A generic, consumer-agnostic evaluation engine for prompt, model, and API outputs."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -20,5 +20,14 @@ from evalcore.graders import (
20
20
  judge,
21
21
  numeric,
22
22
  )
23
+ from evalcore.graders.base import GraderType, category_of
23
24
 
24
- __all__ = ['base', 'classification', 'deterministic', 'judge', 'numeric']
25
+ __all__ = [
26
+ 'GraderType',
27
+ 'base',
28
+ 'category_of',
29
+ 'classification',
30
+ 'deterministic',
31
+ 'judge',
32
+ 'numeric',
33
+ ]
@@ -0,0 +1,145 @@
1
+ """Grader protocols and the type registry.
2
+
3
+ A grader spec is a plain dict from the suite config: ``{type, name, ...}``.
4
+ ``build_graders`` turns a list of specs into grader instances, split into the
5
+ per-case and aggregate buckets the runner needs.
6
+ """
7
+
8
+ import enum
9
+ import typing
10
+
11
+ from evalcore import models
12
+ from evalcore.errors import ConfigError
13
+
14
+
15
+ class GraderType(enum.StrEnum):
16
+ """What kind of check a grader performs, not which one.
17
+
18
+ A closed set, unlike the registry's ``type`` names, which any consumer
19
+ may extend. Declared once per grader at registration and read back by
20
+ ``store.grader_lookups``; a ``StrEnum`` so it needs no serializer of its
21
+ own on the way to a row.
22
+
23
+ The same set is spelled out in three other places, all of which have to
24
+ change together: ``GraderType`` in ``internal-eval-results``, the
25
+ ``grader_type`` ``Enum8`` in that repo's ``schema.sql``, and the deployed
26
+ DDL in ``schemata/clickhouse`` on GHE, which is the source of truth.
27
+
28
+ ``UNKNOWN`` exists for a producer with no registry behind it. Nothing in
29
+ evalcore emits it: ``register`` requires a category, so a grader that
30
+ reaches a suite has always declared one.
31
+
32
+ """
33
+
34
+ UNKNOWN = 'unknown'
35
+ HEURISTIC = 'heuristic'
36
+ STATISTICAL = 'statistical'
37
+ LLM_AS_JUDGE = 'llm_as_judge'
38
+ TRAJECTORY = 'trajectory'
39
+ HUMAN = 'human'
40
+
41
+
42
+ @typing.runtime_checkable
43
+ class Grader(typing.Protocol):
44
+ """Per-case grader. Scores are averaged across cases by the runner."""
45
+
46
+ name: str
47
+
48
+ def grade(
49
+ self, case: models.Case, output: models.Output
50
+ ) -> list[models.Score]: ...
51
+
52
+
53
+ @typing.runtime_checkable
54
+ class AggregateGrader(typing.Protocol):
55
+ """Whole-run grader for set-level metrics (P/R/F1, win-rate, ...)."""
56
+
57
+ name: str
58
+
59
+ def aggregate(
60
+ self, results: list[models.CaseResult]
61
+ ) -> list[models.Score]: ...
62
+
63
+
64
+ _REGISTRY: dict[str, type] = {}
65
+
66
+ _CATEGORIES: dict[str, GraderType] = {}
67
+
68
+
69
+ def register(
70
+ type_name: str, category: GraderType
71
+ ) -> typing.Callable[[type], type]:
72
+ """Class decorator registering a grader under a suite-config ``type``.
73
+
74
+ ``category`` is required rather than defaulting, because it is the only
75
+ source of the row's ``grader_type`` and a default would be the value
76
+ every grader forgets to override. A grader's category belongs to its
77
+ implementation, not to a suite's use of it, so it is declared here and
78
+ not in the suite config.
79
+
80
+ Args:
81
+ type_name: The ``type`` a suite spec names to select this grader.
82
+ category: What kind of check it performs.
83
+
84
+ Returns:
85
+ The decorator.
86
+
87
+ Raises:
88
+ ConfigError: If ``type_name`` is already registered.
89
+
90
+ """
91
+
92
+ def _decorate(cls: type) -> type:
93
+ if type_name in _REGISTRY:
94
+ raise ConfigError(f'grader type {type_name!r} already registered')
95
+ _REGISTRY[type_name] = cls
96
+ _CATEGORIES[type_name] = GraderType(category)
97
+ return cls
98
+
99
+ return _decorate
100
+
101
+
102
+ def category_of(type_name: str) -> GraderType:
103
+ """Return the category a grader type registered under.
104
+
105
+ Args:
106
+ type_name: The suite spec's ``type``.
107
+
108
+ Returns:
109
+ The declared category, or ``UNKNOWN`` for a type no plug-in has
110
+ registered. A suite naming one cannot run - ``build_graders``
111
+ raises - so ``UNKNOWN`` only reaches a row when a caller builds
112
+ rows without loading the plug-ins that produced them.
113
+
114
+ """
115
+ return _CATEGORIES.get(type_name, GraderType.UNKNOWN)
116
+
117
+
118
+ def build_graders(
119
+ specs: list[dict],
120
+ ) -> tuple[list[Grader], list[AggregateGrader]]:
121
+ """Instantiate grader specs, partitioned into per-case and aggregate.
122
+
123
+ Each spec's ``type`` selects a registered class; remaining keys (minus
124
+ ``type``) are passed as keyword arguments to its constructor.
125
+ """
126
+ per_case: list[Grader] = []
127
+ aggregate: list[AggregateGrader] = []
128
+ for spec in specs:
129
+ spec = dict(spec)
130
+ type_name = spec.pop('type')
131
+ if type_name not in _REGISTRY:
132
+ raise ConfigError(
133
+ f'unknown grader type {type_name!r}; '
134
+ f'known: {sorted(_REGISTRY)}'
135
+ )
136
+ grader = _REGISTRY[type_name](**spec)
137
+ if isinstance(grader, AggregateGrader):
138
+ aggregate.append(grader)
139
+ elif isinstance(grader, Grader):
140
+ per_case.append(grader)
141
+ else: # pragma: no cover - defensive
142
+ raise TypeError(
143
+ f'{type_name!r} is neither Grader nor AggregateGrader'
144
+ )
145
+ return per_case, aggregate
@@ -19,7 +19,7 @@ def _safe_div(numerator: float, denominator: float) -> float:
19
19
  return numerator / denominator if denominator else 0.0
20
20
 
21
21
 
22
- @base.register('classification')
22
+ @base.register('classification', base.GraderType.STATISTICAL)
23
23
  class Classification:
24
24
  """Binary precision/recall/F1 + FN/FP rates over a labeled dataset."""
25
25
 
@@ -32,7 +32,7 @@ def _score(name: str, metric: str, case_id: str, ok: bool, detail: str):
32
32
  )
33
33
 
34
34
 
35
- @base.register('max_chars')
35
+ @base.register('max_chars', base.GraderType.HEURISTIC)
36
36
  class MaxChars:
37
37
  """Assert a text field is at most ``maximum`` characters long."""
38
38
 
@@ -58,7 +58,7 @@ class MaxChars:
58
58
  ]
59
59
 
60
60
 
61
- @base.register('regex_absent')
61
+ @base.register('regex_absent', base.GraderType.HEURISTIC)
62
62
  class RegexAbsent:
63
63
  """Assert a text field does NOT match ``pattern`` (e.g. no tokens)."""
64
64
 
@@ -78,7 +78,7 @@ class RegexAbsent:
78
78
  return [_score(self.name, self.name, case.id, ok, detail)]
79
79
 
80
80
 
81
- @base.register('regex_present')
81
+ @base.register('regex_present', base.GraderType.HEURISTIC)
82
82
  class RegexPresent:
83
83
  """Assert a text field matches EVERY pattern in ``patterns`` (all-of).
84
84
 
@@ -111,7 +111,7 @@ class RegexPresent:
111
111
  return [_score(self.name, self.name, case.id, ok, detail)]
112
112
 
113
113
 
114
- @base.register('non_empty')
114
+ @base.register('non_empty', base.GraderType.HEURISTIC)
115
115
  class NonEmpty:
116
116
  """Assert a field resolves to a non-empty value."""
117
117
 
@@ -289,7 +289,7 @@ def _load_images(refs_values: list) -> list[dict]:
289
289
  return images
290
290
 
291
291
 
292
- @base.register('llm_judge')
292
+ @base.register('llm_judge', base.GraderType.LLM_AS_JUDGE)
293
293
  class RubricJudge:
294
294
  """Score an output's text on rubric dimensions with an LLM judge/panel.
295
295
 
@@ -48,7 +48,7 @@ def _as_float(value) -> float | None:
48
48
  return None
49
49
 
50
50
 
51
- @base.register('numeric')
51
+ @base.register('numeric', base.GraderType.HEURISTIC)
52
52
  class Numeric:
53
53
  """Surface numeric output fields as metrics, optionally range-checked."""
54
54