evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Built-in numeric (per-case) grader.
|
|
2
|
+
|
|
3
|
+
Promotes numeric ``Output`` fields onto the scorecard as metrics, so values an
|
|
4
|
+
adapter merely *extracted* (a cost, a tool-call error rate, a token count)
|
|
5
|
+
become first-class metrics the runner averages and ``compare``/guardrails can
|
|
6
|
+
gate on. Without this, an adapter's ``fields`` never reach a scorecard - only
|
|
7
|
+
``Score``s do.
|
|
8
|
+
|
|
9
|
+
Each configured field emits one per-case ``Score`` whose ``value`` is the field
|
|
10
|
+
coerced to float. Give a field a ``min``/``max`` bound and the score also
|
|
11
|
+
carries ``passed`` (in-range), turning it into a pass-rate; leave bounds off
|
|
12
|
+
and it's a pure measurement (``passed`` stays ``None``). Non-numeric or absent
|
|
13
|
+
fields degrade to ``value=None`` rather than raising.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from evalkit import models, refs
|
|
17
|
+
from evalkit.graders import base
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _context(case: models.Case, output: models.Output) -> dict:
|
|
21
|
+
return {
|
|
22
|
+
'input': case.input,
|
|
23
|
+
'expected': case.expected or {},
|
|
24
|
+
'output': output.fields,
|
|
25
|
+
'case': case.model_dump(),
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class _Field:
|
|
30
|
+
"""One resolved field spec: a ``$ref`` selector, a metric name, bounds."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, spec: str | dict):
|
|
33
|
+
if isinstance(spec, str):
|
|
34
|
+
spec = {'ref': spec}
|
|
35
|
+
self.ref = spec['ref']
|
|
36
|
+
self.minimum = spec.get('min')
|
|
37
|
+
self.maximum = spec.get('max')
|
|
38
|
+
# Metric name defaults to the ref's leaf (``output.cost`` -> ``cost``).
|
|
39
|
+
self.metric = spec.get('name') or self.ref.rsplit('.', 1)[-1]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _as_float(value) -> float | None:
|
|
43
|
+
if isinstance(value, bool) or value is None:
|
|
44
|
+
return None
|
|
45
|
+
try:
|
|
46
|
+
return float(value)
|
|
47
|
+
except TypeError, ValueError:
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@base.register('numeric')
|
|
52
|
+
class Numeric:
|
|
53
|
+
"""Surface numeric output fields as metrics, optionally range-checked."""
|
|
54
|
+
|
|
55
|
+
def __init__(self, fields: list[str | dict], name: str = 'numeric'):
|
|
56
|
+
self.name = name
|
|
57
|
+
self.fields = [_Field(spec) for spec in fields]
|
|
58
|
+
|
|
59
|
+
def grade(
|
|
60
|
+
self, case: models.Case, output: models.Output
|
|
61
|
+
) -> list[models.Score]:
|
|
62
|
+
context = _context(case, output)
|
|
63
|
+
scores: list[models.Score] = []
|
|
64
|
+
for field in self.fields:
|
|
65
|
+
value = _as_float(refs.resolve_ref(context, field.ref))
|
|
66
|
+
passed, detail = _check(value, field)
|
|
67
|
+
scores.append(
|
|
68
|
+
models.Score(
|
|
69
|
+
grader=self.name,
|
|
70
|
+
metric=field.metric,
|
|
71
|
+
value=value,
|
|
72
|
+
passed=passed,
|
|
73
|
+
detail=detail,
|
|
74
|
+
case_id=case.id,
|
|
75
|
+
kind='per_case',
|
|
76
|
+
)
|
|
77
|
+
)
|
|
78
|
+
return scores
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _check(value: float | None, field: _Field) -> tuple[bool | None, str]:
|
|
82
|
+
"""Range-check ``value`` against a field's bounds (``None`` if unbound)."""
|
|
83
|
+
bounded = field.minimum is not None or field.maximum is not None
|
|
84
|
+
if value is None:
|
|
85
|
+
return (False if bounded else None), 'absent/non-numeric'
|
|
86
|
+
if not bounded:
|
|
87
|
+
return None, f'{value:g}'
|
|
88
|
+
ok = (field.minimum is None or value >= field.minimum) and (
|
|
89
|
+
field.maximum is None or value <= field.maximum
|
|
90
|
+
)
|
|
91
|
+
return ok, f'{value:g}'
|
evalkit/loader.py
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Loading of suite configs and datasets.
|
|
2
|
+
|
|
3
|
+
Data files are YAML when PyYAML is available (nice for multi-line content
|
|
4
|
+
payloads) and fall back to JSON otherwise, so the engine has no hard non-stdlib
|
|
5
|
+
data dependency.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import hashlib
|
|
9
|
+
import json
|
|
10
|
+
import pathlib
|
|
11
|
+
|
|
12
|
+
import pydantic
|
|
13
|
+
|
|
14
|
+
from evalkit import models
|
|
15
|
+
from evalkit.retry import RetryConfig # re-exported: SuiteConfig.retry type
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def content_hash(data) -> str:
|
|
19
|
+
"""Stable short digest of parsed content (canonical JSON, sha256).
|
|
20
|
+
|
|
21
|
+
Versions content independently of any VCS: two loads hash equal iff
|
|
22
|
+
the parsed data is equal, regardless of file formatting or location.
|
|
23
|
+
"""
|
|
24
|
+
text = json.dumps(data, sort_keys=True, separators=(',', ':'))
|
|
25
|
+
return hashlib.sha256(text.encode('utf-8')).hexdigest()[:12]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def dataset_hash(cases: list[models.Case]) -> str:
|
|
29
|
+
"""Content digest of a loaded case list (order-independent)."""
|
|
30
|
+
dumped = sorted(
|
|
31
|
+
(case.model_dump(mode='json') for case in cases),
|
|
32
|
+
key=lambda data: str(data.get('id')),
|
|
33
|
+
)
|
|
34
|
+
return content_hash(dumped)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def load_data_file(path: str | pathlib.Path) -> dict:
|
|
38
|
+
"""Load a YAML or JSON mapping from ``path``."""
|
|
39
|
+
text = pathlib.Path(path).read_text(encoding='utf-8')
|
|
40
|
+
try:
|
|
41
|
+
import yaml
|
|
42
|
+
|
|
43
|
+
return yaml.safe_load(text) or {}
|
|
44
|
+
except ImportError:
|
|
45
|
+
return json.loads(text) if text.strip() else {}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class SuiteConfig(pydantic.BaseModel):
|
|
49
|
+
"""Parsed ``evals/suites/<name>.yaml`` for one consumer suite."""
|
|
50
|
+
|
|
51
|
+
project: str
|
|
52
|
+
suite: str
|
|
53
|
+
dataset: str
|
|
54
|
+
dataset_version: str = 'v1'
|
|
55
|
+
mode_default: str = 'http'
|
|
56
|
+
replay_fixtures: str | None = None
|
|
57
|
+
adapter: dict
|
|
58
|
+
graders: list[dict] = pydantic.Field(default_factory=list)
|
|
59
|
+
variants: dict[str, dict] = pydantic.Field(default_factory=dict)
|
|
60
|
+
n_samples: int = 1
|
|
61
|
+
# Max concurrent (case, sample) invocations. Above 1, the adapter and
|
|
62
|
+
# per-case graders must tolerate concurrent calls.
|
|
63
|
+
concurrency: int = 1
|
|
64
|
+
# Transient-failure retry policy for adapter calls (default: no retry).
|
|
65
|
+
retry: RetryConfig = pydantic.Field(default_factory=RetryConfig)
|
|
66
|
+
thresholds: dict = pydantic.Field(default_factory=dict)
|
|
67
|
+
# Computed by load_suite (content digest of the raw file); not for
|
|
68
|
+
# authors to set in the suite file.
|
|
69
|
+
suite_hash: str | None = None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def load_suite(path: str | pathlib.Path) -> SuiteConfig:
|
|
73
|
+
"""Load and validate a suite config file.
|
|
74
|
+
|
|
75
|
+
``dataset`` and ``replay_fixtures`` are resolved relative to the suite
|
|
76
|
+
file's own directory (unless already absolute), so a suite is portable and
|
|
77
|
+
runnable from any working directory.
|
|
78
|
+
"""
|
|
79
|
+
path = pathlib.Path(path)
|
|
80
|
+
raw = load_data_file(path)
|
|
81
|
+
config = SuiteConfig.model_validate(raw)
|
|
82
|
+
# Hash the raw content BEFORE resolving paths so the digest is stable
|
|
83
|
+
# across checkout locations.
|
|
84
|
+
config.suite_hash = content_hash(raw)
|
|
85
|
+
base = path.parent
|
|
86
|
+
|
|
87
|
+
def _resolve(value: str) -> str:
|
|
88
|
+
candidate = pathlib.Path(value)
|
|
89
|
+
return value if candidate.is_absolute() else str(base / candidate)
|
|
90
|
+
|
|
91
|
+
config.dataset = _resolve(config.dataset)
|
|
92
|
+
if config.replay_fixtures:
|
|
93
|
+
config.replay_fixtures = _resolve(config.replay_fixtures)
|
|
94
|
+
# Grader specs may carry their own suite-relative fixtures (e.g. the LLM
|
|
95
|
+
# judge's recorded judgments under ``replay_path``) - at the top level
|
|
96
|
+
# for a single judge, or per judge in a ``judges`` panel list.
|
|
97
|
+
for spec in config.graders:
|
|
98
|
+
if isinstance(spec.get('replay_path'), str):
|
|
99
|
+
spec['replay_path'] = _resolve(spec['replay_path'])
|
|
100
|
+
for judge in spec.get('judges', []) or []:
|
|
101
|
+
if isinstance(judge.get('replay_path'), str):
|
|
102
|
+
judge['replay_path'] = _resolve(judge['replay_path'])
|
|
103
|
+
# The pairwise judge's recorded judgments are suite-relative too.
|
|
104
|
+
pairwise = config.thresholds.get('pairwise')
|
|
105
|
+
if isinstance(pairwise, dict) and isinstance(
|
|
106
|
+
pairwise.get('replay_path'), str
|
|
107
|
+
):
|
|
108
|
+
pairwise['replay_path'] = _resolve(pairwise['replay_path'])
|
|
109
|
+
return config
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def load_cases(dataset_dir: str | pathlib.Path) -> list[models.Case]:
|
|
113
|
+
"""Load every ``cases/*.yaml|json`` file under a dataset directory."""
|
|
114
|
+
cases_dir = pathlib.Path(dataset_dir) / 'cases'
|
|
115
|
+
if not cases_dir.is_dir():
|
|
116
|
+
raise FileNotFoundError(f'no cases directory at {cases_dir}')
|
|
117
|
+
cases: list[models.Case] = []
|
|
118
|
+
paths = sorted(
|
|
119
|
+
p
|
|
120
|
+
for p in cases_dir.iterdir()
|
|
121
|
+
if p.suffix in ('.yaml', '.yml', '.json')
|
|
122
|
+
)
|
|
123
|
+
for path in paths:
|
|
124
|
+
data = load_data_file(path)
|
|
125
|
+
data.setdefault('id', path.stem)
|
|
126
|
+
cases.append(models.Case.model_validate(data))
|
|
127
|
+
if not cases:
|
|
128
|
+
raise FileNotFoundError(f'no case files found in {cases_dir}')
|
|
129
|
+
return cases
|
evalkit/models.py
ADDED
|
@@ -0,0 +1,390 @@
|
|
|
1
|
+
"""Core data model for the generic eval engine.
|
|
2
|
+
|
|
3
|
+
None of these types reference any particular consumer. ``Case.input``,
|
|
4
|
+
``Case.expected``, and ``Variant.knobs`` are opaque blobs interpreted only by
|
|
5
|
+
adapters and graders - that opacity is what keeps the engine generic.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import typing
|
|
9
|
+
|
|
10
|
+
import pydantic
|
|
11
|
+
|
|
12
|
+
Json = typing.Any
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class Case(pydantic.BaseModel):
|
|
16
|
+
"""One eval input. ``input``/``expected`` are consumer-defined blobs."""
|
|
17
|
+
|
|
18
|
+
id: str
|
|
19
|
+
input: dict[str, Json] = pydantic.Field(default_factory=dict)
|
|
20
|
+
expected: dict[str, Json] | None = None
|
|
21
|
+
labels: dict[str, Json] = pydantic.Field(default_factory=dict)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Variant(pydantic.BaseModel):
|
|
25
|
+
"""A configuration under test (e.g. baseline vs candidate).
|
|
26
|
+
|
|
27
|
+
``knobs`` is opaque to the engine; the adapter understands it (commonly
|
|
28
|
+
``{model, prompt_version, env}``).
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
name: str
|
|
32
|
+
knobs: dict[str, Json] = pydantic.Field(default_factory=dict)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Output(pydantic.BaseModel):
|
|
36
|
+
"""The system-under-test's response for one case, normalized.
|
|
37
|
+
|
|
38
|
+
``fields`` holds the values the adapter extracted (graders read these via
|
|
39
|
+
``output.<field>`` references). ``error`` is set when the invocation
|
|
40
|
+
failed; graders/aggregators treat errored outputs explicitly rather than
|
|
41
|
+
silently scoring them.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
fields: dict[str, Json] = pydantic.Field(default_factory=dict)
|
|
45
|
+
raw: Json = None
|
|
46
|
+
error: str | None = None
|
|
47
|
+
#: Set by the adapter alongside ``error`` when the failure is *transient*
|
|
48
|
+
#: (a 429, a 5xx, a network timeout) and worth retrying; the runner's
|
|
49
|
+
#: retry loop acts on this. A terminal error (bad request, non-JSON body)
|
|
50
|
+
#: leaves it False so the run doesn't waste attempts on it.
|
|
51
|
+
retryable: bool = False
|
|
52
|
+
latency_ms: float | None = None
|
|
53
|
+
tokens: dict[str, int] | None = None
|
|
54
|
+
cost: float | None = None
|
|
55
|
+
#: Named files the adapter saved for this invocation (screenshots,
|
|
56
|
+
#: rendered HTML, transcripts): name -> filesystem path. Persisted
|
|
57
|
+
#: with the run so downstream review/rating tools can load them.
|
|
58
|
+
artifacts: dict[str, str] = pydantic.Field(default_factory=dict)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class JudgeDetail(pydantic.BaseModel):
|
|
62
|
+
"""One judge's full verdict on a case, retained for review.
|
|
63
|
+
|
|
64
|
+
The runner folds ``points``/``overall`` into the aggregate means; this
|
|
65
|
+
keeps the per-judge breakdown the aggregate hides - the raw 1..scale
|
|
66
|
+
point each dimension got and the free-text ``rationale`` - so a low
|
|
67
|
+
score can be read back to *why* without re-running the judge.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
key: str
|
|
71
|
+
version: str | None = None
|
|
72
|
+
rationale: str | None = None
|
|
73
|
+
#: raw per-dimension points as the judge returned them (1..scale), not
|
|
74
|
+
#: normalized; ``None`` for a dimension the judge did not score.
|
|
75
|
+
points: dict[str, float | None] = pydantic.Field(default_factory=dict)
|
|
76
|
+
#: this judge's own normalized (0..1) overall, so a generous judge shows.
|
|
77
|
+
overall: float | None = None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Score(pydantic.BaseModel):
|
|
81
|
+
"""A single metric emitted by a grader for one case (or aggregate).
|
|
82
|
+
|
|
83
|
+
``kind='per_case'`` scores are averaged across cases by the runner;
|
|
84
|
+
``kind='aggregate'`` scores are computed once over the whole run (e.g.
|
|
85
|
+
precision/recall) and stored as-is.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
grader: str
|
|
89
|
+
metric: str
|
|
90
|
+
value: float | None = None
|
|
91
|
+
passed: bool | None = None
|
|
92
|
+
detail: str | None = None
|
|
93
|
+
case_id: str | None = None
|
|
94
|
+
kind: typing.Literal['per_case', 'aggregate'] = 'per_case'
|
|
95
|
+
#: Populated only on an LLM-judge ``<name>.overall`` score: the per-judge
|
|
96
|
+
#: breakdown (rationale + raw points) behind the aggregated value, so the
|
|
97
|
+
#: store and reporters can surface it. Empty for every other grader.
|
|
98
|
+
judges: list[JudgeDetail] = pydantic.Field(default_factory=list)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class CaseResult(pydantic.BaseModel):
|
|
102
|
+
"""The output + per-case scores for one (case, sample)."""
|
|
103
|
+
|
|
104
|
+
case: Case
|
|
105
|
+
variant_name: str
|
|
106
|
+
sample_idx: int
|
|
107
|
+
output: Output
|
|
108
|
+
scores: list[Score] = pydantic.Field(default_factory=list)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class MetricValue(pydantic.BaseModel):
|
|
112
|
+
"""An aggregated metric on a scorecard.
|
|
113
|
+
|
|
114
|
+
``stdev`` is populated for ``mean`` metrics with 2+ observations
|
|
115
|
+
(e.g. ``n_samples > 1``) so repeat runs carry their spread, not just
|
|
116
|
+
the average.
|
|
117
|
+
"""
|
|
118
|
+
|
|
119
|
+
metric: str
|
|
120
|
+
value: float | None
|
|
121
|
+
kind: typing.Literal['mean', 'aggregate']
|
|
122
|
+
n: int
|
|
123
|
+
stdev: float | None = None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class Scorecard(pydantic.BaseModel):
|
|
127
|
+
"""Aggregated result of running one suite x one variant over a dataset.
|
|
128
|
+
|
|
129
|
+
The key tuple (project, suite, variant, dataset_version, model_id,
|
|
130
|
+
prompt_version, judge_version, revision) makes a scorecard reproducible
|
|
131
|
+
and is exactly the multi-tenant key used by the results store.
|
|
132
|
+
|
|
133
|
+
``revision`` is an opaque consumer-supplied provenance id (a git SHA,
|
|
134
|
+
image digest, package version, release label, ...) - the engine never
|
|
135
|
+
interprets it. ``suite_hash``/``dataset_hash`` are engine-computed
|
|
136
|
+
content digests of the loaded suite config and cases: declared versions
|
|
137
|
+
state intent, the hashes prove the content actually matched.
|
|
138
|
+
"""
|
|
139
|
+
|
|
140
|
+
run_id: str | None = None
|
|
141
|
+
project: str
|
|
142
|
+
suite: str
|
|
143
|
+
variant: Variant
|
|
144
|
+
dataset_version: str
|
|
145
|
+
model_id: str | None = None
|
|
146
|
+
prompt_version: str | None = None
|
|
147
|
+
judge_version: str | None = None
|
|
148
|
+
revision: str | None = None
|
|
149
|
+
suite_hash: str | None = None
|
|
150
|
+
dataset_hash: str | None = None
|
|
151
|
+
mode: str = 'http'
|
|
152
|
+
n_samples: int = 1
|
|
153
|
+
n_cases: int = 0
|
|
154
|
+
created_at: str | None = None
|
|
155
|
+
metrics: dict[str, MetricValue] = pydantic.Field(default_factory=dict)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class RunResult(pydantic.BaseModel):
|
|
159
|
+
"""Everything one run produced: the scorecard plus per-sample results.
|
|
160
|
+
|
|
161
|
+
``results`` is the ground truth the scorecard was aggregated from -
|
|
162
|
+
one entry per (case, sample) with the full output, artifacts, and
|
|
163
|
+
per-case scores. Persisting it is what makes transcript review,
|
|
164
|
+
human rating, and judge-agreement analysis possible after the fact.
|
|
165
|
+
"""
|
|
166
|
+
|
|
167
|
+
run_id: str
|
|
168
|
+
scorecard: Scorecard
|
|
169
|
+
results: list[CaseResult] = pydantic.Field(default_factory=list)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class Rating(pydantic.BaseModel):
|
|
173
|
+
"""One human's blind rating of a single (case, sample) output.
|
|
174
|
+
|
|
175
|
+
The open interchange format: emitted by the ``rate`` web app and
|
|
176
|
+
ingestible from any external tool. ``scores`` maps rubric dimension ->
|
|
177
|
+
an integer on the same 1..scale the judge used, so human and judge are
|
|
178
|
+
directly comparable.
|
|
179
|
+
"""
|
|
180
|
+
|
|
181
|
+
run_id: str
|
|
182
|
+
case_id: str
|
|
183
|
+
sample_idx: int = 0
|
|
184
|
+
rater: str
|
|
185
|
+
scores: dict[str, int] = pydantic.Field(default_factory=dict)
|
|
186
|
+
rated_at: str | None = None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
class Preference(pydantic.BaseModel):
|
|
190
|
+
"""One human's blind side-by-side preference between two variants' outputs
|
|
191
|
+
for a single (case, sample) - the A-vs-B analog of :class:`Rating`.
|
|
192
|
+
|
|
193
|
+
The open interchange format for the ``rank`` web app (one JSON object per
|
|
194
|
+
line). ``winner`` is the overall pick and ``dims`` the per-dimension picks,
|
|
195
|
+
each ``'a'`` / ``'b'`` / ``'tie'`` in *variant* terms: the app
|
|
196
|
+
counterbalances left/right per rater and un-blinds server-side, so a
|
|
197
|
+
stored ``'a'`` always means ``variant_a`` won regardless of which side it
|
|
198
|
+
was shown on.
|
|
199
|
+
"""
|
|
200
|
+
|
|
201
|
+
case_id: str
|
|
202
|
+
sample_idx: int = 0
|
|
203
|
+
variant_a: str
|
|
204
|
+
variant_b: str
|
|
205
|
+
rater: str
|
|
206
|
+
winner: typing.Literal['a', 'b', 'tie'] = 'tie'
|
|
207
|
+
dims: dict[str, typing.Literal['a', 'b', 'tie']] = pydantic.Field(
|
|
208
|
+
default_factory=dict
|
|
209
|
+
)
|
|
210
|
+
rated_at: str | None = None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
class DimensionAgreement(pydantic.BaseModel):
|
|
214
|
+
"""Judge-vs-human agreement on one rubric dimension."""
|
|
215
|
+
|
|
216
|
+
dimension: str
|
|
217
|
+
n: int
|
|
218
|
+
human_mean: float | None = None
|
|
219
|
+
judge_mean: float | None = None
|
|
220
|
+
mae: float | None = None
|
|
221
|
+
correlation: float | None = None
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
class AgreementResult(pydantic.BaseModel):
|
|
225
|
+
"""How well the judge tracks human raters over a run.
|
|
226
|
+
|
|
227
|
+
The calibration gate: a judge whose scores don't agree with human
|
|
228
|
+
ratings isn't trustworthy as a win metric yet. All values compare the
|
|
229
|
+
per-(case,sample) human mean against the judge's normalized 0..1 score.
|
|
230
|
+
"""
|
|
231
|
+
|
|
232
|
+
judge_name: str
|
|
233
|
+
scale: int
|
|
234
|
+
n_ratings: int
|
|
235
|
+
n_raters: int
|
|
236
|
+
dimensions: list[DimensionAgreement] = pydantic.Field(default_factory=list)
|
|
237
|
+
overall_mae: float | None = None
|
|
238
|
+
overall_correlation: float | None = None
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
class SweepEntry(pydantic.BaseModel):
|
|
242
|
+
"""One variant's standing in a sweep, ranked by the win metric."""
|
|
243
|
+
|
|
244
|
+
variant: str
|
|
245
|
+
win_value: float | None
|
|
246
|
+
rank: int
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class SweepResult(pydantic.BaseModel):
|
|
250
|
+
"""N-way comparison of a suite's variants (a model x prompt matrix).
|
|
251
|
+
|
|
252
|
+
``entries`` ranks the variants by the configured win metric;
|
|
253
|
+
``matrix`` is metric -> {variant: value} for the full leaderboard.
|
|
254
|
+
"""
|
|
255
|
+
|
|
256
|
+
project: str
|
|
257
|
+
suite: str
|
|
258
|
+
win_metric: str | None = None
|
|
259
|
+
win_higher_is_better: bool = True
|
|
260
|
+
entries: list[SweepEntry] = pydantic.Field(default_factory=list)
|
|
261
|
+
matrix: dict[str, dict[str, float | None]] = pydantic.Field(
|
|
262
|
+
default_factory=dict
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
class PairwiseOutcome(pydantic.BaseModel):
|
|
267
|
+
"""The head-to-head result for one case (counterbalanced for order)."""
|
|
268
|
+
|
|
269
|
+
case_id: str
|
|
270
|
+
sample_idx: int = 0
|
|
271
|
+
winner: typing.Literal['a', 'b', 'tie'] = 'tie'
|
|
272
|
+
detail: str | None = None
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
class PairwiseResult(pydantic.BaseModel):
|
|
276
|
+
"""A-vs-B win-rate from a judge comparing two variants head-to-head.
|
|
277
|
+
|
|
278
|
+
The headline subjective-regression signal: for each case the judge is
|
|
279
|
+
shown both variants' outputs and picks a winner (order counterbalanced,
|
|
280
|
+
so a position-biased flip becomes a tie). ``win_rate_a`` counts ties as
|
|
281
|
+
half, the standard convention.
|
|
282
|
+
"""
|
|
283
|
+
|
|
284
|
+
project: str
|
|
285
|
+
suite: str
|
|
286
|
+
variant_a: str
|
|
287
|
+
variant_b: str
|
|
288
|
+
judge_name: str = 'pairwise'
|
|
289
|
+
judge_version: str = 'v1'
|
|
290
|
+
n: int = 0
|
|
291
|
+
a_wins: int = 0
|
|
292
|
+
b_wins: int = 0
|
|
293
|
+
ties: int = 0
|
|
294
|
+
win_rate_a: float | None = None
|
|
295
|
+
outcomes: list[PairwiseOutcome] = pydantic.Field(default_factory=list)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
class DimensionPreference(pydantic.BaseModel):
|
|
299
|
+
"""Human A-vs-B win-rate on one rubric dimension (ties count as half)."""
|
|
300
|
+
|
|
301
|
+
dimension: str
|
|
302
|
+
n: int = 0
|
|
303
|
+
a_wins: int = 0
|
|
304
|
+
b_wins: int = 0
|
|
305
|
+
ties: int = 0
|
|
306
|
+
win_rate_a: float | None = None
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
class PreferenceResult(pydantic.BaseModel):
|
|
310
|
+
"""Human side-by-side win-rate of A vs B - the human analog of
|
|
311
|
+
:class:`PairwiseResult`.
|
|
312
|
+
|
|
313
|
+
Reports an overall win-rate plus a per-dimension breakdown; ``win_rate_a``
|
|
314
|
+
counts ties as half, the standard convention the pairwise judge uses.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
project: str
|
|
318
|
+
suite: str
|
|
319
|
+
variant_a: str
|
|
320
|
+
variant_b: str
|
|
321
|
+
n_raters: int = 0
|
|
322
|
+
n: int = 0
|
|
323
|
+
a_wins: int = 0
|
|
324
|
+
b_wins: int = 0
|
|
325
|
+
ties: int = 0
|
|
326
|
+
win_rate_a: float | None = None
|
|
327
|
+
dimensions: list[DimensionPreference] = pydantic.Field(
|
|
328
|
+
default_factory=list
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
class PairwiseAgreementCase(pydantic.BaseModel):
|
|
333
|
+
"""Human-panel vs LLM-judge winner for one case."""
|
|
334
|
+
|
|
335
|
+
case_id: str
|
|
336
|
+
sample_idx: int = 0
|
|
337
|
+
human: typing.Literal['a', 'b', 'tie'] = 'tie'
|
|
338
|
+
judge: typing.Literal['a', 'b', 'tie'] = 'tie'
|
|
339
|
+
agree: bool = False
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class PairwiseAgreement(pydantic.BaseModel):
|
|
343
|
+
"""How often the human panel and the LLM pairwise judge pick the same
|
|
344
|
+
per-case winner - the head-to-head calibration gate (the A-vs-B analog of
|
|
345
|
+
:class:`AgreementResult`).
|
|
346
|
+
"""
|
|
347
|
+
|
|
348
|
+
variant_a: str
|
|
349
|
+
variant_b: str
|
|
350
|
+
judge_name: str = 'pairwise'
|
|
351
|
+
n: int = 0
|
|
352
|
+
agree: int = 0
|
|
353
|
+
human_win_rate_a: float | None = None
|
|
354
|
+
judge_win_rate_a: float | None = None
|
|
355
|
+
agreement_rate: float | None = None
|
|
356
|
+
outcomes: list[PairwiseAgreementCase] = pydantic.Field(
|
|
357
|
+
default_factory=list
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
class MetricDelta(pydantic.BaseModel):
|
|
362
|
+
"""Per-metric baseline->candidate comparison."""
|
|
363
|
+
|
|
364
|
+
metric: str
|
|
365
|
+
baseline: float | None
|
|
366
|
+
candidate: float | None
|
|
367
|
+
delta: float | None
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
class GuardrailResult(pydantic.BaseModel):
|
|
371
|
+
"""Outcome of one guardrail check against the candidate."""
|
|
372
|
+
|
|
373
|
+
metric: str
|
|
374
|
+
passed: bool
|
|
375
|
+
detail: str
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
class Comparison(pydantic.BaseModel):
|
|
379
|
+
"""Candidate-vs-baseline comparison and gate verdict."""
|
|
380
|
+
|
|
381
|
+
project: str
|
|
382
|
+
suite: str
|
|
383
|
+
baseline_variant: str
|
|
384
|
+
candidate_variant: str
|
|
385
|
+
win_metric: str | None = None
|
|
386
|
+
win: typing.Literal['improved', 'regressed', 'neutral'] = 'neutral'
|
|
387
|
+
verdict: typing.Literal['pass', 'warn', 'fail'] = 'pass'
|
|
388
|
+
deltas: list[MetricDelta] = pydantic.Field(default_factory=list)
|
|
389
|
+
guardrails: list[GuardrailResult] = pydantic.Field(default_factory=list)
|
|
390
|
+
summary: str = ''
|