evalcore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evalkit/refs.py ADDED
@@ -0,0 +1,71 @@
1
+ """Reference resolution shared by adapters and graders.
2
+
3
+ Cases, variants, and outputs are *opaque* to the engine - only adapters and
4
+ graders crack them open, and they do so through string references like
5
+ ``$input.content`` (adapter request bodies) or ``output.verdict`` (grader
6
+ field selectors). Keeping this resolution in one place is what lets the core
7
+ stay ignorant of any consumer's data shape.
8
+ """
9
+
10
+ import typing
11
+
12
+ _MISSING = object()
13
+
14
+
15
+ def resolve_path(obj: typing.Any, path: str) -> typing.Any:
16
+ """Traverse ``obj`` along a dotted ``path``.
17
+
18
+ Supports dict keys, list indices (numeric segments), and object
19
+ attributes, in that order of preference. Returns ``None`` when any
20
+ segment is missing rather than raising, so a grader/adapter selector
21
+ that points at an absent field degrades to ``None``.
22
+ """
23
+ current = obj
24
+ for segment in path.split('.'):
25
+ if current is None:
26
+ return None
27
+ if isinstance(current, dict):
28
+ current = current.get(segment, _MISSING)
29
+ elif (
30
+ isinstance(current, (list, tuple))
31
+ and segment.lstrip('-').isdigit()
32
+ ):
33
+ index = int(segment)
34
+ if -len(current) <= index < len(current):
35
+ current = current[index]
36
+ else:
37
+ current = _MISSING
38
+ else:
39
+ current = getattr(current, segment, _MISSING)
40
+ if current is _MISSING:
41
+ return None
42
+ return current
43
+
44
+
45
+ def resolve_ref(context: dict, ref: str) -> typing.Any:
46
+ """Resolve a ``root.path`` reference against a context dict.
47
+
48
+ ``context`` maps root names (``input``, ``expected``, ``variant``,
49
+ ``output``, ``case``) to the corresponding objects.
50
+ """
51
+ root, _, rest = ref.partition('.')
52
+ if root not in context:
53
+ return None
54
+ base = context[root]
55
+ return resolve_path(base, rest) if rest else base
56
+
57
+
58
+ def build_value(template: typing.Any, context: dict) -> typing.Any:
59
+ """Render a (possibly nested) template, substituting ``$ref`` strings.
60
+
61
+ A string beginning with ``$`` is treated as a reference (``$input.foo``)
62
+ and replaced with the resolved value. Any other value - including dicts
63
+ and lists, which are walked recursively - passes through literally.
64
+ """
65
+ if isinstance(template, str) and template.startswith('$'):
66
+ return resolve_ref(context, template[1:])
67
+ if isinstance(template, dict):
68
+ return {k: build_value(v, context) for k, v in template.items()}
69
+ if isinstance(template, list):
70
+ return [build_value(v, context) for v in template]
71
+ return template
evalkit/report.py ADDED
@@ -0,0 +1,186 @@
1
+ """Markdown rendering of scorecards and comparisons (CI / PR comments)."""
2
+
3
+ from evalkit import models
4
+
5
+
6
+ def _fmt(value: float | None) -> str:
7
+ return 'n/a' if value is None else f'{value:.4f}'
8
+
9
+
10
+ def _fmt_delta(value: float | None) -> str:
11
+ return 'n/a' if value is None else f'{value:+.4f}'
12
+
13
+
14
+ def render_scorecard(scorecard: models.Scorecard) -> str:
15
+ """Render a single scorecard as a Markdown section."""
16
+ lines = [
17
+ f'### {scorecard.project}/{scorecard.suite} '
18
+ f'- `{scorecard.variant.name}`',
19
+ '',
20
+ f'- model: `{scorecard.model_id}` '
21
+ f'mode: `{scorecard.mode}` '
22
+ f'dataset: `{scorecard.dataset_version}` '
23
+ f'cases: {scorecard.n_cases}x{scorecard.n_samples}',
24
+ '',
25
+ '| metric | value |',
26
+ '| --- | ---: |',
27
+ ]
28
+ lines += [
29
+ f'| {m.metric} | {_fmt(m.value)} |' for m in scorecard.metrics.values()
30
+ ]
31
+ return '\n'.join(lines)
32
+
33
+
34
+ _VERDICT_BADGE = {'pass': '**PASS**', 'warn': '**WARN**', 'fail': '**FAIL**'}
35
+
36
+
37
+ def render_comparison(comparison: models.Comparison) -> str:
38
+ """Render a candidate-vs-baseline comparison as Markdown."""
39
+ badge = _VERDICT_BADGE.get(comparison.verdict, comparison.verdict)
40
+ lines = [
41
+ f'## {badge} - {comparison.project}/{comparison.suite}',
42
+ '',
43
+ f'`{comparison.candidate_variant}` vs '
44
+ f'`{comparison.baseline_variant}` - {comparison.summary}',
45
+ '',
46
+ '| metric | baseline | candidate | delta |',
47
+ '| --- | ---: | ---: | ---: |',
48
+ ]
49
+ for delta in comparison.deltas:
50
+ marker = (
51
+ f' **({comparison.win})**'
52
+ if delta.metric == comparison.win_metric
53
+ else ''
54
+ )
55
+ lines.append(
56
+ f'| {delta.metric}{marker} | {_fmt(delta.baseline)} '
57
+ f'| {_fmt(delta.candidate)} | {_fmt_delta(delta.delta)} |'
58
+ )
59
+ if comparison.guardrails:
60
+ lines += ['', '**Guardrails**', '']
61
+ for guard in comparison.guardrails:
62
+ mark = 'ok' if guard.passed else 'BREACH'
63
+ lines.append(f'- [{mark}] `{guard.metric}` - {guard.detail}')
64
+ return '\n'.join(lines)
65
+
66
+
67
+ def render_agreement(result: models.AgreementResult) -> str:
68
+ """Render a judge<->human agreement result as Markdown."""
69
+ lines = [
70
+ f'### judge<->human agreement - `{result.judge_name}`',
71
+ '',
72
+ f'- {result.n_ratings} ratings from {result.n_raters} rater(s), '
73
+ f'scale 1..{result.scale}',
74
+ f'- overall: MAE {_fmt(result.overall_mae)}, '
75
+ f'r {_fmt(result.overall_correlation)}',
76
+ '',
77
+ '| dimension | n | human | judge | MAE | corr |',
78
+ '| --- | ---: | ---: | ---: | ---: | ---: |',
79
+ ]
80
+ for dim in result.dimensions:
81
+ lines.append(
82
+ f'| {dim.dimension} | {dim.n} | {_fmt(dim.human_mean)} '
83
+ f'| {_fmt(dim.judge_mean)} | {_fmt(dim.mae)} '
84
+ f'| {_fmt(dim.correlation)} |'
85
+ )
86
+ return '\n'.join(lines)
87
+
88
+
89
+ def render_sweep(result: models.SweepResult) -> str:
90
+ """Render an N-way sweep as a ranking + a metric x variant leaderboard."""
91
+ variants = [c.variant for c in result.entries] or sorted(
92
+ {v for row in result.matrix.values() for v in row}
93
+ )
94
+ lines = [f'## sweep - {result.project}/{result.suite}', '']
95
+ if result.win_metric:
96
+ direction = 'higher' if result.win_higher_is_better else 'lower'
97
+ lines += [
98
+ f'ranked by `{result.win_metric}` ({direction} is better)',
99
+ '',
100
+ '| rank | variant | win |',
101
+ '| ---: | --- | ---: |',
102
+ ]
103
+ lines += [
104
+ f'| {e.rank} | `{e.variant}` | {_fmt(e.win_value)} |'
105
+ for e in result.entries
106
+ ]
107
+ lines.append('')
108
+
109
+ header = '| metric | ' + ' | '.join(variants) + ' |'
110
+ sep = '| --- |' + ' ---: |' * len(variants)
111
+ lines += ['**Leaderboard**', '', header, sep]
112
+ for metric, row in result.matrix.items():
113
+ present = [v for v in row.values() if v is not None]
114
+ best = max(present) if present else None # display hint only
115
+ cells = []
116
+ for variant in variants:
117
+ value = row.get(variant)
118
+ text = _fmt(value)
119
+ if value is not None and best is not None and value == best:
120
+ text = f'**{text}**'
121
+ cells.append(text)
122
+ lines.append(f'| {metric} | ' + ' | '.join(cells) + ' |')
123
+ return '\n'.join(lines)
124
+
125
+
126
+ def render_pairwise(result: models.PairwiseResult) -> str:
127
+ """Render an A-vs-B pairwise win-rate result as Markdown."""
128
+ rate = _fmt(result.win_rate_a)
129
+ return '\n'.join(
130
+ [
131
+ f'## pairwise - {result.project}/{result.suite}',
132
+ '',
133
+ f'`{result.variant_a}` (A) vs `{result.variant_b}` (B), '
134
+ f'judge `{result.judge_name}@{result.judge_version}`',
135
+ '',
136
+ f'- **A win-rate: {rate}** (ties = half) over {result.n} cases',
137
+ f'- A wins {result.a_wins} - B wins {result.b_wins} '
138
+ f'- ties {result.ties}',
139
+ ]
140
+ )
141
+
142
+
143
+ def render_preferences(result: models.PreferenceResult) -> str:
144
+ """Render a human side-by-side win-rate (overall + per dimension)."""
145
+ lines = [
146
+ f'## human preferences - {result.project}/{result.suite}',
147
+ '',
148
+ f'`{result.variant_a}` (A) vs `{result.variant_b}` (B), '
149
+ f'{result.n} preferences from {result.n_raters} rater(s)',
150
+ '',
151
+ f'- **A win-rate: {_fmt(result.win_rate_a)}** (ties = half) '
152
+ f'- A wins {result.a_wins} - B wins {result.b_wins} '
153
+ f'- ties {result.ties}',
154
+ '',
155
+ '| dimension | n | A wins | B wins | ties | A win-rate |',
156
+ '| --- | ---: | ---: | ---: | ---: | ---: |',
157
+ ]
158
+ for dim in result.dimensions:
159
+ lines.append(
160
+ f'| {dim.dimension} | {dim.n} | {dim.a_wins} | {dim.b_wins} '
161
+ f'| {dim.ties} | {_fmt(dim.win_rate_a)} |'
162
+ )
163
+ return '\n'.join(lines)
164
+
165
+
166
+ def render_pairwise_agreement(result: models.PairwiseAgreement) -> str:
167
+ """Render human-panel vs LLM-judge head-to-head agreement as Markdown."""
168
+ lines = [
169
+ f'## pairwise agreement - human vs `{result.judge_name}`',
170
+ '',
171
+ f'`{result.variant_a}` (A) vs `{result.variant_b}` (B)',
172
+ '',
173
+ f'- **agreement: {_fmt(result.agreement_rate)}** '
174
+ f'({result.agree}/{result.n} cases pick the same winner)',
175
+ f'- A win-rate: human {_fmt(result.human_win_rate_a)} '
176
+ f'- judge {_fmt(result.judge_win_rate_a)}',
177
+ '',
178
+ '| case | human | judge | agree |',
179
+ '| --- | :---: | :---: | :---: |',
180
+ ]
181
+ for case in result.outcomes:
182
+ mark = 'yes' if case.agree else 'NO'
183
+ lines.append(
184
+ f'| {case.case_id} | {case.human} | {case.judge} | {mark} |'
185
+ )
186
+ return '\n'.join(lines)
@@ -0,0 +1,33 @@
1
+ """Report rendering, pluggable by ``type`` like adapters and graders.
2
+
3
+ Importing this package registers the built-in ``markdown`` and ``html``
4
+ reporters. Select one with :func:`build_reporter` (the CLI's ``--report``
5
+ flag), and register more with ``@evalkit.reporters.base.register(...)``.
6
+ """
7
+
8
+ from evalkit.reporters import html, markdown # noqa: F401 - register built-ins
9
+ from evalkit.reporters.base import (
10
+ Reporter,
11
+ build_reporter,
12
+ per_case_matrix,
13
+ register,
14
+ render_agreement,
15
+ render_pairwise_agreement,
16
+ render_preferences,
17
+ render_run,
18
+ run_body,
19
+ wrap_document,
20
+ )
21
+
22
+ __all__ = [
23
+ 'Reporter',
24
+ 'build_reporter',
25
+ 'per_case_matrix',
26
+ 'register',
27
+ 'render_agreement',
28
+ 'render_pairwise_agreement',
29
+ 'render_preferences',
30
+ 'render_run',
31
+ 'run_body',
32
+ 'wrap_document',
33
+ ]
@@ -0,0 +1,159 @@
1
+ """Reporter protocol and registry (the report-rendering seam).
2
+
3
+ A reporter turns engine results into a rendered report. Two report kinds are
4
+ first-class, mirroring the two analyses the engine produces:
5
+
6
+ * ``scorecard`` - a SINGLE analysis (one suite x one variant run);
7
+ * ``comparison`` - a COMPARATIVE analysis (candidate vs baseline).
8
+
9
+ Built-ins ``markdown`` and ``html`` cover both. Register more the same way as
10
+ adapters and graders - ``@reporters.base.register('pdf')`` + ``--plugins`` -
11
+ so a consumer can add a format without touching the core.
12
+
13
+ Three further kinds are optional; a reporter is discovered to support each
14
+ via ``getattr`` and only implements the ones it cares about:
15
+
16
+ * ``run`` - a scorecard plus its per-case drill-down (see :func:`render_run`);
17
+ * ``agreement`` - a judge<->human calibration table (see
18
+ :func:`render_agreement`);
19
+ * ``document`` - wrap a rendered body in a standalone document (e.g. an HTML
20
+ page). Reporters that omit it are treated as identity.
21
+
22
+ Every ``*`` method returns a *fragment*; wrapping into a standalone document
23
+ happens once, at the edge, so fragments compose (a ``run`` body and an
24
+ ``agreement`` table can share one page).
25
+ """
26
+
27
+ import typing
28
+
29
+ from evalkit import models
30
+
31
+
32
+ @typing.runtime_checkable
33
+ class Reporter(typing.Protocol):
34
+ """Render a single scorecard and a candidate-vs-baseline comparison."""
35
+
36
+ def scorecard(self, scorecard: models.Scorecard) -> str: ...
37
+
38
+ def comparison(self, comparison: models.Comparison) -> str: ...
39
+
40
+
41
+ _REGISTRY: dict[str, type] = {}
42
+
43
+
44
+ def register(type_name: str) -> typing.Callable[[type], type]:
45
+ """Class decorator registering a reporter under a report ``type``."""
46
+
47
+ def _decorate(cls: type) -> type:
48
+ if type_name in _REGISTRY:
49
+ raise ValueError(f'report type {type_name!r} already registered')
50
+ _REGISTRY[type_name] = cls
51
+ return cls
52
+
53
+ return _decorate
54
+
55
+
56
+ def build_reporter(spec: str | dict) -> Reporter:
57
+ """Instantiate a reporter from a name or a ``{type, ...}`` config spec."""
58
+ spec = {'type': spec} if isinstance(spec, str) else dict(spec)
59
+ type_name = spec.pop('type')
60
+ if type_name not in _REGISTRY:
61
+ raise ValueError(
62
+ f'unknown report type {type_name!r}; known: {sorted(_REGISTRY)}'
63
+ )
64
+ return _REGISTRY[type_name](**spec)
65
+
66
+
67
+ def wrap_document(reporter: Reporter, body: str) -> str:
68
+ """Wrap ``body`` with the reporter's optional ``document`` hook."""
69
+ document = getattr(reporter, 'document', None)
70
+ return document(body) if callable(document) else body
71
+
72
+
73
+ def per_case_matrix(run: models.RunResult) -> tuple[list[str], list[dict]]:
74
+ """Per-case drill-down behind a scorecard's aggregate means.
75
+
76
+ Returns ``(metric_names, rows)`` where each row is
77
+ ``{label, error, output, cells}`` - ``cells`` maps each per-case metric
78
+ to its :class:`~evalkit.models.Score` (or is absent). Reporters format
79
+ this into a case x metric table so a low aggregate points at the case
80
+ that caused it.
81
+ """
82
+ metrics = sorted(
83
+ {
84
+ score.metric
85
+ for result in run.results
86
+ for score in result.scores
87
+ if score.kind == 'per_case'
88
+ }
89
+ )
90
+ multi = run.scorecard.n_samples > 1
91
+ rows: list[dict] = []
92
+ for result in run.results:
93
+ by_metric = {
94
+ s.metric: s for s in result.scores if s.kind == 'per_case'
95
+ }
96
+ label = result.case.id + (f' #{result.sample_idx}' if multi else '')
97
+ rows.append(
98
+ {
99
+ 'label': label,
100
+ 'error': result.output.error,
101
+ 'output': result.output,
102
+ 'cells': by_metric,
103
+ }
104
+ )
105
+ return metrics, rows
106
+
107
+
108
+ def render_run(reporter: Reporter, run: models.RunResult) -> str:
109
+ """Render a full run as a standalone document: the detailed per-case
110
+ report if the reporter supports ``run``, else the aggregate scorecard."""
111
+ return wrap_document(reporter, run_body(reporter, run))
112
+
113
+
114
+ def run_body(reporter: Reporter, run: models.RunResult) -> str:
115
+ """The run's report *fragment* (unwrapped), so it can be composed with
116
+ other fragments before a single :func:`wrap_document` at the edge."""
117
+ detailed = getattr(reporter, 'run', None)
118
+ if callable(detailed):
119
+ return detailed(run)
120
+ return reporter.scorecard(run.scorecard)
121
+
122
+
123
+ def render_agreement(
124
+ reporter: Reporter, result: models.AgreementResult
125
+ ) -> str:
126
+ """A judge<->human agreement *fragment*, using the reporter's ``agreement``
127
+ hook when present and falling back to the Markdown table otherwise."""
128
+ render = getattr(reporter, 'agreement', None)
129
+ if callable(render):
130
+ return render(result)
131
+ from evalkit import report
132
+
133
+ return report.render_agreement(result)
134
+
135
+
136
+ def render_preferences(
137
+ reporter: Reporter, result: models.PreferenceResult
138
+ ) -> str:
139
+ """A human A-vs-B win-rate *fragment*, using the reporter's ``preferences``
140
+ hook when present and falling back to the Markdown table otherwise."""
141
+ render = getattr(reporter, 'preferences', None)
142
+ if callable(render):
143
+ return render(result)
144
+ from evalkit import report
145
+
146
+ return report.render_preferences(result)
147
+
148
+
149
+ def render_pairwise_agreement(
150
+ reporter: Reporter, result: models.PairwiseAgreement
151
+ ) -> str:
152
+ """A human-panel vs LLM-judge agreement *fragment*, using the reporter's
153
+ ``pairwise_agreement`` hook when present, else the Markdown table."""
154
+ render = getattr(reporter, 'pairwise_agreement', None)
155
+ if callable(render):
156
+ return render(result)
157
+ from evalkit import report
158
+
159
+ return report.render_pairwise_agreement(result)