evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
evalkit/refs.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Reference resolution shared by adapters and graders.
|
|
2
|
+
|
|
3
|
+
Cases, variants, and outputs are *opaque* to the engine - only adapters and
|
|
4
|
+
graders crack them open, and they do so through string references like
|
|
5
|
+
``$input.content`` (adapter request bodies) or ``output.verdict`` (grader
|
|
6
|
+
field selectors). Keeping this resolution in one place is what lets the core
|
|
7
|
+
stay ignorant of any consumer's data shape.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import typing
|
|
11
|
+
|
|
12
|
+
_MISSING = object()
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def resolve_path(obj: typing.Any, path: str) -> typing.Any:
|
|
16
|
+
"""Traverse ``obj`` along a dotted ``path``.
|
|
17
|
+
|
|
18
|
+
Supports dict keys, list indices (numeric segments), and object
|
|
19
|
+
attributes, in that order of preference. Returns ``None`` when any
|
|
20
|
+
segment is missing rather than raising, so a grader/adapter selector
|
|
21
|
+
that points at an absent field degrades to ``None``.
|
|
22
|
+
"""
|
|
23
|
+
current = obj
|
|
24
|
+
for segment in path.split('.'):
|
|
25
|
+
if current is None:
|
|
26
|
+
return None
|
|
27
|
+
if isinstance(current, dict):
|
|
28
|
+
current = current.get(segment, _MISSING)
|
|
29
|
+
elif (
|
|
30
|
+
isinstance(current, (list, tuple))
|
|
31
|
+
and segment.lstrip('-').isdigit()
|
|
32
|
+
):
|
|
33
|
+
index = int(segment)
|
|
34
|
+
if -len(current) <= index < len(current):
|
|
35
|
+
current = current[index]
|
|
36
|
+
else:
|
|
37
|
+
current = _MISSING
|
|
38
|
+
else:
|
|
39
|
+
current = getattr(current, segment, _MISSING)
|
|
40
|
+
if current is _MISSING:
|
|
41
|
+
return None
|
|
42
|
+
return current
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def resolve_ref(context: dict, ref: str) -> typing.Any:
|
|
46
|
+
"""Resolve a ``root.path`` reference against a context dict.
|
|
47
|
+
|
|
48
|
+
``context`` maps root names (``input``, ``expected``, ``variant``,
|
|
49
|
+
``output``, ``case``) to the corresponding objects.
|
|
50
|
+
"""
|
|
51
|
+
root, _, rest = ref.partition('.')
|
|
52
|
+
if root not in context:
|
|
53
|
+
return None
|
|
54
|
+
base = context[root]
|
|
55
|
+
return resolve_path(base, rest) if rest else base
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def build_value(template: typing.Any, context: dict) -> typing.Any:
|
|
59
|
+
"""Render a (possibly nested) template, substituting ``$ref`` strings.
|
|
60
|
+
|
|
61
|
+
A string beginning with ``$`` is treated as a reference (``$input.foo``)
|
|
62
|
+
and replaced with the resolved value. Any other value - including dicts
|
|
63
|
+
and lists, which are walked recursively - passes through literally.
|
|
64
|
+
"""
|
|
65
|
+
if isinstance(template, str) and template.startswith('$'):
|
|
66
|
+
return resolve_ref(context, template[1:])
|
|
67
|
+
if isinstance(template, dict):
|
|
68
|
+
return {k: build_value(v, context) for k, v in template.items()}
|
|
69
|
+
if isinstance(template, list):
|
|
70
|
+
return [build_value(v, context) for v in template]
|
|
71
|
+
return template
|
evalkit/report.py
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Markdown rendering of scorecards and comparisons (CI / PR comments)."""
|
|
2
|
+
|
|
3
|
+
from evalkit import models
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _fmt(value: float | None) -> str:
|
|
7
|
+
return 'n/a' if value is None else f'{value:.4f}'
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _fmt_delta(value: float | None) -> str:
|
|
11
|
+
return 'n/a' if value is None else f'{value:+.4f}'
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def render_scorecard(scorecard: models.Scorecard) -> str:
|
|
15
|
+
"""Render a single scorecard as a Markdown section."""
|
|
16
|
+
lines = [
|
|
17
|
+
f'### {scorecard.project}/{scorecard.suite} '
|
|
18
|
+
f'- `{scorecard.variant.name}`',
|
|
19
|
+
'',
|
|
20
|
+
f'- model: `{scorecard.model_id}` '
|
|
21
|
+
f'mode: `{scorecard.mode}` '
|
|
22
|
+
f'dataset: `{scorecard.dataset_version}` '
|
|
23
|
+
f'cases: {scorecard.n_cases}x{scorecard.n_samples}',
|
|
24
|
+
'',
|
|
25
|
+
'| metric | value |',
|
|
26
|
+
'| --- | ---: |',
|
|
27
|
+
]
|
|
28
|
+
lines += [
|
|
29
|
+
f'| {m.metric} | {_fmt(m.value)} |' for m in scorecard.metrics.values()
|
|
30
|
+
]
|
|
31
|
+
return '\n'.join(lines)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
_VERDICT_BADGE = {'pass': '**PASS**', 'warn': '**WARN**', 'fail': '**FAIL**'}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def render_comparison(comparison: models.Comparison) -> str:
|
|
38
|
+
"""Render a candidate-vs-baseline comparison as Markdown."""
|
|
39
|
+
badge = _VERDICT_BADGE.get(comparison.verdict, comparison.verdict)
|
|
40
|
+
lines = [
|
|
41
|
+
f'## {badge} - {comparison.project}/{comparison.suite}',
|
|
42
|
+
'',
|
|
43
|
+
f'`{comparison.candidate_variant}` vs '
|
|
44
|
+
f'`{comparison.baseline_variant}` - {comparison.summary}',
|
|
45
|
+
'',
|
|
46
|
+
'| metric | baseline | candidate | delta |',
|
|
47
|
+
'| --- | ---: | ---: | ---: |',
|
|
48
|
+
]
|
|
49
|
+
for delta in comparison.deltas:
|
|
50
|
+
marker = (
|
|
51
|
+
f' **({comparison.win})**'
|
|
52
|
+
if delta.metric == comparison.win_metric
|
|
53
|
+
else ''
|
|
54
|
+
)
|
|
55
|
+
lines.append(
|
|
56
|
+
f'| {delta.metric}{marker} | {_fmt(delta.baseline)} '
|
|
57
|
+
f'| {_fmt(delta.candidate)} | {_fmt_delta(delta.delta)} |'
|
|
58
|
+
)
|
|
59
|
+
if comparison.guardrails:
|
|
60
|
+
lines += ['', '**Guardrails**', '']
|
|
61
|
+
for guard in comparison.guardrails:
|
|
62
|
+
mark = 'ok' if guard.passed else 'BREACH'
|
|
63
|
+
lines.append(f'- [{mark}] `{guard.metric}` - {guard.detail}')
|
|
64
|
+
return '\n'.join(lines)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def render_agreement(result: models.AgreementResult) -> str:
|
|
68
|
+
"""Render a judge<->human agreement result as Markdown."""
|
|
69
|
+
lines = [
|
|
70
|
+
f'### judge<->human agreement - `{result.judge_name}`',
|
|
71
|
+
'',
|
|
72
|
+
f'- {result.n_ratings} ratings from {result.n_raters} rater(s), '
|
|
73
|
+
f'scale 1..{result.scale}',
|
|
74
|
+
f'- overall: MAE {_fmt(result.overall_mae)}, '
|
|
75
|
+
f'r {_fmt(result.overall_correlation)}',
|
|
76
|
+
'',
|
|
77
|
+
'| dimension | n | human | judge | MAE | corr |',
|
|
78
|
+
'| --- | ---: | ---: | ---: | ---: | ---: |',
|
|
79
|
+
]
|
|
80
|
+
for dim in result.dimensions:
|
|
81
|
+
lines.append(
|
|
82
|
+
f'| {dim.dimension} | {dim.n} | {_fmt(dim.human_mean)} '
|
|
83
|
+
f'| {_fmt(dim.judge_mean)} | {_fmt(dim.mae)} '
|
|
84
|
+
f'| {_fmt(dim.correlation)} |'
|
|
85
|
+
)
|
|
86
|
+
return '\n'.join(lines)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def render_sweep(result: models.SweepResult) -> str:
|
|
90
|
+
"""Render an N-way sweep as a ranking + a metric x variant leaderboard."""
|
|
91
|
+
variants = [c.variant for c in result.entries] or sorted(
|
|
92
|
+
{v for row in result.matrix.values() for v in row}
|
|
93
|
+
)
|
|
94
|
+
lines = [f'## sweep - {result.project}/{result.suite}', '']
|
|
95
|
+
if result.win_metric:
|
|
96
|
+
direction = 'higher' if result.win_higher_is_better else 'lower'
|
|
97
|
+
lines += [
|
|
98
|
+
f'ranked by `{result.win_metric}` ({direction} is better)',
|
|
99
|
+
'',
|
|
100
|
+
'| rank | variant | win |',
|
|
101
|
+
'| ---: | --- | ---: |',
|
|
102
|
+
]
|
|
103
|
+
lines += [
|
|
104
|
+
f'| {e.rank} | `{e.variant}` | {_fmt(e.win_value)} |'
|
|
105
|
+
for e in result.entries
|
|
106
|
+
]
|
|
107
|
+
lines.append('')
|
|
108
|
+
|
|
109
|
+
header = '| metric | ' + ' | '.join(variants) + ' |'
|
|
110
|
+
sep = '| --- |' + ' ---: |' * len(variants)
|
|
111
|
+
lines += ['**Leaderboard**', '', header, sep]
|
|
112
|
+
for metric, row in result.matrix.items():
|
|
113
|
+
present = [v for v in row.values() if v is not None]
|
|
114
|
+
best = max(present) if present else None # display hint only
|
|
115
|
+
cells = []
|
|
116
|
+
for variant in variants:
|
|
117
|
+
value = row.get(variant)
|
|
118
|
+
text = _fmt(value)
|
|
119
|
+
if value is not None and best is not None and value == best:
|
|
120
|
+
text = f'**{text}**'
|
|
121
|
+
cells.append(text)
|
|
122
|
+
lines.append(f'| {metric} | ' + ' | '.join(cells) + ' |')
|
|
123
|
+
return '\n'.join(lines)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def render_pairwise(result: models.PairwiseResult) -> str:
|
|
127
|
+
"""Render an A-vs-B pairwise win-rate result as Markdown."""
|
|
128
|
+
rate = _fmt(result.win_rate_a)
|
|
129
|
+
return '\n'.join(
|
|
130
|
+
[
|
|
131
|
+
f'## pairwise - {result.project}/{result.suite}',
|
|
132
|
+
'',
|
|
133
|
+
f'`{result.variant_a}` (A) vs `{result.variant_b}` (B), '
|
|
134
|
+
f'judge `{result.judge_name}@{result.judge_version}`',
|
|
135
|
+
'',
|
|
136
|
+
f'- **A win-rate: {rate}** (ties = half) over {result.n} cases',
|
|
137
|
+
f'- A wins {result.a_wins} - B wins {result.b_wins} '
|
|
138
|
+
f'- ties {result.ties}',
|
|
139
|
+
]
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def render_preferences(result: models.PreferenceResult) -> str:
|
|
144
|
+
"""Render a human side-by-side win-rate (overall + per dimension)."""
|
|
145
|
+
lines = [
|
|
146
|
+
f'## human preferences - {result.project}/{result.suite}',
|
|
147
|
+
'',
|
|
148
|
+
f'`{result.variant_a}` (A) vs `{result.variant_b}` (B), '
|
|
149
|
+
f'{result.n} preferences from {result.n_raters} rater(s)',
|
|
150
|
+
'',
|
|
151
|
+
f'- **A win-rate: {_fmt(result.win_rate_a)}** (ties = half) '
|
|
152
|
+
f'- A wins {result.a_wins} - B wins {result.b_wins} '
|
|
153
|
+
f'- ties {result.ties}',
|
|
154
|
+
'',
|
|
155
|
+
'| dimension | n | A wins | B wins | ties | A win-rate |',
|
|
156
|
+
'| --- | ---: | ---: | ---: | ---: | ---: |',
|
|
157
|
+
]
|
|
158
|
+
for dim in result.dimensions:
|
|
159
|
+
lines.append(
|
|
160
|
+
f'| {dim.dimension} | {dim.n} | {dim.a_wins} | {dim.b_wins} '
|
|
161
|
+
f'| {dim.ties} | {_fmt(dim.win_rate_a)} |'
|
|
162
|
+
)
|
|
163
|
+
return '\n'.join(lines)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def render_pairwise_agreement(result: models.PairwiseAgreement) -> str:
|
|
167
|
+
"""Render human-panel vs LLM-judge head-to-head agreement as Markdown."""
|
|
168
|
+
lines = [
|
|
169
|
+
f'## pairwise agreement - human vs `{result.judge_name}`',
|
|
170
|
+
'',
|
|
171
|
+
f'`{result.variant_a}` (A) vs `{result.variant_b}` (B)',
|
|
172
|
+
'',
|
|
173
|
+
f'- **agreement: {_fmt(result.agreement_rate)}** '
|
|
174
|
+
f'({result.agree}/{result.n} cases pick the same winner)',
|
|
175
|
+
f'- A win-rate: human {_fmt(result.human_win_rate_a)} '
|
|
176
|
+
f'- judge {_fmt(result.judge_win_rate_a)}',
|
|
177
|
+
'',
|
|
178
|
+
'| case | human | judge | agree |',
|
|
179
|
+
'| --- | :---: | :---: | :---: |',
|
|
180
|
+
]
|
|
181
|
+
for case in result.outcomes:
|
|
182
|
+
mark = 'yes' if case.agree else 'NO'
|
|
183
|
+
lines.append(
|
|
184
|
+
f'| {case.case_id} | {case.human} | {case.judge} | {mark} |'
|
|
185
|
+
)
|
|
186
|
+
return '\n'.join(lines)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Report rendering, pluggable by ``type`` like adapters and graders.
|
|
2
|
+
|
|
3
|
+
Importing this package registers the built-in ``markdown`` and ``html``
|
|
4
|
+
reporters. Select one with :func:`build_reporter` (the CLI's ``--report``
|
|
5
|
+
flag), and register more with ``@evalkit.reporters.base.register(...)``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from evalkit.reporters import html, markdown # noqa: F401 - register built-ins
|
|
9
|
+
from evalkit.reporters.base import (
|
|
10
|
+
Reporter,
|
|
11
|
+
build_reporter,
|
|
12
|
+
per_case_matrix,
|
|
13
|
+
register,
|
|
14
|
+
render_agreement,
|
|
15
|
+
render_pairwise_agreement,
|
|
16
|
+
render_preferences,
|
|
17
|
+
render_run,
|
|
18
|
+
run_body,
|
|
19
|
+
wrap_document,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
'Reporter',
|
|
24
|
+
'build_reporter',
|
|
25
|
+
'per_case_matrix',
|
|
26
|
+
'register',
|
|
27
|
+
'render_agreement',
|
|
28
|
+
'render_pairwise_agreement',
|
|
29
|
+
'render_preferences',
|
|
30
|
+
'render_run',
|
|
31
|
+
'run_body',
|
|
32
|
+
'wrap_document',
|
|
33
|
+
]
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Reporter protocol and registry (the report-rendering seam).
|
|
2
|
+
|
|
3
|
+
A reporter turns engine results into a rendered report. Two report kinds are
|
|
4
|
+
first-class, mirroring the two analyses the engine produces:
|
|
5
|
+
|
|
6
|
+
* ``scorecard`` - a SINGLE analysis (one suite x one variant run);
|
|
7
|
+
* ``comparison`` - a COMPARATIVE analysis (candidate vs baseline).
|
|
8
|
+
|
|
9
|
+
Built-ins ``markdown`` and ``html`` cover both. Register more the same way as
|
|
10
|
+
adapters and graders - ``@reporters.base.register('pdf')`` + ``--plugins`` -
|
|
11
|
+
so a consumer can add a format without touching the core.
|
|
12
|
+
|
|
13
|
+
Three further kinds are optional; a reporter is discovered to support each
|
|
14
|
+
via ``getattr`` and only implements the ones it cares about:
|
|
15
|
+
|
|
16
|
+
* ``run`` - a scorecard plus its per-case drill-down (see :func:`render_run`);
|
|
17
|
+
* ``agreement`` - a judge<->human calibration table (see
|
|
18
|
+
:func:`render_agreement`);
|
|
19
|
+
* ``document`` - wrap a rendered body in a standalone document (e.g. an HTML
|
|
20
|
+
page). Reporters that omit it are treated as identity.
|
|
21
|
+
|
|
22
|
+
Every ``*`` method returns a *fragment*; wrapping into a standalone document
|
|
23
|
+
happens once, at the edge, so fragments compose (a ``run`` body and an
|
|
24
|
+
``agreement`` table can share one page).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import typing
|
|
28
|
+
|
|
29
|
+
from evalkit import models
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@typing.runtime_checkable
|
|
33
|
+
class Reporter(typing.Protocol):
|
|
34
|
+
"""Render a single scorecard and a candidate-vs-baseline comparison."""
|
|
35
|
+
|
|
36
|
+
def scorecard(self, scorecard: models.Scorecard) -> str: ...
|
|
37
|
+
|
|
38
|
+
def comparison(self, comparison: models.Comparison) -> str: ...
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
_REGISTRY: dict[str, type] = {}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def register(type_name: str) -> typing.Callable[[type], type]:
|
|
45
|
+
"""Class decorator registering a reporter under a report ``type``."""
|
|
46
|
+
|
|
47
|
+
def _decorate(cls: type) -> type:
|
|
48
|
+
if type_name in _REGISTRY:
|
|
49
|
+
raise ValueError(f'report type {type_name!r} already registered')
|
|
50
|
+
_REGISTRY[type_name] = cls
|
|
51
|
+
return cls
|
|
52
|
+
|
|
53
|
+
return _decorate
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def build_reporter(spec: str | dict) -> Reporter:
|
|
57
|
+
"""Instantiate a reporter from a name or a ``{type, ...}`` config spec."""
|
|
58
|
+
spec = {'type': spec} if isinstance(spec, str) else dict(spec)
|
|
59
|
+
type_name = spec.pop('type')
|
|
60
|
+
if type_name not in _REGISTRY:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
f'unknown report type {type_name!r}; known: {sorted(_REGISTRY)}'
|
|
63
|
+
)
|
|
64
|
+
return _REGISTRY[type_name](**spec)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def wrap_document(reporter: Reporter, body: str) -> str:
|
|
68
|
+
"""Wrap ``body`` with the reporter's optional ``document`` hook."""
|
|
69
|
+
document = getattr(reporter, 'document', None)
|
|
70
|
+
return document(body) if callable(document) else body
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def per_case_matrix(run: models.RunResult) -> tuple[list[str], list[dict]]:
|
|
74
|
+
"""Per-case drill-down behind a scorecard's aggregate means.
|
|
75
|
+
|
|
76
|
+
Returns ``(metric_names, rows)`` where each row is
|
|
77
|
+
``{label, error, output, cells}`` - ``cells`` maps each per-case metric
|
|
78
|
+
to its :class:`~evalkit.models.Score` (or is absent). Reporters format
|
|
79
|
+
this into a case x metric table so a low aggregate points at the case
|
|
80
|
+
that caused it.
|
|
81
|
+
"""
|
|
82
|
+
metrics = sorted(
|
|
83
|
+
{
|
|
84
|
+
score.metric
|
|
85
|
+
for result in run.results
|
|
86
|
+
for score in result.scores
|
|
87
|
+
if score.kind == 'per_case'
|
|
88
|
+
}
|
|
89
|
+
)
|
|
90
|
+
multi = run.scorecard.n_samples > 1
|
|
91
|
+
rows: list[dict] = []
|
|
92
|
+
for result in run.results:
|
|
93
|
+
by_metric = {
|
|
94
|
+
s.metric: s for s in result.scores if s.kind == 'per_case'
|
|
95
|
+
}
|
|
96
|
+
label = result.case.id + (f' #{result.sample_idx}' if multi else '')
|
|
97
|
+
rows.append(
|
|
98
|
+
{
|
|
99
|
+
'label': label,
|
|
100
|
+
'error': result.output.error,
|
|
101
|
+
'output': result.output,
|
|
102
|
+
'cells': by_metric,
|
|
103
|
+
}
|
|
104
|
+
)
|
|
105
|
+
return metrics, rows
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def render_run(reporter: Reporter, run: models.RunResult) -> str:
|
|
109
|
+
"""Render a full run as a standalone document: the detailed per-case
|
|
110
|
+
report if the reporter supports ``run``, else the aggregate scorecard."""
|
|
111
|
+
return wrap_document(reporter, run_body(reporter, run))
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def run_body(reporter: Reporter, run: models.RunResult) -> str:
|
|
115
|
+
"""The run's report *fragment* (unwrapped), so it can be composed with
|
|
116
|
+
other fragments before a single :func:`wrap_document` at the edge."""
|
|
117
|
+
detailed = getattr(reporter, 'run', None)
|
|
118
|
+
if callable(detailed):
|
|
119
|
+
return detailed(run)
|
|
120
|
+
return reporter.scorecard(run.scorecard)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def render_agreement(
|
|
124
|
+
reporter: Reporter, result: models.AgreementResult
|
|
125
|
+
) -> str:
|
|
126
|
+
"""A judge<->human agreement *fragment*, using the reporter's ``agreement``
|
|
127
|
+
hook when present and falling back to the Markdown table otherwise."""
|
|
128
|
+
render = getattr(reporter, 'agreement', None)
|
|
129
|
+
if callable(render):
|
|
130
|
+
return render(result)
|
|
131
|
+
from evalkit import report
|
|
132
|
+
|
|
133
|
+
return report.render_agreement(result)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def render_preferences(
|
|
137
|
+
reporter: Reporter, result: models.PreferenceResult
|
|
138
|
+
) -> str:
|
|
139
|
+
"""A human A-vs-B win-rate *fragment*, using the reporter's ``preferences``
|
|
140
|
+
hook when present and falling back to the Markdown table otherwise."""
|
|
141
|
+
render = getattr(reporter, 'preferences', None)
|
|
142
|
+
if callable(render):
|
|
143
|
+
return render(result)
|
|
144
|
+
from evalkit import report
|
|
145
|
+
|
|
146
|
+
return report.render_preferences(result)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def render_pairwise_agreement(
|
|
150
|
+
reporter: Reporter, result: models.PairwiseAgreement
|
|
151
|
+
) -> str:
|
|
152
|
+
"""A human-panel vs LLM-judge agreement *fragment*, using the reporter's
|
|
153
|
+
``pairwise_agreement`` hook when present, else the Markdown table."""
|
|
154
|
+
render = getattr(reporter, 'pairwise_agreement', None)
|
|
155
|
+
if callable(render):
|
|
156
|
+
return render(result)
|
|
157
|
+
from evalkit import report
|
|
158
|
+
|
|
159
|
+
return report.render_pairwise_agreement(result)
|