evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
"""HTML reporter - standalone, self-contained report documents.
|
|
2
|
+
|
|
3
|
+
Renders a single scorecard or a candidate-vs-baseline comparison as an HTML
|
|
4
|
+
fragment, and wraps fragments in a minimal styled document via ``document``.
|
|
5
|
+
Values are escaped; the output has no external assets, so it drops straight
|
|
6
|
+
into a CI artifact, a PR attachment, or an inline preview.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import base64
|
|
10
|
+
import html as _html
|
|
11
|
+
import json
|
|
12
|
+
import pathlib
|
|
13
|
+
|
|
14
|
+
from evalkit import models
|
|
15
|
+
from evalkit.reporters import base
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _dump(fields: dict) -> str:
|
|
19
|
+
try:
|
|
20
|
+
return json.dumps(fields, indent=2, default=str)[:6000]
|
|
21
|
+
except TypeError, ValueError:
|
|
22
|
+
return str(fields)[:6000]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
_IMG_EXT = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _embed_artifact(name: str, path: str) -> str:
|
|
29
|
+
"""Embed one saved artifact inline so the report stays self-contained.
|
|
30
|
+
|
|
31
|
+
Images become base64 ``data:`` URIs, HTML renders in a sandboxed iframe,
|
|
32
|
+
PDFs embed via a data URI, and anything else falls back to its decoded
|
|
33
|
+
text. A path that can't be read degrades to a note rather than erroring,
|
|
34
|
+
so a report built on a machine without the artifacts still renders.
|
|
35
|
+
"""
|
|
36
|
+
p = pathlib.Path(str(path))
|
|
37
|
+
cap = (
|
|
38
|
+
f'<figcaption class="meta">{_esc(name)} · '
|
|
39
|
+
f'<code>{_esc(p.name)}</code></figcaption>'
|
|
40
|
+
)
|
|
41
|
+
try:
|
|
42
|
+
raw = p.read_bytes()
|
|
43
|
+
except OSError:
|
|
44
|
+
return (
|
|
45
|
+
f'<figure class="art">{cap}'
|
|
46
|
+
f'<p class="meta">missing: <code>{_esc(path)}</code></p></figure>'
|
|
47
|
+
)
|
|
48
|
+
ext = p.suffix.lower()
|
|
49
|
+
if ext in _IMG_EXT:
|
|
50
|
+
mime = (
|
|
51
|
+
'image/svg+xml'
|
|
52
|
+
if ext == '.svg'
|
|
53
|
+
else 'image/jpeg'
|
|
54
|
+
if ext in ('.jpg', '.jpeg')
|
|
55
|
+
else f'image/{ext.lstrip(".")}'
|
|
56
|
+
)
|
|
57
|
+
b64 = base64.b64encode(raw).decode('ascii')
|
|
58
|
+
media = (
|
|
59
|
+
f'<img class="art-img" alt="{_esc(name)}" '
|
|
60
|
+
f'src="data:{mime};base64,{b64}">'
|
|
61
|
+
)
|
|
62
|
+
elif ext in ('.html', '.htm'):
|
|
63
|
+
media = _iframe(raw.decode('utf-8', 'replace'))
|
|
64
|
+
elif ext == '.pdf':
|
|
65
|
+
b64 = base64.b64encode(raw).decode('ascii')
|
|
66
|
+
media = (
|
|
67
|
+
f'<iframe class="art-frame" '
|
|
68
|
+
f'src="data:application/pdf;base64,{b64}"></iframe>'
|
|
69
|
+
)
|
|
70
|
+
else:
|
|
71
|
+
media = f'<pre>{_esc(raw.decode("utf-8", "replace")[:6000])}</pre>'
|
|
72
|
+
return f'<figure class="art">{cap}{media}</figure>'
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _iframe(content: str) -> str:
|
|
76
|
+
"""A sandboxed iframe rendering an HTML fragment/document."""
|
|
77
|
+
return (
|
|
78
|
+
f'<iframe class="art-frame" sandbox loading="lazy" '
|
|
79
|
+
f'srcdoc="{_html.escape(content, quote=True)}"></iframe>'
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _pt(value) -> str:
|
|
84
|
+
"""A raw judge point: integer-clean when whole (``4``, not ``4.0``)."""
|
|
85
|
+
return '–' if value is None else f'{value:g}'
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _judges_html(score: models.Score) -> str:
|
|
89
|
+
"""The per-judge breakdown behind an LLM-judge ``overall`` score:
|
|
90
|
+
each judge's raw per-dimension points side by side, its own normalized
|
|
91
|
+
overall, and its free-text rationale - the ``why`` the aggregate hides."""
|
|
92
|
+
judges = score.judges
|
|
93
|
+
if not judges:
|
|
94
|
+
return ''
|
|
95
|
+
dims = list(judges[0].points)
|
|
96
|
+
head = ''.join(f'<th class="num">{_esc(j.key)}</th>' for j in judges)
|
|
97
|
+
body = ''
|
|
98
|
+
for dim in dims:
|
|
99
|
+
cells = ''.join(
|
|
100
|
+
f'<td class="num">{_pt(j.points.get(dim))}</td>' for j in judges
|
|
101
|
+
)
|
|
102
|
+
body += f'<tr><td>{_esc(dim)}</td>{cells}</tr>'
|
|
103
|
+
body += (
|
|
104
|
+
'<tr><td><b>overall (0..1)</b></td>'
|
|
105
|
+
+ ''.join(f'<td class="num">{_fmt(j.overall)}</td>' for j in judges)
|
|
106
|
+
+ '</tr>'
|
|
107
|
+
)
|
|
108
|
+
rationales = ''.join(
|
|
109
|
+
f'<p class="rationale"><b>{_esc(j.key)}'
|
|
110
|
+
f'{" @" + _esc(j.version) if j.version else ""}:</b> '
|
|
111
|
+
f'{_esc(j.rationale)}</p>'
|
|
112
|
+
for j in judges
|
|
113
|
+
if j.rationale
|
|
114
|
+
)
|
|
115
|
+
return (
|
|
116
|
+
f'<div class="judgment"><h4>Judge panel — '
|
|
117
|
+
f'<code>{_esc(score.grader)}</code> · raw points</h4>'
|
|
118
|
+
f'<table class="judges"><thead><tr><th>dimension</th>{head}</tr>'
|
|
119
|
+
f'</thead><tbody>{body}</tbody></table>{rationales}</div>'
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _notes_html(cells: dict, error: str | None) -> str:
|
|
124
|
+
"""Surface stored per-case ``detail`` the matrix can't: the ``why`` of
|
|
125
|
+
each failing deterministic check, plus an errored invocation's message."""
|
|
126
|
+
items = []
|
|
127
|
+
if error:
|
|
128
|
+
items.append(f'<li class="bad">errored: {_esc(error)}</li>')
|
|
129
|
+
for sc in cells.values():
|
|
130
|
+
if sc.passed is False and sc.detail:
|
|
131
|
+
items.append(
|
|
132
|
+
f'<li><code>{_esc(sc.metric)}</code> — {_esc(sc.detail)}'
|
|
133
|
+
'</li>'
|
|
134
|
+
)
|
|
135
|
+
if not items:
|
|
136
|
+
return ''
|
|
137
|
+
return f'<h4>Notes</h4><ul class="notes">{"".join(items)}</ul>'
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
_CSS = (
|
|
141
|
+
'body{font:15px/1.5 system-ui,-apple-system,Segoe UI,Roboto,sans-serif;'
|
|
142
|
+
'margin:0;padding:24px;color:#1a1d21;background:#f6f7f9}'
|
|
143
|
+
'.card{max-width:900px;margin:0 auto 20px;background:#fff;'
|
|
144
|
+
'border:1px solid #e3e6ea;border-radius:12px;padding:20px 24px;'
|
|
145
|
+
'box-shadow:0 1px 3px rgba(20,25,35,.06)}'
|
|
146
|
+
'h2{margin:0 0 6px;font-size:20px}h3{margin:18px 0 8px;font-size:15px}'
|
|
147
|
+
'code{background:#eef0f3;border-radius:5px;padding:1px 6px;font-size:.9em}'
|
|
148
|
+
'.meta{color:#68707a;font-size:13px;margin:0 0 14px}'
|
|
149
|
+
'table{width:100%;border-collapse:collapse;font-size:14px}'
|
|
150
|
+
'th,td{text-align:left;padding:7px 10px;border-bottom:1px solid #eceef1}'
|
|
151
|
+
'th{color:#68707a;font-weight:600}.num{text-align:right;'
|
|
152
|
+
'font-variant-numeric:tabular-nums}'
|
|
153
|
+
'.up{color:#1f8f5f}.down{color:#c0392b}'
|
|
154
|
+
'.verdict{display:inline-block;border-radius:6px;padding:2px 10px;color:#fff}'
|
|
155
|
+
'.verdict.pass{background:#1f8f5f}.verdict.warn{background:#d0870b}'
|
|
156
|
+
'.verdict.fail{background:#c0392b}'
|
|
157
|
+
'.guards{list-style:none;padding:0;margin:0}'
|
|
158
|
+
'.guards li{padding:4px 0;font-size:14px}'
|
|
159
|
+
'.guards li.ok::before{content:"\\2713 ";color:#1f8f5f}'
|
|
160
|
+
'.guards li.breach::before{content:"\\2717 ";color:#c0392b}'
|
|
161
|
+
'td.fail{color:#c0392b;font-weight:600}'
|
|
162
|
+
'details{margin:6px 0}summary{cursor:pointer;color:#68707a;font-size:13px}'
|
|
163
|
+
'pre{background:#f0f2f4;border-radius:6px;padding:10px;overflow:auto;'
|
|
164
|
+
'max-height:340px;white-space:pre-wrap;word-break:break-word;'
|
|
165
|
+
'font:12px ui-monospace,SFMono-Regular,Menlo,monospace}'
|
|
166
|
+
'.scroll{overflow-x:auto}.scroll table{min-width:100%;width:max-content}'
|
|
167
|
+
'.arts{display:grid;gap:14px;margin:10px 0}'
|
|
168
|
+
'figure.art{margin:0;border:1px solid #e3e6ea;border-radius:8px;'
|
|
169
|
+
'padding:10px 12px;background:#fafbfc}'
|
|
170
|
+
'figure.art figcaption{margin:0 0 8px}'
|
|
171
|
+
'.art-img{display:block;max-width:100%;height:auto;border-radius:6px;'
|
|
172
|
+
'border:1px solid #e3e6ea;background:#fff}'
|
|
173
|
+
'.art-frame{width:100%;height:520px;border:1px solid #e3e6ea;'
|
|
174
|
+
'border-radius:6px;background:#fff}'
|
|
175
|
+
'details.raw{margin-top:10px}'
|
|
176
|
+
'h4{margin:14px 0 6px;font-size:14px}'
|
|
177
|
+
'.judgment{margin:10px 0}'
|
|
178
|
+
'table.judges{width:auto;min-width:60%}'
|
|
179
|
+
'table.judges td:first-child,table.judges th:first-child{font-weight:600}'
|
|
180
|
+
'.rationale{font-size:13px;color:#333;margin:6px 0;padding:8px 10px;'
|
|
181
|
+
'background:#f0f2f4;border-radius:6px}'
|
|
182
|
+
'.notes{margin:4px 0;padding-left:18px;font-size:13px}'
|
|
183
|
+
'.notes li{padding:2px 0}.notes li.bad{color:#c0392b;font-weight:600}'
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _fmt(value: float | None) -> str:
|
|
188
|
+
return 'n/a' if value is None else f'{value:.4f}'
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _fmt_delta(value: float | None) -> str:
|
|
192
|
+
return 'n/a' if value is None else f'{value:+.4f}'
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _esc(value) -> str:
|
|
196
|
+
return _html.escape(str(value))
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@base.register('html')
|
|
200
|
+
class HtmlReporter:
|
|
201
|
+
"""Render scorecards and comparisons as self-contained HTML."""
|
|
202
|
+
|
|
203
|
+
def scorecard(self, scorecard: models.Scorecard) -> str:
|
|
204
|
+
rows = ''.join(
|
|
205
|
+
f'<tr><td>{_esc(m.metric)}</td>'
|
|
206
|
+
f'<td class="num">{_fmt(m.value)}</td></tr>'
|
|
207
|
+
for m in scorecard.metrics.values()
|
|
208
|
+
)
|
|
209
|
+
meta = (
|
|
210
|
+
f'model <code>{_esc(scorecard.model_id)}</code> · '
|
|
211
|
+
f'mode <code>{_esc(scorecard.mode)}</code> · '
|
|
212
|
+
f'dataset <code>{_esc(scorecard.dataset_version)}</code> · '
|
|
213
|
+
f'{scorecard.n_cases}×{scorecard.n_samples} cases'
|
|
214
|
+
)
|
|
215
|
+
return (
|
|
216
|
+
f'<section class="card"><h2>{_esc(scorecard.project)}/'
|
|
217
|
+
f'{_esc(scorecard.suite)} · '
|
|
218
|
+
f'<code>{_esc(scorecard.variant.name)}</code></h2>'
|
|
219
|
+
f'<p class="meta">{meta}</p>'
|
|
220
|
+
f'<table><thead><tr><th>metric</th>'
|
|
221
|
+
f'<th class="num">value</th></tr></thead>'
|
|
222
|
+
f'<tbody>{rows}</tbody></table></section>'
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
def comparison(self, comparison: models.Comparison) -> str:
|
|
226
|
+
drows = ''
|
|
227
|
+
for delta in comparison.deltas:
|
|
228
|
+
mark = (
|
|
229
|
+
f' <b>({_esc(comparison.win)})</b>'
|
|
230
|
+
if delta.metric == comparison.win_metric
|
|
231
|
+
else ''
|
|
232
|
+
)
|
|
233
|
+
direction = (
|
|
234
|
+
'up'
|
|
235
|
+
if (delta.delta or 0) > 0
|
|
236
|
+
else 'down'
|
|
237
|
+
if (delta.delta or 0) < 0
|
|
238
|
+
else ''
|
|
239
|
+
)
|
|
240
|
+
drows += (
|
|
241
|
+
f'<tr><td>{_esc(delta.metric)}{mark}</td>'
|
|
242
|
+
f'<td class="num">{_fmt(delta.baseline)}</td>'
|
|
243
|
+
f'<td class="num">{_fmt(delta.candidate)}</td>'
|
|
244
|
+
f'<td class="num {direction}">{_fmt_delta(delta.delta)}</td>'
|
|
245
|
+
f'</tr>'
|
|
246
|
+
)
|
|
247
|
+
guards = ''
|
|
248
|
+
if comparison.guardrails:
|
|
249
|
+
items = ''.join(
|
|
250
|
+
f'<li class="{"ok" if g.passed else "breach"}">'
|
|
251
|
+
f'<code>{_esc(g.metric)}</code> — {_esc(g.detail)}</li>'
|
|
252
|
+
for g in comparison.guardrails
|
|
253
|
+
)
|
|
254
|
+
guards = f'<h3>Guardrails</h3><ul class="guards">{items}</ul>'
|
|
255
|
+
return (
|
|
256
|
+
f'<section class="card">'
|
|
257
|
+
f'<h2><span class="verdict {_esc(comparison.verdict)}">'
|
|
258
|
+
f'{_esc(comparison.verdict.upper())}</span> '
|
|
259
|
+
f'{_esc(comparison.project)}/{_esc(comparison.suite)}</h2>'
|
|
260
|
+
f'<p><code>{_esc(comparison.candidate_variant)}</code> vs '
|
|
261
|
+
f'<code>{_esc(comparison.baseline_variant)}</code> — '
|
|
262
|
+
f'{_esc(comparison.summary)}</p>'
|
|
263
|
+
f'<table><thead><tr><th>metric</th>'
|
|
264
|
+
f'<th class="num">baseline</th><th class="num">candidate</th>'
|
|
265
|
+
f'<th class="num">delta</th></tr></thead>'
|
|
266
|
+
f'<tbody>{drows}</tbody></table>{guards}</section>'
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
def agreement(self, result: models.AgreementResult) -> str:
|
|
270
|
+
rows = ''.join(
|
|
271
|
+
f'<tr><td>{_esc(d.dimension)}</td>'
|
|
272
|
+
f'<td class="num">{d.n}</td>'
|
|
273
|
+
f'<td class="num">{_fmt(d.human_mean)}</td>'
|
|
274
|
+
f'<td class="num">{_fmt(d.judge_mean)}</td>'
|
|
275
|
+
f'<td class="num">{_fmt(d.mae)}</td>'
|
|
276
|
+
f'<td class="num">{_fmt(d.correlation)}</td></tr>'
|
|
277
|
+
for d in result.dimensions
|
|
278
|
+
)
|
|
279
|
+
return (
|
|
280
|
+
f'<section class="card"><h2>Judge↔human agreement · '
|
|
281
|
+
f'<code>{_esc(result.judge_name)}</code></h2>'
|
|
282
|
+
f'<p class="meta">{result.n_ratings} ratings from '
|
|
283
|
+
f'{result.n_raters} rater(s), scale 1..{result.scale} · '
|
|
284
|
+
f'overall MAE {_fmt(result.overall_mae)}, '
|
|
285
|
+
f'r {_fmt(result.overall_correlation)}</p>'
|
|
286
|
+
f'<table><thead><tr><th>dimension</th><th class="num">n</th>'
|
|
287
|
+
f'<th class="num">human</th><th class="num">judge</th>'
|
|
288
|
+
f'<th class="num">MAE</th><th class="num">corr</th></tr></thead>'
|
|
289
|
+
f'<tbody>{rows}</tbody></table></section>'
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
def preferences(self, result: models.PreferenceResult) -> str:
|
|
293
|
+
rows = ''.join(
|
|
294
|
+
f'<tr><td>{_esc(d.dimension)}</td>'
|
|
295
|
+
f'<td class="num">{d.n}</td>'
|
|
296
|
+
f'<td class="num">{d.a_wins}</td>'
|
|
297
|
+
f'<td class="num">{d.b_wins}</td>'
|
|
298
|
+
f'<td class="num">{d.ties}</td>'
|
|
299
|
+
f'<td class="num">{_fmt(d.win_rate_a)}</td></tr>'
|
|
300
|
+
for d in result.dimensions
|
|
301
|
+
)
|
|
302
|
+
body = (
|
|
303
|
+
f'<table><thead><tr><th>dimension</th><th class="num">n</th>'
|
|
304
|
+
f'<th class="num">A wins</th><th class="num">B wins</th>'
|
|
305
|
+
f'<th class="num">ties</th><th class="num">A win-rate</th>'
|
|
306
|
+
f'</tr></thead><tbody>{rows}</tbody></table>'
|
|
307
|
+
if rows
|
|
308
|
+
else ''
|
|
309
|
+
)
|
|
310
|
+
return (
|
|
311
|
+
f'<section class="card"><h2>Human preferences · '
|
|
312
|
+
f'{_esc(result.project)}/{_esc(result.suite)}</h2>'
|
|
313
|
+
f'<p class="meta"><code>{_esc(result.variant_a)}</code> (A) vs '
|
|
314
|
+
f'<code>{_esc(result.variant_b)}</code> (B) · '
|
|
315
|
+
f'{result.n} preferences from {result.n_raters} rater(s)</p>'
|
|
316
|
+
f'<p><b>A win-rate {_fmt(result.win_rate_a)}</b> (ties = half) '
|
|
317
|
+
f'· A wins {result.a_wins} · B wins {result.b_wins} '
|
|
318
|
+
f'· ties {result.ties}</p>{body}</section>'
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
def pairwise_agreement(self, result: models.PairwiseAgreement) -> str:
|
|
322
|
+
rows = ''.join(
|
|
323
|
+
f'<tr><td>{_esc(c.case_id)}</td>'
|
|
324
|
+
f'<td class="num">{_esc(c.human)}</td>'
|
|
325
|
+
f'<td class="num">{_esc(c.judge)}</td>'
|
|
326
|
+
f'<td class="{"" if c.agree else "fail"}">'
|
|
327
|
+
f'{"yes" if c.agree else "NO"}</td></tr>'
|
|
328
|
+
for c in result.outcomes
|
|
329
|
+
)
|
|
330
|
+
return (
|
|
331
|
+
f'<section class="card"><h2>Pairwise agreement · human vs '
|
|
332
|
+
f'<code>{_esc(result.judge_name)}</code></h2>'
|
|
333
|
+
f'<p class="meta"><code>{_esc(result.variant_a)}</code> (A) vs '
|
|
334
|
+
f'<code>{_esc(result.variant_b)}</code> (B)</p>'
|
|
335
|
+
f'<p><b>agreement {_fmt(result.agreement_rate)}</b> '
|
|
336
|
+
f'({result.agree}/{result.n} cases pick the same winner) · '
|
|
337
|
+
f'A win-rate: human {_fmt(result.human_win_rate_a)}, '
|
|
338
|
+
f'judge {_fmt(result.judge_win_rate_a)}</p>'
|
|
339
|
+
f'<table><thead><tr><th>case</th><th class="num">human</th>'
|
|
340
|
+
f'<th class="num">judge</th><th>agree</th></tr></thead>'
|
|
341
|
+
f'<tbody>{rows}</tbody></table></section>'
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
def run(self, run: models.RunResult) -> str:
|
|
345
|
+
"""Scorecard summary + a per-case table and collapsible outputs.
|
|
346
|
+
|
|
347
|
+
Returns a fragment; :func:`base.render_run` wraps it in a document."""
|
|
348
|
+
body = self.scorecard(run.scorecard)
|
|
349
|
+
metrics, rows = base.per_case_matrix(run)
|
|
350
|
+
if rows:
|
|
351
|
+
head = ''.join(f'<th class="num">{_esc(m)}</th>' for m in metrics)
|
|
352
|
+
trs = ''
|
|
353
|
+
details = ''
|
|
354
|
+
for row in rows:
|
|
355
|
+
if row['error']:
|
|
356
|
+
cells = (
|
|
357
|
+
f'<td colspan="{len(metrics)}" class="fail">'
|
|
358
|
+
f'error: {_esc(row["error"])}</td>'
|
|
359
|
+
)
|
|
360
|
+
else:
|
|
361
|
+
cells = ''
|
|
362
|
+
for metric in metrics:
|
|
363
|
+
score = row['cells'].get(metric)
|
|
364
|
+
value = _fmt(score.value if score else None)
|
|
365
|
+
cls = (
|
|
366
|
+
'num fail'
|
|
367
|
+
if score and score.passed is False
|
|
368
|
+
else 'num'
|
|
369
|
+
)
|
|
370
|
+
cells += f'<td class="{cls}">{value}</td>'
|
|
371
|
+
trs += f'<tr><td>{_esc(row["label"])}</td>{cells}</tr>'
|
|
372
|
+
out = row['output']
|
|
373
|
+
fields = out.fields or {}
|
|
374
|
+
embeds = [
|
|
375
|
+
_embed_artifact(name, path)
|
|
376
|
+
for name, path in (out.artifacts or {}).items()
|
|
377
|
+
]
|
|
378
|
+
# Fallback: render an inline HTML field when nothing was saved
|
|
379
|
+
# to disk, so offline/replay runs still show the rendered form.
|
|
380
|
+
if not embeds and isinstance(fields.get('html'), str):
|
|
381
|
+
embeds.append(
|
|
382
|
+
'<figure class="art"><figcaption class="meta">'
|
|
383
|
+
'html (inline)</figcaption>'
|
|
384
|
+
f'{_iframe(fields["html"])}</figure>'
|
|
385
|
+
)
|
|
386
|
+
arts = (
|
|
387
|
+
f'<div class="arts">{"".join(embeds)}</div>'
|
|
388
|
+
if embeds
|
|
389
|
+
else ''
|
|
390
|
+
)
|
|
391
|
+
judgments = ''.join(
|
|
392
|
+
_judges_html(sc)
|
|
393
|
+
for sc in row['cells'].values()
|
|
394
|
+
if sc.judges
|
|
395
|
+
)
|
|
396
|
+
notes = _notes_html(row['cells'], row['error'])
|
|
397
|
+
lat = (
|
|
398
|
+
f' · <span class="meta">'
|
|
399
|
+
f'{out.latency_ms / 1000:.1f}s</span>'
|
|
400
|
+
if out.latency_ms is not None
|
|
401
|
+
else ''
|
|
402
|
+
)
|
|
403
|
+
details += (
|
|
404
|
+
f'<details><summary>{_esc(row["label"])}{lat}</summary>'
|
|
405
|
+
+ arts
|
|
406
|
+
+ judgments
|
|
407
|
+
+ notes
|
|
408
|
+
+ '<details class="raw"><summary>output fields (JSON)'
|
|
409
|
+
f'</summary><pre>{_esc(_dump(fields))}</pre></details>'
|
|
410
|
+
+ '</details>'
|
|
411
|
+
)
|
|
412
|
+
body += (
|
|
413
|
+
'<section class="card"><h3>Per-case</h3>'
|
|
414
|
+
f'<div class="scroll"><table><thead><tr><th>case</th>{head}'
|
|
415
|
+
f'</tr></thead><tbody>{trs}</tbody></table></div>'
|
|
416
|
+
f'<h3>Outputs</h3>{details}</section>'
|
|
417
|
+
)
|
|
418
|
+
return body
|
|
419
|
+
|
|
420
|
+
def document(self, body: str) -> str:
|
|
421
|
+
return (
|
|
422
|
+
'<!doctype html><html lang="en"><head><meta charset="utf-8">'
|
|
423
|
+
'<meta name="viewport" content="width=device-width,'
|
|
424
|
+
'initial-scale=1"><title>evalkit report</title>'
|
|
425
|
+
f'<style>{_CSS}</style></head><body>{body}</body></html>'
|
|
426
|
+
)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Markdown reporter - the default. Delegates to :mod:`evalkit.report`.
|
|
2
|
+
|
|
3
|
+
The Markdown renderers live in ``evalkit.report`` (and are re-used directly
|
|
4
|
+
by the CLI's sweep/pairwise/agreement output); this class exposes them
|
|
5
|
+
through the pluggable reporter seam so ``--report markdown`` and a custom
|
|
6
|
+
reporter share one interface.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from evalkit import models, report
|
|
10
|
+
from evalkit.reporters import base
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@base.register('markdown')
|
|
14
|
+
class MarkdownReporter:
|
|
15
|
+
"""Render scorecards and comparisons as Markdown sections."""
|
|
16
|
+
|
|
17
|
+
def scorecard(self, scorecard: models.Scorecard) -> str:
|
|
18
|
+
return report.render_scorecard(scorecard)
|
|
19
|
+
|
|
20
|
+
def comparison(self, comparison: models.Comparison) -> str:
|
|
21
|
+
return report.render_comparison(comparison)
|
|
22
|
+
|
|
23
|
+
def agreement(self, result: models.AgreementResult) -> str:
|
|
24
|
+
return report.render_agreement(result)
|
|
25
|
+
|
|
26
|
+
def preferences(self, result: models.PreferenceResult) -> str:
|
|
27
|
+
return report.render_preferences(result)
|
|
28
|
+
|
|
29
|
+
def pairwise_agreement(self, result: models.PairwiseAgreement) -> str:
|
|
30
|
+
return report.render_pairwise_agreement(result)
|
|
31
|
+
|
|
32
|
+
def run(self, run: models.RunResult) -> str:
|
|
33
|
+
"""Aggregate scorecard + a per-case matrix and failure notes."""
|
|
34
|
+
lines = [report.render_scorecard(run.scorecard)]
|
|
35
|
+
metrics, rows = base.per_case_matrix(run)
|
|
36
|
+
if rows:
|
|
37
|
+
lines += [
|
|
38
|
+
'',
|
|
39
|
+
'#### Per-case',
|
|
40
|
+
'',
|
|
41
|
+
'| case | ' + ' | '.join(metrics) + ' |',
|
|
42
|
+
'| --- |' + ' ---: |' * len(metrics),
|
|
43
|
+
]
|
|
44
|
+
notes: list[str] = []
|
|
45
|
+
for row in rows:
|
|
46
|
+
if row['error']:
|
|
47
|
+
cells = ' | '.join(['err'] * len(metrics))
|
|
48
|
+
notes.append(f'- `{row["label"]}` errored: {row["error"]}')
|
|
49
|
+
else:
|
|
50
|
+
cells = ' | '.join(
|
|
51
|
+
_cell(row['cells'].get(m)) for m in metrics
|
|
52
|
+
)
|
|
53
|
+
for metric in metrics:
|
|
54
|
+
score = row['cells'].get(metric)
|
|
55
|
+
if score and score.passed is False and score.detail:
|
|
56
|
+
notes.append(
|
|
57
|
+
f'- `{row["label"]}` / {metric}: '
|
|
58
|
+
f'{score.detail}'
|
|
59
|
+
)
|
|
60
|
+
lines.append(f'| {row["label"]} | {cells} |')
|
|
61
|
+
if notes:
|
|
62
|
+
lines += ['', '**Notes**', '', *notes]
|
|
63
|
+
return '\n'.join(lines)
|
|
64
|
+
|
|
65
|
+
def document(self, body: str) -> str:
|
|
66
|
+
# Markdown needs no wrapper; a report is just its section(s).
|
|
67
|
+
return body
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _cell(score: models.Score | None) -> str:
|
|
71
|
+
if score is None or score.value is None:
|
|
72
|
+
return 'n/a'
|
|
73
|
+
text = f'{score.value:.3f}'
|
|
74
|
+
return f'**{text}**' if score.passed is False else text
|
evalkit/retry.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Shared transient-failure retry policy and backoff.
|
|
2
|
+
|
|
3
|
+
Two call sites use it with different failure signals but one timing schedule:
|
|
4
|
+
|
|
5
|
+
* the **runner** retries an adapter ``Output`` flagged ``retryable`` (a
|
|
6
|
+
value-signalled failure - the adapter contract never raises);
|
|
7
|
+
* the **LLM judge** retries a client call that *raised* a transient SDK error
|
|
8
|
+
(a 429, a 5xx, a timeout).
|
|
9
|
+
|
|
10
|
+
The exponential-backoff-with-jitter schedule is common, so it lives here; each
|
|
11
|
+
site keeps its own loop and predicate. Defaults are a no-op
|
|
12
|
+
(``max_attempts: 1``), so opting in is per suite.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import asyncio
|
|
16
|
+
import random
|
|
17
|
+
|
|
18
|
+
import pydantic
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RetryConfig(pydantic.BaseModel):
|
|
22
|
+
"""Transient-failure retry policy for a run.
|
|
23
|
+
|
|
24
|
+
Defaults are a no-op (``max_attempts: 1``) so existing suites are
|
|
25
|
+
unchanged; opt in per suite. Backoff before attempt *k+1* is
|
|
26
|
+
``backoff_base * 2**(k-1)`` seconds, capped at ``backoff_max``, with
|
|
27
|
+
+/- ``jitter`` fractional randomization to avoid a thundering herd.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
max_attempts: int = 1
|
|
31
|
+
backoff_base: float = 0.5
|
|
32
|
+
backoff_max: float = 30.0
|
|
33
|
+
jitter: float = 0.1
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def backoff_delay(attempt: int, cfg: RetryConfig) -> float:
|
|
37
|
+
"""Backoff (seconds) before retrying after ``attempt`` (1-based) failed."""
|
|
38
|
+
delay = min(cfg.backoff_max, cfg.backoff_base * 2 ** (attempt - 1))
|
|
39
|
+
if cfg.jitter:
|
|
40
|
+
# +/- jitter fraction; non-crypto, only spreads retry timing.
|
|
41
|
+
delay += delay * cfg.jitter * (random.random() * 2 - 1) # noqa: S311
|
|
42
|
+
return max(0.0, delay)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def is_transient_exc(exc: BaseException) -> bool:
|
|
46
|
+
"""Whether an exception looks transient (worth retrying).
|
|
47
|
+
|
|
48
|
+
SDK-agnostic: an Anthropic/OpenAI ``RateLimitError`` / ``APITimeoutError``
|
|
49
|
+
/ ``APIConnectionError`` / ``InternalServerError`` (and 429/5xx status
|
|
50
|
+
carriers) all match, without importing either SDK. A non-transient error
|
|
51
|
+
(bad request, auth, a programming bug) does not, so it re-raises at once.
|
|
52
|
+
"""
|
|
53
|
+
status = getattr(exc, 'status_code', None)
|
|
54
|
+
if status is None:
|
|
55
|
+
status = getattr(exc, 'status', None)
|
|
56
|
+
if isinstance(status, int) and (status == 429 or status >= 500):
|
|
57
|
+
return True
|
|
58
|
+
name = type(exc).__name__
|
|
59
|
+
markers = (
|
|
60
|
+
'RateLimit',
|
|
61
|
+
'Timeout',
|
|
62
|
+
'Connection',
|
|
63
|
+
'InternalServer',
|
|
64
|
+
'ServiceUnavailable',
|
|
65
|
+
'Overloaded',
|
|
66
|
+
)
|
|
67
|
+
return any(marker in name for marker in markers)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
async def call_with_retry(thunk, cfg: RetryConfig):
|
|
71
|
+
"""Await ``thunk()``, retrying transient exceptions with backoff.
|
|
72
|
+
|
|
73
|
+
Re-raises immediately for a non-transient exception, and re-raises the
|
|
74
|
+
last one once the attempt budget is spent - so a caller's existing
|
|
75
|
+
error handling still sees a genuine, sustained failure.
|
|
76
|
+
"""
|
|
77
|
+
attempt = 1
|
|
78
|
+
while True:
|
|
79
|
+
try:
|
|
80
|
+
return await thunk()
|
|
81
|
+
except Exception as exc:
|
|
82
|
+
if not is_transient_exc(exc) or attempt >= cfg.max_attempts:
|
|
83
|
+
raise
|
|
84
|
+
await asyncio.sleep(backoff_delay(attempt, cfg))
|
|
85
|
+
attempt += 1
|