evalcore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,426 @@
1
+ """HTML reporter - standalone, self-contained report documents.
2
+
3
+ Renders a single scorecard or a candidate-vs-baseline comparison as an HTML
4
+ fragment, and wraps fragments in a minimal styled document via ``document``.
5
+ Values are escaped; the output has no external assets, so it drops straight
6
+ into a CI artifact, a PR attachment, or an inline preview.
7
+ """
8
+
9
+ import base64
10
+ import html as _html
11
+ import json
12
+ import pathlib
13
+
14
+ from evalkit import models
15
+ from evalkit.reporters import base
16
+
17
+
18
+ def _dump(fields: dict) -> str:
19
+ try:
20
+ return json.dumps(fields, indent=2, default=str)[:6000]
21
+ except TypeError, ValueError:
22
+ return str(fields)[:6000]
23
+
24
+
25
+ _IMG_EXT = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
26
+
27
+
28
+ def _embed_artifact(name: str, path: str) -> str:
29
+ """Embed one saved artifact inline so the report stays self-contained.
30
+
31
+ Images become base64 ``data:`` URIs, HTML renders in a sandboxed iframe,
32
+ PDFs embed via a data URI, and anything else falls back to its decoded
33
+ text. A path that can't be read degrades to a note rather than erroring,
34
+ so a report built on a machine without the artifacts still renders.
35
+ """
36
+ p = pathlib.Path(str(path))
37
+ cap = (
38
+ f'<figcaption class="meta">{_esc(name)} &middot; '
39
+ f'<code>{_esc(p.name)}</code></figcaption>'
40
+ )
41
+ try:
42
+ raw = p.read_bytes()
43
+ except OSError:
44
+ return (
45
+ f'<figure class="art">{cap}'
46
+ f'<p class="meta">missing: <code>{_esc(path)}</code></p></figure>'
47
+ )
48
+ ext = p.suffix.lower()
49
+ if ext in _IMG_EXT:
50
+ mime = (
51
+ 'image/svg+xml'
52
+ if ext == '.svg'
53
+ else 'image/jpeg'
54
+ if ext in ('.jpg', '.jpeg')
55
+ else f'image/{ext.lstrip(".")}'
56
+ )
57
+ b64 = base64.b64encode(raw).decode('ascii')
58
+ media = (
59
+ f'<img class="art-img" alt="{_esc(name)}" '
60
+ f'src="data:{mime};base64,{b64}">'
61
+ )
62
+ elif ext in ('.html', '.htm'):
63
+ media = _iframe(raw.decode('utf-8', 'replace'))
64
+ elif ext == '.pdf':
65
+ b64 = base64.b64encode(raw).decode('ascii')
66
+ media = (
67
+ f'<iframe class="art-frame" '
68
+ f'src="data:application/pdf;base64,{b64}"></iframe>'
69
+ )
70
+ else:
71
+ media = f'<pre>{_esc(raw.decode("utf-8", "replace")[:6000])}</pre>'
72
+ return f'<figure class="art">{cap}{media}</figure>'
73
+
74
+
75
+ def _iframe(content: str) -> str:
76
+ """A sandboxed iframe rendering an HTML fragment/document."""
77
+ return (
78
+ f'<iframe class="art-frame" sandbox loading="lazy" '
79
+ f'srcdoc="{_html.escape(content, quote=True)}"></iframe>'
80
+ )
81
+
82
+
83
+ def _pt(value) -> str:
84
+ """A raw judge point: integer-clean when whole (``4``, not ``4.0``)."""
85
+ return '&ndash;' if value is None else f'{value:g}'
86
+
87
+
88
+ def _judges_html(score: models.Score) -> str:
89
+ """The per-judge breakdown behind an LLM-judge ``overall`` score:
90
+ each judge's raw per-dimension points side by side, its own normalized
91
+ overall, and its free-text rationale - the ``why`` the aggregate hides."""
92
+ judges = score.judges
93
+ if not judges:
94
+ return ''
95
+ dims = list(judges[0].points)
96
+ head = ''.join(f'<th class="num">{_esc(j.key)}</th>' for j in judges)
97
+ body = ''
98
+ for dim in dims:
99
+ cells = ''.join(
100
+ f'<td class="num">{_pt(j.points.get(dim))}</td>' for j in judges
101
+ )
102
+ body += f'<tr><td>{_esc(dim)}</td>{cells}</tr>'
103
+ body += (
104
+ '<tr><td><b>overall (0..1)</b></td>'
105
+ + ''.join(f'<td class="num">{_fmt(j.overall)}</td>' for j in judges)
106
+ + '</tr>'
107
+ )
108
+ rationales = ''.join(
109
+ f'<p class="rationale"><b>{_esc(j.key)}'
110
+ f'{" @" + _esc(j.version) if j.version else ""}:</b> '
111
+ f'{_esc(j.rationale)}</p>'
112
+ for j in judges
113
+ if j.rationale
114
+ )
115
+ return (
116
+ f'<div class="judgment"><h4>Judge panel &mdash; '
117
+ f'<code>{_esc(score.grader)}</code> &middot; raw points</h4>'
118
+ f'<table class="judges"><thead><tr><th>dimension</th>{head}</tr>'
119
+ f'</thead><tbody>{body}</tbody></table>{rationales}</div>'
120
+ )
121
+
122
+
123
+ def _notes_html(cells: dict, error: str | None) -> str:
124
+ """Surface stored per-case ``detail`` the matrix can't: the ``why`` of
125
+ each failing deterministic check, plus an errored invocation's message."""
126
+ items = []
127
+ if error:
128
+ items.append(f'<li class="bad">errored: {_esc(error)}</li>')
129
+ for sc in cells.values():
130
+ if sc.passed is False and sc.detail:
131
+ items.append(
132
+ f'<li><code>{_esc(sc.metric)}</code> &mdash; {_esc(sc.detail)}'
133
+ '</li>'
134
+ )
135
+ if not items:
136
+ return ''
137
+ return f'<h4>Notes</h4><ul class="notes">{"".join(items)}</ul>'
138
+
139
+
140
+ _CSS = (
141
+ 'body{font:15px/1.5 system-ui,-apple-system,Segoe UI,Roboto,sans-serif;'
142
+ 'margin:0;padding:24px;color:#1a1d21;background:#f6f7f9}'
143
+ '.card{max-width:900px;margin:0 auto 20px;background:#fff;'
144
+ 'border:1px solid #e3e6ea;border-radius:12px;padding:20px 24px;'
145
+ 'box-shadow:0 1px 3px rgba(20,25,35,.06)}'
146
+ 'h2{margin:0 0 6px;font-size:20px}h3{margin:18px 0 8px;font-size:15px}'
147
+ 'code{background:#eef0f3;border-radius:5px;padding:1px 6px;font-size:.9em}'
148
+ '.meta{color:#68707a;font-size:13px;margin:0 0 14px}'
149
+ 'table{width:100%;border-collapse:collapse;font-size:14px}'
150
+ 'th,td{text-align:left;padding:7px 10px;border-bottom:1px solid #eceef1}'
151
+ 'th{color:#68707a;font-weight:600}.num{text-align:right;'
152
+ 'font-variant-numeric:tabular-nums}'
153
+ '.up{color:#1f8f5f}.down{color:#c0392b}'
154
+ '.verdict{display:inline-block;border-radius:6px;padding:2px 10px;color:#fff}'
155
+ '.verdict.pass{background:#1f8f5f}.verdict.warn{background:#d0870b}'
156
+ '.verdict.fail{background:#c0392b}'
157
+ '.guards{list-style:none;padding:0;margin:0}'
158
+ '.guards li{padding:4px 0;font-size:14px}'
159
+ '.guards li.ok::before{content:"\\2713 ";color:#1f8f5f}'
160
+ '.guards li.breach::before{content:"\\2717 ";color:#c0392b}'
161
+ 'td.fail{color:#c0392b;font-weight:600}'
162
+ 'details{margin:6px 0}summary{cursor:pointer;color:#68707a;font-size:13px}'
163
+ 'pre{background:#f0f2f4;border-radius:6px;padding:10px;overflow:auto;'
164
+ 'max-height:340px;white-space:pre-wrap;word-break:break-word;'
165
+ 'font:12px ui-monospace,SFMono-Regular,Menlo,monospace}'
166
+ '.scroll{overflow-x:auto}.scroll table{min-width:100%;width:max-content}'
167
+ '.arts{display:grid;gap:14px;margin:10px 0}'
168
+ 'figure.art{margin:0;border:1px solid #e3e6ea;border-radius:8px;'
169
+ 'padding:10px 12px;background:#fafbfc}'
170
+ 'figure.art figcaption{margin:0 0 8px}'
171
+ '.art-img{display:block;max-width:100%;height:auto;border-radius:6px;'
172
+ 'border:1px solid #e3e6ea;background:#fff}'
173
+ '.art-frame{width:100%;height:520px;border:1px solid #e3e6ea;'
174
+ 'border-radius:6px;background:#fff}'
175
+ 'details.raw{margin-top:10px}'
176
+ 'h4{margin:14px 0 6px;font-size:14px}'
177
+ '.judgment{margin:10px 0}'
178
+ 'table.judges{width:auto;min-width:60%}'
179
+ 'table.judges td:first-child,table.judges th:first-child{font-weight:600}'
180
+ '.rationale{font-size:13px;color:#333;margin:6px 0;padding:8px 10px;'
181
+ 'background:#f0f2f4;border-radius:6px}'
182
+ '.notes{margin:4px 0;padding-left:18px;font-size:13px}'
183
+ '.notes li{padding:2px 0}.notes li.bad{color:#c0392b;font-weight:600}'
184
+ )
185
+
186
+
187
+ def _fmt(value: float | None) -> str:
188
+ return 'n/a' if value is None else f'{value:.4f}'
189
+
190
+
191
+ def _fmt_delta(value: float | None) -> str:
192
+ return 'n/a' if value is None else f'{value:+.4f}'
193
+
194
+
195
+ def _esc(value) -> str:
196
+ return _html.escape(str(value))
197
+
198
+
199
+ @base.register('html')
200
+ class HtmlReporter:
201
+ """Render scorecards and comparisons as self-contained HTML."""
202
+
203
+ def scorecard(self, scorecard: models.Scorecard) -> str:
204
+ rows = ''.join(
205
+ f'<tr><td>{_esc(m.metric)}</td>'
206
+ f'<td class="num">{_fmt(m.value)}</td></tr>'
207
+ for m in scorecard.metrics.values()
208
+ )
209
+ meta = (
210
+ f'model <code>{_esc(scorecard.model_id)}</code> &middot; '
211
+ f'mode <code>{_esc(scorecard.mode)}</code> &middot; '
212
+ f'dataset <code>{_esc(scorecard.dataset_version)}</code> &middot; '
213
+ f'{scorecard.n_cases}&times;{scorecard.n_samples} cases'
214
+ )
215
+ return (
216
+ f'<section class="card"><h2>{_esc(scorecard.project)}/'
217
+ f'{_esc(scorecard.suite)} &middot; '
218
+ f'<code>{_esc(scorecard.variant.name)}</code></h2>'
219
+ f'<p class="meta">{meta}</p>'
220
+ f'<table><thead><tr><th>metric</th>'
221
+ f'<th class="num">value</th></tr></thead>'
222
+ f'<tbody>{rows}</tbody></table></section>'
223
+ )
224
+
225
+ def comparison(self, comparison: models.Comparison) -> str:
226
+ drows = ''
227
+ for delta in comparison.deltas:
228
+ mark = (
229
+ f' <b>({_esc(comparison.win)})</b>'
230
+ if delta.metric == comparison.win_metric
231
+ else ''
232
+ )
233
+ direction = (
234
+ 'up'
235
+ if (delta.delta or 0) > 0
236
+ else 'down'
237
+ if (delta.delta or 0) < 0
238
+ else ''
239
+ )
240
+ drows += (
241
+ f'<tr><td>{_esc(delta.metric)}{mark}</td>'
242
+ f'<td class="num">{_fmt(delta.baseline)}</td>'
243
+ f'<td class="num">{_fmt(delta.candidate)}</td>'
244
+ f'<td class="num {direction}">{_fmt_delta(delta.delta)}</td>'
245
+ f'</tr>'
246
+ )
247
+ guards = ''
248
+ if comparison.guardrails:
249
+ items = ''.join(
250
+ f'<li class="{"ok" if g.passed else "breach"}">'
251
+ f'<code>{_esc(g.metric)}</code> &mdash; {_esc(g.detail)}</li>'
252
+ for g in comparison.guardrails
253
+ )
254
+ guards = f'<h3>Guardrails</h3><ul class="guards">{items}</ul>'
255
+ return (
256
+ f'<section class="card">'
257
+ f'<h2><span class="verdict {_esc(comparison.verdict)}">'
258
+ f'{_esc(comparison.verdict.upper())}</span> '
259
+ f'{_esc(comparison.project)}/{_esc(comparison.suite)}</h2>'
260
+ f'<p><code>{_esc(comparison.candidate_variant)}</code> vs '
261
+ f'<code>{_esc(comparison.baseline_variant)}</code> &mdash; '
262
+ f'{_esc(comparison.summary)}</p>'
263
+ f'<table><thead><tr><th>metric</th>'
264
+ f'<th class="num">baseline</th><th class="num">candidate</th>'
265
+ f'<th class="num">delta</th></tr></thead>'
266
+ f'<tbody>{drows}</tbody></table>{guards}</section>'
267
+ )
268
+
269
+ def agreement(self, result: models.AgreementResult) -> str:
270
+ rows = ''.join(
271
+ f'<tr><td>{_esc(d.dimension)}</td>'
272
+ f'<td class="num">{d.n}</td>'
273
+ f'<td class="num">{_fmt(d.human_mean)}</td>'
274
+ f'<td class="num">{_fmt(d.judge_mean)}</td>'
275
+ f'<td class="num">{_fmt(d.mae)}</td>'
276
+ f'<td class="num">{_fmt(d.correlation)}</td></tr>'
277
+ for d in result.dimensions
278
+ )
279
+ return (
280
+ f'<section class="card"><h2>Judge&harr;human agreement &middot; '
281
+ f'<code>{_esc(result.judge_name)}</code></h2>'
282
+ f'<p class="meta">{result.n_ratings} ratings from '
283
+ f'{result.n_raters} rater(s), scale 1..{result.scale} &middot; '
284
+ f'overall MAE {_fmt(result.overall_mae)}, '
285
+ f'r {_fmt(result.overall_correlation)}</p>'
286
+ f'<table><thead><tr><th>dimension</th><th class="num">n</th>'
287
+ f'<th class="num">human</th><th class="num">judge</th>'
288
+ f'<th class="num">MAE</th><th class="num">corr</th></tr></thead>'
289
+ f'<tbody>{rows}</tbody></table></section>'
290
+ )
291
+
292
+ def preferences(self, result: models.PreferenceResult) -> str:
293
+ rows = ''.join(
294
+ f'<tr><td>{_esc(d.dimension)}</td>'
295
+ f'<td class="num">{d.n}</td>'
296
+ f'<td class="num">{d.a_wins}</td>'
297
+ f'<td class="num">{d.b_wins}</td>'
298
+ f'<td class="num">{d.ties}</td>'
299
+ f'<td class="num">{_fmt(d.win_rate_a)}</td></tr>'
300
+ for d in result.dimensions
301
+ )
302
+ body = (
303
+ f'<table><thead><tr><th>dimension</th><th class="num">n</th>'
304
+ f'<th class="num">A wins</th><th class="num">B wins</th>'
305
+ f'<th class="num">ties</th><th class="num">A win-rate</th>'
306
+ f'</tr></thead><tbody>{rows}</tbody></table>'
307
+ if rows
308
+ else ''
309
+ )
310
+ return (
311
+ f'<section class="card"><h2>Human preferences &middot; '
312
+ f'{_esc(result.project)}/{_esc(result.suite)}</h2>'
313
+ f'<p class="meta"><code>{_esc(result.variant_a)}</code> (A) vs '
314
+ f'<code>{_esc(result.variant_b)}</code> (B) &middot; '
315
+ f'{result.n} preferences from {result.n_raters} rater(s)</p>'
316
+ f'<p><b>A win-rate {_fmt(result.win_rate_a)}</b> (ties = half) '
317
+ f'&middot; A wins {result.a_wins} &middot; B wins {result.b_wins} '
318
+ f'&middot; ties {result.ties}</p>{body}</section>'
319
+ )
320
+
321
+ def pairwise_agreement(self, result: models.PairwiseAgreement) -> str:
322
+ rows = ''.join(
323
+ f'<tr><td>{_esc(c.case_id)}</td>'
324
+ f'<td class="num">{_esc(c.human)}</td>'
325
+ f'<td class="num">{_esc(c.judge)}</td>'
326
+ f'<td class="{"" if c.agree else "fail"}">'
327
+ f'{"yes" if c.agree else "NO"}</td></tr>'
328
+ for c in result.outcomes
329
+ )
330
+ return (
331
+ f'<section class="card"><h2>Pairwise agreement &middot; human vs '
332
+ f'<code>{_esc(result.judge_name)}</code></h2>'
333
+ f'<p class="meta"><code>{_esc(result.variant_a)}</code> (A) vs '
334
+ f'<code>{_esc(result.variant_b)}</code> (B)</p>'
335
+ f'<p><b>agreement {_fmt(result.agreement_rate)}</b> '
336
+ f'({result.agree}/{result.n} cases pick the same winner) &middot; '
337
+ f'A win-rate: human {_fmt(result.human_win_rate_a)}, '
338
+ f'judge {_fmt(result.judge_win_rate_a)}</p>'
339
+ f'<table><thead><tr><th>case</th><th class="num">human</th>'
340
+ f'<th class="num">judge</th><th>agree</th></tr></thead>'
341
+ f'<tbody>{rows}</tbody></table></section>'
342
+ )
343
+
344
+ def run(self, run: models.RunResult) -> str:
345
+ """Scorecard summary + a per-case table and collapsible outputs.
346
+
347
+ Returns a fragment; :func:`base.render_run` wraps it in a document."""
348
+ body = self.scorecard(run.scorecard)
349
+ metrics, rows = base.per_case_matrix(run)
350
+ if rows:
351
+ head = ''.join(f'<th class="num">{_esc(m)}</th>' for m in metrics)
352
+ trs = ''
353
+ details = ''
354
+ for row in rows:
355
+ if row['error']:
356
+ cells = (
357
+ f'<td colspan="{len(metrics)}" class="fail">'
358
+ f'error: {_esc(row["error"])}</td>'
359
+ )
360
+ else:
361
+ cells = ''
362
+ for metric in metrics:
363
+ score = row['cells'].get(metric)
364
+ value = _fmt(score.value if score else None)
365
+ cls = (
366
+ 'num fail'
367
+ if score and score.passed is False
368
+ else 'num'
369
+ )
370
+ cells += f'<td class="{cls}">{value}</td>'
371
+ trs += f'<tr><td>{_esc(row["label"])}</td>{cells}</tr>'
372
+ out = row['output']
373
+ fields = out.fields or {}
374
+ embeds = [
375
+ _embed_artifact(name, path)
376
+ for name, path in (out.artifacts or {}).items()
377
+ ]
378
+ # Fallback: render an inline HTML field when nothing was saved
379
+ # to disk, so offline/replay runs still show the rendered form.
380
+ if not embeds and isinstance(fields.get('html'), str):
381
+ embeds.append(
382
+ '<figure class="art"><figcaption class="meta">'
383
+ 'html (inline)</figcaption>'
384
+ f'{_iframe(fields["html"])}</figure>'
385
+ )
386
+ arts = (
387
+ f'<div class="arts">{"".join(embeds)}</div>'
388
+ if embeds
389
+ else ''
390
+ )
391
+ judgments = ''.join(
392
+ _judges_html(sc)
393
+ for sc in row['cells'].values()
394
+ if sc.judges
395
+ )
396
+ notes = _notes_html(row['cells'], row['error'])
397
+ lat = (
398
+ f' &middot; <span class="meta">'
399
+ f'{out.latency_ms / 1000:.1f}s</span>'
400
+ if out.latency_ms is not None
401
+ else ''
402
+ )
403
+ details += (
404
+ f'<details><summary>{_esc(row["label"])}{lat}</summary>'
405
+ + arts
406
+ + judgments
407
+ + notes
408
+ + '<details class="raw"><summary>output fields (JSON)'
409
+ f'</summary><pre>{_esc(_dump(fields))}</pre></details>'
410
+ + '</details>'
411
+ )
412
+ body += (
413
+ '<section class="card"><h3>Per-case</h3>'
414
+ f'<div class="scroll"><table><thead><tr><th>case</th>{head}'
415
+ f'</tr></thead><tbody>{trs}</tbody></table></div>'
416
+ f'<h3>Outputs</h3>{details}</section>'
417
+ )
418
+ return body
419
+
420
+ def document(self, body: str) -> str:
421
+ return (
422
+ '<!doctype html><html lang="en"><head><meta charset="utf-8">'
423
+ '<meta name="viewport" content="width=device-width,'
424
+ 'initial-scale=1"><title>evalkit report</title>'
425
+ f'<style>{_CSS}</style></head><body>{body}</body></html>'
426
+ )
@@ -0,0 +1,74 @@
1
+ """Markdown reporter - the default. Delegates to :mod:`evalkit.report`.
2
+
3
+ The Markdown renderers live in ``evalkit.report`` (and are re-used directly
4
+ by the CLI's sweep/pairwise/agreement output); this class exposes them
5
+ through the pluggable reporter seam so ``--report markdown`` and a custom
6
+ reporter share one interface.
7
+ """
8
+
9
+ from evalkit import models, report
10
+ from evalkit.reporters import base
11
+
12
+
13
+ @base.register('markdown')
14
+ class MarkdownReporter:
15
+ """Render scorecards and comparisons as Markdown sections."""
16
+
17
+ def scorecard(self, scorecard: models.Scorecard) -> str:
18
+ return report.render_scorecard(scorecard)
19
+
20
+ def comparison(self, comparison: models.Comparison) -> str:
21
+ return report.render_comparison(comparison)
22
+
23
+ def agreement(self, result: models.AgreementResult) -> str:
24
+ return report.render_agreement(result)
25
+
26
+ def preferences(self, result: models.PreferenceResult) -> str:
27
+ return report.render_preferences(result)
28
+
29
+ def pairwise_agreement(self, result: models.PairwiseAgreement) -> str:
30
+ return report.render_pairwise_agreement(result)
31
+
32
+ def run(self, run: models.RunResult) -> str:
33
+ """Aggregate scorecard + a per-case matrix and failure notes."""
34
+ lines = [report.render_scorecard(run.scorecard)]
35
+ metrics, rows = base.per_case_matrix(run)
36
+ if rows:
37
+ lines += [
38
+ '',
39
+ '#### Per-case',
40
+ '',
41
+ '| case | ' + ' | '.join(metrics) + ' |',
42
+ '| --- |' + ' ---: |' * len(metrics),
43
+ ]
44
+ notes: list[str] = []
45
+ for row in rows:
46
+ if row['error']:
47
+ cells = ' | '.join(['err'] * len(metrics))
48
+ notes.append(f'- `{row["label"]}` errored: {row["error"]}')
49
+ else:
50
+ cells = ' | '.join(
51
+ _cell(row['cells'].get(m)) for m in metrics
52
+ )
53
+ for metric in metrics:
54
+ score = row['cells'].get(metric)
55
+ if score and score.passed is False and score.detail:
56
+ notes.append(
57
+ f'- `{row["label"]}` / {metric}: '
58
+ f'{score.detail}'
59
+ )
60
+ lines.append(f'| {row["label"]} | {cells} |')
61
+ if notes:
62
+ lines += ['', '**Notes**', '', *notes]
63
+ return '\n'.join(lines)
64
+
65
+ def document(self, body: str) -> str:
66
+ # Markdown needs no wrapper; a report is just its section(s).
67
+ return body
68
+
69
+
70
+ def _cell(score: models.Score | None) -> str:
71
+ if score is None or score.value is None:
72
+ return 'n/a'
73
+ text = f'{score.value:.3f}'
74
+ return f'**{text}**' if score.passed is False else text
evalkit/retry.py ADDED
@@ -0,0 +1,85 @@
1
+ """Shared transient-failure retry policy and backoff.
2
+
3
+ Two call sites use it with different failure signals but one timing schedule:
4
+
5
+ * the **runner** retries an adapter ``Output`` flagged ``retryable`` (a
6
+ value-signalled failure - the adapter contract never raises);
7
+ * the **LLM judge** retries a client call that *raised* a transient SDK error
8
+ (a 429, a 5xx, a timeout).
9
+
10
+ The exponential-backoff-with-jitter schedule is common, so it lives here; each
11
+ site keeps its own loop and predicate. Defaults are a no-op
12
+ (``max_attempts: 1``), so opting in is per suite.
13
+ """
14
+
15
+ import asyncio
16
+ import random
17
+
18
+ import pydantic
19
+
20
+
21
+ class RetryConfig(pydantic.BaseModel):
22
+ """Transient-failure retry policy for a run.
23
+
24
+ Defaults are a no-op (``max_attempts: 1``) so existing suites are
25
+ unchanged; opt in per suite. Backoff before attempt *k+1* is
26
+ ``backoff_base * 2**(k-1)`` seconds, capped at ``backoff_max``, with
27
+ +/- ``jitter`` fractional randomization to avoid a thundering herd.
28
+ """
29
+
30
+ max_attempts: int = 1
31
+ backoff_base: float = 0.5
32
+ backoff_max: float = 30.0
33
+ jitter: float = 0.1
34
+
35
+
36
+ def backoff_delay(attempt: int, cfg: RetryConfig) -> float:
37
+ """Backoff (seconds) before retrying after ``attempt`` (1-based) failed."""
38
+ delay = min(cfg.backoff_max, cfg.backoff_base * 2 ** (attempt - 1))
39
+ if cfg.jitter:
40
+ # +/- jitter fraction; non-crypto, only spreads retry timing.
41
+ delay += delay * cfg.jitter * (random.random() * 2 - 1) # noqa: S311
42
+ return max(0.0, delay)
43
+
44
+
45
+ def is_transient_exc(exc: BaseException) -> bool:
46
+ """Whether an exception looks transient (worth retrying).
47
+
48
+ SDK-agnostic: an Anthropic/OpenAI ``RateLimitError`` / ``APITimeoutError``
49
+ / ``APIConnectionError`` / ``InternalServerError`` (and 429/5xx status
50
+ carriers) all match, without importing either SDK. A non-transient error
51
+ (bad request, auth, a programming bug) does not, so it re-raises at once.
52
+ """
53
+ status = getattr(exc, 'status_code', None)
54
+ if status is None:
55
+ status = getattr(exc, 'status', None)
56
+ if isinstance(status, int) and (status == 429 or status >= 500):
57
+ return True
58
+ name = type(exc).__name__
59
+ markers = (
60
+ 'RateLimit',
61
+ 'Timeout',
62
+ 'Connection',
63
+ 'InternalServer',
64
+ 'ServiceUnavailable',
65
+ 'Overloaded',
66
+ )
67
+ return any(marker in name for marker in markers)
68
+
69
+
70
+ async def call_with_retry(thunk, cfg: RetryConfig):
71
+ """Await ``thunk()``, retrying transient exceptions with backoff.
72
+
73
+ Re-raises immediately for a non-transient exception, and re-raises the
74
+ last one once the attempt budget is spent - so a caller's existing
75
+ error handling still sees a genuine, sustained failure.
76
+ """
77
+ attempt = 1
78
+ while True:
79
+ try:
80
+ return await thunk()
81
+ except Exception as exc:
82
+ if not is_transient_exc(exc) or attempt >= cfg.max_attempts:
83
+ raise
84
+ await asyncio.sleep(backoff_delay(attempt, cfg))
85
+ attempt += 1