evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
evalkit/graders/judge.py
ADDED
|
@@ -0,0 +1,635 @@
|
|
|
1
|
+
"""LLM-as-judge (rubric scoring) grader, single judge or a panel.
|
|
2
|
+
|
|
3
|
+
The Tier-2 grader for subjective quality: one or more pinned, strong judge
|
|
4
|
+
models score each output 1..scale on a set of rubric dimensions via
|
|
5
|
+
structured output (Anthropic forced tool call, OpenAI ``json_schema``).
|
|
6
|
+
Scores are normalized to 0..1 and averaged by the runner, so a rubric
|
|
7
|
+
dimension - or the ``overall`` mean - can serve as a suite's win metric.
|
|
8
|
+
|
|
9
|
+
Configure ``judges`` for a **panel**: each judge scores independently, the
|
|
10
|
+
grader emits the panel mean per dimension plus each judge's ``overall``, an
|
|
11
|
+
inter-judge ``disagreement`` magnitude, and a ``flagged`` rate (cases where
|
|
12
|
+
judges disagree by >= ``disagreement_threshold`` raw points - the queue a
|
|
13
|
+
human reviewer should look at). A single judge (the default, back-compatible
|
|
14
|
+
config) emits just the per-dimension scores and ``overall``.
|
|
15
|
+
|
|
16
|
+
Judges take optional **image inputs** (``image_refs``) - screenshots or other
|
|
17
|
+
rendered artifacts the judge should see alongside the text (e.g. the builder's
|
|
18
|
+
rendered page). Images are sent only in live mode; replay is keyed by content.
|
|
19
|
+
|
|
20
|
+
Each judge call goes through a pluggable ``JudgeClient`` so the grader runs
|
|
21
|
+
live (``AnthropicJudgeClient`` / ``OpenAIJudgeClient``) or fully offline
|
|
22
|
+
against recorded judgments (``ReplayJudgeClient``), chosen by the run ``mode``
|
|
23
|
+
like the target adapter. Judge model + version are recorded on every score;
|
|
24
|
+
pin them and treat a judge change as a re-baseline event.
|
|
25
|
+
|
|
26
|
+
Pairwise (A-vs-B win-rate) judging is a cross-variant operation and lives in
|
|
27
|
+
:mod:`evalkit.pairwise`, not here - it needs both variants' per-case outputs
|
|
28
|
+
at once, which a per-case grader never sees.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
import base64
|
|
32
|
+
import functools
|
|
33
|
+
import os
|
|
34
|
+
import pathlib
|
|
35
|
+
import re
|
|
36
|
+
import typing
|
|
37
|
+
|
|
38
|
+
from evalkit import models, refs, retry
|
|
39
|
+
from evalkit.graders import base
|
|
40
|
+
|
|
41
|
+
_ENV_RE = re.compile(r'\$\{([A-Z0-9_]+)\}')
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _env_expand(value):
|
|
45
|
+
"""Expand ``${VAR}`` in a judge model string; passthrough otherwise.
|
|
46
|
+
|
|
47
|
+
Lets a panel judge's model be env-gated - an unset var expands to ''
|
|
48
|
+
and (in live mode) drops that judge, so 1-vs-2 judges is a config knob.
|
|
49
|
+
"""
|
|
50
|
+
if not isinstance(value, str):
|
|
51
|
+
return value
|
|
52
|
+
return _ENV_RE.sub(lambda m: os.environ.get(m.group(1), ''), value)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
_SYSTEM = (
|
|
56
|
+
'You are a strict, consistent evaluator. Score the content on each '
|
|
57
|
+
'dimension using the full 1..{scale} range, judging only against the '
|
|
58
|
+
'rubric and dimension descriptions - never reward length or verbosity. '
|
|
59
|
+
'Return only the structured per-dimension scores.'
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
_MEDIA_TYPES = {
|
|
63
|
+
'.png': 'image/png',
|
|
64
|
+
'.jpg': 'image/jpeg',
|
|
65
|
+
'.jpeg': 'image/jpeg',
|
|
66
|
+
'.webp': 'image/webp',
|
|
67
|
+
'.gif': 'image/gif',
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@typing.runtime_checkable
|
|
72
|
+
class JudgeClient(typing.Protocol):
|
|
73
|
+
"""Score one item; return ``{scores: {dim: int}, rationale: str}``."""
|
|
74
|
+
|
|
75
|
+
async def judge(
|
|
76
|
+
self,
|
|
77
|
+
*,
|
|
78
|
+
key: str,
|
|
79
|
+
system: str,
|
|
80
|
+
user: str,
|
|
81
|
+
dimensions: list[tuple[str, str]],
|
|
82
|
+
scale: int,
|
|
83
|
+
images: list[dict] | None = None,
|
|
84
|
+
) -> dict: ...
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _score_properties(dimensions: list[tuple[str, str]], scale: int) -> dict:
|
|
88
|
+
return {
|
|
89
|
+
key: {
|
|
90
|
+
'type': 'integer',
|
|
91
|
+
'minimum': 1,
|
|
92
|
+
'maximum': scale,
|
|
93
|
+
'description': desc,
|
|
94
|
+
}
|
|
95
|
+
for key, desc in dimensions
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class AnthropicJudgeClient:
|
|
100
|
+
"""Live judge backed by the Anthropic SDK (forced single tool call)."""
|
|
101
|
+
|
|
102
|
+
def __init__(
|
|
103
|
+
self,
|
|
104
|
+
model: str,
|
|
105
|
+
api_key_env: str = 'ANTHROPIC_API_KEY',
|
|
106
|
+
max_tokens: int = 1024,
|
|
107
|
+
timeout: float = 30.0,
|
|
108
|
+
):
|
|
109
|
+
self.model = model
|
|
110
|
+
self.api_key_env = api_key_env
|
|
111
|
+
self.max_tokens = max_tokens
|
|
112
|
+
self.timeout = timeout
|
|
113
|
+
|
|
114
|
+
def _tool(self, dimensions: list[tuple[str, str]], scale: int) -> dict:
|
|
115
|
+
return {
|
|
116
|
+
'name': 'score_rubric',
|
|
117
|
+
'description': 'Report per-dimension rubric scores.',
|
|
118
|
+
'input_schema': {
|
|
119
|
+
'type': 'object',
|
|
120
|
+
'properties': {
|
|
121
|
+
'scores': {
|
|
122
|
+
'type': 'object',
|
|
123
|
+
'properties': _score_properties(dimensions, scale),
|
|
124
|
+
'required': [k for k, _ in dimensions],
|
|
125
|
+
},
|
|
126
|
+
'rationale': {'type': 'string'},
|
|
127
|
+
},
|
|
128
|
+
'required': ['scores'],
|
|
129
|
+
},
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
async def judge(
|
|
133
|
+
self, *, key, system, user, dimensions, scale, images=None
|
|
134
|
+
) -> dict:
|
|
135
|
+
import anthropic
|
|
136
|
+
|
|
137
|
+
tool = self._tool(dimensions, scale)
|
|
138
|
+
content: list[dict] = [{'type': 'text', 'text': user}]
|
|
139
|
+
for image in images or []:
|
|
140
|
+
content.append(
|
|
141
|
+
{
|
|
142
|
+
'type': 'image',
|
|
143
|
+
'source': {
|
|
144
|
+
'type': 'base64',
|
|
145
|
+
'media_type': image['media_type'],
|
|
146
|
+
'data': image['data'],
|
|
147
|
+
},
|
|
148
|
+
}
|
|
149
|
+
)
|
|
150
|
+
# Context-managed so the connection pool closes inside the running
|
|
151
|
+
# event loop instead of at GC time after asyncio.run() tore it down.
|
|
152
|
+
async with anthropic.AsyncAnthropic(
|
|
153
|
+
api_key=os.environ[self.api_key_env]
|
|
154
|
+
) as client:
|
|
155
|
+
response = await client.messages.create(
|
|
156
|
+
model=self.model,
|
|
157
|
+
max_tokens=self.max_tokens,
|
|
158
|
+
temperature=0,
|
|
159
|
+
timeout=self.timeout,
|
|
160
|
+
system=system,
|
|
161
|
+
tools=[tool],
|
|
162
|
+
tool_choice={'type': 'tool', 'name': tool['name']},
|
|
163
|
+
messages=[{'role': 'user', 'content': content}],
|
|
164
|
+
)
|
|
165
|
+
for block in response.content:
|
|
166
|
+
if (
|
|
167
|
+
getattr(block, 'type', None) == 'tool_use'
|
|
168
|
+
and block.name == tool['name']
|
|
169
|
+
):
|
|
170
|
+
return dict(block.input)
|
|
171
|
+
return {}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class OpenAIJudgeClient:
|
|
175
|
+
"""Live judge via the OpenAI SDK (``response_format`` json_schema)."""
|
|
176
|
+
|
|
177
|
+
def __init__(
|
|
178
|
+
self,
|
|
179
|
+
model: str,
|
|
180
|
+
api_key_env: str = 'OPENAI_API_KEY',
|
|
181
|
+
max_tokens: int = 1024,
|
|
182
|
+
timeout: float = 30.0,
|
|
183
|
+
):
|
|
184
|
+
# Accept a 'provider:model' id (e.g. 'openai:gpt-4o'); SDK wants bare.
|
|
185
|
+
self.model = model.split(':', 1)[1] if ':' in model else model
|
|
186
|
+
self.api_key_env = api_key_env
|
|
187
|
+
self.max_tokens = max_tokens
|
|
188
|
+
self.timeout = timeout
|
|
189
|
+
|
|
190
|
+
def _response_format(
|
|
191
|
+
self, dimensions: list[tuple[str, str]], scale: int
|
|
192
|
+
) -> dict:
|
|
193
|
+
return {
|
|
194
|
+
'type': 'json_schema',
|
|
195
|
+
'json_schema': {
|
|
196
|
+
'name': 'score_rubric',
|
|
197
|
+
'strict': True,
|
|
198
|
+
'schema': {
|
|
199
|
+
'type': 'object',
|
|
200
|
+
'properties': {
|
|
201
|
+
'scores': {
|
|
202
|
+
'type': 'object',
|
|
203
|
+
'properties': _score_properties(dimensions, scale),
|
|
204
|
+
'required': [k for k, _ in dimensions],
|
|
205
|
+
'additionalProperties': False,
|
|
206
|
+
},
|
|
207
|
+
'rationale': {'type': 'string'},
|
|
208
|
+
},
|
|
209
|
+
'required': ['scores', 'rationale'],
|
|
210
|
+
'additionalProperties': False,
|
|
211
|
+
},
|
|
212
|
+
},
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
async def judge(
|
|
216
|
+
self, *, key, system, user, dimensions, scale, images=None
|
|
217
|
+
) -> dict:
|
|
218
|
+
import json
|
|
219
|
+
|
|
220
|
+
import openai
|
|
221
|
+
|
|
222
|
+
content: list[dict] = [{'type': 'text', 'text': user}]
|
|
223
|
+
for image in images or []:
|
|
224
|
+
data_url = f'data:{image["media_type"]};base64,{image["data"]}'
|
|
225
|
+
content.append(
|
|
226
|
+
{'type': 'image_url', 'image_url': {'url': data_url}}
|
|
227
|
+
)
|
|
228
|
+
async with openai.AsyncOpenAI(
|
|
229
|
+
api_key=os.environ[self.api_key_env]
|
|
230
|
+
) as client:
|
|
231
|
+
response = await client.chat.completions.create(
|
|
232
|
+
model=self.model,
|
|
233
|
+
max_tokens=self.max_tokens,
|
|
234
|
+
temperature=0,
|
|
235
|
+
timeout=self.timeout,
|
|
236
|
+
response_format=self._response_format(dimensions, scale),
|
|
237
|
+
messages=[
|
|
238
|
+
{'role': 'system', 'content': system},
|
|
239
|
+
{'role': 'user', 'content': content},
|
|
240
|
+
],
|
|
241
|
+
)
|
|
242
|
+
body = response.choices[0].message.content
|
|
243
|
+
try:
|
|
244
|
+
return json.loads(body) if body else {}
|
|
245
|
+
except ValueError as exc:
|
|
246
|
+
raise RuntimeError(f'judge returned non-JSON: {exc}') from exc
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class ReplayJudgeClient:
|
|
250
|
+
"""Offline judge: return recorded judgments keyed by the judged content.
|
|
251
|
+
|
|
252
|
+
Fixtures map the exact content string under judgement to a
|
|
253
|
+
``{scores: {dim: int}, rationale: str}`` dict, so different variants
|
|
254
|
+
(which produce different content) get different scores deterministically.
|
|
255
|
+
"""
|
|
256
|
+
|
|
257
|
+
def __init__(self, fixtures: str | dict):
|
|
258
|
+
if isinstance(fixtures, str):
|
|
259
|
+
from evalkit import loader
|
|
260
|
+
|
|
261
|
+
self._fixtures = loader.load_data_file(fixtures)
|
|
262
|
+
else:
|
|
263
|
+
self._fixtures = fixtures
|
|
264
|
+
|
|
265
|
+
async def judge(
|
|
266
|
+
self, *, key, system, user, dimensions, scale, images=None
|
|
267
|
+
) -> dict:
|
|
268
|
+
return dict(self._fixtures.get(key) or {})
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _load_images(refs_values: list) -> list[dict]:
|
|
272
|
+
"""Turn resolved image refs (file paths or dicts) into base64 blocks."""
|
|
273
|
+
images: list[dict] = []
|
|
274
|
+
for value in refs_values:
|
|
275
|
+
if value is None:
|
|
276
|
+
continue
|
|
277
|
+
if isinstance(value, dict) and 'data' in value:
|
|
278
|
+
images.append(
|
|
279
|
+
{
|
|
280
|
+
'media_type': value.get('media_type', 'image/png'),
|
|
281
|
+
'data': value['data'],
|
|
282
|
+
}
|
|
283
|
+
)
|
|
284
|
+
continue
|
|
285
|
+
if isinstance(value, str):
|
|
286
|
+
path = pathlib.Path(value)
|
|
287
|
+
if not path.is_file():
|
|
288
|
+
continue
|
|
289
|
+
media_type = _MEDIA_TYPES.get(path.suffix.lower(), 'image/png')
|
|
290
|
+
data = base64.b64encode(path.read_bytes()).decode('ascii')
|
|
291
|
+
images.append({'media_type': media_type, 'data': data})
|
|
292
|
+
return images
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@base.register('llm_judge')
|
|
296
|
+
class RubricJudge:
|
|
297
|
+
"""Score an output's text on rubric dimensions with an LLM judge/panel.
|
|
298
|
+
|
|
299
|
+
Config (suite ``graders`` entry):
|
|
300
|
+
content_ref: $ref to the text to judge (e.g. ``output.text``)
|
|
301
|
+
dimensions: list of {key, description}
|
|
302
|
+
scale: max score (default 5)
|
|
303
|
+
rubric: optional rubric text embedded in the judge prompt
|
|
304
|
+
context_refs: optional {label: $ref} of extra context for the judge
|
|
305
|
+
image_refs: optional {label: $ref} of images (paths in
|
|
306
|
+
``output.artifacts`` or inline data) shown to the judge
|
|
307
|
+
judges: optional list of {key, provider(anthropic|openai),
|
|
308
|
+
model, api_key_env?, replay_path?, judge_version?} - a
|
|
309
|
+
PANEL. Omit for a single judge configured by the
|
|
310
|
+
top-level model/replay_path/judge_version.
|
|
311
|
+
disagreement_threshold: raw-point spread at/above which a case is
|
|
312
|
+
flagged for human review (default 2)
|
|
313
|
+
model/replay_path/judge_version: single-judge shorthand
|
|
314
|
+
client: an explicit JudgeClient instance (tests); overrides mode
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
def __init__(
|
|
318
|
+
self,
|
|
319
|
+
content_ref: str,
|
|
320
|
+
dimensions: list[dict],
|
|
321
|
+
name: str = 'llm_judge',
|
|
322
|
+
scale: int = 5,
|
|
323
|
+
rubric: str | None = None,
|
|
324
|
+
context_refs: dict[str, str] | None = None,
|
|
325
|
+
image_refs: dict[str, str] | None = None,
|
|
326
|
+
judges: list[dict] | None = None,
|
|
327
|
+
disagreement_threshold: float = 2.0,
|
|
328
|
+
model: str | None = None,
|
|
329
|
+
judge_version: str = 'v1',
|
|
330
|
+
replay_path: str | None = None,
|
|
331
|
+
client: JudgeClient | None = None,
|
|
332
|
+
max_tokens: int = 1024,
|
|
333
|
+
):
|
|
334
|
+
self.name = name
|
|
335
|
+
self.content_ref = content_ref
|
|
336
|
+
self.dimensions = [
|
|
337
|
+
(d['key'], d.get('description', '')) for d in dimensions
|
|
338
|
+
]
|
|
339
|
+
self.scale = scale
|
|
340
|
+
self.rubric = rubric
|
|
341
|
+
self.context_refs = context_refs or {}
|
|
342
|
+
self.image_refs = image_refs or {}
|
|
343
|
+
self.disagreement_threshold = disagreement_threshold
|
|
344
|
+
self.max_tokens = max_tokens
|
|
345
|
+
self._explicit_client = client
|
|
346
|
+
self._mode = 'http'
|
|
347
|
+
# Normalize to a list of judge specs; a single judge is a panel of 1
|
|
348
|
+
# and produces output identical to the pre-panel grader.
|
|
349
|
+
if judges:
|
|
350
|
+
self.judges = [
|
|
351
|
+
{
|
|
352
|
+
'key': j.get('key') or j.get('provider') or 'judge',
|
|
353
|
+
'provider': j.get('provider', 'anthropic'),
|
|
354
|
+
'model': _env_expand(j.get('model')),
|
|
355
|
+
'api_key_env': j.get('api_key_env'),
|
|
356
|
+
'replay_path': j.get('replay_path') or replay_path,
|
|
357
|
+
'judge_version': j.get('judge_version', judge_version),
|
|
358
|
+
}
|
|
359
|
+
for j in judges
|
|
360
|
+
]
|
|
361
|
+
else:
|
|
362
|
+
self.judges = [
|
|
363
|
+
{
|
|
364
|
+
'key': 'judge',
|
|
365
|
+
'provider': 'anthropic',
|
|
366
|
+
'model': _env_expand(model),
|
|
367
|
+
'api_key_env': None,
|
|
368
|
+
'replay_path': replay_path,
|
|
369
|
+
'judge_version': judge_version,
|
|
370
|
+
}
|
|
371
|
+
]
|
|
372
|
+
self._clients: dict[str, JudgeClient] = {}
|
|
373
|
+
self._retry = retry.RetryConfig()
|
|
374
|
+
|
|
375
|
+
def set_retry(self, config: retry.RetryConfig) -> None:
|
|
376
|
+
"""Runner hook: back off + retry transient judge-client failures."""
|
|
377
|
+
self._retry = config
|
|
378
|
+
|
|
379
|
+
def _active_judges(self) -> list[dict]:
|
|
380
|
+
"""Judges usable in the current mode - a panel may drop some.
|
|
381
|
+
|
|
382
|
+
Only filters a PANEL (2+ judges): a judge with no model (live) or
|
|
383
|
+
no replay_path (replay) is dropped, so an env-gated GPT judge that
|
|
384
|
+
is unset simply leaves a Claude-only panel. A single judge is never
|
|
385
|
+
filtered, so a misconfigured lone judge still raises loudly.
|
|
386
|
+
"""
|
|
387
|
+
judges = self.judges
|
|
388
|
+
if self._explicit_client is not None or len(judges) <= 1:
|
|
389
|
+
return judges
|
|
390
|
+
if self._mode == 'replay':
|
|
391
|
+
return [j for j in judges if j['replay_path']] or judges
|
|
392
|
+
return [j for j in judges if j['model']] or judges
|
|
393
|
+
|
|
394
|
+
@property
|
|
395
|
+
def is_panel(self) -> bool:
|
|
396
|
+
return len(self._active_judges()) > 1
|
|
397
|
+
|
|
398
|
+
@property
|
|
399
|
+
def judge_version(self) -> str:
|
|
400
|
+
"""Provenance pin: ``key@version`` per configured judge (a panel
|
|
401
|
+
joins them, comma-separated).
|
|
402
|
+
|
|
403
|
+
The runner reads this onto ``Scorecard.judge_version`` so a judge
|
|
404
|
+
model / prompt / scale change surfaces as a re-baseline event rather
|
|
405
|
+
than hiding in each score's ``detail``. Uses the configured judges,
|
|
406
|
+
not the mode-filtered active set, so the pin is stable across
|
|
407
|
+
environments.
|
|
408
|
+
"""
|
|
409
|
+
return ','.join(
|
|
410
|
+
f'{j["key"]}@{j["judge_version"]}' for j in self.judges
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
def set_mode(self, mode: str) -> None:
|
|
414
|
+
"""Runner hook: pick the judge client to match the run mode."""
|
|
415
|
+
self._mode = mode
|
|
416
|
+
|
|
417
|
+
def _client_for(self, judge: dict) -> JudgeClient:
|
|
418
|
+
if self._explicit_client is not None:
|
|
419
|
+
return self._explicit_client
|
|
420
|
+
cached = self._clients.get(judge['key'])
|
|
421
|
+
if cached is not None:
|
|
422
|
+
return cached
|
|
423
|
+
if self._mode == 'replay':
|
|
424
|
+
if not judge['replay_path']:
|
|
425
|
+
raise ValueError(
|
|
426
|
+
f'judge {self.name}/{judge["key"]!r} needs a '
|
|
427
|
+
f'replay_path for replay mode'
|
|
428
|
+
)
|
|
429
|
+
client: JudgeClient = ReplayJudgeClient(judge['replay_path'])
|
|
430
|
+
elif judge['provider'] == 'openai':
|
|
431
|
+
if not judge['model']:
|
|
432
|
+
raise ValueError(f'judge {judge["key"]!r} needs a model')
|
|
433
|
+
client = OpenAIJudgeClient(
|
|
434
|
+
judge['model'],
|
|
435
|
+
api_key_env=judge['api_key_env'] or 'OPENAI_API_KEY',
|
|
436
|
+
max_tokens=self.max_tokens,
|
|
437
|
+
)
|
|
438
|
+
else:
|
|
439
|
+
if not judge['model']:
|
|
440
|
+
raise ValueError(f'judge {judge["key"]!r} needs a model')
|
|
441
|
+
client = AnthropicJudgeClient(
|
|
442
|
+
judge['model'],
|
|
443
|
+
api_key_env=judge['api_key_env'] or 'ANTHROPIC_API_KEY',
|
|
444
|
+
max_tokens=self.max_tokens,
|
|
445
|
+
)
|
|
446
|
+
self._clients[judge['key']] = client
|
|
447
|
+
return client
|
|
448
|
+
|
|
449
|
+
def _prompt(self, content: str, context: dict) -> str:
|
|
450
|
+
parts = []
|
|
451
|
+
if self.rubric:
|
|
452
|
+
parts.append(f'<rubric>\n{self.rubric}\n</rubric>')
|
|
453
|
+
for label, value in context.items():
|
|
454
|
+
parts.append(f'<{label}>\n{value}\n</{label}>')
|
|
455
|
+
parts.append(f'<content>\n{content}\n</content>')
|
|
456
|
+
return '\n\n'.join(parts)
|
|
457
|
+
|
|
458
|
+
def _normalize(self, raw) -> float | None:
|
|
459
|
+
if isinstance(raw, (int, float)) and not isinstance(raw, bool):
|
|
460
|
+
return raw / self.scale
|
|
461
|
+
return None
|
|
462
|
+
|
|
463
|
+
async def grade(
|
|
464
|
+
self, case: models.Case, output: models.Output
|
|
465
|
+
) -> list[models.Score]:
|
|
466
|
+
ctx = {
|
|
467
|
+
'input': case.input,
|
|
468
|
+
'expected': case.expected or {},
|
|
469
|
+
'output': output.fields,
|
|
470
|
+
'case': case.model_dump(),
|
|
471
|
+
'artifacts': output.artifacts,
|
|
472
|
+
}
|
|
473
|
+
content = refs.resolve_ref(ctx, self.content_ref)
|
|
474
|
+
if not isinstance(content, str) or not content:
|
|
475
|
+
return self._error_scores('no content to judge')
|
|
476
|
+
|
|
477
|
+
context = {
|
|
478
|
+
label: refs.resolve_ref(ctx, ref)
|
|
479
|
+
for label, ref in self.context_refs.items()
|
|
480
|
+
}
|
|
481
|
+
images = _load_images(
|
|
482
|
+
[refs.resolve_ref(ctx, ref) for ref in self.image_refs.values()]
|
|
483
|
+
)
|
|
484
|
+
system = _SYSTEM.format(scale=self.scale)
|
|
485
|
+
user = self._prompt(content, context)
|
|
486
|
+
|
|
487
|
+
# Build every client up front: a construction failure (e.g. no
|
|
488
|
+
# replay_path in replay mode) is a configuration error and must
|
|
489
|
+
# raise, unlike a failed judge call which degrades to error scores.
|
|
490
|
+
clients = [(j, self._client_for(j)) for j in self._active_judges()]
|
|
491
|
+
|
|
492
|
+
# raw[judge_key][dim] = integer score returned by that judge;
|
|
493
|
+
# rationales[judge_key] = that judge's free-text justification.
|
|
494
|
+
raw: dict[str, dict] = {}
|
|
495
|
+
rationales: dict[str, str | None] = {}
|
|
496
|
+
for judge, client in clients:
|
|
497
|
+
# A transient client failure (429/5xx/timeout) backs off and
|
|
498
|
+
# retries per the suite retry policy; a terminal one degrades to
|
|
499
|
+
# error scores as before. functools.partial (not a closure) keeps
|
|
500
|
+
# the retried call clean of loop-variable capture.
|
|
501
|
+
call = functools.partial(
|
|
502
|
+
client.judge,
|
|
503
|
+
key=content,
|
|
504
|
+
system=system,
|
|
505
|
+
user=user,
|
|
506
|
+
dimensions=self.dimensions,
|
|
507
|
+
scale=self.scale,
|
|
508
|
+
images=images,
|
|
509
|
+
)
|
|
510
|
+
try:
|
|
511
|
+
result = await retry.call_with_retry(call, self._retry)
|
|
512
|
+
except (KeyError, ValueError, RuntimeError) as exc:
|
|
513
|
+
return self._error_scores(f'judge error: {exc}')
|
|
514
|
+
result = result or {}
|
|
515
|
+
raw[judge['key']] = result.get('scores') or {}
|
|
516
|
+
rationales[judge['key']] = result.get('rationale')
|
|
517
|
+
|
|
518
|
+
return self._build_scores(case, raw, rationales)
|
|
519
|
+
|
|
520
|
+
def _build_scores(
|
|
521
|
+
self,
|
|
522
|
+
case: models.Case,
|
|
523
|
+
raw: dict[str, dict],
|
|
524
|
+
rationales: dict[str, str | None] | None = None,
|
|
525
|
+
) -> list[models.Score]:
|
|
526
|
+
active = self._active_judges()
|
|
527
|
+
versions = ','.join(f'{j["key"]}@{j["judge_version"]}' for j in active)
|
|
528
|
+
detail = f'judges={versions}'
|
|
529
|
+
rationales = rationales or {}
|
|
530
|
+
out: list[models.Score] = []
|
|
531
|
+
|
|
532
|
+
def score(
|
|
533
|
+
metric: str,
|
|
534
|
+
value: float | None,
|
|
535
|
+
judges: list[models.JudgeDetail] | None = None,
|
|
536
|
+
) -> models.Score:
|
|
537
|
+
return models.Score(
|
|
538
|
+
grader=self.name,
|
|
539
|
+
metric=metric,
|
|
540
|
+
value=value,
|
|
541
|
+
detail=detail,
|
|
542
|
+
case_id=case.id,
|
|
543
|
+
kind='per_case',
|
|
544
|
+
judges=judges or [],
|
|
545
|
+
)
|
|
546
|
+
|
|
547
|
+
# Per-judge breakdown behind the aggregate means: the raw 1..scale
|
|
548
|
+
# points each dimension got and the free-text rationale. Attached to
|
|
549
|
+
# the overall score so review can read back *why* without re-running.
|
|
550
|
+
judge_details: list[models.JudgeDetail] = []
|
|
551
|
+
for judge in active:
|
|
552
|
+
jkey = judge['key']
|
|
553
|
+
points = {k: raw.get(jkey, {}).get(k) for k, _ in self.dimensions}
|
|
554
|
+
norm = [
|
|
555
|
+
v
|
|
556
|
+
for k, _ in self.dimensions
|
|
557
|
+
if (v := self._normalize(raw.get(jkey, {}).get(k))) is not None
|
|
558
|
+
]
|
|
559
|
+
judge_details.append(
|
|
560
|
+
models.JudgeDetail(
|
|
561
|
+
key=jkey,
|
|
562
|
+
version=judge['judge_version'],
|
|
563
|
+
rationale=rationales.get(jkey),
|
|
564
|
+
points=points,
|
|
565
|
+
overall=sum(norm) / len(norm) if norm else None,
|
|
566
|
+
)
|
|
567
|
+
)
|
|
568
|
+
|
|
569
|
+
# Panel mean per dimension (identical to the single judge's value
|
|
570
|
+
# when there is only one).
|
|
571
|
+
panel_dims: list[float] = []
|
|
572
|
+
for key, _ in self.dimensions:
|
|
573
|
+
per_judge = [
|
|
574
|
+
self._normalize(raw[j['key']].get(key)) for j in active
|
|
575
|
+
]
|
|
576
|
+
present = [v for v in per_judge if v is not None]
|
|
577
|
+
value = sum(present) / len(present) if present else None
|
|
578
|
+
if value is not None:
|
|
579
|
+
panel_dims.append(value)
|
|
580
|
+
out.append(score(f'{self.name}.{key}', value))
|
|
581
|
+
|
|
582
|
+
overall = sum(panel_dims) / len(panel_dims) if panel_dims else None
|
|
583
|
+
out.append(
|
|
584
|
+
score(f'{self.name}.overall', overall, judges=judge_details)
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
if len(active) <= 1:
|
|
588
|
+
return out
|
|
589
|
+
|
|
590
|
+
# Per-judge overall, so a systematically generous judge is visible.
|
|
591
|
+
for jd in judge_details:
|
|
592
|
+
out.append(score(f'{self.name}.{jd.key}.overall', jd.overall))
|
|
593
|
+
|
|
594
|
+
# Inter-judge disagreement, in raw points, per dimension.
|
|
595
|
+
spreads: list[float] = []
|
|
596
|
+
for key, _ in self.dimensions:
|
|
597
|
+
raws = [
|
|
598
|
+
raw[j['key']].get(key)
|
|
599
|
+
for j in active
|
|
600
|
+
if isinstance(raw[j['key']].get(key), (int, float))
|
|
601
|
+
and not isinstance(raw[j['key']].get(key), bool)
|
|
602
|
+
]
|
|
603
|
+
if len(raws) >= 2:
|
|
604
|
+
spreads.append(max(raws) - min(raws))
|
|
605
|
+
mean_spread = sum(spreads) / len(spreads) if spreads else 0.0
|
|
606
|
+
max_spread = max(spreads) if spreads else 0.0
|
|
607
|
+
out.append(score(f'{self.name}.disagreement', mean_spread))
|
|
608
|
+
out.append(
|
|
609
|
+
score(
|
|
610
|
+
f'{self.name}.flagged',
|
|
611
|
+
1.0 if max_spread >= self.disagreement_threshold else 0.0,
|
|
612
|
+
)
|
|
613
|
+
)
|
|
614
|
+
return out
|
|
615
|
+
|
|
616
|
+
def _metric_names(self) -> list[str]:
|
|
617
|
+
active = self._active_judges()
|
|
618
|
+
names = [f'{self.name}.{k}' for k, _ in self.dimensions]
|
|
619
|
+
names.append(f'{self.name}.overall')
|
|
620
|
+
if len(active) > 1:
|
|
621
|
+
names += [f'{self.name}.{j["key"]}.overall' for j in active]
|
|
622
|
+
names += [f'{self.name}.disagreement', f'{self.name}.flagged']
|
|
623
|
+
return names
|
|
624
|
+
|
|
625
|
+
def _error_scores(self, detail: str) -> list[models.Score]:
|
|
626
|
+
return [
|
|
627
|
+
models.Score(
|
|
628
|
+
grader=self.name,
|
|
629
|
+
metric=metric,
|
|
630
|
+
value=None,
|
|
631
|
+
detail=detail,
|
|
632
|
+
kind='per_case',
|
|
633
|
+
)
|
|
634
|
+
for metric in self._metric_names()
|
|
635
|
+
]
|