evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
evalkit/pairwise.py
ADDED
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
"""Pairwise (A-vs-B) judging - the headline subjective win-rate.
|
|
2
|
+
|
|
3
|
+
Rubric scoring asks "how good is this output, 1..5"; pairwise asks the
|
|
4
|
+
sharper question "is A better than B for this case?" and reports A's
|
|
5
|
+
win-rate. It is a cross-variant operation the per-variant runner can't
|
|
6
|
+
express as a grader (it needs both variants' output for the same case at
|
|
7
|
+
once), so it lives here: align two runs by (case_id, sample_idx), ask a
|
|
8
|
+
judge to pick a winner per case, aggregate.
|
|
9
|
+
|
|
10
|
+
Position bias (LLMs favour whichever option they see first) is handled by
|
|
11
|
+
**counterbalancing**: each pair is judged in both orders and a pick that
|
|
12
|
+
flips with order collapses to a tie. The judge client is pluggable and
|
|
13
|
+
mode-selected (Anthropic / OpenAI / replay) exactly like the rubric judge.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import typing
|
|
18
|
+
|
|
19
|
+
from evalkit import models, refs
|
|
20
|
+
|
|
21
|
+
_SYSTEM = (
|
|
22
|
+
'You are a careful, unbiased evaluator comparing two candidate '
|
|
23
|
+
'outputs for the same request. Judge only on quality against the '
|
|
24
|
+
'rubric; ignore which option is shown first. Pick the better option, '
|
|
25
|
+
'or "tie" if they are genuinely equal.'
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
_Pick = typing.Literal['first', 'second', 'tie']
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@typing.runtime_checkable
|
|
32
|
+
class PairwiseClient(typing.Protocol):
|
|
33
|
+
"""Pick the better of two outputs; return 'first' | 'second' | 'tie'."""
|
|
34
|
+
|
|
35
|
+
async def compare(
|
|
36
|
+
self, *, system: str, user: str, first: str, second: str
|
|
37
|
+
) -> str: ...
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _tool() -> dict:
|
|
41
|
+
return {
|
|
42
|
+
'name': 'pick_winner',
|
|
43
|
+
'description': 'Pick the better output for this request.',
|
|
44
|
+
'input_schema': {
|
|
45
|
+
'type': 'object',
|
|
46
|
+
'properties': {
|
|
47
|
+
'winner': {
|
|
48
|
+
'type': 'string',
|
|
49
|
+
'enum': ['first', 'second', 'tie'],
|
|
50
|
+
},
|
|
51
|
+
'rationale': {'type': 'string'},
|
|
52
|
+
},
|
|
53
|
+
'required': ['winner'],
|
|
54
|
+
},
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class AnthropicPairwiseClient:
|
|
59
|
+
"""Live pairwise judge via the Anthropic SDK (forced tool call)."""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
model: str,
|
|
64
|
+
api_key_env: str = 'ANTHROPIC_API_KEY',
|
|
65
|
+
max_tokens: int = 512,
|
|
66
|
+
timeout: float = 30.0,
|
|
67
|
+
):
|
|
68
|
+
self.model = model
|
|
69
|
+
self.api_key_env = api_key_env
|
|
70
|
+
self.max_tokens = max_tokens
|
|
71
|
+
self.timeout = timeout
|
|
72
|
+
|
|
73
|
+
async def compare(self, *, system, user, first, second) -> str:
|
|
74
|
+
import anthropic
|
|
75
|
+
|
|
76
|
+
tool = _tool()
|
|
77
|
+
async with anthropic.AsyncAnthropic(
|
|
78
|
+
api_key=os.environ[self.api_key_env]
|
|
79
|
+
) as client:
|
|
80
|
+
response = await client.messages.create(
|
|
81
|
+
model=self.model,
|
|
82
|
+
max_tokens=self.max_tokens,
|
|
83
|
+
temperature=0,
|
|
84
|
+
timeout=self.timeout,
|
|
85
|
+
system=system,
|
|
86
|
+
tools=[tool],
|
|
87
|
+
tool_choice={'type': 'tool', 'name': tool['name']},
|
|
88
|
+
messages=[{'role': 'user', 'content': user}],
|
|
89
|
+
)
|
|
90
|
+
for block in response.content:
|
|
91
|
+
if (
|
|
92
|
+
getattr(block, 'type', None) == 'tool_use'
|
|
93
|
+
and block.name == tool['name']
|
|
94
|
+
):
|
|
95
|
+
return dict(block.input).get('winner', 'tie')
|
|
96
|
+
return 'tie'
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class OpenAIPairwiseClient:
|
|
100
|
+
"""Live pairwise judge via the OpenAI SDK (``json_schema``)."""
|
|
101
|
+
|
|
102
|
+
def __init__(
|
|
103
|
+
self,
|
|
104
|
+
model: str,
|
|
105
|
+
api_key_env: str = 'OPENAI_API_KEY',
|
|
106
|
+
max_tokens: int = 512,
|
|
107
|
+
timeout: float = 30.0,
|
|
108
|
+
):
|
|
109
|
+
self.model = model.split(':', 1)[1] if ':' in model else model
|
|
110
|
+
self.api_key_env = api_key_env
|
|
111
|
+
self.max_tokens = max_tokens
|
|
112
|
+
self.timeout = timeout
|
|
113
|
+
|
|
114
|
+
async def compare(self, *, system, user, first, second) -> str:
|
|
115
|
+
import json
|
|
116
|
+
|
|
117
|
+
import openai
|
|
118
|
+
|
|
119
|
+
response_format = {
|
|
120
|
+
'type': 'json_schema',
|
|
121
|
+
'json_schema': {
|
|
122
|
+
'name': 'pick_winner',
|
|
123
|
+
'strict': True,
|
|
124
|
+
'schema': {
|
|
125
|
+
'type': 'object',
|
|
126
|
+
'properties': {
|
|
127
|
+
'winner': {
|
|
128
|
+
'type': 'string',
|
|
129
|
+
'enum': ['first', 'second', 'tie'],
|
|
130
|
+
},
|
|
131
|
+
'rationale': {'type': 'string'},
|
|
132
|
+
},
|
|
133
|
+
'required': ['winner', 'rationale'],
|
|
134
|
+
'additionalProperties': False,
|
|
135
|
+
},
|
|
136
|
+
},
|
|
137
|
+
}
|
|
138
|
+
async with openai.AsyncOpenAI(
|
|
139
|
+
api_key=os.environ[self.api_key_env]
|
|
140
|
+
) as client:
|
|
141
|
+
response = await client.chat.completions.create(
|
|
142
|
+
model=self.model,
|
|
143
|
+
max_tokens=self.max_tokens,
|
|
144
|
+
temperature=0,
|
|
145
|
+
timeout=self.timeout,
|
|
146
|
+
response_format=response_format,
|
|
147
|
+
messages=[
|
|
148
|
+
{'role': 'system', 'content': system},
|
|
149
|
+
{'role': 'user', 'content': user},
|
|
150
|
+
],
|
|
151
|
+
)
|
|
152
|
+
body = response.choices[0].message.content
|
|
153
|
+
try:
|
|
154
|
+
return (json.loads(body) if body else {}).get('winner', 'tie')
|
|
155
|
+
except ValueError:
|
|
156
|
+
return 'tie'
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class ReplayPairwiseClient:
|
|
160
|
+
"""Offline pairwise judge keyed by the (unordered) pair of contents.
|
|
161
|
+
|
|
162
|
+
Fixtures are a list of ``{a, b, winner, rationale?}`` where ``winner``
|
|
163
|
+
is the winning *content* string (or ``'tie'``). Lookups are
|
|
164
|
+
order-independent, so counterbalancing returns a stable result offline.
|
|
165
|
+
"""
|
|
166
|
+
|
|
167
|
+
_SEP = '\n<VS>\n'
|
|
168
|
+
|
|
169
|
+
def __init__(self, fixtures: str | list):
|
|
170
|
+
if isinstance(fixtures, str):
|
|
171
|
+
from evalkit import loader
|
|
172
|
+
|
|
173
|
+
fixtures = loader.load_data_file(fixtures)
|
|
174
|
+
# load_data_file returns {} for a bare list file under JSON fallback;
|
|
175
|
+
# accept either a list or a {'pairs': [...]} mapping.
|
|
176
|
+
rows = (
|
|
177
|
+
fixtures
|
|
178
|
+
if isinstance(fixtures, list)
|
|
179
|
+
else (fixtures.get('pairs', []))
|
|
180
|
+
)
|
|
181
|
+
self._winners: dict[str, str] = {}
|
|
182
|
+
for row in rows:
|
|
183
|
+
self._winners[self._key(row['a'], row['b'])] = row['winner']
|
|
184
|
+
|
|
185
|
+
def _key(self, x: str, y: str) -> str:
|
|
186
|
+
return self._SEP.join(sorted([x, y]))
|
|
187
|
+
|
|
188
|
+
async def compare(self, *, system, user, first, second) -> str:
|
|
189
|
+
winner = self._winners.get(self._key(first, second))
|
|
190
|
+
if winner is None or winner == 'tie':
|
|
191
|
+
return 'tie'
|
|
192
|
+
if winner == first:
|
|
193
|
+
return 'first'
|
|
194
|
+
if winner == second:
|
|
195
|
+
return 'second'
|
|
196
|
+
return 'tie'
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def build_pairwise_client(mode: str, config: dict) -> PairwiseClient:
|
|
200
|
+
"""Pick a pairwise client for the run mode from a config mapping."""
|
|
201
|
+
if mode == 'replay':
|
|
202
|
+
if not config.get('replay_path'):
|
|
203
|
+
raise ValueError('pairwise replay mode needs replay_path')
|
|
204
|
+
return ReplayPairwiseClient(config['replay_path'])
|
|
205
|
+
model = config.get('model')
|
|
206
|
+
if not model:
|
|
207
|
+
raise ValueError('pairwise live mode needs a model')
|
|
208
|
+
if config.get('provider') == 'openai':
|
|
209
|
+
return OpenAIPairwiseClient(model)
|
|
210
|
+
return AnthropicPairwiseClient(model)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _content_map(
|
|
214
|
+
run: models.RunResult, content_ref: str
|
|
215
|
+
) -> dict[tuple[str, int], tuple[str, models.CaseResult]]:
|
|
216
|
+
out: dict[tuple[str, int], tuple[str, models.CaseResult]] = {}
|
|
217
|
+
for result in run.results:
|
|
218
|
+
ctx = {
|
|
219
|
+
'input': result.case.input,
|
|
220
|
+
'expected': result.case.expected or {},
|
|
221
|
+
'output': result.output.fields,
|
|
222
|
+
'case': result.case.model_dump(),
|
|
223
|
+
'artifacts': result.output.artifacts,
|
|
224
|
+
}
|
|
225
|
+
content = refs.resolve_ref(ctx, content_ref)
|
|
226
|
+
if isinstance(content, str) and content:
|
|
227
|
+
out[result.case.id, result.sample_idx] = (content, result)
|
|
228
|
+
return out
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _prompt(first: str, second: str, context: dict, rubric: str | None) -> str:
|
|
232
|
+
parts = []
|
|
233
|
+
if rubric:
|
|
234
|
+
parts.append(f'<rubric>\n{rubric}\n</rubric>')
|
|
235
|
+
for label, value in context.items():
|
|
236
|
+
parts.append(f'<{label}>\n{value}\n</{label}>')
|
|
237
|
+
parts.append(f'<option_1>\n{first}\n</option_1>')
|
|
238
|
+
parts.append(f'<option_2>\n{second}\n</option_2>')
|
|
239
|
+
return '\n\n'.join(parts)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
async def judge_pairwise(
|
|
243
|
+
run_a: models.RunResult,
|
|
244
|
+
run_b: models.RunResult,
|
|
245
|
+
*,
|
|
246
|
+
content_ref: str,
|
|
247
|
+
client: PairwiseClient,
|
|
248
|
+
context_refs: dict[str, str] | None = None,
|
|
249
|
+
rubric: str | None = None,
|
|
250
|
+
counterbalance: bool = True,
|
|
251
|
+
judge_name: str = 'pairwise',
|
|
252
|
+
judge_version: str = 'v1',
|
|
253
|
+
) -> models.PairwiseResult:
|
|
254
|
+
"""Compare two runs case-by-case and report A's win-rate."""
|
|
255
|
+
a_map = _content_map(run_a, content_ref)
|
|
256
|
+
b_map = _content_map(run_b, content_ref)
|
|
257
|
+
context_refs = context_refs or {}
|
|
258
|
+
|
|
259
|
+
outcomes: list[models.PairwiseOutcome] = []
|
|
260
|
+
a_wins = b_wins = ties = 0
|
|
261
|
+
for key in sorted(a_map.keys() & b_map.keys()):
|
|
262
|
+
content_a, result_a = a_map[key]
|
|
263
|
+
content_b, _ = b_map[key]
|
|
264
|
+
context = {
|
|
265
|
+
label: refs.resolve_ref(
|
|
266
|
+
{
|
|
267
|
+
'input': result_a.case.input,
|
|
268
|
+
'expected': result_a.case.expected or {},
|
|
269
|
+
'case': result_a.case.model_dump(),
|
|
270
|
+
},
|
|
271
|
+
ref,
|
|
272
|
+
)
|
|
273
|
+
for label, ref in context_refs.items()
|
|
274
|
+
}
|
|
275
|
+
# Order 1: A shown first.
|
|
276
|
+
pick1 = await client.compare(
|
|
277
|
+
system=_SYSTEM,
|
|
278
|
+
user=_prompt(content_a, content_b, context, rubric),
|
|
279
|
+
first=content_a,
|
|
280
|
+
second=content_b,
|
|
281
|
+
)
|
|
282
|
+
winner = {'first': 'a', 'second': 'b', 'tie': 'tie'}[pick1]
|
|
283
|
+
detail = f'order1={pick1}'
|
|
284
|
+
if counterbalance:
|
|
285
|
+
# Order 2: B shown first; a flip vs order 1 means position bias.
|
|
286
|
+
pick2 = await client.compare(
|
|
287
|
+
system=_SYSTEM,
|
|
288
|
+
user=_prompt(content_b, content_a, context, rubric),
|
|
289
|
+
first=content_b,
|
|
290
|
+
second=content_a,
|
|
291
|
+
)
|
|
292
|
+
winner2 = {'first': 'b', 'second': 'a', 'tie': 'tie'}[pick2]
|
|
293
|
+
detail += f' order2={pick2}'
|
|
294
|
+
if winner != winner2:
|
|
295
|
+
winner = 'tie' # inconsistent under swap -> not a real win
|
|
296
|
+
|
|
297
|
+
if winner == 'a':
|
|
298
|
+
a_wins += 1
|
|
299
|
+
elif winner == 'b':
|
|
300
|
+
b_wins += 1
|
|
301
|
+
else:
|
|
302
|
+
ties += 1
|
|
303
|
+
outcomes.append(
|
|
304
|
+
models.PairwiseOutcome(
|
|
305
|
+
case_id=key[0], sample_idx=key[1], winner=winner, detail=detail
|
|
306
|
+
)
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
n = len(outcomes)
|
|
310
|
+
win_rate_a = (a_wins + 0.5 * ties) / n if n else None
|
|
311
|
+
return models.PairwiseResult(
|
|
312
|
+
project=run_a.scorecard.project,
|
|
313
|
+
suite=run_a.scorecard.suite,
|
|
314
|
+
variant_a=run_a.scorecard.variant.name,
|
|
315
|
+
variant_b=run_b.scorecard.variant.name,
|
|
316
|
+
judge_name=judge_name,
|
|
317
|
+
judge_version=judge_version,
|
|
318
|
+
n=n,
|
|
319
|
+
a_wins=a_wins,
|
|
320
|
+
b_wins=b_wins,
|
|
321
|
+
ties=ties,
|
|
322
|
+
win_rate_a=win_rate_a,
|
|
323
|
+
outcomes=outcomes,
|
|
324
|
+
)
|
evalkit/py.typed
ADDED
|
File without changes
|