evalcore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evalkit/pairwise.py ADDED
@@ -0,0 +1,324 @@
1
+ """Pairwise (A-vs-B) judging - the headline subjective win-rate.
2
+
3
+ Rubric scoring asks "how good is this output, 1..5"; pairwise asks the
4
+ sharper question "is A better than B for this case?" and reports A's
5
+ win-rate. It is a cross-variant operation the per-variant runner can't
6
+ express as a grader (it needs both variants' output for the same case at
7
+ once), so it lives here: align two runs by (case_id, sample_idx), ask a
8
+ judge to pick a winner per case, aggregate.
9
+
10
+ Position bias (LLMs favour whichever option they see first) is handled by
11
+ **counterbalancing**: each pair is judged in both orders and a pick that
12
+ flips with order collapses to a tie. The judge client is pluggable and
13
+ mode-selected (Anthropic / OpenAI / replay) exactly like the rubric judge.
14
+ """
15
+
16
+ import os
17
+ import typing
18
+
19
+ from evalkit import models, refs
20
+
21
+ _SYSTEM = (
22
+ 'You are a careful, unbiased evaluator comparing two candidate '
23
+ 'outputs for the same request. Judge only on quality against the '
24
+ 'rubric; ignore which option is shown first. Pick the better option, '
25
+ 'or "tie" if they are genuinely equal.'
26
+ )
27
+
28
+ _Pick = typing.Literal['first', 'second', 'tie']
29
+
30
+
31
+ @typing.runtime_checkable
32
+ class PairwiseClient(typing.Protocol):
33
+ """Pick the better of two outputs; return 'first' | 'second' | 'tie'."""
34
+
35
+ async def compare(
36
+ self, *, system: str, user: str, first: str, second: str
37
+ ) -> str: ...
38
+
39
+
40
+ def _tool() -> dict:
41
+ return {
42
+ 'name': 'pick_winner',
43
+ 'description': 'Pick the better output for this request.',
44
+ 'input_schema': {
45
+ 'type': 'object',
46
+ 'properties': {
47
+ 'winner': {
48
+ 'type': 'string',
49
+ 'enum': ['first', 'second', 'tie'],
50
+ },
51
+ 'rationale': {'type': 'string'},
52
+ },
53
+ 'required': ['winner'],
54
+ },
55
+ }
56
+
57
+
58
+ class AnthropicPairwiseClient:
59
+ """Live pairwise judge via the Anthropic SDK (forced tool call)."""
60
+
61
+ def __init__(
62
+ self,
63
+ model: str,
64
+ api_key_env: str = 'ANTHROPIC_API_KEY',
65
+ max_tokens: int = 512,
66
+ timeout: float = 30.0,
67
+ ):
68
+ self.model = model
69
+ self.api_key_env = api_key_env
70
+ self.max_tokens = max_tokens
71
+ self.timeout = timeout
72
+
73
+ async def compare(self, *, system, user, first, second) -> str:
74
+ import anthropic
75
+
76
+ tool = _tool()
77
+ async with anthropic.AsyncAnthropic(
78
+ api_key=os.environ[self.api_key_env]
79
+ ) as client:
80
+ response = await client.messages.create(
81
+ model=self.model,
82
+ max_tokens=self.max_tokens,
83
+ temperature=0,
84
+ timeout=self.timeout,
85
+ system=system,
86
+ tools=[tool],
87
+ tool_choice={'type': 'tool', 'name': tool['name']},
88
+ messages=[{'role': 'user', 'content': user}],
89
+ )
90
+ for block in response.content:
91
+ if (
92
+ getattr(block, 'type', None) == 'tool_use'
93
+ and block.name == tool['name']
94
+ ):
95
+ return dict(block.input).get('winner', 'tie')
96
+ return 'tie'
97
+
98
+
99
+ class OpenAIPairwiseClient:
100
+ """Live pairwise judge via the OpenAI SDK (``json_schema``)."""
101
+
102
+ def __init__(
103
+ self,
104
+ model: str,
105
+ api_key_env: str = 'OPENAI_API_KEY',
106
+ max_tokens: int = 512,
107
+ timeout: float = 30.0,
108
+ ):
109
+ self.model = model.split(':', 1)[1] if ':' in model else model
110
+ self.api_key_env = api_key_env
111
+ self.max_tokens = max_tokens
112
+ self.timeout = timeout
113
+
114
+ async def compare(self, *, system, user, first, second) -> str:
115
+ import json
116
+
117
+ import openai
118
+
119
+ response_format = {
120
+ 'type': 'json_schema',
121
+ 'json_schema': {
122
+ 'name': 'pick_winner',
123
+ 'strict': True,
124
+ 'schema': {
125
+ 'type': 'object',
126
+ 'properties': {
127
+ 'winner': {
128
+ 'type': 'string',
129
+ 'enum': ['first', 'second', 'tie'],
130
+ },
131
+ 'rationale': {'type': 'string'},
132
+ },
133
+ 'required': ['winner', 'rationale'],
134
+ 'additionalProperties': False,
135
+ },
136
+ },
137
+ }
138
+ async with openai.AsyncOpenAI(
139
+ api_key=os.environ[self.api_key_env]
140
+ ) as client:
141
+ response = await client.chat.completions.create(
142
+ model=self.model,
143
+ max_tokens=self.max_tokens,
144
+ temperature=0,
145
+ timeout=self.timeout,
146
+ response_format=response_format,
147
+ messages=[
148
+ {'role': 'system', 'content': system},
149
+ {'role': 'user', 'content': user},
150
+ ],
151
+ )
152
+ body = response.choices[0].message.content
153
+ try:
154
+ return (json.loads(body) if body else {}).get('winner', 'tie')
155
+ except ValueError:
156
+ return 'tie'
157
+
158
+
159
+ class ReplayPairwiseClient:
160
+ """Offline pairwise judge keyed by the (unordered) pair of contents.
161
+
162
+ Fixtures are a list of ``{a, b, winner, rationale?}`` where ``winner``
163
+ is the winning *content* string (or ``'tie'``). Lookups are
164
+ order-independent, so counterbalancing returns a stable result offline.
165
+ """
166
+
167
+ _SEP = '\n<VS>\n'
168
+
169
+ def __init__(self, fixtures: str | list):
170
+ if isinstance(fixtures, str):
171
+ from evalkit import loader
172
+
173
+ fixtures = loader.load_data_file(fixtures)
174
+ # load_data_file returns {} for a bare list file under JSON fallback;
175
+ # accept either a list or a {'pairs': [...]} mapping.
176
+ rows = (
177
+ fixtures
178
+ if isinstance(fixtures, list)
179
+ else (fixtures.get('pairs', []))
180
+ )
181
+ self._winners: dict[str, str] = {}
182
+ for row in rows:
183
+ self._winners[self._key(row['a'], row['b'])] = row['winner']
184
+
185
+ def _key(self, x: str, y: str) -> str:
186
+ return self._SEP.join(sorted([x, y]))
187
+
188
+ async def compare(self, *, system, user, first, second) -> str:
189
+ winner = self._winners.get(self._key(first, second))
190
+ if winner is None or winner == 'tie':
191
+ return 'tie'
192
+ if winner == first:
193
+ return 'first'
194
+ if winner == second:
195
+ return 'second'
196
+ return 'tie'
197
+
198
+
199
+ def build_pairwise_client(mode: str, config: dict) -> PairwiseClient:
200
+ """Pick a pairwise client for the run mode from a config mapping."""
201
+ if mode == 'replay':
202
+ if not config.get('replay_path'):
203
+ raise ValueError('pairwise replay mode needs replay_path')
204
+ return ReplayPairwiseClient(config['replay_path'])
205
+ model = config.get('model')
206
+ if not model:
207
+ raise ValueError('pairwise live mode needs a model')
208
+ if config.get('provider') == 'openai':
209
+ return OpenAIPairwiseClient(model)
210
+ return AnthropicPairwiseClient(model)
211
+
212
+
213
+ def _content_map(
214
+ run: models.RunResult, content_ref: str
215
+ ) -> dict[tuple[str, int], tuple[str, models.CaseResult]]:
216
+ out: dict[tuple[str, int], tuple[str, models.CaseResult]] = {}
217
+ for result in run.results:
218
+ ctx = {
219
+ 'input': result.case.input,
220
+ 'expected': result.case.expected or {},
221
+ 'output': result.output.fields,
222
+ 'case': result.case.model_dump(),
223
+ 'artifacts': result.output.artifacts,
224
+ }
225
+ content = refs.resolve_ref(ctx, content_ref)
226
+ if isinstance(content, str) and content:
227
+ out[result.case.id, result.sample_idx] = (content, result)
228
+ return out
229
+
230
+
231
+ def _prompt(first: str, second: str, context: dict, rubric: str | None) -> str:
232
+ parts = []
233
+ if rubric:
234
+ parts.append(f'<rubric>\n{rubric}\n</rubric>')
235
+ for label, value in context.items():
236
+ parts.append(f'<{label}>\n{value}\n</{label}>')
237
+ parts.append(f'<option_1>\n{first}\n</option_1>')
238
+ parts.append(f'<option_2>\n{second}\n</option_2>')
239
+ return '\n\n'.join(parts)
240
+
241
+
242
+ async def judge_pairwise(
243
+ run_a: models.RunResult,
244
+ run_b: models.RunResult,
245
+ *,
246
+ content_ref: str,
247
+ client: PairwiseClient,
248
+ context_refs: dict[str, str] | None = None,
249
+ rubric: str | None = None,
250
+ counterbalance: bool = True,
251
+ judge_name: str = 'pairwise',
252
+ judge_version: str = 'v1',
253
+ ) -> models.PairwiseResult:
254
+ """Compare two runs case-by-case and report A's win-rate."""
255
+ a_map = _content_map(run_a, content_ref)
256
+ b_map = _content_map(run_b, content_ref)
257
+ context_refs = context_refs or {}
258
+
259
+ outcomes: list[models.PairwiseOutcome] = []
260
+ a_wins = b_wins = ties = 0
261
+ for key in sorted(a_map.keys() & b_map.keys()):
262
+ content_a, result_a = a_map[key]
263
+ content_b, _ = b_map[key]
264
+ context = {
265
+ label: refs.resolve_ref(
266
+ {
267
+ 'input': result_a.case.input,
268
+ 'expected': result_a.case.expected or {},
269
+ 'case': result_a.case.model_dump(),
270
+ },
271
+ ref,
272
+ )
273
+ for label, ref in context_refs.items()
274
+ }
275
+ # Order 1: A shown first.
276
+ pick1 = await client.compare(
277
+ system=_SYSTEM,
278
+ user=_prompt(content_a, content_b, context, rubric),
279
+ first=content_a,
280
+ second=content_b,
281
+ )
282
+ winner = {'first': 'a', 'second': 'b', 'tie': 'tie'}[pick1]
283
+ detail = f'order1={pick1}'
284
+ if counterbalance:
285
+ # Order 2: B shown first; a flip vs order 1 means position bias.
286
+ pick2 = await client.compare(
287
+ system=_SYSTEM,
288
+ user=_prompt(content_b, content_a, context, rubric),
289
+ first=content_b,
290
+ second=content_a,
291
+ )
292
+ winner2 = {'first': 'b', 'second': 'a', 'tie': 'tie'}[pick2]
293
+ detail += f' order2={pick2}'
294
+ if winner != winner2:
295
+ winner = 'tie' # inconsistent under swap -> not a real win
296
+
297
+ if winner == 'a':
298
+ a_wins += 1
299
+ elif winner == 'b':
300
+ b_wins += 1
301
+ else:
302
+ ties += 1
303
+ outcomes.append(
304
+ models.PairwiseOutcome(
305
+ case_id=key[0], sample_idx=key[1], winner=winner, detail=detail
306
+ )
307
+ )
308
+
309
+ n = len(outcomes)
310
+ win_rate_a = (a_wins + 0.5 * ties) / n if n else None
311
+ return models.PairwiseResult(
312
+ project=run_a.scorecard.project,
313
+ suite=run_a.scorecard.suite,
314
+ variant_a=run_a.scorecard.variant.name,
315
+ variant_b=run_b.scorecard.variant.name,
316
+ judge_name=judge_name,
317
+ judge_version=judge_version,
318
+ n=n,
319
+ a_wins=a_wins,
320
+ b_wins=b_wins,
321
+ ties=ties,
322
+ win_rate_a=win_rate_a,
323
+ outcomes=outcomes,
324
+ )
evalkit/py.typed ADDED
File without changes