evalcore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,635 @@
1
+ """LLM-as-judge (rubric scoring) grader, single judge or a panel.
2
+
3
+ The Tier-2 grader for subjective quality: one or more pinned, strong judge
4
+ models score each output 1..scale on a set of rubric dimensions via
5
+ structured output (Anthropic forced tool call, OpenAI ``json_schema``).
6
+ Scores are normalized to 0..1 and averaged by the runner, so a rubric
7
+ dimension - or the ``overall`` mean - can serve as a suite's win metric.
8
+
9
+ Configure ``judges`` for a **panel**: each judge scores independently, the
10
+ grader emits the panel mean per dimension plus each judge's ``overall``, an
11
+ inter-judge ``disagreement`` magnitude, and a ``flagged`` rate (cases where
12
+ judges disagree by >= ``disagreement_threshold`` raw points - the queue a
13
+ human reviewer should look at). A single judge (the default, back-compatible
14
+ config) emits just the per-dimension scores and ``overall``.
15
+
16
+ Judges take optional **image inputs** (``image_refs``) - screenshots or other
17
+ rendered artifacts the judge should see alongside the text (e.g. the builder's
18
+ rendered page). Images are sent only in live mode; replay is keyed by content.
19
+
20
+ Each judge call goes through a pluggable ``JudgeClient`` so the grader runs
21
+ live (``AnthropicJudgeClient`` / ``OpenAIJudgeClient``) or fully offline
22
+ against recorded judgments (``ReplayJudgeClient``), chosen by the run ``mode``
23
+ like the target adapter. Judge model + version are recorded on every score;
24
+ pin them and treat a judge change as a re-baseline event.
25
+
26
+ Pairwise (A-vs-B win-rate) judging is a cross-variant operation and lives in
27
+ :mod:`evalkit.pairwise`, not here - it needs both variants' per-case outputs
28
+ at once, which a per-case grader never sees.
29
+ """
30
+
31
+ import base64
32
+ import functools
33
+ import os
34
+ import pathlib
35
+ import re
36
+ import typing
37
+
38
+ from evalkit import models, refs, retry
39
+ from evalkit.graders import base
40
+
41
+ _ENV_RE = re.compile(r'\$\{([A-Z0-9_]+)\}')
42
+
43
+
44
+ def _env_expand(value):
45
+ """Expand ``${VAR}`` in a judge model string; passthrough otherwise.
46
+
47
+ Lets a panel judge's model be env-gated - an unset var expands to ''
48
+ and (in live mode) drops that judge, so 1-vs-2 judges is a config knob.
49
+ """
50
+ if not isinstance(value, str):
51
+ return value
52
+ return _ENV_RE.sub(lambda m: os.environ.get(m.group(1), ''), value)
53
+
54
+
55
+ _SYSTEM = (
56
+ 'You are a strict, consistent evaluator. Score the content on each '
57
+ 'dimension using the full 1..{scale} range, judging only against the '
58
+ 'rubric and dimension descriptions - never reward length or verbosity. '
59
+ 'Return only the structured per-dimension scores.'
60
+ )
61
+
62
+ _MEDIA_TYPES = {
63
+ '.png': 'image/png',
64
+ '.jpg': 'image/jpeg',
65
+ '.jpeg': 'image/jpeg',
66
+ '.webp': 'image/webp',
67
+ '.gif': 'image/gif',
68
+ }
69
+
70
+
71
+ @typing.runtime_checkable
72
+ class JudgeClient(typing.Protocol):
73
+ """Score one item; return ``{scores: {dim: int}, rationale: str}``."""
74
+
75
+ async def judge(
76
+ self,
77
+ *,
78
+ key: str,
79
+ system: str,
80
+ user: str,
81
+ dimensions: list[tuple[str, str]],
82
+ scale: int,
83
+ images: list[dict] | None = None,
84
+ ) -> dict: ...
85
+
86
+
87
+ def _score_properties(dimensions: list[tuple[str, str]], scale: int) -> dict:
88
+ return {
89
+ key: {
90
+ 'type': 'integer',
91
+ 'minimum': 1,
92
+ 'maximum': scale,
93
+ 'description': desc,
94
+ }
95
+ for key, desc in dimensions
96
+ }
97
+
98
+
99
+ class AnthropicJudgeClient:
100
+ """Live judge backed by the Anthropic SDK (forced single tool call)."""
101
+
102
+ def __init__(
103
+ self,
104
+ model: str,
105
+ api_key_env: str = 'ANTHROPIC_API_KEY',
106
+ max_tokens: int = 1024,
107
+ timeout: float = 30.0,
108
+ ):
109
+ self.model = model
110
+ self.api_key_env = api_key_env
111
+ self.max_tokens = max_tokens
112
+ self.timeout = timeout
113
+
114
+ def _tool(self, dimensions: list[tuple[str, str]], scale: int) -> dict:
115
+ return {
116
+ 'name': 'score_rubric',
117
+ 'description': 'Report per-dimension rubric scores.',
118
+ 'input_schema': {
119
+ 'type': 'object',
120
+ 'properties': {
121
+ 'scores': {
122
+ 'type': 'object',
123
+ 'properties': _score_properties(dimensions, scale),
124
+ 'required': [k for k, _ in dimensions],
125
+ },
126
+ 'rationale': {'type': 'string'},
127
+ },
128
+ 'required': ['scores'],
129
+ },
130
+ }
131
+
132
+ async def judge(
133
+ self, *, key, system, user, dimensions, scale, images=None
134
+ ) -> dict:
135
+ import anthropic
136
+
137
+ tool = self._tool(dimensions, scale)
138
+ content: list[dict] = [{'type': 'text', 'text': user}]
139
+ for image in images or []:
140
+ content.append(
141
+ {
142
+ 'type': 'image',
143
+ 'source': {
144
+ 'type': 'base64',
145
+ 'media_type': image['media_type'],
146
+ 'data': image['data'],
147
+ },
148
+ }
149
+ )
150
+ # Context-managed so the connection pool closes inside the running
151
+ # event loop instead of at GC time after asyncio.run() tore it down.
152
+ async with anthropic.AsyncAnthropic(
153
+ api_key=os.environ[self.api_key_env]
154
+ ) as client:
155
+ response = await client.messages.create(
156
+ model=self.model,
157
+ max_tokens=self.max_tokens,
158
+ temperature=0,
159
+ timeout=self.timeout,
160
+ system=system,
161
+ tools=[tool],
162
+ tool_choice={'type': 'tool', 'name': tool['name']},
163
+ messages=[{'role': 'user', 'content': content}],
164
+ )
165
+ for block in response.content:
166
+ if (
167
+ getattr(block, 'type', None) == 'tool_use'
168
+ and block.name == tool['name']
169
+ ):
170
+ return dict(block.input)
171
+ return {}
172
+
173
+
174
+ class OpenAIJudgeClient:
175
+ """Live judge via the OpenAI SDK (``response_format`` json_schema)."""
176
+
177
+ def __init__(
178
+ self,
179
+ model: str,
180
+ api_key_env: str = 'OPENAI_API_KEY',
181
+ max_tokens: int = 1024,
182
+ timeout: float = 30.0,
183
+ ):
184
+ # Accept a 'provider:model' id (e.g. 'openai:gpt-4o'); SDK wants bare.
185
+ self.model = model.split(':', 1)[1] if ':' in model else model
186
+ self.api_key_env = api_key_env
187
+ self.max_tokens = max_tokens
188
+ self.timeout = timeout
189
+
190
+ def _response_format(
191
+ self, dimensions: list[tuple[str, str]], scale: int
192
+ ) -> dict:
193
+ return {
194
+ 'type': 'json_schema',
195
+ 'json_schema': {
196
+ 'name': 'score_rubric',
197
+ 'strict': True,
198
+ 'schema': {
199
+ 'type': 'object',
200
+ 'properties': {
201
+ 'scores': {
202
+ 'type': 'object',
203
+ 'properties': _score_properties(dimensions, scale),
204
+ 'required': [k for k, _ in dimensions],
205
+ 'additionalProperties': False,
206
+ },
207
+ 'rationale': {'type': 'string'},
208
+ },
209
+ 'required': ['scores', 'rationale'],
210
+ 'additionalProperties': False,
211
+ },
212
+ },
213
+ }
214
+
215
+ async def judge(
216
+ self, *, key, system, user, dimensions, scale, images=None
217
+ ) -> dict:
218
+ import json
219
+
220
+ import openai
221
+
222
+ content: list[dict] = [{'type': 'text', 'text': user}]
223
+ for image in images or []:
224
+ data_url = f'data:{image["media_type"]};base64,{image["data"]}'
225
+ content.append(
226
+ {'type': 'image_url', 'image_url': {'url': data_url}}
227
+ )
228
+ async with openai.AsyncOpenAI(
229
+ api_key=os.environ[self.api_key_env]
230
+ ) as client:
231
+ response = await client.chat.completions.create(
232
+ model=self.model,
233
+ max_tokens=self.max_tokens,
234
+ temperature=0,
235
+ timeout=self.timeout,
236
+ response_format=self._response_format(dimensions, scale),
237
+ messages=[
238
+ {'role': 'system', 'content': system},
239
+ {'role': 'user', 'content': content},
240
+ ],
241
+ )
242
+ body = response.choices[0].message.content
243
+ try:
244
+ return json.loads(body) if body else {}
245
+ except ValueError as exc:
246
+ raise RuntimeError(f'judge returned non-JSON: {exc}') from exc
247
+
248
+
249
+ class ReplayJudgeClient:
250
+ """Offline judge: return recorded judgments keyed by the judged content.
251
+
252
+ Fixtures map the exact content string under judgement to a
253
+ ``{scores: {dim: int}, rationale: str}`` dict, so different variants
254
+ (which produce different content) get different scores deterministically.
255
+ """
256
+
257
+ def __init__(self, fixtures: str | dict):
258
+ if isinstance(fixtures, str):
259
+ from evalkit import loader
260
+
261
+ self._fixtures = loader.load_data_file(fixtures)
262
+ else:
263
+ self._fixtures = fixtures
264
+
265
+ async def judge(
266
+ self, *, key, system, user, dimensions, scale, images=None
267
+ ) -> dict:
268
+ return dict(self._fixtures.get(key) or {})
269
+
270
+
271
+ def _load_images(refs_values: list) -> list[dict]:
272
+ """Turn resolved image refs (file paths or dicts) into base64 blocks."""
273
+ images: list[dict] = []
274
+ for value in refs_values:
275
+ if value is None:
276
+ continue
277
+ if isinstance(value, dict) and 'data' in value:
278
+ images.append(
279
+ {
280
+ 'media_type': value.get('media_type', 'image/png'),
281
+ 'data': value['data'],
282
+ }
283
+ )
284
+ continue
285
+ if isinstance(value, str):
286
+ path = pathlib.Path(value)
287
+ if not path.is_file():
288
+ continue
289
+ media_type = _MEDIA_TYPES.get(path.suffix.lower(), 'image/png')
290
+ data = base64.b64encode(path.read_bytes()).decode('ascii')
291
+ images.append({'media_type': media_type, 'data': data})
292
+ return images
293
+
294
+
295
+ @base.register('llm_judge')
296
+ class RubricJudge:
297
+ """Score an output's text on rubric dimensions with an LLM judge/panel.
298
+
299
+ Config (suite ``graders`` entry):
300
+ content_ref: $ref to the text to judge (e.g. ``output.text``)
301
+ dimensions: list of {key, description}
302
+ scale: max score (default 5)
303
+ rubric: optional rubric text embedded in the judge prompt
304
+ context_refs: optional {label: $ref} of extra context for the judge
305
+ image_refs: optional {label: $ref} of images (paths in
306
+ ``output.artifacts`` or inline data) shown to the judge
307
+ judges: optional list of {key, provider(anthropic|openai),
308
+ model, api_key_env?, replay_path?, judge_version?} - a
309
+ PANEL. Omit for a single judge configured by the
310
+ top-level model/replay_path/judge_version.
311
+ disagreement_threshold: raw-point spread at/above which a case is
312
+ flagged for human review (default 2)
313
+ model/replay_path/judge_version: single-judge shorthand
314
+ client: an explicit JudgeClient instance (tests); overrides mode
315
+ """
316
+
317
+ def __init__(
318
+ self,
319
+ content_ref: str,
320
+ dimensions: list[dict],
321
+ name: str = 'llm_judge',
322
+ scale: int = 5,
323
+ rubric: str | None = None,
324
+ context_refs: dict[str, str] | None = None,
325
+ image_refs: dict[str, str] | None = None,
326
+ judges: list[dict] | None = None,
327
+ disagreement_threshold: float = 2.0,
328
+ model: str | None = None,
329
+ judge_version: str = 'v1',
330
+ replay_path: str | None = None,
331
+ client: JudgeClient | None = None,
332
+ max_tokens: int = 1024,
333
+ ):
334
+ self.name = name
335
+ self.content_ref = content_ref
336
+ self.dimensions = [
337
+ (d['key'], d.get('description', '')) for d in dimensions
338
+ ]
339
+ self.scale = scale
340
+ self.rubric = rubric
341
+ self.context_refs = context_refs or {}
342
+ self.image_refs = image_refs or {}
343
+ self.disagreement_threshold = disagreement_threshold
344
+ self.max_tokens = max_tokens
345
+ self._explicit_client = client
346
+ self._mode = 'http'
347
+ # Normalize to a list of judge specs; a single judge is a panel of 1
348
+ # and produces output identical to the pre-panel grader.
349
+ if judges:
350
+ self.judges = [
351
+ {
352
+ 'key': j.get('key') or j.get('provider') or 'judge',
353
+ 'provider': j.get('provider', 'anthropic'),
354
+ 'model': _env_expand(j.get('model')),
355
+ 'api_key_env': j.get('api_key_env'),
356
+ 'replay_path': j.get('replay_path') or replay_path,
357
+ 'judge_version': j.get('judge_version', judge_version),
358
+ }
359
+ for j in judges
360
+ ]
361
+ else:
362
+ self.judges = [
363
+ {
364
+ 'key': 'judge',
365
+ 'provider': 'anthropic',
366
+ 'model': _env_expand(model),
367
+ 'api_key_env': None,
368
+ 'replay_path': replay_path,
369
+ 'judge_version': judge_version,
370
+ }
371
+ ]
372
+ self._clients: dict[str, JudgeClient] = {}
373
+ self._retry = retry.RetryConfig()
374
+
375
+ def set_retry(self, config: retry.RetryConfig) -> None:
376
+ """Runner hook: back off + retry transient judge-client failures."""
377
+ self._retry = config
378
+
379
+ def _active_judges(self) -> list[dict]:
380
+ """Judges usable in the current mode - a panel may drop some.
381
+
382
+ Only filters a PANEL (2+ judges): a judge with no model (live) or
383
+ no replay_path (replay) is dropped, so an env-gated GPT judge that
384
+ is unset simply leaves a Claude-only panel. A single judge is never
385
+ filtered, so a misconfigured lone judge still raises loudly.
386
+ """
387
+ judges = self.judges
388
+ if self._explicit_client is not None or len(judges) <= 1:
389
+ return judges
390
+ if self._mode == 'replay':
391
+ return [j for j in judges if j['replay_path']] or judges
392
+ return [j for j in judges if j['model']] or judges
393
+
394
+ @property
395
+ def is_panel(self) -> bool:
396
+ return len(self._active_judges()) > 1
397
+
398
+ @property
399
+ def judge_version(self) -> str:
400
+ """Provenance pin: ``key@version`` per configured judge (a panel
401
+ joins them, comma-separated).
402
+
403
+ The runner reads this onto ``Scorecard.judge_version`` so a judge
404
+ model / prompt / scale change surfaces as a re-baseline event rather
405
+ than hiding in each score's ``detail``. Uses the configured judges,
406
+ not the mode-filtered active set, so the pin is stable across
407
+ environments.
408
+ """
409
+ return ','.join(
410
+ f'{j["key"]}@{j["judge_version"]}' for j in self.judges
411
+ )
412
+
413
+ def set_mode(self, mode: str) -> None:
414
+ """Runner hook: pick the judge client to match the run mode."""
415
+ self._mode = mode
416
+
417
+ def _client_for(self, judge: dict) -> JudgeClient:
418
+ if self._explicit_client is not None:
419
+ return self._explicit_client
420
+ cached = self._clients.get(judge['key'])
421
+ if cached is not None:
422
+ return cached
423
+ if self._mode == 'replay':
424
+ if not judge['replay_path']:
425
+ raise ValueError(
426
+ f'judge {self.name}/{judge["key"]!r} needs a '
427
+ f'replay_path for replay mode'
428
+ )
429
+ client: JudgeClient = ReplayJudgeClient(judge['replay_path'])
430
+ elif judge['provider'] == 'openai':
431
+ if not judge['model']:
432
+ raise ValueError(f'judge {judge["key"]!r} needs a model')
433
+ client = OpenAIJudgeClient(
434
+ judge['model'],
435
+ api_key_env=judge['api_key_env'] or 'OPENAI_API_KEY',
436
+ max_tokens=self.max_tokens,
437
+ )
438
+ else:
439
+ if not judge['model']:
440
+ raise ValueError(f'judge {judge["key"]!r} needs a model')
441
+ client = AnthropicJudgeClient(
442
+ judge['model'],
443
+ api_key_env=judge['api_key_env'] or 'ANTHROPIC_API_KEY',
444
+ max_tokens=self.max_tokens,
445
+ )
446
+ self._clients[judge['key']] = client
447
+ return client
448
+
449
+ def _prompt(self, content: str, context: dict) -> str:
450
+ parts = []
451
+ if self.rubric:
452
+ parts.append(f'<rubric>\n{self.rubric}\n</rubric>')
453
+ for label, value in context.items():
454
+ parts.append(f'<{label}>\n{value}\n</{label}>')
455
+ parts.append(f'<content>\n{content}\n</content>')
456
+ return '\n\n'.join(parts)
457
+
458
+ def _normalize(self, raw) -> float | None:
459
+ if isinstance(raw, (int, float)) and not isinstance(raw, bool):
460
+ return raw / self.scale
461
+ return None
462
+
463
+ async def grade(
464
+ self, case: models.Case, output: models.Output
465
+ ) -> list[models.Score]:
466
+ ctx = {
467
+ 'input': case.input,
468
+ 'expected': case.expected or {},
469
+ 'output': output.fields,
470
+ 'case': case.model_dump(),
471
+ 'artifacts': output.artifacts,
472
+ }
473
+ content = refs.resolve_ref(ctx, self.content_ref)
474
+ if not isinstance(content, str) or not content:
475
+ return self._error_scores('no content to judge')
476
+
477
+ context = {
478
+ label: refs.resolve_ref(ctx, ref)
479
+ for label, ref in self.context_refs.items()
480
+ }
481
+ images = _load_images(
482
+ [refs.resolve_ref(ctx, ref) for ref in self.image_refs.values()]
483
+ )
484
+ system = _SYSTEM.format(scale=self.scale)
485
+ user = self._prompt(content, context)
486
+
487
+ # Build every client up front: a construction failure (e.g. no
488
+ # replay_path in replay mode) is a configuration error and must
489
+ # raise, unlike a failed judge call which degrades to error scores.
490
+ clients = [(j, self._client_for(j)) for j in self._active_judges()]
491
+
492
+ # raw[judge_key][dim] = integer score returned by that judge;
493
+ # rationales[judge_key] = that judge's free-text justification.
494
+ raw: dict[str, dict] = {}
495
+ rationales: dict[str, str | None] = {}
496
+ for judge, client in clients:
497
+ # A transient client failure (429/5xx/timeout) backs off and
498
+ # retries per the suite retry policy; a terminal one degrades to
499
+ # error scores as before. functools.partial (not a closure) keeps
500
+ # the retried call clean of loop-variable capture.
501
+ call = functools.partial(
502
+ client.judge,
503
+ key=content,
504
+ system=system,
505
+ user=user,
506
+ dimensions=self.dimensions,
507
+ scale=self.scale,
508
+ images=images,
509
+ )
510
+ try:
511
+ result = await retry.call_with_retry(call, self._retry)
512
+ except (KeyError, ValueError, RuntimeError) as exc:
513
+ return self._error_scores(f'judge error: {exc}')
514
+ result = result or {}
515
+ raw[judge['key']] = result.get('scores') or {}
516
+ rationales[judge['key']] = result.get('rationale')
517
+
518
+ return self._build_scores(case, raw, rationales)
519
+
520
+ def _build_scores(
521
+ self,
522
+ case: models.Case,
523
+ raw: dict[str, dict],
524
+ rationales: dict[str, str | None] | None = None,
525
+ ) -> list[models.Score]:
526
+ active = self._active_judges()
527
+ versions = ','.join(f'{j["key"]}@{j["judge_version"]}' for j in active)
528
+ detail = f'judges={versions}'
529
+ rationales = rationales or {}
530
+ out: list[models.Score] = []
531
+
532
+ def score(
533
+ metric: str,
534
+ value: float | None,
535
+ judges: list[models.JudgeDetail] | None = None,
536
+ ) -> models.Score:
537
+ return models.Score(
538
+ grader=self.name,
539
+ metric=metric,
540
+ value=value,
541
+ detail=detail,
542
+ case_id=case.id,
543
+ kind='per_case',
544
+ judges=judges or [],
545
+ )
546
+
547
+ # Per-judge breakdown behind the aggregate means: the raw 1..scale
548
+ # points each dimension got and the free-text rationale. Attached to
549
+ # the overall score so review can read back *why* without re-running.
550
+ judge_details: list[models.JudgeDetail] = []
551
+ for judge in active:
552
+ jkey = judge['key']
553
+ points = {k: raw.get(jkey, {}).get(k) for k, _ in self.dimensions}
554
+ norm = [
555
+ v
556
+ for k, _ in self.dimensions
557
+ if (v := self._normalize(raw.get(jkey, {}).get(k))) is not None
558
+ ]
559
+ judge_details.append(
560
+ models.JudgeDetail(
561
+ key=jkey,
562
+ version=judge['judge_version'],
563
+ rationale=rationales.get(jkey),
564
+ points=points,
565
+ overall=sum(norm) / len(norm) if norm else None,
566
+ )
567
+ )
568
+
569
+ # Panel mean per dimension (identical to the single judge's value
570
+ # when there is only one).
571
+ panel_dims: list[float] = []
572
+ for key, _ in self.dimensions:
573
+ per_judge = [
574
+ self._normalize(raw[j['key']].get(key)) for j in active
575
+ ]
576
+ present = [v for v in per_judge if v is not None]
577
+ value = sum(present) / len(present) if present else None
578
+ if value is not None:
579
+ panel_dims.append(value)
580
+ out.append(score(f'{self.name}.{key}', value))
581
+
582
+ overall = sum(panel_dims) / len(panel_dims) if panel_dims else None
583
+ out.append(
584
+ score(f'{self.name}.overall', overall, judges=judge_details)
585
+ )
586
+
587
+ if len(active) <= 1:
588
+ return out
589
+
590
+ # Per-judge overall, so a systematically generous judge is visible.
591
+ for jd in judge_details:
592
+ out.append(score(f'{self.name}.{jd.key}.overall', jd.overall))
593
+
594
+ # Inter-judge disagreement, in raw points, per dimension.
595
+ spreads: list[float] = []
596
+ for key, _ in self.dimensions:
597
+ raws = [
598
+ raw[j['key']].get(key)
599
+ for j in active
600
+ if isinstance(raw[j['key']].get(key), (int, float))
601
+ and not isinstance(raw[j['key']].get(key), bool)
602
+ ]
603
+ if len(raws) >= 2:
604
+ spreads.append(max(raws) - min(raws))
605
+ mean_spread = sum(spreads) / len(spreads) if spreads else 0.0
606
+ max_spread = max(spreads) if spreads else 0.0
607
+ out.append(score(f'{self.name}.disagreement', mean_spread))
608
+ out.append(
609
+ score(
610
+ f'{self.name}.flagged',
611
+ 1.0 if max_spread >= self.disagreement_threshold else 0.0,
612
+ )
613
+ )
614
+ return out
615
+
616
+ def _metric_names(self) -> list[str]:
617
+ active = self._active_judges()
618
+ names = [f'{self.name}.{k}' for k, _ in self.dimensions]
619
+ names.append(f'{self.name}.overall')
620
+ if len(active) > 1:
621
+ names += [f'{self.name}.{j["key"]}.overall' for j in active]
622
+ names += [f'{self.name}.disagreement', f'{self.name}.flagged']
623
+ return names
624
+
625
+ def _error_scores(self, detail: str) -> list[models.Score]:
626
+ return [
627
+ models.Score(
628
+ grader=self.name,
629
+ metric=metric,
630
+ value=None,
631
+ detail=detail,
632
+ kind='per_case',
633
+ )
634
+ for metric in self._metric_names()
635
+ ]