auditkit 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. auditkit/README.md +99 -0
  2. auditkit/__init__.py +177 -0
  3. auditkit/__main__.py +3 -0
  4. auditkit/_bootstrap.py +77 -0
  5. auditkit/_identity_guard.py +99 -0
  6. auditkit/adapter.py +264 -0
  7. auditkit/annotator.py +339 -0
  8. auditkit/api.py +502 -0
  9. auditkit/assets/auditkit_logo.png +0 -0
  10. auditkit/cache.py +47 -0
  11. auditkit/cli.py +417 -0
  12. auditkit/comparison.py +563 -0
  13. auditkit/diff.py +265 -0
  14. auditkit/errors.py +54 -0
  15. auditkit/evaluator.py +20 -0
  16. auditkit/experiment.py +145 -0
  17. auditkit/hf_publish.py +262 -0
  18. auditkit/lmeval_engine.py +550 -0
  19. auditkit/loaders.py +121 -0
  20. auditkit/logs.py +18 -0
  21. auditkit/metric.py +199 -0
  22. auditkit/metrics/README.md +15 -0
  23. auditkit/metrics/__init__.py +0 -0
  24. auditkit/metrics/code.py +222 -0
  25. auditkit/metrics/embedding.py +131 -0
  26. auditkit/metrics/encoder_judge.py +423 -0
  27. auditkit/metrics/generation.py +331 -0
  28. auditkit/metrics/guard.py +412 -0
  29. auditkit/metrics/hallucination.py +45 -0
  30. auditkit/metrics/judge.py +547 -0
  31. auditkit/metrics/pairwise.py +153 -0
  32. auditkit/metrics/perf.py +53 -0
  33. auditkit/metrics/rag.py +149 -0
  34. auditkit/metrics/security.py +64 -0
  35. auditkit/metrics/toxicity.py +238 -0
  36. auditkit/model/README.md +16 -0
  37. auditkit/model/__init__.py +485 -0
  38. auditkit/model/anthropic.py +94 -0
  39. auditkit/model/api_gen.py +133 -0
  40. auditkit/model/groq_gen.py +121 -0
  41. auditkit/model/hf_gen.py +385 -0
  42. auditkit/model/lexsi.py +155 -0
  43. auditkit/model/litellm_gen.py +65 -0
  44. auditkit/model/openai.py +90 -0
  45. auditkit/model/openrouter_gen.py +152 -0
  46. auditkit/model/vllm_gen.py +316 -0
  47. auditkit/model_compare.py +655 -0
  48. auditkit/redteam/README.md +9 -0
  49. auditkit/redteam/__init__.py +26 -0
  50. auditkit/redteam/detector.py +37 -0
  51. auditkit/redteam/detectors/README.md +5 -0
  52. auditkit/redteam/detectors/builtin.py +126 -0
  53. auditkit/redteam/probe.py +39 -0
  54. auditkit/redteam/probes/README.md +5 -0
  55. auditkit/redteam/probes/builtin.py +85 -0
  56. auditkit/redteam/runner.py +206 -0
  57. auditkit/registry.py +65 -0
  58. auditkit/report.py +278 -0
  59. auditkit/report_format.py +52 -0
  60. auditkit/router.py +54 -0
  61. auditkit/runner.py +575 -0
  62. auditkit/runspec.py +159 -0
  63. auditkit/sample.py +40 -0
  64. auditkit/scenario.py +88 -0
  65. auditkit/scenarios/README.md +10 -0
  66. auditkit/scenarios/__init__.py +4 -0
  67. auditkit/scenarios/arc.py +33 -0
  68. auditkit/scenarios/gsm8k.py +32 -0
  69. auditkit/scenarios/hellaswag.py +33 -0
  70. auditkit/scenarios/humaneval.py +32 -0
  71. auditkit/scenarios/mmlu.py +34 -0
  72. auditkit/scenarios/truthfulqa.py +33 -0
  73. auditkit/score.py +165 -0
  74. auditkit/scorers.py +117 -0
  75. auditkit/scoring.py +79 -0
  76. auditkit/types.py +69 -0
  77. auditkit-1.0.0.dist-info/METADATA +396 -0
  78. auditkit-1.0.0.dist-info/RECORD +81 -0
  79. auditkit-1.0.0.dist-info/WHEEL +4 -0
  80. auditkit-1.0.0.dist-info/entry_points.txt +2 -0
  81. auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/runner.py ADDED
@@ -0,0 +1,575 @@
1
+ """The runner: five separately-testable stages that drive one evaluation.
2
+
3
+ ``build_requests`` turns samples into requests, ``execute`` sends them to the
4
+ model in one batch and realigns the results, ``annotate`` runs any annotators,
5
+ ``score_one`` applies the metrics to a sample's output and records a
6
+ :class:`Prediction`, and ``aggregate`` rolls per-sample scores into per-metric
7
+ :class:`~auditkit.score.Stat`. :meth:`Runner.run` chains them into a
8
+ :class:`RunResult`. Keeping the stages independent is what makes each testable in
9
+ isolation and lets techniques swap one stage without touching the others.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ import threading
16
+ import time
17
+ from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeout
18
+ from dataclasses import asdict
19
+ from typing import Any, Optional
20
+
21
+ from .adapter import Adapter
22
+ from .cache import DiskCache
23
+ from .errors import CapabilityError, ExtraNotInstalled, ModelError, ModelTimeout
24
+ from .metric import Metric
25
+ from .metrics.perf import LatencyStats, Throughput
26
+ from .model import Generated, Model, Request, Result_, _prompt_text
27
+ from .report import Prediction, RunResult
28
+ from .runspec import RunConfig, RunSpec
29
+ from .sample import Sample
30
+ from .scenario import Scenario
31
+ from .score import Score, Stat
32
+ from .types import Capability
33
+
34
+ logger = logging.getLogger(__name__)
35
+
36
+ _generate_lock = threading.Lock()
37
+
38
+
39
+ def _score_doc(score: Score) -> dict[str, Any]:
40
+ """Every field on *score* that actually carries information.
41
+
42
+ ``Score`` supports rich, judge-grade output (``reason``, ``threshold``,
43
+ ``label``, per-score ``metadata``, ...), but most deterministic metrics
44
+ never set most of it -- serializing every field unconditionally would
45
+ bury the useful cases (a judge's verdict reasoning) under a wall of
46
+ ``None``s and defaults for the common case (``ExactMatch`` only ever
47
+ sets ``name``/``value``). Keep only fields that aren't ``None`` or an
48
+ empty container, plus the derived ``passed`` (True/False/None against
49
+ ``threshold``), which isn't itself a dataclass field so ``asdict``
50
+ wouldn't pick it up.
51
+ """
52
+ doc = {k: v for k, v in asdict(score).items() if v not in (None, {}, [])}
53
+ if score.passed is not None:
54
+ doc["passed"] = score.passed
55
+ return doc
56
+
57
+
58
+ class Runner:
59
+ """Drives a :class:`RunSpec` to a :class:`RunResult` in five stages."""
60
+
61
+ def build_requests(
62
+ self, scenario: Scenario, adapter: Adapter, config: RunConfig
63
+ ) -> list[tuple[Sample, list[Request]]]:
64
+ samples = list(scenario.samples())
65
+ if config.limit is not None:
66
+ samples = samples[: config.limit]
67
+ split_cfg = config.split
68
+ if split_cfg is not None:
69
+ train, val, test = self._split_samples(samples, split_cfg)
70
+ samples = test or val or train
71
+ if hasattr(adapter, 'pool') and train:
72
+ adapter.pool = train
73
+ return [(s, adapter.adapt(s, config)) for s in samples]
74
+
75
+ def _split_samples(self, samples, split_cfg):
76
+ n = len(samples)
77
+ t = split_cfg
78
+ if t.strategy == "sequential":
79
+ train_end = int(n * t.train_ratio)
80
+ val_end = train_end + int(n * t.val_ratio)
81
+ return samples[:train_end], samples[train_end:val_end], samples[val_end:]
82
+ elif t.strategy == "random":
83
+ import random
84
+ rng = random.Random(t.seed)
85
+ idx = list(range(n))
86
+ rng.shuffle(idx)
87
+ shuffled = [samples[i] for i in idx]
88
+ train_end = int(n * t.train_ratio)
89
+ val_end = train_end + int(n * t.val_ratio)
90
+ return shuffled[:train_end], shuffled[train_end:val_end], shuffled[val_end:]
91
+ elif t.strategy == "stratified":
92
+ from collections import defaultdict
93
+ groups = defaultdict(list)
94
+ for i, s in enumerate(samples):
95
+ key = str(getattr(s, 'kind', None) or str(getattr(s, 'id', None) or i))
96
+ groups[key].append(s)
97
+ train, val, test = [], [], []
98
+ for g in groups.values():
99
+ gn = len(g)
100
+ te = int(gn * t.train_ratio)
101
+ ve = te + int(gn * t.val_ratio)
102
+ train.extend(g[:te])
103
+ val.extend(g[te:ve])
104
+ test.extend(g[ve:])
105
+ return train, val, test
106
+ else:
107
+ return samples, [], []
108
+
109
+ def execute(
110
+ self, model: Model, batch: list[tuple[Sample, list[Request]]],
111
+ concurrency: int = 1, max_retries: int = 3, retry_delay: float = 1.0,
112
+ timeout: float | None = None,
113
+ latency: LatencyStats | None = None, throughput: Throughput | None = None,
114
+ ) -> list[tuple[Sample, list[Result_]]]:
115
+ flat, counts = self._flatten(batch)
116
+ if not flat:
117
+ return [(s, []) for s, _ in batch]
118
+
119
+ # Fan out across threads ONLY when asked for it AND the model declares it
120
+ # is safe to call concurrently. A single local model (HFGenModel/vLLM) is
121
+ # not thread-safe, so calling generate() from many threads at once races
122
+ # the same model — corrupt output or a crash. When we don't parallelize we
123
+ # make a single batched generate() call, which also lets batched backends
124
+ # process every request at once instead of fragmenting the batch.
125
+ parallel = concurrency > 1 and getattr(model, "threadsafe", False)
126
+ if not parallel:
127
+ results = self._retry_generate(model, flat, max_retries, retry_delay, timeout, latency, throughput)
128
+ else:
129
+ # Never make more chunks than there are requests, so we never call
130
+ # generate([]) on empty chunks (wasteful, and some backends choke).
131
+ n_chunks = min(concurrency, len(flat))
132
+ chunks = self._chunk(flat, n_chunks)
133
+ with ThreadPoolExecutor(max_workers=n_chunks) as pool:
134
+ chunk_results = list(pool.map(
135
+ lambda c: self._retry_generate(model, c, max_retries, retry_delay, timeout, latency, throughput),
136
+ chunks,
137
+ ))
138
+ results = [r for cr in chunk_results for r in cr]
139
+
140
+ out: list[tuple[Sample, list[Result_]]] = []
141
+ cursor = 0
142
+ for (sample, _requests), n in zip(batch, counts):
143
+ out.append((sample, results[cursor:cursor + n]))
144
+ cursor += n
145
+ return out
146
+
147
+ def _retry_generate(self, model, requests, max_retries, retry_delay, timeout=None,
148
+ latency: LatencyStats | None = None, throughput: Throughput | None = None):
149
+ """Calls ``model.generate(requests)`` with retry/backoff.
150
+
151
+ When *latency*/*throughput* are given, records one sample per actual
152
+ model-call attempt — success or failure alike, since a failing call
153
+ still occupied real model/network time — but never the artificial
154
+ ``time.sleep()`` backoff between retries, which is our own throttling,
155
+ not the model's latency. This is a *call*-level measurement, not a
156
+ true per-request one: batched backends (the default — see
157
+ ``execute()``'s comment) answer many requests in a single call, so
158
+ there is no per-request timestamp to read. One sample per call,
159
+ tagged with how many requests it served, is the honest granularity
160
+ actually available without breaking batching; it still yields a real
161
+ distribution across a run's several chunks/attempts and a real
162
+ requests/sec figure.
163
+ """
164
+ def _record(elapsed_ms: float, served: bool) -> None:
165
+ # `served` distinguishes a successful call (which actually
166
+ # returned results for every one of `requests`) from a failed
167
+ # attempt: a failing call still occupied real model/network
168
+ # time, so it's a legitimate latency sample -- but it served
169
+ # zero requests, not len(requests), so it must NOT add to
170
+ # Throughput's request count. Previously every attempt credited
171
+ # the full batch size regardless of success, so a run that hit
172
+ # even one retry over-counted total_requests (and therefore
173
+ # inflated requests/sec) by the number of failed attempts.
174
+ if latency is None and throughput is None:
175
+ return
176
+ with _generate_lock:
177
+ if latency is not None:
178
+ latency.record(elapsed_ms)
179
+ if throughput is not None:
180
+ throughput.record(elapsed_ms, len(requests) if served else 0)
181
+
182
+ last_exc = None
183
+ for attempt in range(max_retries + 1):
184
+ start = time.perf_counter()
185
+ try:
186
+ if timeout is not None:
187
+ with ThreadPoolExecutor(max_workers=1) as pool:
188
+ future = pool.submit(model.generate, requests)
189
+ result = future.result(timeout=timeout)
190
+ else:
191
+ result = model.generate(requests)
192
+ except FuturesTimeout:
193
+ _record((time.perf_counter() - start) * 1000.0, served=False)
194
+ raise ModelTimeout(f"generate timed out after {timeout}s")
195
+ except Exception as e:
196
+ _record((time.perf_counter() - start) * 1000.0, served=False)
197
+ last_exc = e
198
+ if attempt < max_retries:
199
+ time.sleep(retry_delay * (2 ** attempt))
200
+ logger.warning("Retry %d/%d: %s", attempt + 1, max_retries, e)
201
+ continue
202
+ _record((time.perf_counter() - start) * 1000.0, served=True)
203
+ return result
204
+ raise ModelError(f"generate failed after {max_retries} retries") from last_exc
205
+
206
+ def _flatten(self, batch):
207
+ flat = []
208
+ counts = []
209
+ for _sample, reqs in batch:
210
+ flat.extend(reqs)
211
+ counts.append(len(reqs))
212
+ return flat, counts
213
+
214
+ def _chunk(self, items, n_chunks):
215
+ k, m = divmod(len(items), n_chunks)
216
+ return [items[i * k + min(i, m):(i + 1) * k + min(i + 1, m)] for i in range(n_chunks)]
217
+
218
+ def annotate(
219
+ self, annotators: list[Any], sample: Sample, results: list[Result_]
220
+ ) -> dict[str, Any]:
221
+ context: dict[str, Any] = {}
222
+ for annotator in annotators:
223
+ context[getattr(annotator, "name", type(annotator).__name__)] = annotator.annotate(
224
+ sample, results
225
+ )
226
+ return context
227
+
228
+ def score_one(
229
+ self,
230
+ metrics: list[Metric],
231
+ sample: Sample,
232
+ output: str,
233
+ context: Any,
234
+ run_id: str,
235
+ errors: list | None = None,
236
+ prompt: Optional[str] = None,
237
+ extracted_by: str | None = None,
238
+ ) -> tuple[list[Score], Prediction]:
239
+ # extracted_by names an annotator whose context["extracted"] value
240
+ # should be scored instead of the raw output (e.g. RegexAnnotator
241
+ # pulling "42" out of "...FINAL ANSWER: 42"). Missing/misspelled
242
+ # name degrades to raw output with a warning, not a crash -- a typo
243
+ # here shouldn't take down a whole run.
244
+ scoring_output = output
245
+ if extracted_by:
246
+ entry = context.get(extracted_by) if isinstance(context, dict) else None
247
+ if isinstance(entry, dict) and "extracted" in entry:
248
+ scoring_output = entry["extracted"]
249
+ else:
250
+ logger.warning("extracted_by=%r has no 'extracted' entry in context", extracted_by)
251
+ scores: list[Score] = []
252
+ for metric in metrics:
253
+ if not metric.applicable(sample):
254
+ continue
255
+ try:
256
+ produced = metric.score(sample, scoring_output, context)
257
+ except ExtraNotInstalled:
258
+ # A missing extra is a run-wide configuration problem, not a
259
+ # per-sample scoring failure -- every remaining sample would
260
+ # hit the exact same error. Swallowing it into a fake 0.0
261
+ # score (the old behavior) is indistinguishable in
262
+ # `headline` from "the model's output genuinely scored
263
+ # zero," silently corrupting the aggregate. Fail loudly and
264
+ # immediately instead, so the missing dependency is obvious.
265
+ raise
266
+ except Exception as e:
267
+ # Don't fake a 0.0 Score here -- that's indistinguishable from
268
+ # "the model's output genuinely scored zero" in every mean/
269
+ # aggregate downstream (the exact corruption ExtraNotInstalled
270
+ # above is deliberately not swallowed into either). A metric
271
+ # crashing on this sample is a computation failure, not a
272
+ # data point -- skip it so this sample's absence from
273
+ # `metric.name`'s count is the honest signal, not a fabricated
274
+ # score. `errors` still records exactly what happened.
275
+ logger.error("Metric %s failed on sample %s: %s", metric.name, sample.id, e)
276
+ if errors is not None:
277
+ errors.append({"sample_id": sample.id, "metric": metric.name, "error": str(e)})
278
+ continue
279
+ produced_list = produced if isinstance(produced, list) else [produced]
280
+ for s in produced_list:
281
+ # Authoritative, not a fallback: metric.direction is required
282
+ # (Metric.__init_subclass__ enforces it), so every Score a
283
+ # metric produces carries ITS metric's declared direction,
284
+ # regardless of what value the metric's own score() happened
285
+ # to set (or forgot to). This is what makes direction
286
+ # actually reliable for RunComparison/compare_models grading.
287
+ s.direction = metric.direction
288
+ scores.extend(produced_list)
289
+
290
+ primary = scores[0] if scores else None
291
+ correct: Optional[bool]
292
+ if primary is None:
293
+ correct = None
294
+ elif primary.passed is not None:
295
+ correct = primary.passed
296
+ else:
297
+ correct = primary.value == 1.0
298
+
299
+ prediction = Prediction(
300
+ run_id=run_id,
301
+ task=sample.task or sample.kind.value,
302
+ sample_id=sample.id or "",
303
+ # The adapter-built prompt actually sent to the model when known
304
+ # (a RAGAdapter's injected context, a ChatAdapter's system-prompt
305
+ # wrapping, ...) -- falls back to the raw sample input only when
306
+ # no request was built (e.g. an errored-out sample).
307
+ prompt=prompt if prompt is not None else sample.input_text,
308
+ raw_output=output,
309
+ parsed_answer=scoring_output,
310
+ expected=sample.target,
311
+ correct=correct,
312
+ score=primary.value if primary else None,
313
+ context=context,
314
+ metadata={"scores": [_score_doc(s) for s in scores]},
315
+ )
316
+ return scores, prediction
317
+
318
+ def aggregate(self, scores: list[Score]) -> dict[str, Stat]:
319
+ stats: dict[str, Stat] = {}
320
+ for score in scores:
321
+ stats.setdefault(score.name, Stat(score.name)).add(score.value)
322
+ return stats
323
+
324
+ def run(self, spec: RunSpec, *, verbose: bool = False) -> RunResult:
325
+ fingerprint = spec.fingerprint()
326
+ config = spec.config
327
+
328
+ cache = DiskCache()
329
+ cached = cache.get(fingerprint)
330
+ if cached is not None:
331
+ return cached
332
+
333
+ run_id = spec.run_name or fingerprint
334
+
335
+ batch = self.build_requests(spec.scenario, spec.adapter, config)
336
+ logger.info("Starting run with %d samples", len(batch))
337
+
338
+ has_loglikelihood = any(
339
+ r.request_type == "loglikelihood"
340
+ for _, reqs in batch
341
+ for r in reqs
342
+ )
343
+
344
+ all_scores: list[Score] = []
345
+ predictions: list[Prediction] = []
346
+ errors: list[dict] = []
347
+ failed_count: int = 0
348
+ latency = LatencyStats()
349
+ throughput = Throughput()
350
+ # Real, provider-reported token counts -- only ever non-zero when a
351
+ # backend actually populates Result_.usage (the hosted-API backends;
352
+ # see openai.py/anthropic.py/groq_gen.py/api_gen.py/litellm_gen.py).
353
+ # Local backends and callables leave this at zero -- never estimated
354
+ # by counting characters or guessing a tokenizer.
355
+ token_usage = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
356
+
357
+ def _accumulate_usage(results: list[Result_]) -> None:
358
+ for r in results:
359
+ for key in token_usage:
360
+ token_usage[key] += r.usage.get(key, 0) or 0
361
+
362
+ if getattr(spec.model, "reads_actual_output", False):
363
+ # Score pre-generated answers (Sample.actual_output) — no model call.
364
+ for index, (sample, _reqs) in enumerate(batch):
365
+ if sample.id is None:
366
+ sample.id = str(index)
367
+ output = sample.actual_output or ""
368
+ context = self.annotate(
369
+ spec.annotators, sample, [Result_(completions=[Generated(text=output)])]
370
+ )
371
+ scores, prediction = self.score_one(
372
+ spec.metrics, sample, output, context, run_id, errors,
373
+ extracted_by=spec.extracted_by,
374
+ )
375
+ all_scores.extend(scores)
376
+ predictions.append(prediction)
377
+ elif has_loglikelihood:
378
+ if not spec.model.supports(Capability.LOGLIKELIHOOD):
379
+ raise CapabilityError(
380
+ f"{spec.model.name} does not declare LOGLIKELIHOOD"
381
+ )
382
+ flat = [r for _, reqs in batch for r in reqs]
383
+ logger.info("Executing %d requests", len(flat))
384
+ last_exc = None
385
+ for attempt in range(config.max_retries + 1):
386
+ start = time.perf_counter()
387
+ try:
388
+ lls = spec.model.loglikelihood(flat)
389
+ except Exception as e:
390
+ # A failing attempt occupied real time (legitimate
391
+ # latency sample) but served zero requests -- crediting
392
+ # it with len(flat) would double-count against the
393
+ # eventual successful retry (see the identical fix in
394
+ # _retry_generate() above).
395
+ elapsed_ms = (time.perf_counter() - start) * 1000.0
396
+ latency.record(elapsed_ms)
397
+ throughput.record(elapsed_ms, 0)
398
+ last_exc = e
399
+ if attempt < config.max_retries:
400
+ time.sleep(config.retry_delay * (2 ** attempt))
401
+ logger.warning("Retry %d/%d for loglikelihood: %s", attempt + 1, config.max_retries, e)
402
+ continue
403
+ elapsed_ms = (time.perf_counter() - start) * 1000.0
404
+ latency.record(elapsed_ms)
405
+ throughput.record(elapsed_ms, len(flat))
406
+ break
407
+ else:
408
+ raise ModelError(f"loglikelihood failed after {config.max_retries} retries") from last_exc
409
+ logger.info("Got %d responses", len(lls))
410
+ cursor = 0
411
+ for index, (sample, reqs) in enumerate(batch):
412
+ try:
413
+ if sample.id is None:
414
+ sample.id = str(index)
415
+ n = len(reqs)
416
+ sample_lls = lls[cursor: cursor + n]
417
+ cursor += n
418
+ choice_probs = [ll.logprob for ll in sample_lls]
419
+ argmax_idx = max(
420
+ range(len(choice_probs)), key=lambda i: choice_probs[i]
421
+ )
422
+ output = sample.choices[argmax_idx] if sample.choices else ""
423
+ context = self.annotate(
424
+ spec.annotators, sample, [Result_(completions=[Generated(text=output)])]
425
+ )
426
+ context["choice_likelihoods"] = choice_probs
427
+ prompt = _prompt_text(reqs[0].prompt) if reqs else None
428
+ # extracted_by deliberately NOT threaded through here:
429
+ # `output` on this path is already sample.choices[argmax_idx]
430
+ # -- the exact, correct, final choice text picked by
431
+ # comparing logprobs, not free-form generated text. There
432
+ # is nothing to clean up, and applying extraction here is
433
+ # actively dangerous for choice-index metrics (Acc/AccNorm
434
+ # interpret a bare digit-string as a CHOICE INDEX, not "a
435
+ # number that happened to appear in the choice text") --
436
+ # a correct pick can silently get scored as a different,
437
+ # wrong choice, or as no match at all. Annotators still
438
+ # run on this branch (context is populated normally, for
439
+ # logging/inspection), only the extraction hookup into
440
+ # scoring is structurally disabled here.
441
+ scores, prediction = self.score_one(
442
+ spec.metrics, sample, output, context, run_id, errors, prompt=prompt,
443
+ )
444
+ all_scores.extend(scores)
445
+ predictions.append(prediction)
446
+ except Exception as e:
447
+ logger.error("Sample %s failed: %s", sample.id, e)
448
+ failed_count += 1
449
+ predictions.append(Prediction(
450
+ run_id=run_id,
451
+ task=getattr(sample, 'task', '') or '',
452
+ sample_id=sample.id or str(index),
453
+ prompt=str(sample.input),
454
+ raw_output="",
455
+ parsed_answer=None,
456
+ expected=None,
457
+ correct=False,
458
+ score=0.0,
459
+ ))
460
+ errors.append({"sample_id": sample.id or str(index), "error": str(e)})
461
+ continue
462
+ logger.debug("Scored %d/%d samples", index + 1, len(batch))
463
+ if verbose and (index + 1) % 10 == 0:
464
+ print(".", end="", flush=True)
465
+ else:
466
+ requests = [r for _, reqs in batch for r in reqs]
467
+ logger.info("Executing %d requests", len(requests))
468
+ executed = self.execute(
469
+ spec.model, batch, concurrency=config.concurrency, max_retries=config.max_retries,
470
+ retry_delay=config.retry_delay, timeout=config.timeout,
471
+ latency=latency, throughput=throughput,
472
+ )
473
+ logger.info("Got %d responses", len(executed))
474
+ for index, (sample, results) in enumerate(executed):
475
+ try:
476
+ if sample.id is None:
477
+ sample.id = str(index)
478
+ output = results[0].text if results else ""
479
+ _accumulate_usage(results)
480
+ context = self.annotate(spec.annotators, sample, results)
481
+ # execute() preserves batch's order/length, so batch[index]
482
+ # is the (sample, requests) pair this result came from.
483
+ reqs = batch[index][1]
484
+ prompt = _prompt_text(reqs[0].prompt) if reqs else None
485
+ scores, prediction = self.score_one(
486
+ spec.metrics, sample, output, context, run_id, errors, prompt=prompt,
487
+ extracted_by=spec.extracted_by,
488
+ )
489
+ all_scores.extend(scores)
490
+ predictions.append(prediction)
491
+ except Exception as e:
492
+ logger.error("Sample %s failed: %s", sample.id, e)
493
+ failed_count += 1
494
+ predictions.append(Prediction(
495
+ run_id=run_id,
496
+ task=getattr(sample, 'task', '') or '',
497
+ sample_id=sample.id or str(index),
498
+ prompt=str(sample.input),
499
+ raw_output="",
500
+ parsed_answer=None,
501
+ expected=None,
502
+ correct=False,
503
+ score=0.0,
504
+ ))
505
+ errors.append({"sample_id": sample.id or str(index), "error": str(e)})
506
+ continue
507
+ logger.debug("Scored %d/%d samples", index + 1, len(executed))
508
+ if verbose and (index + 1) % 10 == 0:
509
+ print(".", end="", flush=True)
510
+
511
+ stats = self.aggregate(all_scores)
512
+ logger.info("Run complete – %d metrics computed", len(stats))
513
+ if verbose:
514
+ print(f" done – {len(stats)} metrics")
515
+ headline = {name: stat.mean for name, stat in stats.items()}
516
+ # All three of perf/model_size/token_usage are populated unless the
517
+ # caller opts out via RunConfig.track_performance=False (default
518
+ # True -- model_info() below can be a real, if best-effort,
519
+ # introspection cost for local backends, so callers who don't want
520
+ # that can disable it). This is a *reporting* gate, not a timing
521
+ # one -- latency/throughput/usage were already measured for free
522
+ # above as part of running the requests; we just don't surface them
523
+ # when disabled.
524
+ if config.track_performance:
525
+ # Empty on the `reads_actual_output` path (no model call was made
526
+ # at all -- nothing to time) and on the loglikelihood/generative
527
+ # paths when there were zero requests; LatencyStats/Throughput
528
+ # .stats() both degrade to zeroed dicts rather than raising.
529
+ perf = {"latency_ms": latency.stats(), "throughput": throughput.stats()}
530
+ # Token throughput: derived from two numbers that already exist
531
+ # separately (real elapsed call time, real token counts from the
532
+ # backend's own API response) but were never combined into a
533
+ # tokens/sec figure. total_time_ms is the same denominator
534
+ # throughput["rps"] already uses -- real wall-clock time spent in
535
+ # model calls, not the whole run (annotation/scoring time excluded).
536
+ total_time_s = perf["throughput"].get("total_time_ms", 0.0) / 1000.0
537
+ if total_time_s > 0:
538
+ perf["throughput"]["output_tokens_per_sec"] = token_usage["completion_tokens"] / total_time_s
539
+ perf["throughput"]["total_tokens_per_sec"] = token_usage["total_tokens"] / total_time_s
540
+ else:
541
+ perf["throughput"]["output_tokens_per_sec"] = 0.0
542
+ perf["throughput"]["total_tokens_per_sec"] = 0.0
543
+ # Never caller-supplied: introspected for local backends (real
544
+ # parameter count/sparsity/size), identity-only for hosted-API
545
+ # backends (nothing to measure -- see Model.model_info()).
546
+ # Guarded because a backend's own introspection (e.g. loading a
547
+ # model just to inspect it) is best-effort and shouldn't take
548
+ # down an otherwise successful run. Skipped entirely (not just
549
+ # hidden) when tracking is off, since this is the one field here
550
+ # that can cost real time to compute, not just to report.
551
+ try:
552
+ model_size = spec.model.model_info() if spec.model else {}
553
+ except Exception as e:
554
+ logger.warning("model_info() failed for %s: %s", getattr(spec.model, "name", spec.model), e)
555
+ model_size = {}
556
+ else:
557
+ perf = None
558
+ model_size = None
559
+ token_usage = None
560
+ result = RunResult(
561
+ run_id=run_id,
562
+ fingerprint=fingerprint,
563
+ stats=stats,
564
+ predictions=predictions,
565
+ headline=headline,
566
+ config=spec.config,
567
+ model_spec=str(spec.model) if spec.model else None,
568
+ errors=errors,
569
+ failed_count=failed_count,
570
+ model_size=model_size,
571
+ token_usage=token_usage,
572
+ perf=perf,
573
+ )
574
+ cache.set(fingerprint, result)
575
+ return result