auditkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auditkit/README.md +99 -0
- auditkit/__init__.py +177 -0
- auditkit/__main__.py +3 -0
- auditkit/_bootstrap.py +77 -0
- auditkit/_identity_guard.py +99 -0
- auditkit/adapter.py +264 -0
- auditkit/annotator.py +339 -0
- auditkit/api.py +502 -0
- auditkit/assets/auditkit_logo.png +0 -0
- auditkit/cache.py +47 -0
- auditkit/cli.py +417 -0
- auditkit/comparison.py +563 -0
- auditkit/diff.py +265 -0
- auditkit/errors.py +54 -0
- auditkit/evaluator.py +20 -0
- auditkit/experiment.py +145 -0
- auditkit/hf_publish.py +262 -0
- auditkit/lmeval_engine.py +550 -0
- auditkit/loaders.py +121 -0
- auditkit/logs.py +18 -0
- auditkit/metric.py +199 -0
- auditkit/metrics/README.md +15 -0
- auditkit/metrics/__init__.py +0 -0
- auditkit/metrics/code.py +222 -0
- auditkit/metrics/embedding.py +131 -0
- auditkit/metrics/encoder_judge.py +423 -0
- auditkit/metrics/generation.py +331 -0
- auditkit/metrics/guard.py +412 -0
- auditkit/metrics/hallucination.py +45 -0
- auditkit/metrics/judge.py +547 -0
- auditkit/metrics/pairwise.py +153 -0
- auditkit/metrics/perf.py +53 -0
- auditkit/metrics/rag.py +149 -0
- auditkit/metrics/security.py +64 -0
- auditkit/metrics/toxicity.py +238 -0
- auditkit/model/README.md +16 -0
- auditkit/model/__init__.py +485 -0
- auditkit/model/anthropic.py +94 -0
- auditkit/model/api_gen.py +133 -0
- auditkit/model/groq_gen.py +121 -0
- auditkit/model/hf_gen.py +385 -0
- auditkit/model/lexsi.py +155 -0
- auditkit/model/litellm_gen.py +65 -0
- auditkit/model/openai.py +90 -0
- auditkit/model/openrouter_gen.py +152 -0
- auditkit/model/vllm_gen.py +316 -0
- auditkit/model_compare.py +655 -0
- auditkit/redteam/README.md +9 -0
- auditkit/redteam/__init__.py +26 -0
- auditkit/redteam/detector.py +37 -0
- auditkit/redteam/detectors/README.md +5 -0
- auditkit/redteam/detectors/builtin.py +126 -0
- auditkit/redteam/probe.py +39 -0
- auditkit/redteam/probes/README.md +5 -0
- auditkit/redteam/probes/builtin.py +85 -0
- auditkit/redteam/runner.py +206 -0
- auditkit/registry.py +65 -0
- auditkit/report.py +278 -0
- auditkit/report_format.py +52 -0
- auditkit/router.py +54 -0
- auditkit/runner.py +575 -0
- auditkit/runspec.py +159 -0
- auditkit/sample.py +40 -0
- auditkit/scenario.py +88 -0
- auditkit/scenarios/README.md +10 -0
- auditkit/scenarios/__init__.py +4 -0
- auditkit/scenarios/arc.py +33 -0
- auditkit/scenarios/gsm8k.py +32 -0
- auditkit/scenarios/hellaswag.py +33 -0
- auditkit/scenarios/humaneval.py +32 -0
- auditkit/scenarios/mmlu.py +34 -0
- auditkit/scenarios/truthfulqa.py +33 -0
- auditkit/score.py +165 -0
- auditkit/scorers.py +117 -0
- auditkit/scoring.py +79 -0
- auditkit/types.py +69 -0
- auditkit-1.0.0.dist-info/METADATA +396 -0
- auditkit-1.0.0.dist-info/RECORD +81 -0
- auditkit-1.0.0.dist-info/WHEEL +4 -0
- auditkit-1.0.0.dist-info/entry_points.txt +2 -0
- auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/runner.py
ADDED
|
@@ -0,0 +1,575 @@
|
|
|
1
|
+
"""The runner: five separately-testable stages that drive one evaluation.
|
|
2
|
+
|
|
3
|
+
``build_requests`` turns samples into requests, ``execute`` sends them to the
|
|
4
|
+
model in one batch and realigns the results, ``annotate`` runs any annotators,
|
|
5
|
+
``score_one`` applies the metrics to a sample's output and records a
|
|
6
|
+
:class:`Prediction`, and ``aggregate`` rolls per-sample scores into per-metric
|
|
7
|
+
:class:`~auditkit.score.Stat`. :meth:`Runner.run` chains them into a
|
|
8
|
+
:class:`RunResult`. Keeping the stages independent is what makes each testable in
|
|
9
|
+
isolation and lets techniques swap one stage without touching the others.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import threading
|
|
16
|
+
import time
|
|
17
|
+
from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeout
|
|
18
|
+
from dataclasses import asdict
|
|
19
|
+
from typing import Any, Optional
|
|
20
|
+
|
|
21
|
+
from .adapter import Adapter
|
|
22
|
+
from .cache import DiskCache
|
|
23
|
+
from .errors import CapabilityError, ExtraNotInstalled, ModelError, ModelTimeout
|
|
24
|
+
from .metric import Metric
|
|
25
|
+
from .metrics.perf import LatencyStats, Throughput
|
|
26
|
+
from .model import Generated, Model, Request, Result_, _prompt_text
|
|
27
|
+
from .report import Prediction, RunResult
|
|
28
|
+
from .runspec import RunConfig, RunSpec
|
|
29
|
+
from .sample import Sample
|
|
30
|
+
from .scenario import Scenario
|
|
31
|
+
from .score import Score, Stat
|
|
32
|
+
from .types import Capability
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
_generate_lock = threading.Lock()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _score_doc(score: Score) -> dict[str, Any]:
|
|
40
|
+
"""Every field on *score* that actually carries information.
|
|
41
|
+
|
|
42
|
+
``Score`` supports rich, judge-grade output (``reason``, ``threshold``,
|
|
43
|
+
``label``, per-score ``metadata``, ...), but most deterministic metrics
|
|
44
|
+
never set most of it -- serializing every field unconditionally would
|
|
45
|
+
bury the useful cases (a judge's verdict reasoning) under a wall of
|
|
46
|
+
``None``s and defaults for the common case (``ExactMatch`` only ever
|
|
47
|
+
sets ``name``/``value``). Keep only fields that aren't ``None`` or an
|
|
48
|
+
empty container, plus the derived ``passed`` (True/False/None against
|
|
49
|
+
``threshold``), which isn't itself a dataclass field so ``asdict``
|
|
50
|
+
wouldn't pick it up.
|
|
51
|
+
"""
|
|
52
|
+
doc = {k: v for k, v in asdict(score).items() if v not in (None, {}, [])}
|
|
53
|
+
if score.passed is not None:
|
|
54
|
+
doc["passed"] = score.passed
|
|
55
|
+
return doc
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Runner:
|
|
59
|
+
"""Drives a :class:`RunSpec` to a :class:`RunResult` in five stages."""
|
|
60
|
+
|
|
61
|
+
def build_requests(
|
|
62
|
+
self, scenario: Scenario, adapter: Adapter, config: RunConfig
|
|
63
|
+
) -> list[tuple[Sample, list[Request]]]:
|
|
64
|
+
samples = list(scenario.samples())
|
|
65
|
+
if config.limit is not None:
|
|
66
|
+
samples = samples[: config.limit]
|
|
67
|
+
split_cfg = config.split
|
|
68
|
+
if split_cfg is not None:
|
|
69
|
+
train, val, test = self._split_samples(samples, split_cfg)
|
|
70
|
+
samples = test or val or train
|
|
71
|
+
if hasattr(adapter, 'pool') and train:
|
|
72
|
+
adapter.pool = train
|
|
73
|
+
return [(s, adapter.adapt(s, config)) for s in samples]
|
|
74
|
+
|
|
75
|
+
def _split_samples(self, samples, split_cfg):
|
|
76
|
+
n = len(samples)
|
|
77
|
+
t = split_cfg
|
|
78
|
+
if t.strategy == "sequential":
|
|
79
|
+
train_end = int(n * t.train_ratio)
|
|
80
|
+
val_end = train_end + int(n * t.val_ratio)
|
|
81
|
+
return samples[:train_end], samples[train_end:val_end], samples[val_end:]
|
|
82
|
+
elif t.strategy == "random":
|
|
83
|
+
import random
|
|
84
|
+
rng = random.Random(t.seed)
|
|
85
|
+
idx = list(range(n))
|
|
86
|
+
rng.shuffle(idx)
|
|
87
|
+
shuffled = [samples[i] for i in idx]
|
|
88
|
+
train_end = int(n * t.train_ratio)
|
|
89
|
+
val_end = train_end + int(n * t.val_ratio)
|
|
90
|
+
return shuffled[:train_end], shuffled[train_end:val_end], shuffled[val_end:]
|
|
91
|
+
elif t.strategy == "stratified":
|
|
92
|
+
from collections import defaultdict
|
|
93
|
+
groups = defaultdict(list)
|
|
94
|
+
for i, s in enumerate(samples):
|
|
95
|
+
key = str(getattr(s, 'kind', None) or str(getattr(s, 'id', None) or i))
|
|
96
|
+
groups[key].append(s)
|
|
97
|
+
train, val, test = [], [], []
|
|
98
|
+
for g in groups.values():
|
|
99
|
+
gn = len(g)
|
|
100
|
+
te = int(gn * t.train_ratio)
|
|
101
|
+
ve = te + int(gn * t.val_ratio)
|
|
102
|
+
train.extend(g[:te])
|
|
103
|
+
val.extend(g[te:ve])
|
|
104
|
+
test.extend(g[ve:])
|
|
105
|
+
return train, val, test
|
|
106
|
+
else:
|
|
107
|
+
return samples, [], []
|
|
108
|
+
|
|
109
|
+
def execute(
|
|
110
|
+
self, model: Model, batch: list[tuple[Sample, list[Request]]],
|
|
111
|
+
concurrency: int = 1, max_retries: int = 3, retry_delay: float = 1.0,
|
|
112
|
+
timeout: float | None = None,
|
|
113
|
+
latency: LatencyStats | None = None, throughput: Throughput | None = None,
|
|
114
|
+
) -> list[tuple[Sample, list[Result_]]]:
|
|
115
|
+
flat, counts = self._flatten(batch)
|
|
116
|
+
if not flat:
|
|
117
|
+
return [(s, []) for s, _ in batch]
|
|
118
|
+
|
|
119
|
+
# Fan out across threads ONLY when asked for it AND the model declares it
|
|
120
|
+
# is safe to call concurrently. A single local model (HFGenModel/vLLM) is
|
|
121
|
+
# not thread-safe, so calling generate() from many threads at once races
|
|
122
|
+
# the same model — corrupt output or a crash. When we don't parallelize we
|
|
123
|
+
# make a single batched generate() call, which also lets batched backends
|
|
124
|
+
# process every request at once instead of fragmenting the batch.
|
|
125
|
+
parallel = concurrency > 1 and getattr(model, "threadsafe", False)
|
|
126
|
+
if not parallel:
|
|
127
|
+
results = self._retry_generate(model, flat, max_retries, retry_delay, timeout, latency, throughput)
|
|
128
|
+
else:
|
|
129
|
+
# Never make more chunks than there are requests, so we never call
|
|
130
|
+
# generate([]) on empty chunks (wasteful, and some backends choke).
|
|
131
|
+
n_chunks = min(concurrency, len(flat))
|
|
132
|
+
chunks = self._chunk(flat, n_chunks)
|
|
133
|
+
with ThreadPoolExecutor(max_workers=n_chunks) as pool:
|
|
134
|
+
chunk_results = list(pool.map(
|
|
135
|
+
lambda c: self._retry_generate(model, c, max_retries, retry_delay, timeout, latency, throughput),
|
|
136
|
+
chunks,
|
|
137
|
+
))
|
|
138
|
+
results = [r for cr in chunk_results for r in cr]
|
|
139
|
+
|
|
140
|
+
out: list[tuple[Sample, list[Result_]]] = []
|
|
141
|
+
cursor = 0
|
|
142
|
+
for (sample, _requests), n in zip(batch, counts):
|
|
143
|
+
out.append((sample, results[cursor:cursor + n]))
|
|
144
|
+
cursor += n
|
|
145
|
+
return out
|
|
146
|
+
|
|
147
|
+
def _retry_generate(self, model, requests, max_retries, retry_delay, timeout=None,
|
|
148
|
+
latency: LatencyStats | None = None, throughput: Throughput | None = None):
|
|
149
|
+
"""Calls ``model.generate(requests)`` with retry/backoff.
|
|
150
|
+
|
|
151
|
+
When *latency*/*throughput* are given, records one sample per actual
|
|
152
|
+
model-call attempt — success or failure alike, since a failing call
|
|
153
|
+
still occupied real model/network time — but never the artificial
|
|
154
|
+
``time.sleep()`` backoff between retries, which is our own throttling,
|
|
155
|
+
not the model's latency. This is a *call*-level measurement, not a
|
|
156
|
+
true per-request one: batched backends (the default — see
|
|
157
|
+
``execute()``'s comment) answer many requests in a single call, so
|
|
158
|
+
there is no per-request timestamp to read. One sample per call,
|
|
159
|
+
tagged with how many requests it served, is the honest granularity
|
|
160
|
+
actually available without breaking batching; it still yields a real
|
|
161
|
+
distribution across a run's several chunks/attempts and a real
|
|
162
|
+
requests/sec figure.
|
|
163
|
+
"""
|
|
164
|
+
def _record(elapsed_ms: float, served: bool) -> None:
|
|
165
|
+
# `served` distinguishes a successful call (which actually
|
|
166
|
+
# returned results for every one of `requests`) from a failed
|
|
167
|
+
# attempt: a failing call still occupied real model/network
|
|
168
|
+
# time, so it's a legitimate latency sample -- but it served
|
|
169
|
+
# zero requests, not len(requests), so it must NOT add to
|
|
170
|
+
# Throughput's request count. Previously every attempt credited
|
|
171
|
+
# the full batch size regardless of success, so a run that hit
|
|
172
|
+
# even one retry over-counted total_requests (and therefore
|
|
173
|
+
# inflated requests/sec) by the number of failed attempts.
|
|
174
|
+
if latency is None and throughput is None:
|
|
175
|
+
return
|
|
176
|
+
with _generate_lock:
|
|
177
|
+
if latency is not None:
|
|
178
|
+
latency.record(elapsed_ms)
|
|
179
|
+
if throughput is not None:
|
|
180
|
+
throughput.record(elapsed_ms, len(requests) if served else 0)
|
|
181
|
+
|
|
182
|
+
last_exc = None
|
|
183
|
+
for attempt in range(max_retries + 1):
|
|
184
|
+
start = time.perf_counter()
|
|
185
|
+
try:
|
|
186
|
+
if timeout is not None:
|
|
187
|
+
with ThreadPoolExecutor(max_workers=1) as pool:
|
|
188
|
+
future = pool.submit(model.generate, requests)
|
|
189
|
+
result = future.result(timeout=timeout)
|
|
190
|
+
else:
|
|
191
|
+
result = model.generate(requests)
|
|
192
|
+
except FuturesTimeout:
|
|
193
|
+
_record((time.perf_counter() - start) * 1000.0, served=False)
|
|
194
|
+
raise ModelTimeout(f"generate timed out after {timeout}s")
|
|
195
|
+
except Exception as e:
|
|
196
|
+
_record((time.perf_counter() - start) * 1000.0, served=False)
|
|
197
|
+
last_exc = e
|
|
198
|
+
if attempt < max_retries:
|
|
199
|
+
time.sleep(retry_delay * (2 ** attempt))
|
|
200
|
+
logger.warning("Retry %d/%d: %s", attempt + 1, max_retries, e)
|
|
201
|
+
continue
|
|
202
|
+
_record((time.perf_counter() - start) * 1000.0, served=True)
|
|
203
|
+
return result
|
|
204
|
+
raise ModelError(f"generate failed after {max_retries} retries") from last_exc
|
|
205
|
+
|
|
206
|
+
def _flatten(self, batch):
|
|
207
|
+
flat = []
|
|
208
|
+
counts = []
|
|
209
|
+
for _sample, reqs in batch:
|
|
210
|
+
flat.extend(reqs)
|
|
211
|
+
counts.append(len(reqs))
|
|
212
|
+
return flat, counts
|
|
213
|
+
|
|
214
|
+
def _chunk(self, items, n_chunks):
|
|
215
|
+
k, m = divmod(len(items), n_chunks)
|
|
216
|
+
return [items[i * k + min(i, m):(i + 1) * k + min(i + 1, m)] for i in range(n_chunks)]
|
|
217
|
+
|
|
218
|
+
def annotate(
|
|
219
|
+
self, annotators: list[Any], sample: Sample, results: list[Result_]
|
|
220
|
+
) -> dict[str, Any]:
|
|
221
|
+
context: dict[str, Any] = {}
|
|
222
|
+
for annotator in annotators:
|
|
223
|
+
context[getattr(annotator, "name", type(annotator).__name__)] = annotator.annotate(
|
|
224
|
+
sample, results
|
|
225
|
+
)
|
|
226
|
+
return context
|
|
227
|
+
|
|
228
|
+
def score_one(
|
|
229
|
+
self,
|
|
230
|
+
metrics: list[Metric],
|
|
231
|
+
sample: Sample,
|
|
232
|
+
output: str,
|
|
233
|
+
context: Any,
|
|
234
|
+
run_id: str,
|
|
235
|
+
errors: list | None = None,
|
|
236
|
+
prompt: Optional[str] = None,
|
|
237
|
+
extracted_by: str | None = None,
|
|
238
|
+
) -> tuple[list[Score], Prediction]:
|
|
239
|
+
# extracted_by names an annotator whose context["extracted"] value
|
|
240
|
+
# should be scored instead of the raw output (e.g. RegexAnnotator
|
|
241
|
+
# pulling "42" out of "...FINAL ANSWER: 42"). Missing/misspelled
|
|
242
|
+
# name degrades to raw output with a warning, not a crash -- a typo
|
|
243
|
+
# here shouldn't take down a whole run.
|
|
244
|
+
scoring_output = output
|
|
245
|
+
if extracted_by:
|
|
246
|
+
entry = context.get(extracted_by) if isinstance(context, dict) else None
|
|
247
|
+
if isinstance(entry, dict) and "extracted" in entry:
|
|
248
|
+
scoring_output = entry["extracted"]
|
|
249
|
+
else:
|
|
250
|
+
logger.warning("extracted_by=%r has no 'extracted' entry in context", extracted_by)
|
|
251
|
+
scores: list[Score] = []
|
|
252
|
+
for metric in metrics:
|
|
253
|
+
if not metric.applicable(sample):
|
|
254
|
+
continue
|
|
255
|
+
try:
|
|
256
|
+
produced = metric.score(sample, scoring_output, context)
|
|
257
|
+
except ExtraNotInstalled:
|
|
258
|
+
# A missing extra is a run-wide configuration problem, not a
|
|
259
|
+
# per-sample scoring failure -- every remaining sample would
|
|
260
|
+
# hit the exact same error. Swallowing it into a fake 0.0
|
|
261
|
+
# score (the old behavior) is indistinguishable in
|
|
262
|
+
# `headline` from "the model's output genuinely scored
|
|
263
|
+
# zero," silently corrupting the aggregate. Fail loudly and
|
|
264
|
+
# immediately instead, so the missing dependency is obvious.
|
|
265
|
+
raise
|
|
266
|
+
except Exception as e:
|
|
267
|
+
# Don't fake a 0.0 Score here -- that's indistinguishable from
|
|
268
|
+
# "the model's output genuinely scored zero" in every mean/
|
|
269
|
+
# aggregate downstream (the exact corruption ExtraNotInstalled
|
|
270
|
+
# above is deliberately not swallowed into either). A metric
|
|
271
|
+
# crashing on this sample is a computation failure, not a
|
|
272
|
+
# data point -- skip it so this sample's absence from
|
|
273
|
+
# `metric.name`'s count is the honest signal, not a fabricated
|
|
274
|
+
# score. `errors` still records exactly what happened.
|
|
275
|
+
logger.error("Metric %s failed on sample %s: %s", metric.name, sample.id, e)
|
|
276
|
+
if errors is not None:
|
|
277
|
+
errors.append({"sample_id": sample.id, "metric": metric.name, "error": str(e)})
|
|
278
|
+
continue
|
|
279
|
+
produced_list = produced if isinstance(produced, list) else [produced]
|
|
280
|
+
for s in produced_list:
|
|
281
|
+
# Authoritative, not a fallback: metric.direction is required
|
|
282
|
+
# (Metric.__init_subclass__ enforces it), so every Score a
|
|
283
|
+
# metric produces carries ITS metric's declared direction,
|
|
284
|
+
# regardless of what value the metric's own score() happened
|
|
285
|
+
# to set (or forgot to). This is what makes direction
|
|
286
|
+
# actually reliable for RunComparison/compare_models grading.
|
|
287
|
+
s.direction = metric.direction
|
|
288
|
+
scores.extend(produced_list)
|
|
289
|
+
|
|
290
|
+
primary = scores[0] if scores else None
|
|
291
|
+
correct: Optional[bool]
|
|
292
|
+
if primary is None:
|
|
293
|
+
correct = None
|
|
294
|
+
elif primary.passed is not None:
|
|
295
|
+
correct = primary.passed
|
|
296
|
+
else:
|
|
297
|
+
correct = primary.value == 1.0
|
|
298
|
+
|
|
299
|
+
prediction = Prediction(
|
|
300
|
+
run_id=run_id,
|
|
301
|
+
task=sample.task or sample.kind.value,
|
|
302
|
+
sample_id=sample.id or "",
|
|
303
|
+
# The adapter-built prompt actually sent to the model when known
|
|
304
|
+
# (a RAGAdapter's injected context, a ChatAdapter's system-prompt
|
|
305
|
+
# wrapping, ...) -- falls back to the raw sample input only when
|
|
306
|
+
# no request was built (e.g. an errored-out sample).
|
|
307
|
+
prompt=prompt if prompt is not None else sample.input_text,
|
|
308
|
+
raw_output=output,
|
|
309
|
+
parsed_answer=scoring_output,
|
|
310
|
+
expected=sample.target,
|
|
311
|
+
correct=correct,
|
|
312
|
+
score=primary.value if primary else None,
|
|
313
|
+
context=context,
|
|
314
|
+
metadata={"scores": [_score_doc(s) for s in scores]},
|
|
315
|
+
)
|
|
316
|
+
return scores, prediction
|
|
317
|
+
|
|
318
|
+
def aggregate(self, scores: list[Score]) -> dict[str, Stat]:
|
|
319
|
+
stats: dict[str, Stat] = {}
|
|
320
|
+
for score in scores:
|
|
321
|
+
stats.setdefault(score.name, Stat(score.name)).add(score.value)
|
|
322
|
+
return stats
|
|
323
|
+
|
|
324
|
+
def run(self, spec: RunSpec, *, verbose: bool = False) -> RunResult:
|
|
325
|
+
fingerprint = spec.fingerprint()
|
|
326
|
+
config = spec.config
|
|
327
|
+
|
|
328
|
+
cache = DiskCache()
|
|
329
|
+
cached = cache.get(fingerprint)
|
|
330
|
+
if cached is not None:
|
|
331
|
+
return cached
|
|
332
|
+
|
|
333
|
+
run_id = spec.run_name or fingerprint
|
|
334
|
+
|
|
335
|
+
batch = self.build_requests(spec.scenario, spec.adapter, config)
|
|
336
|
+
logger.info("Starting run with %d samples", len(batch))
|
|
337
|
+
|
|
338
|
+
has_loglikelihood = any(
|
|
339
|
+
r.request_type == "loglikelihood"
|
|
340
|
+
for _, reqs in batch
|
|
341
|
+
for r in reqs
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
all_scores: list[Score] = []
|
|
345
|
+
predictions: list[Prediction] = []
|
|
346
|
+
errors: list[dict] = []
|
|
347
|
+
failed_count: int = 0
|
|
348
|
+
latency = LatencyStats()
|
|
349
|
+
throughput = Throughput()
|
|
350
|
+
# Real, provider-reported token counts -- only ever non-zero when a
|
|
351
|
+
# backend actually populates Result_.usage (the hosted-API backends;
|
|
352
|
+
# see openai.py/anthropic.py/groq_gen.py/api_gen.py/litellm_gen.py).
|
|
353
|
+
# Local backends and callables leave this at zero -- never estimated
|
|
354
|
+
# by counting characters or guessing a tokenizer.
|
|
355
|
+
token_usage = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
|
|
356
|
+
|
|
357
|
+
def _accumulate_usage(results: list[Result_]) -> None:
|
|
358
|
+
for r in results:
|
|
359
|
+
for key in token_usage:
|
|
360
|
+
token_usage[key] += r.usage.get(key, 0) or 0
|
|
361
|
+
|
|
362
|
+
if getattr(spec.model, "reads_actual_output", False):
|
|
363
|
+
# Score pre-generated answers (Sample.actual_output) — no model call.
|
|
364
|
+
for index, (sample, _reqs) in enumerate(batch):
|
|
365
|
+
if sample.id is None:
|
|
366
|
+
sample.id = str(index)
|
|
367
|
+
output = sample.actual_output or ""
|
|
368
|
+
context = self.annotate(
|
|
369
|
+
spec.annotators, sample, [Result_(completions=[Generated(text=output)])]
|
|
370
|
+
)
|
|
371
|
+
scores, prediction = self.score_one(
|
|
372
|
+
spec.metrics, sample, output, context, run_id, errors,
|
|
373
|
+
extracted_by=spec.extracted_by,
|
|
374
|
+
)
|
|
375
|
+
all_scores.extend(scores)
|
|
376
|
+
predictions.append(prediction)
|
|
377
|
+
elif has_loglikelihood:
|
|
378
|
+
if not spec.model.supports(Capability.LOGLIKELIHOOD):
|
|
379
|
+
raise CapabilityError(
|
|
380
|
+
f"{spec.model.name} does not declare LOGLIKELIHOOD"
|
|
381
|
+
)
|
|
382
|
+
flat = [r for _, reqs in batch for r in reqs]
|
|
383
|
+
logger.info("Executing %d requests", len(flat))
|
|
384
|
+
last_exc = None
|
|
385
|
+
for attempt in range(config.max_retries + 1):
|
|
386
|
+
start = time.perf_counter()
|
|
387
|
+
try:
|
|
388
|
+
lls = spec.model.loglikelihood(flat)
|
|
389
|
+
except Exception as e:
|
|
390
|
+
# A failing attempt occupied real time (legitimate
|
|
391
|
+
# latency sample) but served zero requests -- crediting
|
|
392
|
+
# it with len(flat) would double-count against the
|
|
393
|
+
# eventual successful retry (see the identical fix in
|
|
394
|
+
# _retry_generate() above).
|
|
395
|
+
elapsed_ms = (time.perf_counter() - start) * 1000.0
|
|
396
|
+
latency.record(elapsed_ms)
|
|
397
|
+
throughput.record(elapsed_ms, 0)
|
|
398
|
+
last_exc = e
|
|
399
|
+
if attempt < config.max_retries:
|
|
400
|
+
time.sleep(config.retry_delay * (2 ** attempt))
|
|
401
|
+
logger.warning("Retry %d/%d for loglikelihood: %s", attempt + 1, config.max_retries, e)
|
|
402
|
+
continue
|
|
403
|
+
elapsed_ms = (time.perf_counter() - start) * 1000.0
|
|
404
|
+
latency.record(elapsed_ms)
|
|
405
|
+
throughput.record(elapsed_ms, len(flat))
|
|
406
|
+
break
|
|
407
|
+
else:
|
|
408
|
+
raise ModelError(f"loglikelihood failed after {config.max_retries} retries") from last_exc
|
|
409
|
+
logger.info("Got %d responses", len(lls))
|
|
410
|
+
cursor = 0
|
|
411
|
+
for index, (sample, reqs) in enumerate(batch):
|
|
412
|
+
try:
|
|
413
|
+
if sample.id is None:
|
|
414
|
+
sample.id = str(index)
|
|
415
|
+
n = len(reqs)
|
|
416
|
+
sample_lls = lls[cursor: cursor + n]
|
|
417
|
+
cursor += n
|
|
418
|
+
choice_probs = [ll.logprob for ll in sample_lls]
|
|
419
|
+
argmax_idx = max(
|
|
420
|
+
range(len(choice_probs)), key=lambda i: choice_probs[i]
|
|
421
|
+
)
|
|
422
|
+
output = sample.choices[argmax_idx] if sample.choices else ""
|
|
423
|
+
context = self.annotate(
|
|
424
|
+
spec.annotators, sample, [Result_(completions=[Generated(text=output)])]
|
|
425
|
+
)
|
|
426
|
+
context["choice_likelihoods"] = choice_probs
|
|
427
|
+
prompt = _prompt_text(reqs[0].prompt) if reqs else None
|
|
428
|
+
# extracted_by deliberately NOT threaded through here:
|
|
429
|
+
# `output` on this path is already sample.choices[argmax_idx]
|
|
430
|
+
# -- the exact, correct, final choice text picked by
|
|
431
|
+
# comparing logprobs, not free-form generated text. There
|
|
432
|
+
# is nothing to clean up, and applying extraction here is
|
|
433
|
+
# actively dangerous for choice-index metrics (Acc/AccNorm
|
|
434
|
+
# interpret a bare digit-string as a CHOICE INDEX, not "a
|
|
435
|
+
# number that happened to appear in the choice text") --
|
|
436
|
+
# a correct pick can silently get scored as a different,
|
|
437
|
+
# wrong choice, or as no match at all. Annotators still
|
|
438
|
+
# run on this branch (context is populated normally, for
|
|
439
|
+
# logging/inspection), only the extraction hookup into
|
|
440
|
+
# scoring is structurally disabled here.
|
|
441
|
+
scores, prediction = self.score_one(
|
|
442
|
+
spec.metrics, sample, output, context, run_id, errors, prompt=prompt,
|
|
443
|
+
)
|
|
444
|
+
all_scores.extend(scores)
|
|
445
|
+
predictions.append(prediction)
|
|
446
|
+
except Exception as e:
|
|
447
|
+
logger.error("Sample %s failed: %s", sample.id, e)
|
|
448
|
+
failed_count += 1
|
|
449
|
+
predictions.append(Prediction(
|
|
450
|
+
run_id=run_id,
|
|
451
|
+
task=getattr(sample, 'task', '') or '',
|
|
452
|
+
sample_id=sample.id or str(index),
|
|
453
|
+
prompt=str(sample.input),
|
|
454
|
+
raw_output="",
|
|
455
|
+
parsed_answer=None,
|
|
456
|
+
expected=None,
|
|
457
|
+
correct=False,
|
|
458
|
+
score=0.0,
|
|
459
|
+
))
|
|
460
|
+
errors.append({"sample_id": sample.id or str(index), "error": str(e)})
|
|
461
|
+
continue
|
|
462
|
+
logger.debug("Scored %d/%d samples", index + 1, len(batch))
|
|
463
|
+
if verbose and (index + 1) % 10 == 0:
|
|
464
|
+
print(".", end="", flush=True)
|
|
465
|
+
else:
|
|
466
|
+
requests = [r for _, reqs in batch for r in reqs]
|
|
467
|
+
logger.info("Executing %d requests", len(requests))
|
|
468
|
+
executed = self.execute(
|
|
469
|
+
spec.model, batch, concurrency=config.concurrency, max_retries=config.max_retries,
|
|
470
|
+
retry_delay=config.retry_delay, timeout=config.timeout,
|
|
471
|
+
latency=latency, throughput=throughput,
|
|
472
|
+
)
|
|
473
|
+
logger.info("Got %d responses", len(executed))
|
|
474
|
+
for index, (sample, results) in enumerate(executed):
|
|
475
|
+
try:
|
|
476
|
+
if sample.id is None:
|
|
477
|
+
sample.id = str(index)
|
|
478
|
+
output = results[0].text if results else ""
|
|
479
|
+
_accumulate_usage(results)
|
|
480
|
+
context = self.annotate(spec.annotators, sample, results)
|
|
481
|
+
# execute() preserves batch's order/length, so batch[index]
|
|
482
|
+
# is the (sample, requests) pair this result came from.
|
|
483
|
+
reqs = batch[index][1]
|
|
484
|
+
prompt = _prompt_text(reqs[0].prompt) if reqs else None
|
|
485
|
+
scores, prediction = self.score_one(
|
|
486
|
+
spec.metrics, sample, output, context, run_id, errors, prompt=prompt,
|
|
487
|
+
extracted_by=spec.extracted_by,
|
|
488
|
+
)
|
|
489
|
+
all_scores.extend(scores)
|
|
490
|
+
predictions.append(prediction)
|
|
491
|
+
except Exception as e:
|
|
492
|
+
logger.error("Sample %s failed: %s", sample.id, e)
|
|
493
|
+
failed_count += 1
|
|
494
|
+
predictions.append(Prediction(
|
|
495
|
+
run_id=run_id,
|
|
496
|
+
task=getattr(sample, 'task', '') or '',
|
|
497
|
+
sample_id=sample.id or str(index),
|
|
498
|
+
prompt=str(sample.input),
|
|
499
|
+
raw_output="",
|
|
500
|
+
parsed_answer=None,
|
|
501
|
+
expected=None,
|
|
502
|
+
correct=False,
|
|
503
|
+
score=0.0,
|
|
504
|
+
))
|
|
505
|
+
errors.append({"sample_id": sample.id or str(index), "error": str(e)})
|
|
506
|
+
continue
|
|
507
|
+
logger.debug("Scored %d/%d samples", index + 1, len(executed))
|
|
508
|
+
if verbose and (index + 1) % 10 == 0:
|
|
509
|
+
print(".", end="", flush=True)
|
|
510
|
+
|
|
511
|
+
stats = self.aggregate(all_scores)
|
|
512
|
+
logger.info("Run complete – %d metrics computed", len(stats))
|
|
513
|
+
if verbose:
|
|
514
|
+
print(f" done – {len(stats)} metrics")
|
|
515
|
+
headline = {name: stat.mean for name, stat in stats.items()}
|
|
516
|
+
# All three of perf/model_size/token_usage are populated unless the
|
|
517
|
+
# caller opts out via RunConfig.track_performance=False (default
|
|
518
|
+
# True -- model_info() below can be a real, if best-effort,
|
|
519
|
+
# introspection cost for local backends, so callers who don't want
|
|
520
|
+
# that can disable it). This is a *reporting* gate, not a timing
|
|
521
|
+
# one -- latency/throughput/usage were already measured for free
|
|
522
|
+
# above as part of running the requests; we just don't surface them
|
|
523
|
+
# when disabled.
|
|
524
|
+
if config.track_performance:
|
|
525
|
+
# Empty on the `reads_actual_output` path (no model call was made
|
|
526
|
+
# at all -- nothing to time) and on the loglikelihood/generative
|
|
527
|
+
# paths when there were zero requests; LatencyStats/Throughput
|
|
528
|
+
# .stats() both degrade to zeroed dicts rather than raising.
|
|
529
|
+
perf = {"latency_ms": latency.stats(), "throughput": throughput.stats()}
|
|
530
|
+
# Token throughput: derived from two numbers that already exist
|
|
531
|
+
# separately (real elapsed call time, real token counts from the
|
|
532
|
+
# backend's own API response) but were never combined into a
|
|
533
|
+
# tokens/sec figure. total_time_ms is the same denominator
|
|
534
|
+
# throughput["rps"] already uses -- real wall-clock time spent in
|
|
535
|
+
# model calls, not the whole run (annotation/scoring time excluded).
|
|
536
|
+
total_time_s = perf["throughput"].get("total_time_ms", 0.0) / 1000.0
|
|
537
|
+
if total_time_s > 0:
|
|
538
|
+
perf["throughput"]["output_tokens_per_sec"] = token_usage["completion_tokens"] / total_time_s
|
|
539
|
+
perf["throughput"]["total_tokens_per_sec"] = token_usage["total_tokens"] / total_time_s
|
|
540
|
+
else:
|
|
541
|
+
perf["throughput"]["output_tokens_per_sec"] = 0.0
|
|
542
|
+
perf["throughput"]["total_tokens_per_sec"] = 0.0
|
|
543
|
+
# Never caller-supplied: introspected for local backends (real
|
|
544
|
+
# parameter count/sparsity/size), identity-only for hosted-API
|
|
545
|
+
# backends (nothing to measure -- see Model.model_info()).
|
|
546
|
+
# Guarded because a backend's own introspection (e.g. loading a
|
|
547
|
+
# model just to inspect it) is best-effort and shouldn't take
|
|
548
|
+
# down an otherwise successful run. Skipped entirely (not just
|
|
549
|
+
# hidden) when tracking is off, since this is the one field here
|
|
550
|
+
# that can cost real time to compute, not just to report.
|
|
551
|
+
try:
|
|
552
|
+
model_size = spec.model.model_info() if spec.model else {}
|
|
553
|
+
except Exception as e:
|
|
554
|
+
logger.warning("model_info() failed for %s: %s", getattr(spec.model, "name", spec.model), e)
|
|
555
|
+
model_size = {}
|
|
556
|
+
else:
|
|
557
|
+
perf = None
|
|
558
|
+
model_size = None
|
|
559
|
+
token_usage = None
|
|
560
|
+
result = RunResult(
|
|
561
|
+
run_id=run_id,
|
|
562
|
+
fingerprint=fingerprint,
|
|
563
|
+
stats=stats,
|
|
564
|
+
predictions=predictions,
|
|
565
|
+
headline=headline,
|
|
566
|
+
config=spec.config,
|
|
567
|
+
model_spec=str(spec.model) if spec.model else None,
|
|
568
|
+
errors=errors,
|
|
569
|
+
failed_count=failed_count,
|
|
570
|
+
model_size=model_size,
|
|
571
|
+
token_usage=token_usage,
|
|
572
|
+
perf=perf,
|
|
573
|
+
)
|
|
574
|
+
cache.set(fingerprint, result)
|
|
575
|
+
return result
|