errorbars 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
errorbars/__init__.py ADDED
@@ -0,0 +1,28 @@
1
+ """errorbars: statistical rigor for LLM eval results."""
2
+
3
+ from errorbars.stats import (
4
+ ClusterDiagnostics,
5
+ MeanEstimate,
6
+ bootstrap_ci,
7
+ cluster_robust_se,
8
+ design_effect,
9
+ intraclass_correlation,
10
+ mean_ci_clt,
11
+ wilson_ci,
12
+ within_between_variance,
13
+ )
14
+
15
+ __version__ = "0.1.2"
16
+
17
+ __all__ = [
18
+ "__version__",
19
+ "MeanEstimate",
20
+ "ClusterDiagnostics",
21
+ "mean_ci_clt",
22
+ "wilson_ci",
23
+ "bootstrap_ci",
24
+ "cluster_robust_se",
25
+ "intraclass_correlation",
26
+ "design_effect",
27
+ "within_between_variance",
28
+ ]
@@ -0,0 +1 @@
1
+ """Adapters that turn other eval tools' logs into errorbars' long format."""
@@ -0,0 +1,83 @@
1
+ """Adapter for Inspect AI eval logs.
2
+
3
+ Verified against a real log produced by **inspect-ai 0.3.268**, running a
4
+ 5-sample task through the built-in ``mockllm/model`` provider (no network
5
+ model calls). See ``tests/fixtures/inspect_tiny_qa.eval`` for the exact,
6
+ unedited log this adapter is built and tested against.
7
+
8
+ This reads logs with Inspect's own ``inspect_ai.log.read_eval_log`` and
9
+ converts score values with its own ``inspect_ai.scorer.value_to_float``
10
+ (the same conversion Inspect uses for its own metrics: correct/incorrect/
11
+ partial/no-answer string codes -> 1.0/0.0/0.5/0.0, numbers and bools pass
12
+ through) rather than hand-parsing the ``.eval`` file, which is a versioned
13
+ binary/zip format not meant to be read directly. Requires the ``inspect``
14
+ extra: ``pip install "errorbars[inspect]"``.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from pathlib import Path
20
+
21
+ from errorbars.io import EvalData
22
+
23
+ __all__ = ["load_inspect_log"]
24
+
25
+
26
+ def load_inspect_log(path: str | Path, scorer: str | None = None) -> EvalData:
27
+ """Load per-sample scores from an Inspect AI ``.eval`` or ``.json`` log.
28
+
29
+ Args:
30
+ path: path to the log file (as written by ``inspect eval``).
31
+ scorer: which scorer's score to use, for tasks with more than one
32
+ scorer. Defaults to the only scorer, and raises if a sample has
33
+ more than one and none was specified.
34
+
35
+ Repeated ``--epochs`` sampling of the same input is captured in the
36
+ ``sample`` column (Inspect's per-sample ``epoch`` number), so
37
+ ``errorbars summarize`` can decompose within/between-question variance.
38
+ """
39
+ try:
40
+ from inspect_ai.log import read_eval_log
41
+ from inspect_ai.scorer import value_to_float
42
+ except ImportError as exc: # pragma: no cover - exercised only without the extra
43
+ raise ImportError(
44
+ "load_inspect_log requires inspect-ai (the 'inspect' extra): pip install inspect-ai"
45
+ ) from exc
46
+
47
+ log = read_eval_log(str(path))
48
+ if not log.samples:
49
+ raise ValueError(f"{path}: log has no samples (status={log.status!r})")
50
+
51
+ model = str(log.eval.model)
52
+ to_float = value_to_float()
53
+
54
+ question_id: list[str] = []
55
+ model_col: list[str] = []
56
+ score: list[float] = []
57
+ sample_col: list[str] = []
58
+
59
+ for s in log.samples:
60
+ if not s.scores:
61
+ continue
62
+ if scorer is not None:
63
+ if scorer not in s.scores:
64
+ raise ValueError(
65
+ f"{path}: sample {s.id!r} has no scorer {scorer!r} (has: {list(s.scores)})"
66
+ )
67
+ key = scorer
68
+ elif len(s.scores) == 1:
69
+ key = next(iter(s.scores))
70
+ else:
71
+ raise ValueError(
72
+ f"{path}: sample {s.id!r} has multiple scorers {list(s.scores)}; "
73
+ "pass `scorer=...` to pick one"
74
+ )
75
+ question_id.append(str(s.id))
76
+ model_col.append(model)
77
+ score.append(float(to_float(s.scores[key].value)))
78
+ sample_col.append(str(s.epoch))
79
+
80
+ if not question_id:
81
+ raise ValueError(f"{path}: no scored samples found")
82
+
83
+ return EvalData(question_id=question_id, model=model_col, score=score, sample=sample_col)
@@ -0,0 +1,103 @@
1
+ """Adapter for lm-evaluation-harness `--log_samples` output.
2
+
3
+ Verified against real output from lm-eval **0.4.13** (the latest stable
4
+ release as of 2026-09-24; there is also a 0.5.0.dev1 prerelease with a
5
+ different CLI we did not target), generated with:
6
+
7
+ lm_eval run --model dummy --tasks copa --limit 20 \\
8
+ --log_samples --output_path <dir>
9
+
10
+ See ``tests/fixtures/lm_eval_samples_copa.jsonl`` (single metric: ``acc``)
11
+ and ``tests/fixtures/lm_eval_samples_arc_easy.jsonl`` (two metrics: ``acc``,
12
+ ``acc_norm``) for the exact, unedited records this adapter is built and
13
+ tested against.
14
+
15
+ Each line of a `--log_samples` file is a JSON object with (at least)::
16
+
17
+ {"doc_id": 0, "doc": {...}, "target": "...", "arguments": {...},
18
+ "resps": [...], "filtered_resps": [...], "filter": "none",
19
+ "metrics": ["acc"], "acc": 1.0, "doc_hash": "...", ...}
20
+
21
+ ``metrics`` names which top-level keys on the record hold computed scores
22
+ (a multiple-choice task can report more than one, e.g. ``acc`` and
23
+ ``acc_norm``). The samples file has no model name in it (lm-eval's
24
+ ``--output_path`` subdirectory can be a content hash rather than a model
25
+ name, e.g. for the ``dummy`` model used here, which has no identifying
26
+ ``model_args``), so the caller supplies one.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import json
32
+ import re
33
+ from pathlib import Path
34
+
35
+ from errorbars.io import EvalData
36
+
37
+ __all__ = ["load_lm_eval_samples"]
38
+
39
+ # lm-eval's own convention (loggers/evaluation_tracker.py): samples files are
40
+ # named "samples_<task>_<ISO timestamp>.jsonl". Searched rather than anchored
41
+ # to the start so a path prefix (or a renamed-but-suffixed fixture) still matches.
42
+ _TASK_NAME_RE = re.compile(r"samples_(?P<task>.+?)_\d{4}-\d{2}-\d{2}T")
43
+
44
+
45
+ def _infer_task_name(path: Path) -> str | None:
46
+ m = _TASK_NAME_RE.search(path.name)
47
+ return m.group("task") if m else None
48
+
49
+
50
+ def load_lm_eval_samples(path: str | Path, model: str, metric: str | None = None) -> EvalData:
51
+ """Load an lm-evaluation-harness ``--log_samples`` JSONL file.
52
+
53
+ Args:
54
+ path: path to a ``samples_<task>_<timestamp>.jsonl`` file.
55
+ model: model name to record (lm-eval's samples file doesn't embed
56
+ one usable name for every model type; pass whatever you'd want
57
+ in the ``model`` column).
58
+ metric: which computed metric to use as the score, e.g. ``"acc"``
59
+ or ``"acc_norm"``. Defaults to the first name in each record's
60
+ ``metrics`` list.
61
+
62
+ ``question_id`` is the task name (inferred from the filename, if it
63
+ matches lm-eval's naming convention) plus ``doc_id``, e.g.
64
+ ``"copa-3"``, so files from different tasks can be safely concatenated.
65
+ """
66
+ path = Path(path)
67
+ task_name = _infer_task_name(path)
68
+
69
+ question_id: list[str] = []
70
+ model_col: list[str] = []
71
+ score: list[float] = []
72
+
73
+ with open(path, encoding="utf-8") as f:
74
+ for lineno, raw_line in enumerate(f, start=1):
75
+ line = raw_line.strip()
76
+ if not line:
77
+ continue
78
+ try:
79
+ rec = json.loads(line)
80
+ except json.JSONDecodeError as exc:
81
+ raise ValueError(f"{path}:{lineno}: invalid JSON: {exc}") from exc
82
+
83
+ metrics = rec.get("metrics")
84
+ if not metrics:
85
+ raise ValueError(f"{path}:{lineno}: record has no non-empty 'metrics' list")
86
+ key = metric or metrics[0]
87
+ if key not in rec:
88
+ raise ValueError(
89
+ f"{path}:{lineno}: metric {key!r} not present on this record "
90
+ f"(available: {metrics})"
91
+ )
92
+
93
+ qid = str(rec.get("doc_id", lineno - 1))
94
+ if task_name:
95
+ qid = f"{task_name}-{qid}"
96
+ question_id.append(qid)
97
+ model_col.append(model)
98
+ score.append(float(rec[key]))
99
+
100
+ if not question_id:
101
+ raise ValueError(f"{path}: no records found")
102
+
103
+ return EvalData(question_id=question_id, model=model_col, score=score)
errorbars/cli.py ADDED
@@ -0,0 +1,426 @@
1
+ """Command-line interface: ``errorbars summarize|compare|leaderboard|power``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ from errorbars import __version__
12
+ from errorbars.compare import paired_compare
13
+ from errorbars.io import ColumnMap, EvalData, load_csv, load_jsonl, write_csv
14
+ from errorbars.leaderboard import build_leaderboard
15
+ from errorbars.plot import forest_plot_svg
16
+ from errorbars.power import minimum_detectable_effect, questions_needed
17
+ from errorbars.stats import bootstrap_ci, is_binary, mean_ci_clt, wilson_ci
18
+
19
+ try:
20
+ from rich.console import Console
21
+ from rich.table import Table
22
+
23
+ _HAS_RICH = True
24
+ except ImportError: # pragma: no cover
25
+ _HAS_RICH = False
26
+
27
+
28
+ def _require_models(data: EvalData, *names: str) -> None:
29
+ known = data.models()
30
+ for name in names:
31
+ if name not in known:
32
+ listed = ", ".join(repr(m) for m in known[:10]) + (", ..." if len(known) > 10 else "")
33
+ raise SystemExit(f"error: no model {name!r} in the data (models: {listed})")
34
+
35
+
36
+ def _load(path: str, columns: ColumnMap) -> EvalData:
37
+ p = Path(path)
38
+ if p.suffix.lower() in (".jsonl", ".ndjson"):
39
+ return load_jsonl(p, columns)
40
+ if p.suffix.lower() == ".csv":
41
+ return load_csv(p, columns)
42
+ raise SystemExit(f"error: unrecognized file extension {p.suffix!r} (use .csv or .jsonl)")
43
+
44
+
45
+ def _column_args(sp: argparse.ArgumentParser) -> None:
46
+ sp.add_argument("--question-col", default="question_id")
47
+ sp.add_argument("--model-col", default="model")
48
+ sp.add_argument("--score-col", default="score")
49
+ sp.add_argument("--cluster-col", default="cluster_id")
50
+ sp.add_argument("--sample-col", default="sample")
51
+
52
+
53
+ def _columns_from(args: argparse.Namespace) -> ColumnMap:
54
+ return ColumnMap(
55
+ question_id=args.question_col,
56
+ model=args.model_col,
57
+ score=args.score_col,
58
+ cluster_id=args.cluster_col,
59
+ sample=args.sample_col,
60
+ )
61
+
62
+
63
+ def _print_json(obj: Any) -> None:
64
+ print(json.dumps(obj, indent=2, default=str))
65
+
66
+
67
+ def _console() -> Console:
68
+ if not _HAS_RICH:
69
+ raise SystemExit("error: rich is required for table output; install errorbars[cli] or use --json")
70
+ return Console()
71
+
72
+
73
+ def cmd_summarize(args: argparse.Namespace) -> None:
74
+ data = _load(args.file, _columns_from(args))
75
+ if args.model:
76
+ _require_models(data, args.model)
77
+ sub = data.filter_model(args.model) if args.model else data
78
+ if not sub.score:
79
+ raise SystemExit(f"error: no rows for model {args.model!r}" if args.model else "error: empty data")
80
+
81
+ scores = sub.score
82
+ binary = is_binary(scores)
83
+ if args.ci == "bootstrap":
84
+ est = bootstrap_ci(scores, confidence=args.confidence, seed=args.seed)
85
+ elif args.ci == "wilson" or (args.ci == "auto" and binary and len(scores) < 30):
86
+ if not binary:
87
+ raise SystemExit("error: --ci wilson requires binary (0/1) scores")
88
+ est = wilson_ci(int(round(sum(scores))), len(scores), confidence=args.confidence)
89
+ else:
90
+ est = mean_ci_clt(scores, confidence=args.confidence)
91
+
92
+ result: dict[str, Any] = est.as_dict()
93
+ result["model"] = args.model or "(all)"
94
+ result["is_binary"] = binary
95
+
96
+ cluster_info = None
97
+ if sub.cluster_id and len(set(sub.cluster_id)) > 1:
98
+ from errorbars.stats import (
99
+ cluster_robust_se,
100
+ design_effect,
101
+ intraclass_correlation,
102
+ t_for_confidence,
103
+ )
104
+
105
+ n_clusters = len(set(sub.cluster_id))
106
+ icc = intraclass_correlation(scores, sub.cluster_id)
107
+ avg_size = len(scores) / n_clusters
108
+ deff = design_effect(icc, avg_size)
109
+ se_c = cluster_robust_se(scores, sub.cluster_id)
110
+ # A clustered mean has G - 1 degrees of freedom, not n - 1.
111
+ t_crit = t_for_confidence(est.confidence, n_clusters - 1)
112
+ cluster_info = {
113
+ "clustered_se": se_c,
114
+ "clustered_ci_low": est.mean - t_crit * se_c,
115
+ "clustered_ci_high": est.mean + t_crit * se_c,
116
+ "icc": icc,
117
+ "design_effect": deff,
118
+ "n_clusters": len(set(sub.cluster_id)),
119
+ }
120
+ result["clustered"] = cluster_info
121
+
122
+ within_between = None
123
+ if sub.sample and len(set(sub.sample)) > 1:
124
+ from errorbars.stats import within_between_variance
125
+
126
+ var_w, var_b = within_between_variance(scores, sub.question_id)
127
+ within_between = {"var_within": var_w, "var_between": var_b}
128
+ result["within_between"] = within_between
129
+
130
+ if args.json:
131
+ _print_json(result)
132
+ return
133
+
134
+ console = _console()
135
+ table = Table(title=f"summarize: {result['model']}", show_header=True, header_style="bold cyan")
136
+ table.add_column("metric")
137
+ table.add_column("value", justify="right")
138
+ table.add_row("n", str(est.n))
139
+ table.add_row("mean", f"{est.mean:.4f}")
140
+ table.add_row("SE", f"{est.se:.4f}")
141
+ table.add_row(f"{int(args.confidence * 100)}% CI", f"[{est.ci_low:.4f}, {est.ci_high:.4f}]")
142
+ table.add_row("method", est.method)
143
+ console.print(table)
144
+ if cluster_info:
145
+ ct = Table(title="clustering diagnostics", header_style="bold magenta")
146
+ ct.add_column("metric")
147
+ ct.add_column("value", justify="right")
148
+ ct.add_row("n clusters", str(cluster_info["n_clusters"]))
149
+ ct.add_row("ICC", f"{cluster_info['icc']:.4f}")
150
+ ct.add_row("design effect", f"{cluster_info['design_effect']:.3f}")
151
+ ct.add_row("clustered SE", f"{cluster_info['clustered_se']:.4f}")
152
+ ct.add_row(
153
+ "clustered CI",
154
+ f"[{cluster_info['clustered_ci_low']:.4f}, {cluster_info['clustered_ci_high']:.4f}]",
155
+ )
156
+ console.print(ct)
157
+ if within_between:
158
+ wt = Table(title="within/between-question variance", header_style="bold magenta")
159
+ wt.add_column("component")
160
+ wt.add_column("variance", justify="right")
161
+ wt.add_row("within-question (sampling noise)", f"{within_between['var_within']:.4f}")
162
+ wt.add_row("between-question (item difficulty)", f"{within_between['var_between']:.4f}")
163
+ console.print(wt)
164
+
165
+
166
+ def cmd_compare(args: argparse.Namespace) -> None:
167
+ columns = _columns_from(args)
168
+ data = _load(args.file, columns)
169
+ _require_models(data, args.model_a, args.model_b)
170
+ a = data.filter_model(args.model_a).scores_by_question()
171
+ b = data.filter_model(args.model_b).scores_by_question()
172
+ common = sorted(set(a) & set(b))
173
+ if len(common) < 2:
174
+ raise SystemExit("error: fewer than 2 shared question_ids between the two models")
175
+ sa = [a[q] for q in common]
176
+ sb = [b[q] for q in common]
177
+ clusters = None
178
+ cmap = data.cluster_by_question()
179
+ if cmap and any(cmap.get(q) != q for q in common):
180
+ clusters = [cmap.get(q, q) for q in common]
181
+ comp = paired_compare(sa, sb, clusters=clusters, confidence=args.confidence)
182
+
183
+ if args.json:
184
+ _print_json(comp.as_dict())
185
+ return
186
+
187
+ console = _console()
188
+ table = Table(title=f"compare: {args.model_a} vs {args.model_b}", header_style="bold cyan")
189
+ table.add_column("metric")
190
+ table.add_column("value", justify="right")
191
+ table.add_row("n (shared questions)", str(comp.n))
192
+ table.add_row(f"mean({args.model_a})", f"{comp.mean_a:.4f}")
193
+ table.add_row(f"mean({args.model_b})", f"{comp.mean_b:.4f}")
194
+ table.add_row("mean diff (A - B)", f"{comp.mean_diff:.4f}")
195
+ table.add_row("paired SE", f"{comp.se_paired:.4f}")
196
+ table.add_row(f"{int(args.confidence * 100)}% CI", f"[{comp.ci_low:.4f}, {comp.ci_high:.4f}]")
197
+ table.add_row("p-value", f"{comp.p_value:.4g}")
198
+ table.add_row("correlation(A, B)", f"{comp.correlation:.4f}")
199
+ table.add_row("unpaired SE (for reference)", f"{comp.se_unpaired:.4f}")
200
+ table.add_row("variance reduction from pairing", f"{comp.variance_reduction:.1%}")
201
+ if comp.se_clustered is not None:
202
+ table.add_row("clustered paired SE", f"{comp.se_clustered:.4f}")
203
+ table.add_row(
204
+ "clustered CI", f"[{comp.ci_low_clustered:.4f}, {comp.ci_high_clustered:.4f}]"
205
+ )
206
+ if comp.mcnemar is not None:
207
+ table.add_row(
208
+ "McNemar discordant (A wrong/B right, A right/B wrong)",
209
+ f"{comp.mcnemar.n01} / {comp.mcnemar.n10}",
210
+ )
211
+ table.add_row("McNemar exact p-value", f"{comp.mcnemar.p_value:.4g}")
212
+ console.print(table)
213
+
214
+
215
+ def cmd_leaderboard(args: argparse.Namespace) -> None:
216
+ data = _load(args.file, _columns_from(args))
217
+ lb = build_leaderboard(data, confidence=args.confidence, alpha=args.alpha)
218
+
219
+ if args.plot:
220
+ svg = forest_plot_svg(lb, title=args.plot_title)
221
+ Path(args.plot).write_text(svg, encoding="utf-8")
222
+
223
+ if args.json:
224
+ _print_json(lb.as_dict())
225
+ return
226
+
227
+ console = _console()
228
+ table = Table(title="leaderboard", header_style="bold cyan")
229
+ table.add_column("rank", justify="right")
230
+ table.add_column("model")
231
+ table.add_column("mean", justify="right")
232
+ table.add_column(f"{int(args.confidence * 100)}% CI", justify="right")
233
+ table.add_column("n", justify="right")
234
+ table.add_column("group")
235
+ letter_of: dict[str, str] = {}
236
+ for i, group in enumerate(lb.groups):
237
+ letter = chr(ord("a") + i)
238
+ for m in group:
239
+ letter_of[m] = letter_of.get(m, "") + letter
240
+ for rank, e in enumerate(lb.entries, start=1):
241
+ table.add_row(
242
+ str(rank),
243
+ e.model,
244
+ f"{e.mean:.4f}",
245
+ f"[{e.ci_low:.4f}, {e.ci_high:.4f}]",
246
+ str(e.n),
247
+ letter_of.get(e.model, ""),
248
+ )
249
+ console.print(table)
250
+ console.print(
251
+ "[dim]Models sharing a group letter are not statistically distinguishable "
252
+ f"(Holm-corrected paired test, alpha={args.alpha}).[/dim]"
253
+ )
254
+
255
+ pt = Table(title="pairwise paired tests (Holm-corrected)", header_style="bold magenta")
256
+ pt.add_column("A")
257
+ pt.add_column("B")
258
+ pt.add_column("mean diff", justify="right")
259
+ pt.add_column("p (unclustered)", justify="right")
260
+ pt.add_column("p (used, Holm)", justify="right")
261
+ pt.add_column("significant?")
262
+ for pr in lb.pairwise:
263
+ sig = "yes" if pr.p_holm < args.alpha else "no"
264
+ pt.add_row(
265
+ pr.model_a,
266
+ pr.model_b,
267
+ f"{pr.comparison.mean_diff:+.4f}",
268
+ f"{pr.comparison.p_value:.4g}",
269
+ f"{pr.p_holm:.4g}",
270
+ sig,
271
+ )
272
+ console.print(pt)
273
+ console.print(
274
+ "[dim]'p (used, Holm)' is the cluster-robust paired p-value (when clusters are "
275
+ "present) after Holm correction across all pairs; otherwise the unclustered paired p-value.[/dim]"
276
+ )
277
+
278
+
279
+ def cmd_power(args: argparse.Namespace) -> None:
280
+ if args.n is not None and args.delta is not None:
281
+ raise SystemExit("error: pass either --delta (solve for n) or --n (solve for MDE), not both")
282
+ if args.n is None and args.delta is None:
283
+ raise SystemExit("error: pass one of --delta or --n")
284
+
285
+ kwargs: dict[str, Any] = dict(
286
+ baseline_accuracy=args.baseline,
287
+ variance=args.variance,
288
+ alpha=args.alpha,
289
+ power=args.power,
290
+ rho=args.rho,
291
+ samples_per_question=args.samples_per_question,
292
+ cluster_design_effect=args.cluster_deff,
293
+ )
294
+
295
+ if args.delta is not None:
296
+ result = questions_needed(delta=args.delta, **kwargs)
297
+ payload = result.as_dict()
298
+ if args.json:
299
+ _print_json(payload)
300
+ return
301
+ console = _console()
302
+ table = Table(title="power: questions needed", header_style="bold cyan")
303
+ table.add_column("input")
304
+ table.add_column("value", justify="right")
305
+ table.add_row("delta", f"{args.delta}")
306
+ table.add_row("alpha", f"{args.alpha}")
307
+ table.add_row("power", f"{args.power}")
308
+ table.add_row("rho (paired correlation)", f"{args.rho}")
309
+ table.add_row("samples/question", str(args.samples_per_question))
310
+ table.add_row("cluster design effect", f"{args.cluster_deff}")
311
+ console.print(table)
312
+ console.print(f"[bold green]Questions needed: {result.n_questions}[/bold green]")
313
+ else:
314
+ mde = minimum_detectable_effect(n_questions=args.n, **kwargs)
315
+ payload = {"n_questions": args.n, "mde": mde, **{k: v for k, v in kwargs.items()}}
316
+ if args.json:
317
+ _print_json(payload)
318
+ return
319
+ console = _console()
320
+ console.print(f"[bold green]Minimum detectable effect at n={args.n}: {mde:.4f}[/bold green]")
321
+
322
+
323
+ def cmd_import(args: argparse.Namespace) -> None:
324
+ if args.adapter == "lm-eval":
325
+ if not args.model:
326
+ raise SystemExit("error: --model is required for the lm-eval adapter")
327
+ from errorbars.adapters.lm_eval import load_lm_eval_samples
328
+
329
+ data = load_lm_eval_samples(args.file, model=args.model, metric=args.metric)
330
+ elif args.adapter == "inspect":
331
+ try:
332
+ from errorbars.adapters.inspect_ai import load_inspect_log
333
+ except ImportError as exc:
334
+ raise SystemExit(f"error: {exc}") from exc
335
+
336
+ data = load_inspect_log(args.file, scorer=args.scorer)
337
+ else: # pragma: no cover - argparse `choices` already prevents this
338
+ raise SystemExit(f"error: unknown adapter {args.adapter!r}")
339
+
340
+ write_csv(data, args.output)
341
+ n_models = len(data.models())
342
+ print(f"wrote {len(data)} rows ({n_models} model{'s' if n_models != 1 else ''}) to {args.output}")
343
+
344
+
345
+ def build_parser() -> argparse.ArgumentParser:
346
+ parser = argparse.ArgumentParser(prog="errorbars", description="Error bars for LLM evals.")
347
+ parser.add_argument("--version", action="version", version=f"errorbars {__version__}")
348
+ sub = parser.add_subparsers(dest="command", required=True)
349
+
350
+ p_sum = sub.add_parser("summarize", help="mean, SE, and CI for one model")
351
+ p_sum.add_argument("file")
352
+ p_sum.add_argument("--model", default=None, help="filter to this model (default: use all rows)")
353
+ p_sum.add_argument("--confidence", type=float, default=0.95)
354
+ p_sum.add_argument("--ci", choices=["auto", "clt", "wilson", "bootstrap"], default="auto")
355
+ p_sum.add_argument("--seed", type=int, default=0, help="bootstrap RNG seed")
356
+ p_sum.add_argument("--json", action="store_true")
357
+ _column_args(p_sum)
358
+ p_sum.set_defaults(func=cmd_summarize)
359
+
360
+ p_cmp = sub.add_parser("compare", help="paired comparison of two models")
361
+ p_cmp.add_argument("file")
362
+ p_cmp.add_argument("--model-a", required=True)
363
+ p_cmp.add_argument("--model-b", required=True)
364
+ p_cmp.add_argument("--confidence", type=float, default=0.95)
365
+ p_cmp.add_argument("--json", action="store_true")
366
+ _column_args(p_cmp)
367
+ p_cmp.set_defaults(func=cmd_compare)
368
+
369
+ p_lb = sub.add_parser("leaderboard", help="rank all models with pairwise tests")
370
+ p_lb.add_argument("file")
371
+ p_lb.add_argument("--confidence", type=float, default=0.95)
372
+ p_lb.add_argument("--alpha", type=float, default=0.05)
373
+ p_lb.add_argument("--plot", default=None, help="write an SVG forest plot to this path")
374
+ p_lb.add_argument("--plot-title", default=None)
375
+ p_lb.add_argument("--json", action="store_true")
376
+ _column_args(p_lb)
377
+ p_lb.set_defaults(func=cmd_leaderboard)
378
+
379
+ p_pow = sub.add_parser("power", help="sample-size / minimum-detectable-effect planning")
380
+ p_pow.add_argument("--delta", type=float, default=None, help="effect size to detect (solve for n)")
381
+ p_pow.add_argument("--n", type=int, default=None, help="number of questions (solve for MDE)")
382
+ p_pow.add_argument("--baseline", type=float, default=None, help="baseline accuracy (binary metric)")
383
+ p_pow.add_argument(
384
+ "--variance", type=float, default=None, help="raw per-sample variance (continuous metric)"
385
+ )
386
+ p_pow.add_argument("--alpha", type=float, default=0.05)
387
+ p_pow.add_argument("--power", type=float, default=0.8)
388
+ p_pow.add_argument(
389
+ "--rho", type=float, default=0.0, help="correlation between models' per-question scores"
390
+ )
391
+ p_pow.add_argument("--samples-per-question", type=int, default=1)
392
+ p_pow.add_argument("--cluster-deff", type=float, default=1.0, help="cluster design effect (>=1)")
393
+ p_pow.add_argument("--json", action="store_true")
394
+ p_pow.set_defaults(func=cmd_power)
395
+
396
+ p_imp = sub.add_parser(
397
+ "import", help="convert an lm-evaluation-harness or Inspect AI log to the canonical CSV"
398
+ )
399
+ p_imp.add_argument("adapter", choices=["lm-eval", "inspect"])
400
+ p_imp.add_argument("file", help="lm-eval samples_*.jsonl file, or an Inspect .eval/.json log")
401
+ p_imp.add_argument("-o", "--output", required=True, help="path to write the canonical CSV to")
402
+ p_imp.add_argument(
403
+ "--model", default=None, help="model name to record (lm-eval adapter only; required for it)"
404
+ )
405
+ p_imp.add_argument(
406
+ "--metric", default=None, help="lm-eval metric to use as the score (default: first available)"
407
+ )
408
+ p_imp.add_argument(
409
+ "--scorer", default=None, help="Inspect scorer to use (default: the only one, if unambiguous)"
410
+ )
411
+ p_imp.set_defaults(func=cmd_import)
412
+
413
+ return parser
414
+
415
+
416
+ def main(argv: list[str] | None = None) -> None:
417
+ parser = build_parser()
418
+ args = parser.parse_args(argv)
419
+ try:
420
+ args.func(args)
421
+ except ValueError as exc:
422
+ raise SystemExit(f"error: {exc}") from exc
423
+
424
+
425
+ if __name__ == "__main__":
426
+ main(sys.argv[1:])