deqio 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
deqio/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Deqio package."""
2
+
3
+ __version__ = "0.1.0"
deqio/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ from .cli import main
2
+
3
+
4
+ if __name__ == "__main__":
5
+ raise SystemExit(main())
deqio/backends.py ADDED
@@ -0,0 +1,163 @@
1
+ from __future__ import annotations
2
+
3
+ import gc
4
+ from collections.abc import Callable
5
+ from typing import Any
6
+
7
+ from .config import Settings
8
+
9
+
10
+ class BackendRuntime:
11
+ """Stable server-facing scoring interface for a loaded decision runtime."""
12
+
13
+ def __init__(
14
+ self,
15
+ *,
16
+ settings: Settings,
17
+ model: Any,
18
+ tokenizer: Any,
19
+ metadata: dict[str, Any],
20
+ direct_score: Callable[..., dict],
21
+ serial_factory: Callable[..., Any],
22
+ shared_score: Callable[..., tuple[list[dict], dict]],
23
+ ) -> None:
24
+ self.settings = settings
25
+ self.name = settings.backend
26
+ self.engine = "semif"
27
+ self.model = model
28
+ self.tokenizer = tokenizer
29
+ self.metadata = metadata
30
+ self._direct_score = direct_score
31
+ self._serial_factory = serial_factory
32
+ self._shared_score = shared_score
33
+ self.serial_scorer = self._new_serial_scorer()
34
+
35
+ @classmethod
36
+ def load(cls, settings: Settings) -> "BackendRuntime":
37
+ # Every model engine, including SemIf, is executed through an isolated
38
+ # runtime. This keeps the PyPI package free of engine-specific VCS
39
+ # dependencies and gives all engines the same lifecycle semantics.
40
+ from .systemone_runtime import SystemOneRuntime
41
+
42
+ return SystemOneRuntime.load(settings) # type: ignore[return-value]
43
+
44
+ def _new_serial_scorer(self):
45
+ return self._serial_factory(
46
+ self.model,
47
+ self.tokenizer,
48
+ self.metadata,
49
+ self.settings.max_tokens,
50
+ )
51
+
52
+ def score(self, row: dict, mode: str) -> dict:
53
+ if mode == "serial":
54
+ return self.serial_scorer.score(row)
55
+ if mode == "direct":
56
+ return self._direct_score(
57
+ self.model,
58
+ self.tokenizer,
59
+ row,
60
+ self.metadata,
61
+ self.settings.max_tokens,
62
+ )
63
+ raise ValueError(f"Unsupported scoring mode: {mode}")
64
+
65
+ def score_noul(self, row: dict, mode: str) -> dict:
66
+ choice_row = {
67
+ **row,
68
+ "options": [
69
+ {"id": "yes", "description": "Yes. The evidence supports the criterion or question."},
70
+ {"id": "no", "description": "No. The evidence does not support the criterion or question."},
71
+ ],
72
+ }
73
+ return self.score(choice_row, mode)
74
+
75
+ def score_shared(self, rows: list[dict]) -> tuple[list[dict], dict]:
76
+ return self._shared_score(
77
+ self.model,
78
+ self.tokenizer,
79
+ rows,
80
+ self.metadata,
81
+ self.settings.max_tokens,
82
+ )
83
+
84
+ def clear_cache(self) -> dict[str, Any]:
85
+ """Drop reusable prefix state and release backend allocator caches when available."""
86
+ self.serial_scorer = self._new_serial_scorer()
87
+ gc.collect()
88
+
89
+ details: dict[str, Any] = {
90
+ "engine": self.engine,
91
+ "backend": self.name,
92
+ "prefix_cache": "cleared",
93
+ "model_loaded": True,
94
+ }
95
+
96
+ if self.name == "mlx":
97
+ import mlx.core as mx
98
+
99
+ before = int(mx.get_cache_memory())
100
+ mx.clear_cache()
101
+ after = int(mx.get_cache_memory())
102
+ details.update(
103
+ allocator_cache="cleared",
104
+ allocator_cache_bytes_before=before,
105
+ allocator_cache_bytes_after=after,
106
+ )
107
+ elif self.name == "cuda":
108
+ import torch
109
+
110
+ before = int(torch.cuda.memory_reserved())
111
+ torch.cuda.empty_cache()
112
+ after = int(torch.cuda.memory_reserved())
113
+ details.update(
114
+ allocator_cache="cleared",
115
+ allocator_cache_bytes_before=before,
116
+ allocator_cache_bytes_after=after,
117
+ )
118
+ elif self.name == "mps":
119
+ import torch
120
+
121
+ before = int(torch.mps.current_allocated_memory()) if hasattr(torch.mps, "current_allocated_memory") else None
122
+ torch.mps.empty_cache()
123
+ after = int(torch.mps.current_allocated_memory()) if hasattr(torch.mps, "current_allocated_memory") else None
124
+ details.update(
125
+ allocator_cache="cleared",
126
+ allocator_cache_bytes_before=before,
127
+ allocator_cache_bytes_after=after,
128
+ )
129
+
130
+ return details
131
+
132
+ def close(self) -> None:
133
+ """Release native model references before another runtime is loaded.
134
+
135
+ Live model switching calls ``close()`` before loading the replacement.
136
+ Dropping the model, tokenizer, metadata, and serial scorer here avoids
137
+ temporarily keeping two native models resident in accelerator/unified
138
+ memory during the switch. Allocator cleanup is best-effort so shutdown
139
+ cannot fail solely because a backend cache API is unavailable.
140
+ """
141
+ self.serial_scorer = None
142
+ self.model = None
143
+ self.tokenizer = None
144
+ self.metadata = {}
145
+ gc.collect()
146
+
147
+ try:
148
+ if self.name == "mlx":
149
+ import mlx.core as mx
150
+
151
+ mx.clear_cache()
152
+ elif self.name == "cuda":
153
+ import torch
154
+
155
+ torch.cuda.empty_cache()
156
+ elif self.name == "mps":
157
+ import torch
158
+
159
+ torch.mps.empty_cache()
160
+ except Exception:
161
+ # References are already dropped above. Cache cleanup is only an
162
+ # allocator hint and must not make model switching/shutdown fail.
163
+ pass
deqio/benchmark.py ADDED
@@ -0,0 +1,353 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import statistics
6
+ import sys
7
+ import time
8
+ from dataclasses import dataclass
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from .backends import BackendRuntime
14
+ from .catalog import apply_selection, get_model, get_profile, load_catalog
15
+ from .config import read_config_data, settings_from_data
16
+ from .installations import installed_profiles
17
+
18
+
19
+ DEFAULT_SUITE = Path("benchmarks/basic.json")
20
+
21
+
22
+ @dataclass
23
+ class BenchResult:
24
+ model_id: str
25
+ backend: str
26
+ engine: str
27
+ case_id: str
28
+ kind: str
29
+ passed: bool
30
+ expected: Any
31
+ actual: Any
32
+ latency_ms: float | None
33
+ assertions: int
34
+ correct: int
35
+ top_probability: float | None = None
36
+ error: str | None = None
37
+
38
+ def as_dict(self) -> dict[str, Any]:
39
+ return self.__dict__.copy()
40
+
41
+
42
+ def load_suite(path: Path) -> dict[str, Any]:
43
+ data = json.loads(path.read_text(encoding="utf-8"))
44
+ if not isinstance(data, dict) or not isinstance(data.get("cases"), list):
45
+ raise RuntimeError("Benchmark suite must contain a top-level cases array")
46
+ if int(data.get("schema_version", 0)) != 1:
47
+ raise RuntimeError("Unsupported benchmark schema_version")
48
+ seen: set[str] = set()
49
+ counts = {"noul": 0, "choice": 0, "shared": 0}
50
+ for case in data["cases"]:
51
+ if not isinstance(case, dict):
52
+ raise RuntimeError("Every benchmark case must be an object")
53
+ case_id = str(case.get("id", ""))
54
+ kind = str(case.get("type", ""))
55
+ if not case_id or case_id in seen:
56
+ raise RuntimeError(f"Benchmark case id is missing or duplicated: {case_id!r}")
57
+ if kind not in counts:
58
+ raise RuntimeError(f"Unsupported benchmark case type: {kind!r}")
59
+ seen.add(case_id)
60
+ counts[kind] += 1
61
+ return data
62
+
63
+
64
+ def _percentile(values: list[float], p: float) -> float | None:
65
+ if not values:
66
+ return None
67
+ ordered = sorted(values)
68
+ index = round((len(ordered) - 1) * p)
69
+ return ordered[index]
70
+
71
+
72
+ def _choice(raw: dict[str, Any]) -> tuple[str, float]:
73
+ ids = [str(value) for value in raw["option_ids"]]
74
+ probs = [float(value) for value in raw["probabilities"]]
75
+ index = max(range(len(probs)), key=probs.__getitem__)
76
+ return ids[index], probs[index]
77
+
78
+
79
+ def _choice_row(case: dict[str, Any]) -> dict[str, Any]:
80
+ return {
81
+ "id": str(case["id"]),
82
+ "state": case["state"],
83
+ "question": str(case["question"]),
84
+ "options": list(case["options"]),
85
+ }
86
+
87
+
88
+ def _run_case(runtime: BackendRuntime, model: dict[str, str], case: dict[str, Any]) -> BenchResult:
89
+ kind = str(case["type"])
90
+ started = time.perf_counter()
91
+ try:
92
+ if kind == "noul":
93
+ row = {"id": str(case["id"]), "state": case["state"], "question": str(case["question"])}
94
+ raw = runtime.score_noul(row, "serial")
95
+ actual, top = _choice(raw)
96
+ expected = str(case["expected"])
97
+ latency_ms = float(raw.get("total_seconds", time.perf_counter() - started)) * 1000.0
98
+ return BenchResult(**model, case_id=str(case["id"]), kind=kind, passed=actual == expected,
99
+ expected=expected, actual=actual, latency_ms=latency_ms,
100
+ assertions=1, correct=int(actual == expected), top_probability=top)
101
+
102
+ if kind == "choice":
103
+ raw = runtime.score(_choice_row(case), "serial")
104
+ actual, top = _choice(raw)
105
+ expected = str(case["expected"])
106
+ latency_ms = float(raw.get("total_seconds", time.perf_counter() - started)) * 1000.0
107
+ return BenchResult(**model, case_id=str(case["id"]), kind=kind, passed=actual == expected,
108
+ expected=expected, actual=actual, latency_ms=latency_ms,
109
+ assertions=1, correct=int(actual == expected), top_probability=top)
110
+
111
+ rows = []
112
+ expected: list[str] = []
113
+ for decision in case["decisions"]:
114
+ rows.append({
115
+ "id": str(decision["id"]),
116
+ "state": case["state"],
117
+ "question": str(decision["question"]),
118
+ "options": list(decision["options"]),
119
+ })
120
+ expected.append(str(decision["expected"]))
121
+ raw_results, timing = runtime.score_shared(rows)
122
+ actual = [_choice(raw)[0] for raw in raw_results]
123
+ correct = sum(a == e for a, e in zip(actual, expected))
124
+ latency_ms = float(timing.get("total_seconds", time.perf_counter() - started)) * 1000.0
125
+ return BenchResult(**model, case_id=str(case["id"]), kind=kind, passed=correct == len(expected),
126
+ expected=expected, actual=actual, latency_ms=latency_ms,
127
+ assertions=len(expected), correct=correct)
128
+ except Exception as error:
129
+ return BenchResult(**model, case_id=str(case["id"]), kind=kind, passed=False,
130
+ expected=case.get("expected") or [d.get("expected") for d in case.get("decisions", [])],
131
+ actual=None, latency_ms=(time.perf_counter() - started) * 1000.0,
132
+ assertions=max(1, len(case.get("decisions", []))), correct=0, error=str(error))
133
+
134
+
135
+ def summarize(results: list[BenchResult]) -> dict[str, Any]:
136
+ summary: dict[str, Any] = {}
137
+ groups: dict[str, list[BenchResult]] = {}
138
+ for result in results:
139
+ groups.setdefault(result.kind, []).append(result)
140
+ all_latencies = [r.latency_ms for r in results if r.latency_ms is not None]
141
+ assertions = sum(r.assertions for r in results)
142
+ correct = sum(r.correct for r in results)
143
+ summary["overall"] = {
144
+ "cases": len(results),
145
+ "passed_cases": sum(r.passed for r in results),
146
+ "case_accuracy": (sum(r.passed for r in results) / len(results)) if results else 0.0,
147
+ "assertions": assertions,
148
+ "correct": correct,
149
+ "decision_accuracy": (correct / assertions) if assertions else 0.0,
150
+ "mean_ms": statistics.fmean(all_latencies) if all_latencies else None,
151
+ "median_ms": statistics.median(all_latencies) if all_latencies else None,
152
+ "p95_ms": _percentile(all_latencies, 0.95),
153
+ }
154
+ for kind, rows in groups.items():
155
+ latencies = [r.latency_ms for r in rows if r.latency_ms is not None]
156
+ kind_assertions = sum(r.assertions for r in rows)
157
+ kind_correct = sum(r.correct for r in rows)
158
+ summary[kind] = {
159
+ "cases": len(rows),
160
+ "passed_cases": sum(r.passed for r in rows),
161
+ "case_accuracy": sum(r.passed for r in rows) / len(rows),
162
+ "assertions": kind_assertions,
163
+ "correct": kind_correct,
164
+ "decision_accuracy": kind_correct / kind_assertions if kind_assertions else 0.0,
165
+ "mean_ms": statistics.fmean(latencies) if latencies else None,
166
+ "median_ms": statistics.median(latencies) if latencies else None,
167
+ "p95_ms": _percentile(latencies, 0.95),
168
+ }
169
+ return summary
170
+
171
+
172
+ def _installed(config_path: Path, config_data: dict[str, Any], catalog: dict[str, Any]) -> list[dict[str, Any]]:
173
+ return [row for row in installed_profiles(
174
+ config_path=config_path,
175
+ config_data=config_data,
176
+ catalog=catalog,
177
+ active_model_id=str(config_data.get("model_id", "")),
178
+ active_backend=str(config_data.get("backend", "")),
179
+ ) if row["installed"] and row["host_compatible"]]
180
+
181
+
182
+ def _select_profiles(rows: list[dict[str, Any]], args: argparse.Namespace) -> list[dict[str, Any]]:
183
+ if not rows:
184
+ raise RuntimeError("No installed model profiles are available on this host")
185
+ if args.all:
186
+ return rows
187
+ if args.model:
188
+ selected = []
189
+ for value in args.model:
190
+ if ":" not in value:
191
+ raise RuntimeError("--model must use MODEL_ID:BACKEND, for example decider-0.8b:mps")
192
+ model_id, backend = value.rsplit(":", 1)
193
+ match = next((row for row in rows if row["model_id"] == model_id and row["backend"] == backend), None)
194
+ if match is None:
195
+ raise RuntimeError(f"Installed profile not found: {value}")
196
+ selected.append(match)
197
+ return selected
198
+
199
+ print("Installed model profiles:")
200
+ for index, row in enumerate(rows, start=1):
201
+ print(f" {index}. {row['label']} — {row['backend']} ({row['engine']})")
202
+ print(" A. all installed profiles")
203
+ value = input("Select models [A or comma-separated numbers]: ").strip().lower()
204
+ if value in {"", "a", "all"}:
205
+ return rows
206
+ indexes = []
207
+ for token in value.split(","):
208
+ index = int(token.strip()) - 1
209
+ if index not in range(len(rows)):
210
+ raise RuntimeError("Invalid benchmark model selection")
211
+ indexes.append(index)
212
+ return [rows[index] for index in dict.fromkeys(indexes)]
213
+
214
+
215
+ def _runtime_settings(config_path: Path, config_data: dict[str, Any], catalog: dict[str, Any], row: dict[str, Any]):
216
+ entry = get_model(catalog, str(row["model_id"]))
217
+ profile = get_profile(catalog, str(row["model_id"]), str(row["backend"]))
218
+ selected = apply_selection(config_data, entry, profile, str(row["backend"]))
219
+ return settings_from_data(config_path, selected, apply_environment=False)
220
+
221
+
222
+ def _warmup(runtime: BackendRuntime) -> None:
223
+ row = {
224
+ "id": "benchmark-warmup",
225
+ "state": "The service is ready and healthy.",
226
+ "question": "What is the service state?",
227
+ "options": [
228
+ {"id": "ready", "description": "The service is ready."},
229
+ {"id": "not_ready", "description": "The service is not ready."},
230
+ ],
231
+ }
232
+ runtime.score(row, "serial")
233
+ runtime.score(row, "serial")
234
+
235
+
236
+ def _fmt_ms(value: Any) -> str:
237
+ return "-" if value is None else f"{float(value):.1f}"
238
+
239
+
240
+ def run(args: argparse.Namespace) -> int:
241
+ config_path, config_data = read_config_data(args.config)
242
+ catalog_path = Path(str(config_data.get("model_catalog", "models.json")))
243
+ if not catalog_path.is_absolute():
244
+ catalog_path = config_path.parent / catalog_path
245
+ catalog = load_catalog(catalog_path)
246
+ suite_path = Path(args.suite).expanduser().resolve()
247
+ suite = load_suite(suite_path)
248
+ profiles = _select_profiles(_installed(config_path, config_data, catalog), args)
249
+
250
+ stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
251
+ output_dir = Path(args.output).expanduser().resolve() if args.output else (config_path.parent / ".deqio" / "benchmarks" / stamp)
252
+ output_dir.mkdir(parents=True, exist_ok=True)
253
+ results_path = output_dir / "results.jsonl"
254
+ summary_path = output_dir / "summary.json"
255
+
256
+ print(f"[bench] suite={suite.get('name', suite_path.name)} cases={len(suite['cases'])}")
257
+ print(f"[bench] models={len(profiles)} output={output_dir}")
258
+
259
+ all_summaries: list[dict[str, Any]] = []
260
+ with results_path.open("w", encoding="utf-8") as log:
261
+ for profile_row in profiles:
262
+ settings = _runtime_settings(config_path, config_data, catalog, profile_row)
263
+ label = f"{settings.model_id}/{settings.backend}"
264
+ print(f"\n[bench] Loading {label} ({settings.engine})")
265
+ runtime = None
266
+ model_results: list[BenchResult] = []
267
+ load_started = time.perf_counter()
268
+ try:
269
+ runtime = BackendRuntime.load(settings)
270
+ load_ms = (time.perf_counter() - load_started) * 1000.0
271
+ _warmup(runtime)
272
+ print(f"[bench] Ready {label} load={load_ms:.1f}ms; warmup complete")
273
+ totals = {kind: sum(c["type"] == kind for c in suite["cases"]) for kind in ("noul", "choice", "shared")}
274
+ seen = {"noul": 0, "choice": 0, "shared": 0}
275
+ for case in suite["cases"]:
276
+ kind = str(case["type"])
277
+ seen[kind] += 1
278
+ result = _run_case(runtime, {"model_id": settings.model_id, "backend": settings.backend, "engine": settings.engine}, case)
279
+ model_results.append(result)
280
+ log.write(json.dumps(result.as_dict(), ensure_ascii=False) + "\n")
281
+ log.flush()
282
+ status = "PASS" if result.passed else "FAIL"
283
+ detail = f"{result.correct}/{result.assertions}" if kind == "shared" else f"actual={result.actual} expected={result.expected}"
284
+ error = f" error={result.error}" if result.error else ""
285
+ print(f"[bench] {label:30} {kind:6} {seen[kind]:02}/{totals[kind]:02} {status:4} {detail} {_fmt_ms(result.latency_ms)}ms{error}")
286
+ except Exception as error:
287
+ load_ms = (time.perf_counter() - load_started) * 1000.0
288
+ print(f"[bench] ERROR loading {label}: {error}")
289
+ finally:
290
+ if runtime is not None:
291
+ runtime.close()
292
+
293
+ model_summary = summarize(model_results)
294
+ all_summaries.append({
295
+ "model_id": settings.model_id,
296
+ "backend": settings.backend,
297
+ "engine": settings.engine,
298
+ "load_ms": load_ms,
299
+ "summary": model_summary,
300
+ })
301
+
302
+ document = {
303
+ "schema_version": 1,
304
+ "created_at": datetime.now(timezone.utc).isoformat(),
305
+ "suite": str(suite_path),
306
+ "suite_name": suite.get("name"),
307
+ "models": all_summaries,
308
+ }
309
+ summary_path.write_text(json.dumps(document, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
310
+
311
+ print("\nBenchmark summary")
312
+ print(f"{'MODEL':30} {'TYPE':7} {'CASES':>7} {'ACC':>8} {'DEC ACC':>8} {'MEDIAN':>9} {'P95':>9}")
313
+ print("-" * 91)
314
+ for item in all_summaries:
315
+ model = f"{item['model_id']}:{item['backend']}"
316
+ for kind in ("overall", "noul", "choice", "shared"):
317
+ stats = item["summary"].get(kind, {})
318
+ cases = int(stats.get("cases", 0))
319
+ acc = float(stats.get("case_accuracy", 0.0)) * 100.0
320
+ dacc = float(stats.get("decision_accuracy", 0.0)) * 100.0
321
+ label = "all" if kind == "overall" else kind
322
+ print(
323
+ f"{model:30} {label:7} {cases:7d} {acc:7.1f}% {dacc:7.1f}% "
324
+ f"{_fmt_ms(stats.get('median_ms')):>7}ms {_fmt_ms(stats.get('p95_ms')):>7}ms"
325
+ )
326
+ print(f"\n[bench] results: {results_path}")
327
+ print(f"[bench] summary: {summary_path}")
328
+ return 0
329
+
330
+
331
+ def build_parser() -> argparse.ArgumentParser:
332
+ parser = argparse.ArgumentParser(prog="deqio benchmark", description="Benchmark installed Deqio model profiles.")
333
+ parser.add_argument("--config", help="Path to config.json (default: ./config.json)")
334
+ parser.add_argument("--suite", default=str(DEFAULT_SUITE), help="Benchmark suite JSON file")
335
+ parser.add_argument("--output", help="Output directory (default: .deqio/benchmarks/<timestamp>)")
336
+ group = parser.add_mutually_exclusive_group()
337
+ group.add_argument("--all", action="store_true", help="Benchmark every installed host-compatible profile")
338
+ group.add_argument("--model", action="append", help="Benchmark MODEL_ID:BACKEND; repeat to select several")
339
+ return parser
340
+
341
+
342
+ def main(argv: list[str] | None = None) -> int:
343
+ parser = build_parser()
344
+ args = parser.parse_args(argv)
345
+ try:
346
+ return run(args)
347
+ except (RuntimeError, ValueError, OSError, json.JSONDecodeError) as error:
348
+ print(f"error: {error}", file=sys.stderr)
349
+ return 2
350
+
351
+
352
+ if __name__ == "__main__":
353
+ raise SystemExit(main())
@@ -0,0 +1,120 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from datetime import datetime, timezone
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+
9
+ def benchmark_root(config_path: Path) -> Path:
10
+ return config_path.parent / ".deqio" / "benchmarks"
11
+
12
+
13
+ def _read_json(path: Path) -> dict[str, Any]:
14
+ try:
15
+ data = json.loads(path.read_text(encoding="utf-8"))
16
+ except json.JSONDecodeError as error:
17
+ raise RuntimeError(f"Invalid JSON in benchmark file {path}: {error}") from error
18
+ if not isinstance(data, dict):
19
+ raise RuntimeError(f"Benchmark file must contain a JSON object: {path}")
20
+ return data
21
+
22
+
23
+ def _run_dir(config_path: Path, run_id: str) -> Path:
24
+ if not run_id or run_id in {".", ".."} or Path(run_id).name != run_id:
25
+ raise FileNotFoundError(f"Benchmark run not found: {run_id}")
26
+ root = benchmark_root(config_path)
27
+ candidate = root / run_id
28
+ if not candidate.is_dir():
29
+ raise FileNotFoundError(f"Benchmark run not found: {run_id}")
30
+ return candidate
31
+
32
+
33
+ def _file_iso(path: Path) -> str:
34
+ return datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc).isoformat()
35
+
36
+
37
+ def _count_jsonl(path: Path) -> int:
38
+ if not path.is_file():
39
+ return 0
40
+ with path.open("r", encoding="utf-8") as handle:
41
+ return sum(1 for line in handle if line.strip())
42
+
43
+
44
+ def list_benchmark_runs(config_path: Path) -> list[dict[str, Any]]:
45
+ root = benchmark_root(config_path)
46
+ if not root.is_dir():
47
+ return []
48
+
49
+ runs: list[dict[str, Any]] = []
50
+ for directory in root.iterdir():
51
+ if not directory.is_dir():
52
+ continue
53
+ summary_path = directory / "summary.json"
54
+ results_path = directory / "results.jsonl"
55
+ if not summary_path.is_file() and not results_path.is_file():
56
+ continue
57
+
58
+ summary: dict[str, Any] | None = None
59
+ summary_error: str | None = None
60
+ if summary_path.is_file():
61
+ try:
62
+ summary = _read_json(summary_path)
63
+ except RuntimeError as error:
64
+ summary_error = str(error)
65
+
66
+ created_at = None
67
+ suite_name = None
68
+ model_count = None
69
+ if summary is not None:
70
+ created_at = summary.get("created_at")
71
+ suite_name = summary.get("suite_name")
72
+ models = summary.get("models")
73
+ if isinstance(models, list):
74
+ model_count = len(models)
75
+
76
+ fallback_path = summary_path if summary_path.is_file() else results_path
77
+ runs.append({
78
+ "id": directory.name,
79
+ "created_at": str(created_at or _file_iso(fallback_path)),
80
+ "suite_name": suite_name,
81
+ "models": model_count,
82
+ "results": _count_jsonl(results_path),
83
+ "has_summary": summary_path.is_file(),
84
+ "has_results": results_path.is_file(),
85
+ "summary_error": summary_error,
86
+ })
87
+
88
+ runs.sort(key=lambda row: (str(row.get("created_at", "")), str(row["id"])), reverse=True)
89
+ return runs
90
+
91
+
92
+ def read_benchmark_summary(config_path: Path, run_id: str) -> dict[str, Any]:
93
+ path = _run_dir(config_path, run_id) / "summary.json"
94
+ if not path.is_file():
95
+ raise FileNotFoundError(f"summary.json is missing for benchmark run: {run_id}")
96
+ return _read_json(path)
97
+
98
+
99
+ def read_benchmark_results(config_path: Path, run_id: str) -> list[dict[str, Any]]:
100
+ path = _run_dir(config_path, run_id) / "results.jsonl"
101
+ if not path.is_file():
102
+ raise FileNotFoundError(f"results.jsonl is missing for benchmark run: {run_id}")
103
+
104
+ results: list[dict[str, Any]] = []
105
+ with path.open("r", encoding="utf-8") as handle:
106
+ for line_number, line in enumerate(handle, start=1):
107
+ if not line.strip():
108
+ continue
109
+ try:
110
+ row = json.loads(line)
111
+ except json.JSONDecodeError as error:
112
+ raise RuntimeError(
113
+ f"Invalid JSONL in benchmark run {run_id} at line {line_number}: {error}"
114
+ ) from error
115
+ if not isinstance(row, dict):
116
+ raise RuntimeError(
117
+ f"Benchmark result at line {line_number} must be a JSON object"
118
+ )
119
+ results.append(row)
120
+ return results