quantcost 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgellm/__init__.py +7 -0
- edgellm/base.py +44 -0
- edgellm/benchmark.py +226 -0
- edgellm/card.py +234 -0
- edgellm/cli.py +370 -0
- edgellm/cli_bench.py +195 -0
- edgellm/config.py +120 -0
- edgellm/data/SOURCE.md +17 -0
- edgellm/data/eval_wikitext2.txt +205 -0
- edgellm/eval_lite.py +128 -0
- edgellm/export.py +65 -0
- edgellm/hub.py +125 -0
- edgellm/leaderboard.py +210 -0
- edgellm/models.py +91 -0
- edgellm/ort_lite.py +236 -0
- edgellm/quantize.py +134 -0
- edgellm/render.py +164 -0
- edgellm/report.py +53 -0
- edgellm/runners.py +170 -0
- edgellm/submit.py +225 -0
- edgellm/sweep.py +309 -0
- edgellm/validate.py +274 -0
- quantcost-0.2.0.dist-info/METADATA +265 -0
- quantcost-0.2.0.dist-info/RECORD +27 -0
- quantcost-0.2.0.dist-info/WHEEL +4 -0
- quantcost-0.2.0.dist-info/entry_points.txt +3 -0
- quantcost-0.2.0.dist-info/licenses/LICENSE +21 -0
edgellm/__init__.py
ADDED
edgellm/base.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Backend-agnostic interfaces, shared by the torch and the torch-free paths.
|
|
2
|
+
|
|
3
|
+
These used to live in :mod:`edgellm.runners`, which imports ``torch`` at module
|
|
4
|
+
level. The lightweight benchmark path (ONNX Runtime + NumPy, no torch, no
|
|
5
|
+
transformers) needs the same interface without paying for that import, so the
|
|
6
|
+
parts that carry no torch dependency live here and ``runners`` re-exports them.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from abc import ABC, abstractmethod
|
|
12
|
+
from dataclasses import asdict, dataclass
|
|
13
|
+
|
|
14
|
+
from edgellm.config import GenerationConfig
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class GenerationResult:
|
|
19
|
+
"""The text plus the timing/throughput numbers for one generation call."""
|
|
20
|
+
|
|
21
|
+
backend: str
|
|
22
|
+
prompt: str
|
|
23
|
+
text: str
|
|
24
|
+
prompt_tokens: int
|
|
25
|
+
generated_tokens: int
|
|
26
|
+
latency_s: float
|
|
27
|
+
tokens_per_second: float
|
|
28
|
+
|
|
29
|
+
def as_dict(self) -> dict:
|
|
30
|
+
return asdict(self)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class InferenceRunner(ABC):
|
|
34
|
+
"""Common interface for all inference backends."""
|
|
35
|
+
|
|
36
|
+
name: str = "base"
|
|
37
|
+
|
|
38
|
+
@abstractmethod
|
|
39
|
+
def generate(self, prompt: str, generation: GenerationConfig) -> GenerationResult:
|
|
40
|
+
"""Generate a completion for ``prompt`` and report timing."""
|
|
41
|
+
|
|
42
|
+
def close(self) -> None:
|
|
43
|
+
"""Release whatever the backend holds (sessions, memory). Optional."""
|
|
44
|
+
return None
|
edgellm/benchmark.py
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""Benchmark harness: latency, throughput, peak RAM, on-disk size, perplexity.
|
|
2
|
+
|
|
3
|
+
Every metric here is measured on the real machine. Nothing is estimated or
|
|
4
|
+
scaled. The harness is backend-agnostic: it drives any :class:`InferenceRunner`
|
|
5
|
+
and computes perplexity from any callable that returns logits, so PyTorch and
|
|
6
|
+
ONNX Runtime (FP32 or quantized) are measured identically.
|
|
7
|
+
|
|
8
|
+
Perplexity method (documented so results are reproducible): the eval text is a
|
|
9
|
+
slice of WikiText-2, tokenized once and split into non-overlapping windows of
|
|
10
|
+
``eval_max_length`` tokens. For each window we compute the summed
|
|
11
|
+
next-token cross-entropy; perplexity is ``exp(total_nll / total_tokens)``.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import statistics
|
|
18
|
+
import threading
|
|
19
|
+
import time
|
|
20
|
+
from collections.abc import Callable
|
|
21
|
+
from dataclasses import asdict, dataclass
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import TYPE_CHECKING
|
|
24
|
+
|
|
25
|
+
import psutil
|
|
26
|
+
|
|
27
|
+
from edgellm.base import InferenceRunner
|
|
28
|
+
from edgellm.config import BenchmarkConfig, GenerationConfig
|
|
29
|
+
|
|
30
|
+
if TYPE_CHECKING: # ``torch`` is only needed by PerplexityEvaluator (the heavy path).
|
|
31
|
+
import torch
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class BenchmarkResult:
|
|
36
|
+
"""All measured numbers for one (backend, precision) combination."""
|
|
37
|
+
|
|
38
|
+
backend: str
|
|
39
|
+
precision: str
|
|
40
|
+
device: str
|
|
41
|
+
model_id: str
|
|
42
|
+
size_mb: float | None
|
|
43
|
+
latency_s_mean: float
|
|
44
|
+
latency_s_std: float
|
|
45
|
+
tokens_per_second: float
|
|
46
|
+
peak_ram_mb: float
|
|
47
|
+
perplexity: float | None
|
|
48
|
+
generated_tokens: int
|
|
49
|
+
measured_runs: int
|
|
50
|
+
|
|
51
|
+
def as_dict(self) -> dict:
|
|
52
|
+
return asdict(self)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class _PeakRSSSampler:
|
|
56
|
+
"""Context manager sampling process RSS on a background thread to find its peak."""
|
|
57
|
+
|
|
58
|
+
def __init__(self, interval_s: float = 0.02) -> None:
|
|
59
|
+
self.interval_s = interval_s
|
|
60
|
+
self.peak_bytes = 0
|
|
61
|
+
self._stop = threading.Event()
|
|
62
|
+
self._thread: threading.Thread | None = None
|
|
63
|
+
self._proc = psutil.Process()
|
|
64
|
+
|
|
65
|
+
def _run(self) -> None:
|
|
66
|
+
while not self._stop.is_set():
|
|
67
|
+
self.peak_bytes = max(self.peak_bytes, self._proc.memory_info().rss)
|
|
68
|
+
time.sleep(self.interval_s)
|
|
69
|
+
|
|
70
|
+
def __enter__(self) -> _PeakRSSSampler:
|
|
71
|
+
self.peak_bytes = self._proc.memory_info().rss
|
|
72
|
+
self._thread = threading.Thread(target=self._run, daemon=True)
|
|
73
|
+
self._thread.start()
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
def __exit__(self, *exc) -> None:
|
|
77
|
+
self._stop.set()
|
|
78
|
+
if self._thread is not None:
|
|
79
|
+
self._thread.join()
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def peak_mb(self) -> float:
|
|
83
|
+
return self.peak_bytes / (1024 * 1024)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def measure_size_mb(path: Path, patterns: tuple[str, ...]) -> float:
|
|
87
|
+
"""Sum the on-disk size (MB) of files under ``path`` matching any glob pattern."""
|
|
88
|
+
total = 0
|
|
89
|
+
for pattern in patterns:
|
|
90
|
+
for f in path.glob(pattern):
|
|
91
|
+
if f.is_file():
|
|
92
|
+
total += f.stat().st_size
|
|
93
|
+
return total / (1024 * 1024)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class PerplexityEvaluator:
|
|
97
|
+
"""Compute perplexity on a WikiText slice from any logits-producing callable."""
|
|
98
|
+
|
|
99
|
+
def __init__(self, config: BenchmarkConfig) -> None:
|
|
100
|
+
self.config = config
|
|
101
|
+
|
|
102
|
+
def _load_text(self) -> str:
|
|
103
|
+
from datasets import load_dataset
|
|
104
|
+
|
|
105
|
+
ds = load_dataset(self.config.eval_dataset, self.config.eval_config, split="test")
|
|
106
|
+
lines = [t for t in ds["text"] if t.strip()]
|
|
107
|
+
return "\n\n".join(lines[: self.config.eval_num_samples])
|
|
108
|
+
|
|
109
|
+
def evaluate(
|
|
110
|
+
self,
|
|
111
|
+
forward_fn: Callable[[torch.Tensor, torch.Tensor], torch.Tensor],
|
|
112
|
+
tokenizer,
|
|
113
|
+
) -> float:
|
|
114
|
+
"""``forward_fn(input_ids, attention_mask) -> logits`` (all torch tensors)."""
|
|
115
|
+
import torch
|
|
116
|
+
|
|
117
|
+
text = self._load_text()
|
|
118
|
+
input_ids = tokenizer(text, return_tensors="pt").input_ids
|
|
119
|
+
max_len = self.config.eval_max_length
|
|
120
|
+
|
|
121
|
+
total_nll = torch.zeros((), dtype=torch.float64)
|
|
122
|
+
total_tokens = 0
|
|
123
|
+
for start in range(0, input_ids.shape[1], max_len):
|
|
124
|
+
window = input_ids[:, start : start + max_len]
|
|
125
|
+
if window.shape[1] < 2:
|
|
126
|
+
continue
|
|
127
|
+
attn = torch.ones_like(window)
|
|
128
|
+
logits = forward_fn(window, attn).float()
|
|
129
|
+
shift_logits = logits[:, :-1, :].reshape(-1, logits.size(-1))
|
|
130
|
+
shift_labels = window[:, 1:].reshape(-1).to(shift_logits.device)
|
|
131
|
+
nll = torch.nn.functional.cross_entropy(shift_logits, shift_labels, reduction="sum")
|
|
132
|
+
total_nll += nll.double().cpu()
|
|
133
|
+
total_tokens += int(shift_labels.numel())
|
|
134
|
+
|
|
135
|
+
return float(torch.exp(total_nll / total_tokens))
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
class BenchmarkHarness:
|
|
139
|
+
"""Drive a runner through warmup + measured runs and collect all metrics."""
|
|
140
|
+
|
|
141
|
+
def __init__(self, config: BenchmarkConfig) -> None:
|
|
142
|
+
self.config = config
|
|
143
|
+
|
|
144
|
+
def _bench_generation_config(self, base: GenerationConfig) -> GenerationConfig:
|
|
145
|
+
"""Deterministic, fixed-length decoding so runs are comparable."""
|
|
146
|
+
n = self.config.gen_tokens
|
|
147
|
+
return GenerationConfig(
|
|
148
|
+
max_new_tokens=n,
|
|
149
|
+
min_new_tokens=n, # force full length -> stable token count across backends
|
|
150
|
+
do_sample=False,
|
|
151
|
+
seed=base.seed,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def run(
|
|
155
|
+
self,
|
|
156
|
+
runner: InferenceRunner,
|
|
157
|
+
*,
|
|
158
|
+
precision: str,
|
|
159
|
+
device: str,
|
|
160
|
+
model_id: str,
|
|
161
|
+
generation: GenerationConfig,
|
|
162
|
+
size_mb: float | None = None,
|
|
163
|
+
perplexity: float | None = None,
|
|
164
|
+
) -> BenchmarkResult:
|
|
165
|
+
gen = self._bench_generation_config(generation)
|
|
166
|
+
prompt = self.config.prompt
|
|
167
|
+
|
|
168
|
+
for _ in range(self.config.warmup_runs):
|
|
169
|
+
runner.generate(prompt, gen)
|
|
170
|
+
|
|
171
|
+
latencies: list[float] = []
|
|
172
|
+
throughputs: list[float] = []
|
|
173
|
+
generated_tokens = 0
|
|
174
|
+
with _PeakRSSSampler() as sampler:
|
|
175
|
+
for _ in range(self.config.measured_runs):
|
|
176
|
+
result = runner.generate(prompt, gen)
|
|
177
|
+
latencies.append(result.latency_s)
|
|
178
|
+
throughputs.append(result.tokens_per_second)
|
|
179
|
+
generated_tokens = result.generated_tokens
|
|
180
|
+
|
|
181
|
+
return BenchmarkResult(
|
|
182
|
+
backend=runner.name,
|
|
183
|
+
precision=precision,
|
|
184
|
+
device=device,
|
|
185
|
+
model_id=model_id,
|
|
186
|
+
size_mb=round(size_mb, 2) if size_mb is not None else None,
|
|
187
|
+
latency_s_mean=round(statistics.fmean(latencies), 4),
|
|
188
|
+
latency_s_std=round(statistics.pstdev(latencies), 4),
|
|
189
|
+
tokens_per_second=round(statistics.fmean(throughputs), 2),
|
|
190
|
+
peak_ram_mb=round(sampler.peak_mb, 1),
|
|
191
|
+
perplexity=round(perplexity, 3) if perplexity is not None else None,
|
|
192
|
+
generated_tokens=generated_tokens,
|
|
193
|
+
measured_runs=self.config.measured_runs,
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def save_results(results: list[BenchmarkResult], path: Path) -> None:
|
|
198
|
+
"""Merge results into a JSON file keyed by (backend, precision)."""
|
|
199
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
200
|
+
existing: dict[str, dict] = {}
|
|
201
|
+
if path.exists():
|
|
202
|
+
for row in json.loads(path.read_text()):
|
|
203
|
+
existing[f"{row['backend']}|{row['precision']}"] = row
|
|
204
|
+
for r in results:
|
|
205
|
+
existing[f"{r.backend}|{r.precision}"] = r.as_dict()
|
|
206
|
+
path.write_text(json.dumps(list(existing.values()), indent=2) + "\n")
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def render_markdown(results_json: Path) -> str:
|
|
210
|
+
"""Render the JSON results as a Markdown table (real numbers only)."""
|
|
211
|
+
rows = json.loads(results_json.read_text()) if results_json.exists() else []
|
|
212
|
+
header = (
|
|
213
|
+
"| Backend | Precision | Device | Size (MB) | Latency (s) | "
|
|
214
|
+
"Throughput (tok/s) | Peak RAM (MB) | Perplexity |\n"
|
|
215
|
+
"| --- | --- | --- | --- | --- | --- | --- | --- |\n"
|
|
216
|
+
)
|
|
217
|
+
lines = []
|
|
218
|
+
for r in rows:
|
|
219
|
+
size = f"{r['size_mb']}" if r.get("size_mb") is not None else "—"
|
|
220
|
+
ppl = f"{r['perplexity']}" if r.get("perplexity") is not None else "—"
|
|
221
|
+
lat = f"{r['latency_s_mean']} ± {r['latency_s_std']}"
|
|
222
|
+
lines.append(
|
|
223
|
+
f"| {r['backend']} | {r['precision']} | {r['device']} | {size} | "
|
|
224
|
+
f"{lat} | {r['tokens_per_second']} | {r['peak_ram_mb']} | {ppl} |"
|
|
225
|
+
)
|
|
226
|
+
return header + "\n".join(lines) + "\n"
|
edgellm/card.py
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""The result card: what one machine measured, in a shape that can be compared.
|
|
2
|
+
|
|
3
|
+
A card is the unit a contributor submits. It pairs the measured rows with enough
|
|
4
|
+
hardware and software context to make them meaningful, and with the knobs that
|
|
5
|
+
would otherwise silently change the numbers (thread count, token budget, eval
|
|
6
|
+
window count, corpus hash).
|
|
7
|
+
|
|
8
|
+
**On privacy.** Cards get committed to a public repository, so the fingerprint is
|
|
9
|
+
deliberately narrow: CPU model string, architecture, core count, rounded RAM, OS
|
|
10
|
+
name and release, and version strings. It never collects hostname, username,
|
|
11
|
+
local paths, MAC address, IP, serial number or any machine ID. ``fingerprint()``
|
|
12
|
+
is the only thing that reads the host, so that promise is auditable in one place.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import platform
|
|
18
|
+
import re
|
|
19
|
+
import subprocess
|
|
20
|
+
from dataclasses import asdict, dataclass, field
|
|
21
|
+
from datetime import datetime, timezone
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
#: Bumped when the card layout changes in a way a validator must notice.
|
|
26
|
+
CARD_SCHEMA_VERSION = 2
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _cpu_model() -> str:
|
|
30
|
+
"""A human-readable CPU model string, or 'unknown' if the OS will not say."""
|
|
31
|
+
system = platform.system()
|
|
32
|
+
try:
|
|
33
|
+
if system == "Darwin":
|
|
34
|
+
out = subprocess.run(
|
|
35
|
+
["sysctl", "-n", "machdep.cpu.brand_string"],
|
|
36
|
+
capture_output=True,
|
|
37
|
+
text=True,
|
|
38
|
+
timeout=5,
|
|
39
|
+
)
|
|
40
|
+
if out.returncode == 0 and out.stdout.strip():
|
|
41
|
+
return out.stdout.strip()
|
|
42
|
+
elif system == "Linux":
|
|
43
|
+
with open("/proc/cpuinfo") as fh:
|
|
44
|
+
for line in fh:
|
|
45
|
+
if line.startswith(("model name", "Model")):
|
|
46
|
+
return line.split(":", 1)[1].strip()
|
|
47
|
+
except (OSError, subprocess.SubprocessError):
|
|
48
|
+
pass
|
|
49
|
+
return platform.processor() or platform.machine() or "unknown"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def performance_cores() -> tuple[int, str]:
|
|
53
|
+
"""Count the machine's *fast* cores, and say how that was determined.
|
|
54
|
+
|
|
55
|
+
This matters far more than it looks. On a heterogeneous CPU — Apple Silicon,
|
|
56
|
+
ARM big.LITTLE, Intel hybrid — an ONNX Runtime parallel region ends on a
|
|
57
|
+
barrier, so one thread scheduled onto an efficiency core gates the whole
|
|
58
|
+
region. Measured on an M4 (4 performance + 6 efficiency), pinning all ten
|
|
59
|
+
physical cores cost int8 **66% of its throughput** (24.1 vs 40.0 tok/s) and
|
|
60
|
+
doubled run-to-run spread, while barely moving fp32: the quantized kernel is
|
|
61
|
+
compute-bound and waits on the slow thread, fp32 is bandwidth-bound and does
|
|
62
|
+
not. Defaulting to every physical core therefore understates quantized
|
|
63
|
+
performance and makes results unreproducible.
|
|
64
|
+
|
|
65
|
+
Returns ``(count, policy)`` so the card records which rule produced it.
|
|
66
|
+
"""
|
|
67
|
+
system = platform.system()
|
|
68
|
+
|
|
69
|
+
if system == "Darwin":
|
|
70
|
+
# perflevel0 is the fastest tier; the key is absent on uniform chips.
|
|
71
|
+
try:
|
|
72
|
+
out = subprocess.run(
|
|
73
|
+
["sysctl", "-n", "hw.perflevel0.physicalcpu"],
|
|
74
|
+
capture_output=True,
|
|
75
|
+
text=True,
|
|
76
|
+
timeout=5,
|
|
77
|
+
)
|
|
78
|
+
if out.returncode == 0 and out.stdout.strip().isdigit():
|
|
79
|
+
count = int(out.stdout.strip())
|
|
80
|
+
if count > 0:
|
|
81
|
+
return count, "macos-perflevel0"
|
|
82
|
+
except (OSError, subprocess.SubprocessError):
|
|
83
|
+
pass
|
|
84
|
+
|
|
85
|
+
elif system == "Linux":
|
|
86
|
+
# ARM big.LITTLE exposes a per-cpu capacity; the fast tier is the max.
|
|
87
|
+
try:
|
|
88
|
+
caps: dict[int, int] = {}
|
|
89
|
+
for path in Path("/sys/devices/system/cpu").glob("cpu[0-9]*/cpu_capacity"):
|
|
90
|
+
try:
|
|
91
|
+
caps[int(path.parent.name[3:])] = int(path.read_text().strip())
|
|
92
|
+
except (OSError, ValueError):
|
|
93
|
+
continue
|
|
94
|
+
if caps:
|
|
95
|
+
peak = max(caps.values())
|
|
96
|
+
count = sum(1 for value in caps.values() if value == peak)
|
|
97
|
+
if 0 < count < len(caps):
|
|
98
|
+
return count, "linux-cpu-capacity"
|
|
99
|
+
except OSError:
|
|
100
|
+
pass
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
import psutil
|
|
104
|
+
|
|
105
|
+
return (psutil.cpu_count(logical=False) or psutil.cpu_count() or 1), "physical-cores"
|
|
106
|
+
except Exception:
|
|
107
|
+
return 1, "fallback"
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _total_ram_gb() -> float:
|
|
111
|
+
"""Installed RAM, rounded to 1 GB — precise enough to compare, too coarse to identify."""
|
|
112
|
+
try:
|
|
113
|
+
import psutil
|
|
114
|
+
|
|
115
|
+
return round(psutil.virtual_memory().total / (1024**3))
|
|
116
|
+
except Exception:
|
|
117
|
+
return 0.0
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def slugify(text: str, *, max_length: int = 48) -> str:
|
|
121
|
+
"""Lowercase, hyphenated, filesystem-safe — used to name the card file."""
|
|
122
|
+
slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
|
|
123
|
+
return slug[:max_length].rstrip("-") or "unknown"
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass
|
|
127
|
+
class Fingerprint:
|
|
128
|
+
"""Non-identifying description of the machine and software stack."""
|
|
129
|
+
|
|
130
|
+
cpu: str
|
|
131
|
+
arch: str
|
|
132
|
+
physical_cores: int
|
|
133
|
+
logical_cores: int
|
|
134
|
+
ram_gb: float
|
|
135
|
+
os: str
|
|
136
|
+
os_release: str
|
|
137
|
+
python: str
|
|
138
|
+
onnxruntime: str
|
|
139
|
+
provider: str
|
|
140
|
+
intra_op_threads: int
|
|
141
|
+
thread_policy: str = "unknown"
|
|
142
|
+
|
|
143
|
+
@property
|
|
144
|
+
def slug(self) -> str:
|
|
145
|
+
return f"{slugify(self.cpu, max_length=40)}-{slugify(self.os, max_length=10)}"
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def fingerprint(
|
|
149
|
+
*, provider: str, intra_op_threads: int, thread_policy: str = "unknown"
|
|
150
|
+
) -> Fingerprint:
|
|
151
|
+
"""Collect the host description. The only function in the project that reads the machine."""
|
|
152
|
+
import onnxruntime as ort
|
|
153
|
+
import psutil
|
|
154
|
+
|
|
155
|
+
return Fingerprint(
|
|
156
|
+
cpu=_cpu_model(),
|
|
157
|
+
arch=platform.machine(),
|
|
158
|
+
physical_cores=psutil.cpu_count(logical=False) or 0,
|
|
159
|
+
logical_cores=psutil.cpu_count(logical=True) or 0,
|
|
160
|
+
ram_gb=_total_ram_gb(),
|
|
161
|
+
os=platform.system(),
|
|
162
|
+
os_release=platform.release(),
|
|
163
|
+
python=platform.python_version(),
|
|
164
|
+
onnxruntime=ort.__version__,
|
|
165
|
+
provider=provider,
|
|
166
|
+
intra_op_threads=intra_op_threads,
|
|
167
|
+
thread_policy=thread_policy,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
@dataclass
|
|
172
|
+
class PrecisionRow:
|
|
173
|
+
"""One measured (model, precision) combination."""
|
|
174
|
+
|
|
175
|
+
precision: str
|
|
176
|
+
size_mb: float
|
|
177
|
+
latency_s_mean: float
|
|
178
|
+
latency_s_std: float
|
|
179
|
+
tokens_per_second: float
|
|
180
|
+
peak_ram_mb: float
|
|
181
|
+
perplexity: float | None
|
|
182
|
+
generated_tokens: int
|
|
183
|
+
measured_runs: int
|
|
184
|
+
|
|
185
|
+
def as_dict(self) -> dict[str, Any]:
|
|
186
|
+
return asdict(self)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
@dataclass
|
|
190
|
+
class ResultCard:
|
|
191
|
+
"""A full submission: one machine, one model, every precision it measured."""
|
|
192
|
+
|
|
193
|
+
schema_version: int
|
|
194
|
+
model_id: str
|
|
195
|
+
revision: str
|
|
196
|
+
created_utc: str
|
|
197
|
+
tool_version: str
|
|
198
|
+
machine: Fingerprint
|
|
199
|
+
rows: list[PrecisionRow]
|
|
200
|
+
eval_windows: int
|
|
201
|
+
eval_corpus_sha256: str
|
|
202
|
+
warmup_runs: int
|
|
203
|
+
gen_tokens: int
|
|
204
|
+
prompt_sha256: str
|
|
205
|
+
notes: str = ""
|
|
206
|
+
submitted_by: str = ""
|
|
207
|
+
extra: dict[str, Any] = field(default_factory=dict)
|
|
208
|
+
|
|
209
|
+
def as_dict(self) -> dict[str, Any]:
|
|
210
|
+
payload = asdict(self)
|
|
211
|
+
payload["machine"] = asdict(self.machine)
|
|
212
|
+
payload["rows"] = [r.as_dict() for r in self.rows]
|
|
213
|
+
return payload
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def filename(self) -> str:
|
|
217
|
+
"""``<cpu>-<os>--<model>.json`` — stable, so a re-run updates its own card."""
|
|
218
|
+
return f"{self.machine.slug}--{slugify(self.model_id.split('/')[-1], max_length=32)}.json"
|
|
219
|
+
|
|
220
|
+
def speedup_vs(self, baseline: str = "fp32") -> dict[str, float]:
|
|
221
|
+
"""Throughput of each precision relative to the baseline precision."""
|
|
222
|
+
by_precision = {r.precision: r for r in self.rows}
|
|
223
|
+
base = by_precision.get(baseline)
|
|
224
|
+
if base is None or base.tokens_per_second <= 0:
|
|
225
|
+
return {}
|
|
226
|
+
return {
|
|
227
|
+
r.precision: round(r.tokens_per_second / base.tokens_per_second, 3)
|
|
228
|
+
for r in self.rows
|
|
229
|
+
if r.precision != baseline
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def now_utc_iso() -> str:
|
|
234
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|