quantcost 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
edgellm/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """quantcost: measure what quantization actually costs on your own hardware."""
2
+
3
+ from __future__ import annotations
4
+
5
+ __version__ = "0.2.0"
6
+
7
+ __all__ = ["__version__"]
edgellm/base.py ADDED
@@ -0,0 +1,44 @@
1
+ """Backend-agnostic interfaces, shared by the torch and the torch-free paths.
2
+
3
+ These used to live in :mod:`edgellm.runners`, which imports ``torch`` at module
4
+ level. The lightweight benchmark path (ONNX Runtime + NumPy, no torch, no
5
+ transformers) needs the same interface without paying for that import, so the
6
+ parts that carry no torch dependency live here and ``runners`` re-exports them.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from abc import ABC, abstractmethod
12
+ from dataclasses import asdict, dataclass
13
+
14
+ from edgellm.config import GenerationConfig
15
+
16
+
17
+ @dataclass
18
+ class GenerationResult:
19
+ """The text plus the timing/throughput numbers for one generation call."""
20
+
21
+ backend: str
22
+ prompt: str
23
+ text: str
24
+ prompt_tokens: int
25
+ generated_tokens: int
26
+ latency_s: float
27
+ tokens_per_second: float
28
+
29
+ def as_dict(self) -> dict:
30
+ return asdict(self)
31
+
32
+
33
+ class InferenceRunner(ABC):
34
+ """Common interface for all inference backends."""
35
+
36
+ name: str = "base"
37
+
38
+ @abstractmethod
39
+ def generate(self, prompt: str, generation: GenerationConfig) -> GenerationResult:
40
+ """Generate a completion for ``prompt`` and report timing."""
41
+
42
+ def close(self) -> None:
43
+ """Release whatever the backend holds (sessions, memory). Optional."""
44
+ return None
edgellm/benchmark.py ADDED
@@ -0,0 +1,226 @@
1
+ """Benchmark harness: latency, throughput, peak RAM, on-disk size, perplexity.
2
+
3
+ Every metric here is measured on the real machine. Nothing is estimated or
4
+ scaled. The harness is backend-agnostic: it drives any :class:`InferenceRunner`
5
+ and computes perplexity from any callable that returns logits, so PyTorch and
6
+ ONNX Runtime (FP32 or quantized) are measured identically.
7
+
8
+ Perplexity method (documented so results are reproducible): the eval text is a
9
+ slice of WikiText-2, tokenized once and split into non-overlapping windows of
10
+ ``eval_max_length`` tokens. For each window we compute the summed
11
+ next-token cross-entropy; perplexity is ``exp(total_nll / total_tokens)``.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import statistics
18
+ import threading
19
+ import time
20
+ from collections.abc import Callable
21
+ from dataclasses import asdict, dataclass
22
+ from pathlib import Path
23
+ from typing import TYPE_CHECKING
24
+
25
+ import psutil
26
+
27
+ from edgellm.base import InferenceRunner
28
+ from edgellm.config import BenchmarkConfig, GenerationConfig
29
+
30
+ if TYPE_CHECKING: # ``torch`` is only needed by PerplexityEvaluator (the heavy path).
31
+ import torch
32
+
33
+
34
+ @dataclass
35
+ class BenchmarkResult:
36
+ """All measured numbers for one (backend, precision) combination."""
37
+
38
+ backend: str
39
+ precision: str
40
+ device: str
41
+ model_id: str
42
+ size_mb: float | None
43
+ latency_s_mean: float
44
+ latency_s_std: float
45
+ tokens_per_second: float
46
+ peak_ram_mb: float
47
+ perplexity: float | None
48
+ generated_tokens: int
49
+ measured_runs: int
50
+
51
+ def as_dict(self) -> dict:
52
+ return asdict(self)
53
+
54
+
55
+ class _PeakRSSSampler:
56
+ """Context manager sampling process RSS on a background thread to find its peak."""
57
+
58
+ def __init__(self, interval_s: float = 0.02) -> None:
59
+ self.interval_s = interval_s
60
+ self.peak_bytes = 0
61
+ self._stop = threading.Event()
62
+ self._thread: threading.Thread | None = None
63
+ self._proc = psutil.Process()
64
+
65
+ def _run(self) -> None:
66
+ while not self._stop.is_set():
67
+ self.peak_bytes = max(self.peak_bytes, self._proc.memory_info().rss)
68
+ time.sleep(self.interval_s)
69
+
70
+ def __enter__(self) -> _PeakRSSSampler:
71
+ self.peak_bytes = self._proc.memory_info().rss
72
+ self._thread = threading.Thread(target=self._run, daemon=True)
73
+ self._thread.start()
74
+ return self
75
+
76
+ def __exit__(self, *exc) -> None:
77
+ self._stop.set()
78
+ if self._thread is not None:
79
+ self._thread.join()
80
+
81
+ @property
82
+ def peak_mb(self) -> float:
83
+ return self.peak_bytes / (1024 * 1024)
84
+
85
+
86
+ def measure_size_mb(path: Path, patterns: tuple[str, ...]) -> float:
87
+ """Sum the on-disk size (MB) of files under ``path`` matching any glob pattern."""
88
+ total = 0
89
+ for pattern in patterns:
90
+ for f in path.glob(pattern):
91
+ if f.is_file():
92
+ total += f.stat().st_size
93
+ return total / (1024 * 1024)
94
+
95
+
96
+ class PerplexityEvaluator:
97
+ """Compute perplexity on a WikiText slice from any logits-producing callable."""
98
+
99
+ def __init__(self, config: BenchmarkConfig) -> None:
100
+ self.config = config
101
+
102
+ def _load_text(self) -> str:
103
+ from datasets import load_dataset
104
+
105
+ ds = load_dataset(self.config.eval_dataset, self.config.eval_config, split="test")
106
+ lines = [t for t in ds["text"] if t.strip()]
107
+ return "\n\n".join(lines[: self.config.eval_num_samples])
108
+
109
+ def evaluate(
110
+ self,
111
+ forward_fn: Callable[[torch.Tensor, torch.Tensor], torch.Tensor],
112
+ tokenizer,
113
+ ) -> float:
114
+ """``forward_fn(input_ids, attention_mask) -> logits`` (all torch tensors)."""
115
+ import torch
116
+
117
+ text = self._load_text()
118
+ input_ids = tokenizer(text, return_tensors="pt").input_ids
119
+ max_len = self.config.eval_max_length
120
+
121
+ total_nll = torch.zeros((), dtype=torch.float64)
122
+ total_tokens = 0
123
+ for start in range(0, input_ids.shape[1], max_len):
124
+ window = input_ids[:, start : start + max_len]
125
+ if window.shape[1] < 2:
126
+ continue
127
+ attn = torch.ones_like(window)
128
+ logits = forward_fn(window, attn).float()
129
+ shift_logits = logits[:, :-1, :].reshape(-1, logits.size(-1))
130
+ shift_labels = window[:, 1:].reshape(-1).to(shift_logits.device)
131
+ nll = torch.nn.functional.cross_entropy(shift_logits, shift_labels, reduction="sum")
132
+ total_nll += nll.double().cpu()
133
+ total_tokens += int(shift_labels.numel())
134
+
135
+ return float(torch.exp(total_nll / total_tokens))
136
+
137
+
138
+ class BenchmarkHarness:
139
+ """Drive a runner through warmup + measured runs and collect all metrics."""
140
+
141
+ def __init__(self, config: BenchmarkConfig) -> None:
142
+ self.config = config
143
+
144
+ def _bench_generation_config(self, base: GenerationConfig) -> GenerationConfig:
145
+ """Deterministic, fixed-length decoding so runs are comparable."""
146
+ n = self.config.gen_tokens
147
+ return GenerationConfig(
148
+ max_new_tokens=n,
149
+ min_new_tokens=n, # force full length -> stable token count across backends
150
+ do_sample=False,
151
+ seed=base.seed,
152
+ )
153
+
154
+ def run(
155
+ self,
156
+ runner: InferenceRunner,
157
+ *,
158
+ precision: str,
159
+ device: str,
160
+ model_id: str,
161
+ generation: GenerationConfig,
162
+ size_mb: float | None = None,
163
+ perplexity: float | None = None,
164
+ ) -> BenchmarkResult:
165
+ gen = self._bench_generation_config(generation)
166
+ prompt = self.config.prompt
167
+
168
+ for _ in range(self.config.warmup_runs):
169
+ runner.generate(prompt, gen)
170
+
171
+ latencies: list[float] = []
172
+ throughputs: list[float] = []
173
+ generated_tokens = 0
174
+ with _PeakRSSSampler() as sampler:
175
+ for _ in range(self.config.measured_runs):
176
+ result = runner.generate(prompt, gen)
177
+ latencies.append(result.latency_s)
178
+ throughputs.append(result.tokens_per_second)
179
+ generated_tokens = result.generated_tokens
180
+
181
+ return BenchmarkResult(
182
+ backend=runner.name,
183
+ precision=precision,
184
+ device=device,
185
+ model_id=model_id,
186
+ size_mb=round(size_mb, 2) if size_mb is not None else None,
187
+ latency_s_mean=round(statistics.fmean(latencies), 4),
188
+ latency_s_std=round(statistics.pstdev(latencies), 4),
189
+ tokens_per_second=round(statistics.fmean(throughputs), 2),
190
+ peak_ram_mb=round(sampler.peak_mb, 1),
191
+ perplexity=round(perplexity, 3) if perplexity is not None else None,
192
+ generated_tokens=generated_tokens,
193
+ measured_runs=self.config.measured_runs,
194
+ )
195
+
196
+
197
+ def save_results(results: list[BenchmarkResult], path: Path) -> None:
198
+ """Merge results into a JSON file keyed by (backend, precision)."""
199
+ path.parent.mkdir(parents=True, exist_ok=True)
200
+ existing: dict[str, dict] = {}
201
+ if path.exists():
202
+ for row in json.loads(path.read_text()):
203
+ existing[f"{row['backend']}|{row['precision']}"] = row
204
+ for r in results:
205
+ existing[f"{r.backend}|{r.precision}"] = r.as_dict()
206
+ path.write_text(json.dumps(list(existing.values()), indent=2) + "\n")
207
+
208
+
209
+ def render_markdown(results_json: Path) -> str:
210
+ """Render the JSON results as a Markdown table (real numbers only)."""
211
+ rows = json.loads(results_json.read_text()) if results_json.exists() else []
212
+ header = (
213
+ "| Backend | Precision | Device | Size (MB) | Latency (s) | "
214
+ "Throughput (tok/s) | Peak RAM (MB) | Perplexity |\n"
215
+ "| --- | --- | --- | --- | --- | --- | --- | --- |\n"
216
+ )
217
+ lines = []
218
+ for r in rows:
219
+ size = f"{r['size_mb']}" if r.get("size_mb") is not None else "—"
220
+ ppl = f"{r['perplexity']}" if r.get("perplexity") is not None else "—"
221
+ lat = f"{r['latency_s_mean']} ± {r['latency_s_std']}"
222
+ lines.append(
223
+ f"| {r['backend']} | {r['precision']} | {r['device']} | {size} | "
224
+ f"{lat} | {r['tokens_per_second']} | {r['peak_ram_mb']} | {ppl} |"
225
+ )
226
+ return header + "\n".join(lines) + "\n"
edgellm/card.py ADDED
@@ -0,0 +1,234 @@
1
+ """The result card: what one machine measured, in a shape that can be compared.
2
+
3
+ A card is the unit a contributor submits. It pairs the measured rows with enough
4
+ hardware and software context to make them meaningful, and with the knobs that
5
+ would otherwise silently change the numbers (thread count, token budget, eval
6
+ window count, corpus hash).
7
+
8
+ **On privacy.** Cards get committed to a public repository, so the fingerprint is
9
+ deliberately narrow: CPU model string, architecture, core count, rounded RAM, OS
10
+ name and release, and version strings. It never collects hostname, username,
11
+ local paths, MAC address, IP, serial number or any machine ID. ``fingerprint()``
12
+ is the only thing that reads the host, so that promise is auditable in one place.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import platform
18
+ import re
19
+ import subprocess
20
+ from dataclasses import asdict, dataclass, field
21
+ from datetime import datetime, timezone
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ #: Bumped when the card layout changes in a way a validator must notice.
26
+ CARD_SCHEMA_VERSION = 2
27
+
28
+
29
+ def _cpu_model() -> str:
30
+ """A human-readable CPU model string, or 'unknown' if the OS will not say."""
31
+ system = platform.system()
32
+ try:
33
+ if system == "Darwin":
34
+ out = subprocess.run(
35
+ ["sysctl", "-n", "machdep.cpu.brand_string"],
36
+ capture_output=True,
37
+ text=True,
38
+ timeout=5,
39
+ )
40
+ if out.returncode == 0 and out.stdout.strip():
41
+ return out.stdout.strip()
42
+ elif system == "Linux":
43
+ with open("/proc/cpuinfo") as fh:
44
+ for line in fh:
45
+ if line.startswith(("model name", "Model")):
46
+ return line.split(":", 1)[1].strip()
47
+ except (OSError, subprocess.SubprocessError):
48
+ pass
49
+ return platform.processor() or platform.machine() or "unknown"
50
+
51
+
52
+ def performance_cores() -> tuple[int, str]:
53
+ """Count the machine's *fast* cores, and say how that was determined.
54
+
55
+ This matters far more than it looks. On a heterogeneous CPU — Apple Silicon,
56
+ ARM big.LITTLE, Intel hybrid — an ONNX Runtime parallel region ends on a
57
+ barrier, so one thread scheduled onto an efficiency core gates the whole
58
+ region. Measured on an M4 (4 performance + 6 efficiency), pinning all ten
59
+ physical cores cost int8 **66% of its throughput** (24.1 vs 40.0 tok/s) and
60
+ doubled run-to-run spread, while barely moving fp32: the quantized kernel is
61
+ compute-bound and waits on the slow thread, fp32 is bandwidth-bound and does
62
+ not. Defaulting to every physical core therefore understates quantized
63
+ performance and makes results unreproducible.
64
+
65
+ Returns ``(count, policy)`` so the card records which rule produced it.
66
+ """
67
+ system = platform.system()
68
+
69
+ if system == "Darwin":
70
+ # perflevel0 is the fastest tier; the key is absent on uniform chips.
71
+ try:
72
+ out = subprocess.run(
73
+ ["sysctl", "-n", "hw.perflevel0.physicalcpu"],
74
+ capture_output=True,
75
+ text=True,
76
+ timeout=5,
77
+ )
78
+ if out.returncode == 0 and out.stdout.strip().isdigit():
79
+ count = int(out.stdout.strip())
80
+ if count > 0:
81
+ return count, "macos-perflevel0"
82
+ except (OSError, subprocess.SubprocessError):
83
+ pass
84
+
85
+ elif system == "Linux":
86
+ # ARM big.LITTLE exposes a per-cpu capacity; the fast tier is the max.
87
+ try:
88
+ caps: dict[int, int] = {}
89
+ for path in Path("/sys/devices/system/cpu").glob("cpu[0-9]*/cpu_capacity"):
90
+ try:
91
+ caps[int(path.parent.name[3:])] = int(path.read_text().strip())
92
+ except (OSError, ValueError):
93
+ continue
94
+ if caps:
95
+ peak = max(caps.values())
96
+ count = sum(1 for value in caps.values() if value == peak)
97
+ if 0 < count < len(caps):
98
+ return count, "linux-cpu-capacity"
99
+ except OSError:
100
+ pass
101
+
102
+ try:
103
+ import psutil
104
+
105
+ return (psutil.cpu_count(logical=False) or psutil.cpu_count() or 1), "physical-cores"
106
+ except Exception:
107
+ return 1, "fallback"
108
+
109
+
110
+ def _total_ram_gb() -> float:
111
+ """Installed RAM, rounded to 1 GB — precise enough to compare, too coarse to identify."""
112
+ try:
113
+ import psutil
114
+
115
+ return round(psutil.virtual_memory().total / (1024**3))
116
+ except Exception:
117
+ return 0.0
118
+
119
+
120
+ def slugify(text: str, *, max_length: int = 48) -> str:
121
+ """Lowercase, hyphenated, filesystem-safe — used to name the card file."""
122
+ slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
123
+ return slug[:max_length].rstrip("-") or "unknown"
124
+
125
+
126
+ @dataclass
127
+ class Fingerprint:
128
+ """Non-identifying description of the machine and software stack."""
129
+
130
+ cpu: str
131
+ arch: str
132
+ physical_cores: int
133
+ logical_cores: int
134
+ ram_gb: float
135
+ os: str
136
+ os_release: str
137
+ python: str
138
+ onnxruntime: str
139
+ provider: str
140
+ intra_op_threads: int
141
+ thread_policy: str = "unknown"
142
+
143
+ @property
144
+ def slug(self) -> str:
145
+ return f"{slugify(self.cpu, max_length=40)}-{slugify(self.os, max_length=10)}"
146
+
147
+
148
+ def fingerprint(
149
+ *, provider: str, intra_op_threads: int, thread_policy: str = "unknown"
150
+ ) -> Fingerprint:
151
+ """Collect the host description. The only function in the project that reads the machine."""
152
+ import onnxruntime as ort
153
+ import psutil
154
+
155
+ return Fingerprint(
156
+ cpu=_cpu_model(),
157
+ arch=platform.machine(),
158
+ physical_cores=psutil.cpu_count(logical=False) or 0,
159
+ logical_cores=psutil.cpu_count(logical=True) or 0,
160
+ ram_gb=_total_ram_gb(),
161
+ os=platform.system(),
162
+ os_release=platform.release(),
163
+ python=platform.python_version(),
164
+ onnxruntime=ort.__version__,
165
+ provider=provider,
166
+ intra_op_threads=intra_op_threads,
167
+ thread_policy=thread_policy,
168
+ )
169
+
170
+
171
+ @dataclass
172
+ class PrecisionRow:
173
+ """One measured (model, precision) combination."""
174
+
175
+ precision: str
176
+ size_mb: float
177
+ latency_s_mean: float
178
+ latency_s_std: float
179
+ tokens_per_second: float
180
+ peak_ram_mb: float
181
+ perplexity: float | None
182
+ generated_tokens: int
183
+ measured_runs: int
184
+
185
+ def as_dict(self) -> dict[str, Any]:
186
+ return asdict(self)
187
+
188
+
189
+ @dataclass
190
+ class ResultCard:
191
+ """A full submission: one machine, one model, every precision it measured."""
192
+
193
+ schema_version: int
194
+ model_id: str
195
+ revision: str
196
+ created_utc: str
197
+ tool_version: str
198
+ machine: Fingerprint
199
+ rows: list[PrecisionRow]
200
+ eval_windows: int
201
+ eval_corpus_sha256: str
202
+ warmup_runs: int
203
+ gen_tokens: int
204
+ prompt_sha256: str
205
+ notes: str = ""
206
+ submitted_by: str = ""
207
+ extra: dict[str, Any] = field(default_factory=dict)
208
+
209
+ def as_dict(self) -> dict[str, Any]:
210
+ payload = asdict(self)
211
+ payload["machine"] = asdict(self.machine)
212
+ payload["rows"] = [r.as_dict() for r in self.rows]
213
+ return payload
214
+
215
+ @property
216
+ def filename(self) -> str:
217
+ """``<cpu>-<os>--<model>.json`` — stable, so a re-run updates its own card."""
218
+ return f"{self.machine.slug}--{slugify(self.model_id.split('/')[-1], max_length=32)}.json"
219
+
220
+ def speedup_vs(self, baseline: str = "fp32") -> dict[str, float]:
221
+ """Throughput of each precision relative to the baseline precision."""
222
+ by_precision = {r.precision: r for r in self.rows}
223
+ base = by_precision.get(baseline)
224
+ if base is None or base.tokens_per_second <= 0:
225
+ return {}
226
+ return {
227
+ r.precision: round(r.tokens_per_second / base.tokens_per_second, 3)
228
+ for r in self.rows
229
+ if r.precision != baseline
230
+ }
231
+
232
+
233
+ def now_utc_iso() -> str:
234
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")