quantcost 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgellm/__init__.py +7 -0
- edgellm/base.py +44 -0
- edgellm/benchmark.py +226 -0
- edgellm/card.py +234 -0
- edgellm/cli.py +370 -0
- edgellm/cli_bench.py +195 -0
- edgellm/config.py +120 -0
- edgellm/data/SOURCE.md +17 -0
- edgellm/data/eval_wikitext2.txt +205 -0
- edgellm/eval_lite.py +128 -0
- edgellm/export.py +65 -0
- edgellm/hub.py +125 -0
- edgellm/leaderboard.py +210 -0
- edgellm/models.py +91 -0
- edgellm/ort_lite.py +236 -0
- edgellm/quantize.py +134 -0
- edgellm/render.py +164 -0
- edgellm/report.py +53 -0
- edgellm/runners.py +170 -0
- edgellm/submit.py +225 -0
- edgellm/sweep.py +309 -0
- edgellm/validate.py +274 -0
- quantcost-0.2.0.dist-info/METADATA +265 -0
- quantcost-0.2.0.dist-info/RECORD +27 -0
- quantcost-0.2.0.dist-info/WHEEL +4 -0
- quantcost-0.2.0.dist-info/entry_points.txt +3 -0
- quantcost-0.2.0.dist-info/licenses/LICENSE +21 -0
edgellm/validate.py
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""Validate a submitted result card.
|
|
2
|
+
|
|
3
|
+
What this can and cannot do, stated plainly, because a leaderboard that overclaims
|
|
4
|
+
its own rigour is worse than one that is honest about being trust-based:
|
|
5
|
+
|
|
6
|
+
**It can** reject cards that are malformed, internally inconsistent, scored
|
|
7
|
+
against a different corpus, or run with settings that make them incomparable to
|
|
8
|
+
everyone else's — and it can flag numbers that are physically implausible.
|
|
9
|
+
|
|
10
|
+
**It cannot** prove a number came from real hardware. Nothing short of attested
|
|
11
|
+
execution could, and this is a benchmark run on strangers' laptops. The defence
|
|
12
|
+
is that the inputs are pinned (model revision, artifact bytes, corpus hash,
|
|
13
|
+
prompt hash, token budget), so a fabricated card has to be *internally
|
|
14
|
+
consistent* across every one of those to pass — and a reviewer can re-run the
|
|
15
|
+
exact configuration from the card itself. Cards are reviewed, not trusted.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from edgellm.card import CARD_SCHEMA_VERSION
|
|
25
|
+
from edgellm.eval_lite import EVAL_CORPUS_SHA256
|
|
26
|
+
from edgellm.hub import PRECISION_FILES
|
|
27
|
+
from edgellm.sweep import prompt_sha256
|
|
28
|
+
|
|
29
|
+
#: A card whose settings differ from these is still valid, but is not comparable
|
|
30
|
+
#: with the main leaderboard, so it is listed separately rather than ranked.
|
|
31
|
+
COMPARABLE_GEN_TOKENS = 32
|
|
32
|
+
COMPARABLE_EVAL_WINDOWS = 8
|
|
33
|
+
|
|
34
|
+
REQUIRED_TOP_LEVEL = {
|
|
35
|
+
"schema_version",
|
|
36
|
+
"model_id",
|
|
37
|
+
"revision",
|
|
38
|
+
"created_utc",
|
|
39
|
+
"tool_version",
|
|
40
|
+
"machine",
|
|
41
|
+
"rows",
|
|
42
|
+
"eval_windows",
|
|
43
|
+
"eval_corpus_sha256",
|
|
44
|
+
"warmup_runs",
|
|
45
|
+
"gen_tokens",
|
|
46
|
+
"prompt_sha256",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
REQUIRED_MACHINE = {
|
|
50
|
+
"cpu",
|
|
51
|
+
"arch",
|
|
52
|
+
"physical_cores",
|
|
53
|
+
"logical_cores",
|
|
54
|
+
"ram_gb",
|
|
55
|
+
"os",
|
|
56
|
+
"os_release",
|
|
57
|
+
"python",
|
|
58
|
+
"onnxruntime",
|
|
59
|
+
"provider",
|
|
60
|
+
"intra_op_threads",
|
|
61
|
+
"thread_policy",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
REQUIRED_ROW = {
|
|
65
|
+
"precision",
|
|
66
|
+
"size_mb",
|
|
67
|
+
"latency_s_mean",
|
|
68
|
+
"latency_s_std",
|
|
69
|
+
"tokens_per_second",
|
|
70
|
+
"peak_ram_mb",
|
|
71
|
+
"perplexity",
|
|
72
|
+
"generated_tokens",
|
|
73
|
+
"measured_runs",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
#: Loose physical bounds. Deliberately wide — the point is to catch a typo or a
|
|
77
|
+
#: fabricated order of magnitude, not to second-guess unusual but real hardware.
|
|
78
|
+
MAX_PLAUSIBLE_TOK_S = 100_000.0
|
|
79
|
+
MAX_PLAUSIBLE_PERPLEXITY = 1e6
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class ValidationReport:
|
|
84
|
+
path: Path
|
|
85
|
+
errors: list[str]
|
|
86
|
+
warnings: list[str]
|
|
87
|
+
comparable: bool
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def ok(self) -> bool:
|
|
91
|
+
return not self.errors
|
|
92
|
+
|
|
93
|
+
def render(self) -> str:
|
|
94
|
+
status = "PASS" if self.ok else "FAIL"
|
|
95
|
+
flag = "" if self.comparable else " (not comparable — will be listed, not ranked)"
|
|
96
|
+
lines = [f"[{status}] {self.path.name}{flag}"]
|
|
97
|
+
lines += [f" error: {e}" for e in self.errors]
|
|
98
|
+
lines += [f" warning: {w}" for w in self.warnings]
|
|
99
|
+
return "\n".join(lines)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def validate_card(path: Path) -> ValidationReport:
|
|
103
|
+
"""Check one card file. Never raises for bad content — it reports."""
|
|
104
|
+
errors: list[str] = []
|
|
105
|
+
warnings: list[str] = []
|
|
106
|
+
|
|
107
|
+
try:
|
|
108
|
+
card = json.loads(path.read_text())
|
|
109
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
110
|
+
return ValidationReport(path, [f"unreadable JSON: {exc}"], [], False)
|
|
111
|
+
|
|
112
|
+
if not isinstance(card, dict):
|
|
113
|
+
return ValidationReport(path, ["top level must be a JSON object"], [], False)
|
|
114
|
+
|
|
115
|
+
missing = REQUIRED_TOP_LEVEL - set(card)
|
|
116
|
+
if missing:
|
|
117
|
+
errors.append(f"missing field(s): {', '.join(sorted(missing))}")
|
|
118
|
+
|
|
119
|
+
if card.get("schema_version") != CARD_SCHEMA_VERSION:
|
|
120
|
+
errors.append(
|
|
121
|
+
f"schema_version is {card.get('schema_version')!r}, expected {CARD_SCHEMA_VERSION}. "
|
|
122
|
+
"Re-run with the current quantcost."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
machine = card.get("machine")
|
|
126
|
+
if not isinstance(machine, dict):
|
|
127
|
+
errors.append("'machine' must be an object")
|
|
128
|
+
else:
|
|
129
|
+
missing_machine = REQUIRED_MACHINE - set(machine)
|
|
130
|
+
if missing_machine:
|
|
131
|
+
errors.append(f"machine missing: {', '.join(sorted(missing_machine))}")
|
|
132
|
+
if machine.get("intra_op_threads") in (0, None):
|
|
133
|
+
errors.append(
|
|
134
|
+
"machine.intra_op_threads is 0/null — thread count must be pinned, "
|
|
135
|
+
"or throughput is not comparable. Re-run with the current version."
|
|
136
|
+
)
|
|
137
|
+
for leak in ("hostname", "user", "username", "mac", "ip", "serial"):
|
|
138
|
+
if leak in machine:
|
|
139
|
+
errors.append(f"machine.{leak} must not be submitted (identifying information)")
|
|
140
|
+
|
|
141
|
+
rows = card.get("rows")
|
|
142
|
+
if not isinstance(rows, list) or not rows:
|
|
143
|
+
errors.append("'rows' must be a non-empty list")
|
|
144
|
+
rows = []
|
|
145
|
+
|
|
146
|
+
seen: set[str] = set()
|
|
147
|
+
scores_perplexity = False
|
|
148
|
+
for index, row in enumerate(rows):
|
|
149
|
+
where = f"rows[{index}]"
|
|
150
|
+
if not isinstance(row, dict):
|
|
151
|
+
errors.append(f"{where} must be an object")
|
|
152
|
+
continue
|
|
153
|
+
|
|
154
|
+
missing_row = REQUIRED_ROW - set(row)
|
|
155
|
+
if missing_row:
|
|
156
|
+
errors.append(f"{where} missing: {', '.join(sorted(missing_row))}")
|
|
157
|
+
continue
|
|
158
|
+
|
|
159
|
+
precision = row["precision"]
|
|
160
|
+
if precision not in PRECISION_FILES:
|
|
161
|
+
errors.append(f"{where}: unknown precision {precision!r}")
|
|
162
|
+
if precision in seen:
|
|
163
|
+
errors.append(f"{where}: duplicate precision {precision!r}")
|
|
164
|
+
seen.add(precision)
|
|
165
|
+
|
|
166
|
+
for field in ("size_mb", "latency_s_mean", "tokens_per_second", "peak_ram_mb"):
|
|
167
|
+
value = row[field]
|
|
168
|
+
if not isinstance(value, (int, float)) or value <= 0:
|
|
169
|
+
errors.append(f"{where}.{field} must be a positive number, got {value!r}")
|
|
170
|
+
|
|
171
|
+
if isinstance(row.get("tokens_per_second"), (int, float)):
|
|
172
|
+
if row["tokens_per_second"] > MAX_PLAUSIBLE_TOK_S:
|
|
173
|
+
errors.append(
|
|
174
|
+
f"{where}.tokens_per_second = {row['tokens_per_second']} is not physically "
|
|
175
|
+
f"plausible (cap {MAX_PLAUSIBLE_TOK_S:.0f})"
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
# Throughput must agree with latency and the token count it claims.
|
|
179
|
+
lat, tps, gen = row["latency_s_mean"], row["tokens_per_second"], row["generated_tokens"]
|
|
180
|
+
if all(isinstance(v, (int, float)) and v > 0 for v in (lat, tps, gen)):
|
|
181
|
+
implied = gen / lat
|
|
182
|
+
if abs(implied - tps) / max(implied, tps) > 0.02:
|
|
183
|
+
errors.append(
|
|
184
|
+
f"{where}: tokens_per_second ({tps}) disagrees with "
|
|
185
|
+
f"generated_tokens/latency_s_mean ({implied:.2f}). "
|
|
186
|
+
"These are derived from the same measurement and must match."
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
if row.get("generated_tokens") != card.get("gen_tokens"):
|
|
190
|
+
errors.append(
|
|
191
|
+
f"{where}.generated_tokens ({row.get('generated_tokens')}) does not match "
|
|
192
|
+
f"the card's gen_tokens ({card.get('gen_tokens')})"
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
if row.get("perplexity") is not None:
|
|
196
|
+
scores_perplexity = True
|
|
197
|
+
ppl = row["perplexity"]
|
|
198
|
+
if not isinstance(ppl, (int, float)) or ppl <= 1.0:
|
|
199
|
+
errors.append(f"{where}.perplexity must be > 1.0, got {ppl!r}")
|
|
200
|
+
elif ppl > MAX_PLAUSIBLE_PERPLEXITY:
|
|
201
|
+
errors.append(f"{where}.perplexity = {ppl} is implausible")
|
|
202
|
+
|
|
203
|
+
if isinstance(row.get("latency_s_std"), (int, float)) and isinstance(lat, (int, float)):
|
|
204
|
+
if lat > 0 and row["latency_s_std"] > lat * 0.5:
|
|
205
|
+
warnings.append(
|
|
206
|
+
f"{where}: latency varied by more than 50% of the mean — the machine was "
|
|
207
|
+
"probably busy. Consider re-running on an idle system."
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
if scores_perplexity and card.get("eval_corpus_sha256") != EVAL_CORPUS_SHA256:
|
|
211
|
+
errors.append(
|
|
212
|
+
f"eval_corpus_sha256 {card.get('eval_corpus_sha256')!r} does not match the bundled "
|
|
213
|
+
f"corpus {EVAL_CORPUS_SHA256!r}. Perplexity scored against different text is not "
|
|
214
|
+
"comparable."
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
if card.get("prompt_sha256") != prompt_sha256():
|
|
218
|
+
errors.append(
|
|
219
|
+
"prompt_sha256 does not match the pinned benchmark prompt. Latency depends on "
|
|
220
|
+
"prompt length, so a different prompt is not comparable."
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
if "fp32" not in seen and rows:
|
|
224
|
+
warnings.append("no fp32 baseline in this card, so speedups cannot be computed for it")
|
|
225
|
+
|
|
226
|
+
comparable = (
|
|
227
|
+
not errors
|
|
228
|
+
and card.get("gen_tokens") == COMPARABLE_GEN_TOKENS
|
|
229
|
+
and (not scores_perplexity or card.get("eval_windows") == COMPARABLE_EVAL_WINDOWS)
|
|
230
|
+
)
|
|
231
|
+
if not comparable and not errors:
|
|
232
|
+
warnings.append(
|
|
233
|
+
f"settings differ from the comparable defaults "
|
|
234
|
+
f"(gen_tokens={COMPARABLE_GEN_TOKENS}, eval_windows={COMPARABLE_EVAL_WINDOWS}); "
|
|
235
|
+
"this card will be listed but not ranked"
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
return ValidationReport(path, errors, warnings, comparable)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def validate_all(paths: list[Path]) -> list[ValidationReport]:
|
|
242
|
+
return [validate_card(p) for p in sorted(paths)]
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _main(argv: list[str] | None = None) -> int:
|
|
246
|
+
"""``python -m edgellm.validate results/community/*.json`` — the CI entry point."""
|
|
247
|
+
import argparse
|
|
248
|
+
|
|
249
|
+
parser = argparse.ArgumentParser(prog="python -m edgellm.validate")
|
|
250
|
+
parser.add_argument("paths", nargs="+", type=Path)
|
|
251
|
+
parser.add_argument("--strict", action="store_true", help="Treat warnings as failures as well.")
|
|
252
|
+
args = parser.parse_args(argv)
|
|
253
|
+
|
|
254
|
+
files: list[Path] = []
|
|
255
|
+
for path in args.paths:
|
|
256
|
+
files.extend(sorted(path.glob("*.json")) if path.is_dir() else [path])
|
|
257
|
+
|
|
258
|
+
reports = validate_all(files)
|
|
259
|
+
for report in reports:
|
|
260
|
+
print(report.render())
|
|
261
|
+
|
|
262
|
+
failed = [r for r in reports if not r.ok]
|
|
263
|
+
warned = [r for r in reports if r.warnings]
|
|
264
|
+
print(
|
|
265
|
+
f"\n{len(reports) - len(failed)}/{len(reports)} card(s) passed"
|
|
266
|
+
+ (f", {len(warned)} with warnings" if warned else "")
|
|
267
|
+
)
|
|
268
|
+
if failed:
|
|
269
|
+
return 1
|
|
270
|
+
return 1 if (args.strict and warned) else 0
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
if __name__ == "__main__":
|
|
274
|
+
raise SystemExit(_main())
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: quantcost
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Measure what quantization actually costs on your own hardware: speed, size, memory and quality.
|
|
5
|
+
Project-URL: Homepage, https://github.com/vijay-kapse/EdgeLLM
|
|
6
|
+
Project-URL: Leaderboard, https://github.com/vijay-kapse/EdgeLLM/blob/main/results/LEADERBOARD.md
|
|
7
|
+
Project-URL: Issues, https://github.com/vijay-kapse/EdgeLLM/issues
|
|
8
|
+
Author: Vijay Kapse
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: benchmark,edge,inference,int4,int8,llm,npu,onnx,onnxruntime,perplexity,qualcomm,quantization
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Topic :: System :: Benchmark
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: huggingface-hub>=0.23
|
|
26
|
+
Requires-Dist: numpy>=1.24
|
|
27
|
+
Requires-Dist: onnxruntime>=1.18
|
|
28
|
+
Requires-Dist: psutil>=5.9
|
|
29
|
+
Requires-Dist: pyyaml>=6.0
|
|
30
|
+
Requires-Dist: tokenizers>=0.19
|
|
31
|
+
Requires-Dist: typer>=0.12
|
|
32
|
+
Provides-Extra: aihub
|
|
33
|
+
Requires-Dist: qai-hub>=0.9; extra == 'aihub'
|
|
34
|
+
Provides-Extra: dev
|
|
35
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
36
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
37
|
+
Provides-Extra: quantize
|
|
38
|
+
Requires-Dist: datasets>=2.19; extra == 'quantize'
|
|
39
|
+
Requires-Dist: onnx-ir>=0.2; extra == 'quantize'
|
|
40
|
+
Requires-Dist: onnx>=1.16; extra == 'quantize'
|
|
41
|
+
Requires-Dist: optimum[onnxruntime]>=1.20; extra == 'quantize'
|
|
42
|
+
Requires-Dist: torch>=2.2; extra == 'quantize'
|
|
43
|
+
Requires-Dist: transformers>=4.44; extra == 'quantize'
|
|
44
|
+
Provides-Extra: report
|
|
45
|
+
Requires-Dist: matplotlib>=3.8; extra == 'report'
|
|
46
|
+
Description-Content-Type: text/markdown
|
|
47
|
+
|
|
48
|
+
# quantcost
|
|
49
|
+
|
|
50
|
+
**Find out what quantization actually costs you — on your machine, not someone else's.**
|
|
51
|
+
|
|
52
|
+
[](https://github.com/vijay-kapse/EdgeLLM/actions/workflows/ci.yml)
|
|
53
|
+
[](https://pypi.org/project/quantcost/)
|
|
54
|
+
[](https://pypi.org/project/quantcost/)
|
|
55
|
+
[](LICENSE)
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install quantcost
|
|
59
|
+
quantcost run
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
That's it. Two to three minutes later you get speed, size, memory and *quality*
|
|
63
|
+
for fp32 / int8 / int4 on your own CPU, with a plain-language verdict.
|
|
64
|
+
|
|
65
|
+
No PyTorch. No GPU. No quantizing anything yourself — the pre-quantized ONNX
|
|
66
|
+
already published on the Hugging Face Hub gets downloaded and measured, so the
|
|
67
|
+
whole install is ONNX Runtime, NumPy and a tokenizer, and finishes in about ten
|
|
68
|
+
seconds.
|
|
69
|
+
|
|
70
|
+
---
|
|
71
|
+
|
|
72
|
+
## What quantization actually costs
|
|
73
|
+
|
|
74
|
+
"Quantize it, it'll be smaller and faster" is half right. Two models, one Apple
|
|
75
|
+
M4, measured with `quantcost run` at its defaults:
|
|
76
|
+
|
|
77
|
+
**SmolLM2-135M-Instruct**
|
|
78
|
+
|
|
79
|
+
| Precision | Size (MB) | tok/s | Speedup | Peak RAM (MB) | Perplexity | PPL change |
|
|
80
|
+
| --- | --- | --- | --- | --- | --- | --- |
|
|
81
|
+
| fp32 | 515 | 83.65 | baseline | 1360 | 23.07 | baseline |
|
|
82
|
+
| int8 | 131 | 39.36 | **0.47x** | 1002 | 25.02 | +8.5% |
|
|
83
|
+
| q4 | 174 | 57.40 | **0.69x** | 906 | 28.35 | +22.9% |
|
|
84
|
+
|
|
85
|
+
**Qwen2.5-0.5B-Instruct**
|
|
86
|
+
|
|
87
|
+
| Precision | Size (MB) | tok/s | Speedup | Peak RAM (MB) | Perplexity | PPL change |
|
|
88
|
+
| --- | --- | --- | --- | --- | --- | --- |
|
|
89
|
+
| fp32 | 1901 | 21.49 | baseline | 3282 | 19.26 | baseline |
|
|
90
|
+
| int8 | 488 | 10.17 | **0.47x** | 3052 | 21.19 | +10.0% |
|
|
91
|
+
| q4 | 750 | 18.37 | **0.85x** | 2303 | 22.28 | +15.7% |
|
|
92
|
+
|
|
93
|
+
Smaller: reliably, 3–4x. Less memory: yes, 25–30% off the peak. Faster: **no** —
|
|
94
|
+
int8 generates tokens at 0.47x the fp32 rate on *both* models. That the figure
|
|
95
|
+
lands on 0.47 twice, across a 135M and a 500M model, is what makes it look like
|
|
96
|
+
a property of the runtime rather than an accident of one benchmark.
|
|
97
|
+
|
|
98
|
+
### Why, and why it is not "quantized maths is slow"
|
|
99
|
+
|
|
100
|
+
Run the same artifacts over a 512-token prefill instead of one token at a time:
|
|
101
|
+
|
|
102
|
+
| | prefill (512 tok) | decode (1 tok/step) |
|
|
103
|
+
| --- | --- | --- |
|
|
104
|
+
| SmolLM2 int8 | 0.79x | 0.43x |
|
|
105
|
+
| SmolLM2 q4 | 0.31x | 0.68x |
|
|
106
|
+
| Qwen int8 | 0.97x | 0.59x |
|
|
107
|
+
| Qwen q4 | 0.36x | 1.02x |
|
|
108
|
+
|
|
109
|
+
The int8 penalty roughly **halves** once there is a batch to amortise over
|
|
110
|
+
(0.43 → 0.79, 0.59 → 0.97). That is the signature of a fixed per-step cost:
|
|
111
|
+
ONNX Runtime unpacks the weights back to float inside each matmul, and a single
|
|
112
|
+
token cannot amortise unpacking a whole weight matrix. fp32 decode is
|
|
113
|
+
bandwidth-bound on *reading* weights; int8 decode is compute-bound on
|
|
114
|
+
*unpacking* them, so it does strictly more work despite being 4x smaller.
|
|
115
|
+
|
|
116
|
+
Note also that **int8 and q4 invert**: q4 is the better choice for decode
|
|
117
|
+
(0.68x / 1.02x vs int8's 0.43x / 0.59x) and much the worse for prefill
|
|
118
|
+
(0.31x / 0.36x vs 0.79x / 0.97x). They use different kernels —
|
|
119
|
+
`MatMulNBits` for q4, dynamic-quantize + `MatMulInteger` for int8 — with
|
|
120
|
+
opposite strengths. Which format is right depends on whether your workload is
|
|
121
|
+
prompt-heavy or generation-heavy.
|
|
122
|
+
|
|
123
|
+
### One number that is easy to get wrong
|
|
124
|
+
|
|
125
|
+
Pin threads to your machine's **performance** cores, not every physical core.
|
|
126
|
+
On this M4 (4 performance + 6 efficiency) pinning all ten cost int8 66% of its
|
|
127
|
+
throughput — 24.1 vs 40.0 tok/s — and doubled run-to-run spread, because an ONNX
|
|
128
|
+
Runtime parallel region ends on a barrier and one thread on an efficiency core
|
|
129
|
+
gates the whole thing. `quantcost` does this by default and records how it
|
|
130
|
+
decided in every card. It is the single easiest way to publish a wrong number,
|
|
131
|
+
and it is how the first draft of this README got the figures above wrong.
|
|
132
|
+
|
|
133
|
+
So the honest answer to "should I quantize?" is **measure it on your hardware**,
|
|
134
|
+
which is why this is a tool rather than a blog post. On this machine you quantize
|
|
135
|
+
to fit in memory, not to go faster. On yours it may differ — that is the thing
|
|
136
|
+
worth finding out.
|
|
137
|
+
|
|
138
|
+
**[See what other machines measured →](results/LEADERBOARD.md)**
|
|
139
|
+
|
|
140
|
+
## Add your machine
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
quantcost run
|
|
144
|
+
quantcost submit
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
`submit` validates your result, forks this repo, commits the card and opens the
|
|
148
|
+
pull request for you. No GitHub CLI? It prints a prefilled link instead.
|
|
149
|
+
|
|
150
|
+
Unusual hardware is the most valuable kind: Raspberry Pi, old ThinkPads,
|
|
151
|
+
Snapdragon laptops, bare-metal ARM. The interesting question is not who has the
|
|
152
|
+
fastest laptop — it is *where quantization pays off and where it backfires*, and
|
|
153
|
+
that only becomes visible across many real machines.
|
|
154
|
+
|
|
155
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for what makes a good submission and
|
|
156
|
+
exactly what a card contains (no hostname, no username, no paths — the
|
|
157
|
+
fingerprint is one auditable function).
|
|
158
|
+
|
|
159
|
+
## Usage
|
|
160
|
+
|
|
161
|
+
```bash
|
|
162
|
+
# A different model — anything with ONNX on the Hub works
|
|
163
|
+
quantcost run --model onnx-community/Qwen2.5-0.5B-Instruct
|
|
164
|
+
|
|
165
|
+
# Which precisions does a model actually publish?
|
|
166
|
+
quantcost models --model onnx-community/Qwen2.5-0.5B-Instruct
|
|
167
|
+
|
|
168
|
+
# Speed and size only; skips the perplexity pass and is much faster
|
|
169
|
+
quantcost run --precisions fp32,int8 --skip-perplexity
|
|
170
|
+
|
|
171
|
+
# Pin threads to compare against a specific configuration
|
|
172
|
+
quantcost run --threads 4
|
|
173
|
+
|
|
174
|
+
# Rebuild the leaderboard from every submitted card
|
|
175
|
+
quantcost leaderboard
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Any Hub repo following the `onnx-community` / `transformers.js` layout works
|
|
179
|
+
unchanged — that is thousands of models, including everything under
|
|
180
|
+
[onnx-community](https://huggingface.co/onnx-community).
|
|
181
|
+
|
|
182
|
+
## How the numbers are produced
|
|
183
|
+
|
|
184
|
+
The headline claim of this project is that its numbers are real, so the
|
|
185
|
+
methodology is worth stating plainly.
|
|
186
|
+
|
|
187
|
+
**Every precision is measured in its own subprocess.** Peak RSS is a per-process
|
|
188
|
+
high-water mark and an ONNX Runtime session does not return all of its arenas
|
|
189
|
+
when dropped, so measuring several precisions in one process reports the *union*
|
|
190
|
+
of their footprints and blames whichever ran last. Isolation is the only way the
|
|
191
|
+
peak-RAM column means what it says.
|
|
192
|
+
|
|
193
|
+
**The file cache is warmed before anything is timed.** ONNX Runtime mmaps
|
|
194
|
+
weights and faults them in lazily. Measured cold — straight after the download —
|
|
195
|
+
an fp32 baseline came in at 17 tok/s; warm, the same machine and build measured
|
|
196
|
+
64. A 4x error on the number every other row is divided by is not a rounding
|
|
197
|
+
detail, so the model file is read through once before the session is built.
|
|
198
|
+
|
|
199
|
+
**Throughput is total tokens over total time**, derived from the same mean
|
|
200
|
+
latency that is reported, never the average of each run's own rate. Those are
|
|
201
|
+
different statistics (ratio of means vs. mean of ratios) and they disagree by a
|
|
202
|
+
few percent, which would make two columns of the same card contradict each other.
|
|
203
|
+
|
|
204
|
+
**Decoding is greedy and fixed-length.** No sampling, no early EOS stop, so every
|
|
205
|
+
precision does exactly the same amount of work and throughput does not depend on
|
|
206
|
+
the RNG.
|
|
207
|
+
|
|
208
|
+
**Perplexity is scored against a corpus pinned inside the package** — a fixed
|
|
209
|
+
65,342-byte slice of WikiText-2, hash-checked on load and recorded in every card.
|
|
210
|
+
A leaderboard where machines score different text is not a leaderboard, and this
|
|
211
|
+
also removes a heavy `datasets` dependency. Method: non-overlapping 512-token
|
|
212
|
+
windows, summed next-token cross-entropy, `exp(total_nll / total_tokens)`. The
|
|
213
|
+
NumPy implementation is tested against the uniform-distribution case, where the
|
|
214
|
+
right answer is analytically the vocabulary size, and agrees with PyTorch's
|
|
215
|
+
`cross_entropy` to 4e-07 relative.
|
|
216
|
+
|
|
217
|
+
**Thread count is always pinned** to physical cores (SMT siblings contend for the
|
|
218
|
+
same vector units) and recorded, because leaving it at ORT's default stores "0 —
|
|
219
|
+
decide for me" and makes two very different runs look identical.
|
|
220
|
+
|
|
221
|
+
**Instability is reported, not hidden.** If run-to-run latency varies by more
|
|
222
|
+
than 15%, the report says the machine was too busy and asks you to re-run.
|
|
223
|
+
|
|
224
|
+
### What validation can and cannot do
|
|
225
|
+
|
|
226
|
+
CI rejects cards that are malformed, internally inconsistent, scored against a
|
|
227
|
+
modified corpus, run with an unpinned thread count, or carrying identifying
|
|
228
|
+
information. It **cannot** prove a number came from real silicon — nothing short
|
|
229
|
+
of attested execution could. The defence is that every input is pinned, so a
|
|
230
|
+
fabricated card must be self-consistent across all of them, and anyone can re-run
|
|
231
|
+
the exact configuration recorded in the card. Cards are reviewed, not trusted.
|
|
232
|
+
|
|
233
|
+
## Authoring quantized artifacts yourself
|
|
234
|
+
|
|
235
|
+
Everything above *consumes* pre-quantized ONNX. The original project also
|
|
236
|
+
*produces* it — export, quantize, a C++ inference harness, a SIMD INT8 GEMM
|
|
237
|
+
kernel, an Android app and a Qualcomm Snapdragon NPU path. That path needs the
|
|
238
|
+
heavy dependencies:
|
|
239
|
+
|
|
240
|
+
```bash
|
|
241
|
+
pip install -e ".[quantize]"
|
|
242
|
+
edgellm export --help
|
|
243
|
+
edgellm quantize --help
|
|
244
|
+
edgellm bench --help # the original torch + Optimum harness
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
See [docs/AUTHORING.md](docs/AUTHORING.md) for the full pipeline, the C++ harness
|
|
248
|
+
and the Snapdragon notes.
|
|
249
|
+
|
|
250
|
+
## Development
|
|
251
|
+
|
|
252
|
+
```bash
|
|
253
|
+
pip install -e ".[dev]"
|
|
254
|
+
ruff check . && ruff format --check . && pytest
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
CI asserts that importing the CLI pulls in no heavy dependency. If you need
|
|
258
|
+
torch, import it inside the function that uses it — the fast install is what
|
|
259
|
+
makes a one-command benchmark viable for someone who has never heard of this
|
|
260
|
+
project.
|
|
261
|
+
|
|
262
|
+
## License
|
|
263
|
+
|
|
264
|
+
MIT — see [LICENSE](LICENSE). The bundled eval corpus is WikiText-2, CC BY-SA 4.0;
|
|
265
|
+
see [edgellm/data/SOURCE.md](edgellm/data/SOURCE.md).
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
edgellm/__init__.py,sha256=WEF4Psnkp-LKq1B5VNPpWb7gcuv4Whh_Gjk9cd9R5IU,166
|
|
2
|
+
edgellm/base.py,sha256=hMVqPbnZFV9rymRfX6CtKQ4UPWGxZb_zEqVgqaIU7fs,1294
|
|
3
|
+
edgellm/benchmark.py,sha256=BYzMzcTv3gcTraajBBAW2JAzjDueCe_7BoXbctFM2V4,8181
|
|
4
|
+
edgellm/card.py,sha256=ev-3LJHimFf14OyKC1_ARZYSEfb7ozJJqBFa3Qd4lYM,7972
|
|
5
|
+
edgellm/cli.py,sha256=u131oDKUgm1ucZeS2LRU923qyaIIHLw_gHGeGNGI_G4,13706
|
|
6
|
+
edgellm/cli_bench.py,sha256=grLG-iRsBnMD8uSeTrzIBZMvUZYQdB2deSpkjzxGTtU,7849
|
|
7
|
+
edgellm/config.py,sha256=GuuloueR0y5_8drBYyOj-oLj2PN5D6Z_g_wiJ9fHVjc,3570
|
|
8
|
+
edgellm/eval_lite.py,sha256=eNM_EGMFr7O10l2xtYbkFTMGnPsARfK_aJeovpMzI8I,4806
|
|
9
|
+
edgellm/export.py,sha256=EaI3gWNJqL0KL1-Eso-FgcMmdqlN0oN1SWpRR8JfDp0,2344
|
|
10
|
+
edgellm/hub.py,sha256=djTbeoUryiRqqoKaFtMtke6VWN9UDMCu5xU-PY7ixLk,4598
|
|
11
|
+
edgellm/leaderboard.py,sha256=9zPUnarUzzGM8rciDu54r-2FPgpgf9NTpvxYTp92vAY,8443
|
|
12
|
+
edgellm/models.py,sha256=JmmQRjR3t1sNl8Kz-SX9NLf5bX1iJdbWNLJKusTDXjE,2745
|
|
13
|
+
edgellm/ort_lite.py,sha256=vw71IqlIUQ8AfKA2gtVbWVFXdHPEqcD79n2QHOxNITU,9564
|
|
14
|
+
edgellm/quantize.py,sha256=gxC0YxxQaXX_c1uq9h1HT94EgMyQvofk9nFWkmPZ7a0,5075
|
|
15
|
+
edgellm/render.py,sha256=Nu6iUBnhKueXEmiSk0eVs7hRPl_txAGFlnv5NWxklis,6998
|
|
16
|
+
edgellm/report.py,sha256=YSp5_3xAeWgKUCIz0xeZYxBKobXZsRu3RKlPDNok0j4,1807
|
|
17
|
+
edgellm/runners.py,sha256=D7ICj5PIsWj6ExCmyllon7VyNgrT92aCbjWzVLRMK7g,6227
|
|
18
|
+
edgellm/submit.py,sha256=dSmzvIseg3hCckxyvzANJBRNuwcnGDIG9YX1Grv1jDs,8118
|
|
19
|
+
edgellm/sweep.py,sha256=kaBCIsKwm_xM2_i2kYLoIp9O0JO9do6SiYLy2zzpCfI,11540
|
|
20
|
+
edgellm/validate.py,sha256=7GYm7gYhn15bxgOVGcEG10HujaVtv5PP0CWpo7TIkgU,10068
|
|
21
|
+
edgellm/data/SOURCE.md,sha256=9fv2dBtucKis_UjyYF2yZh_puSqyc9JDHWD0iHez-bI,919
|
|
22
|
+
edgellm/data/eval_wikitext2.txt,sha256=NKKkP89JIX96QhM_lhkw2MgwrH_jzHkn9CP0Qwg_l78,65342
|
|
23
|
+
quantcost-0.2.0.dist-info/METADATA,sha256=ND-s2JTWgdrg0WQmtTNXRsJTz9-qQj9WR1cTo8qdbow,11582
|
|
24
|
+
quantcost-0.2.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
25
|
+
quantcost-0.2.0.dist-info/entry_points.txt,sha256=1Hv957MAOBjb0MT3nJdko_Cqhrtfsxxxf7Oax4DQNBs,74
|
|
26
|
+
quantcost-0.2.0.dist-info/licenses/LICENSE,sha256=jBAq79ONijdbPnqSEEL9fDEzsZp8TCOxNdvmMhT6EzU,1068
|
|
27
|
+
quantcost-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vijay Kapse
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|