quantcost 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
edgellm/validate.py ADDED
@@ -0,0 +1,274 @@
1
+ """Validate a submitted result card.
2
+
3
+ What this can and cannot do, stated plainly, because a leaderboard that overclaims
4
+ its own rigour is worse than one that is honest about being trust-based:
5
+
6
+ **It can** reject cards that are malformed, internally inconsistent, scored
7
+ against a different corpus, or run with settings that make them incomparable to
8
+ everyone else's — and it can flag numbers that are physically implausible.
9
+
10
+ **It cannot** prove a number came from real hardware. Nothing short of attested
11
+ execution could, and this is a benchmark run on strangers' laptops. The defence
12
+ is that the inputs are pinned (model revision, artifact bytes, corpus hash,
13
+ prompt hash, token budget), so a fabricated card has to be *internally
14
+ consistent* across every one of those to pass — and a reviewer can re-run the
15
+ exact configuration from the card itself. Cards are reviewed, not trusted.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ from dataclasses import dataclass
22
+ from pathlib import Path
23
+
24
+ from edgellm.card import CARD_SCHEMA_VERSION
25
+ from edgellm.eval_lite import EVAL_CORPUS_SHA256
26
+ from edgellm.hub import PRECISION_FILES
27
+ from edgellm.sweep import prompt_sha256
28
+
29
+ #: A card whose settings differ from these is still valid, but is not comparable
30
+ #: with the main leaderboard, so it is listed separately rather than ranked.
31
+ COMPARABLE_GEN_TOKENS = 32
32
+ COMPARABLE_EVAL_WINDOWS = 8
33
+
34
+ REQUIRED_TOP_LEVEL = {
35
+ "schema_version",
36
+ "model_id",
37
+ "revision",
38
+ "created_utc",
39
+ "tool_version",
40
+ "machine",
41
+ "rows",
42
+ "eval_windows",
43
+ "eval_corpus_sha256",
44
+ "warmup_runs",
45
+ "gen_tokens",
46
+ "prompt_sha256",
47
+ }
48
+
49
+ REQUIRED_MACHINE = {
50
+ "cpu",
51
+ "arch",
52
+ "physical_cores",
53
+ "logical_cores",
54
+ "ram_gb",
55
+ "os",
56
+ "os_release",
57
+ "python",
58
+ "onnxruntime",
59
+ "provider",
60
+ "intra_op_threads",
61
+ "thread_policy",
62
+ }
63
+
64
+ REQUIRED_ROW = {
65
+ "precision",
66
+ "size_mb",
67
+ "latency_s_mean",
68
+ "latency_s_std",
69
+ "tokens_per_second",
70
+ "peak_ram_mb",
71
+ "perplexity",
72
+ "generated_tokens",
73
+ "measured_runs",
74
+ }
75
+
76
+ #: Loose physical bounds. Deliberately wide — the point is to catch a typo or a
77
+ #: fabricated order of magnitude, not to second-guess unusual but real hardware.
78
+ MAX_PLAUSIBLE_TOK_S = 100_000.0
79
+ MAX_PLAUSIBLE_PERPLEXITY = 1e6
80
+
81
+
82
+ @dataclass
83
+ class ValidationReport:
84
+ path: Path
85
+ errors: list[str]
86
+ warnings: list[str]
87
+ comparable: bool
88
+
89
+ @property
90
+ def ok(self) -> bool:
91
+ return not self.errors
92
+
93
+ def render(self) -> str:
94
+ status = "PASS" if self.ok else "FAIL"
95
+ flag = "" if self.comparable else " (not comparable — will be listed, not ranked)"
96
+ lines = [f"[{status}] {self.path.name}{flag}"]
97
+ lines += [f" error: {e}" for e in self.errors]
98
+ lines += [f" warning: {w}" for w in self.warnings]
99
+ return "\n".join(lines)
100
+
101
+
102
+ def validate_card(path: Path) -> ValidationReport:
103
+ """Check one card file. Never raises for bad content — it reports."""
104
+ errors: list[str] = []
105
+ warnings: list[str] = []
106
+
107
+ try:
108
+ card = json.loads(path.read_text())
109
+ except (OSError, json.JSONDecodeError) as exc:
110
+ return ValidationReport(path, [f"unreadable JSON: {exc}"], [], False)
111
+
112
+ if not isinstance(card, dict):
113
+ return ValidationReport(path, ["top level must be a JSON object"], [], False)
114
+
115
+ missing = REQUIRED_TOP_LEVEL - set(card)
116
+ if missing:
117
+ errors.append(f"missing field(s): {', '.join(sorted(missing))}")
118
+
119
+ if card.get("schema_version") != CARD_SCHEMA_VERSION:
120
+ errors.append(
121
+ f"schema_version is {card.get('schema_version')!r}, expected {CARD_SCHEMA_VERSION}. "
122
+ "Re-run with the current quantcost."
123
+ )
124
+
125
+ machine = card.get("machine")
126
+ if not isinstance(machine, dict):
127
+ errors.append("'machine' must be an object")
128
+ else:
129
+ missing_machine = REQUIRED_MACHINE - set(machine)
130
+ if missing_machine:
131
+ errors.append(f"machine missing: {', '.join(sorted(missing_machine))}")
132
+ if machine.get("intra_op_threads") in (0, None):
133
+ errors.append(
134
+ "machine.intra_op_threads is 0/null — thread count must be pinned, "
135
+ "or throughput is not comparable. Re-run with the current version."
136
+ )
137
+ for leak in ("hostname", "user", "username", "mac", "ip", "serial"):
138
+ if leak in machine:
139
+ errors.append(f"machine.{leak} must not be submitted (identifying information)")
140
+
141
+ rows = card.get("rows")
142
+ if not isinstance(rows, list) or not rows:
143
+ errors.append("'rows' must be a non-empty list")
144
+ rows = []
145
+
146
+ seen: set[str] = set()
147
+ scores_perplexity = False
148
+ for index, row in enumerate(rows):
149
+ where = f"rows[{index}]"
150
+ if not isinstance(row, dict):
151
+ errors.append(f"{where} must be an object")
152
+ continue
153
+
154
+ missing_row = REQUIRED_ROW - set(row)
155
+ if missing_row:
156
+ errors.append(f"{where} missing: {', '.join(sorted(missing_row))}")
157
+ continue
158
+
159
+ precision = row["precision"]
160
+ if precision not in PRECISION_FILES:
161
+ errors.append(f"{where}: unknown precision {precision!r}")
162
+ if precision in seen:
163
+ errors.append(f"{where}: duplicate precision {precision!r}")
164
+ seen.add(precision)
165
+
166
+ for field in ("size_mb", "latency_s_mean", "tokens_per_second", "peak_ram_mb"):
167
+ value = row[field]
168
+ if not isinstance(value, (int, float)) or value <= 0:
169
+ errors.append(f"{where}.{field} must be a positive number, got {value!r}")
170
+
171
+ if isinstance(row.get("tokens_per_second"), (int, float)):
172
+ if row["tokens_per_second"] > MAX_PLAUSIBLE_TOK_S:
173
+ errors.append(
174
+ f"{where}.tokens_per_second = {row['tokens_per_second']} is not physically "
175
+ f"plausible (cap {MAX_PLAUSIBLE_TOK_S:.0f})"
176
+ )
177
+
178
+ # Throughput must agree with latency and the token count it claims.
179
+ lat, tps, gen = row["latency_s_mean"], row["tokens_per_second"], row["generated_tokens"]
180
+ if all(isinstance(v, (int, float)) and v > 0 for v in (lat, tps, gen)):
181
+ implied = gen / lat
182
+ if abs(implied - tps) / max(implied, tps) > 0.02:
183
+ errors.append(
184
+ f"{where}: tokens_per_second ({tps}) disagrees with "
185
+ f"generated_tokens/latency_s_mean ({implied:.2f}). "
186
+ "These are derived from the same measurement and must match."
187
+ )
188
+
189
+ if row.get("generated_tokens") != card.get("gen_tokens"):
190
+ errors.append(
191
+ f"{where}.generated_tokens ({row.get('generated_tokens')}) does not match "
192
+ f"the card's gen_tokens ({card.get('gen_tokens')})"
193
+ )
194
+
195
+ if row.get("perplexity") is not None:
196
+ scores_perplexity = True
197
+ ppl = row["perplexity"]
198
+ if not isinstance(ppl, (int, float)) or ppl <= 1.0:
199
+ errors.append(f"{where}.perplexity must be > 1.0, got {ppl!r}")
200
+ elif ppl > MAX_PLAUSIBLE_PERPLEXITY:
201
+ errors.append(f"{where}.perplexity = {ppl} is implausible")
202
+
203
+ if isinstance(row.get("latency_s_std"), (int, float)) and isinstance(lat, (int, float)):
204
+ if lat > 0 and row["latency_s_std"] > lat * 0.5:
205
+ warnings.append(
206
+ f"{where}: latency varied by more than 50% of the mean — the machine was "
207
+ "probably busy. Consider re-running on an idle system."
208
+ )
209
+
210
+ if scores_perplexity and card.get("eval_corpus_sha256") != EVAL_CORPUS_SHA256:
211
+ errors.append(
212
+ f"eval_corpus_sha256 {card.get('eval_corpus_sha256')!r} does not match the bundled "
213
+ f"corpus {EVAL_CORPUS_SHA256!r}. Perplexity scored against different text is not "
214
+ "comparable."
215
+ )
216
+
217
+ if card.get("prompt_sha256") != prompt_sha256():
218
+ errors.append(
219
+ "prompt_sha256 does not match the pinned benchmark prompt. Latency depends on "
220
+ "prompt length, so a different prompt is not comparable."
221
+ )
222
+
223
+ if "fp32" not in seen and rows:
224
+ warnings.append("no fp32 baseline in this card, so speedups cannot be computed for it")
225
+
226
+ comparable = (
227
+ not errors
228
+ and card.get("gen_tokens") == COMPARABLE_GEN_TOKENS
229
+ and (not scores_perplexity or card.get("eval_windows") == COMPARABLE_EVAL_WINDOWS)
230
+ )
231
+ if not comparable and not errors:
232
+ warnings.append(
233
+ f"settings differ from the comparable defaults "
234
+ f"(gen_tokens={COMPARABLE_GEN_TOKENS}, eval_windows={COMPARABLE_EVAL_WINDOWS}); "
235
+ "this card will be listed but not ranked"
236
+ )
237
+
238
+ return ValidationReport(path, errors, warnings, comparable)
239
+
240
+
241
+ def validate_all(paths: list[Path]) -> list[ValidationReport]:
242
+ return [validate_card(p) for p in sorted(paths)]
243
+
244
+
245
+ def _main(argv: list[str] | None = None) -> int:
246
+ """``python -m edgellm.validate results/community/*.json`` — the CI entry point."""
247
+ import argparse
248
+
249
+ parser = argparse.ArgumentParser(prog="python -m edgellm.validate")
250
+ parser.add_argument("paths", nargs="+", type=Path)
251
+ parser.add_argument("--strict", action="store_true", help="Treat warnings as failures as well.")
252
+ args = parser.parse_args(argv)
253
+
254
+ files: list[Path] = []
255
+ for path in args.paths:
256
+ files.extend(sorted(path.glob("*.json")) if path.is_dir() else [path])
257
+
258
+ reports = validate_all(files)
259
+ for report in reports:
260
+ print(report.render())
261
+
262
+ failed = [r for r in reports if not r.ok]
263
+ warned = [r for r in reports if r.warnings]
264
+ print(
265
+ f"\n{len(reports) - len(failed)}/{len(reports)} card(s) passed"
266
+ + (f", {len(warned)} with warnings" if warned else "")
267
+ )
268
+ if failed:
269
+ return 1
270
+ return 1 if (args.strict and warned) else 0
271
+
272
+
273
+ if __name__ == "__main__":
274
+ raise SystemExit(_main())
@@ -0,0 +1,265 @@
1
+ Metadata-Version: 2.5
2
+ Name: quantcost
3
+ Version: 0.2.0
4
+ Summary: Measure what quantization actually costs on your own hardware: speed, size, memory and quality.
5
+ Project-URL: Homepage, https://github.com/vijay-kapse/EdgeLLM
6
+ Project-URL: Leaderboard, https://github.com/vijay-kapse/EdgeLLM/blob/main/results/LEADERBOARD.md
7
+ Project-URL: Issues, https://github.com/vijay-kapse/EdgeLLM/issues
8
+ Author: Vijay Kapse
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: benchmark,edge,inference,int4,int8,llm,npu,onnx,onnxruntime,perplexity,qualcomm,quantization
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: System :: Benchmark
24
+ Requires-Python: >=3.10
25
+ Requires-Dist: huggingface-hub>=0.23
26
+ Requires-Dist: numpy>=1.24
27
+ Requires-Dist: onnxruntime>=1.18
28
+ Requires-Dist: psutil>=5.9
29
+ Requires-Dist: pyyaml>=6.0
30
+ Requires-Dist: tokenizers>=0.19
31
+ Requires-Dist: typer>=0.12
32
+ Provides-Extra: aihub
33
+ Requires-Dist: qai-hub>=0.9; extra == 'aihub'
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=8.0; extra == 'dev'
36
+ Requires-Dist: ruff>=0.5; extra == 'dev'
37
+ Provides-Extra: quantize
38
+ Requires-Dist: datasets>=2.19; extra == 'quantize'
39
+ Requires-Dist: onnx-ir>=0.2; extra == 'quantize'
40
+ Requires-Dist: onnx>=1.16; extra == 'quantize'
41
+ Requires-Dist: optimum[onnxruntime]>=1.20; extra == 'quantize'
42
+ Requires-Dist: torch>=2.2; extra == 'quantize'
43
+ Requires-Dist: transformers>=4.44; extra == 'quantize'
44
+ Provides-Extra: report
45
+ Requires-Dist: matplotlib>=3.8; extra == 'report'
46
+ Description-Content-Type: text/markdown
47
+
48
+ # quantcost
49
+
50
+ **Find out what quantization actually costs you — on your machine, not someone else's.**
51
+
52
+ [![CI](https://github.com/vijay-kapse/EdgeLLM/actions/workflows/ci.yml/badge.svg)](https://github.com/vijay-kapse/EdgeLLM/actions/workflows/ci.yml)
53
+ [![PyPI](https://img.shields.io/pypi/v/quantcost)](https://pypi.org/project/quantcost/)
54
+ [![Python](https://img.shields.io/pypi/pyversions/quantcost)](https://pypi.org/project/quantcost/)
55
+ [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE)
56
+
57
+ ```bash
58
+ pip install quantcost
59
+ quantcost run
60
+ ```
61
+
62
+ That's it. Two to three minutes later you get speed, size, memory and *quality*
63
+ for fp32 / int8 / int4 on your own CPU, with a plain-language verdict.
64
+
65
+ No PyTorch. No GPU. No quantizing anything yourself — the pre-quantized ONNX
66
+ already published on the Hugging Face Hub gets downloaded and measured, so the
67
+ whole install is ONNX Runtime, NumPy and a tokenizer, and finishes in about ten
68
+ seconds.
69
+
70
+ ---
71
+
72
+ ## What quantization actually costs
73
+
74
+ "Quantize it, it'll be smaller and faster" is half right. Two models, one Apple
75
+ M4, measured with `quantcost run` at its defaults:
76
+
77
+ **SmolLM2-135M-Instruct**
78
+
79
+ | Precision | Size (MB) | tok/s | Speedup | Peak RAM (MB) | Perplexity | PPL change |
80
+ | --- | --- | --- | --- | --- | --- | --- |
81
+ | fp32 | 515 | 83.65 | baseline | 1360 | 23.07 | baseline |
82
+ | int8 | 131 | 39.36 | **0.47x** | 1002 | 25.02 | +8.5% |
83
+ | q4 | 174 | 57.40 | **0.69x** | 906 | 28.35 | +22.9% |
84
+
85
+ **Qwen2.5-0.5B-Instruct**
86
+
87
+ | Precision | Size (MB) | tok/s | Speedup | Peak RAM (MB) | Perplexity | PPL change |
88
+ | --- | --- | --- | --- | --- | --- | --- |
89
+ | fp32 | 1901 | 21.49 | baseline | 3282 | 19.26 | baseline |
90
+ | int8 | 488 | 10.17 | **0.47x** | 3052 | 21.19 | +10.0% |
91
+ | q4 | 750 | 18.37 | **0.85x** | 2303 | 22.28 | +15.7% |
92
+
93
+ Smaller: reliably, 3–4x. Less memory: yes, 25–30% off the peak. Faster: **no** —
94
+ int8 generates tokens at 0.47x the fp32 rate on *both* models. That the figure
95
+ lands on 0.47 twice, across a 135M and a 500M model, is what makes it look like
96
+ a property of the runtime rather than an accident of one benchmark.
97
+
98
+ ### Why, and why it is not "quantized maths is slow"
99
+
100
+ Run the same artifacts over a 512-token prefill instead of one token at a time:
101
+
102
+ | | prefill (512 tok) | decode (1 tok/step) |
103
+ | --- | --- | --- |
104
+ | SmolLM2 int8 | 0.79x | 0.43x |
105
+ | SmolLM2 q4 | 0.31x | 0.68x |
106
+ | Qwen int8 | 0.97x | 0.59x |
107
+ | Qwen q4 | 0.36x | 1.02x |
108
+
109
+ The int8 penalty roughly **halves** once there is a batch to amortise over
110
+ (0.43 → 0.79, 0.59 → 0.97). That is the signature of a fixed per-step cost:
111
+ ONNX Runtime unpacks the weights back to float inside each matmul, and a single
112
+ token cannot amortise unpacking a whole weight matrix. fp32 decode is
113
+ bandwidth-bound on *reading* weights; int8 decode is compute-bound on
114
+ *unpacking* them, so it does strictly more work despite being 4x smaller.
115
+
116
+ Note also that **int8 and q4 invert**: q4 is the better choice for decode
117
+ (0.68x / 1.02x vs int8's 0.43x / 0.59x) and much the worse for prefill
118
+ (0.31x / 0.36x vs 0.79x / 0.97x). They use different kernels —
119
+ `MatMulNBits` for q4, dynamic-quantize + `MatMulInteger` for int8 — with
120
+ opposite strengths. Which format is right depends on whether your workload is
121
+ prompt-heavy or generation-heavy.
122
+
123
+ ### One number that is easy to get wrong
124
+
125
+ Pin threads to your machine's **performance** cores, not every physical core.
126
+ On this M4 (4 performance + 6 efficiency) pinning all ten cost int8 66% of its
127
+ throughput — 24.1 vs 40.0 tok/s — and doubled run-to-run spread, because an ONNX
128
+ Runtime parallel region ends on a barrier and one thread on an efficiency core
129
+ gates the whole thing. `quantcost` does this by default and records how it
130
+ decided in every card. It is the single easiest way to publish a wrong number,
131
+ and it is how the first draft of this README got the figures above wrong.
132
+
133
+ So the honest answer to "should I quantize?" is **measure it on your hardware**,
134
+ which is why this is a tool rather than a blog post. On this machine you quantize
135
+ to fit in memory, not to go faster. On yours it may differ — that is the thing
136
+ worth finding out.
137
+
138
+ **[See what other machines measured →](results/LEADERBOARD.md)**
139
+
140
+ ## Add your machine
141
+
142
+ ```bash
143
+ quantcost run
144
+ quantcost submit
145
+ ```
146
+
147
+ `submit` validates your result, forks this repo, commits the card and opens the
148
+ pull request for you. No GitHub CLI? It prints a prefilled link instead.
149
+
150
+ Unusual hardware is the most valuable kind: Raspberry Pi, old ThinkPads,
151
+ Snapdragon laptops, bare-metal ARM. The interesting question is not who has the
152
+ fastest laptop — it is *where quantization pays off and where it backfires*, and
153
+ that only becomes visible across many real machines.
154
+
155
+ See [CONTRIBUTING.md](CONTRIBUTING.md) for what makes a good submission and
156
+ exactly what a card contains (no hostname, no username, no paths — the
157
+ fingerprint is one auditable function).
158
+
159
+ ## Usage
160
+
161
+ ```bash
162
+ # A different model — anything with ONNX on the Hub works
163
+ quantcost run --model onnx-community/Qwen2.5-0.5B-Instruct
164
+
165
+ # Which precisions does a model actually publish?
166
+ quantcost models --model onnx-community/Qwen2.5-0.5B-Instruct
167
+
168
+ # Speed and size only; skips the perplexity pass and is much faster
169
+ quantcost run --precisions fp32,int8 --skip-perplexity
170
+
171
+ # Pin threads to compare against a specific configuration
172
+ quantcost run --threads 4
173
+
174
+ # Rebuild the leaderboard from every submitted card
175
+ quantcost leaderboard
176
+ ```
177
+
178
+ Any Hub repo following the `onnx-community` / `transformers.js` layout works
179
+ unchanged — that is thousands of models, including everything under
180
+ [onnx-community](https://huggingface.co/onnx-community).
181
+
182
+ ## How the numbers are produced
183
+
184
+ The headline claim of this project is that its numbers are real, so the
185
+ methodology is worth stating plainly.
186
+
187
+ **Every precision is measured in its own subprocess.** Peak RSS is a per-process
188
+ high-water mark and an ONNX Runtime session does not return all of its arenas
189
+ when dropped, so measuring several precisions in one process reports the *union*
190
+ of their footprints and blames whichever ran last. Isolation is the only way the
191
+ peak-RAM column means what it says.
192
+
193
+ **The file cache is warmed before anything is timed.** ONNX Runtime mmaps
194
+ weights and faults them in lazily. Measured cold — straight after the download —
195
+ an fp32 baseline came in at 17 tok/s; warm, the same machine and build measured
196
+ 64. A 4x error on the number every other row is divided by is not a rounding
197
+ detail, so the model file is read through once before the session is built.
198
+
199
+ **Throughput is total tokens over total time**, derived from the same mean
200
+ latency that is reported, never the average of each run's own rate. Those are
201
+ different statistics (ratio of means vs. mean of ratios) and they disagree by a
202
+ few percent, which would make two columns of the same card contradict each other.
203
+
204
+ **Decoding is greedy and fixed-length.** No sampling, no early EOS stop, so every
205
+ precision does exactly the same amount of work and throughput does not depend on
206
+ the RNG.
207
+
208
+ **Perplexity is scored against a corpus pinned inside the package** — a fixed
209
+ 65,342-byte slice of WikiText-2, hash-checked on load and recorded in every card.
210
+ A leaderboard where machines score different text is not a leaderboard, and this
211
+ also removes a heavy `datasets` dependency. Method: non-overlapping 512-token
212
+ windows, summed next-token cross-entropy, `exp(total_nll / total_tokens)`. The
213
+ NumPy implementation is tested against the uniform-distribution case, where the
214
+ right answer is analytically the vocabulary size, and agrees with PyTorch's
215
+ `cross_entropy` to 4e-07 relative.
216
+
217
+ **Thread count is always pinned** to physical cores (SMT siblings contend for the
218
+ same vector units) and recorded, because leaving it at ORT's default stores "0 —
219
+ decide for me" and makes two very different runs look identical.
220
+
221
+ **Instability is reported, not hidden.** If run-to-run latency varies by more
222
+ than 15%, the report says the machine was too busy and asks you to re-run.
223
+
224
+ ### What validation can and cannot do
225
+
226
+ CI rejects cards that are malformed, internally inconsistent, scored against a
227
+ modified corpus, run with an unpinned thread count, or carrying identifying
228
+ information. It **cannot** prove a number came from real silicon — nothing short
229
+ of attested execution could. The defence is that every input is pinned, so a
230
+ fabricated card must be self-consistent across all of them, and anyone can re-run
231
+ the exact configuration recorded in the card. Cards are reviewed, not trusted.
232
+
233
+ ## Authoring quantized artifacts yourself
234
+
235
+ Everything above *consumes* pre-quantized ONNX. The original project also
236
+ *produces* it — export, quantize, a C++ inference harness, a SIMD INT8 GEMM
237
+ kernel, an Android app and a Qualcomm Snapdragon NPU path. That path needs the
238
+ heavy dependencies:
239
+
240
+ ```bash
241
+ pip install -e ".[quantize]"
242
+ edgellm export --help
243
+ edgellm quantize --help
244
+ edgellm bench --help # the original torch + Optimum harness
245
+ ```
246
+
247
+ See [docs/AUTHORING.md](docs/AUTHORING.md) for the full pipeline, the C++ harness
248
+ and the Snapdragon notes.
249
+
250
+ ## Development
251
+
252
+ ```bash
253
+ pip install -e ".[dev]"
254
+ ruff check . && ruff format --check . && pytest
255
+ ```
256
+
257
+ CI asserts that importing the CLI pulls in no heavy dependency. If you need
258
+ torch, import it inside the function that uses it — the fast install is what
259
+ makes a one-command benchmark viable for someone who has never heard of this
260
+ project.
261
+
262
+ ## License
263
+
264
+ MIT — see [LICENSE](LICENSE). The bundled eval corpus is WikiText-2, CC BY-SA 4.0;
265
+ see [edgellm/data/SOURCE.md](edgellm/data/SOURCE.md).
@@ -0,0 +1,27 @@
1
+ edgellm/__init__.py,sha256=WEF4Psnkp-LKq1B5VNPpWb7gcuv4Whh_Gjk9cd9R5IU,166
2
+ edgellm/base.py,sha256=hMVqPbnZFV9rymRfX6CtKQ4UPWGxZb_zEqVgqaIU7fs,1294
3
+ edgellm/benchmark.py,sha256=BYzMzcTv3gcTraajBBAW2JAzjDueCe_7BoXbctFM2V4,8181
4
+ edgellm/card.py,sha256=ev-3LJHimFf14OyKC1_ARZYSEfb7ozJJqBFa3Qd4lYM,7972
5
+ edgellm/cli.py,sha256=u131oDKUgm1ucZeS2LRU923qyaIIHLw_gHGeGNGI_G4,13706
6
+ edgellm/cli_bench.py,sha256=grLG-iRsBnMD8uSeTrzIBZMvUZYQdB2deSpkjzxGTtU,7849
7
+ edgellm/config.py,sha256=GuuloueR0y5_8drBYyOj-oLj2PN5D6Z_g_wiJ9fHVjc,3570
8
+ edgellm/eval_lite.py,sha256=eNM_EGMFr7O10l2xtYbkFTMGnPsARfK_aJeovpMzI8I,4806
9
+ edgellm/export.py,sha256=EaI3gWNJqL0KL1-Eso-FgcMmdqlN0oN1SWpRR8JfDp0,2344
10
+ edgellm/hub.py,sha256=djTbeoUryiRqqoKaFtMtke6VWN9UDMCu5xU-PY7ixLk,4598
11
+ edgellm/leaderboard.py,sha256=9zPUnarUzzGM8rciDu54r-2FPgpgf9NTpvxYTp92vAY,8443
12
+ edgellm/models.py,sha256=JmmQRjR3t1sNl8Kz-SX9NLf5bX1iJdbWNLJKusTDXjE,2745
13
+ edgellm/ort_lite.py,sha256=vw71IqlIUQ8AfKA2gtVbWVFXdHPEqcD79n2QHOxNITU,9564
14
+ edgellm/quantize.py,sha256=gxC0YxxQaXX_c1uq9h1HT94EgMyQvofk9nFWkmPZ7a0,5075
15
+ edgellm/render.py,sha256=Nu6iUBnhKueXEmiSk0eVs7hRPl_txAGFlnv5NWxklis,6998
16
+ edgellm/report.py,sha256=YSp5_3xAeWgKUCIz0xeZYxBKobXZsRu3RKlPDNok0j4,1807
17
+ edgellm/runners.py,sha256=D7ICj5PIsWj6ExCmyllon7VyNgrT92aCbjWzVLRMK7g,6227
18
+ edgellm/submit.py,sha256=dSmzvIseg3hCckxyvzANJBRNuwcnGDIG9YX1Grv1jDs,8118
19
+ edgellm/sweep.py,sha256=kaBCIsKwm_xM2_i2kYLoIp9O0JO9do6SiYLy2zzpCfI,11540
20
+ edgellm/validate.py,sha256=7GYm7gYhn15bxgOVGcEG10HujaVtv5PP0CWpo7TIkgU,10068
21
+ edgellm/data/SOURCE.md,sha256=9fv2dBtucKis_UjyYF2yZh_puSqyc9JDHWD0iHez-bI,919
22
+ edgellm/data/eval_wikitext2.txt,sha256=NKKkP89JIX96QhM_lhkw2MgwrH_jzHkn9CP0Qwg_l78,65342
23
+ quantcost-0.2.0.dist-info/METADATA,sha256=ND-s2JTWgdrg0WQmtTNXRsJTz9-qQj9WR1cTo8qdbow,11582
24
+ quantcost-0.2.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
25
+ quantcost-0.2.0.dist-info/entry_points.txt,sha256=1Hv957MAOBjb0MT3nJdko_Cqhrtfsxxxf7Oax4DQNBs,74
26
+ quantcost-0.2.0.dist-info/licenses/LICENSE,sha256=jBAq79ONijdbPnqSEEL9fDEzsZp8TCOxNdvmMhT6EzU,1068
27
+ quantcost-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ edgellm = edgellm.cli:main
3
+ quantcost = edgellm.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Vijay Kapse
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.