quantcost 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgellm/__init__.py +7 -0
- edgellm/base.py +44 -0
- edgellm/benchmark.py +226 -0
- edgellm/card.py +234 -0
- edgellm/cli.py +370 -0
- edgellm/cli_bench.py +195 -0
- edgellm/config.py +120 -0
- edgellm/data/SOURCE.md +17 -0
- edgellm/data/eval_wikitext2.txt +205 -0
- edgellm/eval_lite.py +128 -0
- edgellm/export.py +65 -0
- edgellm/hub.py +125 -0
- edgellm/leaderboard.py +210 -0
- edgellm/models.py +91 -0
- edgellm/ort_lite.py +236 -0
- edgellm/quantize.py +134 -0
- edgellm/render.py +164 -0
- edgellm/report.py +53 -0
- edgellm/runners.py +170 -0
- edgellm/submit.py +225 -0
- edgellm/sweep.py +309 -0
- edgellm/validate.py +274 -0
- quantcost-0.2.0.dist-info/METADATA +265 -0
- quantcost-0.2.0.dist-info/RECORD +27 -0
- quantcost-0.2.0.dist-info/WHEEL +4 -0
- quantcost-0.2.0.dist-info/entry_points.txt +3 -0
- quantcost-0.2.0.dist-info/licenses/LICENSE +21 -0
edgellm/cli.py
ADDED
|
@@ -0,0 +1,370 @@
|
|
|
1
|
+
"""The ``edgellm`` command-line interface (Typer)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from dataclasses import replace
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import typer
|
|
10
|
+
|
|
11
|
+
from edgellm import __version__, cli_bench
|
|
12
|
+
from edgellm.config import Config
|
|
13
|
+
|
|
14
|
+
app = typer.Typer(
|
|
15
|
+
add_completion=False,
|
|
16
|
+
help="EdgeLLM: quantize a small LLM and run it on-device with honest benchmarks.",
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
DEFAULT_CONFIG = Path("configs/default.yaml")
|
|
20
|
+
|
|
21
|
+
# The lightweight benchmark/submit/leaderboard commands live in their own module.
|
|
22
|
+
cli_bench.register(app)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _load_config(config_path: Path) -> Config:
|
|
26
|
+
if config_path.exists():
|
|
27
|
+
return Config.from_yaml(config_path)
|
|
28
|
+
typer.echo(f"[warn] config '{config_path}' not found; using built-in defaults.", err=True)
|
|
29
|
+
return Config()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@app.command()
|
|
33
|
+
def version() -> None:
|
|
34
|
+
"""Print the EdgeLLM version."""
|
|
35
|
+
typer.echo(__version__)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@app.command()
|
|
39
|
+
def info(
|
|
40
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
41
|
+
) -> None:
|
|
42
|
+
"""Show the resolved model + device configuration for this machine."""
|
|
43
|
+
from edgellm.models import ModelLoader
|
|
44
|
+
|
|
45
|
+
cfg = _load_config(config_path)
|
|
46
|
+
loader = ModelLoader(cfg.model)
|
|
47
|
+
typer.echo(f"model: {cfg.model.id}")
|
|
48
|
+
typer.echo(f"dtype: {cfg.model.dtype}")
|
|
49
|
+
typer.echo(f"device: {loader.resolve_device()} (requested: {cfg.model.device})")
|
|
50
|
+
typer.echo(f"backends: {', '.join(cfg.backends)}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@app.command()
|
|
54
|
+
def generate(
|
|
55
|
+
prompt: str = typer.Option(..., "--prompt", "-p", help="Prompt to generate from."),
|
|
56
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
57
|
+
max_new_tokens: int | None = typer.Option(None, help="Override max new tokens."),
|
|
58
|
+
temperature: float | None = typer.Option(None, help="Override sampling temperature."),
|
|
59
|
+
seed: int | None = typer.Option(None, help="Override random seed."),
|
|
60
|
+
greedy: bool = typer.Option(False, "--greedy", help="Disable sampling (deterministic)."),
|
|
61
|
+
) -> None:
|
|
62
|
+
"""Generate text from PROMPT with the PyTorch backend and print timing."""
|
|
63
|
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
|
|
64
|
+
|
|
65
|
+
from edgellm.models import ModelLoader
|
|
66
|
+
from edgellm.runners import PyTorchRunner
|
|
67
|
+
|
|
68
|
+
cfg = _load_config(config_path)
|
|
69
|
+
if max_new_tokens is not None:
|
|
70
|
+
cfg.generation.max_new_tokens = max_new_tokens
|
|
71
|
+
if temperature is not None:
|
|
72
|
+
cfg.generation.temperature = temperature
|
|
73
|
+
if seed is not None:
|
|
74
|
+
cfg.generation.seed = seed
|
|
75
|
+
if greedy:
|
|
76
|
+
cfg.generation.do_sample = False
|
|
77
|
+
|
|
78
|
+
loaded = ModelLoader(cfg.model).load()
|
|
79
|
+
runner = PyTorchRunner(loaded)
|
|
80
|
+
result = runner.generate(prompt, cfg.generation)
|
|
81
|
+
|
|
82
|
+
typer.echo("\n=== output ===")
|
|
83
|
+
typer.echo(result.text.strip())
|
|
84
|
+
typer.echo("\n=== stats ===")
|
|
85
|
+
typer.echo(f"backend: {result.backend}")
|
|
86
|
+
typer.echo(f"device: {loaded.device}")
|
|
87
|
+
typer.echo(f"prompt tokens: {result.prompt_tokens}")
|
|
88
|
+
typer.echo(f"generated tokens: {result.generated_tokens}")
|
|
89
|
+
typer.echo(f"latency: {result.latency_s:.3f} s")
|
|
90
|
+
typer.echo(f"throughput: {result.tokens_per_second:.2f} tok/s")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@app.command()
|
|
94
|
+
def export(
|
|
95
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
96
|
+
force: bool = typer.Option(False, "--force", help="Re-export even if it already exists."),
|
|
97
|
+
) -> None:
|
|
98
|
+
"""Export the configured model to ONNX (FP32) via Optimum."""
|
|
99
|
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
|
|
100
|
+
from edgellm.export import ONNXExporter
|
|
101
|
+
|
|
102
|
+
cfg = _load_config(config_path)
|
|
103
|
+
out_dir = ONNXExporter(cfg.model, cfg.export).export(force=force)
|
|
104
|
+
typer.echo(f"ONNX export: {out_dir}")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@app.command()
|
|
108
|
+
def snapdragon(
|
|
109
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
110
|
+
precision: str = typer.Option(
|
|
111
|
+
"int8", "--precision", help="Artifact to profile: int8|int4|fp32."
|
|
112
|
+
),
|
|
113
|
+
device: str = typer.Option("Snapdragon 8 Elite QRD", "--device", help="AI Hub device name."),
|
|
114
|
+
seq: int = typer.Option(64, "--seq", help="Fixed sequence length for compilation."),
|
|
115
|
+
) -> None:
|
|
116
|
+
"""Compile + profile the model on a real Snapdragon NPU via Qualcomm AI Hub."""
|
|
117
|
+
import sys
|
|
118
|
+
|
|
119
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "aihub"))
|
|
120
|
+
from run_on_snapdragon import SnapdragonProfiler, check_auth
|
|
121
|
+
|
|
122
|
+
cfg = _load_config(config_path)
|
|
123
|
+
safe = cfg.model.id.replace("/", "__")
|
|
124
|
+
suffix = {"int8": "-int8-dynamic", "int4": "-int4", "fp32": "-fp32"}.get(
|
|
125
|
+
precision, "-int8-dynamic"
|
|
126
|
+
)
|
|
127
|
+
model_dir = Path(cfg.export.output_dir) / f"{safe}{suffix}"
|
|
128
|
+
if not check_auth():
|
|
129
|
+
raise typer.Exit(0)
|
|
130
|
+
SnapdragonProfiler(model_dir, device, seq, 0).run(Path("results/benchmarks.json"))
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@app.command()
|
|
134
|
+
def encode(
|
|
135
|
+
prompt: str = typer.Option(..., "--prompt", "-p", help="Prompt to tokenize."),
|
|
136
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
137
|
+
) -> None:
|
|
138
|
+
"""Print space-separated token ids for PROMPT (feeds the C++ harness)."""
|
|
139
|
+
from transformers import AutoTokenizer
|
|
140
|
+
|
|
141
|
+
from edgellm.runners import encode_prompt
|
|
142
|
+
|
|
143
|
+
cfg = _load_config(config_path)
|
|
144
|
+
tok = AutoTokenizer.from_pretrained(cfg.model.id, revision=cfg.model.revision)
|
|
145
|
+
ids = encode_prompt(tok, prompt)["input_ids"][0].tolist()
|
|
146
|
+
typer.echo(" ".join(str(i) for i in ids))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@app.command()
|
|
150
|
+
def decode(
|
|
151
|
+
ids: str = typer.Option(..., "--ids", help="Space/comma-separated token ids to decode."),
|
|
152
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
153
|
+
) -> None:
|
|
154
|
+
"""Decode token ids back to text (e.g. the C++ harness's GENERATED_IDS)."""
|
|
155
|
+
from transformers import AutoTokenizer
|
|
156
|
+
|
|
157
|
+
cfg = _load_config(config_path)
|
|
158
|
+
tok = AutoTokenizer.from_pretrained(cfg.model.id, revision=cfg.model.revision)
|
|
159
|
+
id_list = [int(x) for x in ids.replace(",", " ").split()]
|
|
160
|
+
typer.echo(tok.decode(id_list, skip_special_tokens=True))
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
@app.command()
|
|
164
|
+
def report(
|
|
165
|
+
results_dir: Path = typer.Option(Path("results"), "--results", help="Results directory."),
|
|
166
|
+
) -> None:
|
|
167
|
+
"""Regenerate the Markdown table + bar charts from benchmarks.json."""
|
|
168
|
+
from edgellm.benchmark import render_markdown
|
|
169
|
+
from edgellm.report import render_charts
|
|
170
|
+
|
|
171
|
+
json_path = results_dir / "benchmarks.json"
|
|
172
|
+
if not json_path.exists():
|
|
173
|
+
typer.echo(f"no results at {json_path}; run 'edgellm benchmark' first.", err=True)
|
|
174
|
+
raise typer.Exit(1)
|
|
175
|
+
(results_dir / "benchmark.md").write_text(render_markdown(json_path))
|
|
176
|
+
chart = render_charts(json_path, results_dir / "benchmark_chart.png")
|
|
177
|
+
typer.echo(f"wrote {results_dir / 'benchmark.md'} and {chart}")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
@app.command()
|
|
181
|
+
def quantize(
|
|
182
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
183
|
+
) -> None:
|
|
184
|
+
"""Produce INT8 (ONNX Runtime) and INT4 (block-wise) quantized artifacts."""
|
|
185
|
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
|
|
186
|
+
from edgellm.export import ONNXExporter
|
|
187
|
+
from edgellm.quantize import Quantizer
|
|
188
|
+
|
|
189
|
+
cfg = _load_config(config_path)
|
|
190
|
+
fp32_dir = ONNXExporter(cfg.model, cfg.export).export()
|
|
191
|
+
q = Quantizer(cfg.quantize)
|
|
192
|
+
base = fp32_dir.parent / fp32_dir.name.replace("-fp32", "")
|
|
193
|
+
int8_dir = q.ort_dynamic_int8(fp32_dir, Path(f"{base}-int8-dynamic"))
|
|
194
|
+
int4_dir = q.ort_int4(fp32_dir, Path(f"{base}-int4"))
|
|
195
|
+
typer.echo(f"INT8: {int8_dir}")
|
|
196
|
+
typer.echo(f"INT4: {int4_dir}")
|
|
197
|
+
typer.echo("\nINT4 backend availability on this machine:")
|
|
198
|
+
for backend, state in q.int4_availability().items():
|
|
199
|
+
typer.echo(f" {backend}: {state}")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
@app.command()
|
|
203
|
+
def benchmark(
|
|
204
|
+
config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
|
|
205
|
+
skip_ppl: bool = typer.Option(False, "--skip-ppl", help="Skip perplexity (faster)."),
|
|
206
|
+
results_dir: Path = typer.Option(Path("results"), "--results", help="Output directory."),
|
|
207
|
+
only: str = typer.Option(
|
|
208
|
+
"pt-fp32,ort-fp32,ort-int8,ort-int4,pt-int8",
|
|
209
|
+
"--only",
|
|
210
|
+
help="Comma list of backends to benchmark.",
|
|
211
|
+
),
|
|
212
|
+
) -> None:
|
|
213
|
+
"""Benchmark FP32/INT8/INT4 across PyTorch + ONNX Runtime and write real numbers."""
|
|
214
|
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
|
|
215
|
+
|
|
216
|
+
from edgellm.benchmark import (
|
|
217
|
+
BenchmarkHarness,
|
|
218
|
+
PerplexityEvaluator,
|
|
219
|
+
measure_size_mb,
|
|
220
|
+
render_markdown,
|
|
221
|
+
save_results,
|
|
222
|
+
)
|
|
223
|
+
from edgellm.export import ONNXExporter
|
|
224
|
+
from edgellm.models import ModelLoader
|
|
225
|
+
from edgellm.quantize import Quantizer
|
|
226
|
+
from edgellm.runners import ORTRunner, PyTorchRunner
|
|
227
|
+
|
|
228
|
+
cfg = _load_config(config_path)
|
|
229
|
+
selected = {s.strip() for s in only.split(",") if s.strip()}
|
|
230
|
+
harness = BenchmarkHarness(cfg.benchmark)
|
|
231
|
+
ppl_eval = None if skip_ppl else PerplexityEvaluator(cfg.benchmark)
|
|
232
|
+
json_path = results_dir / "benchmarks.json"
|
|
233
|
+
|
|
234
|
+
loaded = ModelLoader(cfg.model).load() # FP32 on the best local device (mps/cuda/cpu)
|
|
235
|
+
fp32_dir = ONNXExporter(cfg.model, cfg.export).export()
|
|
236
|
+
quantizer = Quantizer(cfg.quantize)
|
|
237
|
+
base = fp32_dir.parent / fp32_dir.name.replace("-fp32", "")
|
|
238
|
+
|
|
239
|
+
def bench_ort(model_dir: Path, precision: str, name: str):
|
|
240
|
+
runner = ORTRunner(str(model_dir), loaded.tokenizer, name=name)
|
|
241
|
+
ppl = _ppl_ort(runner, ppl_eval, loaded.tokenizer) if ppl_eval else None
|
|
242
|
+
size = measure_size_mb(model_dir, ("*.onnx", "*.onnx_data"))
|
|
243
|
+
return harness.run(
|
|
244
|
+
runner,
|
|
245
|
+
precision=precision,
|
|
246
|
+
device="cpu",
|
|
247
|
+
model_id=cfg.model.id,
|
|
248
|
+
generation=cfg.generation,
|
|
249
|
+
size_mb=size,
|
|
250
|
+
perplexity=ppl,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
def bench_pt_fp32():
|
|
254
|
+
ppl = (
|
|
255
|
+
_ppl_torch(loaded.model, loaded.device, ppl_eval, loaded.tokenizer)
|
|
256
|
+
if ppl_eval
|
|
257
|
+
else None
|
|
258
|
+
)
|
|
259
|
+
return harness.run(
|
|
260
|
+
PyTorchRunner(loaded),
|
|
261
|
+
precision="fp32",
|
|
262
|
+
device=loaded.device,
|
|
263
|
+
model_id=cfg.model.id,
|
|
264
|
+
generation=cfg.generation,
|
|
265
|
+
size_mb=_pytorch_size_mb(cfg.model.id, cfg.model.revision),
|
|
266
|
+
perplexity=ppl,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
def bench_pt_int8():
|
|
270
|
+
cpu_loaded = ModelLoader(replace(cfg.model, device="cpu")).load()
|
|
271
|
+
qmodel = Quantizer.pytorch_dynamic_int8(cpu_loaded.model)
|
|
272
|
+
cpu_loaded = replace(cpu_loaded, model=qmodel, device="cpu")
|
|
273
|
+
ppl = _ppl_torch(qmodel, "cpu", ppl_eval, cpu_loaded.tokenizer) if ppl_eval else None
|
|
274
|
+
return harness.run(
|
|
275
|
+
PyTorchRunner(cpu_loaded),
|
|
276
|
+
precision="int8",
|
|
277
|
+
device="cpu",
|
|
278
|
+
model_id=cfg.model.id,
|
|
279
|
+
generation=cfg.generation,
|
|
280
|
+
size_mb=None,
|
|
281
|
+
perplexity=ppl,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
# (key, human label, thunk). Each runs guarded so one failure can't discard the rest.
|
|
285
|
+
steps = [
|
|
286
|
+
("pt-fp32", "pytorch fp32", bench_pt_fp32),
|
|
287
|
+
("ort-fp32", "ort-cpu fp32", lambda: bench_ort(fp32_dir, "fp32", "ort-cpu")),
|
|
288
|
+
(
|
|
289
|
+
"ort-int8",
|
|
290
|
+
"ort-cpu int8",
|
|
291
|
+
lambda: bench_ort(
|
|
292
|
+
quantizer.ort_dynamic_int8(fp32_dir, Path(f"{base}-int8-dynamic")),
|
|
293
|
+
"int8",
|
|
294
|
+
"ort-cpu-int8",
|
|
295
|
+
),
|
|
296
|
+
),
|
|
297
|
+
(
|
|
298
|
+
"ort-int4",
|
|
299
|
+
"ort-cpu int4",
|
|
300
|
+
lambda: bench_ort(
|
|
301
|
+
quantizer.ort_int4(fp32_dir, Path(f"{base}-int4")), "int4", "ort-cpu-int4"
|
|
302
|
+
),
|
|
303
|
+
),
|
|
304
|
+
("pt-int8", "pytorch int8 (cpu)", bench_pt_int8),
|
|
305
|
+
]
|
|
306
|
+
|
|
307
|
+
for key, label, thunk in steps:
|
|
308
|
+
if key not in selected:
|
|
309
|
+
continue
|
|
310
|
+
typer.echo(f"[bench] {label}...")
|
|
311
|
+
try:
|
|
312
|
+
result = thunk()
|
|
313
|
+
except Exception as exc: # noqa: BLE001 - one backend must not sink the others
|
|
314
|
+
typer.echo(f"[bench] {label} FAILED: {type(exc).__name__}: {exc}", err=True)
|
|
315
|
+
continue
|
|
316
|
+
save_results([result], json_path) # persist incrementally
|
|
317
|
+
|
|
318
|
+
md = render_markdown(json_path)
|
|
319
|
+
(results_dir / "benchmark.md").write_text(md)
|
|
320
|
+
typer.echo("\n=== results ===")
|
|
321
|
+
typer.echo(md)
|
|
322
|
+
typer.echo(f"wrote {json_path} and {results_dir / 'benchmark.md'}")
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _ppl_torch(model, device: str, ppl_eval, tokenizer) -> float:
|
|
326
|
+
import torch
|
|
327
|
+
|
|
328
|
+
def forward(ids: torch.Tensor, attn: torch.Tensor) -> torch.Tensor:
|
|
329
|
+
with torch.inference_mode():
|
|
330
|
+
out = model(input_ids=ids.to(device), attention_mask=attn.to(device))
|
|
331
|
+
return out.logits.detach().to("cpu")
|
|
332
|
+
|
|
333
|
+
return ppl_eval.evaluate(forward, tokenizer)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _ppl_ort(runner, ppl_eval, tokenizer) -> float:
|
|
337
|
+
import torch
|
|
338
|
+
|
|
339
|
+
def forward(ids: torch.Tensor, attn: torch.Tensor) -> torch.Tensor:
|
|
340
|
+
out = runner.model(input_ids=ids, attention_mask=attn)
|
|
341
|
+
return out.logits.detach().to("cpu")
|
|
342
|
+
|
|
343
|
+
return ppl_eval.evaluate(forward, tokenizer)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _pytorch_size_mb(model_id: str, revision: str) -> float | None:
|
|
347
|
+
"""On-disk size (MB) of the model weights in the local HF cache."""
|
|
348
|
+
from huggingface_hub import snapshot_download
|
|
349
|
+
|
|
350
|
+
from edgellm.benchmark import measure_size_mb
|
|
351
|
+
|
|
352
|
+
try:
|
|
353
|
+
snap = Path(
|
|
354
|
+
snapshot_download(
|
|
355
|
+
model_id, revision=revision, allow_patterns=["*.safetensors", "*.bin"]
|
|
356
|
+
)
|
|
357
|
+
)
|
|
358
|
+
except Exception:
|
|
359
|
+
return None
|
|
360
|
+
size = measure_size_mb(snap, ("*.safetensors", "*.bin"))
|
|
361
|
+
return size or None
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def main() -> None:
|
|
365
|
+
"""Console-script entry point."""
|
|
366
|
+
app()
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
if __name__ == "__main__":
|
|
370
|
+
main()
|
edgellm/cli_bench.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""The ``run`` / ``submit`` / ``leaderboard`` commands — the lightweight path.
|
|
2
|
+
|
|
3
|
+
Registered onto the main Typer app by :mod:`edgellm.cli`. Kept in its own module
|
|
4
|
+
because this is the path a contributor touches, and it should be readable without
|
|
5
|
+
scrolling past the export/quantize authoring commands.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import logging
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import typer
|
|
15
|
+
|
|
16
|
+
from edgellm.eval_lite import DEFAULT_EVAL_WINDOWS
|
|
17
|
+
from edgellm.hub import DEFAULT_MODEL, DEFAULT_PRECISIONS
|
|
18
|
+
from edgellm.sweep import (
|
|
19
|
+
DEFAULT_GEN_TOKENS,
|
|
20
|
+
DEFAULT_MEASURED_RUNS,
|
|
21
|
+
DEFAULT_WARMUP_RUNS,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
COMMUNITY_DIR = Path("results/community")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _parse_precisions(value: str) -> tuple[str, ...]:
|
|
28
|
+
items = tuple(p.strip() for p in value.split(",") if p.strip())
|
|
29
|
+
if not items:
|
|
30
|
+
raise typer.BadParameter("Give at least one precision, e.g. --precisions fp32,int8,q4")
|
|
31
|
+
return items
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def register(app: typer.Typer) -> None:
|
|
35
|
+
"""Attach the lightweight commands to ``app``."""
|
|
36
|
+
|
|
37
|
+
@app.command()
|
|
38
|
+
def run(
|
|
39
|
+
model: str = typer.Option(DEFAULT_MODEL, "--model", "-m", help="Hub model id with ONNX."),
|
|
40
|
+
precisions: str = typer.Option(
|
|
41
|
+
",".join(DEFAULT_PRECISIONS), "--precisions", help="Comma-separated precision list."
|
|
42
|
+
),
|
|
43
|
+
revision: str = typer.Option("main", "--revision", help="Hub revision to pin."),
|
|
44
|
+
provider: str = typer.Option(
|
|
45
|
+
"CPUExecutionProvider", "--provider", help="ONNX Runtime execution provider."
|
|
46
|
+
),
|
|
47
|
+
threads: int | None = typer.Option(
|
|
48
|
+
None, "--threads", help="intra-op threads (default: physical core count)."
|
|
49
|
+
),
|
|
50
|
+
gen_tokens: int = typer.Option(
|
|
51
|
+
DEFAULT_GEN_TOKENS, "--gen-tokens", help="Tokens to generate per run."
|
|
52
|
+
),
|
|
53
|
+
measured_runs: int = typer.Option(
|
|
54
|
+
DEFAULT_MEASURED_RUNS, "--measured-runs", help="Timed runs per precision."
|
|
55
|
+
),
|
|
56
|
+
warmup_runs: int = typer.Option(
|
|
57
|
+
DEFAULT_WARMUP_RUNS, "--warmup-runs", help="Untimed warmup runs."
|
|
58
|
+
),
|
|
59
|
+
eval_windows: int = typer.Option(
|
|
60
|
+
DEFAULT_EVAL_WINDOWS, "--eval-windows", help="512-token perplexity windows."
|
|
61
|
+
),
|
|
62
|
+
skip_perplexity: bool = typer.Option(
|
|
63
|
+
False, "--skip-perplexity", help="Measure speed and size only (much faster)."
|
|
64
|
+
),
|
|
65
|
+
out: Path = typer.Option(
|
|
66
|
+
COMMUNITY_DIR, "--out", help="Directory to write the result card into."
|
|
67
|
+
),
|
|
68
|
+
no_isolate: bool = typer.Option(
|
|
69
|
+
False,
|
|
70
|
+
"--no-isolate",
|
|
71
|
+
help="Measure all precisions in one process (faster, but peak-RAM becomes unreliable).",
|
|
72
|
+
),
|
|
73
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Debug logging."),
|
|
74
|
+
) -> None:
|
|
75
|
+
"""Benchmark pre-quantized ONNX precisions of MODEL on this machine.
|
|
76
|
+
|
|
77
|
+
Downloads already-quantized artifacts from the Hub — nothing is quantized
|
|
78
|
+
locally, so this needs no PyTorch and no GPU.
|
|
79
|
+
"""
|
|
80
|
+
from edgellm.render import console_report
|
|
81
|
+
from edgellm.sweep import SweepSettings, run_sweep, save_card
|
|
82
|
+
|
|
83
|
+
logging.basicConfig(
|
|
84
|
+
level=logging.DEBUG if verbose else logging.WARNING,
|
|
85
|
+
format="%(levelname)s %(name)s: %(message)s",
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
settings = SweepSettings(
|
|
89
|
+
model_id=model,
|
|
90
|
+
precisions=_parse_precisions(precisions),
|
|
91
|
+
revision=revision,
|
|
92
|
+
provider=provider,
|
|
93
|
+
intra_op_threads=threads,
|
|
94
|
+
warmup_runs=warmup_runs,
|
|
95
|
+
measured_runs=measured_runs,
|
|
96
|
+
gen_tokens=gen_tokens,
|
|
97
|
+
eval_windows=eval_windows,
|
|
98
|
+
skip_perplexity=skip_perplexity,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
typer.echo(f"Benchmarking {model} on this machine ({len(settings.precisions)} precisions).")
|
|
102
|
+
typer.echo("First run downloads the ONNX artifacts; later runs reuse the HF cache.\n")
|
|
103
|
+
|
|
104
|
+
def progress(precision: str, state: str) -> None:
|
|
105
|
+
if state == "start":
|
|
106
|
+
typer.echo(f" {precision:>6} ... ", nl=False)
|
|
107
|
+
elif state == "ok":
|
|
108
|
+
typer.echo("done")
|
|
109
|
+
else:
|
|
110
|
+
typer.echo("skipped")
|
|
111
|
+
|
|
112
|
+
card, errors = run_sweep(settings, on_progress=progress, isolate=not no_isolate)
|
|
113
|
+
|
|
114
|
+
if not card.rows:
|
|
115
|
+
typer.echo("\nNo precision could be measured. Reasons:", err=True)
|
|
116
|
+
for precision, why in errors.items():
|
|
117
|
+
typer.echo(f" {precision}: {why}", err=True)
|
|
118
|
+
raise typer.Exit(1)
|
|
119
|
+
|
|
120
|
+
typer.echo(console_report(card, errors))
|
|
121
|
+
|
|
122
|
+
path = save_card(card, out)
|
|
123
|
+
typer.echo(f"Result card: {path}")
|
|
124
|
+
typer.echo("Share it on the public leaderboard with: quantcost submit")
|
|
125
|
+
|
|
126
|
+
@app.command()
|
|
127
|
+
def models(
|
|
128
|
+
model: str = typer.Option(DEFAULT_MODEL, "--model", "-m", help="Hub model id to inspect."),
|
|
129
|
+
revision: str = typer.Option("main", "--revision"),
|
|
130
|
+
) -> None:
|
|
131
|
+
"""List which ONNX precisions a Hub model actually publishes."""
|
|
132
|
+
from edgellm.hub import available_precisions
|
|
133
|
+
|
|
134
|
+
found = available_precisions(model, revision=revision)
|
|
135
|
+
if not found:
|
|
136
|
+
typer.echo(
|
|
137
|
+
f"{model} publishes no ONNX files under onnx/. Try a model from "
|
|
138
|
+
"https://huggingface.co/onnx-community",
|
|
139
|
+
err=True,
|
|
140
|
+
)
|
|
141
|
+
raise typer.Exit(1)
|
|
142
|
+
typer.echo(f"{model} publishes: {', '.join(found)}")
|
|
143
|
+
|
|
144
|
+
@app.command()
|
|
145
|
+
def leaderboard(
|
|
146
|
+
community: Path = typer.Option(COMMUNITY_DIR, "--dir", help="Directory of result cards."),
|
|
147
|
+
out: Path = typer.Option(
|
|
148
|
+
Path("results/LEADERBOARD.md"), "--out", help="Markdown file to write."
|
|
149
|
+
),
|
|
150
|
+
json_out: Path | None = typer.Option(
|
|
151
|
+
Path("site/leaderboard.json"), "--json-out", help="JSON for the site to render."
|
|
152
|
+
),
|
|
153
|
+
) -> None:
|
|
154
|
+
"""Rebuild the leaderboard from every card in the community directory."""
|
|
155
|
+
from edgellm.leaderboard import build_leaderboard, render_leaderboard_markdown
|
|
156
|
+
|
|
157
|
+
cards = sorted(community.glob("*.json"))
|
|
158
|
+
if not cards:
|
|
159
|
+
typer.echo(f"No result cards found in {community}.", err=True)
|
|
160
|
+
raise typer.Exit(1)
|
|
161
|
+
|
|
162
|
+
board = build_leaderboard(cards)
|
|
163
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
out.write_text(render_leaderboard_markdown(board))
|
|
165
|
+
typer.echo(
|
|
166
|
+
f"Wrote {out} from {len(cards)} card(s), {len(board['entries'])} machine-model rows."
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
if json_out is not None:
|
|
170
|
+
json_out.parent.mkdir(parents=True, exist_ok=True)
|
|
171
|
+
json_out.write_text(json.dumps(board, indent=2) + "\n")
|
|
172
|
+
typer.echo(f"Wrote {json_out}")
|
|
173
|
+
|
|
174
|
+
@app.command()
|
|
175
|
+
def submit(
|
|
176
|
+
card: Path | None = typer.Option(
|
|
177
|
+
None, "--card", help="Card to submit (default: the newest in results/community)."
|
|
178
|
+
),
|
|
179
|
+
community: Path = typer.Option(COMMUNITY_DIR, "--dir"),
|
|
180
|
+
name: str = typer.Option("", "--name", help="Credit line for the leaderboard (optional)."),
|
|
181
|
+
notes: str = typer.Option("", "--notes", help="Anything unusual about this machine."),
|
|
182
|
+
dry_run: bool = typer.Option(False, "--dry-run", help="Validate and print, do not push."),
|
|
183
|
+
) -> None:
|
|
184
|
+
"""Open a pull request adding your result card to the public leaderboard."""
|
|
185
|
+
from edgellm.submit import submit_card
|
|
186
|
+
|
|
187
|
+
target = card
|
|
188
|
+
if target is None:
|
|
189
|
+
cards = sorted(community.glob("*.json"), key=lambda p: p.stat().st_mtime)
|
|
190
|
+
if not cards:
|
|
191
|
+
typer.echo(f"No result card in {community}. Run `quantcost run` first.", err=True)
|
|
192
|
+
raise typer.Exit(1)
|
|
193
|
+
target = cards[-1]
|
|
194
|
+
|
|
195
|
+
submit_card(target, name=name, notes=notes, dry_run=dry_run, echo=typer.echo)
|
edgellm/config.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Typed configuration objects loaded from YAML.
|
|
2
|
+
|
|
3
|
+
The whole project is driven by a single :class:`Config` tree so that every phase
|
|
4
|
+
(loading, export, quantization, benchmarking) reads its settings from one place
|
|
5
|
+
and the CLI can override any field.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field, fields, is_dataclass
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any, get_type_hints
|
|
13
|
+
|
|
14
|
+
import yaml
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class ModelConfig:
|
|
19
|
+
"""Which model to load and how to place it on hardware."""
|
|
20
|
+
|
|
21
|
+
id: str = "Qwen/Qwen2.5-0.5B-Instruct"
|
|
22
|
+
revision: str = "main"
|
|
23
|
+
dtype: str = "float32"
|
|
24
|
+
device: str = "auto"
|
|
25
|
+
trust_remote_code: bool = False
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class GenerationConfig:
|
|
30
|
+
"""Text-generation decoding parameters."""
|
|
31
|
+
|
|
32
|
+
max_new_tokens: int = 128
|
|
33
|
+
min_new_tokens: int | None = None
|
|
34
|
+
temperature: float = 0.7
|
|
35
|
+
top_p: float = 0.9
|
|
36
|
+
do_sample: bool = True
|
|
37
|
+
seed: int = 0
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class BenchmarkConfig:
|
|
42
|
+
"""Settings for the benchmark harness (used from Phase 2 onward)."""
|
|
43
|
+
|
|
44
|
+
warmup_runs: int = 2
|
|
45
|
+
measured_runs: int = 5
|
|
46
|
+
gen_tokens: int = 64
|
|
47
|
+
prompt: str = "Explain what neural network quantization is, in one short paragraph."
|
|
48
|
+
eval_dataset: str = "wikitext"
|
|
49
|
+
eval_config: str = "wikitext-2-raw-v1"
|
|
50
|
+
eval_num_samples: int = 32
|
|
51
|
+
eval_max_length: int = 512
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class ExportConfig:
|
|
56
|
+
"""ONNX export settings (used from Phase 2 onward)."""
|
|
57
|
+
|
|
58
|
+
opset: int = 17
|
|
59
|
+
output_dir: str = "artifacts/onnx"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class Int8Config:
|
|
64
|
+
scheme: str = "dynamic"
|
|
65
|
+
per_channel: bool = True
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class Int4Config:
|
|
70
|
+
method: str = "gptq"
|
|
71
|
+
group_size: int = 128
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass
|
|
75
|
+
class QuantizeConfig:
|
|
76
|
+
"""Quantization settings (used from Phase 3 onward)."""
|
|
77
|
+
|
|
78
|
+
int8: Int8Config = field(default_factory=Int8Config)
|
|
79
|
+
int4: Int4Config = field(default_factory=Int4Config)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class Config:
|
|
84
|
+
"""Root configuration tree."""
|
|
85
|
+
|
|
86
|
+
model: ModelConfig = field(default_factory=ModelConfig)
|
|
87
|
+
generation: GenerationConfig = field(default_factory=GenerationConfig)
|
|
88
|
+
benchmark: BenchmarkConfig = field(default_factory=BenchmarkConfig)
|
|
89
|
+
export: ExportConfig = field(default_factory=ExportConfig)
|
|
90
|
+
quantize: QuantizeConfig = field(default_factory=QuantizeConfig)
|
|
91
|
+
backends: list[str] = field(default_factory=lambda: ["pytorch", "ort-cpu"])
|
|
92
|
+
|
|
93
|
+
@classmethod
|
|
94
|
+
def from_yaml(cls, path: str | Path) -> Config:
|
|
95
|
+
"""Load a :class:`Config` from a YAML file, filling in defaults."""
|
|
96
|
+
raw = yaml.safe_load(Path(path).read_text()) or {}
|
|
97
|
+
if not isinstance(raw, dict):
|
|
98
|
+
raise ValueError(f"Config root must be a mapping, got {type(raw).__name__}")
|
|
99
|
+
return _from_dict(cls, raw)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _from_dict(cls: type, data: dict[str, Any]) -> Any:
|
|
103
|
+
"""Recursively build a (possibly nested) dataclass from a plain dict.
|
|
104
|
+
|
|
105
|
+
Unknown keys raise, so typos in the YAML fail loudly instead of being ignored.
|
|
106
|
+
"""
|
|
107
|
+
kwargs: dict[str, Any] = {}
|
|
108
|
+
known = {f.name for f in fields(cls)}
|
|
109
|
+
# Resolve annotations to real types (needed because `from __future__ import
|
|
110
|
+
# annotations` turns dataclass field types into strings).
|
|
111
|
+
hints = get_type_hints(cls)
|
|
112
|
+
for key, value in data.items():
|
|
113
|
+
if key not in known:
|
|
114
|
+
raise ValueError(f"Unknown config key '{key}' for {cls.__name__}")
|
|
115
|
+
field_type = hints[key]
|
|
116
|
+
if is_dataclass(field_type) and isinstance(value, dict):
|
|
117
|
+
kwargs[key] = _from_dict(field_type, value)
|
|
118
|
+
else:
|
|
119
|
+
kwargs[key] = value
|
|
120
|
+
return cls(**kwargs)
|