quantcost 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
edgellm/cli.py ADDED
@@ -0,0 +1,370 @@
1
+ """The ``edgellm`` command-line interface (Typer)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from dataclasses import replace
7
+ from pathlib import Path
8
+
9
+ import typer
10
+
11
+ from edgellm import __version__, cli_bench
12
+ from edgellm.config import Config
13
+
14
+ app = typer.Typer(
15
+ add_completion=False,
16
+ help="EdgeLLM: quantize a small LLM and run it on-device with honest benchmarks.",
17
+ )
18
+
19
+ DEFAULT_CONFIG = Path("configs/default.yaml")
20
+
21
+ # The lightweight benchmark/submit/leaderboard commands live in their own module.
22
+ cli_bench.register(app)
23
+
24
+
25
+ def _load_config(config_path: Path) -> Config:
26
+ if config_path.exists():
27
+ return Config.from_yaml(config_path)
28
+ typer.echo(f"[warn] config '{config_path}' not found; using built-in defaults.", err=True)
29
+ return Config()
30
+
31
+
32
+ @app.command()
33
+ def version() -> None:
34
+ """Print the EdgeLLM version."""
35
+ typer.echo(__version__)
36
+
37
+
38
+ @app.command()
39
+ def info(
40
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
41
+ ) -> None:
42
+ """Show the resolved model + device configuration for this machine."""
43
+ from edgellm.models import ModelLoader
44
+
45
+ cfg = _load_config(config_path)
46
+ loader = ModelLoader(cfg.model)
47
+ typer.echo(f"model: {cfg.model.id}")
48
+ typer.echo(f"dtype: {cfg.model.dtype}")
49
+ typer.echo(f"device: {loader.resolve_device()} (requested: {cfg.model.device})")
50
+ typer.echo(f"backends: {', '.join(cfg.backends)}")
51
+
52
+
53
+ @app.command()
54
+ def generate(
55
+ prompt: str = typer.Option(..., "--prompt", "-p", help="Prompt to generate from."),
56
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
57
+ max_new_tokens: int | None = typer.Option(None, help="Override max new tokens."),
58
+ temperature: float | None = typer.Option(None, help="Override sampling temperature."),
59
+ seed: int | None = typer.Option(None, help="Override random seed."),
60
+ greedy: bool = typer.Option(False, "--greedy", help="Disable sampling (deterministic)."),
61
+ ) -> None:
62
+ """Generate text from PROMPT with the PyTorch backend and print timing."""
63
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
64
+
65
+ from edgellm.models import ModelLoader
66
+ from edgellm.runners import PyTorchRunner
67
+
68
+ cfg = _load_config(config_path)
69
+ if max_new_tokens is not None:
70
+ cfg.generation.max_new_tokens = max_new_tokens
71
+ if temperature is not None:
72
+ cfg.generation.temperature = temperature
73
+ if seed is not None:
74
+ cfg.generation.seed = seed
75
+ if greedy:
76
+ cfg.generation.do_sample = False
77
+
78
+ loaded = ModelLoader(cfg.model).load()
79
+ runner = PyTorchRunner(loaded)
80
+ result = runner.generate(prompt, cfg.generation)
81
+
82
+ typer.echo("\n=== output ===")
83
+ typer.echo(result.text.strip())
84
+ typer.echo("\n=== stats ===")
85
+ typer.echo(f"backend: {result.backend}")
86
+ typer.echo(f"device: {loaded.device}")
87
+ typer.echo(f"prompt tokens: {result.prompt_tokens}")
88
+ typer.echo(f"generated tokens: {result.generated_tokens}")
89
+ typer.echo(f"latency: {result.latency_s:.3f} s")
90
+ typer.echo(f"throughput: {result.tokens_per_second:.2f} tok/s")
91
+
92
+
93
+ @app.command()
94
+ def export(
95
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
96
+ force: bool = typer.Option(False, "--force", help="Re-export even if it already exists."),
97
+ ) -> None:
98
+ """Export the configured model to ONNX (FP32) via Optimum."""
99
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
100
+ from edgellm.export import ONNXExporter
101
+
102
+ cfg = _load_config(config_path)
103
+ out_dir = ONNXExporter(cfg.model, cfg.export).export(force=force)
104
+ typer.echo(f"ONNX export: {out_dir}")
105
+
106
+
107
+ @app.command()
108
+ def snapdragon(
109
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
110
+ precision: str = typer.Option(
111
+ "int8", "--precision", help="Artifact to profile: int8|int4|fp32."
112
+ ),
113
+ device: str = typer.Option("Snapdragon 8 Elite QRD", "--device", help="AI Hub device name."),
114
+ seq: int = typer.Option(64, "--seq", help="Fixed sequence length for compilation."),
115
+ ) -> None:
116
+ """Compile + profile the model on a real Snapdragon NPU via Qualcomm AI Hub."""
117
+ import sys
118
+
119
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "aihub"))
120
+ from run_on_snapdragon import SnapdragonProfiler, check_auth
121
+
122
+ cfg = _load_config(config_path)
123
+ safe = cfg.model.id.replace("/", "__")
124
+ suffix = {"int8": "-int8-dynamic", "int4": "-int4", "fp32": "-fp32"}.get(
125
+ precision, "-int8-dynamic"
126
+ )
127
+ model_dir = Path(cfg.export.output_dir) / f"{safe}{suffix}"
128
+ if not check_auth():
129
+ raise typer.Exit(0)
130
+ SnapdragonProfiler(model_dir, device, seq, 0).run(Path("results/benchmarks.json"))
131
+
132
+
133
+ @app.command()
134
+ def encode(
135
+ prompt: str = typer.Option(..., "--prompt", "-p", help="Prompt to tokenize."),
136
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
137
+ ) -> None:
138
+ """Print space-separated token ids for PROMPT (feeds the C++ harness)."""
139
+ from transformers import AutoTokenizer
140
+
141
+ from edgellm.runners import encode_prompt
142
+
143
+ cfg = _load_config(config_path)
144
+ tok = AutoTokenizer.from_pretrained(cfg.model.id, revision=cfg.model.revision)
145
+ ids = encode_prompt(tok, prompt)["input_ids"][0].tolist()
146
+ typer.echo(" ".join(str(i) for i in ids))
147
+
148
+
149
+ @app.command()
150
+ def decode(
151
+ ids: str = typer.Option(..., "--ids", help="Space/comma-separated token ids to decode."),
152
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
153
+ ) -> None:
154
+ """Decode token ids back to text (e.g. the C++ harness's GENERATED_IDS)."""
155
+ from transformers import AutoTokenizer
156
+
157
+ cfg = _load_config(config_path)
158
+ tok = AutoTokenizer.from_pretrained(cfg.model.id, revision=cfg.model.revision)
159
+ id_list = [int(x) for x in ids.replace(",", " ").split()]
160
+ typer.echo(tok.decode(id_list, skip_special_tokens=True))
161
+
162
+
163
+ @app.command()
164
+ def report(
165
+ results_dir: Path = typer.Option(Path("results"), "--results", help="Results directory."),
166
+ ) -> None:
167
+ """Regenerate the Markdown table + bar charts from benchmarks.json."""
168
+ from edgellm.benchmark import render_markdown
169
+ from edgellm.report import render_charts
170
+
171
+ json_path = results_dir / "benchmarks.json"
172
+ if not json_path.exists():
173
+ typer.echo(f"no results at {json_path}; run 'edgellm benchmark' first.", err=True)
174
+ raise typer.Exit(1)
175
+ (results_dir / "benchmark.md").write_text(render_markdown(json_path))
176
+ chart = render_charts(json_path, results_dir / "benchmark_chart.png")
177
+ typer.echo(f"wrote {results_dir / 'benchmark.md'} and {chart}")
178
+
179
+
180
+ @app.command()
181
+ def quantize(
182
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
183
+ ) -> None:
184
+ """Produce INT8 (ONNX Runtime) and INT4 (block-wise) quantized artifacts."""
185
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
186
+ from edgellm.export import ONNXExporter
187
+ from edgellm.quantize import Quantizer
188
+
189
+ cfg = _load_config(config_path)
190
+ fp32_dir = ONNXExporter(cfg.model, cfg.export).export()
191
+ q = Quantizer(cfg.quantize)
192
+ base = fp32_dir.parent / fp32_dir.name.replace("-fp32", "")
193
+ int8_dir = q.ort_dynamic_int8(fp32_dir, Path(f"{base}-int8-dynamic"))
194
+ int4_dir = q.ort_int4(fp32_dir, Path(f"{base}-int4"))
195
+ typer.echo(f"INT8: {int8_dir}")
196
+ typer.echo(f"INT4: {int4_dir}")
197
+ typer.echo("\nINT4 backend availability on this machine:")
198
+ for backend, state in q.int4_availability().items():
199
+ typer.echo(f" {backend}: {state}")
200
+
201
+
202
+ @app.command()
203
+ def benchmark(
204
+ config_path: Path = typer.Option(DEFAULT_CONFIG, "--config", "-c", help="Path to YAML config."),
205
+ skip_ppl: bool = typer.Option(False, "--skip-ppl", help="Skip perplexity (faster)."),
206
+ results_dir: Path = typer.Option(Path("results"), "--results", help="Output directory."),
207
+ only: str = typer.Option(
208
+ "pt-fp32,ort-fp32,ort-int8,ort-int4,pt-int8",
209
+ "--only",
210
+ help="Comma list of backends to benchmark.",
211
+ ),
212
+ ) -> None:
213
+ """Benchmark FP32/INT8/INT4 across PyTorch + ONNX Runtime and write real numbers."""
214
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
215
+
216
+ from edgellm.benchmark import (
217
+ BenchmarkHarness,
218
+ PerplexityEvaluator,
219
+ measure_size_mb,
220
+ render_markdown,
221
+ save_results,
222
+ )
223
+ from edgellm.export import ONNXExporter
224
+ from edgellm.models import ModelLoader
225
+ from edgellm.quantize import Quantizer
226
+ from edgellm.runners import ORTRunner, PyTorchRunner
227
+
228
+ cfg = _load_config(config_path)
229
+ selected = {s.strip() for s in only.split(",") if s.strip()}
230
+ harness = BenchmarkHarness(cfg.benchmark)
231
+ ppl_eval = None if skip_ppl else PerplexityEvaluator(cfg.benchmark)
232
+ json_path = results_dir / "benchmarks.json"
233
+
234
+ loaded = ModelLoader(cfg.model).load() # FP32 on the best local device (mps/cuda/cpu)
235
+ fp32_dir = ONNXExporter(cfg.model, cfg.export).export()
236
+ quantizer = Quantizer(cfg.quantize)
237
+ base = fp32_dir.parent / fp32_dir.name.replace("-fp32", "")
238
+
239
+ def bench_ort(model_dir: Path, precision: str, name: str):
240
+ runner = ORTRunner(str(model_dir), loaded.tokenizer, name=name)
241
+ ppl = _ppl_ort(runner, ppl_eval, loaded.tokenizer) if ppl_eval else None
242
+ size = measure_size_mb(model_dir, ("*.onnx", "*.onnx_data"))
243
+ return harness.run(
244
+ runner,
245
+ precision=precision,
246
+ device="cpu",
247
+ model_id=cfg.model.id,
248
+ generation=cfg.generation,
249
+ size_mb=size,
250
+ perplexity=ppl,
251
+ )
252
+
253
+ def bench_pt_fp32():
254
+ ppl = (
255
+ _ppl_torch(loaded.model, loaded.device, ppl_eval, loaded.tokenizer)
256
+ if ppl_eval
257
+ else None
258
+ )
259
+ return harness.run(
260
+ PyTorchRunner(loaded),
261
+ precision="fp32",
262
+ device=loaded.device,
263
+ model_id=cfg.model.id,
264
+ generation=cfg.generation,
265
+ size_mb=_pytorch_size_mb(cfg.model.id, cfg.model.revision),
266
+ perplexity=ppl,
267
+ )
268
+
269
+ def bench_pt_int8():
270
+ cpu_loaded = ModelLoader(replace(cfg.model, device="cpu")).load()
271
+ qmodel = Quantizer.pytorch_dynamic_int8(cpu_loaded.model)
272
+ cpu_loaded = replace(cpu_loaded, model=qmodel, device="cpu")
273
+ ppl = _ppl_torch(qmodel, "cpu", ppl_eval, cpu_loaded.tokenizer) if ppl_eval else None
274
+ return harness.run(
275
+ PyTorchRunner(cpu_loaded),
276
+ precision="int8",
277
+ device="cpu",
278
+ model_id=cfg.model.id,
279
+ generation=cfg.generation,
280
+ size_mb=None,
281
+ perplexity=ppl,
282
+ )
283
+
284
+ # (key, human label, thunk). Each runs guarded so one failure can't discard the rest.
285
+ steps = [
286
+ ("pt-fp32", "pytorch fp32", bench_pt_fp32),
287
+ ("ort-fp32", "ort-cpu fp32", lambda: bench_ort(fp32_dir, "fp32", "ort-cpu")),
288
+ (
289
+ "ort-int8",
290
+ "ort-cpu int8",
291
+ lambda: bench_ort(
292
+ quantizer.ort_dynamic_int8(fp32_dir, Path(f"{base}-int8-dynamic")),
293
+ "int8",
294
+ "ort-cpu-int8",
295
+ ),
296
+ ),
297
+ (
298
+ "ort-int4",
299
+ "ort-cpu int4",
300
+ lambda: bench_ort(
301
+ quantizer.ort_int4(fp32_dir, Path(f"{base}-int4")), "int4", "ort-cpu-int4"
302
+ ),
303
+ ),
304
+ ("pt-int8", "pytorch int8 (cpu)", bench_pt_int8),
305
+ ]
306
+
307
+ for key, label, thunk in steps:
308
+ if key not in selected:
309
+ continue
310
+ typer.echo(f"[bench] {label}...")
311
+ try:
312
+ result = thunk()
313
+ except Exception as exc: # noqa: BLE001 - one backend must not sink the others
314
+ typer.echo(f"[bench] {label} FAILED: {type(exc).__name__}: {exc}", err=True)
315
+ continue
316
+ save_results([result], json_path) # persist incrementally
317
+
318
+ md = render_markdown(json_path)
319
+ (results_dir / "benchmark.md").write_text(md)
320
+ typer.echo("\n=== results ===")
321
+ typer.echo(md)
322
+ typer.echo(f"wrote {json_path} and {results_dir / 'benchmark.md'}")
323
+
324
+
325
+ def _ppl_torch(model, device: str, ppl_eval, tokenizer) -> float:
326
+ import torch
327
+
328
+ def forward(ids: torch.Tensor, attn: torch.Tensor) -> torch.Tensor:
329
+ with torch.inference_mode():
330
+ out = model(input_ids=ids.to(device), attention_mask=attn.to(device))
331
+ return out.logits.detach().to("cpu")
332
+
333
+ return ppl_eval.evaluate(forward, tokenizer)
334
+
335
+
336
+ def _ppl_ort(runner, ppl_eval, tokenizer) -> float:
337
+ import torch
338
+
339
+ def forward(ids: torch.Tensor, attn: torch.Tensor) -> torch.Tensor:
340
+ out = runner.model(input_ids=ids, attention_mask=attn)
341
+ return out.logits.detach().to("cpu")
342
+
343
+ return ppl_eval.evaluate(forward, tokenizer)
344
+
345
+
346
+ def _pytorch_size_mb(model_id: str, revision: str) -> float | None:
347
+ """On-disk size (MB) of the model weights in the local HF cache."""
348
+ from huggingface_hub import snapshot_download
349
+
350
+ from edgellm.benchmark import measure_size_mb
351
+
352
+ try:
353
+ snap = Path(
354
+ snapshot_download(
355
+ model_id, revision=revision, allow_patterns=["*.safetensors", "*.bin"]
356
+ )
357
+ )
358
+ except Exception:
359
+ return None
360
+ size = measure_size_mb(snap, ("*.safetensors", "*.bin"))
361
+ return size or None
362
+
363
+
364
+ def main() -> None:
365
+ """Console-script entry point."""
366
+ app()
367
+
368
+
369
+ if __name__ == "__main__":
370
+ main()
edgellm/cli_bench.py ADDED
@@ -0,0 +1,195 @@
1
+ """The ``run`` / ``submit`` / ``leaderboard`` commands — the lightweight path.
2
+
3
+ Registered onto the main Typer app by :mod:`edgellm.cli`. Kept in its own module
4
+ because this is the path a contributor touches, and it should be readable without
5
+ scrolling past the export/quantize authoring commands.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import logging
12
+ from pathlib import Path
13
+
14
+ import typer
15
+
16
+ from edgellm.eval_lite import DEFAULT_EVAL_WINDOWS
17
+ from edgellm.hub import DEFAULT_MODEL, DEFAULT_PRECISIONS
18
+ from edgellm.sweep import (
19
+ DEFAULT_GEN_TOKENS,
20
+ DEFAULT_MEASURED_RUNS,
21
+ DEFAULT_WARMUP_RUNS,
22
+ )
23
+
24
+ COMMUNITY_DIR = Path("results/community")
25
+
26
+
27
+ def _parse_precisions(value: str) -> tuple[str, ...]:
28
+ items = tuple(p.strip() for p in value.split(",") if p.strip())
29
+ if not items:
30
+ raise typer.BadParameter("Give at least one precision, e.g. --precisions fp32,int8,q4")
31
+ return items
32
+
33
+
34
+ def register(app: typer.Typer) -> None:
35
+ """Attach the lightweight commands to ``app``."""
36
+
37
+ @app.command()
38
+ def run(
39
+ model: str = typer.Option(DEFAULT_MODEL, "--model", "-m", help="Hub model id with ONNX."),
40
+ precisions: str = typer.Option(
41
+ ",".join(DEFAULT_PRECISIONS), "--precisions", help="Comma-separated precision list."
42
+ ),
43
+ revision: str = typer.Option("main", "--revision", help="Hub revision to pin."),
44
+ provider: str = typer.Option(
45
+ "CPUExecutionProvider", "--provider", help="ONNX Runtime execution provider."
46
+ ),
47
+ threads: int | None = typer.Option(
48
+ None, "--threads", help="intra-op threads (default: physical core count)."
49
+ ),
50
+ gen_tokens: int = typer.Option(
51
+ DEFAULT_GEN_TOKENS, "--gen-tokens", help="Tokens to generate per run."
52
+ ),
53
+ measured_runs: int = typer.Option(
54
+ DEFAULT_MEASURED_RUNS, "--measured-runs", help="Timed runs per precision."
55
+ ),
56
+ warmup_runs: int = typer.Option(
57
+ DEFAULT_WARMUP_RUNS, "--warmup-runs", help="Untimed warmup runs."
58
+ ),
59
+ eval_windows: int = typer.Option(
60
+ DEFAULT_EVAL_WINDOWS, "--eval-windows", help="512-token perplexity windows."
61
+ ),
62
+ skip_perplexity: bool = typer.Option(
63
+ False, "--skip-perplexity", help="Measure speed and size only (much faster)."
64
+ ),
65
+ out: Path = typer.Option(
66
+ COMMUNITY_DIR, "--out", help="Directory to write the result card into."
67
+ ),
68
+ no_isolate: bool = typer.Option(
69
+ False,
70
+ "--no-isolate",
71
+ help="Measure all precisions in one process (faster, but peak-RAM becomes unreliable).",
72
+ ),
73
+ verbose: bool = typer.Option(False, "--verbose", "-v", help="Debug logging."),
74
+ ) -> None:
75
+ """Benchmark pre-quantized ONNX precisions of MODEL on this machine.
76
+
77
+ Downloads already-quantized artifacts from the Hub — nothing is quantized
78
+ locally, so this needs no PyTorch and no GPU.
79
+ """
80
+ from edgellm.render import console_report
81
+ from edgellm.sweep import SweepSettings, run_sweep, save_card
82
+
83
+ logging.basicConfig(
84
+ level=logging.DEBUG if verbose else logging.WARNING,
85
+ format="%(levelname)s %(name)s: %(message)s",
86
+ )
87
+
88
+ settings = SweepSettings(
89
+ model_id=model,
90
+ precisions=_parse_precisions(precisions),
91
+ revision=revision,
92
+ provider=provider,
93
+ intra_op_threads=threads,
94
+ warmup_runs=warmup_runs,
95
+ measured_runs=measured_runs,
96
+ gen_tokens=gen_tokens,
97
+ eval_windows=eval_windows,
98
+ skip_perplexity=skip_perplexity,
99
+ )
100
+
101
+ typer.echo(f"Benchmarking {model} on this machine ({len(settings.precisions)} precisions).")
102
+ typer.echo("First run downloads the ONNX artifacts; later runs reuse the HF cache.\n")
103
+
104
+ def progress(precision: str, state: str) -> None:
105
+ if state == "start":
106
+ typer.echo(f" {precision:>6} ... ", nl=False)
107
+ elif state == "ok":
108
+ typer.echo("done")
109
+ else:
110
+ typer.echo("skipped")
111
+
112
+ card, errors = run_sweep(settings, on_progress=progress, isolate=not no_isolate)
113
+
114
+ if not card.rows:
115
+ typer.echo("\nNo precision could be measured. Reasons:", err=True)
116
+ for precision, why in errors.items():
117
+ typer.echo(f" {precision}: {why}", err=True)
118
+ raise typer.Exit(1)
119
+
120
+ typer.echo(console_report(card, errors))
121
+
122
+ path = save_card(card, out)
123
+ typer.echo(f"Result card: {path}")
124
+ typer.echo("Share it on the public leaderboard with: quantcost submit")
125
+
126
+ @app.command()
127
+ def models(
128
+ model: str = typer.Option(DEFAULT_MODEL, "--model", "-m", help="Hub model id to inspect."),
129
+ revision: str = typer.Option("main", "--revision"),
130
+ ) -> None:
131
+ """List which ONNX precisions a Hub model actually publishes."""
132
+ from edgellm.hub import available_precisions
133
+
134
+ found = available_precisions(model, revision=revision)
135
+ if not found:
136
+ typer.echo(
137
+ f"{model} publishes no ONNX files under onnx/. Try a model from "
138
+ "https://huggingface.co/onnx-community",
139
+ err=True,
140
+ )
141
+ raise typer.Exit(1)
142
+ typer.echo(f"{model} publishes: {', '.join(found)}")
143
+
144
+ @app.command()
145
+ def leaderboard(
146
+ community: Path = typer.Option(COMMUNITY_DIR, "--dir", help="Directory of result cards."),
147
+ out: Path = typer.Option(
148
+ Path("results/LEADERBOARD.md"), "--out", help="Markdown file to write."
149
+ ),
150
+ json_out: Path | None = typer.Option(
151
+ Path("site/leaderboard.json"), "--json-out", help="JSON for the site to render."
152
+ ),
153
+ ) -> None:
154
+ """Rebuild the leaderboard from every card in the community directory."""
155
+ from edgellm.leaderboard import build_leaderboard, render_leaderboard_markdown
156
+
157
+ cards = sorted(community.glob("*.json"))
158
+ if not cards:
159
+ typer.echo(f"No result cards found in {community}.", err=True)
160
+ raise typer.Exit(1)
161
+
162
+ board = build_leaderboard(cards)
163
+ out.parent.mkdir(parents=True, exist_ok=True)
164
+ out.write_text(render_leaderboard_markdown(board))
165
+ typer.echo(
166
+ f"Wrote {out} from {len(cards)} card(s), {len(board['entries'])} machine-model rows."
167
+ )
168
+
169
+ if json_out is not None:
170
+ json_out.parent.mkdir(parents=True, exist_ok=True)
171
+ json_out.write_text(json.dumps(board, indent=2) + "\n")
172
+ typer.echo(f"Wrote {json_out}")
173
+
174
+ @app.command()
175
+ def submit(
176
+ card: Path | None = typer.Option(
177
+ None, "--card", help="Card to submit (default: the newest in results/community)."
178
+ ),
179
+ community: Path = typer.Option(COMMUNITY_DIR, "--dir"),
180
+ name: str = typer.Option("", "--name", help="Credit line for the leaderboard (optional)."),
181
+ notes: str = typer.Option("", "--notes", help="Anything unusual about this machine."),
182
+ dry_run: bool = typer.Option(False, "--dry-run", help="Validate and print, do not push."),
183
+ ) -> None:
184
+ """Open a pull request adding your result card to the public leaderboard."""
185
+ from edgellm.submit import submit_card
186
+
187
+ target = card
188
+ if target is None:
189
+ cards = sorted(community.glob("*.json"), key=lambda p: p.stat().st_mtime)
190
+ if not cards:
191
+ typer.echo(f"No result card in {community}. Run `quantcost run` first.", err=True)
192
+ raise typer.Exit(1)
193
+ target = cards[-1]
194
+
195
+ submit_card(target, name=name, notes=notes, dry_run=dry_run, echo=typer.echo)
edgellm/config.py ADDED
@@ -0,0 +1,120 @@
1
+ """Typed configuration objects loaded from YAML.
2
+
3
+ The whole project is driven by a single :class:`Config` tree so that every phase
4
+ (loading, export, quantization, benchmarking) reads its settings from one place
5
+ and the CLI can override any field.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field, fields, is_dataclass
11
+ from pathlib import Path
12
+ from typing import Any, get_type_hints
13
+
14
+ import yaml
15
+
16
+
17
+ @dataclass
18
+ class ModelConfig:
19
+ """Which model to load and how to place it on hardware."""
20
+
21
+ id: str = "Qwen/Qwen2.5-0.5B-Instruct"
22
+ revision: str = "main"
23
+ dtype: str = "float32"
24
+ device: str = "auto"
25
+ trust_remote_code: bool = False
26
+
27
+
28
+ @dataclass
29
+ class GenerationConfig:
30
+ """Text-generation decoding parameters."""
31
+
32
+ max_new_tokens: int = 128
33
+ min_new_tokens: int | None = None
34
+ temperature: float = 0.7
35
+ top_p: float = 0.9
36
+ do_sample: bool = True
37
+ seed: int = 0
38
+
39
+
40
+ @dataclass
41
+ class BenchmarkConfig:
42
+ """Settings for the benchmark harness (used from Phase 2 onward)."""
43
+
44
+ warmup_runs: int = 2
45
+ measured_runs: int = 5
46
+ gen_tokens: int = 64
47
+ prompt: str = "Explain what neural network quantization is, in one short paragraph."
48
+ eval_dataset: str = "wikitext"
49
+ eval_config: str = "wikitext-2-raw-v1"
50
+ eval_num_samples: int = 32
51
+ eval_max_length: int = 512
52
+
53
+
54
+ @dataclass
55
+ class ExportConfig:
56
+ """ONNX export settings (used from Phase 2 onward)."""
57
+
58
+ opset: int = 17
59
+ output_dir: str = "artifacts/onnx"
60
+
61
+
62
+ @dataclass
63
+ class Int8Config:
64
+ scheme: str = "dynamic"
65
+ per_channel: bool = True
66
+
67
+
68
+ @dataclass
69
+ class Int4Config:
70
+ method: str = "gptq"
71
+ group_size: int = 128
72
+
73
+
74
+ @dataclass
75
+ class QuantizeConfig:
76
+ """Quantization settings (used from Phase 3 onward)."""
77
+
78
+ int8: Int8Config = field(default_factory=Int8Config)
79
+ int4: Int4Config = field(default_factory=Int4Config)
80
+
81
+
82
+ @dataclass
83
+ class Config:
84
+ """Root configuration tree."""
85
+
86
+ model: ModelConfig = field(default_factory=ModelConfig)
87
+ generation: GenerationConfig = field(default_factory=GenerationConfig)
88
+ benchmark: BenchmarkConfig = field(default_factory=BenchmarkConfig)
89
+ export: ExportConfig = field(default_factory=ExportConfig)
90
+ quantize: QuantizeConfig = field(default_factory=QuantizeConfig)
91
+ backends: list[str] = field(default_factory=lambda: ["pytorch", "ort-cpu"])
92
+
93
+ @classmethod
94
+ def from_yaml(cls, path: str | Path) -> Config:
95
+ """Load a :class:`Config` from a YAML file, filling in defaults."""
96
+ raw = yaml.safe_load(Path(path).read_text()) or {}
97
+ if not isinstance(raw, dict):
98
+ raise ValueError(f"Config root must be a mapping, got {type(raw).__name__}")
99
+ return _from_dict(cls, raw)
100
+
101
+
102
+ def _from_dict(cls: type, data: dict[str, Any]) -> Any:
103
+ """Recursively build a (possibly nested) dataclass from a plain dict.
104
+
105
+ Unknown keys raise, so typos in the YAML fail loudly instead of being ignored.
106
+ """
107
+ kwargs: dict[str, Any] = {}
108
+ known = {f.name for f in fields(cls)}
109
+ # Resolve annotations to real types (needed because `from __future__ import
110
+ # annotations` turns dataclass field types into strings).
111
+ hints = get_type_hints(cls)
112
+ for key, value in data.items():
113
+ if key not in known:
114
+ raise ValueError(f"Unknown config key '{key}' for {cls.__name__}")
115
+ field_type = hints[key]
116
+ if is_dataclass(field_type) and isinstance(value, dict):
117
+ kwargs[key] = _from_dict(field_type, value)
118
+ else:
119
+ kwargs[key] = value
120
+ return cls(**kwargs)