quantcost 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantcost-0.2.0/.dockerignore +30 -0
- quantcost-0.2.0/.github/pull_request_template.md +23 -0
- quantcost-0.2.0/.github/workflows/ci.yml +73 -0
- quantcost-0.2.0/.github/workflows/leaderboard.yml +51 -0
- quantcost-0.2.0/.github/workflows/release.yml +89 -0
- quantcost-0.2.0/.gitignore +44 -0
- quantcost-0.2.0/.mailmap +11 -0
- quantcost-0.2.0/CONTRIBUTING.md +76 -0
- quantcost-0.2.0/Dockerfile +119 -0
- quantcost-0.2.0/LICENSE +21 -0
- quantcost-0.2.0/PKG-INFO +265 -0
- quantcost-0.2.0/README.md +218 -0
- quantcost-0.2.0/aihub/run_on_snapdragon.py +183 -0
- quantcost-0.2.0/android/app/build.gradle.kts +44 -0
- quantcost-0.2.0/android/app/src/main/AndroidManifest.xml +19 -0
- quantcost-0.2.0/android/app/src/main/assets/README.md +15 -0
- quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/HfTokenizer.kt +36 -0
- quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/MainActivity.kt +63 -0
- quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/OnnxLlm.kt +130 -0
- quantcost-0.2.0/android/app/src/main/res/layout/activity_main.xml +38 -0
- quantcost-0.2.0/android/app/src/main/res/values/strings.xml +7 -0
- quantcost-0.2.0/android/build.gradle.kts +5 -0
- quantcost-0.2.0/android/gradle.properties +3 -0
- quantcost-0.2.0/android/settings.gradle.kts +17 -0
- quantcost-0.2.0/configs/default.yaml +43 -0
- quantcost-0.2.0/cpp/CMakeLists.txt +37 -0
- quantcost-0.2.0/cpp/src/main.cpp +254 -0
- quantcost-0.2.0/docker-compose.yml +73 -0
- quantcost-0.2.0/docs/AUTHORING.md +310 -0
- quantcost-0.2.0/edgellm/__init__.py +7 -0
- quantcost-0.2.0/edgellm/base.py +44 -0
- quantcost-0.2.0/edgellm/benchmark.py +226 -0
- quantcost-0.2.0/edgellm/card.py +234 -0
- quantcost-0.2.0/edgellm/cli.py +370 -0
- quantcost-0.2.0/edgellm/cli_bench.py +195 -0
- quantcost-0.2.0/edgellm/config.py +120 -0
- quantcost-0.2.0/edgellm/data/SOURCE.md +17 -0
- quantcost-0.2.0/edgellm/data/eval_wikitext2.txt +205 -0
- quantcost-0.2.0/edgellm/eval_lite.py +128 -0
- quantcost-0.2.0/edgellm/export.py +65 -0
- quantcost-0.2.0/edgellm/hub.py +125 -0
- quantcost-0.2.0/edgellm/leaderboard.py +210 -0
- quantcost-0.2.0/edgellm/models.py +91 -0
- quantcost-0.2.0/edgellm/ort_lite.py +236 -0
- quantcost-0.2.0/edgellm/quantize.py +134 -0
- quantcost-0.2.0/edgellm/render.py +164 -0
- quantcost-0.2.0/edgellm/report.py +53 -0
- quantcost-0.2.0/edgellm/runners.py +170 -0
- quantcost-0.2.0/edgellm/submit.py +225 -0
- quantcost-0.2.0/edgellm/sweep.py +309 -0
- quantcost-0.2.0/edgellm/validate.py +274 -0
- quantcost-0.2.0/infra/terraform/README.md +96 -0
- quantcost-0.2.0/infra/terraform/main.tf +129 -0
- quantcost-0.2.0/infra/terraform/outputs.tf +39 -0
- quantcost-0.2.0/infra/terraform/user_data.sh +100 -0
- quantcost-0.2.0/infra/terraform/variables.tf +76 -0
- quantcost-0.2.0/kernels/CMakeLists.txt +25 -0
- quantcost-0.2.0/kernels/int8_gemm.cpp +157 -0
- quantcost-0.2.0/pyproject.toml +95 -0
- quantcost-0.2.0/results/LEADERBOARD.md +40 -0
- quantcost-0.2.0/results/benchmark.md +7 -0
- quantcost-0.2.0/results/benchmark_chart.png +0 -0
- quantcost-0.2.0/results/benchmarks.json +72 -0
- quantcost-0.2.0/results/community/apple-m4-darwin--qwen2-5-0-5b-instruct.json +64 -0
- quantcost-0.2.0/results/community/apple-m4-darwin--smollm2-135m-instruct.json +64 -0
- quantcost-0.2.0/scripts/run_all.sh +56 -0
- quantcost-0.2.0/site/.gitignore +1 -0
- quantcost-0.2.0/site/index.html +1601 -0
- quantcost-0.2.0/site/leaderboard.json +171 -0
- quantcost-0.2.0/site/vercel.json +5 -0
- quantcost-0.2.0/tests/test_bench_lite.py +358 -0
- quantcost-0.2.0/tests/test_benchmark.py +48 -0
- quantcost-0.2.0/tests/test_cli.py +23 -0
- quantcost-0.2.0/tests/test_config.py +43 -0
- quantcost-0.2.0/tests/test_report.py +48 -0
- quantcost-0.2.0/tests/test_validate.py +181 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Keep the build context small and hermetic. Anything the image genuinely needs
|
|
2
|
+
# is COPY'd explicitly in the Dockerfile; anything generated is mounted at run
|
|
3
|
+
# time via docker-compose.yml.
|
|
4
|
+
|
|
5
|
+
.git
|
|
6
|
+
.github
|
|
7
|
+
.venv
|
|
8
|
+
venv
|
|
9
|
+
__pycache__
|
|
10
|
+
**/__pycache__
|
|
11
|
+
*.pyc
|
|
12
|
+
.pytest_cache
|
|
13
|
+
.ruff_cache
|
|
14
|
+
.mypy_cache
|
|
15
|
+
|
|
16
|
+
# Generated artifacts and measurements: mounted, never baked in. Baking them
|
|
17
|
+
# would let a stale model or an old benchmark table ride along inside the image.
|
|
18
|
+
artifacts/
|
|
19
|
+
results/
|
|
20
|
+
|
|
21
|
+
# Not part of the Linux container build.
|
|
22
|
+
android/
|
|
23
|
+
aihub/
|
|
24
|
+
site/
|
|
25
|
+
tests/
|
|
26
|
+
|
|
27
|
+
*.onnx
|
|
28
|
+
*.safetensors
|
|
29
|
+
*.gguf
|
|
30
|
+
.DS_Store
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
Submitting benchmark results? `quantcost submit` fills this in for you.
|
|
3
|
+
If you ran it, you can delete this template — your numbers are already below.
|
|
4
|
+
|
|
5
|
+
Submitting a code change instead? Delete this and describe the change.
|
|
6
|
+
-->
|
|
7
|
+
|
|
8
|
+
## Benchmark results
|
|
9
|
+
|
|
10
|
+
- **CPU**:
|
|
11
|
+
- **OS**:
|
|
12
|
+
- **Model**:
|
|
13
|
+
|
|
14
|
+
<!-- Paste the table `quantcost run` printed, or just leave the card file. -->
|
|
15
|
+
|
|
16
|
+
### Checklist
|
|
17
|
+
|
|
18
|
+
- [ ] The card was produced by `quantcost run` (not edited by hand)
|
|
19
|
+
- [ ] The machine was reasonably idle — no "varied by more than 15%" warning
|
|
20
|
+
- [ ] `results/community/` is the only directory this PR touches
|
|
21
|
+
|
|
22
|
+
Numbers are never edited after the fact. If something looks wrong, re-run and
|
|
23
|
+
submit the new card — a corrected measurement is welcome, a hand-tuned one is not.
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
lint-and-test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
matrix:
|
|
13
|
+
# The default install must keep working on every Python it claims to
|
|
14
|
+
# support, including the newest — people run this on whatever they have.
|
|
15
|
+
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
|
|
19
|
+
- name: Set up Python
|
|
20
|
+
uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: ${{ matrix.python-version }}
|
|
23
|
+
|
|
24
|
+
- name: Install (default deps only — no torch)
|
|
25
|
+
run: |
|
|
26
|
+
python -m pip install --upgrade pip
|
|
27
|
+
pip install -e ".[dev]"
|
|
28
|
+
|
|
29
|
+
- name: Ruff (lint)
|
|
30
|
+
run: ruff check .
|
|
31
|
+
|
|
32
|
+
- name: Ruff (format check)
|
|
33
|
+
run: ruff format --check .
|
|
34
|
+
|
|
35
|
+
- name: Pytest
|
|
36
|
+
run: pytest
|
|
37
|
+
|
|
38
|
+
- name: Assert the default install stays torch-free
|
|
39
|
+
# The whole contribution path depends on this: if importing the CLI ever
|
|
40
|
+
# drags in torch, a one-command benchmark becomes a multi-gigabyte one.
|
|
41
|
+
run: |
|
|
42
|
+
python - <<'PY'
|
|
43
|
+
import sys
|
|
44
|
+
import edgellm.cli # noqa: F401
|
|
45
|
+
heavy = {m.split(".")[0] for m in sys.modules} & {
|
|
46
|
+
"torch", "transformers", "optimum", "datasets", "onnx", "matplotlib"
|
|
47
|
+
}
|
|
48
|
+
assert not heavy, f"CLI import pulled in heavy dependencies: {sorted(heavy)}"
|
|
49
|
+
print("CLI import is torch-free")
|
|
50
|
+
PY
|
|
51
|
+
|
|
52
|
+
validate-submitted-cards:
|
|
53
|
+
name: Validate result cards
|
|
54
|
+
runs-on: ubuntu-latest
|
|
55
|
+
steps:
|
|
56
|
+
- uses: actions/checkout@v4
|
|
57
|
+
|
|
58
|
+
- uses: actions/setup-python@v5
|
|
59
|
+
with:
|
|
60
|
+
python-version: "3.12"
|
|
61
|
+
|
|
62
|
+
- name: Install
|
|
63
|
+
run: |
|
|
64
|
+
python -m pip install --upgrade pip
|
|
65
|
+
pip install -e .
|
|
66
|
+
|
|
67
|
+
- name: Validate every card in results/community
|
|
68
|
+
run: python -m edgellm.validate results/community
|
|
69
|
+
|
|
70
|
+
- name: Confirm the leaderboard still builds from these cards
|
|
71
|
+
# Only that it *builds*. A contributor's PR adds a card and nothing else;
|
|
72
|
+
# regenerating LEADERBOARD.md is the merge job's responsibility, not theirs.
|
|
73
|
+
run: quantcost leaderboard --out /tmp/leaderboard.md --json-out /tmp/leaderboard.json
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
name: Rebuild leaderboard
|
|
2
|
+
|
|
3
|
+
# Contributors submit a result card and nothing else. The leaderboard is a
|
|
4
|
+
# generated artifact, so it is rebuilt here after a merge and committed back —
|
|
5
|
+
# asking a first-time contributor to also regenerate a Markdown table is the
|
|
6
|
+
# kind of friction that loses the submission.
|
|
7
|
+
on:
|
|
8
|
+
push:
|
|
9
|
+
branches: [main]
|
|
10
|
+
paths:
|
|
11
|
+
- "results/community/**"
|
|
12
|
+
- "edgellm/leaderboard.py"
|
|
13
|
+
- "edgellm/validate.py"
|
|
14
|
+
workflow_dispatch:
|
|
15
|
+
|
|
16
|
+
permissions:
|
|
17
|
+
contents: write
|
|
18
|
+
|
|
19
|
+
concurrency:
|
|
20
|
+
group: leaderboard
|
|
21
|
+
cancel-in-progress: false
|
|
22
|
+
|
|
23
|
+
jobs:
|
|
24
|
+
rebuild:
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
steps:
|
|
27
|
+
- uses: actions/checkout@v4
|
|
28
|
+
|
|
29
|
+
- uses: actions/setup-python@v5
|
|
30
|
+
with:
|
|
31
|
+
python-version: "3.12"
|
|
32
|
+
|
|
33
|
+
- name: Install
|
|
34
|
+
run: |
|
|
35
|
+
python -m pip install --upgrade pip
|
|
36
|
+
pip install -e .
|
|
37
|
+
|
|
38
|
+
- name: Rebuild from every submitted card
|
|
39
|
+
run: quantcost leaderboard
|
|
40
|
+
|
|
41
|
+
- name: Commit if it changed
|
|
42
|
+
run: |
|
|
43
|
+
if git diff --quiet -- results/LEADERBOARD.md site/leaderboard.json; then
|
|
44
|
+
echo "Leaderboard unchanged."
|
|
45
|
+
exit 0
|
|
46
|
+
fi
|
|
47
|
+
git config user.name "github-actions[bot]"
|
|
48
|
+
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
|
49
|
+
git add results/LEADERBOARD.md site/leaderboard.json
|
|
50
|
+
git commit -m "Rebuild leaderboard from submitted cards"
|
|
51
|
+
git push
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
name: Release to PyPI
|
|
2
|
+
|
|
3
|
+
# Publishes on a version tag. Uses PyPI Trusted Publishing (OIDC), so there is no
|
|
4
|
+
# API token stored in this repository at all.
|
|
5
|
+
#
|
|
6
|
+
# One-time setup on PyPI, before the first tag. The project does not exist yet,
|
|
7
|
+
# so this is the *pending* publisher form, not the per-project settings page:
|
|
8
|
+
# https://pypi.org/manage/account/publishing/
|
|
9
|
+
# PyPI Project Name: quantcost
|
|
10
|
+
# Owner: vijay-kapse
|
|
11
|
+
# Repository name: EdgeLLM (the repo, not the package)
|
|
12
|
+
# Workflow name: release.yml
|
|
13
|
+
# Environment name: pypi (required - this job declares it)
|
|
14
|
+
#
|
|
15
|
+
# Then: git tag v0.2.0 && git push origin v0.2.0
|
|
16
|
+
on:
|
|
17
|
+
push:
|
|
18
|
+
tags: ["v*"]
|
|
19
|
+
workflow_dispatch:
|
|
20
|
+
|
|
21
|
+
jobs:
|
|
22
|
+
build:
|
|
23
|
+
runs-on: ubuntu-latest
|
|
24
|
+
steps:
|
|
25
|
+
- uses: actions/checkout@v4
|
|
26
|
+
|
|
27
|
+
- uses: actions/setup-python@v5
|
|
28
|
+
with:
|
|
29
|
+
python-version: "3.12"
|
|
30
|
+
|
|
31
|
+
- name: Build sdist and wheel
|
|
32
|
+
run: |
|
|
33
|
+
python -m pip install --upgrade pip build
|
|
34
|
+
python -m build
|
|
35
|
+
|
|
36
|
+
- name: Verify the eval corpus is inside the wheel
|
|
37
|
+
# A wheel missing the pinned corpus installs fine and then fails the hash
|
|
38
|
+
# check on first use, so this is checked before anything is published.
|
|
39
|
+
run: |
|
|
40
|
+
python - <<'PY'
|
|
41
|
+
import glob, zipfile
|
|
42
|
+
wheel = sorted(glob.glob("dist/*.whl"))[-1]
|
|
43
|
+
names = zipfile.ZipFile(wheel).namelist()
|
|
44
|
+
assert "edgellm/data/eval_wikitext2.txt" in names, f"corpus missing from {wheel}"
|
|
45
|
+
print(f"{wheel}: corpus bundled")
|
|
46
|
+
PY
|
|
47
|
+
|
|
48
|
+
- name: Check the tag matches the packaged version
|
|
49
|
+
if: startsWith(github.ref, 'refs/tags/v')
|
|
50
|
+
run: |
|
|
51
|
+
python - <<'PY'
|
|
52
|
+
import glob, os, re
|
|
53
|
+
tag = os.environ["GITHUB_REF_NAME"].lstrip("v")
|
|
54
|
+
wheel = os.path.basename(sorted(glob.glob("dist/*.whl"))[-1])
|
|
55
|
+
version = re.match(r"quantcost-([^-]+)-", wheel).group(1)
|
|
56
|
+
assert version == tag, f"tag {tag!r} does not match built version {version!r}"
|
|
57
|
+
print(f"tag and version agree: {version}")
|
|
58
|
+
PY
|
|
59
|
+
|
|
60
|
+
- name: Smoke-test the built wheel in a clean environment
|
|
61
|
+
run: |
|
|
62
|
+
python -m venv /tmp/smoke
|
|
63
|
+
/tmp/smoke/bin/pip install --quiet dist/*.whl
|
|
64
|
+
/tmp/smoke/bin/quantcost --help > /dev/null
|
|
65
|
+
/tmp/smoke/bin/python -c "
|
|
66
|
+
import sys, edgellm.cli
|
|
67
|
+
heavy = {m.split('.')[0] for m in sys.modules} & {'torch', 'transformers'}
|
|
68
|
+
assert not heavy, heavy
|
|
69
|
+
print('wheel installs clean and imports torch-free')
|
|
70
|
+
"
|
|
71
|
+
|
|
72
|
+
- uses: actions/upload-artifact@v4
|
|
73
|
+
with:
|
|
74
|
+
name: dist
|
|
75
|
+
path: dist/
|
|
76
|
+
|
|
77
|
+
publish:
|
|
78
|
+
needs: build
|
|
79
|
+
runs-on: ubuntu-latest
|
|
80
|
+
environment: pypi
|
|
81
|
+
permissions:
|
|
82
|
+
id-token: write # required for Trusted Publishing
|
|
83
|
+
steps:
|
|
84
|
+
- uses: actions/download-artifact@v4
|
|
85
|
+
with:
|
|
86
|
+
name: dist
|
|
87
|
+
path: dist/
|
|
88
|
+
|
|
89
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
.eggs/
|
|
6
|
+
build/
|
|
7
|
+
dist/
|
|
8
|
+
.venv/
|
|
9
|
+
venv/
|
|
10
|
+
.pytest_cache/
|
|
11
|
+
.ruff_cache/
|
|
12
|
+
.mypy_cache/
|
|
13
|
+
|
|
14
|
+
# Project artifacts (models, exports, quantized weights are large — never commit)
|
|
15
|
+
artifacts/
|
|
16
|
+
*.onnx
|
|
17
|
+
*.onnx_data
|
|
18
|
+
*.gguf
|
|
19
|
+
*.bin
|
|
20
|
+
*.safetensors
|
|
21
|
+
hf_cache/
|
|
22
|
+
.cache/
|
|
23
|
+
|
|
24
|
+
# Results: commit curated tables/charts, ignore bulky raw dumps
|
|
25
|
+
results/raw/
|
|
26
|
+
|
|
27
|
+
# C++ build
|
|
28
|
+
cpp/build/
|
|
29
|
+
kernels/build/
|
|
30
|
+
*.o
|
|
31
|
+
*.obj
|
|
32
|
+
|
|
33
|
+
# Android
|
|
34
|
+
android/.gradle/
|
|
35
|
+
android/build/
|
|
36
|
+
android/app/build/
|
|
37
|
+
android/local.properties
|
|
38
|
+
android/.idea/
|
|
39
|
+
*.apk
|
|
40
|
+
|
|
41
|
+
# OS / editor
|
|
42
|
+
.DS_Store
|
|
43
|
+
.idea/
|
|
44
|
+
.vscode/
|
quantcost-0.2.0/.mailmap
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Canonical identity for every commit in this repository.
|
|
2
|
+
# All of these are the same author; git split them across machines,
|
|
3
|
+
# GitHub noreply addresses, university accounts and assistant tooling.
|
|
4
|
+
Vijay Kapse <vijayskofficial@gmail.com> <118760102+vijay-kapse@users.noreply.github.com>
|
|
5
|
+
Vijay Kapse <vijayskofficial@gmail.com> <vijay-kapse@users.noreply.github.com>
|
|
6
|
+
Vijay Kapse <vijayskofficial@gmail.com> <vj@Vijays-MacBook-Air.local>
|
|
7
|
+
Vijay Kapse <vijayskofficial@gmail.com> <vsk@Vijays-MacBook-Pro.local>
|
|
8
|
+
Vijay Kapse <vijayskofficial@gmail.com> <vkapse@binghamton.edu>
|
|
9
|
+
Vijay Kapse <vijayskofficial@gmail.com> <21f1000202@ds.study.iitm.ac.in>
|
|
10
|
+
Vijay Kapse <vijayskofficial@gmail.com> <noreply@anthropic.com>
|
|
11
|
+
Vijay Kapse <vijayskofficial@gmail.com> Vijay Suryakant Kapse <vijayskofficial@gmail.com>
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
## Submitting benchmark results
|
|
4
|
+
|
|
5
|
+
This is the most useful thing you can contribute, and it takes about two minutes.
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install quantcost
|
|
9
|
+
quantcost run
|
|
10
|
+
quantcost submit
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
`submit` validates the card, forks this repo, commits your result and opens the
|
|
14
|
+
pull request. If you do not have the [GitHub CLI](https://cli.github.com)
|
|
15
|
+
installed it prints a prefilled link instead, so you are never stuck.
|
|
16
|
+
|
|
17
|
+
### What makes a good submission
|
|
18
|
+
|
|
19
|
+
- **An idle machine.** Close the browser and the build you have running. If the
|
|
20
|
+
report prints a *"varied by more than 15%"* warning, re-run before submitting.
|
|
21
|
+
- **Default settings.** `quantcost run` with no flags produces a card that
|
|
22
|
+
is comparable with everyone else's. Non-default runs are still accepted and
|
|
23
|
+
listed — they just are not ranked, because a 64-token run and a 32-token run
|
|
24
|
+
do not measure the same thing.
|
|
25
|
+
- **Leave `--threads` alone** unless you are deliberately investigating it. The
|
|
26
|
+
default pins ONNX Runtime to your machine's *performance* cores, and the card
|
|
27
|
+
records how that was determined (`thread_policy`). This matters more than it
|
|
28
|
+
sounds: on a heterogeneous CPU one thread scheduled onto an efficiency core
|
|
29
|
+
gates the whole parallel region. Measured on an Apple M4 (4 performance + 6
|
|
30
|
+
efficiency), pinning all ten physical cores cost int8 66% of its throughput
|
|
31
|
+
and doubled run-to-run spread, while barely moving fp32. A card run across
|
|
32
|
+
both tiers is measuring something different from everyone else's.
|
|
33
|
+
- **Interesting hardware especially welcome.** Raspberry Pi, old ThinkPads,
|
|
34
|
+
Snapdragon laptops, bare-metal ARM servers, anything unusual. The point of the
|
|
35
|
+
leaderboard is the *spread* across real machines, not the top of the list.
|
|
36
|
+
|
|
37
|
+
### What gets rejected, and why
|
|
38
|
+
|
|
39
|
+
CI runs `python -m edgellm.validate results/community` on every PR. It rejects
|
|
40
|
+
cards that are internally inconsistent (throughput that disagrees with the
|
|
41
|
+
latency and token count it claims), scored against a modified eval corpus, run
|
|
42
|
+
with an unpinned thread count, or carrying identifying information about your
|
|
43
|
+
machine. The intent is to keep the leaderboard comparable, not to accuse anyone:
|
|
44
|
+
if your card is rejected, the message says what to fix, and re-running almost
|
|
45
|
+
always fixes it.
|
|
46
|
+
|
|
47
|
+
Results are reviewed, not blindly trusted. Nothing here can prove a number came
|
|
48
|
+
from real silicon — but every input is pinned (model revision, artifact bytes,
|
|
49
|
+
corpus hash, prompt hash, token budget), so anyone can re-run the exact
|
|
50
|
+
configuration recorded in your card and compare.
|
|
51
|
+
|
|
52
|
+
### Privacy
|
|
53
|
+
|
|
54
|
+
A card contains your CPU model string, architecture, core count, RAM rounded to
|
|
55
|
+
the nearest gigabyte, OS name and release, and version strings. It does **not**
|
|
56
|
+
contain your hostname, username, file paths, IP or MAC address, or any machine
|
|
57
|
+
identifier. `edgellm/card.py::fingerprint` is the only function that reads your
|
|
58
|
+
machine, so you can check that claim in one place. The `--name` flag is optional
|
|
59
|
+
and the only thing that attaches an identity to a submission.
|
|
60
|
+
|
|
61
|
+
## Code changes
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install -e ".[dev]"
|
|
65
|
+
ruff check . && ruff format --check . && pytest
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Adding the quantization/export path (PyTorch, Optimum, ONNX) needs the extra:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install -e ".[dev,quantize]"
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Keep the default install torch-free. CI asserts that importing the CLI pulls in
|
|
75
|
+
no heavy dependency, because the contribution flow above depends on a fast
|
|
76
|
+
install. If you need torch, import it inside the function that uses it.
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# syntax=docker/dockerfile:1.7
|
|
2
|
+
#
|
|
3
|
+
# EdgeLLM container image.
|
|
4
|
+
#
|
|
5
|
+
# Builds both native binaries against a pinned ONNX Runtime release, then ships
|
|
6
|
+
# them alongside the Python pipeline in a slim runtime image:
|
|
7
|
+
#
|
|
8
|
+
# edgellm_infer - C++17 harness, KV-cache greedy decode via the ORT C++ API
|
|
9
|
+
# int8_gemm - scalar vs. hand-vectorized INT8 GEMM microbenchmark
|
|
10
|
+
#
|
|
11
|
+
# The image is architecture-aware. On arm64 the kernel compiles the ARM NEON
|
|
12
|
+
# SDOT path; on amd64 it compiles the AVX2 path. That is deliberate: running the
|
|
13
|
+
# same image on an Apple Silicon host and on an x86 cloud instance produces two
|
|
14
|
+
# genuinely different SIMD backends to compare, rather than one emulated number.
|
|
15
|
+
#
|
|
16
|
+
# Build:
|
|
17
|
+
# docker build -t edgellm .
|
|
18
|
+
# docker build -t edgellm --build-arg ORT_VERSION=1.20.1 .
|
|
19
|
+
#
|
|
20
|
+
# See docker-compose.yml for the pipeline / inference / benchmark entry points.
|
|
21
|
+
|
|
22
|
+
ARG UBUNTU_VERSION=22.04
|
|
23
|
+
ARG PYTHON_VERSION=3.11
|
|
24
|
+
|
|
25
|
+
# ---------------------------------------------------------------- builder ----
|
|
26
|
+
FROM ubuntu:${UBUNTU_VERSION} AS builder
|
|
27
|
+
|
|
28
|
+
# Pinned rather than "latest": benchmark numbers are only comparable across runs
|
|
29
|
+
# if the inference runtime is held fixed.
|
|
30
|
+
ARG ORT_VERSION=1.22.0
|
|
31
|
+
ARG TARGETARCH
|
|
32
|
+
|
|
33
|
+
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
34
|
+
build-essential \
|
|
35
|
+
cmake \
|
|
36
|
+
ca-certificates \
|
|
37
|
+
curl \
|
|
38
|
+
&& rm -rf /var/lib/apt/lists/*
|
|
39
|
+
|
|
40
|
+
# ONNX Runtime publishes prebuilt Linux archives as x64 / aarch64; Docker's
|
|
41
|
+
# TARGETARCH is amd64 / arm64. Translate, with a uname fallback for the legacy
|
|
42
|
+
# (non-BuildKit) builder where TARGETARCH is not populated.
|
|
43
|
+
RUN set -eux; \
|
|
44
|
+
arch="${TARGETARCH:-}"; \
|
|
45
|
+
if [ -z "${arch}" ]; then \
|
|
46
|
+
case "$(uname -m)" in \
|
|
47
|
+
x86_64) arch=amd64 ;; \
|
|
48
|
+
aarch64|arm64) arch=arm64 ;; \
|
|
49
|
+
esac; \
|
|
50
|
+
fi; \
|
|
51
|
+
case "${arch}" in \
|
|
52
|
+
amd64) ort_arch=x64 ;; \
|
|
53
|
+
arm64) ort_arch=aarch64 ;; \
|
|
54
|
+
*) echo "unsupported architecture: '${arch}'" >&2; exit 1 ;; \
|
|
55
|
+
esac; \
|
|
56
|
+
url="https://github.com/microsoft/onnxruntime/releases/download/v${ORT_VERSION}/onnxruntime-linux-${ort_arch}-${ORT_VERSION}.tgz"; \
|
|
57
|
+
echo "fetching ${url}"; \
|
|
58
|
+
curl -fsSL -o /tmp/ort.tgz "${url}"; \
|
|
59
|
+
mkdir -p /opt/onnxruntime; \
|
|
60
|
+
tar -xzf /tmp/ort.tgz -C /opt/onnxruntime --strip-components=1; \
|
|
61
|
+
rm /tmp/ort.tgz; \
|
|
62
|
+
test -f /opt/onnxruntime/include/onnxruntime_cxx_api.h; \
|
|
63
|
+
test -e /opt/onnxruntime/lib/libonnxruntime.so
|
|
64
|
+
|
|
65
|
+
# cpp/CMakeLists.txt resolves ONNX Runtime from ORT_HOME (its documented escape
|
|
66
|
+
# hatch for non-Homebrew hosts), so no CMake changes are needed to build here.
|
|
67
|
+
ENV ORT_HOME=/opt/onnxruntime
|
|
68
|
+
|
|
69
|
+
WORKDIR /src
|
|
70
|
+
COPY cpp/ cpp/
|
|
71
|
+
COPY kernels/ kernels/
|
|
72
|
+
|
|
73
|
+
RUN cmake -S cpp -B build/cpp -DCMAKE_BUILD_TYPE=Release \
|
|
74
|
+
&& cmake --build build/cpp -j "$(nproc)"
|
|
75
|
+
|
|
76
|
+
RUN cmake -S kernels -B build/kernels -DCMAKE_BUILD_TYPE=Release \
|
|
77
|
+
&& cmake --build build/kernels -j "$(nproc)"
|
|
78
|
+
|
|
79
|
+
# ---------------------------------------------------------------- runtime ----
|
|
80
|
+
FROM python:${PYTHON_VERSION}-slim-bookworm AS runtime
|
|
81
|
+
|
|
82
|
+
# libgomp is ONNX Runtime's OpenMP dependency; without it the .so fails to load.
|
|
83
|
+
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
84
|
+
libgomp1 \
|
|
85
|
+
ca-certificates \
|
|
86
|
+
&& rm -rf /var/lib/apt/lists/*
|
|
87
|
+
|
|
88
|
+
COPY --from=builder /opt/onnxruntime/lib/ /opt/onnxruntime/lib/
|
|
89
|
+
COPY --from=builder /src/build/cpp/edgellm_infer /usr/local/bin/edgellm_infer
|
|
90
|
+
COPY --from=builder /src/build/kernels/int8_gemm /usr/local/bin/int8_gemm
|
|
91
|
+
|
|
92
|
+
ENV LD_LIBRARY_PATH=/opt/onnxruntime/lib \
|
|
93
|
+
PYTHONUNBUFFERED=1 \
|
|
94
|
+
HF_HOME=/cache/huggingface
|
|
95
|
+
|
|
96
|
+
WORKDIR /app
|
|
97
|
+
|
|
98
|
+
# Dependency metadata first so the (slow) install layer caches independently of
|
|
99
|
+
# source edits. hatchling reads README.md at build time, so it has to come too.
|
|
100
|
+
COPY pyproject.toml README.md ./
|
|
101
|
+
COPY edgellm/ edgellm/
|
|
102
|
+
|
|
103
|
+
# CPU-only torch on x86 avoids pulling ~2.5 GB of CUDA wheels that this image
|
|
104
|
+
# has no GPU to use. The CPU index has no aarch64 wheels, so arm64 takes the
|
|
105
|
+
# ordinary PyPI build, which is already CPU-only.
|
|
106
|
+
RUN set -eux; \
|
|
107
|
+
case "$(uname -m)" in \
|
|
108
|
+
x86_64) pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu ;; \
|
|
109
|
+
*) pip install --no-cache-dir torch ;; \
|
|
110
|
+
esac; \
|
|
111
|
+
pip install --no-cache-dir ".[onnx,eval,report]"
|
|
112
|
+
|
|
113
|
+
COPY configs/ configs/
|
|
114
|
+
COPY scripts/ scripts/
|
|
115
|
+
|
|
116
|
+
# Written to by the pipeline; compose mounts host directories over both.
|
|
117
|
+
RUN mkdir -p /app/artifacts /app/results /cache/huggingface
|
|
118
|
+
|
|
119
|
+
CMD ["edgellm", "--help"]
|
quantcost-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vijay Kapse
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|