quantcost 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. quantcost-0.2.0/.dockerignore +30 -0
  2. quantcost-0.2.0/.github/pull_request_template.md +23 -0
  3. quantcost-0.2.0/.github/workflows/ci.yml +73 -0
  4. quantcost-0.2.0/.github/workflows/leaderboard.yml +51 -0
  5. quantcost-0.2.0/.github/workflows/release.yml +89 -0
  6. quantcost-0.2.0/.gitignore +44 -0
  7. quantcost-0.2.0/.mailmap +11 -0
  8. quantcost-0.2.0/CONTRIBUTING.md +76 -0
  9. quantcost-0.2.0/Dockerfile +119 -0
  10. quantcost-0.2.0/LICENSE +21 -0
  11. quantcost-0.2.0/PKG-INFO +265 -0
  12. quantcost-0.2.0/README.md +218 -0
  13. quantcost-0.2.0/aihub/run_on_snapdragon.py +183 -0
  14. quantcost-0.2.0/android/app/build.gradle.kts +44 -0
  15. quantcost-0.2.0/android/app/src/main/AndroidManifest.xml +19 -0
  16. quantcost-0.2.0/android/app/src/main/assets/README.md +15 -0
  17. quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/HfTokenizer.kt +36 -0
  18. quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/MainActivity.kt +63 -0
  19. quantcost-0.2.0/android/app/src/main/java/com/edgellm/app/OnnxLlm.kt +130 -0
  20. quantcost-0.2.0/android/app/src/main/res/layout/activity_main.xml +38 -0
  21. quantcost-0.2.0/android/app/src/main/res/values/strings.xml +7 -0
  22. quantcost-0.2.0/android/build.gradle.kts +5 -0
  23. quantcost-0.2.0/android/gradle.properties +3 -0
  24. quantcost-0.2.0/android/settings.gradle.kts +17 -0
  25. quantcost-0.2.0/configs/default.yaml +43 -0
  26. quantcost-0.2.0/cpp/CMakeLists.txt +37 -0
  27. quantcost-0.2.0/cpp/src/main.cpp +254 -0
  28. quantcost-0.2.0/docker-compose.yml +73 -0
  29. quantcost-0.2.0/docs/AUTHORING.md +310 -0
  30. quantcost-0.2.0/edgellm/__init__.py +7 -0
  31. quantcost-0.2.0/edgellm/base.py +44 -0
  32. quantcost-0.2.0/edgellm/benchmark.py +226 -0
  33. quantcost-0.2.0/edgellm/card.py +234 -0
  34. quantcost-0.2.0/edgellm/cli.py +370 -0
  35. quantcost-0.2.0/edgellm/cli_bench.py +195 -0
  36. quantcost-0.2.0/edgellm/config.py +120 -0
  37. quantcost-0.2.0/edgellm/data/SOURCE.md +17 -0
  38. quantcost-0.2.0/edgellm/data/eval_wikitext2.txt +205 -0
  39. quantcost-0.2.0/edgellm/eval_lite.py +128 -0
  40. quantcost-0.2.0/edgellm/export.py +65 -0
  41. quantcost-0.2.0/edgellm/hub.py +125 -0
  42. quantcost-0.2.0/edgellm/leaderboard.py +210 -0
  43. quantcost-0.2.0/edgellm/models.py +91 -0
  44. quantcost-0.2.0/edgellm/ort_lite.py +236 -0
  45. quantcost-0.2.0/edgellm/quantize.py +134 -0
  46. quantcost-0.2.0/edgellm/render.py +164 -0
  47. quantcost-0.2.0/edgellm/report.py +53 -0
  48. quantcost-0.2.0/edgellm/runners.py +170 -0
  49. quantcost-0.2.0/edgellm/submit.py +225 -0
  50. quantcost-0.2.0/edgellm/sweep.py +309 -0
  51. quantcost-0.2.0/edgellm/validate.py +274 -0
  52. quantcost-0.2.0/infra/terraform/README.md +96 -0
  53. quantcost-0.2.0/infra/terraform/main.tf +129 -0
  54. quantcost-0.2.0/infra/terraform/outputs.tf +39 -0
  55. quantcost-0.2.0/infra/terraform/user_data.sh +100 -0
  56. quantcost-0.2.0/infra/terraform/variables.tf +76 -0
  57. quantcost-0.2.0/kernels/CMakeLists.txt +25 -0
  58. quantcost-0.2.0/kernels/int8_gemm.cpp +157 -0
  59. quantcost-0.2.0/pyproject.toml +95 -0
  60. quantcost-0.2.0/results/LEADERBOARD.md +40 -0
  61. quantcost-0.2.0/results/benchmark.md +7 -0
  62. quantcost-0.2.0/results/benchmark_chart.png +0 -0
  63. quantcost-0.2.0/results/benchmarks.json +72 -0
  64. quantcost-0.2.0/results/community/apple-m4-darwin--qwen2-5-0-5b-instruct.json +64 -0
  65. quantcost-0.2.0/results/community/apple-m4-darwin--smollm2-135m-instruct.json +64 -0
  66. quantcost-0.2.0/scripts/run_all.sh +56 -0
  67. quantcost-0.2.0/site/.gitignore +1 -0
  68. quantcost-0.2.0/site/index.html +1601 -0
  69. quantcost-0.2.0/site/leaderboard.json +171 -0
  70. quantcost-0.2.0/site/vercel.json +5 -0
  71. quantcost-0.2.0/tests/test_bench_lite.py +358 -0
  72. quantcost-0.2.0/tests/test_benchmark.py +48 -0
  73. quantcost-0.2.0/tests/test_cli.py +23 -0
  74. quantcost-0.2.0/tests/test_config.py +43 -0
  75. quantcost-0.2.0/tests/test_report.py +48 -0
  76. quantcost-0.2.0/tests/test_validate.py +181 -0
@@ -0,0 +1,30 @@
1
+ # Keep the build context small and hermetic. Anything the image genuinely needs
2
+ # is COPY'd explicitly in the Dockerfile; anything generated is mounted at run
3
+ # time via docker-compose.yml.
4
+
5
+ .git
6
+ .github
7
+ .venv
8
+ venv
9
+ __pycache__
10
+ **/__pycache__
11
+ *.pyc
12
+ .pytest_cache
13
+ .ruff_cache
14
+ .mypy_cache
15
+
16
+ # Generated artifacts and measurements: mounted, never baked in. Baking them
17
+ # would let a stale model or an old benchmark table ride along inside the image.
18
+ artifacts/
19
+ results/
20
+
21
+ # Not part of the Linux container build.
22
+ android/
23
+ aihub/
24
+ site/
25
+ tests/
26
+
27
+ *.onnx
28
+ *.safetensors
29
+ *.gguf
30
+ .DS_Store
@@ -0,0 +1,23 @@
1
+ <!--
2
+ Submitting benchmark results? `quantcost submit` fills this in for you.
3
+ If you ran it, you can delete this template — your numbers are already below.
4
+
5
+ Submitting a code change instead? Delete this and describe the change.
6
+ -->
7
+
8
+ ## Benchmark results
9
+
10
+ - **CPU**:
11
+ - **OS**:
12
+ - **Model**:
13
+
14
+ <!-- Paste the table `quantcost run` printed, or just leave the card file. -->
15
+
16
+ ### Checklist
17
+
18
+ - [ ] The card was produced by `quantcost run` (not edited by hand)
19
+ - [ ] The machine was reasonably idle — no "varied by more than 15%" warning
20
+ - [ ] `results/community/` is the only directory this PR touches
21
+
22
+ Numbers are never edited after the fact. If something looks wrong, re-run and
23
+ submit the new card — a corrected measurement is welcome, a hand-tuned one is not.
@@ -0,0 +1,73 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ lint-and-test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ matrix:
13
+ # The default install must keep working on every Python it claims to
14
+ # support, including the newest — people run this on whatever they have.
15
+ python-version: ["3.10", "3.11", "3.12", "3.13"]
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+
19
+ - name: Set up Python
20
+ uses: actions/setup-python@v5
21
+ with:
22
+ python-version: ${{ matrix.python-version }}
23
+
24
+ - name: Install (default deps only — no torch)
25
+ run: |
26
+ python -m pip install --upgrade pip
27
+ pip install -e ".[dev]"
28
+
29
+ - name: Ruff (lint)
30
+ run: ruff check .
31
+
32
+ - name: Ruff (format check)
33
+ run: ruff format --check .
34
+
35
+ - name: Pytest
36
+ run: pytest
37
+
38
+ - name: Assert the default install stays torch-free
39
+ # The whole contribution path depends on this: if importing the CLI ever
40
+ # drags in torch, a one-command benchmark becomes a multi-gigabyte one.
41
+ run: |
42
+ python - <<'PY'
43
+ import sys
44
+ import edgellm.cli # noqa: F401
45
+ heavy = {m.split(".")[0] for m in sys.modules} & {
46
+ "torch", "transformers", "optimum", "datasets", "onnx", "matplotlib"
47
+ }
48
+ assert not heavy, f"CLI import pulled in heavy dependencies: {sorted(heavy)}"
49
+ print("CLI import is torch-free")
50
+ PY
51
+
52
+ validate-submitted-cards:
53
+ name: Validate result cards
54
+ runs-on: ubuntu-latest
55
+ steps:
56
+ - uses: actions/checkout@v4
57
+
58
+ - uses: actions/setup-python@v5
59
+ with:
60
+ python-version: "3.12"
61
+
62
+ - name: Install
63
+ run: |
64
+ python -m pip install --upgrade pip
65
+ pip install -e .
66
+
67
+ - name: Validate every card in results/community
68
+ run: python -m edgellm.validate results/community
69
+
70
+ - name: Confirm the leaderboard still builds from these cards
71
+ # Only that it *builds*. A contributor's PR adds a card and nothing else;
72
+ # regenerating LEADERBOARD.md is the merge job's responsibility, not theirs.
73
+ run: quantcost leaderboard --out /tmp/leaderboard.md --json-out /tmp/leaderboard.json
@@ -0,0 +1,51 @@
1
+ name: Rebuild leaderboard
2
+
3
+ # Contributors submit a result card and nothing else. The leaderboard is a
4
+ # generated artifact, so it is rebuilt here after a merge and committed back —
5
+ # asking a first-time contributor to also regenerate a Markdown table is the
6
+ # kind of friction that loses the submission.
7
+ on:
8
+ push:
9
+ branches: [main]
10
+ paths:
11
+ - "results/community/**"
12
+ - "edgellm/leaderboard.py"
13
+ - "edgellm/validate.py"
14
+ workflow_dispatch:
15
+
16
+ permissions:
17
+ contents: write
18
+
19
+ concurrency:
20
+ group: leaderboard
21
+ cancel-in-progress: false
22
+
23
+ jobs:
24
+ rebuild:
25
+ runs-on: ubuntu-latest
26
+ steps:
27
+ - uses: actions/checkout@v4
28
+
29
+ - uses: actions/setup-python@v5
30
+ with:
31
+ python-version: "3.12"
32
+
33
+ - name: Install
34
+ run: |
35
+ python -m pip install --upgrade pip
36
+ pip install -e .
37
+
38
+ - name: Rebuild from every submitted card
39
+ run: quantcost leaderboard
40
+
41
+ - name: Commit if it changed
42
+ run: |
43
+ if git diff --quiet -- results/LEADERBOARD.md site/leaderboard.json; then
44
+ echo "Leaderboard unchanged."
45
+ exit 0
46
+ fi
47
+ git config user.name "github-actions[bot]"
48
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
49
+ git add results/LEADERBOARD.md site/leaderboard.json
50
+ git commit -m "Rebuild leaderboard from submitted cards"
51
+ git push
@@ -0,0 +1,89 @@
1
+ name: Release to PyPI
2
+
3
+ # Publishes on a version tag. Uses PyPI Trusted Publishing (OIDC), so there is no
4
+ # API token stored in this repository at all.
5
+ #
6
+ # One-time setup on PyPI, before the first tag. The project does not exist yet,
7
+ # so this is the *pending* publisher form, not the per-project settings page:
8
+ # https://pypi.org/manage/account/publishing/
9
+ # PyPI Project Name: quantcost
10
+ # Owner: vijay-kapse
11
+ # Repository name: EdgeLLM (the repo, not the package)
12
+ # Workflow name: release.yml
13
+ # Environment name: pypi (required - this job declares it)
14
+ #
15
+ # Then: git tag v0.2.0 && git push origin v0.2.0
16
+ on:
17
+ push:
18
+ tags: ["v*"]
19
+ workflow_dispatch:
20
+
21
+ jobs:
22
+ build:
23
+ runs-on: ubuntu-latest
24
+ steps:
25
+ - uses: actions/checkout@v4
26
+
27
+ - uses: actions/setup-python@v5
28
+ with:
29
+ python-version: "3.12"
30
+
31
+ - name: Build sdist and wheel
32
+ run: |
33
+ python -m pip install --upgrade pip build
34
+ python -m build
35
+
36
+ - name: Verify the eval corpus is inside the wheel
37
+ # A wheel missing the pinned corpus installs fine and then fails the hash
38
+ # check on first use, so this is checked before anything is published.
39
+ run: |
40
+ python - <<'PY'
41
+ import glob, zipfile
42
+ wheel = sorted(glob.glob("dist/*.whl"))[-1]
43
+ names = zipfile.ZipFile(wheel).namelist()
44
+ assert "edgellm/data/eval_wikitext2.txt" in names, f"corpus missing from {wheel}"
45
+ print(f"{wheel}: corpus bundled")
46
+ PY
47
+
48
+ - name: Check the tag matches the packaged version
49
+ if: startsWith(github.ref, 'refs/tags/v')
50
+ run: |
51
+ python - <<'PY'
52
+ import glob, os, re
53
+ tag = os.environ["GITHUB_REF_NAME"].lstrip("v")
54
+ wheel = os.path.basename(sorted(glob.glob("dist/*.whl"))[-1])
55
+ version = re.match(r"quantcost-([^-]+)-", wheel).group(1)
56
+ assert version == tag, f"tag {tag!r} does not match built version {version!r}"
57
+ print(f"tag and version agree: {version}")
58
+ PY
59
+
60
+ - name: Smoke-test the built wheel in a clean environment
61
+ run: |
62
+ python -m venv /tmp/smoke
63
+ /tmp/smoke/bin/pip install --quiet dist/*.whl
64
+ /tmp/smoke/bin/quantcost --help > /dev/null
65
+ /tmp/smoke/bin/python -c "
66
+ import sys, edgellm.cli
67
+ heavy = {m.split('.')[0] for m in sys.modules} & {'torch', 'transformers'}
68
+ assert not heavy, heavy
69
+ print('wheel installs clean and imports torch-free')
70
+ "
71
+
72
+ - uses: actions/upload-artifact@v4
73
+ with:
74
+ name: dist
75
+ path: dist/
76
+
77
+ publish:
78
+ needs: build
79
+ runs-on: ubuntu-latest
80
+ environment: pypi
81
+ permissions:
82
+ id-token: write # required for Trusted Publishing
83
+ steps:
84
+ - uses: actions/download-artifact@v4
85
+ with:
86
+ name: dist
87
+ path: dist/
88
+
89
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,44 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ .eggs/
6
+ build/
7
+ dist/
8
+ .venv/
9
+ venv/
10
+ .pytest_cache/
11
+ .ruff_cache/
12
+ .mypy_cache/
13
+
14
+ # Project artifacts (models, exports, quantized weights are large — never commit)
15
+ artifacts/
16
+ *.onnx
17
+ *.onnx_data
18
+ *.gguf
19
+ *.bin
20
+ *.safetensors
21
+ hf_cache/
22
+ .cache/
23
+
24
+ # Results: commit curated tables/charts, ignore bulky raw dumps
25
+ results/raw/
26
+
27
+ # C++ build
28
+ cpp/build/
29
+ kernels/build/
30
+ *.o
31
+ *.obj
32
+
33
+ # Android
34
+ android/.gradle/
35
+ android/build/
36
+ android/app/build/
37
+ android/local.properties
38
+ android/.idea/
39
+ *.apk
40
+
41
+ # OS / editor
42
+ .DS_Store
43
+ .idea/
44
+ .vscode/
@@ -0,0 +1,11 @@
1
+ # Canonical identity for every commit in this repository.
2
+ # All of these are the same author; git split them across machines,
3
+ # GitHub noreply addresses, university accounts and assistant tooling.
4
+ Vijay Kapse <vijayskofficial@gmail.com> <118760102+vijay-kapse@users.noreply.github.com>
5
+ Vijay Kapse <vijayskofficial@gmail.com> <vijay-kapse@users.noreply.github.com>
6
+ Vijay Kapse <vijayskofficial@gmail.com> <vj@Vijays-MacBook-Air.local>
7
+ Vijay Kapse <vijayskofficial@gmail.com> <vsk@Vijays-MacBook-Pro.local>
8
+ Vijay Kapse <vijayskofficial@gmail.com> <vkapse@binghamton.edu>
9
+ Vijay Kapse <vijayskofficial@gmail.com> <21f1000202@ds.study.iitm.ac.in>
10
+ Vijay Kapse <vijayskofficial@gmail.com> <noreply@anthropic.com>
11
+ Vijay Kapse <vijayskofficial@gmail.com> Vijay Suryakant Kapse <vijayskofficial@gmail.com>
@@ -0,0 +1,76 @@
1
+ # Contributing
2
+
3
+ ## Submitting benchmark results
4
+
5
+ This is the most useful thing you can contribute, and it takes about two minutes.
6
+
7
+ ```bash
8
+ pip install quantcost
9
+ quantcost run
10
+ quantcost submit
11
+ ```
12
+
13
+ `submit` validates the card, forks this repo, commits your result and opens the
14
+ pull request. If you do not have the [GitHub CLI](https://cli.github.com)
15
+ installed it prints a prefilled link instead, so you are never stuck.
16
+
17
+ ### What makes a good submission
18
+
19
+ - **An idle machine.** Close the browser and the build you have running. If the
20
+ report prints a *"varied by more than 15%"* warning, re-run before submitting.
21
+ - **Default settings.** `quantcost run` with no flags produces a card that
22
+ is comparable with everyone else's. Non-default runs are still accepted and
23
+ listed — they just are not ranked, because a 64-token run and a 32-token run
24
+ do not measure the same thing.
25
+ - **Leave `--threads` alone** unless you are deliberately investigating it. The
26
+ default pins ONNX Runtime to your machine's *performance* cores, and the card
27
+ records how that was determined (`thread_policy`). This matters more than it
28
+ sounds: on a heterogeneous CPU one thread scheduled onto an efficiency core
29
+ gates the whole parallel region. Measured on an Apple M4 (4 performance + 6
30
+ efficiency), pinning all ten physical cores cost int8 66% of its throughput
31
+ and doubled run-to-run spread, while barely moving fp32. A card run across
32
+ both tiers is measuring something different from everyone else's.
33
+ - **Interesting hardware especially welcome.** Raspberry Pi, old ThinkPads,
34
+ Snapdragon laptops, bare-metal ARM servers, anything unusual. The point of the
35
+ leaderboard is the *spread* across real machines, not the top of the list.
36
+
37
+ ### What gets rejected, and why
38
+
39
+ CI runs `python -m edgellm.validate results/community` on every PR. It rejects
40
+ cards that are internally inconsistent (throughput that disagrees with the
41
+ latency and token count it claims), scored against a modified eval corpus, run
42
+ with an unpinned thread count, or carrying identifying information about your
43
+ machine. The intent is to keep the leaderboard comparable, not to accuse anyone:
44
+ if your card is rejected, the message says what to fix, and re-running almost
45
+ always fixes it.
46
+
47
+ Results are reviewed, not blindly trusted. Nothing here can prove a number came
48
+ from real silicon — but every input is pinned (model revision, artifact bytes,
49
+ corpus hash, prompt hash, token budget), so anyone can re-run the exact
50
+ configuration recorded in your card and compare.
51
+
52
+ ### Privacy
53
+
54
+ A card contains your CPU model string, architecture, core count, RAM rounded to
55
+ the nearest gigabyte, OS name and release, and version strings. It does **not**
56
+ contain your hostname, username, file paths, IP or MAC address, or any machine
57
+ identifier. `edgellm/card.py::fingerprint` is the only function that reads your
58
+ machine, so you can check that claim in one place. The `--name` flag is optional
59
+ and the only thing that attaches an identity to a submission.
60
+
61
+ ## Code changes
62
+
63
+ ```bash
64
+ pip install -e ".[dev]"
65
+ ruff check . && ruff format --check . && pytest
66
+ ```
67
+
68
+ Adding the quantization/export path (PyTorch, Optimum, ONNX) needs the extra:
69
+
70
+ ```bash
71
+ pip install -e ".[dev,quantize]"
72
+ ```
73
+
74
+ Keep the default install torch-free. CI asserts that importing the CLI pulls in
75
+ no heavy dependency, because the contribution flow above depends on a fast
76
+ install. If you need torch, import it inside the function that uses it.
@@ -0,0 +1,119 @@
1
+ # syntax=docker/dockerfile:1.7
2
+ #
3
+ # EdgeLLM container image.
4
+ #
5
+ # Builds both native binaries against a pinned ONNX Runtime release, then ships
6
+ # them alongside the Python pipeline in a slim runtime image:
7
+ #
8
+ # edgellm_infer - C++17 harness, KV-cache greedy decode via the ORT C++ API
9
+ # int8_gemm - scalar vs. hand-vectorized INT8 GEMM microbenchmark
10
+ #
11
+ # The image is architecture-aware. On arm64 the kernel compiles the ARM NEON
12
+ # SDOT path; on amd64 it compiles the AVX2 path. That is deliberate: running the
13
+ # same image on an Apple Silicon host and on an x86 cloud instance produces two
14
+ # genuinely different SIMD backends to compare, rather than one emulated number.
15
+ #
16
+ # Build:
17
+ # docker build -t edgellm .
18
+ # docker build -t edgellm --build-arg ORT_VERSION=1.20.1 .
19
+ #
20
+ # See docker-compose.yml for the pipeline / inference / benchmark entry points.
21
+
22
+ ARG UBUNTU_VERSION=22.04
23
+ ARG PYTHON_VERSION=3.11
24
+
25
+ # ---------------------------------------------------------------- builder ----
26
+ FROM ubuntu:${UBUNTU_VERSION} AS builder
27
+
28
+ # Pinned rather than "latest": benchmark numbers are only comparable across runs
29
+ # if the inference runtime is held fixed.
30
+ ARG ORT_VERSION=1.22.0
31
+ ARG TARGETARCH
32
+
33
+ RUN apt-get update && apt-get install -y --no-install-recommends \
34
+ build-essential \
35
+ cmake \
36
+ ca-certificates \
37
+ curl \
38
+ && rm -rf /var/lib/apt/lists/*
39
+
40
+ # ONNX Runtime publishes prebuilt Linux archives as x64 / aarch64; Docker's
41
+ # TARGETARCH is amd64 / arm64. Translate, with a uname fallback for the legacy
42
+ # (non-BuildKit) builder where TARGETARCH is not populated.
43
+ RUN set -eux; \
44
+ arch="${TARGETARCH:-}"; \
45
+ if [ -z "${arch}" ]; then \
46
+ case "$(uname -m)" in \
47
+ x86_64) arch=amd64 ;; \
48
+ aarch64|arm64) arch=arm64 ;; \
49
+ esac; \
50
+ fi; \
51
+ case "${arch}" in \
52
+ amd64) ort_arch=x64 ;; \
53
+ arm64) ort_arch=aarch64 ;; \
54
+ *) echo "unsupported architecture: '${arch}'" >&2; exit 1 ;; \
55
+ esac; \
56
+ url="https://github.com/microsoft/onnxruntime/releases/download/v${ORT_VERSION}/onnxruntime-linux-${ort_arch}-${ORT_VERSION}.tgz"; \
57
+ echo "fetching ${url}"; \
58
+ curl -fsSL -o /tmp/ort.tgz "${url}"; \
59
+ mkdir -p /opt/onnxruntime; \
60
+ tar -xzf /tmp/ort.tgz -C /opt/onnxruntime --strip-components=1; \
61
+ rm /tmp/ort.tgz; \
62
+ test -f /opt/onnxruntime/include/onnxruntime_cxx_api.h; \
63
+ test -e /opt/onnxruntime/lib/libonnxruntime.so
64
+
65
+ # cpp/CMakeLists.txt resolves ONNX Runtime from ORT_HOME (its documented escape
66
+ # hatch for non-Homebrew hosts), so no CMake changes are needed to build here.
67
+ ENV ORT_HOME=/opt/onnxruntime
68
+
69
+ WORKDIR /src
70
+ COPY cpp/ cpp/
71
+ COPY kernels/ kernels/
72
+
73
+ RUN cmake -S cpp -B build/cpp -DCMAKE_BUILD_TYPE=Release \
74
+ && cmake --build build/cpp -j "$(nproc)"
75
+
76
+ RUN cmake -S kernels -B build/kernels -DCMAKE_BUILD_TYPE=Release \
77
+ && cmake --build build/kernels -j "$(nproc)"
78
+
79
+ # ---------------------------------------------------------------- runtime ----
80
+ FROM python:${PYTHON_VERSION}-slim-bookworm AS runtime
81
+
82
+ # libgomp is ONNX Runtime's OpenMP dependency; without it the .so fails to load.
83
+ RUN apt-get update && apt-get install -y --no-install-recommends \
84
+ libgomp1 \
85
+ ca-certificates \
86
+ && rm -rf /var/lib/apt/lists/*
87
+
88
+ COPY --from=builder /opt/onnxruntime/lib/ /opt/onnxruntime/lib/
89
+ COPY --from=builder /src/build/cpp/edgellm_infer /usr/local/bin/edgellm_infer
90
+ COPY --from=builder /src/build/kernels/int8_gemm /usr/local/bin/int8_gemm
91
+
92
+ ENV LD_LIBRARY_PATH=/opt/onnxruntime/lib \
93
+ PYTHONUNBUFFERED=1 \
94
+ HF_HOME=/cache/huggingface
95
+
96
+ WORKDIR /app
97
+
98
+ # Dependency metadata first so the (slow) install layer caches independently of
99
+ # source edits. hatchling reads README.md at build time, so it has to come too.
100
+ COPY pyproject.toml README.md ./
101
+ COPY edgellm/ edgellm/
102
+
103
+ # CPU-only torch on x86 avoids pulling ~2.5 GB of CUDA wheels that this image
104
+ # has no GPU to use. The CPU index has no aarch64 wheels, so arm64 takes the
105
+ # ordinary PyPI build, which is already CPU-only.
106
+ RUN set -eux; \
107
+ case "$(uname -m)" in \
108
+ x86_64) pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu ;; \
109
+ *) pip install --no-cache-dir torch ;; \
110
+ esac; \
111
+ pip install --no-cache-dir ".[onnx,eval,report]"
112
+
113
+ COPY configs/ configs/
114
+ COPY scripts/ scripts/
115
+
116
+ # Written to by the pipeline; compose mounts host directories over both.
117
+ RUN mkdir -p /app/artifacts /app/results /cache/huggingface
118
+
119
+ CMD ["edgellm", "--help"]
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Vijay Kapse
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.