vramcalc 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. vramcalc-0.1.0/.github/workflows/ci.yml +34 -0
  2. vramcalc-0.1.0/.gitignore +9 -0
  3. vramcalc-0.1.0/CHANGELOG.md +8 -0
  4. vramcalc-0.1.0/LICENSE +21 -0
  5. vramcalc-0.1.0/PKG-INFO +237 -0
  6. vramcalc-0.1.0/README.md +215 -0
  7. vramcalc-0.1.0/benchmarks/make_table.py +42 -0
  8. vramcalc-0.1.0/benchmarks/results.jsonl +19 -0
  9. vramcalc-0.1.0/docs/design.md +522 -0
  10. vramcalc-0.1.0/pyproject.toml +35 -0
  11. vramcalc-0.1.0/src/vramcalc/__init__.py +9 -0
  12. vramcalc-0.1.0/src/vramcalc/arch.py +111 -0
  13. vramcalc-0.1.0/src/vramcalc/bench.py +220 -0
  14. vramcalc-0.1.0/src/vramcalc/cli.py +126 -0
  15. vramcalc-0.1.0/src/vramcalc/estimate.py +219 -0
  16. vramcalc-0.1.0/src/vramcalc/gpus.py +76 -0
  17. vramcalc-0.1.0/src/vramcalc/loaders.py +57 -0
  18. vramcalc-0.1.0/src/vramcalc/memory.py +114 -0
  19. vramcalc-0.1.0/src/vramcalc/suggest.py +45 -0
  20. vramcalc-0.1.0/tests/__init__.py +0 -0
  21. vramcalc-0.1.0/tests/configs/gemma-2-2b.json +1 -0
  22. vramcalc-0.1.0/tests/configs/llama-2-7b.json +1 -0
  23. vramcalc-0.1.0/tests/configs/llama-3.1-8b.json +1 -0
  24. vramcalc-0.1.0/tests/configs/llama-3.2-1b.json +1 -0
  25. vramcalc-0.1.0/tests/configs/llama-3.2-3b.json +1 -0
  26. vramcalc-0.1.0/tests/configs/mistral-7b-v0.3.json +1 -0
  27. vramcalc-0.1.0/tests/configs/qwen2.5-1.5b.json +1 -0
  28. vramcalc-0.1.0/tests/configs/qwen2.5-7b.json +1 -0
  29. vramcalc-0.1.0/tests/conftest.py +26 -0
  30. vramcalc-0.1.0/tests/test_arch.py +74 -0
  31. vramcalc-0.1.0/tests/test_bench_format.py +79 -0
  32. vramcalc-0.1.0/tests/test_cli.py +64 -0
  33. vramcalc-0.1.0/tests/test_estimate.py +107 -0
  34. vramcalc-0.1.0/tests/test_gpus.py +60 -0
  35. vramcalc-0.1.0/tests/test_loaders.py +76 -0
  36. vramcalc-0.1.0/tests/test_make_table.py +29 -0
  37. vramcalc-0.1.0/tests/test_memory.py +96 -0
  38. vramcalc-0.1.0/tests/test_suggest.py +59 -0
  39. vramcalc-0.1.0/uv.lock +1600 -0
@@ -0,0 +1,34 @@
1
+ name: ci
2
+ on:
3
+ push:
4
+ branches: [main]
5
+ tags: ["v*"]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ matrix:
13
+ python: ["3.10", "3.12"]
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: astral-sh/setup-uv@v5
17
+ with:
18
+ python-version: ${{ matrix.python }}
19
+ - run: uv sync --extra dev
20
+ - run: uv run ruff check .
21
+ - run: uv run pytest
22
+
23
+ publish:
24
+ needs: test
25
+ if: startsWith(github.ref, 'refs/tags/v')
26
+ runs-on: ubuntu-latest
27
+ environment: pypi
28
+ permissions:
29
+ id-token: write
30
+ steps:
31
+ - uses: actions/checkout@v4
32
+ - uses: astral-sh/setup-uv@v5
33
+ - run: uv build
34
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,9 @@
1
+ .serena/
2
+ .env
3
+ __pycache__/
4
+ *.egg-info/
5
+ dist/
6
+ .venv/
7
+ .pytest_cache/
8
+ .ruff_cache/
9
+ models/
@@ -0,0 +1,8 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 (2026-10-11)
4
+
5
+ - Inference, full fine-tuning and LoRA memory estimation for Llama, Mistral, Qwen2 and Gemma dense models
6
+ - `fits()` judgment with single-adjustment suggestions
7
+ - `bench` subcommand measuring real memory and comparing per bucket
8
+ - CLI: `vramcalc MODEL`, `vramcalc bench`, `vramcalc gpus`
vramcalc-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Seonho Hong
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,237 @@
1
+ Metadata-Version: 2.5
2
+ Name: vramcalc
3
+ Version: 0.1.0
4
+ Summary: Predict GPU VRAM usage of Llama-family models from config.json and check whether they fit a GPU
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Classifier: License :: OSI Approved :: MIT License
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
10
+ Requires-Python: >=3.10
11
+ Provides-Extra: bench
12
+ Requires-Dist: nvidia-ml-py; extra == 'bench'
13
+ Requires-Dist: peft>=0.12; extra == 'bench'
14
+ Requires-Dist: torch>=2.3; extra == 'bench'
15
+ Requires-Dist: transformers<5,>=4.46; extra == 'bench'
16
+ Provides-Extra: dev
17
+ Requires-Dist: pytest; extra == 'dev'
18
+ Requires-Dist: ruff; extra == 'dev'
19
+ Provides-Extra: hf
20
+ Requires-Dist: huggingface-hub; extra == 'hf'
21
+ Description-Content-Type: text/markdown
22
+
23
+ # vramcalc
24
+
25
+ [English](#english) | [한국어](#한국어)
26
+
27
+ ---
28
+
29
+ ## English
30
+
31
+ Predict how much GPU memory an LLM needs before you download the weights or rent the GPU. vramcalc reads only the Hugging Face `config.json`, tells you whether the model fits a given GPU, and when it does not, which single change would make it fit. A `bench` subcommand measures real memory on your GPU and compares it with the prediction, bucket by bucket.
32
+
33
+ ```python
34
+ from vramcalc import estimate
35
+
36
+ e = estimate("meta-llama/Llama-3.1-8B", mode="train", lora_rank=16,
37
+ precision="pure", seq_len=4096)
38
+ print(e.breakdown())
39
+ r = e.fits("RTX 3090")
40
+ print(r.message)
41
+ for s in r.suggestions:
42
+ print(s.change, s.margin_bytes)
43
+ ```
44
+
45
+ ```
46
+ $ vramcalc meta-llama/Llama-3.1-8B --train --lora 16 --precision pure --seq 4096 --gpu "RTX 3090"
47
+ weights 16.12 GB
48
+ gradients 0.05 GB
49
+ optimizer 0.11 GB
50
+ activations 36.92 GB
51
+ step_workspace 0.05 GB
52
+ total 53.19 GB
53
+ overhead 0.75 GB
54
+
55
+ short by 28.9 GB on RTX 3090; 1 single adjustment fits
56
+ -> seq_len=512: 21.6 GB (changes training objective)
57
+ ```
58
+
59
+ ### Install
60
+
61
+ | Command | What you get |
62
+ |---|---|
63
+ | `pip install vramcalc` | Estimator and CLI. No dependencies. Takes a local `config.json` or a model directory |
64
+ | `pip install "vramcalc[hf]"` | Hugging Face Hub repo ids as input (`huggingface_hub`) |
65
+ | `pip install "vramcalc[bench]"` | `vramcalc bench` for real measurements (`torch`, `transformers`, `peft`, `nvidia-ml-py`) |
66
+
67
+ Offline or on-premises: pass a model directory, `estimate("/path/to/model")`. With `HF_HUB_OFFLINE=1`, Hub ids resolve from the local cache only.
68
+
69
+ ### Why
70
+
71
+ "Will this model run on this GPU with these settings?" is a question you want answered before downloading 16 GB of weights or paying for an H100 hour. Existing calculators mostly size the weights and treat activations, optimizer state and autocast copies loosely, which is where training estimates drift by several gigabytes. vramcalc keeps one formula per memory term and ships a `bench` command that measures the same configuration on a real GPU, so every term can be checked against the allocator. The formulas are fixed; there is no learned correction factor. Measurements are evidence for fixing a formula, not a fudge applied on top.
72
+
73
+ ### Scope
74
+
75
+ - Models: dense decoder-only models with `model_type` of `llama`, `mistral`, `qwen2`, `gemma` or `gemma2`. GQA, QKV bias (Qwen2), tied embeddings and Gemma 2's four norms per layer are read from the config.
76
+ - Modes: inference (with KV cache), full fine-tuning, LoRA on `q_proj`, `k_proj`, `v_proj`, `o_proj`.
77
+ - Knobs: `dtype` (fp32/bf16/fp16), `batch`, `seq_len`, `grad_checkpoint`, `optimizer` (adamw/sgd), `precision` (amp/pure), `attn_impl` (eager/sdpa/flash).
78
+
79
+ `precision="amp"` models HF Trainer with `bf16=True` (fp32 weights, bf16 autocast). `precision="pure"` models loading with `torch_dtype=bf16` and training as is. The two differ by almost 2x in memory, so pick the one you actually run.
80
+
81
+ Out of scope for 0.1: MoE, MLA, quantization (int8/int4/QLoRA), distributed training (ZeRO/FSDP), multimodal and encoder models, and sliding-window KV caches (currently counted as global attention, which is conservative).
82
+
83
+ ### What is counted
84
+
85
+ The prediction targets PyTorch's caching allocator (`allocated` memory). CUDA context and other memory outside the allocator is kept separate as `overhead` (default 0.75 GB) and only added in `fits()`.
86
+
87
+ | Term | Meaning |
88
+ |---|---|
89
+ | `weights` | parameter count x storage dtype; for LoRA, frozen base plus fp32 adapters |
90
+ | `kv_cache` | inference only, `2 x L x b x s x kv x bytes` |
91
+ | `gradients`, `optimizer` | over trainable parameters; AdamW keeps two moments, SGD one |
92
+ | `activations` | per-layer tensors saved for backward (SwiGLU, RMSNorm) plus the fp32 logits chain at the `log_softmax` backward peak |
93
+ | `autocast_cache` | bf16 weight copies alive during an amp forward pass |
94
+ | `step_workspace` | temporary buffer of the foreach optimizer during `step()` |
95
+
96
+ `total = weights + kv_cache + gradients + optimizer + max(activations + autocast_cache, step_workspace)`. The full formulas are in `docs/design.md`, section 5.
97
+
98
+ ### Validation
99
+
100
+ `vramcalc bench` loads the model with the same arguments, records `torch.cuda.max_memory_allocated()`, and compares five buckets: weights, states, peak, total, overhead.
101
+
102
+ ```
103
+ vramcalc bench /path/to/Llama-3.2-1B --train --precision pure --record benchmarks/results.jsonl
104
+ ```
105
+
106
+ Environment: NVIDIA RTX A6000 48 GB, torch 2.14.1+cu130, transformers 4.57.6, peft 0.21.2. Llama weights are gated, so the `unsloth/` mirrors of the same checkpoints were used. All 19 runs land within +/-4.3% on `total`; `weights` is exact in every run.
107
+
108
+ | model | mode | setting | GPU | predicted | measured | error |
109
+ |---|---|---|---|---|---|---|
110
+ | Llama-3.2-1B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 3.08 GB | 3.08 GB | -0.1% |
111
+ | Llama-3.2-1B | inference | bf16 b=1 s=4096 sdpa | RTX A6000 | 3.69 GB | 3.68 GB | +0.2% |
112
+ | Llama-3.2-1B | inference | bf16 b=1 s=2048 eager | RTX A6000 | 3.90 GB | 3.95 GB | -1.2% |
113
+ | Llama-3.2-3B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 7.21 GB | 7.21 GB | +0.0% |
114
+ | Llama-3.1-8B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 16.89 GB | 16.88 GB | +0.0% |
115
+ | Qwen2.5-1.5B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 3.78 GB | 3.78 GB | -0.1% |
116
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full pure | RTX A6000 | 17.52 GB | 17.51 GB | +0.0% |
117
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full pure ckpt | RTX A6000 | 13.95 GB | 13.76 GB | +1.4% |
118
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full amp | RTX A6000 | 29.88 GB | 30.55 GB | -2.2% |
119
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full amp ckpt | RTX A6000 | 26.30 GB | 25.22 GB | +4.3% |
120
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 eager full pure | RTX A6000 | 30.41 GB | 30.26 GB | +0.5% |
121
+ | Llama-3.2-3B | train | bf16 b=1 s=2048 sdpa full pure ckpt | RTX A6000 | 32.13 GB | 32.15 GB | -0.1% |
122
+ | Qwen2.5-1.5B | train | bf16 b=1 s=2048 sdpa full pure | RTX A6000 | 23.17 GB | 23.40 GB | -1.0% |
123
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 10.17 GB | 10.13 GB | +0.4% |
124
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 amp | RTX A6000 | 15.11 GB | 14.52 GB | +4.1% |
125
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 amp ckpt | RTX A6000 | 9.22 GB | 9.51 GB | -3.1% |
126
+ | Llama-3.2-3B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 18.84 GB | 18.98 GB | -0.8% |
127
+ | Llama-3.1-8B | train | bf16 b=1 s=2048 sdpa lora=16 pure ckpt | RTX A6000 | 20.97 GB | 20.55 GB | +2.0% |
128
+ | Qwen2.5-1.5B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 13.99 GB | 13.78 GB | +1.5% |
129
+
130
+ `pure` is `torch_dtype=bf16` training, `amp` is fp32 weights with bf16 autocast, `ckpt` is gradient checkpointing. Per-bucket numbers are in `benchmarks/results.jsonl`; regenerate the table with `python benchmarks/make_table.py`.
131
+
132
+ What the measurements changed in the formulas: the inference peak is just `kv_cache + logits` (per-layer buffers are freed as the forward proceeds); the training logits peak is 14 bytes per logit, not 10, because `log_softmax` backward holds its saved output and both gradient tensors at once; and each layer saves about two more hidden-sized tensors than the textbook count.
133
+
134
+ ### Limits
135
+
136
+ - `bench` uses `torch.autocast` directly, not HF Trainer. Extra conversions done by Accelerate are not modelled.
137
+ - `fits()` assumes nothing else is using the GPU and takes 97% of the nominal capacity as usable.
138
+ - `attn_impl="sdpa"` assumes the memory-efficient or flash backend. If your environment falls back to the math backend, use `eager`.
139
+ - Overhead was measured on an A6000 with CUDA 13; other cards and driver versions may differ by a few hundred megabytes.
140
+
141
+ ### License
142
+
143
+ MIT
144
+
145
+ ---
146
+
147
+ ## 한국어
148
+
149
+ 모델 가중치를 받기 전, GPU를 빌리기 전에 LLM이 얼마만큼의 GPU 메모리를 필요로 하는지 예측하는 라이브러리입니다. Hugging Face `config.json`만 읽어 지정한 GPU에 들어가는지 판정하고, 들어가지 않으면 무엇을 하나 바꾸면 들어가는지 제안합니다. `bench` 명령은 실제 GPU에서 메모리를 측정해 예측과 항목별로 비교합니다.
150
+
151
+ ```python
152
+ from vramcalc import estimate
153
+
154
+ e = estimate("meta-llama/Llama-3.1-8B", mode="train", lora_rank=16,
155
+ precision="pure", seq_len=4096)
156
+ print(e.breakdown())
157
+ r = e.fits("RTX 3090")
158
+ print(r.message)
159
+ for s in r.suggestions:
160
+ print(s.change, s.margin_bytes)
161
+ ```
162
+
163
+ ```
164
+ $ vramcalc meta-llama/Llama-3.1-8B --train --lora 16 --precision pure --seq 4096 --gpu "RTX 3090"
165
+ weights 16.12 GB
166
+ gradients 0.05 GB
167
+ optimizer 0.11 GB
168
+ activations 36.92 GB
169
+ step_workspace 0.05 GB
170
+ total 53.19 GB
171
+ overhead 0.75 GB
172
+
173
+ short by 28.9 GB on RTX 3090; 1 single adjustment fits
174
+ -> seq_len=512: 21.6 GB (changes training objective)
175
+ ```
176
+
177
+ ### 설치
178
+
179
+ | 명령 | 포함 내용 |
180
+ |---|---|
181
+ | `pip install vramcalc` | 계산기와 CLI. 의존성 없음. 로컬 `config.json` 또는 모델 디렉터리 입력 |
182
+ | `pip install "vramcalc[hf]"` | Hugging Face Hub repo id 입력 (`huggingface_hub`) |
183
+ | `pip install "vramcalc[bench]"` | 실측 비교 `vramcalc bench` (`torch`, `transformers`, `peft`, `nvidia-ml-py`) |
184
+
185
+ 오프라인이나 온프레미스 환경에서는 `estimate("/path/to/model")`처럼 모델 디렉터리를 넘기면 됩니다. `HF_HUB_OFFLINE=1`이면 Hub id도 로컬 캐시만 조회합니다.
186
+
187
+ ### 만든 이유
188
+
189
+ "이 모델을 이 설정으로 이 GPU에서 돌릴 수 있는가"는 16GB 가중치를 받거나 H100을 한 시간 빌리기 전에 알아야 하는 질문입니다. 기존 계산기는 대부분 가중치 크기 위주이고 activation, optimizer 상태, autocast 복사본은 대략적으로만 다루기 때문에 학습 모드에서 수 GB씩 어긋납니다. vramcalc는 메모리 항목마다 식을 하나씩 두고, 같은 설정을 실제 GPU에서 측정하는 `bench` 명령을 함께 제공해 항목별로 allocator 값과 대조할 수 있게 했습니다. 식은 고정이며 보정 계수를 학습하지 않습니다. 실측은 식을 고치는 근거이지, 위에 덧씌우는 보정이 아닙니다.
190
+
191
+ ### 지원 범위
192
+
193
+ - 모델: `model_type`이 `llama`, `mistral`, `qwen2`, `gemma`, `gemma2`인 dense decoder-only 모델. GQA, QKV bias(Qwen2), 임베딩 공유, Gemma 2의 레이어당 norm 4개를 config에서 읽어 반영합니다.
194
+ - 모드: 추론(KV cache 포함), 전체 학습, LoRA(`q_proj`, `k_proj`, `v_proj`, `o_proj` 대상)
195
+ - 옵션: `dtype`(fp32/bf16/fp16), `batch`, `seq_len`, `grad_checkpoint`, `optimizer`(adamw/sgd), `precision`(amp/pure), `attn_impl`(eager/sdpa/flash)
196
+
197
+ `precision="amp"`는 HF Trainer `bf16=True` 방식(fp32 가중치, bf16 autocast)이고, `precision="pure"`는 `torch_dtype=bf16`으로 로드해 그대로 학습하는 방식입니다. 둘의 메모리는 2배 가까이 차이가 나므로 실제로 사용하는 방식을 지정해야 합니다.
198
+
199
+ 0.1에서 제외된 항목: MoE, MLA, 양자화(int8/int4/QLoRA), 분산 학습(ZeRO/FSDP), 멀티모달과 encoder 모델, sliding window KV cache(현재는 전부 global attention으로 보수적으로 계산).
200
+
201
+ ### 계산 항목
202
+
203
+ 예측 대상은 PyTorch caching allocator 기준 `allocated` 메모리입니다. CUDA context 등 allocator 밖 메모리는 `overhead`(기본 0.75GB)로 분리해 `fits()` 판정에서만 더합니다.
204
+
205
+ | 항목 | 내용 |
206
+ |---|---|
207
+ | `weights` | 파라미터 수 × 저장 dtype. LoRA는 frozen base + fp32 어댑터 |
208
+ | `kv_cache` | 추론 전용. `2 × L × b × s × kv × bytes` |
209
+ | `gradients`, `optimizer` | 학습 대상 파라미터 기준. AdamW는 상태 2개, SGD는 momentum 1개 |
210
+ | `activations` | 레이어당 역전파 저장분(SwiGLU, RMSNorm) + `log_softmax` 역전파 시점의 fp32 logits 체인 |
211
+ | `autocast_cache` | amp에서 forward 동안 살아 있는 bf16 가중치 복사본 |
212
+ | `step_workspace` | foreach optimizer의 `step()` 중 임시 버퍼 |
213
+
214
+ `total = weights + kv_cache + gradients + optimizer + max(activations + autocast_cache, step_workspace)`. 상세 식은 `docs/design.md` 5절에 있습니다.
215
+
216
+ ### 실측 검증
217
+
218
+ `vramcalc bench`는 같은 인자로 모델을 실제로 로드해 `torch.cuda.max_memory_allocated()`를 기록하고 weights, states, peak, total, overhead 다섯 묶음으로 예측과 비교합니다.
219
+
220
+ ```
221
+ vramcalc bench /path/to/Llama-3.2-1B --train --precision pure --record benchmarks/results.jsonl
222
+ ```
223
+
224
+ 측정 환경: NVIDIA RTX A6000 48GB, torch 2.14.1+cu130, transformers 4.57.6, peft 0.21.2. Llama 가중치는 gated라 동일 가중치의 `unsloth/` 미러를 사용했습니다. 19개 조합 모두 `total` 오차 ±4.3% 안이며 `weights`는 전부 0.0%입니다. 표는 위 영어 절과 같고, 항목별 수치는 `benchmarks/results.jsonl`에 있습니다. `python benchmarks/make_table.py`로 표를 다시 생성할 수 있습니다.
225
+
226
+ 실측으로 바뀐 식: 추론 peak는 `kv_cache + logits`만으로 설명되고(레이어 버퍼는 forward 진행 중 해제됨), 학습 logits peak는 logit당 10바이트가 아니라 14바이트이며(`log_softmax` 역전파가 저장 출력과 두 gradient를 동시에 보유), 레이어마다 교과서 계산보다 hidden 크기 텐서 약 2개가 더 저장됩니다.
227
+
228
+ ### 한계
229
+
230
+ - `bench`는 HF Trainer가 아니라 `torch.autocast`를 직접 사용합니다. Accelerate가 추가로 수행하는 변환은 반영하지 않습니다.
231
+ - `fits()`는 GPU에 다른 프로세스가 없다고 가정하고 표기 용량의 97%를 사용 가능 용량으로 봅니다.
232
+ - `attn_impl="sdpa"`는 memory-efficient 또는 flash backend를 전제로 합니다. math backend로 떨어지는 환경이면 `eager`로 지정해야 합니다.
233
+ - overhead는 A6000과 CUDA 13에서 측정한 값이라 다른 카드와 드라이버에서는 수백 MB 차이가 날 수 있습니다.
234
+
235
+ ### 라이선스
236
+
237
+ MIT
@@ -0,0 +1,215 @@
1
+ # vramcalc
2
+
3
+ [English](#english) | [한국어](#한국어)
4
+
5
+ ---
6
+
7
+ ## English
8
+
9
+ Predict how much GPU memory an LLM needs before you download the weights or rent the GPU. vramcalc reads only the Hugging Face `config.json`, tells you whether the model fits a given GPU, and when it does not, which single change would make it fit. A `bench` subcommand measures real memory on your GPU and compares it with the prediction, bucket by bucket.
10
+
11
+ ```python
12
+ from vramcalc import estimate
13
+
14
+ e = estimate("meta-llama/Llama-3.1-8B", mode="train", lora_rank=16,
15
+ precision="pure", seq_len=4096)
16
+ print(e.breakdown())
17
+ r = e.fits("RTX 3090")
18
+ print(r.message)
19
+ for s in r.suggestions:
20
+ print(s.change, s.margin_bytes)
21
+ ```
22
+
23
+ ```
24
+ $ vramcalc meta-llama/Llama-3.1-8B --train --lora 16 --precision pure --seq 4096 --gpu "RTX 3090"
25
+ weights 16.12 GB
26
+ gradients 0.05 GB
27
+ optimizer 0.11 GB
28
+ activations 36.92 GB
29
+ step_workspace 0.05 GB
30
+ total 53.19 GB
31
+ overhead 0.75 GB
32
+
33
+ short by 28.9 GB on RTX 3090; 1 single adjustment fits
34
+ -> seq_len=512: 21.6 GB (changes training objective)
35
+ ```
36
+
37
+ ### Install
38
+
39
+ | Command | What you get |
40
+ |---|---|
41
+ | `pip install vramcalc` | Estimator and CLI. No dependencies. Takes a local `config.json` or a model directory |
42
+ | `pip install "vramcalc[hf]"` | Hugging Face Hub repo ids as input (`huggingface_hub`) |
43
+ | `pip install "vramcalc[bench]"` | `vramcalc bench` for real measurements (`torch`, `transformers`, `peft`, `nvidia-ml-py`) |
44
+
45
+ Offline or on-premises: pass a model directory, `estimate("/path/to/model")`. With `HF_HUB_OFFLINE=1`, Hub ids resolve from the local cache only.
46
+
47
+ ### Why
48
+
49
+ "Will this model run on this GPU with these settings?" is a question you want answered before downloading 16 GB of weights or paying for an H100 hour. Existing calculators mostly size the weights and treat activations, optimizer state and autocast copies loosely, which is where training estimates drift by several gigabytes. vramcalc keeps one formula per memory term and ships a `bench` command that measures the same configuration on a real GPU, so every term can be checked against the allocator. The formulas are fixed; there is no learned correction factor. Measurements are evidence for fixing a formula, not a fudge applied on top.
50
+
51
+ ### Scope
52
+
53
+ - Models: dense decoder-only models with `model_type` of `llama`, `mistral`, `qwen2`, `gemma` or `gemma2`. GQA, QKV bias (Qwen2), tied embeddings and Gemma 2's four norms per layer are read from the config.
54
+ - Modes: inference (with KV cache), full fine-tuning, LoRA on `q_proj`, `k_proj`, `v_proj`, `o_proj`.
55
+ - Knobs: `dtype` (fp32/bf16/fp16), `batch`, `seq_len`, `grad_checkpoint`, `optimizer` (adamw/sgd), `precision` (amp/pure), `attn_impl` (eager/sdpa/flash).
56
+
57
+ `precision="amp"` models HF Trainer with `bf16=True` (fp32 weights, bf16 autocast). `precision="pure"` models loading with `torch_dtype=bf16` and training as is. The two differ by almost 2x in memory, so pick the one you actually run.
58
+
59
+ Out of scope for 0.1: MoE, MLA, quantization (int8/int4/QLoRA), distributed training (ZeRO/FSDP), multimodal and encoder models, and sliding-window KV caches (currently counted as global attention, which is conservative).
60
+
61
+ ### What is counted
62
+
63
+ The prediction targets PyTorch's caching allocator (`allocated` memory). CUDA context and other memory outside the allocator is kept separate as `overhead` (default 0.75 GB) and only added in `fits()`.
64
+
65
+ | Term | Meaning |
66
+ |---|---|
67
+ | `weights` | parameter count x storage dtype; for LoRA, frozen base plus fp32 adapters |
68
+ | `kv_cache` | inference only, `2 x L x b x s x kv x bytes` |
69
+ | `gradients`, `optimizer` | over trainable parameters; AdamW keeps two moments, SGD one |
70
+ | `activations` | per-layer tensors saved for backward (SwiGLU, RMSNorm) plus the fp32 logits chain at the `log_softmax` backward peak |
71
+ | `autocast_cache` | bf16 weight copies alive during an amp forward pass |
72
+ | `step_workspace` | temporary buffer of the foreach optimizer during `step()` |
73
+
74
+ `total = weights + kv_cache + gradients + optimizer + max(activations + autocast_cache, step_workspace)`. The full formulas are in `docs/design.md`, section 5.
75
+
76
+ ### Validation
77
+
78
+ `vramcalc bench` loads the model with the same arguments, records `torch.cuda.max_memory_allocated()`, and compares five buckets: weights, states, peak, total, overhead.
79
+
80
+ ```
81
+ vramcalc bench /path/to/Llama-3.2-1B --train --precision pure --record benchmarks/results.jsonl
82
+ ```
83
+
84
+ Environment: NVIDIA RTX A6000 48 GB, torch 2.14.1+cu130, transformers 4.57.6, peft 0.21.2. Llama weights are gated, so the `unsloth/` mirrors of the same checkpoints were used. All 19 runs land within +/-4.3% on `total`; `weights` is exact in every run.
85
+
86
+ | model | mode | setting | GPU | predicted | measured | error |
87
+ |---|---|---|---|---|---|---|
88
+ | Llama-3.2-1B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 3.08 GB | 3.08 GB | -0.1% |
89
+ | Llama-3.2-1B | inference | bf16 b=1 s=4096 sdpa | RTX A6000 | 3.69 GB | 3.68 GB | +0.2% |
90
+ | Llama-3.2-1B | inference | bf16 b=1 s=2048 eager | RTX A6000 | 3.90 GB | 3.95 GB | -1.2% |
91
+ | Llama-3.2-3B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 7.21 GB | 7.21 GB | +0.0% |
92
+ | Llama-3.1-8B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 16.89 GB | 16.88 GB | +0.0% |
93
+ | Qwen2.5-1.5B | inference | bf16 b=1 s=2048 sdpa | RTX A6000 | 3.78 GB | 3.78 GB | -0.1% |
94
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full pure | RTX A6000 | 17.52 GB | 17.51 GB | +0.0% |
95
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full pure ckpt | RTX A6000 | 13.95 GB | 13.76 GB | +1.4% |
96
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full amp | RTX A6000 | 29.88 GB | 30.55 GB | -2.2% |
97
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa full amp ckpt | RTX A6000 | 26.30 GB | 25.22 GB | +4.3% |
98
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 eager full pure | RTX A6000 | 30.41 GB | 30.26 GB | +0.5% |
99
+ | Llama-3.2-3B | train | bf16 b=1 s=2048 sdpa full pure ckpt | RTX A6000 | 32.13 GB | 32.15 GB | -0.1% |
100
+ | Qwen2.5-1.5B | train | bf16 b=1 s=2048 sdpa full pure | RTX A6000 | 23.17 GB | 23.40 GB | -1.0% |
101
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 10.17 GB | 10.13 GB | +0.4% |
102
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 amp | RTX A6000 | 15.11 GB | 14.52 GB | +4.1% |
103
+ | Llama-3.2-1B | train | bf16 b=1 s=2048 sdpa lora=16 amp ckpt | RTX A6000 | 9.22 GB | 9.51 GB | -3.1% |
104
+ | Llama-3.2-3B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 18.84 GB | 18.98 GB | -0.8% |
105
+ | Llama-3.1-8B | train | bf16 b=1 s=2048 sdpa lora=16 pure ckpt | RTX A6000 | 20.97 GB | 20.55 GB | +2.0% |
106
+ | Qwen2.5-1.5B | train | bf16 b=1 s=2048 sdpa lora=16 pure | RTX A6000 | 13.99 GB | 13.78 GB | +1.5% |
107
+
108
+ `pure` is `torch_dtype=bf16` training, `amp` is fp32 weights with bf16 autocast, `ckpt` is gradient checkpointing. Per-bucket numbers are in `benchmarks/results.jsonl`; regenerate the table with `python benchmarks/make_table.py`.
109
+
110
+ What the measurements changed in the formulas: the inference peak is just `kv_cache + logits` (per-layer buffers are freed as the forward proceeds); the training logits peak is 14 bytes per logit, not 10, because `log_softmax` backward holds its saved output and both gradient tensors at once; and each layer saves about two more hidden-sized tensors than the textbook count.
111
+
112
+ ### Limits
113
+
114
+ - `bench` uses `torch.autocast` directly, not HF Trainer. Extra conversions done by Accelerate are not modelled.
115
+ - `fits()` assumes nothing else is using the GPU and takes 97% of the nominal capacity as usable.
116
+ - `attn_impl="sdpa"` assumes the memory-efficient or flash backend. If your environment falls back to the math backend, use `eager`.
117
+ - Overhead was measured on an A6000 with CUDA 13; other cards and driver versions may differ by a few hundred megabytes.
118
+
119
+ ### License
120
+
121
+ MIT
122
+
123
+ ---
124
+
125
+ ## 한국어
126
+
127
+ 모델 가중치를 받기 전, GPU를 빌리기 전에 LLM이 얼마만큼의 GPU 메모리를 필요로 하는지 예측하는 라이브러리입니다. Hugging Face `config.json`만 읽어 지정한 GPU에 들어가는지 판정하고, 들어가지 않으면 무엇을 하나 바꾸면 들어가는지 제안합니다. `bench` 명령은 실제 GPU에서 메모리를 측정해 예측과 항목별로 비교합니다.
128
+
129
+ ```python
130
+ from vramcalc import estimate
131
+
132
+ e = estimate("meta-llama/Llama-3.1-8B", mode="train", lora_rank=16,
133
+ precision="pure", seq_len=4096)
134
+ print(e.breakdown())
135
+ r = e.fits("RTX 3090")
136
+ print(r.message)
137
+ for s in r.suggestions:
138
+ print(s.change, s.margin_bytes)
139
+ ```
140
+
141
+ ```
142
+ $ vramcalc meta-llama/Llama-3.1-8B --train --lora 16 --precision pure --seq 4096 --gpu "RTX 3090"
143
+ weights 16.12 GB
144
+ gradients 0.05 GB
145
+ optimizer 0.11 GB
146
+ activations 36.92 GB
147
+ step_workspace 0.05 GB
148
+ total 53.19 GB
149
+ overhead 0.75 GB
150
+
151
+ short by 28.9 GB on RTX 3090; 1 single adjustment fits
152
+ -> seq_len=512: 21.6 GB (changes training objective)
153
+ ```
154
+
155
+ ### 설치
156
+
157
+ | 명령 | 포함 내용 |
158
+ |---|---|
159
+ | `pip install vramcalc` | 계산기와 CLI. 의존성 없음. 로컬 `config.json` 또는 모델 디렉터리 입력 |
160
+ | `pip install "vramcalc[hf]"` | Hugging Face Hub repo id 입력 (`huggingface_hub`) |
161
+ | `pip install "vramcalc[bench]"` | 실측 비교 `vramcalc bench` (`torch`, `transformers`, `peft`, `nvidia-ml-py`) |
162
+
163
+ 오프라인이나 온프레미스 환경에서는 `estimate("/path/to/model")`처럼 모델 디렉터리를 넘기면 됩니다. `HF_HUB_OFFLINE=1`이면 Hub id도 로컬 캐시만 조회합니다.
164
+
165
+ ### 만든 이유
166
+
167
+ "이 모델을 이 설정으로 이 GPU에서 돌릴 수 있는가"는 16GB 가중치를 받거나 H100을 한 시간 빌리기 전에 알아야 하는 질문입니다. 기존 계산기는 대부분 가중치 크기 위주이고 activation, optimizer 상태, autocast 복사본은 대략적으로만 다루기 때문에 학습 모드에서 수 GB씩 어긋납니다. vramcalc는 메모리 항목마다 식을 하나씩 두고, 같은 설정을 실제 GPU에서 측정하는 `bench` 명령을 함께 제공해 항목별로 allocator 값과 대조할 수 있게 했습니다. 식은 고정이며 보정 계수를 학습하지 않습니다. 실측은 식을 고치는 근거이지, 위에 덧씌우는 보정이 아닙니다.
168
+
169
+ ### 지원 범위
170
+
171
+ - 모델: `model_type`이 `llama`, `mistral`, `qwen2`, `gemma`, `gemma2`인 dense decoder-only 모델. GQA, QKV bias(Qwen2), 임베딩 공유, Gemma 2의 레이어당 norm 4개를 config에서 읽어 반영합니다.
172
+ - 모드: 추론(KV cache 포함), 전체 학습, LoRA(`q_proj`, `k_proj`, `v_proj`, `o_proj` 대상)
173
+ - 옵션: `dtype`(fp32/bf16/fp16), `batch`, `seq_len`, `grad_checkpoint`, `optimizer`(adamw/sgd), `precision`(amp/pure), `attn_impl`(eager/sdpa/flash)
174
+
175
+ `precision="amp"`는 HF Trainer `bf16=True` 방식(fp32 가중치, bf16 autocast)이고, `precision="pure"`는 `torch_dtype=bf16`으로 로드해 그대로 학습하는 방식입니다. 둘의 메모리는 2배 가까이 차이가 나므로 실제로 사용하는 방식을 지정해야 합니다.
176
+
177
+ 0.1에서 제외된 항목: MoE, MLA, 양자화(int8/int4/QLoRA), 분산 학습(ZeRO/FSDP), 멀티모달과 encoder 모델, sliding window KV cache(현재는 전부 global attention으로 보수적으로 계산).
178
+
179
+ ### 계산 항목
180
+
181
+ 예측 대상은 PyTorch caching allocator 기준 `allocated` 메모리입니다. CUDA context 등 allocator 밖 메모리는 `overhead`(기본 0.75GB)로 분리해 `fits()` 판정에서만 더합니다.
182
+
183
+ | 항목 | 내용 |
184
+ |---|---|
185
+ | `weights` | 파라미터 수 × 저장 dtype. LoRA는 frozen base + fp32 어댑터 |
186
+ | `kv_cache` | 추론 전용. `2 × L × b × s × kv × bytes` |
187
+ | `gradients`, `optimizer` | 학습 대상 파라미터 기준. AdamW는 상태 2개, SGD는 momentum 1개 |
188
+ | `activations` | 레이어당 역전파 저장분(SwiGLU, RMSNorm) + `log_softmax` 역전파 시점의 fp32 logits 체인 |
189
+ | `autocast_cache` | amp에서 forward 동안 살아 있는 bf16 가중치 복사본 |
190
+ | `step_workspace` | foreach optimizer의 `step()` 중 임시 버퍼 |
191
+
192
+ `total = weights + kv_cache + gradients + optimizer + max(activations + autocast_cache, step_workspace)`. 상세 식은 `docs/design.md` 5절에 있습니다.
193
+
194
+ ### 실측 검증
195
+
196
+ `vramcalc bench`는 같은 인자로 모델을 실제로 로드해 `torch.cuda.max_memory_allocated()`를 기록하고 weights, states, peak, total, overhead 다섯 묶음으로 예측과 비교합니다.
197
+
198
+ ```
199
+ vramcalc bench /path/to/Llama-3.2-1B --train --precision pure --record benchmarks/results.jsonl
200
+ ```
201
+
202
+ 측정 환경: NVIDIA RTX A6000 48GB, torch 2.14.1+cu130, transformers 4.57.6, peft 0.21.2. Llama 가중치는 gated라 동일 가중치의 `unsloth/` 미러를 사용했습니다. 19개 조합 모두 `total` 오차 ±4.3% 안이며 `weights`는 전부 0.0%입니다. 표는 위 영어 절과 같고, 항목별 수치는 `benchmarks/results.jsonl`에 있습니다. `python benchmarks/make_table.py`로 표를 다시 생성할 수 있습니다.
203
+
204
+ 실측으로 바뀐 식: 추론 peak는 `kv_cache + logits`만으로 설명되고(레이어 버퍼는 forward 진행 중 해제됨), 학습 logits peak는 logit당 10바이트가 아니라 14바이트이며(`log_softmax` 역전파가 저장 출력과 두 gradient를 동시에 보유), 레이어마다 교과서 계산보다 hidden 크기 텐서 약 2개가 더 저장됩니다.
205
+
206
+ ### 한계
207
+
208
+ - `bench`는 HF Trainer가 아니라 `torch.autocast`를 직접 사용합니다. Accelerate가 추가로 수행하는 변환은 반영하지 않습니다.
209
+ - `fits()`는 GPU에 다른 프로세스가 없다고 가정하고 표기 용량의 97%를 사용 가능 용량으로 봅니다.
210
+ - `attn_impl="sdpa"`는 memory-efficient 또는 flash backend를 전제로 합니다. math backend로 떨어지는 환경이면 `eager`로 지정해야 합니다.
211
+ - overhead는 A6000과 CUDA 13에서 측정한 값이라 다른 카드와 드라이버에서는 수백 MB 차이가 날 수 있습니다.
212
+
213
+ ### 라이선스
214
+
215
+ MIT
@@ -0,0 +1,42 @@
1
+ """Render benchmarks/results.jsonl as a Markdown table for the README.
2
+
3
+ Usage: python benchmarks/make_table.py [results.jsonl]
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ GB = 10**9
13
+
14
+
15
+ def main(path: str) -> int:
16
+ rows = [json.loads(line) for line in Path(path).read_text().splitlines() if line.strip()]
17
+ print("| model | mode | setting | GPU | predicted | measured | error |")
18
+ print("|---|---|---|---|---|---|---|")
19
+ for r in rows:
20
+ p = r["params"]
21
+ parts = [p["dtype"], f"b={p['batch']}", f"s={p['seq_len']}", p["attn_impl"]]
22
+ if p["mode"] == "train":
23
+ parts.append(f"lora={p['lora_rank']}" if p["lora_rank"] else "full")
24
+ parts.append(p["precision"])
25
+ if p["grad_checkpoint"]:
26
+ parts.append("ckpt")
27
+ pred = r["predicted"]["total"] / GB
28
+ if r["status"] == "ok":
29
+ meas = r["measured"]["total"] / GB
30
+ err = f"{(pred - meas) / meas * 100:+.1f}%"
31
+ meas_txt = f"{meas:.2f} GB"
32
+ else:
33
+ meas_txt, err = r["status"].upper(), "-"
34
+ gpu = r["gpu_name"].replace("NVIDIA GeForce ", "").replace("NVIDIA ", "")
35
+ name = p["arch"]["name"].removeprefix("Meta-").removeprefix("meta-")
36
+ print(f"| {name} | {p['mode']} | {' '.join(parts)} | {gpu} | "
37
+ f"{pred:.2f} GB | {meas_txt} | {err} |")
38
+ return 0
39
+
40
+
41
+ if __name__ == "__main__":
42
+ sys.exit(main(sys.argv[1] if len(sys.argv) > 1 else "benchmarks/results.jsonl"))