hfit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hfit-0.1.0/LICENSE +21 -0
- hfit-0.1.0/PKG-INFO +113 -0
- hfit-0.1.0/README.md +86 -0
- hfit-0.1.0/hfit/__init__.py +3 -0
- hfit-0.1.0/hfit/__main__.py +5 -0
- hfit-0.1.0/hfit/cli.py +65 -0
- hfit-0.1.0/hfit/estimate.py +132 -0
- hfit-0.1.0/hfit/hw.py +231 -0
- hfit-0.1.0/hfit/model.py +200 -0
- hfit-0.1.0/hfit/report.py +143 -0
- hfit-0.1.0/hfit/tui.py +111 -0
- hfit-0.1.0/hfit.egg-info/PKG-INFO +113 -0
- hfit-0.1.0/hfit.egg-info/SOURCES.txt +17 -0
- hfit-0.1.0/hfit.egg-info/dependency_links.txt +1 -0
- hfit-0.1.0/hfit.egg-info/entry_points.txt +2 -0
- hfit-0.1.0/hfit.egg-info/requires.txt +6 -0
- hfit-0.1.0/hfit.egg-info/top_level.txt +1 -0
- hfit-0.1.0/pyproject.toml +41 -0
- hfit-0.1.0/setup.cfg +4 -0
hfit-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Samir Nuri
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
hfit-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hfit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Check whether a Hugging Face model fits your local hardware — without downloading the weights.
|
|
5
|
+
Author-email: Samir Nuri <samirnuri714@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/snuri00/hfit
|
|
8
|
+
Project-URL: Issues, https://github.com/snuri00/hfit/issues
|
|
9
|
+
Keywords: huggingface,llm,vram,gpu,memory,tui,hardware,inference
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: System :: Hardware
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: huggingface_hub>=0.23
|
|
22
|
+
Requires-Dist: rich>=13.0
|
|
23
|
+
Requires-Dist: textual>=0.60
|
|
24
|
+
Provides-Extra: full
|
|
25
|
+
Requires-Dist: psutil>=5.9; extra == "full"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# hfit
|
|
29
|
+
|
|
30
|
+
*Hugging Face + fit.*
|
|
31
|
+
|
|
32
|
+
**Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
|
|
33
|
+
|
|
34
|
+
Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
|
|
35
|
+
parameter count and architecture from the Hub's metadata (a few KB of API
|
|
36
|
+
calls), detects your local hardware (GPU VRAM, RAM), and tells you which
|
|
37
|
+
precisions fit:
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
|
|
41
|
+
┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
|
|
42
|
+
┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
|
|
43
|
+
│ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
|
|
44
|
+
│ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
|
|
45
|
+
│ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
|
|
46
|
+
└─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Unlike `accelerate estimate-memory`, this compares the model against **your
|
|
50
|
+
actual hardware** and includes a **KV-cache estimate** at your chosen context
|
|
51
|
+
length computed exactly from the model's architecture (layers × KV heads ×
|
|
52
|
+
head dim), not a flat percentage.
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install hfit
|
|
58
|
+
# or from a clone of this repo:
|
|
59
|
+
pip install .
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
|
|
64
|
+
|
|
65
|
+
## Usage
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
hfit # interactive TUI
|
|
69
|
+
hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
|
|
70
|
+
hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
|
|
71
|
+
hfit some/gated-model --token hf_xxx
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
In the TUI: type a model id, press Enter; switch the context length from the
|
|
75
|
+
dropdown to see the KV cache impact instantly (no refetch). `q` quits.
|
|
76
|
+
|
|
77
|
+
Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
|
|
78
|
+
error, `3` = does not fit at all.
|
|
79
|
+
|
|
80
|
+
## What it understands
|
|
81
|
+
|
|
82
|
+
- **Safetensors repos** — exact parameter count from Hub metadata.
|
|
83
|
+
- **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
|
|
84
|
+
individually, for llama.cpp / Ollama / LM Studio users.
|
|
85
|
+
- **Multi-component repos** (speech/vision pipelines with several weight
|
|
86
|
+
folders) reported as a stored total instead of a fake parameter count.
|
|
87
|
+
- Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
|
|
88
|
+
are deduplicated, not double-counted.
|
|
89
|
+
|
|
90
|
+
## Platform support
|
|
91
|
+
|
|
92
|
+
| Platform | GPU detection | RAM detection |
|
|
93
|
+
|---|---|---|
|
|
94
|
+
| Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
|
|
95
|
+
| Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
|
|
96
|
+
| macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
|
|
97
|
+
|
|
98
|
+
`psutil` is used when available (`pip install .[full]`) but is not required.
|
|
99
|
+
|
|
100
|
+
## How the estimate works
|
|
101
|
+
|
|
102
|
+
```
|
|
103
|
+
required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
- Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
|
|
107
|
+
FP32 / FP16 / INT8 / INT4).
|
|
108
|
+
- KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
|
|
109
|
+
from `config.json`; a 20% margin is used when the architecture is unknown.
|
|
110
|
+
- Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
|
|
111
|
+
|
|
112
|
+
These are pre-flight estimates. For a definitive runtime answer, vLLM's
|
|
113
|
+
`--dry-run` on the actual machine remains the ground truth.
|
hfit-0.1.0/README.md
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# hfit
|
|
2
|
+
|
|
3
|
+
*Hugging Face + fit.*
|
|
4
|
+
|
|
5
|
+
**Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
|
|
6
|
+
|
|
7
|
+
Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
|
|
8
|
+
parameter count and architecture from the Hub's metadata (a few KB of API
|
|
9
|
+
calls), detects your local hardware (GPU VRAM, RAM), and tells you which
|
|
10
|
+
precisions fit:
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
|
|
14
|
+
┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
|
|
15
|
+
┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
|
|
16
|
+
│ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
|
|
17
|
+
│ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
|
|
18
|
+
│ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
|
|
19
|
+
└─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Unlike `accelerate estimate-memory`, this compares the model against **your
|
|
23
|
+
actual hardware** and includes a **KV-cache estimate** at your chosen context
|
|
24
|
+
length computed exactly from the model's architecture (layers × KV heads ×
|
|
25
|
+
head dim), not a flat percentage.
|
|
26
|
+
|
|
27
|
+
## Install
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install hfit
|
|
31
|
+
# or from a clone of this repo:
|
|
32
|
+
pip install .
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
|
|
37
|
+
|
|
38
|
+
## Usage
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
hfit # interactive TUI
|
|
42
|
+
hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
|
|
43
|
+
hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
|
|
44
|
+
hfit some/gated-model --token hf_xxx
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
In the TUI: type a model id, press Enter; switch the context length from the
|
|
48
|
+
dropdown to see the KV cache impact instantly (no refetch). `q` quits.
|
|
49
|
+
|
|
50
|
+
Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
|
|
51
|
+
error, `3` = does not fit at all.
|
|
52
|
+
|
|
53
|
+
## What it understands
|
|
54
|
+
|
|
55
|
+
- **Safetensors repos** — exact parameter count from Hub metadata.
|
|
56
|
+
- **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
|
|
57
|
+
individually, for llama.cpp / Ollama / LM Studio users.
|
|
58
|
+
- **Multi-component repos** (speech/vision pipelines with several weight
|
|
59
|
+
folders) reported as a stored total instead of a fake parameter count.
|
|
60
|
+
- Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
|
|
61
|
+
are deduplicated, not double-counted.
|
|
62
|
+
|
|
63
|
+
## Platform support
|
|
64
|
+
|
|
65
|
+
| Platform | GPU detection | RAM detection |
|
|
66
|
+
|---|---|---|
|
|
67
|
+
| Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
|
|
68
|
+
| Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
|
|
69
|
+
| macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
|
|
70
|
+
|
|
71
|
+
`psutil` is used when available (`pip install .[full]`) but is not required.
|
|
72
|
+
|
|
73
|
+
## How the estimate works
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
- Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
|
|
80
|
+
FP32 / FP16 / INT8 / INT4).
|
|
81
|
+
- KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
|
|
82
|
+
from `config.json`; a 20% margin is used when the architecture is unknown.
|
|
83
|
+
- Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
|
|
84
|
+
|
|
85
|
+
These are pre-flight estimates. For a definitive runtime answer, vLLM's
|
|
86
|
+
`--dry-run` on the actual machine remains the ground truth.
|
hfit-0.1.0/hfit/cli.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Entry point: `hfit MODEL_ID` prints a report, no args opens the TUI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from typing import List, Optional
|
|
8
|
+
|
|
9
|
+
from rich.console import Console
|
|
10
|
+
|
|
11
|
+
from . import __version__
|
|
12
|
+
from .estimate import Verdict, assess, best_verdict
|
|
13
|
+
from .hw import detect
|
|
14
|
+
from .model import ModelLookupError, fetch_model
|
|
15
|
+
from .report import full_report, system_panel
|
|
16
|
+
|
|
17
|
+
EXIT_OK, EXIT_ERROR, EXIT_NO_FIT = 0, 1, 3
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def run_check(model_id: str, ctx_tokens: int, token: Optional[str]) -> int:
|
|
21
|
+
console = Console()
|
|
22
|
+
specs = detect()
|
|
23
|
+
console.print(system_panel(specs))
|
|
24
|
+
|
|
25
|
+
with console.status(f"Fetching metadata for [bold]{model_id}[/] …"):
|
|
26
|
+
try:
|
|
27
|
+
model = fetch_model(model_id, token=token)
|
|
28
|
+
except ModelLookupError as e:
|
|
29
|
+
console.print(f"[bold red]{e}[/]")
|
|
30
|
+
return EXIT_ERROR
|
|
31
|
+
|
|
32
|
+
results = assess(model, specs, ctx_tokens)
|
|
33
|
+
console.print(full_report(model, specs, results, ctx_tokens))
|
|
34
|
+
|
|
35
|
+
verdict = best_verdict(results)
|
|
36
|
+
return EXIT_NO_FIT if verdict in (Verdict.NO_FIT, None) else EXIT_OK
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
40
|
+
parser = argparse.ArgumentParser(
|
|
41
|
+
prog="hfit",
|
|
42
|
+
description="Check whether a Hugging Face model fits your local hardware, "
|
|
43
|
+
"without downloading the weights.",
|
|
44
|
+
)
|
|
45
|
+
parser.add_argument("model", nargs="?",
|
|
46
|
+
help="Hugging Face model id (org/name). Omit to open the TUI.")
|
|
47
|
+
parser.add_argument("--ctx", type=int, default=4096, metavar="TOKENS",
|
|
48
|
+
help="context length for the KV-cache estimate (default: 4096)")
|
|
49
|
+
parser.add_argument("--token", default=None,
|
|
50
|
+
help="HF access token for gated/private models "
|
|
51
|
+
"(defaults to your cached huggingface-cli login)")
|
|
52
|
+
parser.add_argument("--version", action="version",
|
|
53
|
+
version=f"%(prog)s {__version__}")
|
|
54
|
+
args = parser.parse_args(argv)
|
|
55
|
+
|
|
56
|
+
if args.model:
|
|
57
|
+
return run_check(args.model.strip().strip("/"), args.ctx, args.token)
|
|
58
|
+
|
|
59
|
+
from .tui import run_tui # imported lazily so plain CLI use stays snappy
|
|
60
|
+
run_tui(ctx_tokens=args.ctx, token=args.token)
|
|
61
|
+
return EXIT_OK
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
sys.exit(main())
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Memory math and fit verdicts.
|
|
2
|
+
|
|
3
|
+
Required memory = weights + KV cache at the chosen context + runtime overhead.
|
|
4
|
+
The KV cache is computed exactly from the architecture when config.json is
|
|
5
|
+
available, otherwise approximated as 20% of the weights.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from typing import List, Optional
|
|
13
|
+
|
|
14
|
+
from .hw import SystemSpecs
|
|
15
|
+
from .model import HFModel
|
|
16
|
+
|
|
17
|
+
GiB = 1024**3
|
|
18
|
+
|
|
19
|
+
# label, bytes per parameter, needs a quantized checkpoint?
|
|
20
|
+
PRECISIONS = [
|
|
21
|
+
("FP32", 4.0, False),
|
|
22
|
+
("FP16 / BF16", 2.0, False),
|
|
23
|
+
("INT8 (quantized)", 1.0, True),
|
|
24
|
+
("INT4 (quantized)", 0.5, True),
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
# Fractions of reported memory treated as actually usable: leave headroom for
|
|
28
|
+
# the CUDA context and fragmentation on GPUs, and for the OS on system RAM.
|
|
29
|
+
GPU_HEADROOM = 0.93
|
|
30
|
+
RAM_HEADROOM = 0.85
|
|
31
|
+
FRAMEWORK_OVERHEAD_GB = 0.8 # CUDA/Metal context, runtime buffers
|
|
32
|
+
ACTIVATION_FRACTION = 0.05 # inference activations, scales with model size
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Verdict(Enum):
|
|
36
|
+
GPU_OK = "gpu_ok"
|
|
37
|
+
GPU_TIGHT = "gpu_tight"
|
|
38
|
+
CPU_OK = "cpu_ok"
|
|
39
|
+
NO_FIT = "no_fit"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Assessment:
|
|
44
|
+
label: str
|
|
45
|
+
weights_gb: float
|
|
46
|
+
kv_gb: Optional[float] # None -> folded into overhead as a rough margin
|
|
47
|
+
required_gb: float
|
|
48
|
+
verdict: Verdict
|
|
49
|
+
note: str = ""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _kv_gb(model: HFModel, ctx_tokens: int) -> Optional[float]:
|
|
53
|
+
if model.kv is None:
|
|
54
|
+
return None
|
|
55
|
+
return model.kv.bytes_per_token * ctx_tokens / GiB
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _required(weights_gb: float, kv_gb: Optional[float]) -> float:
|
|
59
|
+
overhead = FRAMEWORK_OVERHEAD_GB + weights_gb * ACTIVATION_FRACTION
|
|
60
|
+
if kv_gb is None:
|
|
61
|
+
overhead += weights_gb * 0.20 # no architecture info: rough KV margin
|
|
62
|
+
kv_gb = 0.0
|
|
63
|
+
return weights_gb + kv_gb + overhead
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _judge(required_gb: float, specs: SystemSpecs) -> Verdict:
|
|
67
|
+
ram_budget = (specs.ram_available_gb or specs.ram_total_gb * 0.8) * RAM_HEADROOM
|
|
68
|
+
|
|
69
|
+
if specs.unified_memory:
|
|
70
|
+
# One memory pool: the GPU can address (most of) system RAM.
|
|
71
|
+
if required_gb <= ram_budget:
|
|
72
|
+
return Verdict.GPU_OK
|
|
73
|
+
if required_gb <= specs.ram_total_gb * RAM_HEADROOM:
|
|
74
|
+
return Verdict.GPU_TIGHT
|
|
75
|
+
return Verdict.NO_FIT
|
|
76
|
+
|
|
77
|
+
if specs.gpus:
|
|
78
|
+
free = specs.vram_free_gb
|
|
79
|
+
vram_budget = (free if free is not None else specs.vram_total_gb) * GPU_HEADROOM
|
|
80
|
+
if required_gb <= vram_budget:
|
|
81
|
+
return Verdict.GPU_OK
|
|
82
|
+
if required_gb <= specs.vram_total_gb:
|
|
83
|
+
return Verdict.GPU_TIGHT
|
|
84
|
+
if required_gb <= ram_budget:
|
|
85
|
+
return Verdict.CPU_OK
|
|
86
|
+
return Verdict.NO_FIT
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def assess(model: HFModel, specs: SystemSpecs, ctx_tokens: int = 4096) -> List[Assessment]:
|
|
90
|
+
results: List[Assessment] = []
|
|
91
|
+
kv = _kv_gb(model, ctx_tokens)
|
|
92
|
+
|
|
93
|
+
if model.params:
|
|
94
|
+
native = (model.native_dtype or "").upper()
|
|
95
|
+
for label, bytes_pp, needs_quant in PRECISIONS:
|
|
96
|
+
weights = model.params * bytes_pp / GiB
|
|
97
|
+
required = _required(weights, kv)
|
|
98
|
+
note = ""
|
|
99
|
+
if needs_quant:
|
|
100
|
+
note = "needs a quantized build (AWQ/GPTQ/bitsandbytes)"
|
|
101
|
+
elif "16" in label and native in ("F16", "BF16", "FP16"):
|
|
102
|
+
label += " (native)"
|
|
103
|
+
elif label == "FP32" and native in ("F16", "BF16", "FP16"):
|
|
104
|
+
note = "upcast — rarely useful for inference"
|
|
105
|
+
results.append(Assessment(label, weights, kv, required,
|
|
106
|
+
_judge(required, specs), note))
|
|
107
|
+
|
|
108
|
+
if model.params is None and model.weight_bytes:
|
|
109
|
+
# Multi-component pipeline repo: report the stored total, no dtype rows.
|
|
110
|
+
weights = model.weight_bytes / GiB
|
|
111
|
+
required = _required(weights, kv)
|
|
112
|
+
results.append(Assessment(
|
|
113
|
+
f"Full pipeline as stored ({model.weight_components} components)",
|
|
114
|
+
weights, kv, required, _judge(required, specs),
|
|
115
|
+
"sum of all weight folders in the repo",
|
|
116
|
+
))
|
|
117
|
+
|
|
118
|
+
for name, size in model.gguf_files:
|
|
119
|
+
weights = size / GiB
|
|
120
|
+
required = _required(weights, kv)
|
|
121
|
+
results.append(Assessment(f"GGUF: {name}", weights, kv, required,
|
|
122
|
+
_judge(required, specs),
|
|
123
|
+
"runs with llama.cpp / Ollama / LM Studio"))
|
|
124
|
+
return results
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def best_verdict(results: List[Assessment]) -> Optional[Verdict]:
|
|
128
|
+
order = [Verdict.GPU_OK, Verdict.GPU_TIGHT, Verdict.CPU_OK, Verdict.NO_FIT]
|
|
129
|
+
for v in order:
|
|
130
|
+
if any(r.verdict == v for r in results):
|
|
131
|
+
return v
|
|
132
|
+
return None
|
hfit-0.1.0/hfit/hw.py
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Cross-platform hardware detection with no hard dependencies.
|
|
2
|
+
|
|
3
|
+
Detection order per vendor, cheapest reliable source first:
|
|
4
|
+
NVIDIA -> nvidia-smi (Linux/Windows/WSL)
|
|
5
|
+
AMD -> /sys/class/drm sysfs (Linux, no ROCm needed), then rocm-smi
|
|
6
|
+
Apple -> unified memory on arm64 macOS (VRAM budget == system RAM)
|
|
7
|
+
RAM -> psutil if installed, else /proc/meminfo, sysctl, or WinAPI
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import ctypes
|
|
13
|
+
import glob
|
|
14
|
+
import os
|
|
15
|
+
import platform
|
|
16
|
+
import re
|
|
17
|
+
import subprocess
|
|
18
|
+
import sys
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from typing import List, Optional, Tuple
|
|
21
|
+
|
|
22
|
+
GiB = 1024**3
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _run(cmd: List[str], timeout: float = 10.0) -> Optional[str]:
|
|
26
|
+
try:
|
|
27
|
+
out = subprocess.run(
|
|
28
|
+
cmd, capture_output=True, text=True, timeout=timeout,
|
|
29
|
+
creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0),
|
|
30
|
+
)
|
|
31
|
+
except (OSError, subprocess.SubprocessError):
|
|
32
|
+
return None
|
|
33
|
+
if out.returncode != 0:
|
|
34
|
+
return None
|
|
35
|
+
return out.stdout
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class GPU:
|
|
40
|
+
name: str
|
|
41
|
+
vendor: str # "nvidia" | "amd" | "apple"
|
|
42
|
+
total_gb: float
|
|
43
|
+
free_gb: Optional[float] = None # None when the driver can't report it
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class SystemSpecs:
|
|
48
|
+
os_name: str
|
|
49
|
+
arch: str
|
|
50
|
+
cpu: str
|
|
51
|
+
cores: int
|
|
52
|
+
ram_total_gb: float
|
|
53
|
+
ram_available_gb: Optional[float]
|
|
54
|
+
gpus: List[GPU] = field(default_factory=list)
|
|
55
|
+
unified_memory: bool = False # Apple Silicon: RAM and VRAM are one pool
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def vram_total_gb(self) -> float:
|
|
59
|
+
return sum(g.total_gb for g in self.gpus)
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def vram_free_gb(self) -> Optional[float]:
|
|
63
|
+
if not self.gpus or any(g.free_gb is None for g in self.gpus):
|
|
64
|
+
return None
|
|
65
|
+
return sum(g.free_gb for g in self.gpus)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --------------------------------------------------------------------------- #
|
|
69
|
+
# GPU detection
|
|
70
|
+
# --------------------------------------------------------------------------- #
|
|
71
|
+
|
|
72
|
+
def _detect_nvidia() -> List[GPU]:
|
|
73
|
+
out = _run([
|
|
74
|
+
"nvidia-smi",
|
|
75
|
+
"--query-gpu=name,memory.total,memory.free",
|
|
76
|
+
"--format=csv,noheader,nounits",
|
|
77
|
+
])
|
|
78
|
+
if out is None and sys.platform == "win32":
|
|
79
|
+
smi = os.path.join(os.environ.get("SystemRoot", r"C:\Windows"),
|
|
80
|
+
"System32", "nvidia-smi.exe")
|
|
81
|
+
if os.path.exists(smi):
|
|
82
|
+
out = _run([smi, "--query-gpu=name,memory.total,memory.free",
|
|
83
|
+
"--format=csv,noheader,nounits"])
|
|
84
|
+
gpus: List[GPU] = []
|
|
85
|
+
for line in (out or "").strip().splitlines():
|
|
86
|
+
parts = [p.strip() for p in line.split(",")]
|
|
87
|
+
if len(parts) < 3:
|
|
88
|
+
continue
|
|
89
|
+
try:
|
|
90
|
+
total, free = float(parts[-2]) / 1024, float(parts[-1]) / 1024
|
|
91
|
+
except ValueError:
|
|
92
|
+
continue
|
|
93
|
+
gpus.append(GPU(name=parts[0], vendor="nvidia", total_gb=total, free_gb=free))
|
|
94
|
+
return gpus
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _detect_amd_sysfs() -> List[GPU]:
|
|
98
|
+
"""amdgpu kernel driver exposes VRAM in sysfs — works without ROCm."""
|
|
99
|
+
gpus: List[GPU] = []
|
|
100
|
+
for dev in sorted(glob.glob("/sys/class/drm/card[0-9]*/device")):
|
|
101
|
+
try:
|
|
102
|
+
with open(os.path.join(dev, "vendor")) as f:
|
|
103
|
+
if f.read().strip().lower() != "0x1002": # AMD PCI vendor id
|
|
104
|
+
continue
|
|
105
|
+
with open(os.path.join(dev, "mem_info_vram_total")) as f:
|
|
106
|
+
total = int(f.read().strip()) / GiB
|
|
107
|
+
free = None
|
|
108
|
+
used_path = os.path.join(dev, "mem_info_vram_used")
|
|
109
|
+
if os.path.exists(used_path):
|
|
110
|
+
with open(used_path) as f:
|
|
111
|
+
free = total - int(f.read().strip()) / GiB
|
|
112
|
+
except (OSError, ValueError):
|
|
113
|
+
continue
|
|
114
|
+
if total < 0.25: # skip tiny iGPU carve-outs reported by some APUs
|
|
115
|
+
continue
|
|
116
|
+
gpus.append(GPU(name="AMD GPU (amdgpu)", vendor="amd",
|
|
117
|
+
total_gb=total, free_gb=free))
|
|
118
|
+
return gpus
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _detect_amd_rocm() -> List[GPU]:
|
|
122
|
+
out = _run(["rocm-smi", "--showmeminfo", "vram", "--csv"])
|
|
123
|
+
gpus: List[GPU] = []
|
|
124
|
+
for line in (out or "").strip().splitlines():
|
|
125
|
+
m = re.match(r"card(\d+),(\d+),(\d+)", line.replace(" ", ""))
|
|
126
|
+
if not m:
|
|
127
|
+
continue
|
|
128
|
+
total = int(m.group(2)) / GiB
|
|
129
|
+
free = total - int(m.group(3)) / GiB
|
|
130
|
+
gpus.append(GPU(name=f"AMD GPU {m.group(1)} (ROCm)", vendor="amd",
|
|
131
|
+
total_gb=total, free_gb=free))
|
|
132
|
+
return gpus
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _detect_apple(ram_total_gb: float) -> List[GPU]:
|
|
136
|
+
if sys.platform != "darwin" or platform.machine() != "arm64":
|
|
137
|
+
return []
|
|
138
|
+
chip = (_run(["sysctl", "-n", "machdep.cpu.brand_string"]) or "Apple Silicon").strip()
|
|
139
|
+
return [GPU(name=f"{chip} (unified memory)", vendor="apple",
|
|
140
|
+
total_gb=ram_total_gb, free_gb=None)]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# --------------------------------------------------------------------------- #
|
|
144
|
+
# RAM / CPU detection
|
|
145
|
+
# --------------------------------------------------------------------------- #
|
|
146
|
+
|
|
147
|
+
def _ram() -> Tuple[float, Optional[float]]:
|
|
148
|
+
try:
|
|
149
|
+
import psutil # optional dependency
|
|
150
|
+
vm = psutil.virtual_memory()
|
|
151
|
+
return vm.total / GiB, vm.available / GiB
|
|
152
|
+
except ImportError:
|
|
153
|
+
pass
|
|
154
|
+
|
|
155
|
+
if sys.platform.startswith("linux"):
|
|
156
|
+
info = {}
|
|
157
|
+
try:
|
|
158
|
+
with open("/proc/meminfo") as f:
|
|
159
|
+
for line in f:
|
|
160
|
+
key, _, rest = line.partition(":")
|
|
161
|
+
info[key] = int(rest.strip().split()[0]) * 1024
|
|
162
|
+
except (OSError, ValueError, IndexError):
|
|
163
|
+
return 0.0, None
|
|
164
|
+
total = info.get("MemTotal", 0) / GiB
|
|
165
|
+
avail = info.get("MemAvailable")
|
|
166
|
+
return total, (avail / GiB if avail is not None else None)
|
|
167
|
+
|
|
168
|
+
if sys.platform == "darwin":
|
|
169
|
+
out = _run(["sysctl", "-n", "hw.memsize"])
|
|
170
|
+
total = int(out.strip()) / GiB if out else 0.0
|
|
171
|
+
return total, None
|
|
172
|
+
|
|
173
|
+
if sys.platform == "win32":
|
|
174
|
+
class MEMORYSTATUSEX(ctypes.Structure):
|
|
175
|
+
_fields_ = [
|
|
176
|
+
("dwLength", ctypes.c_ulong),
|
|
177
|
+
("dwMemoryLoad", ctypes.c_ulong),
|
|
178
|
+
("ullTotalPhys", ctypes.c_ulonglong),
|
|
179
|
+
("ullAvailPhys", ctypes.c_ulonglong),
|
|
180
|
+
("ullTotalPageFile", ctypes.c_ulonglong),
|
|
181
|
+
("ullAvailPageFile", ctypes.c_ulonglong),
|
|
182
|
+
("ullTotalVirtual", ctypes.c_ulonglong),
|
|
183
|
+
("ullAvailVirtual", ctypes.c_ulonglong),
|
|
184
|
+
("ullAvailExtendedVirtual", ctypes.c_ulonglong),
|
|
185
|
+
]
|
|
186
|
+
|
|
187
|
+
stat = MEMORYSTATUSEX()
|
|
188
|
+
stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
|
|
189
|
+
if ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat)):
|
|
190
|
+
return stat.ullTotalPhys / GiB, stat.ullAvailPhys / GiB
|
|
191
|
+
|
|
192
|
+
return 0.0, None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _cpu_name() -> str:
|
|
196
|
+
if sys.platform.startswith("linux"):
|
|
197
|
+
try:
|
|
198
|
+
with open("/proc/cpuinfo") as f:
|
|
199
|
+
for line in f:
|
|
200
|
+
if line.lower().startswith("model name"):
|
|
201
|
+
return line.split(":", 1)[1].strip()
|
|
202
|
+
except OSError:
|
|
203
|
+
pass
|
|
204
|
+
elif sys.platform == "darwin":
|
|
205
|
+
out = _run(["sysctl", "-n", "machdep.cpu.brand_string"])
|
|
206
|
+
if out:
|
|
207
|
+
return out.strip()
|
|
208
|
+
return platform.processor() or platform.machine() or "Unknown CPU"
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def detect() -> SystemSpecs:
|
|
212
|
+
ram_total, ram_avail = _ram()
|
|
213
|
+
|
|
214
|
+
gpus = _detect_nvidia()
|
|
215
|
+
if not gpus and sys.platform.startswith("linux"):
|
|
216
|
+
gpus = _detect_amd_sysfs() or _detect_amd_rocm()
|
|
217
|
+
unified = False
|
|
218
|
+
if not gpus:
|
|
219
|
+
gpus = _detect_apple(ram_total)
|
|
220
|
+
unified = bool(gpus)
|
|
221
|
+
|
|
222
|
+
return SystemSpecs(
|
|
223
|
+
os_name=f"{platform.system()} {platform.release()}",
|
|
224
|
+
arch=platform.machine(),
|
|
225
|
+
cpu=_cpu_name(),
|
|
226
|
+
cores=os.cpu_count() or 1,
|
|
227
|
+
ram_total_gb=ram_total,
|
|
228
|
+
ram_available_gb=ram_avail,
|
|
229
|
+
gpus=gpus,
|
|
230
|
+
unified_memory=unified,
|
|
231
|
+
)
|
hfit-0.1.0/hfit/model.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""Fetch model metadata from the Hugging Face Hub without downloading weights.
|
|
2
|
+
|
|
3
|
+
Parameter counts come from the Hub's safetensors index (exact, free).
|
|
4
|
+
Architecture details for the KV-cache estimate come from config.json (~1 KB).
|
|
5
|
+
GGUF-only repos are handled per quant file, using file sizes.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import List, Optional, Tuple
|
|
13
|
+
|
|
14
|
+
from huggingface_hub import HfApi, hf_hub_download
|
|
15
|
+
from huggingface_hub.utils import (
|
|
16
|
+
EntryNotFoundError,
|
|
17
|
+
GatedRepoError,
|
|
18
|
+
HfHubHTTPError,
|
|
19
|
+
RepositoryNotFoundError,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
WEIGHT_EXTS = (".safetensors", ".bin", ".pt", ".pth", ".msgpack", ".h5")
|
|
23
|
+
|
|
24
|
+
# bytes per parameter for the native checkpoint, keyed by Hub dtype names
|
|
25
|
+
DTYPE_BYTES = {
|
|
26
|
+
"F64": 8, "F32": 4, "F16": 2, "BF16": 2, "FP16": 2,
|
|
27
|
+
"I64": 8, "I32": 4, "I16": 2, "I8": 1, "U8": 1,
|
|
28
|
+
"F8_E4M3": 1, "F8_E5M2": 1, "I4": 0.5, "U4": 0.5,
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ModelLookupError(Exception):
|
|
33
|
+
"""User-presentable failure while resolving a model."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class KVConfig:
|
|
38
|
+
num_layers: int
|
|
39
|
+
kv_heads: int
|
|
40
|
+
head_dim: int
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def bytes_per_token(self) -> int:
|
|
44
|
+
# K and V, fp16 cache: 2 tensors * 2 bytes
|
|
45
|
+
return 2 * self.num_layers * self.kv_heads * self.head_dim * 2
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class HFModel:
|
|
50
|
+
model_id: str
|
|
51
|
+
params: Optional[int] # total parameter count, None if unknown
|
|
52
|
+
params_estimated: bool # True when derived from file sizes
|
|
53
|
+
native_dtype: Optional[str] # e.g. "BF16"
|
|
54
|
+
weight_bytes: Optional[int] # sum of transformer-format weight files
|
|
55
|
+
gguf_files: List[Tuple[str, int]] = field(default_factory=list)
|
|
56
|
+
gated: bool = False
|
|
57
|
+
pipeline_tag: Optional[str] = None
|
|
58
|
+
downloads: Optional[int] = None
|
|
59
|
+
kv: Optional[KVConfig] = None
|
|
60
|
+
max_ctx: Optional[int] = None
|
|
61
|
+
weight_components: int = 0 # distinct repo folders holding weights
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def params_b(self) -> Optional[float]:
|
|
65
|
+
return self.params / 1e9 if self.params else None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _dig(cfg: dict, *keys, default=None):
|
|
69
|
+
for key in keys:
|
|
70
|
+
if key in cfg and cfg[key] is not None:
|
|
71
|
+
return cfg[key]
|
|
72
|
+
return default
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _text_config(cfg: dict) -> dict:
|
|
76
|
+
"""Return the sub-config that actually describes the language model."""
|
|
77
|
+
if _dig(cfg, "num_hidden_layers", "n_layer", "num_layers") is not None:
|
|
78
|
+
return cfg
|
|
79
|
+
for key in ("text_config", "llm_config", "language_config", "decoder"):
|
|
80
|
+
sub = cfg.get(key)
|
|
81
|
+
if isinstance(sub, dict) and _dig(sub, "num_hidden_layers", "n_layer") is not None:
|
|
82
|
+
return sub
|
|
83
|
+
return cfg
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _parse_config(model_id: str, token: Optional[str]) -> Tuple[Optional[KVConfig], Optional[int], Optional[str]]:
|
|
87
|
+
try:
|
|
88
|
+
path = hf_hub_download(model_id, "config.json", token=token)
|
|
89
|
+
with open(path) as f:
|
|
90
|
+
raw = json.load(f)
|
|
91
|
+
except Exception:
|
|
92
|
+
return None, None, None
|
|
93
|
+
|
|
94
|
+
cfg = _text_config(raw)
|
|
95
|
+
dtype = _dig(cfg, "dtype", "torch_dtype") or _dig(raw, "dtype", "torch_dtype")
|
|
96
|
+
max_ctx = _dig(cfg, "max_position_embeddings", "n_positions", "max_sequence_length")
|
|
97
|
+
|
|
98
|
+
layers = _dig(cfg, "num_hidden_layers", "n_layer", "num_layers")
|
|
99
|
+
heads = _dig(cfg, "num_attention_heads", "n_head")
|
|
100
|
+
hidden = _dig(cfg, "hidden_size", "n_embd", "d_model")
|
|
101
|
+
if not (layers and heads and hidden):
|
|
102
|
+
return None, max_ctx, dtype
|
|
103
|
+
|
|
104
|
+
kv_heads = _dig(cfg, "num_key_value_heads", default=heads)
|
|
105
|
+
head_dim = _dig(cfg, "head_dim", default=hidden // heads)
|
|
106
|
+
return KVConfig(int(layers), int(kv_heads), int(head_dim)), max_ctx, dtype
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def fetch_model(model_id: str, token: Optional[str] = None) -> HFModel:
|
|
110
|
+
api = HfApi(token=token)
|
|
111
|
+
try:
|
|
112
|
+
info = api.model_info(model_id, files_metadata=True)
|
|
113
|
+
except RepositoryNotFoundError:
|
|
114
|
+
raise ModelLookupError(
|
|
115
|
+
f"Model '{model_id}' not found on the Hugging Face Hub.\n"
|
|
116
|
+
"Check the spelling — the id must be 'organization/model-name'."
|
|
117
|
+
) from None
|
|
118
|
+
except GatedRepoError:
|
|
119
|
+
raise ModelLookupError(
|
|
120
|
+
f"'{model_id}' is a gated model. Accept its license on huggingface.co, "
|
|
121
|
+
"then log in with 'huggingface-cli login' or pass --token."
|
|
122
|
+
) from None
|
|
123
|
+
except HfHubHTTPError as e:
|
|
124
|
+
raise ModelLookupError(f"Hub request failed for '{model_id}': {e}") from None
|
|
125
|
+
except Exception as e:
|
|
126
|
+
raise ModelLookupError(
|
|
127
|
+
f"Could not reach the Hugging Face Hub: {e}\nCheck your internet connection."
|
|
128
|
+
) from None
|
|
129
|
+
|
|
130
|
+
gguf_files: List[Tuple[str, int]] = []
|
|
131
|
+
by_dir: dict = {}
|
|
132
|
+
for sib in info.siblings or []:
|
|
133
|
+
name, size = sib.rfilename, sib.size or 0
|
|
134
|
+
low = name.lower()
|
|
135
|
+
if low.endswith(".gguf"):
|
|
136
|
+
gguf_files.append((name, size))
|
|
137
|
+
elif low.endswith(WEIGHT_EXTS) and "optimizer" not in low:
|
|
138
|
+
dirname = name.rsplit("/", 1)[0] if "/" in name else ""
|
|
139
|
+
by_dir.setdefault(dirname, []).append((name, size))
|
|
140
|
+
gguf_files.sort(key=lambda x: x[1])
|
|
141
|
+
|
|
142
|
+
weight_bytes = 0
|
|
143
|
+
for files in by_dir.values():
|
|
144
|
+
weight_bytes += sum(size for _, size in _dedup_formats(files))
|
|
145
|
+
|
|
146
|
+
params = info.safetensors.total if info.safetensors else None
|
|
147
|
+
native_dtype = None
|
|
148
|
+
if info.safetensors and info.safetensors.parameters:
|
|
149
|
+
native_dtype = max(info.safetensors.parameters, key=info.safetensors.parameters.get)
|
|
150
|
+
|
|
151
|
+
kv, max_ctx, cfg_dtype = _parse_config(model_id, token)
|
|
152
|
+
if native_dtype is None and cfg_dtype:
|
|
153
|
+
native_dtype = str(cfg_dtype).upper().replace("FLOAT", "F").replace("BFLOAT", "BF")
|
|
154
|
+
|
|
155
|
+
params_estimated = False
|
|
156
|
+
if params is None and weight_bytes and len(by_dir) == 1:
|
|
157
|
+
# Single-model repo: estimate the count from checkpoint size and dtype.
|
|
158
|
+
# Multi-component repos (speech pipelines, VAE+LLM combos) keep params
|
|
159
|
+
# unset — a single parameter count would be misleading there.
|
|
160
|
+
per_param = DTYPE_BYTES.get(native_dtype or "", 2)
|
|
161
|
+
params = int(weight_bytes / per_param)
|
|
162
|
+
params_estimated = True
|
|
163
|
+
|
|
164
|
+
return HFModel(
|
|
165
|
+
model_id=info.id,
|
|
166
|
+
params=params,
|
|
167
|
+
params_estimated=params_estimated,
|
|
168
|
+
native_dtype=native_dtype,
|
|
169
|
+
weight_bytes=weight_bytes or None,
|
|
170
|
+
gguf_files=gguf_files,
|
|
171
|
+
gated=bool(info.gated),
|
|
172
|
+
pipeline_tag=info.pipeline_tag,
|
|
173
|
+
downloads=info.downloads,
|
|
174
|
+
kv=kv,
|
|
175
|
+
max_ctx=max_ctx,
|
|
176
|
+
weight_components=len(by_dir),
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _dedup_formats(files: List[Tuple[str, int]]) -> List[Tuple[str, int]]:
|
|
181
|
+
"""Drop redundant copies of the same weights within one repo folder.
|
|
182
|
+
|
|
183
|
+
Repos often ship several serializations side by side (safetensors + bin +
|
|
184
|
+
h5, or sharded shards + a consolidated file). Keep only the best format.
|
|
185
|
+
"""
|
|
186
|
+
def fmt_rank(name: str) -> int:
|
|
187
|
+
low = name.lower()
|
|
188
|
+
if low.endswith(".safetensors"):
|
|
189
|
+
return 0
|
|
190
|
+
if low.endswith((".bin", ".pt", ".pth")):
|
|
191
|
+
return 1
|
|
192
|
+
return 2 # .h5, .msgpack
|
|
193
|
+
|
|
194
|
+
best = min(fmt_rank(n) for n, _ in files)
|
|
195
|
+
kept = [(n, s) for n, s in files if fmt_rank(n) == best]
|
|
196
|
+
|
|
197
|
+
sharded = [x for x in kept if "-of-" in x[0].rsplit("/", 1)[-1]]
|
|
198
|
+
if sharded and len(sharded) < len(kept):
|
|
199
|
+
kept = sharded # drop consolidated.* duplicates next to shards
|
|
200
|
+
return kept
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""Rich renderables shared by the CLI and the TUI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import List, Optional
|
|
6
|
+
|
|
7
|
+
from rich.console import Group, RenderableType
|
|
8
|
+
from rich.panel import Panel
|
|
9
|
+
from rich.table import Table
|
|
10
|
+
from rich.text import Text
|
|
11
|
+
|
|
12
|
+
from .estimate import Assessment, Verdict, best_verdict
|
|
13
|
+
from .hw import SystemSpecs
|
|
14
|
+
from .model import HFModel
|
|
15
|
+
|
|
16
|
+
VERDICT_STYLE = {
|
|
17
|
+
Verdict.GPU_OK: ("✔ fits on GPU", "bold green"),
|
|
18
|
+
Verdict.GPU_TIGHT: ("● tight on GPU", "bold yellow"),
|
|
19
|
+
Verdict.CPU_OK: ("◐ CPU/RAM only (slow)", "bold dark_orange"),
|
|
20
|
+
Verdict.NO_FIT: ("✘ does not fit", "bold red"),
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
SUMMARY_LINE = {
|
|
24
|
+
Verdict.GPU_OK: ("This model can run comfortably on your GPU.", "green"),
|
|
25
|
+
Verdict.GPU_TIGHT: ("It fits on your GPU, but barely — close other GPU apps "
|
|
26
|
+
"or lower gpu-memory-utilization.", "yellow"),
|
|
27
|
+
Verdict.CPU_OK: ("Too big for your GPU. It can run from system RAM "
|
|
28
|
+
"(CPU or partial offload) — expect low speed.", "dark_orange"),
|
|
29
|
+
Verdict.NO_FIT: ("This model does not fit this machine in any listed precision.",
|
|
30
|
+
"red"),
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _gb(x: Optional[float]) -> str:
|
|
35
|
+
return f"{x:.2f} GB" if x is not None else "—"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def system_panel(specs: SystemSpecs) -> Panel:
|
|
39
|
+
t = Table.grid(padding=(0, 2))
|
|
40
|
+
t.add_column(style="bold cyan", justify="right")
|
|
41
|
+
t.add_column()
|
|
42
|
+
t.add_row("OS", f"{specs.os_name} ({specs.arch})")
|
|
43
|
+
t.add_row("CPU", f"{specs.cpu} ×{specs.cores}")
|
|
44
|
+
ram = f"{specs.ram_total_gb:.1f} GB total"
|
|
45
|
+
if specs.ram_available_gb is not None:
|
|
46
|
+
ram += f", {specs.ram_available_gb:.1f} GB available"
|
|
47
|
+
t.add_row("RAM", ram)
|
|
48
|
+
if specs.gpus:
|
|
49
|
+
for g in specs.gpus:
|
|
50
|
+
vram = f"{g.total_gb:.1f} GB VRAM"
|
|
51
|
+
if g.free_gb is not None:
|
|
52
|
+
vram += f", {g.free_gb:.1f} GB free"
|
|
53
|
+
if specs.unified_memory:
|
|
54
|
+
vram = "shares system RAM (unified)"
|
|
55
|
+
t.add_row("GPU", f"{g.name} — {vram}")
|
|
56
|
+
else:
|
|
57
|
+
t.add_row("GPU", Text("none detected — verdicts use system RAM only", "dim"))
|
|
58
|
+
return Panel(t, title="[bold]Your machine", border_style="cyan")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def model_panel(model: HFModel) -> Panel:
|
|
62
|
+
t = Table.grid(padding=(0, 2))
|
|
63
|
+
t.add_column(style="bold magenta", justify="right")
|
|
64
|
+
t.add_column()
|
|
65
|
+
if model.params_b:
|
|
66
|
+
est = " (estimated from file sizes)" if model.params_estimated else ""
|
|
67
|
+
t.add_row("Parameters", f"{model.params_b:.2f} B{est}")
|
|
68
|
+
if model.weight_bytes:
|
|
69
|
+
size = f"{model.weight_bytes / 1024**3:.2f} GB"
|
|
70
|
+
if model.weight_components > 1:
|
|
71
|
+
size += f" ({model.weight_components} weight components)"
|
|
72
|
+
t.add_row("On disk", size)
|
|
73
|
+
if model.native_dtype:
|
|
74
|
+
t.add_row("Native dtype", model.native_dtype)
|
|
75
|
+
if model.pipeline_tag:
|
|
76
|
+
t.add_row("Task", model.pipeline_tag)
|
|
77
|
+
if model.max_ctx:
|
|
78
|
+
t.add_row("Max context", f"{model.max_ctx:,} tokens")
|
|
79
|
+
if model.downloads is not None:
|
|
80
|
+
t.add_row("Downloads", f"{model.downloads:,}/month")
|
|
81
|
+
if model.gguf_files:
|
|
82
|
+
t.add_row("GGUF quants", f"{len(model.gguf_files)} file(s) in repo")
|
|
83
|
+
if model.gated:
|
|
84
|
+
t.add_row("Access", Text("gated — license acceptance required", "yellow"))
|
|
85
|
+
return Panel(t, title=f"[bold]{model.model_id}", border_style="magenta")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def results_table(results: List[Assessment], ctx_tokens: int) -> Table:
|
|
89
|
+
table = Table(title=f"Fit assessment @ {ctx_tokens:,}-token context",
|
|
90
|
+
title_style="bold", header_style="bold")
|
|
91
|
+
table.add_column("Precision", overflow="fold")
|
|
92
|
+
table.add_column("Weights", justify="right")
|
|
93
|
+
table.add_column("KV cache", justify="right")
|
|
94
|
+
table.add_column("Required*", justify="right")
|
|
95
|
+
table.add_column("Verdict")
|
|
96
|
+
table.add_column("Note", style="dim", overflow="fold")
|
|
97
|
+
|
|
98
|
+
for r in results:
|
|
99
|
+
label, style = VERDICT_STYLE[r.verdict]
|
|
100
|
+
table.add_row(
|
|
101
|
+
r.label,
|
|
102
|
+
_gb(r.weights_gb),
|
|
103
|
+
_gb(r.kv_gb),
|
|
104
|
+
_gb(r.required_gb),
|
|
105
|
+
Text(label, style=style),
|
|
106
|
+
r.note,
|
|
107
|
+
)
|
|
108
|
+
return table
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def full_report(model: HFModel, specs: SystemSpecs,
|
|
112
|
+
results: List[Assessment], ctx_tokens: int) -> RenderableType:
|
|
113
|
+
parts: List[RenderableType] = [model_panel(model)]
|
|
114
|
+
|
|
115
|
+
if not results:
|
|
116
|
+
parts.append(Text(
|
|
117
|
+
"No parameter count or weight files found for this repo — "
|
|
118
|
+
"cannot estimate memory (it may be a dataset-style or adapter repo).",
|
|
119
|
+
style="yellow",
|
|
120
|
+
))
|
|
121
|
+
return Group(*parts)
|
|
122
|
+
|
|
123
|
+
parts.append(results_table(results, ctx_tokens))
|
|
124
|
+
|
|
125
|
+
if model.max_ctx and ctx_tokens > model.max_ctx:
|
|
126
|
+
parts.append(Text(
|
|
127
|
+
f"Note: the chosen context ({ctx_tokens:,}) exceeds this model's "
|
|
128
|
+
f"maximum ({model.max_ctx:,} tokens).", style="yellow",
|
|
129
|
+
))
|
|
130
|
+
|
|
131
|
+
verdict = best_verdict(results)
|
|
132
|
+
if verdict is not None:
|
|
133
|
+
line, style = SUMMARY_LINE[verdict]
|
|
134
|
+
parts.append(Text(f"\n{line}", style=f"bold {style}"))
|
|
135
|
+
|
|
136
|
+
footnotes = ("*Required = weights + KV cache + runtime overhead "
|
|
137
|
+
f"(~{0.8:.1f} GB + 5% activations).")
|
|
138
|
+
if any(r.kv_gb is None for r in results):
|
|
139
|
+
footnotes += " KV cache unknown for this architecture — a 20% margin was used."
|
|
140
|
+
footnotes += (" INT8/INT4 rows are theoretical: they need an actual quantized "
|
|
141
|
+
"checkpoint or on-the-fly quantization.")
|
|
142
|
+
parts.append(Text(footnotes, style="dim"))
|
|
143
|
+
return Group(*parts)
|
hfit-0.1.0/hfit/tui.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Textual TUI: type a model id, get a fit report against this machine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from rich.text import Text
|
|
8
|
+
from textual import work
|
|
9
|
+
from textual.app import App, ComposeResult
|
|
10
|
+
from textual.containers import Horizontal, VerticalScroll
|
|
11
|
+
from textual.widgets import Footer, Header, Input, Select, Static
|
|
12
|
+
|
|
13
|
+
from . import __version__
|
|
14
|
+
from .estimate import assess
|
|
15
|
+
from .hw import detect
|
|
16
|
+
from .model import HFModel, ModelLookupError, fetch_model
|
|
17
|
+
from .report import full_report, system_panel
|
|
18
|
+
|
|
19
|
+
CTX_CHOICES = [2048, 4096, 8192, 16384, 32768, 65536, 131072]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _ctx_label(tokens: int) -> str:
|
|
23
|
+
return f"{tokens // 1024}k context"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class SpecCheckerApp(App):
|
|
27
|
+
TITLE = "hfit"
|
|
28
|
+
SUB_TITLE = f"v{__version__} — will it run on this machine?"
|
|
29
|
+
|
|
30
|
+
CSS = """
|
|
31
|
+
#topbar { height: 3; dock: top; }
|
|
32
|
+
#model-input { width: 1fr; }
|
|
33
|
+
#ctx-select { width: 24; }
|
|
34
|
+
#results { padding: 0 1; }
|
|
35
|
+
#system-panel { margin-bottom: 1; }
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
BINDINGS = [
|
|
39
|
+
("q", "quit", "Quit"),
|
|
40
|
+
("ctrl+l", "focus_input", "New model"),
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
def __init__(self, ctx_tokens: int = 4096, token: Optional[str] = None):
|
|
44
|
+
super().__init__()
|
|
45
|
+
self.ctx_tokens = ctx_tokens if ctx_tokens in CTX_CHOICES else 4096
|
|
46
|
+
self.hf_token = token
|
|
47
|
+
self.specs = detect()
|
|
48
|
+
self.current_model: Optional[HFModel] = None
|
|
49
|
+
|
|
50
|
+
def compose(self) -> ComposeResult:
|
|
51
|
+
yield Header()
|
|
52
|
+
with Horizontal(id="topbar"):
|
|
53
|
+
yield Input(
|
|
54
|
+
placeholder="org/model-name (e.g. Qwen/Qwen2.5-7B-Instruct) — Enter to check",
|
|
55
|
+
id="model-input",
|
|
56
|
+
)
|
|
57
|
+
yield Select(
|
|
58
|
+
[(_ctx_label(c), c) for c in CTX_CHOICES],
|
|
59
|
+
value=self.ctx_tokens, allow_blank=False, id="ctx-select",
|
|
60
|
+
)
|
|
61
|
+
with VerticalScroll(id="results"):
|
|
62
|
+
yield Static(system_panel(self.specs), id="system-panel")
|
|
63
|
+
yield Static(
|
|
64
|
+
Text("Enter a Hugging Face model id above to start.", style="dim"),
|
|
65
|
+
id="report",
|
|
66
|
+
)
|
|
67
|
+
yield Footer()
|
|
68
|
+
|
|
69
|
+
def on_mount(self) -> None:
|
|
70
|
+
self.query_one("#model-input", Input).focus()
|
|
71
|
+
|
|
72
|
+
def action_focus_input(self) -> None:
|
|
73
|
+
inp = self.query_one("#model-input", Input)
|
|
74
|
+
inp.focus()
|
|
75
|
+
inp.selection = (0, len(inp.value))
|
|
76
|
+
|
|
77
|
+
def on_input_submitted(self, event: Input.Submitted) -> None:
|
|
78
|
+
model_id = event.value.strip().strip("/")
|
|
79
|
+
if model_id:
|
|
80
|
+
self.check_model(model_id)
|
|
81
|
+
|
|
82
|
+
def on_select_changed(self, event: Select.Changed) -> None:
|
|
83
|
+
if event.select.id != "ctx-select" or event.value is None:
|
|
84
|
+
return
|
|
85
|
+
self.ctx_tokens = int(event.value)
|
|
86
|
+
if self.current_model is not None:
|
|
87
|
+
self._render(self.current_model)
|
|
88
|
+
|
|
89
|
+
@work(thread=True, exclusive=True)
|
|
90
|
+
def check_model(self, model_id: str) -> None:
|
|
91
|
+
report = self.query_one("#report", Static)
|
|
92
|
+
self.call_from_thread(
|
|
93
|
+
report.update, Text(f"Fetching metadata for {model_id} …", style="italic yellow")
|
|
94
|
+
)
|
|
95
|
+
try:
|
|
96
|
+
model = fetch_model(model_id, token=self.hf_token)
|
|
97
|
+
except ModelLookupError as e:
|
|
98
|
+
self.call_from_thread(report.update, Text(str(e), style="bold red"))
|
|
99
|
+
return
|
|
100
|
+
self.current_model = model
|
|
101
|
+
self.call_from_thread(self._render, model)
|
|
102
|
+
|
|
103
|
+
def _render(self, model: HFModel) -> None:
|
|
104
|
+
results = assess(model, self.specs, self.ctx_tokens)
|
|
105
|
+
self.query_one("#report", Static).update(
|
|
106
|
+
full_report(model, self.specs, results, self.ctx_tokens)
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def run_tui(ctx_tokens: int = 4096, token: Optional[str] = None) -> None:
|
|
111
|
+
SpecCheckerApp(ctx_tokens=ctx_tokens, token=token).run()
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hfit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Check whether a Hugging Face model fits your local hardware — without downloading the weights.
|
|
5
|
+
Author-email: Samir Nuri <samirnuri714@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/snuri00/hfit
|
|
8
|
+
Project-URL: Issues, https://github.com/snuri00/hfit/issues
|
|
9
|
+
Keywords: huggingface,llm,vram,gpu,memory,tui,hardware,inference
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: System :: Hardware
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: huggingface_hub>=0.23
|
|
22
|
+
Requires-Dist: rich>=13.0
|
|
23
|
+
Requires-Dist: textual>=0.60
|
|
24
|
+
Provides-Extra: full
|
|
25
|
+
Requires-Dist: psutil>=5.9; extra == "full"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# hfit
|
|
29
|
+
|
|
30
|
+
*Hugging Face + fit.*
|
|
31
|
+
|
|
32
|
+
**Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
|
|
33
|
+
|
|
34
|
+
Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
|
|
35
|
+
parameter count and architecture from the Hub's metadata (a few KB of API
|
|
36
|
+
calls), detects your local hardware (GPU VRAM, RAM), and tells you which
|
|
37
|
+
precisions fit:
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
|
|
41
|
+
┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
|
|
42
|
+
┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
|
|
43
|
+
│ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
|
|
44
|
+
│ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
|
|
45
|
+
│ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
|
|
46
|
+
└─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Unlike `accelerate estimate-memory`, this compares the model against **your
|
|
50
|
+
actual hardware** and includes a **KV-cache estimate** at your chosen context
|
|
51
|
+
length computed exactly from the model's architecture (layers × KV heads ×
|
|
52
|
+
head dim), not a flat percentage.
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install hfit
|
|
58
|
+
# or from a clone of this repo:
|
|
59
|
+
pip install .
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
|
|
64
|
+
|
|
65
|
+
## Usage
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
hfit # interactive TUI
|
|
69
|
+
hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
|
|
70
|
+
hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
|
|
71
|
+
hfit some/gated-model --token hf_xxx
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
In the TUI: type a model id, press Enter; switch the context length from the
|
|
75
|
+
dropdown to see the KV cache impact instantly (no refetch). `q` quits.
|
|
76
|
+
|
|
77
|
+
Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
|
|
78
|
+
error, `3` = does not fit at all.
|
|
79
|
+
|
|
80
|
+
## What it understands
|
|
81
|
+
|
|
82
|
+
- **Safetensors repos** — exact parameter count from Hub metadata.
|
|
83
|
+
- **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
|
|
84
|
+
individually, for llama.cpp / Ollama / LM Studio users.
|
|
85
|
+
- **Multi-component repos** (speech/vision pipelines with several weight
|
|
86
|
+
folders) reported as a stored total instead of a fake parameter count.
|
|
87
|
+
- Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
|
|
88
|
+
are deduplicated, not double-counted.
|
|
89
|
+
|
|
90
|
+
## Platform support
|
|
91
|
+
|
|
92
|
+
| Platform | GPU detection | RAM detection |
|
|
93
|
+
|---|---|---|
|
|
94
|
+
| Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
|
|
95
|
+
| Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
|
|
96
|
+
| macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
|
|
97
|
+
|
|
98
|
+
`psutil` is used when available (`pip install .[full]`) but is not required.
|
|
99
|
+
|
|
100
|
+
## How the estimate works
|
|
101
|
+
|
|
102
|
+
```
|
|
103
|
+
required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
- Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
|
|
107
|
+
FP32 / FP16 / INT8 / INT4).
|
|
108
|
+
- KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
|
|
109
|
+
from `config.json`; a 20% margin is used when the architecture is unknown.
|
|
110
|
+
- Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
|
|
111
|
+
|
|
112
|
+
These are pre-flight estimates. For a definitive runtime answer, vLLM's
|
|
113
|
+
`--dry-run` on the actual machine remains the ground truth.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
hfit/__init__.py
|
|
5
|
+
hfit/__main__.py
|
|
6
|
+
hfit/cli.py
|
|
7
|
+
hfit/estimate.py
|
|
8
|
+
hfit/hw.py
|
|
9
|
+
hfit/model.py
|
|
10
|
+
hfit/report.py
|
|
11
|
+
hfit/tui.py
|
|
12
|
+
hfit.egg-info/PKG-INFO
|
|
13
|
+
hfit.egg-info/SOURCES.txt
|
|
14
|
+
hfit.egg-info/dependency_links.txt
|
|
15
|
+
hfit.egg-info/entry_points.txt
|
|
16
|
+
hfit.egg-info/requires.txt
|
|
17
|
+
hfit.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
hfit
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "hfit"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Check whether a Hugging Face model fits your local hardware — without downloading the weights."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.9"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
authors = [{ name = "Samir Nuri", email = "samirnuri714@gmail.com" }]
|
|
9
|
+
keywords = ["huggingface", "llm", "vram", "gpu", "memory", "tui", "hardware", "inference"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 4 - Beta",
|
|
12
|
+
"Environment :: Console",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"License :: OSI Approved :: MIT License",
|
|
15
|
+
"Operating System :: OS Independent",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
18
|
+
"Topic :: System :: Hardware",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"huggingface_hub>=0.23",
|
|
22
|
+
"rich>=13.0",
|
|
23
|
+
"textual>=0.60",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
full = ["psutil>=5.9"]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/snuri00/hfit"
|
|
31
|
+
Issues = "https://github.com/snuri00/hfit/issues"
|
|
32
|
+
|
|
33
|
+
[project.scripts]
|
|
34
|
+
hfit = "hfit.cli:main"
|
|
35
|
+
|
|
36
|
+
[build-system]
|
|
37
|
+
requires = ["setuptools>=61"]
|
|
38
|
+
build-backend = "setuptools.build_meta"
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.packages.find]
|
|
41
|
+
include = ["hfit*"]
|
hfit-0.1.0/setup.cfg
ADDED