hfit 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hfit-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Samir Nuri
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
hfit-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,113 @@
1
+ Metadata-Version: 2.4
2
+ Name: hfit
3
+ Version: 0.1.0
4
+ Summary: Check whether a Hugging Face model fits your local hardware — without downloading the weights.
5
+ Author-email: Samir Nuri <samirnuri714@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/snuri00/hfit
8
+ Project-URL: Issues, https://github.com/snuri00/hfit/issues
9
+ Keywords: huggingface,llm,vram,gpu,memory,tui,hardware,inference
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Classifier: Topic :: System :: Hardware
18
+ Requires-Python: >=3.9
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: huggingface_hub>=0.23
22
+ Requires-Dist: rich>=13.0
23
+ Requires-Dist: textual>=0.60
24
+ Provides-Extra: full
25
+ Requires-Dist: psutil>=5.9; extra == "full"
26
+ Dynamic: license-file
27
+
28
+ # hfit
29
+
30
+ *Hugging Face + fit.*
31
+
32
+ **Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
33
+
34
+ Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
35
+ parameter count and architecture from the Hub's metadata (a few KB of API
36
+ calls), detects your local hardware (GPU VRAM, RAM), and tells you which
37
+ precisions fit:
38
+
39
+ ```
40
+ ┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
41
+ ┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
42
+ ┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
43
+ │ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
44
+ │ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
45
+ │ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
46
+ └─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
47
+ ```
48
+
49
+ Unlike `accelerate estimate-memory`, this compares the model against **your
50
+ actual hardware** and includes a **KV-cache estimate** at your chosen context
51
+ length computed exactly from the model's architecture (layers × KV heads ×
52
+ head dim), not a flat percentage.
53
+
54
+ ## Install
55
+
56
+ ```bash
57
+ pip install hfit
58
+ # or from a clone of this repo:
59
+ pip install .
60
+
61
+ ```
62
+
63
+ Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
64
+
65
+ ## Usage
66
+
67
+ ```bash
68
+ hfit # interactive TUI
69
+ hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
70
+ hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
71
+ hfit some/gated-model --token hf_xxx
72
+ ```
73
+
74
+ In the TUI: type a model id, press Enter; switch the context length from the
75
+ dropdown to see the KV cache impact instantly (no refetch). `q` quits.
76
+
77
+ Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
78
+ error, `3` = does not fit at all.
79
+
80
+ ## What it understands
81
+
82
+ - **Safetensors repos** — exact parameter count from Hub metadata.
83
+ - **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
84
+ individually, for llama.cpp / Ollama / LM Studio users.
85
+ - **Multi-component repos** (speech/vision pipelines with several weight
86
+ folders) reported as a stored total instead of a fake parameter count.
87
+ - Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
88
+ are deduplicated, not double-counted.
89
+
90
+ ## Platform support
91
+
92
+ | Platform | GPU detection | RAM detection |
93
+ |---|---|---|
94
+ | Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
95
+ | Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
96
+ | macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
97
+
98
+ `psutil` is used when available (`pip install .[full]`) but is not required.
99
+
100
+ ## How the estimate works
101
+
102
+ ```
103
+ required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
104
+ ```
105
+
106
+ - Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
107
+ FP32 / FP16 / INT8 / INT4).
108
+ - KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
109
+ from `config.json`; a 20% margin is used when the architecture is unknown.
110
+ - Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
111
+
112
+ These are pre-flight estimates. For a definitive runtime answer, vLLM's
113
+ `--dry-run` on the actual machine remains the ground truth.
hfit-0.1.0/README.md ADDED
@@ -0,0 +1,86 @@
1
+ # hfit
2
+
3
+ *Hugging Face + fit.*
4
+
5
+ **Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
6
+
7
+ Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
8
+ parameter count and architecture from the Hub's metadata (a few KB of API
9
+ calls), detects your local hardware (GPU VRAM, RAM), and tells you which
10
+ precisions fit:
11
+
12
+ ```
13
+ ┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
14
+ ┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
15
+ ┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
16
+ │ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
17
+ │ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
18
+ │ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
19
+ └─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
20
+ ```
21
+
22
+ Unlike `accelerate estimate-memory`, this compares the model against **your
23
+ actual hardware** and includes a **KV-cache estimate** at your chosen context
24
+ length computed exactly from the model's architecture (layers × KV heads ×
25
+ head dim), not a flat percentage.
26
+
27
+ ## Install
28
+
29
+ ```bash
30
+ pip install hfit
31
+ # or from a clone of this repo:
32
+ pip install .
33
+
34
+ ```
35
+
36
+ Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
37
+
38
+ ## Usage
39
+
40
+ ```bash
41
+ hfit # interactive TUI
42
+ hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
43
+ hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
44
+ hfit some/gated-model --token hf_xxx
45
+ ```
46
+
47
+ In the TUI: type a model id, press Enter; switch the context length from the
48
+ dropdown to see the KV cache impact instantly (no refetch). `q` quits.
49
+
50
+ Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
51
+ error, `3` = does not fit at all.
52
+
53
+ ## What it understands
54
+
55
+ - **Safetensors repos** — exact parameter count from Hub metadata.
56
+ - **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
57
+ individually, for llama.cpp / Ollama / LM Studio users.
58
+ - **Multi-component repos** (speech/vision pipelines with several weight
59
+ folders) reported as a stored total instead of a fake parameter count.
60
+ - Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
61
+ are deduplicated, not double-counted.
62
+
63
+ ## Platform support
64
+
65
+ | Platform | GPU detection | RAM detection |
66
+ |---|---|---|
67
+ | Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
68
+ | Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
69
+ | macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
70
+
71
+ `psutil` is used when available (`pip install .[full]`) but is not required.
72
+
73
+ ## How the estimate works
74
+
75
+ ```
76
+ required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
77
+ ```
78
+
79
+ - Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
80
+ FP32 / FP16 / INT8 / INT4).
81
+ - KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
82
+ from `config.json`; a 20% margin is used when the architecture is unknown.
83
+ - Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
84
+
85
+ These are pre-flight estimates. For a definitive runtime answer, vLLM's
86
+ `--dry-run` on the actual machine remains the ground truth.
@@ -0,0 +1,3 @@
1
+ """hfit — will this Hugging Face model fit on my machine?"""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
hfit-0.1.0/hfit/cli.py ADDED
@@ -0,0 +1,65 @@
1
+ """Entry point: `hfit MODEL_ID` prints a report, no args opens the TUI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from typing import List, Optional
8
+
9
+ from rich.console import Console
10
+
11
+ from . import __version__
12
+ from .estimate import Verdict, assess, best_verdict
13
+ from .hw import detect
14
+ from .model import ModelLookupError, fetch_model
15
+ from .report import full_report, system_panel
16
+
17
+ EXIT_OK, EXIT_ERROR, EXIT_NO_FIT = 0, 1, 3
18
+
19
+
20
+ def run_check(model_id: str, ctx_tokens: int, token: Optional[str]) -> int:
21
+ console = Console()
22
+ specs = detect()
23
+ console.print(system_panel(specs))
24
+
25
+ with console.status(f"Fetching metadata for [bold]{model_id}[/] …"):
26
+ try:
27
+ model = fetch_model(model_id, token=token)
28
+ except ModelLookupError as e:
29
+ console.print(f"[bold red]{e}[/]")
30
+ return EXIT_ERROR
31
+
32
+ results = assess(model, specs, ctx_tokens)
33
+ console.print(full_report(model, specs, results, ctx_tokens))
34
+
35
+ verdict = best_verdict(results)
36
+ return EXIT_NO_FIT if verdict in (Verdict.NO_FIT, None) else EXIT_OK
37
+
38
+
39
+ def main(argv: Optional[List[str]] = None) -> int:
40
+ parser = argparse.ArgumentParser(
41
+ prog="hfit",
42
+ description="Check whether a Hugging Face model fits your local hardware, "
43
+ "without downloading the weights.",
44
+ )
45
+ parser.add_argument("model", nargs="?",
46
+ help="Hugging Face model id (org/name). Omit to open the TUI.")
47
+ parser.add_argument("--ctx", type=int, default=4096, metavar="TOKENS",
48
+ help="context length for the KV-cache estimate (default: 4096)")
49
+ parser.add_argument("--token", default=None,
50
+ help="HF access token for gated/private models "
51
+ "(defaults to your cached huggingface-cli login)")
52
+ parser.add_argument("--version", action="version",
53
+ version=f"%(prog)s {__version__}")
54
+ args = parser.parse_args(argv)
55
+
56
+ if args.model:
57
+ return run_check(args.model.strip().strip("/"), args.ctx, args.token)
58
+
59
+ from .tui import run_tui # imported lazily so plain CLI use stays snappy
60
+ run_tui(ctx_tokens=args.ctx, token=args.token)
61
+ return EXIT_OK
62
+
63
+
64
+ if __name__ == "__main__":
65
+ sys.exit(main())
@@ -0,0 +1,132 @@
1
+ """Memory math and fit verdicts.
2
+
3
+ Required memory = weights + KV cache at the chosen context + runtime overhead.
4
+ The KV cache is computed exactly from the architecture when config.json is
5
+ available, otherwise approximated as 20% of the weights.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ from enum import Enum
12
+ from typing import List, Optional
13
+
14
+ from .hw import SystemSpecs
15
+ from .model import HFModel
16
+
17
+ GiB = 1024**3
18
+
19
+ # label, bytes per parameter, needs a quantized checkpoint?
20
+ PRECISIONS = [
21
+ ("FP32", 4.0, False),
22
+ ("FP16 / BF16", 2.0, False),
23
+ ("INT8 (quantized)", 1.0, True),
24
+ ("INT4 (quantized)", 0.5, True),
25
+ ]
26
+
27
+ # Fractions of reported memory treated as actually usable: leave headroom for
28
+ # the CUDA context and fragmentation on GPUs, and for the OS on system RAM.
29
+ GPU_HEADROOM = 0.93
30
+ RAM_HEADROOM = 0.85
31
+ FRAMEWORK_OVERHEAD_GB = 0.8 # CUDA/Metal context, runtime buffers
32
+ ACTIVATION_FRACTION = 0.05 # inference activations, scales with model size
33
+
34
+
35
+ class Verdict(Enum):
36
+ GPU_OK = "gpu_ok"
37
+ GPU_TIGHT = "gpu_tight"
38
+ CPU_OK = "cpu_ok"
39
+ NO_FIT = "no_fit"
40
+
41
+
42
+ @dataclass
43
+ class Assessment:
44
+ label: str
45
+ weights_gb: float
46
+ kv_gb: Optional[float] # None -> folded into overhead as a rough margin
47
+ required_gb: float
48
+ verdict: Verdict
49
+ note: str = ""
50
+
51
+
52
+ def _kv_gb(model: HFModel, ctx_tokens: int) -> Optional[float]:
53
+ if model.kv is None:
54
+ return None
55
+ return model.kv.bytes_per_token * ctx_tokens / GiB
56
+
57
+
58
+ def _required(weights_gb: float, kv_gb: Optional[float]) -> float:
59
+ overhead = FRAMEWORK_OVERHEAD_GB + weights_gb * ACTIVATION_FRACTION
60
+ if kv_gb is None:
61
+ overhead += weights_gb * 0.20 # no architecture info: rough KV margin
62
+ kv_gb = 0.0
63
+ return weights_gb + kv_gb + overhead
64
+
65
+
66
+ def _judge(required_gb: float, specs: SystemSpecs) -> Verdict:
67
+ ram_budget = (specs.ram_available_gb or specs.ram_total_gb * 0.8) * RAM_HEADROOM
68
+
69
+ if specs.unified_memory:
70
+ # One memory pool: the GPU can address (most of) system RAM.
71
+ if required_gb <= ram_budget:
72
+ return Verdict.GPU_OK
73
+ if required_gb <= specs.ram_total_gb * RAM_HEADROOM:
74
+ return Verdict.GPU_TIGHT
75
+ return Verdict.NO_FIT
76
+
77
+ if specs.gpus:
78
+ free = specs.vram_free_gb
79
+ vram_budget = (free if free is not None else specs.vram_total_gb) * GPU_HEADROOM
80
+ if required_gb <= vram_budget:
81
+ return Verdict.GPU_OK
82
+ if required_gb <= specs.vram_total_gb:
83
+ return Verdict.GPU_TIGHT
84
+ if required_gb <= ram_budget:
85
+ return Verdict.CPU_OK
86
+ return Verdict.NO_FIT
87
+
88
+
89
+ def assess(model: HFModel, specs: SystemSpecs, ctx_tokens: int = 4096) -> List[Assessment]:
90
+ results: List[Assessment] = []
91
+ kv = _kv_gb(model, ctx_tokens)
92
+
93
+ if model.params:
94
+ native = (model.native_dtype or "").upper()
95
+ for label, bytes_pp, needs_quant in PRECISIONS:
96
+ weights = model.params * bytes_pp / GiB
97
+ required = _required(weights, kv)
98
+ note = ""
99
+ if needs_quant:
100
+ note = "needs a quantized build (AWQ/GPTQ/bitsandbytes)"
101
+ elif "16" in label and native in ("F16", "BF16", "FP16"):
102
+ label += " (native)"
103
+ elif label == "FP32" and native in ("F16", "BF16", "FP16"):
104
+ note = "upcast — rarely useful for inference"
105
+ results.append(Assessment(label, weights, kv, required,
106
+ _judge(required, specs), note))
107
+
108
+ if model.params is None and model.weight_bytes:
109
+ # Multi-component pipeline repo: report the stored total, no dtype rows.
110
+ weights = model.weight_bytes / GiB
111
+ required = _required(weights, kv)
112
+ results.append(Assessment(
113
+ f"Full pipeline as stored ({model.weight_components} components)",
114
+ weights, kv, required, _judge(required, specs),
115
+ "sum of all weight folders in the repo",
116
+ ))
117
+
118
+ for name, size in model.gguf_files:
119
+ weights = size / GiB
120
+ required = _required(weights, kv)
121
+ results.append(Assessment(f"GGUF: {name}", weights, kv, required,
122
+ _judge(required, specs),
123
+ "runs with llama.cpp / Ollama / LM Studio"))
124
+ return results
125
+
126
+
127
+ def best_verdict(results: List[Assessment]) -> Optional[Verdict]:
128
+ order = [Verdict.GPU_OK, Verdict.GPU_TIGHT, Verdict.CPU_OK, Verdict.NO_FIT]
129
+ for v in order:
130
+ if any(r.verdict == v for r in results):
131
+ return v
132
+ return None
hfit-0.1.0/hfit/hw.py ADDED
@@ -0,0 +1,231 @@
1
+ """Cross-platform hardware detection with no hard dependencies.
2
+
3
+ Detection order per vendor, cheapest reliable source first:
4
+ NVIDIA -> nvidia-smi (Linux/Windows/WSL)
5
+ AMD -> /sys/class/drm sysfs (Linux, no ROCm needed), then rocm-smi
6
+ Apple -> unified memory on arm64 macOS (VRAM budget == system RAM)
7
+ RAM -> psutil if installed, else /proc/meminfo, sysctl, or WinAPI
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import ctypes
13
+ import glob
14
+ import os
15
+ import platform
16
+ import re
17
+ import subprocess
18
+ import sys
19
+ from dataclasses import dataclass, field
20
+ from typing import List, Optional, Tuple
21
+
22
+ GiB = 1024**3
23
+
24
+
25
+ def _run(cmd: List[str], timeout: float = 10.0) -> Optional[str]:
26
+ try:
27
+ out = subprocess.run(
28
+ cmd, capture_output=True, text=True, timeout=timeout,
29
+ creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0),
30
+ )
31
+ except (OSError, subprocess.SubprocessError):
32
+ return None
33
+ if out.returncode != 0:
34
+ return None
35
+ return out.stdout
36
+
37
+
38
+ @dataclass
39
+ class GPU:
40
+ name: str
41
+ vendor: str # "nvidia" | "amd" | "apple"
42
+ total_gb: float
43
+ free_gb: Optional[float] = None # None when the driver can't report it
44
+
45
+
46
+ @dataclass
47
+ class SystemSpecs:
48
+ os_name: str
49
+ arch: str
50
+ cpu: str
51
+ cores: int
52
+ ram_total_gb: float
53
+ ram_available_gb: Optional[float]
54
+ gpus: List[GPU] = field(default_factory=list)
55
+ unified_memory: bool = False # Apple Silicon: RAM and VRAM are one pool
56
+
57
+ @property
58
+ def vram_total_gb(self) -> float:
59
+ return sum(g.total_gb for g in self.gpus)
60
+
61
+ @property
62
+ def vram_free_gb(self) -> Optional[float]:
63
+ if not self.gpus or any(g.free_gb is None for g in self.gpus):
64
+ return None
65
+ return sum(g.free_gb for g in self.gpus)
66
+
67
+
68
+ # --------------------------------------------------------------------------- #
69
+ # GPU detection
70
+ # --------------------------------------------------------------------------- #
71
+
72
+ def _detect_nvidia() -> List[GPU]:
73
+ out = _run([
74
+ "nvidia-smi",
75
+ "--query-gpu=name,memory.total,memory.free",
76
+ "--format=csv,noheader,nounits",
77
+ ])
78
+ if out is None and sys.platform == "win32":
79
+ smi = os.path.join(os.environ.get("SystemRoot", r"C:\Windows"),
80
+ "System32", "nvidia-smi.exe")
81
+ if os.path.exists(smi):
82
+ out = _run([smi, "--query-gpu=name,memory.total,memory.free",
83
+ "--format=csv,noheader,nounits"])
84
+ gpus: List[GPU] = []
85
+ for line in (out or "").strip().splitlines():
86
+ parts = [p.strip() for p in line.split(",")]
87
+ if len(parts) < 3:
88
+ continue
89
+ try:
90
+ total, free = float(parts[-2]) / 1024, float(parts[-1]) / 1024
91
+ except ValueError:
92
+ continue
93
+ gpus.append(GPU(name=parts[0], vendor="nvidia", total_gb=total, free_gb=free))
94
+ return gpus
95
+
96
+
97
+ def _detect_amd_sysfs() -> List[GPU]:
98
+ """amdgpu kernel driver exposes VRAM in sysfs — works without ROCm."""
99
+ gpus: List[GPU] = []
100
+ for dev in sorted(glob.glob("/sys/class/drm/card[0-9]*/device")):
101
+ try:
102
+ with open(os.path.join(dev, "vendor")) as f:
103
+ if f.read().strip().lower() != "0x1002": # AMD PCI vendor id
104
+ continue
105
+ with open(os.path.join(dev, "mem_info_vram_total")) as f:
106
+ total = int(f.read().strip()) / GiB
107
+ free = None
108
+ used_path = os.path.join(dev, "mem_info_vram_used")
109
+ if os.path.exists(used_path):
110
+ with open(used_path) as f:
111
+ free = total - int(f.read().strip()) / GiB
112
+ except (OSError, ValueError):
113
+ continue
114
+ if total < 0.25: # skip tiny iGPU carve-outs reported by some APUs
115
+ continue
116
+ gpus.append(GPU(name="AMD GPU (amdgpu)", vendor="amd",
117
+ total_gb=total, free_gb=free))
118
+ return gpus
119
+
120
+
121
+ def _detect_amd_rocm() -> List[GPU]:
122
+ out = _run(["rocm-smi", "--showmeminfo", "vram", "--csv"])
123
+ gpus: List[GPU] = []
124
+ for line in (out or "").strip().splitlines():
125
+ m = re.match(r"card(\d+),(\d+),(\d+)", line.replace(" ", ""))
126
+ if not m:
127
+ continue
128
+ total = int(m.group(2)) / GiB
129
+ free = total - int(m.group(3)) / GiB
130
+ gpus.append(GPU(name=f"AMD GPU {m.group(1)} (ROCm)", vendor="amd",
131
+ total_gb=total, free_gb=free))
132
+ return gpus
133
+
134
+
135
+ def _detect_apple(ram_total_gb: float) -> List[GPU]:
136
+ if sys.platform != "darwin" or platform.machine() != "arm64":
137
+ return []
138
+ chip = (_run(["sysctl", "-n", "machdep.cpu.brand_string"]) or "Apple Silicon").strip()
139
+ return [GPU(name=f"{chip} (unified memory)", vendor="apple",
140
+ total_gb=ram_total_gb, free_gb=None)]
141
+
142
+
143
+ # --------------------------------------------------------------------------- #
144
+ # RAM / CPU detection
145
+ # --------------------------------------------------------------------------- #
146
+
147
+ def _ram() -> Tuple[float, Optional[float]]:
148
+ try:
149
+ import psutil # optional dependency
150
+ vm = psutil.virtual_memory()
151
+ return vm.total / GiB, vm.available / GiB
152
+ except ImportError:
153
+ pass
154
+
155
+ if sys.platform.startswith("linux"):
156
+ info = {}
157
+ try:
158
+ with open("/proc/meminfo") as f:
159
+ for line in f:
160
+ key, _, rest = line.partition(":")
161
+ info[key] = int(rest.strip().split()[0]) * 1024
162
+ except (OSError, ValueError, IndexError):
163
+ return 0.0, None
164
+ total = info.get("MemTotal", 0) / GiB
165
+ avail = info.get("MemAvailable")
166
+ return total, (avail / GiB if avail is not None else None)
167
+
168
+ if sys.platform == "darwin":
169
+ out = _run(["sysctl", "-n", "hw.memsize"])
170
+ total = int(out.strip()) / GiB if out else 0.0
171
+ return total, None
172
+
173
+ if sys.platform == "win32":
174
+ class MEMORYSTATUSEX(ctypes.Structure):
175
+ _fields_ = [
176
+ ("dwLength", ctypes.c_ulong),
177
+ ("dwMemoryLoad", ctypes.c_ulong),
178
+ ("ullTotalPhys", ctypes.c_ulonglong),
179
+ ("ullAvailPhys", ctypes.c_ulonglong),
180
+ ("ullTotalPageFile", ctypes.c_ulonglong),
181
+ ("ullAvailPageFile", ctypes.c_ulonglong),
182
+ ("ullTotalVirtual", ctypes.c_ulonglong),
183
+ ("ullAvailVirtual", ctypes.c_ulonglong),
184
+ ("ullAvailExtendedVirtual", ctypes.c_ulonglong),
185
+ ]
186
+
187
+ stat = MEMORYSTATUSEX()
188
+ stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
189
+ if ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat)):
190
+ return stat.ullTotalPhys / GiB, stat.ullAvailPhys / GiB
191
+
192
+ return 0.0, None
193
+
194
+
195
+ def _cpu_name() -> str:
196
+ if sys.platform.startswith("linux"):
197
+ try:
198
+ with open("/proc/cpuinfo") as f:
199
+ for line in f:
200
+ if line.lower().startswith("model name"):
201
+ return line.split(":", 1)[1].strip()
202
+ except OSError:
203
+ pass
204
+ elif sys.platform == "darwin":
205
+ out = _run(["sysctl", "-n", "machdep.cpu.brand_string"])
206
+ if out:
207
+ return out.strip()
208
+ return platform.processor() or platform.machine() or "Unknown CPU"
209
+
210
+
211
+ def detect() -> SystemSpecs:
212
+ ram_total, ram_avail = _ram()
213
+
214
+ gpus = _detect_nvidia()
215
+ if not gpus and sys.platform.startswith("linux"):
216
+ gpus = _detect_amd_sysfs() or _detect_amd_rocm()
217
+ unified = False
218
+ if not gpus:
219
+ gpus = _detect_apple(ram_total)
220
+ unified = bool(gpus)
221
+
222
+ return SystemSpecs(
223
+ os_name=f"{platform.system()} {platform.release()}",
224
+ arch=platform.machine(),
225
+ cpu=_cpu_name(),
226
+ cores=os.cpu_count() or 1,
227
+ ram_total_gb=ram_total,
228
+ ram_available_gb=ram_avail,
229
+ gpus=gpus,
230
+ unified_memory=unified,
231
+ )
@@ -0,0 +1,200 @@
1
+ """Fetch model metadata from the Hugging Face Hub without downloading weights.
2
+
3
+ Parameter counts come from the Hub's safetensors index (exact, free).
4
+ Architecture details for the KV-cache estimate come from config.json (~1 KB).
5
+ GGUF-only repos are handled per quant file, using file sizes.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from dataclasses import dataclass, field
12
+ from typing import List, Optional, Tuple
13
+
14
+ from huggingface_hub import HfApi, hf_hub_download
15
+ from huggingface_hub.utils import (
16
+ EntryNotFoundError,
17
+ GatedRepoError,
18
+ HfHubHTTPError,
19
+ RepositoryNotFoundError,
20
+ )
21
+
22
+ WEIGHT_EXTS = (".safetensors", ".bin", ".pt", ".pth", ".msgpack", ".h5")
23
+
24
+ # bytes per parameter for the native checkpoint, keyed by Hub dtype names
25
+ DTYPE_BYTES = {
26
+ "F64": 8, "F32": 4, "F16": 2, "BF16": 2, "FP16": 2,
27
+ "I64": 8, "I32": 4, "I16": 2, "I8": 1, "U8": 1,
28
+ "F8_E4M3": 1, "F8_E5M2": 1, "I4": 0.5, "U4": 0.5,
29
+ }
30
+
31
+
32
+ class ModelLookupError(Exception):
33
+ """User-presentable failure while resolving a model."""
34
+
35
+
36
+ @dataclass
37
+ class KVConfig:
38
+ num_layers: int
39
+ kv_heads: int
40
+ head_dim: int
41
+
42
+ @property
43
+ def bytes_per_token(self) -> int:
44
+ # K and V, fp16 cache: 2 tensors * 2 bytes
45
+ return 2 * self.num_layers * self.kv_heads * self.head_dim * 2
46
+
47
+
48
+ @dataclass
49
+ class HFModel:
50
+ model_id: str
51
+ params: Optional[int] # total parameter count, None if unknown
52
+ params_estimated: bool # True when derived from file sizes
53
+ native_dtype: Optional[str] # e.g. "BF16"
54
+ weight_bytes: Optional[int] # sum of transformer-format weight files
55
+ gguf_files: List[Tuple[str, int]] = field(default_factory=list)
56
+ gated: bool = False
57
+ pipeline_tag: Optional[str] = None
58
+ downloads: Optional[int] = None
59
+ kv: Optional[KVConfig] = None
60
+ max_ctx: Optional[int] = None
61
+ weight_components: int = 0 # distinct repo folders holding weights
62
+
63
+ @property
64
+ def params_b(self) -> Optional[float]:
65
+ return self.params / 1e9 if self.params else None
66
+
67
+
68
+ def _dig(cfg: dict, *keys, default=None):
69
+ for key in keys:
70
+ if key in cfg and cfg[key] is not None:
71
+ return cfg[key]
72
+ return default
73
+
74
+
75
+ def _text_config(cfg: dict) -> dict:
76
+ """Return the sub-config that actually describes the language model."""
77
+ if _dig(cfg, "num_hidden_layers", "n_layer", "num_layers") is not None:
78
+ return cfg
79
+ for key in ("text_config", "llm_config", "language_config", "decoder"):
80
+ sub = cfg.get(key)
81
+ if isinstance(sub, dict) and _dig(sub, "num_hidden_layers", "n_layer") is not None:
82
+ return sub
83
+ return cfg
84
+
85
+
86
+ def _parse_config(model_id: str, token: Optional[str]) -> Tuple[Optional[KVConfig], Optional[int], Optional[str]]:
87
+ try:
88
+ path = hf_hub_download(model_id, "config.json", token=token)
89
+ with open(path) as f:
90
+ raw = json.load(f)
91
+ except Exception:
92
+ return None, None, None
93
+
94
+ cfg = _text_config(raw)
95
+ dtype = _dig(cfg, "dtype", "torch_dtype") or _dig(raw, "dtype", "torch_dtype")
96
+ max_ctx = _dig(cfg, "max_position_embeddings", "n_positions", "max_sequence_length")
97
+
98
+ layers = _dig(cfg, "num_hidden_layers", "n_layer", "num_layers")
99
+ heads = _dig(cfg, "num_attention_heads", "n_head")
100
+ hidden = _dig(cfg, "hidden_size", "n_embd", "d_model")
101
+ if not (layers and heads and hidden):
102
+ return None, max_ctx, dtype
103
+
104
+ kv_heads = _dig(cfg, "num_key_value_heads", default=heads)
105
+ head_dim = _dig(cfg, "head_dim", default=hidden // heads)
106
+ return KVConfig(int(layers), int(kv_heads), int(head_dim)), max_ctx, dtype
107
+
108
+
109
+ def fetch_model(model_id: str, token: Optional[str] = None) -> HFModel:
110
+ api = HfApi(token=token)
111
+ try:
112
+ info = api.model_info(model_id, files_metadata=True)
113
+ except RepositoryNotFoundError:
114
+ raise ModelLookupError(
115
+ f"Model '{model_id}' not found on the Hugging Face Hub.\n"
116
+ "Check the spelling — the id must be 'organization/model-name'."
117
+ ) from None
118
+ except GatedRepoError:
119
+ raise ModelLookupError(
120
+ f"'{model_id}' is a gated model. Accept its license on huggingface.co, "
121
+ "then log in with 'huggingface-cli login' or pass --token."
122
+ ) from None
123
+ except HfHubHTTPError as e:
124
+ raise ModelLookupError(f"Hub request failed for '{model_id}': {e}") from None
125
+ except Exception as e:
126
+ raise ModelLookupError(
127
+ f"Could not reach the Hugging Face Hub: {e}\nCheck your internet connection."
128
+ ) from None
129
+
130
+ gguf_files: List[Tuple[str, int]] = []
131
+ by_dir: dict = {}
132
+ for sib in info.siblings or []:
133
+ name, size = sib.rfilename, sib.size or 0
134
+ low = name.lower()
135
+ if low.endswith(".gguf"):
136
+ gguf_files.append((name, size))
137
+ elif low.endswith(WEIGHT_EXTS) and "optimizer" not in low:
138
+ dirname = name.rsplit("/", 1)[0] if "/" in name else ""
139
+ by_dir.setdefault(dirname, []).append((name, size))
140
+ gguf_files.sort(key=lambda x: x[1])
141
+
142
+ weight_bytes = 0
143
+ for files in by_dir.values():
144
+ weight_bytes += sum(size for _, size in _dedup_formats(files))
145
+
146
+ params = info.safetensors.total if info.safetensors else None
147
+ native_dtype = None
148
+ if info.safetensors and info.safetensors.parameters:
149
+ native_dtype = max(info.safetensors.parameters, key=info.safetensors.parameters.get)
150
+
151
+ kv, max_ctx, cfg_dtype = _parse_config(model_id, token)
152
+ if native_dtype is None and cfg_dtype:
153
+ native_dtype = str(cfg_dtype).upper().replace("FLOAT", "F").replace("BFLOAT", "BF")
154
+
155
+ params_estimated = False
156
+ if params is None and weight_bytes and len(by_dir) == 1:
157
+ # Single-model repo: estimate the count from checkpoint size and dtype.
158
+ # Multi-component repos (speech pipelines, VAE+LLM combos) keep params
159
+ # unset — a single parameter count would be misleading there.
160
+ per_param = DTYPE_BYTES.get(native_dtype or "", 2)
161
+ params = int(weight_bytes / per_param)
162
+ params_estimated = True
163
+
164
+ return HFModel(
165
+ model_id=info.id,
166
+ params=params,
167
+ params_estimated=params_estimated,
168
+ native_dtype=native_dtype,
169
+ weight_bytes=weight_bytes or None,
170
+ gguf_files=gguf_files,
171
+ gated=bool(info.gated),
172
+ pipeline_tag=info.pipeline_tag,
173
+ downloads=info.downloads,
174
+ kv=kv,
175
+ max_ctx=max_ctx,
176
+ weight_components=len(by_dir),
177
+ )
178
+
179
+
180
+ def _dedup_formats(files: List[Tuple[str, int]]) -> List[Tuple[str, int]]:
181
+ """Drop redundant copies of the same weights within one repo folder.
182
+
183
+ Repos often ship several serializations side by side (safetensors + bin +
184
+ h5, or sharded shards + a consolidated file). Keep only the best format.
185
+ """
186
+ def fmt_rank(name: str) -> int:
187
+ low = name.lower()
188
+ if low.endswith(".safetensors"):
189
+ return 0
190
+ if low.endswith((".bin", ".pt", ".pth")):
191
+ return 1
192
+ return 2 # .h5, .msgpack
193
+
194
+ best = min(fmt_rank(n) for n, _ in files)
195
+ kept = [(n, s) for n, s in files if fmt_rank(n) == best]
196
+
197
+ sharded = [x for x in kept if "-of-" in x[0].rsplit("/", 1)[-1]]
198
+ if sharded and len(sharded) < len(kept):
199
+ kept = sharded # drop consolidated.* duplicates next to shards
200
+ return kept
@@ -0,0 +1,143 @@
1
+ """Rich renderables shared by the CLI and the TUI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import List, Optional
6
+
7
+ from rich.console import Group, RenderableType
8
+ from rich.panel import Panel
9
+ from rich.table import Table
10
+ from rich.text import Text
11
+
12
+ from .estimate import Assessment, Verdict, best_verdict
13
+ from .hw import SystemSpecs
14
+ from .model import HFModel
15
+
16
+ VERDICT_STYLE = {
17
+ Verdict.GPU_OK: ("✔ fits on GPU", "bold green"),
18
+ Verdict.GPU_TIGHT: ("● tight on GPU", "bold yellow"),
19
+ Verdict.CPU_OK: ("◐ CPU/RAM only (slow)", "bold dark_orange"),
20
+ Verdict.NO_FIT: ("✘ does not fit", "bold red"),
21
+ }
22
+
23
+ SUMMARY_LINE = {
24
+ Verdict.GPU_OK: ("This model can run comfortably on your GPU.", "green"),
25
+ Verdict.GPU_TIGHT: ("It fits on your GPU, but barely — close other GPU apps "
26
+ "or lower gpu-memory-utilization.", "yellow"),
27
+ Verdict.CPU_OK: ("Too big for your GPU. It can run from system RAM "
28
+ "(CPU or partial offload) — expect low speed.", "dark_orange"),
29
+ Verdict.NO_FIT: ("This model does not fit this machine in any listed precision.",
30
+ "red"),
31
+ }
32
+
33
+
34
+ def _gb(x: Optional[float]) -> str:
35
+ return f"{x:.2f} GB" if x is not None else "—"
36
+
37
+
38
+ def system_panel(specs: SystemSpecs) -> Panel:
39
+ t = Table.grid(padding=(0, 2))
40
+ t.add_column(style="bold cyan", justify="right")
41
+ t.add_column()
42
+ t.add_row("OS", f"{specs.os_name} ({specs.arch})")
43
+ t.add_row("CPU", f"{specs.cpu} ×{specs.cores}")
44
+ ram = f"{specs.ram_total_gb:.1f} GB total"
45
+ if specs.ram_available_gb is not None:
46
+ ram += f", {specs.ram_available_gb:.1f} GB available"
47
+ t.add_row("RAM", ram)
48
+ if specs.gpus:
49
+ for g in specs.gpus:
50
+ vram = f"{g.total_gb:.1f} GB VRAM"
51
+ if g.free_gb is not None:
52
+ vram += f", {g.free_gb:.1f} GB free"
53
+ if specs.unified_memory:
54
+ vram = "shares system RAM (unified)"
55
+ t.add_row("GPU", f"{g.name} — {vram}")
56
+ else:
57
+ t.add_row("GPU", Text("none detected — verdicts use system RAM only", "dim"))
58
+ return Panel(t, title="[bold]Your machine", border_style="cyan")
59
+
60
+
61
+ def model_panel(model: HFModel) -> Panel:
62
+ t = Table.grid(padding=(0, 2))
63
+ t.add_column(style="bold magenta", justify="right")
64
+ t.add_column()
65
+ if model.params_b:
66
+ est = " (estimated from file sizes)" if model.params_estimated else ""
67
+ t.add_row("Parameters", f"{model.params_b:.2f} B{est}")
68
+ if model.weight_bytes:
69
+ size = f"{model.weight_bytes / 1024**3:.2f} GB"
70
+ if model.weight_components > 1:
71
+ size += f" ({model.weight_components} weight components)"
72
+ t.add_row("On disk", size)
73
+ if model.native_dtype:
74
+ t.add_row("Native dtype", model.native_dtype)
75
+ if model.pipeline_tag:
76
+ t.add_row("Task", model.pipeline_tag)
77
+ if model.max_ctx:
78
+ t.add_row("Max context", f"{model.max_ctx:,} tokens")
79
+ if model.downloads is not None:
80
+ t.add_row("Downloads", f"{model.downloads:,}/month")
81
+ if model.gguf_files:
82
+ t.add_row("GGUF quants", f"{len(model.gguf_files)} file(s) in repo")
83
+ if model.gated:
84
+ t.add_row("Access", Text("gated — license acceptance required", "yellow"))
85
+ return Panel(t, title=f"[bold]{model.model_id}", border_style="magenta")
86
+
87
+
88
+ def results_table(results: List[Assessment], ctx_tokens: int) -> Table:
89
+ table = Table(title=f"Fit assessment @ {ctx_tokens:,}-token context",
90
+ title_style="bold", header_style="bold")
91
+ table.add_column("Precision", overflow="fold")
92
+ table.add_column("Weights", justify="right")
93
+ table.add_column("KV cache", justify="right")
94
+ table.add_column("Required*", justify="right")
95
+ table.add_column("Verdict")
96
+ table.add_column("Note", style="dim", overflow="fold")
97
+
98
+ for r in results:
99
+ label, style = VERDICT_STYLE[r.verdict]
100
+ table.add_row(
101
+ r.label,
102
+ _gb(r.weights_gb),
103
+ _gb(r.kv_gb),
104
+ _gb(r.required_gb),
105
+ Text(label, style=style),
106
+ r.note,
107
+ )
108
+ return table
109
+
110
+
111
+ def full_report(model: HFModel, specs: SystemSpecs,
112
+ results: List[Assessment], ctx_tokens: int) -> RenderableType:
113
+ parts: List[RenderableType] = [model_panel(model)]
114
+
115
+ if not results:
116
+ parts.append(Text(
117
+ "No parameter count or weight files found for this repo — "
118
+ "cannot estimate memory (it may be a dataset-style or adapter repo).",
119
+ style="yellow",
120
+ ))
121
+ return Group(*parts)
122
+
123
+ parts.append(results_table(results, ctx_tokens))
124
+
125
+ if model.max_ctx and ctx_tokens > model.max_ctx:
126
+ parts.append(Text(
127
+ f"Note: the chosen context ({ctx_tokens:,}) exceeds this model's "
128
+ f"maximum ({model.max_ctx:,} tokens).", style="yellow",
129
+ ))
130
+
131
+ verdict = best_verdict(results)
132
+ if verdict is not None:
133
+ line, style = SUMMARY_LINE[verdict]
134
+ parts.append(Text(f"\n{line}", style=f"bold {style}"))
135
+
136
+ footnotes = ("*Required = weights + KV cache + runtime overhead "
137
+ f"(~{0.8:.1f} GB + 5% activations).")
138
+ if any(r.kv_gb is None for r in results):
139
+ footnotes += " KV cache unknown for this architecture — a 20% margin was used."
140
+ footnotes += (" INT8/INT4 rows are theoretical: they need an actual quantized "
141
+ "checkpoint or on-the-fly quantization.")
142
+ parts.append(Text(footnotes, style="dim"))
143
+ return Group(*parts)
hfit-0.1.0/hfit/tui.py ADDED
@@ -0,0 +1,111 @@
1
+ """Textual TUI: type a model id, get a fit report against this machine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional
6
+
7
+ from rich.text import Text
8
+ from textual import work
9
+ from textual.app import App, ComposeResult
10
+ from textual.containers import Horizontal, VerticalScroll
11
+ from textual.widgets import Footer, Header, Input, Select, Static
12
+
13
+ from . import __version__
14
+ from .estimate import assess
15
+ from .hw import detect
16
+ from .model import HFModel, ModelLookupError, fetch_model
17
+ from .report import full_report, system_panel
18
+
19
+ CTX_CHOICES = [2048, 4096, 8192, 16384, 32768, 65536, 131072]
20
+
21
+
22
+ def _ctx_label(tokens: int) -> str:
23
+ return f"{tokens // 1024}k context"
24
+
25
+
26
+ class SpecCheckerApp(App):
27
+ TITLE = "hfit"
28
+ SUB_TITLE = f"v{__version__} — will it run on this machine?"
29
+
30
+ CSS = """
31
+ #topbar { height: 3; dock: top; }
32
+ #model-input { width: 1fr; }
33
+ #ctx-select { width: 24; }
34
+ #results { padding: 0 1; }
35
+ #system-panel { margin-bottom: 1; }
36
+ """
37
+
38
+ BINDINGS = [
39
+ ("q", "quit", "Quit"),
40
+ ("ctrl+l", "focus_input", "New model"),
41
+ ]
42
+
43
+ def __init__(self, ctx_tokens: int = 4096, token: Optional[str] = None):
44
+ super().__init__()
45
+ self.ctx_tokens = ctx_tokens if ctx_tokens in CTX_CHOICES else 4096
46
+ self.hf_token = token
47
+ self.specs = detect()
48
+ self.current_model: Optional[HFModel] = None
49
+
50
+ def compose(self) -> ComposeResult:
51
+ yield Header()
52
+ with Horizontal(id="topbar"):
53
+ yield Input(
54
+ placeholder="org/model-name (e.g. Qwen/Qwen2.5-7B-Instruct) — Enter to check",
55
+ id="model-input",
56
+ )
57
+ yield Select(
58
+ [(_ctx_label(c), c) for c in CTX_CHOICES],
59
+ value=self.ctx_tokens, allow_blank=False, id="ctx-select",
60
+ )
61
+ with VerticalScroll(id="results"):
62
+ yield Static(system_panel(self.specs), id="system-panel")
63
+ yield Static(
64
+ Text("Enter a Hugging Face model id above to start.", style="dim"),
65
+ id="report",
66
+ )
67
+ yield Footer()
68
+
69
+ def on_mount(self) -> None:
70
+ self.query_one("#model-input", Input).focus()
71
+
72
+ def action_focus_input(self) -> None:
73
+ inp = self.query_one("#model-input", Input)
74
+ inp.focus()
75
+ inp.selection = (0, len(inp.value))
76
+
77
+ def on_input_submitted(self, event: Input.Submitted) -> None:
78
+ model_id = event.value.strip().strip("/")
79
+ if model_id:
80
+ self.check_model(model_id)
81
+
82
+ def on_select_changed(self, event: Select.Changed) -> None:
83
+ if event.select.id != "ctx-select" or event.value is None:
84
+ return
85
+ self.ctx_tokens = int(event.value)
86
+ if self.current_model is not None:
87
+ self._render(self.current_model)
88
+
89
+ @work(thread=True, exclusive=True)
90
+ def check_model(self, model_id: str) -> None:
91
+ report = self.query_one("#report", Static)
92
+ self.call_from_thread(
93
+ report.update, Text(f"Fetching metadata for {model_id} …", style="italic yellow")
94
+ )
95
+ try:
96
+ model = fetch_model(model_id, token=self.hf_token)
97
+ except ModelLookupError as e:
98
+ self.call_from_thread(report.update, Text(str(e), style="bold red"))
99
+ return
100
+ self.current_model = model
101
+ self.call_from_thread(self._render, model)
102
+
103
+ def _render(self, model: HFModel) -> None:
104
+ results = assess(model, self.specs, self.ctx_tokens)
105
+ self.query_one("#report", Static).update(
106
+ full_report(model, self.specs, results, self.ctx_tokens)
107
+ )
108
+
109
+
110
+ def run_tui(ctx_tokens: int = 4096, token: Optional[str] = None) -> None:
111
+ SpecCheckerApp(ctx_tokens=ctx_tokens, token=token).run()
@@ -0,0 +1,113 @@
1
+ Metadata-Version: 2.4
2
+ Name: hfit
3
+ Version: 0.1.0
4
+ Summary: Check whether a Hugging Face model fits your local hardware — without downloading the weights.
5
+ Author-email: Samir Nuri <samirnuri714@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/snuri00/hfit
8
+ Project-URL: Issues, https://github.com/snuri00/hfit/issues
9
+ Keywords: huggingface,llm,vram,gpu,memory,tui,hardware,inference
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Classifier: Topic :: System :: Hardware
18
+ Requires-Python: >=3.9
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: huggingface_hub>=0.23
22
+ Requires-Dist: rich>=13.0
23
+ Requires-Dist: textual>=0.60
24
+ Provides-Extra: full
25
+ Requires-Dist: psutil>=5.9; extra == "full"
26
+ Dynamic: license-file
27
+
28
+ # hfit
29
+
30
+ *Hugging Face + fit.*
31
+
32
+ **Will this Hugging Face model run on my machine?** Find out in seconds without downloading a single weight file.
33
+
34
+ Give it a model id like `Qwen/Qwen2.5-7B-Instruct`. It reads the exact
35
+ parameter count and architecture from the Hub's metadata (a few KB of API
36
+ calls), detects your local hardware (GPU VRAM, RAM), and tells you which
37
+ precisions fit:
38
+
39
+ ```
40
+ ┏━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━┳━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━┓
41
+ ┃ Precision ┃ Weights ┃ KV cache ┃ Required* ┃ Verdict ┃
42
+ ┡━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━╇━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━┩
43
+ │ FP16 / BF16 (native)│ 14.19 GB │ 0.22 GB │ 15.91 GB │ ◐ CPU/RAM only (slow) │
44
+ │ INT8 (quantized) │ 7.09 GB │ 0.22 GB │ 8.47 GB │ ◐ CPU/RAM only (slow) │
45
+ │ INT4 (quantized) │ 3.55 GB │ 0.22 GB │ 4.74 GB │ ● tight on GPU │
46
+ └─────────────────────┴──────────┴──────────┴───────────┴───────────────────────┘
47
+ ```
48
+
49
+ Unlike `accelerate estimate-memory`, this compares the model against **your
50
+ actual hardware** and includes a **KV-cache estimate** at your chosen context
51
+ length computed exactly from the model's architecture (layers × KV heads ×
52
+ head dim), not a flat percentage.
53
+
54
+ ## Install
55
+
56
+ ```bash
57
+ pip install hfit
58
+ # or from a clone of this repo:
59
+ pip install .
60
+
61
+ ```
62
+
63
+ Dependencies: `huggingface_hub`, `rich`, `textual`. No torch, no CUDA needed.
64
+
65
+ ## Usage
66
+
67
+ ```bash
68
+ hfit # interactive TUI
69
+ hfit Qwen/Qwen2.5-7B-Instruct # one-shot report
70
+ hfit meta-llama/Llama-3.1-8B-Instruct --ctx 32768
71
+ hfit some/gated-model --token hf_xxx
72
+ ```
73
+
74
+ In the TUI: type a model id, press Enter; switch the context length from the
75
+ dropdown to see the KV cache impact instantly (no refetch). `q` quits.
76
+
77
+ Exit codes for scripting: `0` = fits somewhere (GPU or RAM), `1` = lookup
78
+ error, `3` = does not fit at all.
79
+
80
+ ## What it understands
81
+
82
+ - **Safetensors repos** — exact parameter count from Hub metadata.
83
+ - **GGUF repos** (e.g. `bartowski/...-GGUF`) every quant file assessed
84
+ individually, for llama.cpp / Ollama / LM Studio users.
85
+ - **Multi-component repos** (speech/vision pipelines with several weight
86
+ folders) reported as a stored total instead of a fake parameter count.
87
+ - Duplicate serializations (safetensors + bin + h5, sharded + consolidated)
88
+ are deduplicated, not double-counted.
89
+
90
+ ## Platform support
91
+
92
+ | Platform | GPU detection | RAM detection |
93
+ |---|---|---|
94
+ | Linux | `nvidia-smi`; AMD via sysfs (no ROCm needed), then `rocm-smi` | `/proc/meminfo` |
95
+ | Windows | `nvidia-smi` (PATH or System32) | WinAPI via ctypes |
96
+ | macOS (Apple Silicon) | unified memory (RAM = VRAM budget) | `sysctl` |
97
+
98
+ `psutil` is used when available (`pip install .[full]`) but is not required.
99
+
100
+ ## How the estimate works
101
+
102
+ ```
103
+ required = weights + KV cache(context) + overhead(0.8 GB + 5% of weights)
104
+ ```
105
+
106
+ - Weights: parameters × bytes-per-parameter (4 / 2 / 1 / 0.5 for
107
+ FP32 / FP16 / INT8 / INT4).
108
+ - KV cache: `2 × layers × kv_heads × head_dim × 2 bytes × context_tokens`
109
+ from `config.json`; a 20% margin is used when the architecture is unknown.
110
+ - Verdicts leave ~7% VRAM and ~15% RAM headroom for the driver and OS.
111
+
112
+ These are pre-flight estimates. For a definitive runtime answer, vLLM's
113
+ `--dry-run` on the actual machine remains the ground truth.
@@ -0,0 +1,17 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ hfit/__init__.py
5
+ hfit/__main__.py
6
+ hfit/cli.py
7
+ hfit/estimate.py
8
+ hfit/hw.py
9
+ hfit/model.py
10
+ hfit/report.py
11
+ hfit/tui.py
12
+ hfit.egg-info/PKG-INFO
13
+ hfit.egg-info/SOURCES.txt
14
+ hfit.egg-info/dependency_links.txt
15
+ hfit.egg-info/entry_points.txt
16
+ hfit.egg-info/requires.txt
17
+ hfit.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ hfit = hfit.cli:main
@@ -0,0 +1,6 @@
1
+ huggingface_hub>=0.23
2
+ rich>=13.0
3
+ textual>=0.60
4
+
5
+ [full]
6
+ psutil>=5.9
@@ -0,0 +1 @@
1
+ hfit
@@ -0,0 +1,41 @@
1
+ [project]
2
+ name = "hfit"
3
+ version = "0.1.0"
4
+ description = "Check whether a Hugging Face model fits your local hardware — without downloading the weights."
5
+ readme = "README.md"
6
+ requires-python = ">=3.9"
7
+ license = { text = "MIT" }
8
+ authors = [{ name = "Samir Nuri", email = "samirnuri714@gmail.com" }]
9
+ keywords = ["huggingface", "llm", "vram", "gpu", "memory", "tui", "hardware", "inference"]
10
+ classifiers = [
11
+ "Development Status :: 4 - Beta",
12
+ "Environment :: Console",
13
+ "Intended Audience :: Developers",
14
+ "License :: OSI Approved :: MIT License",
15
+ "Operating System :: OS Independent",
16
+ "Programming Language :: Python :: 3",
17
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
18
+ "Topic :: System :: Hardware",
19
+ ]
20
+ dependencies = [
21
+ "huggingface_hub>=0.23",
22
+ "rich>=13.0",
23
+ "textual>=0.60",
24
+ ]
25
+
26
+ [project.optional-dependencies]
27
+ full = ["psutil>=5.9"]
28
+
29
+ [project.urls]
30
+ Homepage = "https://github.com/snuri00/hfit"
31
+ Issues = "https://github.com/snuri00/hfit/issues"
32
+
33
+ [project.scripts]
34
+ hfit = "hfit.cli:main"
35
+
36
+ [build-system]
37
+ requires = ["setuptools>=61"]
38
+ build-backend = "setuptools.build_meta"
39
+
40
+ [tool.setuptools.packages.find]
41
+ include = ["hfit*"]
hfit-0.1.0/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+