canirunllm 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- canirunllm-0.1.0/LICENSE +21 -0
- canirunllm-0.1.0/PKG-INFO +152 -0
- canirunllm-0.1.0/README.md +121 -0
- canirunllm-0.1.0/pyproject.toml +51 -0
- canirunllm-0.1.0/setup.cfg +4 -0
- canirunllm-0.1.0/src/canirunllm/__init__.py +1 -0
- canirunllm-0.1.0/src/canirunllm/api/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/api/presentation.py +192 -0
- canirunllm-0.1.0/src/canirunllm/api/schemas.py +116 -0
- canirunllm-0.1.0/src/canirunllm/api/server.py +293 -0
- canirunllm-0.1.0/src/canirunllm/application/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/application/scanner_service.py +88 -0
- canirunllm-0.1.0/src/canirunllm/cli.py +299 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/confidence.py +39 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/config.py +10 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/decision.py +45 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/engine.py +164 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/kv_cache.py +29 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/memory.py +58 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/memory_planner.py +91 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/requirements.py +59 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/runtime.py +90 -0
- canirunllm-0.1.0/src/canirunllm/compatibility/verdict.py +8 -0
- canirunllm-0.1.0/src/canirunllm/hardware/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/hardware/cpu.py +13 -0
- canirunllm-0.1.0/src/canirunllm/hardware/gpu.py +23 -0
- canirunllm-0.1.0/src/canirunllm/hardware/memory.py +12 -0
- canirunllm-0.1.0/src/canirunllm/hardware/os.py +10 -0
- canirunllm-0.1.0/src/canirunllm/hardware/scanner.py +38 -0
- canirunllm-0.1.0/src/canirunllm/models/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/models/hardware.py +43 -0
- canirunllm-0.1.0/src/canirunllm/models/model.py +34 -0
- canirunllm-0.1.0/src/canirunllm/models/resolver.py +53 -0
- canirunllm-0.1.0/src/canirunllm/models/results.py +13 -0
- canirunllm-0.1.0/src/canirunllm/performance/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/performance/prediction.py +112 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/engine.py +312 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/profiles.py +88 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/results.py +40 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/scoring.py +92 -0
- canirunllm-0.1.0/src/canirunllm/recommendation/tier.py +26 -0
- canirunllm-0.1.0/src/canirunllm/registry/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/registry/models.json +814 -0
- canirunllm-0.1.0/src/canirunllm/registry/models.py +18 -0
- canirunllm-0.1.0/src/canirunllm/scanner.py +30 -0
- canirunllm-0.1.0/src/canirunllm/web/__init__.py +0 -0
- canirunllm-0.1.0/src/canirunllm/web/launcher.py +49 -0
- canirunllm-0.1.0/src/canirunllm/web/static/app.js +547 -0
- canirunllm-0.1.0/src/canirunllm/web/static/style.css +644 -0
- canirunllm-0.1.0/src/canirunllm/web/templates/index.html +113 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/PKG-INFO +152 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/SOURCES.txt +77 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/dependency_links.txt +1 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/entry_points.txt +2 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/requires.txt +5 -0
- canirunllm-0.1.0/src/canirunllm.egg-info/top_level.txt +1 -0
- canirunllm-0.1.0/tests/test_api.py +210 -0
- canirunllm-0.1.0/tests/test_compatibility.py +178 -0
- canirunllm-0.1.0/tests/test_confidence.py +73 -0
- canirunllm-0.1.0/tests/test_decision.py +70 -0
- canirunllm-0.1.0/tests/test_hardware.py +13 -0
- canirunllm-0.1.0/tests/test_kv_cache.py +33 -0
- canirunllm-0.1.0/tests/test_memory.py +22 -0
- canirunllm-0.1.0/tests/test_memory_planner.py +185 -0
- canirunllm-0.1.0/tests/test_memory_strategy.py +70 -0
- canirunllm-0.1.0/tests/test_model.py +25 -0
- canirunllm-0.1.0/tests/test_performance_prediction.py +129 -0
- canirunllm-0.1.0/tests/test_presentation.py +219 -0
- canirunllm-0.1.0/tests/test_recommendation.py +233 -0
- canirunllm-0.1.0/tests/test_recommendation_engine.py +281 -0
- canirunllm-0.1.0/tests/test_registry.py +16 -0
- canirunllm-0.1.0/tests/test_requirements.py +40 -0
- canirunllm-0.1.0/tests/test_resolver.py +85 -0
- canirunllm-0.1.0/tests/test_results.py +85 -0
- canirunllm-0.1.0/tests/test_runtime_capabilities.py +54 -0
- canirunllm-0.1.0/tests/test_scanner_service.py +90 -0
- canirunllm-0.1.0/tests/test_web_launcher.py +76 -0
canirunllm-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 jashwanthsai678
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: canirunllm
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Find out which open-source LLMs can run on your hardware.
|
|
5
|
+
Author: jashwanthsai678
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/jashwanthsai678/CaniRunLLM
|
|
8
|
+
Project-URL: Repository, https://github.com/jashwanthsai678/CaniRunLLM
|
|
9
|
+
Project-URL: Issues, https://github.com/jashwanthsai678/CaniRunLLM/issues
|
|
10
|
+
Keywords: llm,local-llm,hardware,gguf,llama.cpp,quantization
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Environment :: Web Environment
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
21
|
+
Classifier: Topic :: System :: Hardware
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: fastapi>=0.110
|
|
26
|
+
Requires-Dist: uvicorn>=0.29
|
|
27
|
+
Requires-Dist: pydantic>=2.0
|
|
28
|
+
Requires-Dist: psutil>=5.9
|
|
29
|
+
Requires-Dist: GPUtil>=1.4
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# CanIRunLLM
|
|
33
|
+
|
|
34
|
+
Find out which open-source LLMs your machine can actually run — and which
|
|
35
|
+
one you should pick — with an automatic hardware scan and a local dashboard.
|
|
36
|
+
|
|
37
|
+
CanIRunLLM inspects your real CPU, RAM, and GPU/VRAM, checks that against a
|
|
38
|
+
registry of real open-weight models (Qwen, Llama, Mistral, Gemma, Phi,
|
|
39
|
+
DeepSeek, and more), and gives you a plain-language verdict — not just a raw
|
|
40
|
+
"fits/doesn't fit" table. It runs entirely on your machine: no account, no
|
|
41
|
+
cloud calls, no telemetry.
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
$ canirunllm scan
|
|
45
|
+
|
|
46
|
+
CanIRunLLM
|
|
47
|
+
|
|
48
|
+
Checking your computer...
|
|
49
|
+
CPU detected
|
|
50
|
+
RAM detected
|
|
51
|
+
GPU detected
|
|
52
|
+
VRAM detected
|
|
53
|
+
|
|
54
|
+
Analyzing local AI models...
|
|
55
|
+
58 models analyzed
|
|
56
|
+
|
|
57
|
+
You can run 14 model(s) comfortably.
|
|
58
|
+
7 more can run with CPU/RAM offload (slower).
|
|
59
|
+
|
|
60
|
+
Dashboard:
|
|
61
|
+
http://127.0.0.1:8765
|
|
62
|
+
|
|
63
|
+
Opening browser...
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The browser dashboard shows a friendly "what can I run / what should I run"
|
|
67
|
+
report, with the full technical breakdown (VRAM, RAM, KV cache, quantization,
|
|
68
|
+
runtime, confidence) available behind a "Technical details" toggle for anyone
|
|
69
|
+
who wants it.
|
|
70
|
+
|
|
71
|
+
## Install
|
|
72
|
+
|
|
73
|
+
Requires Python 3.10+.
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
git clone https://github.com/jashwanthsai678/CaniRunLLM.git
|
|
77
|
+
cd CaniRunLLM
|
|
78
|
+
pip install -e .
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Quick start
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
canirunllm scan # scan hardware, evaluate models, open the dashboard
|
|
85
|
+
canirunllm scan --no-browser # same, but skip the dashboard (good for CI/headless)
|
|
86
|
+
canirunllm scan --technical # also print the full technical breakdown in the terminal
|
|
87
|
+
|
|
88
|
+
canirunllm search qwen # search the model registry
|
|
89
|
+
canirunllm check Qwen3-8B # check one model or a whole family
|
|
90
|
+
canirunllm recommend # ranked list of models for your hardware
|
|
91
|
+
|
|
92
|
+
canirunllm web # launch the dashboard on its own
|
|
93
|
+
canirunllm web --port 9000 # on a custom port
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## What it actually checks
|
|
97
|
+
|
|
98
|
+
For every model + quantization pair, CanIRunLLM estimates:
|
|
99
|
+
|
|
100
|
+
- **Weight memory** from parameter count and quantization (Q4_K_M, Q8_0, etc.)
|
|
101
|
+
- **KV cache size**, which grows with context length — a model isn't just
|
|
102
|
+
"fits" or "doesn't," it fits *at a given context length*
|
|
103
|
+
- **Runtime overhead and a safety margin**, not just the raw weight size
|
|
104
|
+
- **Memory strategy**: does it fit on a single GPU, does it need to be split
|
|
105
|
+
across multiple GPUs, or does it need CPU/RAM offloading — each of these
|
|
106
|
+
is a real, different scenario, not a single generic "offload" verdict
|
|
107
|
+
- **Runtime compatibility**: is the declared runtime (llama.cpp, etc.)
|
|
108
|
+
actually known to support that memory strategy
|
|
109
|
+
|
|
110
|
+
The result is always a verdict *plus* a confidence level and a plain-English
|
|
111
|
+
reason — never a bare "cannot run" with no explanation.
|
|
112
|
+
|
|
113
|
+
## Architecture
|
|
114
|
+
|
|
115
|
+
```
|
|
116
|
+
CLI / Web Dashboard (presentation only)
|
|
117
|
+
│
|
|
118
|
+
Application Layer (ScannerService, RecommendationEngine)
|
|
119
|
+
│
|
|
120
|
+
┌────┴─────┬──────────────┬─────────────┐
|
|
121
|
+
▼ ▼ ▼ ▼
|
|
122
|
+
Hardware Model Registry Compatibility Performance
|
|
123
|
+
Scanner + Resolver Engine Prediction
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
The CLI and the local web dashboard are two presentation layers over the
|
|
127
|
+
same Python core — nothing about compatibility is calculated twice, and the
|
|
128
|
+
frontend never re-derives a verdict on its own.
|
|
129
|
+
|
|
130
|
+
## Current limitations (being upfront about them)
|
|
131
|
+
|
|
132
|
+
- GPU detection currently only recognizes NVIDIA GPUs (via `GPUtil`/
|
|
133
|
+
`nvidia-smi`). On AMD/Intel/Apple Silicon machines it safely falls back to
|
|
134
|
+
CPU-only mode rather than crashing, but it won't report real GPU numbers yet.
|
|
135
|
+
- The model registry is a curated set of well-known open-weight models, not
|
|
136
|
+
an exhaustive mirror of every model on Hugging Face — see
|
|
137
|
+
`src/canirunllm/registry/SOURCES.md` for exactly where every number in it
|
|
138
|
+
came from.
|
|
139
|
+
- Performance numbers are a coarse, clearly-labeled *estimate* based on
|
|
140
|
+
parameter count and memory strategy — there is no real benchmarking yet,
|
|
141
|
+
and the tool never presents an estimate as a measurement.
|
|
142
|
+
|
|
143
|
+
## Testing
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
pip install -e .
|
|
147
|
+
pytest -v
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
## License
|
|
151
|
+
|
|
152
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# CanIRunLLM
|
|
2
|
+
|
|
3
|
+
Find out which open-source LLMs your machine can actually run — and which
|
|
4
|
+
one you should pick — with an automatic hardware scan and a local dashboard.
|
|
5
|
+
|
|
6
|
+
CanIRunLLM inspects your real CPU, RAM, and GPU/VRAM, checks that against a
|
|
7
|
+
registry of real open-weight models (Qwen, Llama, Mistral, Gemma, Phi,
|
|
8
|
+
DeepSeek, and more), and gives you a plain-language verdict — not just a raw
|
|
9
|
+
"fits/doesn't fit" table. It runs entirely on your machine: no account, no
|
|
10
|
+
cloud calls, no telemetry.
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
$ canirunllm scan
|
|
14
|
+
|
|
15
|
+
CanIRunLLM
|
|
16
|
+
|
|
17
|
+
Checking your computer...
|
|
18
|
+
CPU detected
|
|
19
|
+
RAM detected
|
|
20
|
+
GPU detected
|
|
21
|
+
VRAM detected
|
|
22
|
+
|
|
23
|
+
Analyzing local AI models...
|
|
24
|
+
58 models analyzed
|
|
25
|
+
|
|
26
|
+
You can run 14 model(s) comfortably.
|
|
27
|
+
7 more can run with CPU/RAM offload (slower).
|
|
28
|
+
|
|
29
|
+
Dashboard:
|
|
30
|
+
http://127.0.0.1:8765
|
|
31
|
+
|
|
32
|
+
Opening browser...
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
The browser dashboard shows a friendly "what can I run / what should I run"
|
|
36
|
+
report, with the full technical breakdown (VRAM, RAM, KV cache, quantization,
|
|
37
|
+
runtime, confidence) available behind a "Technical details" toggle for anyone
|
|
38
|
+
who wants it.
|
|
39
|
+
|
|
40
|
+
## Install
|
|
41
|
+
|
|
42
|
+
Requires Python 3.10+.
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
git clone https://github.com/jashwanthsai678/CaniRunLLM.git
|
|
46
|
+
cd CaniRunLLM
|
|
47
|
+
pip install -e .
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Quick start
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
canirunllm scan # scan hardware, evaluate models, open the dashboard
|
|
54
|
+
canirunllm scan --no-browser # same, but skip the dashboard (good for CI/headless)
|
|
55
|
+
canirunllm scan --technical # also print the full technical breakdown in the terminal
|
|
56
|
+
|
|
57
|
+
canirunllm search qwen # search the model registry
|
|
58
|
+
canirunllm check Qwen3-8B # check one model or a whole family
|
|
59
|
+
canirunllm recommend # ranked list of models for your hardware
|
|
60
|
+
|
|
61
|
+
canirunllm web # launch the dashboard on its own
|
|
62
|
+
canirunllm web --port 9000 # on a custom port
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## What it actually checks
|
|
66
|
+
|
|
67
|
+
For every model + quantization pair, CanIRunLLM estimates:
|
|
68
|
+
|
|
69
|
+
- **Weight memory** from parameter count and quantization (Q4_K_M, Q8_0, etc.)
|
|
70
|
+
- **KV cache size**, which grows with context length — a model isn't just
|
|
71
|
+
"fits" or "doesn't," it fits *at a given context length*
|
|
72
|
+
- **Runtime overhead and a safety margin**, not just the raw weight size
|
|
73
|
+
- **Memory strategy**: does it fit on a single GPU, does it need to be split
|
|
74
|
+
across multiple GPUs, or does it need CPU/RAM offloading — each of these
|
|
75
|
+
is a real, different scenario, not a single generic "offload" verdict
|
|
76
|
+
- **Runtime compatibility**: is the declared runtime (llama.cpp, etc.)
|
|
77
|
+
actually known to support that memory strategy
|
|
78
|
+
|
|
79
|
+
The result is always a verdict *plus* a confidence level and a plain-English
|
|
80
|
+
reason — never a bare "cannot run" with no explanation.
|
|
81
|
+
|
|
82
|
+
## Architecture
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
CLI / Web Dashboard (presentation only)
|
|
86
|
+
│
|
|
87
|
+
Application Layer (ScannerService, RecommendationEngine)
|
|
88
|
+
│
|
|
89
|
+
┌────┴─────┬──────────────┬─────────────┐
|
|
90
|
+
▼ ▼ ▼ ▼
|
|
91
|
+
Hardware Model Registry Compatibility Performance
|
|
92
|
+
Scanner + Resolver Engine Prediction
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The CLI and the local web dashboard are two presentation layers over the
|
|
96
|
+
same Python core — nothing about compatibility is calculated twice, and the
|
|
97
|
+
frontend never re-derives a verdict on its own.
|
|
98
|
+
|
|
99
|
+
## Current limitations (being upfront about them)
|
|
100
|
+
|
|
101
|
+
- GPU detection currently only recognizes NVIDIA GPUs (via `GPUtil`/
|
|
102
|
+
`nvidia-smi`). On AMD/Intel/Apple Silicon machines it safely falls back to
|
|
103
|
+
CPU-only mode rather than crashing, but it won't report real GPU numbers yet.
|
|
104
|
+
- The model registry is a curated set of well-known open-weight models, not
|
|
105
|
+
an exhaustive mirror of every model on Hugging Face — see
|
|
106
|
+
`src/canirunllm/registry/SOURCES.md` for exactly where every number in it
|
|
107
|
+
came from.
|
|
108
|
+
- Performance numbers are a coarse, clearly-labeled *estimate* based on
|
|
109
|
+
parameter count and memory strategy — there is no real benchmarking yet,
|
|
110
|
+
and the tool never presents an estimate as a measurement.
|
|
111
|
+
|
|
112
|
+
## Testing
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
pip install -e .
|
|
116
|
+
pytest -v
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## License
|
|
120
|
+
|
|
121
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "canirunllm"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Find out which open-source LLMs can run on your hardware."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "jashwanthsai678" },
|
|
15
|
+
]
|
|
16
|
+
keywords = ["llm", "local-llm", "hardware", "gguf", "llama.cpp", "quantization"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 3 - Alpha",
|
|
19
|
+
"Environment :: Console",
|
|
20
|
+
"Environment :: Web Environment",
|
|
21
|
+
"Intended Audience :: Developers",
|
|
22
|
+
"Operating System :: OS Independent",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Programming Language :: Python :: 3.10",
|
|
25
|
+
"Programming Language :: Python :: 3.11",
|
|
26
|
+
"Programming Language :: Python :: 3.12",
|
|
27
|
+
"Topic :: Software Development :: Libraries",
|
|
28
|
+
"Topic :: System :: Hardware",
|
|
29
|
+
]
|
|
30
|
+
dependencies = [
|
|
31
|
+
"fastapi>=0.110",
|
|
32
|
+
"uvicorn>=0.29",
|
|
33
|
+
"pydantic>=2.0",
|
|
34
|
+
"psutil>=5.9",
|
|
35
|
+
"GPUtil>=1.4",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Homepage = "https://github.com/jashwanthsai678/CaniRunLLM"
|
|
40
|
+
Repository = "https://github.com/jashwanthsai678/CaniRunLLM"
|
|
41
|
+
Issues = "https://github.com/jashwanthsai678/CaniRunLLM/issues"
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
canirunllm = "canirunllm.cli:main"
|
|
45
|
+
|
|
46
|
+
[tool.setuptools.packages.find]
|
|
47
|
+
where = ["src"]
|
|
48
|
+
|
|
49
|
+
[tool.setuptools.package-data]
|
|
50
|
+
"canirunllm.registry" = ["*.json"]
|
|
51
|
+
"canirunllm.web" = ["static/*", "templates/*"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
File without changes
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Translates raw core objects into human-readable product language.
|
|
2
|
+
|
|
3
|
+
This is deliberately kept server-side (Python), not JavaScript: the
|
|
4
|
+
frontend must never re-derive compatibility meaning on its own. It
|
|
5
|
+
only renders the strings and flags this module produces.
|
|
6
|
+
|
|
7
|
+
Nothing here recalculates compatibility — it only interprets fields
|
|
8
|
+
that CompatibilityResult/RankedModel already computed.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
from canirunllm.compatibility.engine import CompatibilityResult
|
|
14
|
+
from canirunllm.compatibility.decision import OverallVerdict
|
|
15
|
+
from canirunllm.compatibility.runtime import RuntimeVerdict
|
|
16
|
+
from canirunllm.models.model import ModelSpec
|
|
17
|
+
from canirunllm.recommendation.engine import RankedModel
|
|
18
|
+
from canirunllm.recommendation.tier import RecommendationTier
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
GB = 1024 ** 3
|
|
22
|
+
|
|
23
|
+
FAVORABLE_TIERS = {RecommendationTier.BEST_MATCH, RecommendationTier.GOOD}
|
|
24
|
+
|
|
25
|
+
# (friendly label, icon) — icon is a UI-neutral token, not an emoji,
|
|
26
|
+
# so the frontend controls the actual glyph/color.
|
|
27
|
+
VERDICT_LABELS: dict[OverallVerdict, tuple[str, str]] = {
|
|
28
|
+
OverallVerdict.CAN_RUN: ("Runs comfortably", "check"),
|
|
29
|
+
OverallVerdict.CAN_RUN_WITH_OFFLOAD: ("Runs, but slower", "warn"),
|
|
30
|
+
OverallVerdict.NEEDS_VALIDATION: ("Needs more info to confirm", "unknown"),
|
|
31
|
+
OverallVerdict.CANNOT_RUN: ("Not enough resources", "cross"),
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class ReasonItem:
|
|
37
|
+
ok: bool
|
|
38
|
+
text: str
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class RunCommandInfo:
|
|
43
|
+
runtime: str
|
|
44
|
+
command: str
|
|
45
|
+
note: str
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def friendly_verdict(overall_verdict: OverallVerdict) -> tuple[str, str]:
|
|
49
|
+
return VERDICT_LABELS[overall_verdict]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_reasons(
|
|
53
|
+
model: ModelSpec,
|
|
54
|
+
compatibility: CompatibilityResult,
|
|
55
|
+
) -> list[ReasonItem]:
|
|
56
|
+
"""Human-readable checklist explaining a verdict — built only from
|
|
57
|
+
fields the compatibility engine already computed, never invented."""
|
|
58
|
+
|
|
59
|
+
reasons: list[ReasonItem] = []
|
|
60
|
+
|
|
61
|
+
required_gb = compatibility.required_memory_bytes / GB
|
|
62
|
+
vram_gb = compatibility.available_vram_bytes / GB
|
|
63
|
+
ram_gb = compatibility.available_ram_bytes / GB
|
|
64
|
+
|
|
65
|
+
strategy = compatibility.memory_strategy
|
|
66
|
+
|
|
67
|
+
if strategy == "SINGLE_GPU":
|
|
68
|
+
reasons.append(ReasonItem(
|
|
69
|
+
True,
|
|
70
|
+
f"Model weights fit within your available GPU memory "
|
|
71
|
+
f"(~{vram_gb:.1f} GB free).",
|
|
72
|
+
))
|
|
73
|
+
|
|
74
|
+
elif strategy == "MULTI_GPU":
|
|
75
|
+
reasons.append(ReasonItem(
|
|
76
|
+
True,
|
|
77
|
+
"Model fits by splitting across your multiple GPUs.",
|
|
78
|
+
))
|
|
79
|
+
reasons.append(ReasonItem(
|
|
80
|
+
False,
|
|
81
|
+
"Splitting a model across GPUs depends on the runtime "
|
|
82
|
+
"supporting it correctly, and can be slower than a single "
|
|
83
|
+
"GPU that fits the whole model.",
|
|
84
|
+
))
|
|
85
|
+
|
|
86
|
+
elif strategy == "CPU_OFFLOAD":
|
|
87
|
+
reasons.append(ReasonItem(
|
|
88
|
+
True,
|
|
89
|
+
f"Model doesn't fully fit in GPU memory, but fits when "
|
|
90
|
+
f"combined with system RAM (~{ram_gb:.1f} GB available).",
|
|
91
|
+
))
|
|
92
|
+
reasons.append(ReasonItem(
|
|
93
|
+
False,
|
|
94
|
+
"Running part of the model on CPU/RAM is typically much "
|
|
95
|
+
"slower than running fully on GPU.",
|
|
96
|
+
))
|
|
97
|
+
|
|
98
|
+
else:
|
|
99
|
+
reasons.append(ReasonItem(
|
|
100
|
+
False,
|
|
101
|
+
f"This model needs about {required_gb:.1f} GB, but your "
|
|
102
|
+
f"machine currently has ~{vram_gb:.1f} GB free GPU memory "
|
|
103
|
+
f"and ~{ram_gb:.1f} GB free RAM.",
|
|
104
|
+
))
|
|
105
|
+
|
|
106
|
+
if compatibility.runtime_verdict == RuntimeVerdict.SUPPORTED:
|
|
107
|
+
reasons.append(ReasonItem(
|
|
108
|
+
True,
|
|
109
|
+
f"Runtime '{model.runtime}' is supported.",
|
|
110
|
+
))
|
|
111
|
+
elif compatibility.runtime_verdict == RuntimeVerdict.UNKNOWN:
|
|
112
|
+
reasons.append(ReasonItem(
|
|
113
|
+
False,
|
|
114
|
+
"This model doesn't specify a runtime, so compatibility "
|
|
115
|
+
"can't be fully confirmed.",
|
|
116
|
+
))
|
|
117
|
+
else:
|
|
118
|
+
reasons.append(ReasonItem(
|
|
119
|
+
False,
|
|
120
|
+
f"Runtime '{model.runtime}' is not one of the runtimes "
|
|
121
|
+
"this tool currently recognizes.",
|
|
122
|
+
))
|
|
123
|
+
|
|
124
|
+
return reasons
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# Command templates for runtimes this tool actually knows about. These
|
|
128
|
+
# are illustrative examples, not verified working commands — this tool
|
|
129
|
+
# does not download, store, or locate model files on disk.
|
|
130
|
+
_RUNTIME_COMMAND_TEMPLATES: dict[str, tuple[str, str]] = {
|
|
131
|
+
"llama.cpp": (
|
|
132
|
+
"llama-cli -m /path/to/{filename} -c {context_length}",
|
|
133
|
+
"Point this at the GGUF file you've downloaded for this "
|
|
134
|
+
"model - this tool does not download or store model files.",
|
|
135
|
+
),
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def build_run_command(model: ModelSpec) -> RunCommandInfo | None:
|
|
140
|
+
|
|
141
|
+
if model.runtime is None:
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
template = _RUNTIME_COMMAND_TEMPLATES.get(model.runtime.lower())
|
|
145
|
+
|
|
146
|
+
if template is None:
|
|
147
|
+
return None
|
|
148
|
+
|
|
149
|
+
command_template, note = template
|
|
150
|
+
|
|
151
|
+
command = command_template.format(
|
|
152
|
+
filename=f"{model.name}.gguf",
|
|
153
|
+
context_length=model.context_length,
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
return RunCommandInfo(
|
|
157
|
+
runtime=model.runtime,
|
|
158
|
+
command=command,
|
|
159
|
+
note=note,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def pick_best_for_you(ranked: list[RankedModel]) -> RankedModel | None:
|
|
164
|
+
|
|
165
|
+
if not ranked:
|
|
166
|
+
return None
|
|
167
|
+
|
|
168
|
+
top = ranked[0]
|
|
169
|
+
|
|
170
|
+
if top.tier in FAVORABLE_TIERS:
|
|
171
|
+
return top
|
|
172
|
+
|
|
173
|
+
return None
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def pick_alternative(
|
|
177
|
+
ranked: list[RankedModel],
|
|
178
|
+
exclude_model_name: str,
|
|
179
|
+
) -> ModelSpec | None:
|
|
180
|
+
"""The best currently-recommendable model other than the one being
|
|
181
|
+
looked at — shown when that model can't run, so the user always
|
|
182
|
+
has a next step instead of a dead end."""
|
|
183
|
+
|
|
184
|
+
for entry in ranked:
|
|
185
|
+
|
|
186
|
+
if entry.model.name == exclude_model_name:
|
|
187
|
+
continue
|
|
188
|
+
|
|
189
|
+
if entry.tier in FAVORABLE_TIERS:
|
|
190
|
+
return entry.model
|
|
191
|
+
|
|
192
|
+
return None
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
from pydantic import BaseModel
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class CPUResponse(BaseModel):
|
|
5
|
+
name: str
|
|
6
|
+
architecture: str
|
|
7
|
+
physical_cores: int | None
|
|
8
|
+
logical_cores: int | None
|
|
9
|
+
frequency_mhz: float | None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class MemoryResponse(BaseModel):
|
|
13
|
+
total_bytes: int
|
|
14
|
+
available_bytes: int
|
|
15
|
+
used_bytes: int
|
|
16
|
+
usage_percent: float
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class GPUResponse(BaseModel):
|
|
20
|
+
name: str
|
|
21
|
+
memory_total_bytes: int
|
|
22
|
+
memory_used_bytes: int
|
|
23
|
+
memory_free_bytes: int
|
|
24
|
+
utilization_percent: float
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class OSResponse(BaseModel):
|
|
28
|
+
system: str
|
|
29
|
+
release: str
|
|
30
|
+
version: str
|
|
31
|
+
machine: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class HardwareResponse(BaseModel):
|
|
35
|
+
cpu: CPUResponse
|
|
36
|
+
memory: MemoryResponse
|
|
37
|
+
os: OSResponse
|
|
38
|
+
gpus: list[GPUResponse]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ModelResponse(BaseModel):
|
|
42
|
+
name: str
|
|
43
|
+
family: str
|
|
44
|
+
architecture: str
|
|
45
|
+
parameters: int
|
|
46
|
+
quantization: str
|
|
47
|
+
context_length: int
|
|
48
|
+
runtime: str | None
|
|
49
|
+
file_size_bytes: int | None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class ReasonItemResponse(BaseModel):
|
|
53
|
+
ok: bool
|
|
54
|
+
text: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class CompatibilityResponse(BaseModel):
|
|
58
|
+
memory_verdict: str
|
|
59
|
+
runtime_verdict: str
|
|
60
|
+
overall_verdict: str
|
|
61
|
+
confidence: str
|
|
62
|
+
memory_strategy: str
|
|
63
|
+
required_memory_bytes: int
|
|
64
|
+
available_vram_bytes: int
|
|
65
|
+
available_ram_bytes: int
|
|
66
|
+
reason: str
|
|
67
|
+
|
|
68
|
+
# Human-facing interpretation of the fields above — generated
|
|
69
|
+
# server-side (see api/presentation.py), never in the frontend.
|
|
70
|
+
friendly_verdict: str
|
|
71
|
+
friendly_icon: str
|
|
72
|
+
reasons: list[ReasonItemResponse]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class ModelResultResponse(BaseModel):
|
|
76
|
+
model: ModelResponse
|
|
77
|
+
compatibility: CompatibilityResponse
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class SummaryResponse(BaseModel):
|
|
81
|
+
total: int
|
|
82
|
+
can_run: int
|
|
83
|
+
can_run_with_offload: int
|
|
84
|
+
needs_validation: int
|
|
85
|
+
cannot_run: int
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class ScanResponse(BaseModel):
|
|
89
|
+
hardware: HardwareResponse
|
|
90
|
+
summary: SummaryResponse
|
|
91
|
+
results: list[ModelResultResponse]
|
|
92
|
+
recommended: list[ModelResultResponse]
|
|
93
|
+
best_for_you: ModelResultResponse | None
|
|
94
|
+
scanned_at: str
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class MemoryBreakdownResponse(BaseModel):
|
|
98
|
+
weight_memory_bytes: int
|
|
99
|
+
kv_cache_bytes: int
|
|
100
|
+
runtime_overhead_bytes: int
|
|
101
|
+
safety_margin_bytes: int
|
|
102
|
+
total_required_bytes: int
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class RunCommandResponse(BaseModel):
|
|
106
|
+
runtime: str
|
|
107
|
+
command: str
|
|
108
|
+
note: str
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class ModelDetailResponse(BaseModel):
|
|
112
|
+
model: ModelResponse
|
|
113
|
+
compatibility: CompatibilityResponse
|
|
114
|
+
memory_breakdown: MemoryBreakdownResponse
|
|
115
|
+
run_command: RunCommandResponse | None
|
|
116
|
+
alternative: ModelResponse | None
|