make-localllm-easier 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- make_localllm_easier-0.1.0/LICENSE +21 -0
- make_localllm_easier-0.1.0/PKG-INFO +130 -0
- make_localllm_easier-0.1.0/README.md +114 -0
- make_localllm_easier-0.1.0/localllm/__init__.py +1 -0
- make_localllm_easier-0.1.0/localllm/__main__.py +3 -0
- make_localllm_easier-0.1.0/localllm/bench.py +135 -0
- make_localllm_easier-0.1.0/localllm/catalog.py +53 -0
- make_localllm_easier-0.1.0/localllm/cli.py +193 -0
- make_localllm_easier-0.1.0/localllm/runtime.py +102 -0
- make_localllm_easier-0.1.0/localllm/sizing.py +76 -0
- make_localllm_easier-0.1.0/make_localllm_easier.egg-info/PKG-INFO +130 -0
- make_localllm_easier-0.1.0/make_localllm_easier.egg-info/SOURCES.txt +16 -0
- make_localllm_easier-0.1.0/make_localllm_easier.egg-info/dependency_links.txt +1 -0
- make_localllm_easier-0.1.0/make_localllm_easier.egg-info/entry_points.txt +3 -0
- make_localllm_easier-0.1.0/make_localllm_easier.egg-info/top_level.txt +1 -0
- make_localllm_easier-0.1.0/pyproject.toml +24 -0
- make_localllm_easier-0.1.0/setup.cfg +4 -0
- make_localllm_easier-0.1.0/tests/test_core.py +48 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 homellm contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: make-localllm-easier
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: One command runs the best local AI your PC can handle: measured model picks + a tuned llama.cpp (AMD, NVIDIA, Intel, Apple)
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/phonology024/make-localllm-easier
|
|
7
|
+
Keywords: llm,local-llm,llama.cpp,vulkan,amd,gguf,thai,offline-ai
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
12
|
+
Requires-Python: >=3.9
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# make-localllm-easier — run the best local LLM your GPU can handle, in one command
|
|
18
|
+
|
|
19
|
+
**`localllm` picks, downloads and runs the most accurate local AI model for your PC and your language, chosen from real
|
|
20
|
+
benchmark measurements, with llama.cpp tuned for AMD, NVIDIA, Intel and Apple GPUs.**
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
pip install make-localllm-easier
|
|
24
|
+
localllm
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
That's it. `localllm` checks your GPU and RAM, picks the most accurate model we have *measured* for your language that
|
|
28
|
+
fits your card, downloads llama.cpp and the model, starts it with settings profiled op by op, and opens the chat page.
|
|
29
|
+
You also get an OpenAI-compatible API at `http://127.0.0.1:8080/v1` for any app that speaks it. Offline, private, free.
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
localllm doctor # what this GPU is good for: model sizes, speed, how much text it can hold
|
|
33
|
+
localllm list # every model we have measured, with scores per language
|
|
34
|
+
localllm serve # API only, no browser
|
|
35
|
+
localllm eval # score any running server in English + your language
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## FAQ
|
|
39
|
+
|
|
40
|
+
**Which local LLM should I run on my GPU?** Run `localllm doctor`. It lists which model sizes fit your card (4B up to
|
|
41
|
+
120B MoE), at which quantization, how fast they should run, and the most accurate measured model for your language.
|
|
42
|
+
|
|
43
|
+
**Can a 16 GB GPU run a 27B model?** Yes. Qwen3.8-27B at ~3.5 bits (12.2 GB) runs at ~50 tok/s on an RX 9070 XT and
|
|
44
|
+
keeps 81.8% on English Global-MMLU-Lite. gemma-4-26B-A4B (13.3 GB) runs at ~69 tok/s with similar accuracy.
|
|
45
|
+
|
|
46
|
+
**Is a 2-bit quantized model good enough?** Usually not for non-English use: 2-bit costs 8-13 accuracy points, and
|
|
47
|
+
Hindi, Arabic and Thai lose the most (13 points).
|
|
48
|
+
|
|
49
|
+
**Why is llama.cpp slow on my AMD (or Intel) GPU on Windows?** If Resizable BAR is off, llama.cpp's Vulkan backend puts
|
|
50
|
+
buffers in a 256 MB host-visible heap backed by system RAM and decode drops up to 1.7x. `localllm` sets
|
|
51
|
+
`GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM=1` for you ([llama.cpp#27097](https://github.com/ggml-org/llama.cpp/issues/27097)).
|
|
52
|
+
|
|
53
|
+
**Does it work offline?** After the first download, yes. Nothing leaves your PC.
|
|
54
|
+
|
|
55
|
+
**Which languages are measured?** 23 languages on Global-MMLU-Lite, 44 countries' own exams on INCLUDE, plus Thai
|
|
56
|
+
(ThaiExam). `localllm eval --langs ...` measures any of them on your hardware.
|
|
57
|
+
|
|
58
|
+
## What `localllm doctor` tells you
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
GPU AMD Radeon RX 9070 XT 15.9 GB (640 GB/s) RAM 32 GB language: th
|
|
62
|
+
|
|
63
|
+
Model sizes for this PC (whole model on the GPU = fast):
|
|
64
|
+
[OK ] 4B Q8 4.2 GB ~90 tok/s (est.)
|
|
65
|
+
[OK ] 8B Q8 8.5 GB ~45 tok/s (est.)
|
|
66
|
+
[OK ] 14B Q6 11.5 GB ~33 tok/s (est.)
|
|
67
|
+
[OK ] 24-32B Q3 13.2 GB ~29 tok/s (est.)
|
|
68
|
+
[SLOW] 30B MoE (3B active) Q4 18.0 GB with experts in RAM - works, ~10-25 tok/s
|
|
69
|
+
[NO ] 70B needs ~42.0 GB - too big for this PC
|
|
70
|
+
|
|
71
|
+
Best measured model for you: gemma4-26b-a4b-qat (MoE with ~4B active params: fastest)
|
|
72
|
+
What it can do here:
|
|
73
|
+
TH real local school/licence exams 65.7% correct <- your language
|
|
74
|
+
holds ~78k tokens at once (~130 pages of text) next to the model
|
|
75
|
+
answers at ~69 tok/s
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Speeds marked *est.* come from your card's memory bandwidth, calibrated on measured runs. Everything else is measured.
|
|
79
|
+
|
|
80
|
+
## Measured results (RX 9070 XT 16 GB, Windows 11, llama.cpp Vulkan)
|
|
81
|
+
|
|
82
|
+
Accuracy (%) on multiple-choice exams, zero-shot. **global** = Global-MMLU-Lite: the same 400 questions translated, so
|
|
83
|
+
languages compare like for like. **regional** = INCLUDE: real exams written in each country (ThaiExam for Thai).
|
|
84
|
+
|
|
85
|
+
| | Qwen3.8-27B Q3 (12.2 GB) | gemma-4-26B-A4B QAT Q4 (13.3 GB) | Qwen3.8-27B 2-bit (7.8 GB) |
|
|
86
|
+
|---|---|---|---|
|
|
87
|
+
| English | 81.8 | 82.2 | 74.2 |
|
|
88
|
+
| Chinese | 76.2 / 74.7 | 73.5 / 66.5 | 67.8 / 67.8 |
|
|
89
|
+
| Spanish | 80.2 / 76.8 | 74.5 / 75.2 | 70.8 / 69.2 |
|
|
90
|
+
| Japanese | 73.5 / 87.6 | 74.5 / 81.9 | 65.8 / 77.9 |
|
|
91
|
+
| Arabic | 70.8 / 71.2 | 71.5 / 73.6 | 60.8 / 57.2 |
|
|
92
|
+
| Hindi | 69.0 / 74.3 | 69.5 / 71.0 | 56.2 / 55.5 |
|
|
93
|
+
| Thai | – / 67.1 | – / 65.7 | – / 54.2 |
|
|
94
|
+
| **decode speed** | **50 tok/s** (MTP) | **69 tok/s** | 40 tok/s |
|
|
95
|
+
|
|
96
|
+
Cells are global / regional. Margins are about ±4 (global) and ±5 (regional) points at 95%, so `localllm` treats gaps
|
|
97
|
+
under 2 points as a tie and picks the faster model.
|
|
98
|
+
|
|
99
|
+
## Findings worth knowing
|
|
100
|
+
|
|
101
|
+
1. **2-bit costs 8-13 points, and lower-resource languages pay the most.** Hindi, Arabic and Thai lose 13; English,
|
|
102
|
+
Chinese and Spanish about 8-9. A 177B MoE squeezed to 1.6 bits scored *below* a 27B at 3 bits.
|
|
103
|
+
2. **Calibrating the quantization on your language doesn't help at ~3.5 bits.** A Thai-text importance matrix scored the
|
|
104
|
+
same as the stock one in Thai, English and Chinese (64.8 vs 64.6 Thai). At this level the number of bits matters,
|
|
105
|
+
the calibration text doesn't.
|
|
106
|
+
3. **AMD/Intel cards without Resizable BAR lose up to 1.7x decode speed** in llama.cpp's Vulkan backend. Hybrid DeltaNet
|
|
107
|
+
models (Qwen3.5/3.8) suffer most: they rewrite a 3 MB state per layer per token.
|
|
108
|
+
4. **Qwen3.8 GGUFs ship a multi-token-prediction head.** Drafting 2 tokens with it adds ~40% decode speed for free;
|
|
109
|
+
drafting 3 is slower.
|
|
110
|
+
5. **The first run of a new llama.cpp build is slow** while the GPU driver compiles its shaders once (~15 s).
|
|
111
|
+
|
|
112
|
+
## How the benchmark works
|
|
113
|
+
|
|
114
|
+
`localllm eval` asks each question with thinking off and reads the log-probability of every answer letter from the first
|
|
115
|
+
generated token, then takes the most likely one. It's prompt processing only, so a language takes a few minutes, and the
|
|
116
|
+
result is deterministic. Data is downloaded at eval time from the original Apache-2.0 datasets
|
|
117
|
+
([Global-MMLU-Lite](https://huggingface.co/datasets/CohereLabs/Global-MMLU-Lite),
|
|
118
|
+
[INCLUDE](https://huggingface.co/datasets/CohereLabs/include-lite-44),
|
|
119
|
+
[ThaiExam](https://huggingface.co/datasets/typhoon-ai/thai_exam)) and never redistributed. It measures knowledge and
|
|
120
|
+
reasoning in multiple choice, not writing quality.
|
|
121
|
+
|
|
122
|
+
## Contributing
|
|
123
|
+
|
|
124
|
+
The catalog only grows with measurements. Run `localllm eval --langs en,<yours>` on your GPU and open a PR with
|
|
125
|
+
`~/.localllm/results.json` and your GPU name. Other languages' local exams are very welcome. See [ROADMAP.md](ROADMAP.md)
|
|
126
|
+
for what's next: using less system RAM (0.2), working alongside cloud provider APIs (0.3), and a speed-only release (0.4).
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
MIT. Models keep their own licenses; benchmark data keeps its own (Apache-2.0).
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# make-localllm-easier — run the best local LLM your GPU can handle, in one command
|
|
2
|
+
|
|
3
|
+
**`localllm` picks, downloads and runs the most accurate local AI model for your PC and your language, chosen from real
|
|
4
|
+
benchmark measurements, with llama.cpp tuned for AMD, NVIDIA, Intel and Apple GPUs.**
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
pip install make-localllm-easier
|
|
8
|
+
localllm
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
That's it. `localllm` checks your GPU and RAM, picks the most accurate model we have *measured* for your language that
|
|
12
|
+
fits your card, downloads llama.cpp and the model, starts it with settings profiled op by op, and opens the chat page.
|
|
13
|
+
You also get an OpenAI-compatible API at `http://127.0.0.1:8080/v1` for any app that speaks it. Offline, private, free.
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
localllm doctor # what this GPU is good for: model sizes, speed, how much text it can hold
|
|
17
|
+
localllm list # every model we have measured, with scores per language
|
|
18
|
+
localllm serve # API only, no browser
|
|
19
|
+
localllm eval # score any running server in English + your language
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## FAQ
|
|
23
|
+
|
|
24
|
+
**Which local LLM should I run on my GPU?** Run `localllm doctor`. It lists which model sizes fit your card (4B up to
|
|
25
|
+
120B MoE), at which quantization, how fast they should run, and the most accurate measured model for your language.
|
|
26
|
+
|
|
27
|
+
**Can a 16 GB GPU run a 27B model?** Yes. Qwen3.8-27B at ~3.5 bits (12.2 GB) runs at ~50 tok/s on an RX 9070 XT and
|
|
28
|
+
keeps 81.8% on English Global-MMLU-Lite. gemma-4-26B-A4B (13.3 GB) runs at ~69 tok/s with similar accuracy.
|
|
29
|
+
|
|
30
|
+
**Is a 2-bit quantized model good enough?** Usually not for non-English use: 2-bit costs 8-13 accuracy points, and
|
|
31
|
+
Hindi, Arabic and Thai lose the most (13 points).
|
|
32
|
+
|
|
33
|
+
**Why is llama.cpp slow on my AMD (or Intel) GPU on Windows?** If Resizable BAR is off, llama.cpp's Vulkan backend puts
|
|
34
|
+
buffers in a 256 MB host-visible heap backed by system RAM and decode drops up to 1.7x. `localllm` sets
|
|
35
|
+
`GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM=1` for you ([llama.cpp#27097](https://github.com/ggml-org/llama.cpp/issues/27097)).
|
|
36
|
+
|
|
37
|
+
**Does it work offline?** After the first download, yes. Nothing leaves your PC.
|
|
38
|
+
|
|
39
|
+
**Which languages are measured?** 23 languages on Global-MMLU-Lite, 44 countries' own exams on INCLUDE, plus Thai
|
|
40
|
+
(ThaiExam). `localllm eval --langs ...` measures any of them on your hardware.
|
|
41
|
+
|
|
42
|
+
## What `localllm doctor` tells you
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
GPU AMD Radeon RX 9070 XT 15.9 GB (640 GB/s) RAM 32 GB language: th
|
|
46
|
+
|
|
47
|
+
Model sizes for this PC (whole model on the GPU = fast):
|
|
48
|
+
[OK ] 4B Q8 4.2 GB ~90 tok/s (est.)
|
|
49
|
+
[OK ] 8B Q8 8.5 GB ~45 tok/s (est.)
|
|
50
|
+
[OK ] 14B Q6 11.5 GB ~33 tok/s (est.)
|
|
51
|
+
[OK ] 24-32B Q3 13.2 GB ~29 tok/s (est.)
|
|
52
|
+
[SLOW] 30B MoE (3B active) Q4 18.0 GB with experts in RAM - works, ~10-25 tok/s
|
|
53
|
+
[NO ] 70B needs ~42.0 GB - too big for this PC
|
|
54
|
+
|
|
55
|
+
Best measured model for you: gemma4-26b-a4b-qat (MoE with ~4B active params: fastest)
|
|
56
|
+
What it can do here:
|
|
57
|
+
TH real local school/licence exams 65.7% correct <- your language
|
|
58
|
+
holds ~78k tokens at once (~130 pages of text) next to the model
|
|
59
|
+
answers at ~69 tok/s
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Speeds marked *est.* come from your card's memory bandwidth, calibrated on measured runs. Everything else is measured.
|
|
63
|
+
|
|
64
|
+
## Measured results (RX 9070 XT 16 GB, Windows 11, llama.cpp Vulkan)
|
|
65
|
+
|
|
66
|
+
Accuracy (%) on multiple-choice exams, zero-shot. **global** = Global-MMLU-Lite: the same 400 questions translated, so
|
|
67
|
+
languages compare like for like. **regional** = INCLUDE: real exams written in each country (ThaiExam for Thai).
|
|
68
|
+
|
|
69
|
+
| | Qwen3.8-27B Q3 (12.2 GB) | gemma-4-26B-A4B QAT Q4 (13.3 GB) | Qwen3.8-27B 2-bit (7.8 GB) |
|
|
70
|
+
|---|---|---|---|
|
|
71
|
+
| English | 81.8 | 82.2 | 74.2 |
|
|
72
|
+
| Chinese | 76.2 / 74.7 | 73.5 / 66.5 | 67.8 / 67.8 |
|
|
73
|
+
| Spanish | 80.2 / 76.8 | 74.5 / 75.2 | 70.8 / 69.2 |
|
|
74
|
+
| Japanese | 73.5 / 87.6 | 74.5 / 81.9 | 65.8 / 77.9 |
|
|
75
|
+
| Arabic | 70.8 / 71.2 | 71.5 / 73.6 | 60.8 / 57.2 |
|
|
76
|
+
| Hindi | 69.0 / 74.3 | 69.5 / 71.0 | 56.2 / 55.5 |
|
|
77
|
+
| Thai | – / 67.1 | – / 65.7 | – / 54.2 |
|
|
78
|
+
| **decode speed** | **50 tok/s** (MTP) | **69 tok/s** | 40 tok/s |
|
|
79
|
+
|
|
80
|
+
Cells are global / regional. Margins are about ±4 (global) and ±5 (regional) points at 95%, so `localllm` treats gaps
|
|
81
|
+
under 2 points as a tie and picks the faster model.
|
|
82
|
+
|
|
83
|
+
## Findings worth knowing
|
|
84
|
+
|
|
85
|
+
1. **2-bit costs 8-13 points, and lower-resource languages pay the most.** Hindi, Arabic and Thai lose 13; English,
|
|
86
|
+
Chinese and Spanish about 8-9. A 177B MoE squeezed to 1.6 bits scored *below* a 27B at 3 bits.
|
|
87
|
+
2. **Calibrating the quantization on your language doesn't help at ~3.5 bits.** A Thai-text importance matrix scored the
|
|
88
|
+
same as the stock one in Thai, English and Chinese (64.8 vs 64.6 Thai). At this level the number of bits matters,
|
|
89
|
+
the calibration text doesn't.
|
|
90
|
+
3. **AMD/Intel cards without Resizable BAR lose up to 1.7x decode speed** in llama.cpp's Vulkan backend. Hybrid DeltaNet
|
|
91
|
+
models (Qwen3.5/3.8) suffer most: they rewrite a 3 MB state per layer per token.
|
|
92
|
+
4. **Qwen3.8 GGUFs ship a multi-token-prediction head.** Drafting 2 tokens with it adds ~40% decode speed for free;
|
|
93
|
+
drafting 3 is slower.
|
|
94
|
+
5. **The first run of a new llama.cpp build is slow** while the GPU driver compiles its shaders once (~15 s).
|
|
95
|
+
|
|
96
|
+
## How the benchmark works
|
|
97
|
+
|
|
98
|
+
`localllm eval` asks each question with thinking off and reads the log-probability of every answer letter from the first
|
|
99
|
+
generated token, then takes the most likely one. It's prompt processing only, so a language takes a few minutes, and the
|
|
100
|
+
result is deterministic. Data is downloaded at eval time from the original Apache-2.0 datasets
|
|
101
|
+
([Global-MMLU-Lite](https://huggingface.co/datasets/CohereLabs/Global-MMLU-Lite),
|
|
102
|
+
[INCLUDE](https://huggingface.co/datasets/CohereLabs/include-lite-44),
|
|
103
|
+
[ThaiExam](https://huggingface.co/datasets/typhoon-ai/thai_exam)) and never redistributed. It measures knowledge and
|
|
104
|
+
reasoning in multiple choice, not writing quality.
|
|
105
|
+
|
|
106
|
+
## Contributing
|
|
107
|
+
|
|
108
|
+
The catalog only grows with measurements. Run `localllm eval --langs en,<yours>` on your GPU and open a PR with
|
|
109
|
+
`~/.localllm/results.json` and your GPU name. Other languages' local exams are very welcome. See [ROADMAP.md](ROADMAP.md)
|
|
110
|
+
for what's next: using less system RAM (0.2), working alongside cloud provider APIs (0.3), and a speed-only release (0.4).
|
|
111
|
+
|
|
112
|
+
## License
|
|
113
|
+
|
|
114
|
+
MIT. Models keep their own licenses; benchmark data keeps its own (Apache-2.0).
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Multilingual multiple-choice benchmark for any OpenAI-compatible server (llama-server, Ollama, LM Studio, vLLM ...).
|
|
2
|
+
|
|
3
|
+
Two kinds of test per language, so a score means something wherever you live:
|
|
4
|
+
global Global-MMLU-Lite (CohereLabs, Apache-2.0): the same 400 questions translated into 23 languages,
|
|
5
|
+
so languages and models compare like for like
|
|
6
|
+
regional INCLUDE-lite-44 (CohereLabs, Apache-2.0): real exams written in each country (~250 per language,
|
|
7
|
+
44 languages); ThaiExam (typhoon-ai, Apache-2.0) fills in Thai
|
|
8
|
+
Zero-shot, thinking off, one token: the answer is the option letter with the highest log-probability. Fast
|
|
9
|
+
(prompt processing only) and deterministic. Data is downloaded at eval time and cached, never redistributed.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import ast
|
|
14
|
+
import json
|
|
15
|
+
import locale
|
|
16
|
+
import time
|
|
17
|
+
import urllib.parse
|
|
18
|
+
import urllib.request
|
|
19
|
+
|
|
20
|
+
from .runtime import HOME
|
|
21
|
+
|
|
22
|
+
ROWS = "https://datasets-server.huggingface.co/rows?dataset={ds}&config={cfg}&split={split}&offset={off}&length=100"
|
|
23
|
+
GLOBAL_LANGS = ["ar", "bn", "cs", "cy", "de", "en", "es", "fr", "hi", "hu", "id", "it", "ja", "ko", "my", "or", "pt",
|
|
24
|
+
"sk", "sq", "sw", "tg", "yo", "zh"]
|
|
25
|
+
INCLUDE = {"sq": "Albanian", "ar": "Arabic", "hy": "Armenian", "az": "Azerbaijani", "eu": "Basque", "be": "Belarusian",
|
|
26
|
+
"bn": "Bengali", "bg": "Bulgarian", "zh": "Chinese", "hr": "Croatian", "nl": "Dutch", "et": "Estonian",
|
|
27
|
+
"fi": "Finnish", "fr": "French", "ka": "Georgian", "de": "German", "el": "Greek", "he": "Hebrew",
|
|
28
|
+
"hi": "Hindi", "hu": "Hungarian", "id": "Indonesian", "it": "Italian", "ja": "Japanese", "kk": "Kazakh",
|
|
29
|
+
"ko": "Korean", "lt": "Lithuanian", "ms": "Malay", "ml": "Malayalam", "ne": "Nepali",
|
|
30
|
+
"mk": "North Macedonian", "fa": "Persian", "pl": "Polish", "pt": "Portuguese", "ru": "Russian",
|
|
31
|
+
"sr": "Serbian", "es": "Spanish", "tl": "Tagalog", "ta": "Tamil", "te": "Telugu", "tr": "Turkish",
|
|
32
|
+
"uk": "Ukrainian", "ur": "Urdu", "uz": "Uzbek", "vi": "Vietnamese"}
|
|
33
|
+
THAIEXAM = "https://huggingface.co/datasets/typhoon-ai/thai_exam/resolve/main/data/{s}/{s}_test.jsonl"
|
|
34
|
+
SYSTEM = "Answer the multiple-choice question. Reply with only the letter of the correct option."
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def system_language() -> str:
|
|
38
|
+
try:
|
|
39
|
+
loc = locale.getlocale()[0] or ""
|
|
40
|
+
except ValueError:
|
|
41
|
+
loc = ""
|
|
42
|
+
if "_" in loc or len(loc) == 2:
|
|
43
|
+
return loc.split("_")[0].lower()[:2]
|
|
44
|
+
names = {n.lower(): c for c, n in INCLUDE.items()} | {"thai": "th", "english": "en"}
|
|
45
|
+
return names.get(loc.split("_")[0].lower(), "en")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def available(lang: str) -> list[str]:
|
|
49
|
+
return [s for s, ok in (("global", lang in GLOBAL_LANGS), ("regional", lang in INCLUDE or lang == "th")) if ok]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _rows(ds: str, cfg: str, split: str = "test") -> list[dict]:
|
|
53
|
+
out, off = [], 0
|
|
54
|
+
while True:
|
|
55
|
+
u = ROWS.format(ds=urllib.parse.quote(ds), cfg=urllib.parse.quote(cfg), split=split, off=off)
|
|
56
|
+
d = json.load(urllib.request.urlopen(u, timeout=120))
|
|
57
|
+
out += [r["row"] for r in d["rows"]]
|
|
58
|
+
off += 100
|
|
59
|
+
if off >= d.get("num_rows_total", 0) or not d["rows"]:
|
|
60
|
+
return out
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def load(suite: str, lang: str) -> list[dict]:
|
|
64
|
+
"""[{'q': question, 'opts': [..], 'ans': index}] cached under ~/.localllm/bench/."""
|
|
65
|
+
cache = HOME / "bench" / f"{suite}-{lang}.json"
|
|
66
|
+
if cache.exists():
|
|
67
|
+
return json.loads(cache.read_text(encoding="utf-8"))
|
|
68
|
+
items = []
|
|
69
|
+
if suite == "global":
|
|
70
|
+
for r in _rows("CohereLabs/Global-MMLU-Lite", lang):
|
|
71
|
+
opts = [r[f"option_{c}"] for c in "abcd"]
|
|
72
|
+
items.append({"q": r["question"], "opts": opts, "ans": "ABCD".index(r["answer"].strip().upper())})
|
|
73
|
+
elif suite == "regional" and lang == "th":
|
|
74
|
+
for s in ["onet", "ic", "tgat", "tpat1", "a_level"]:
|
|
75
|
+
for line in urllib.request.urlopen(THAIEXAM.format(s=s)).read().decode("utf-8").splitlines():
|
|
76
|
+
if line.strip():
|
|
77
|
+
r = json.loads(line)
|
|
78
|
+
keys = [c for c in "abcde" if r.get(c)]
|
|
79
|
+
items.append({"q": r["question"], "opts": [r[c] for c in keys],
|
|
80
|
+
"ans": keys.index(r["answer"].strip().lower())})
|
|
81
|
+
elif suite == "regional":
|
|
82
|
+
for r in _rows("CohereLabs/include-lite-44", INCLUDE[lang]):
|
|
83
|
+
opts = r["choices"] if isinstance(r["choices"], list) else ast.literal_eval(r["choices"])
|
|
84
|
+
items.append({"q": r["question"], "opts": opts, "ans": int(r["answer"])})
|
|
85
|
+
else:
|
|
86
|
+
raise ValueError(f"no {suite} test for '{lang}'")
|
|
87
|
+
cache.parent.mkdir(parents=True, exist_ok=True)
|
|
88
|
+
cache.write_text(json.dumps(items, ensure_ascii=False), encoding="utf-8")
|
|
89
|
+
return items
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def ask(url: str, item: dict) -> int | None:
|
|
93
|
+
letters = "ABCDEFGHIJ"[: len(item["opts"])]
|
|
94
|
+
opts = "\n".join(f"{l}. {o}" for l, o in zip(letters, item["opts"]))
|
|
95
|
+
body = {"messages": [{"role": "system", "content": SYSTEM},
|
|
96
|
+
{"role": "user", "content": f"{item['q']}\n\n{opts}\n\nAnswer ({'/'.join(letters)}):"}],
|
|
97
|
+
"max_tokens": 1, "temperature": 0, "logprobs": True, "top_logprobs": 20,
|
|
98
|
+
"chat_template_kwargs": {"enable_thinking": False}}
|
|
99
|
+
req = urllib.request.Request(url.rstrip("/") + "/v1/chat/completions", data=json.dumps(body).encode(),
|
|
100
|
+
headers={"Content-Type": "application/json"})
|
|
101
|
+
d = json.load(urllib.request.urlopen(req, timeout=600))
|
|
102
|
+
best, best_lp = None, -1e9
|
|
103
|
+
for t in d["choices"][0]["logprobs"]["content"][0]["top_logprobs"]:
|
|
104
|
+
tok = t["token"].strip().upper().strip(".()")
|
|
105
|
+
if len(tok) == 1 and tok in letters and t["logprob"] > best_lp:
|
|
106
|
+
best, best_lp = letters.index(tok), t["logprob"]
|
|
107
|
+
return best
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _save(name: str, res: dict) -> None:
|
|
111
|
+
out = HOME / "results.json"
|
|
112
|
+
allres = json.loads(out.read_text(encoding="utf-8")) if out.exists() else {}
|
|
113
|
+
allres[name] = {**allres.get(name, {}), **res}
|
|
114
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
115
|
+
out.write_text(json.dumps(allres, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def run(url: str, name: str, langs: list[str], limit: int = 0) -> dict:
|
|
119
|
+
"""Scores are saved after every test, so an interrupted run keeps what it finished."""
|
|
120
|
+
res, t0 = {}, time.time()
|
|
121
|
+
for lang in langs:
|
|
122
|
+
suites = available(lang)
|
|
123
|
+
if not suites:
|
|
124
|
+
print(f" {lang}: no benchmark yet (contributions welcome)")
|
|
125
|
+
for suite in suites:
|
|
126
|
+
items = load(suite, lang)
|
|
127
|
+
items = items[:limit] if limit else items
|
|
128
|
+
ok = sum(ask(url, it) == it["ans"] for it in items)
|
|
129
|
+
acc = round(100 * ok / len(items), 1)
|
|
130
|
+
res[f"{lang}/{suite}"] = {"acc": acc, "correct": ok, "n": len(items)}
|
|
131
|
+
margin = round(196 * (acc / 100 * (1 - acc / 100) / len(items)) ** 0.5, 1)
|
|
132
|
+
print(f" {lang:3} {suite:9} {acc:5.1f}% ±{margin} ({ok}/{len(items)})", flush=True)
|
|
133
|
+
_save(name, {**res, "_meta": {"model": name, "seconds": round(time.time() - t0), "limit": limit}})
|
|
134
|
+
print(f"saved to {HOME / 'results.json'} (share it: open a PR adding your GPU's numbers)")
|
|
135
|
+
return res
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Models we have measured end to end. `scores` = accuracy (%) from `localllm eval` keyed "lang/suite" (see bench.py).
|
|
2
|
+
gb = weights in GiB; kv_kb_per_token = KV cache per token at q8 (from the GGUF attention layout).
|
|
3
|
+
Speeds: llama-server decode tok/s on an RX 9070 XT 16 GB (Windows, Vulkan) with the tuned launch in runtime.py.
|
|
4
|
+
Add a model only after measuring it with `localllm eval`."""
|
|
5
|
+
|
|
6
|
+
MODELS = {
|
|
7
|
+
"qwen3.8-27b-q3": {
|
|
8
|
+
"repo": "unsloth/Qwen3.8-27B-GGUF", "file": "Qwen3.8-27B-UD-Q3_K_XL.gguf", "gb": 12.2,
|
|
9
|
+
"kv_kb_per_token": 34.8, "fixed_cache_gb": 0.15, "max_ctx": 262144, "tok_s_9070xt": 50, "mtp": True,
|
|
10
|
+
"scores": {"en/global": 81.5, "zh/global": 76.2, "zh/regional": 74.7, "es/global": 80.2, "es/regional": 76.8, "hi/global": 69.0, "hi/regional": 74.3, "ar/global": 70.8, "ar/regional": 71.2, "ja/global": 73.5, "ja/regional": 87.6, "th/regional": 67.1},
|
|
11
|
+
"note": "dense 27B; built-in MTP head drafts 2 tokens",
|
|
12
|
+
},
|
|
13
|
+
"gemma4-26b-a4b-qat": {
|
|
14
|
+
"repo": "unsloth/gemma-4-26B-A4B-it-qat-GGUF", "file": "gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf", "gb": 13.3,
|
|
15
|
+
"kv_kb_per_token": 10.9, "fixed_cache_gb": 0.11, "max_ctx": 262144, "tok_s_9070xt": 69, "mtp": False,
|
|
16
|
+
"scores": {"en/global": 82.2, "zh/global": 73.5, "zh/regional": 66.5, "es/global": 74.5, "es/regional": 75.2, "hi/global": 69.5, "hi/regional": 71.0, "ar/global": 71.5, "ar/regional": 73.6, "ja/global": 74.5, "ja/regional": 81.9, "th/regional": 65.7},
|
|
17
|
+
"note": "MoE with ~4B active params: fastest",
|
|
18
|
+
},
|
|
19
|
+
"qwen3.8-27b-iq2": {
|
|
20
|
+
"repo": "unsloth/Qwen3.8-27B-GGUF", "file": "Qwen3.8-27B-UD-IQ2_S.gguf", "gb": 7.8,
|
|
21
|
+
"kv_kb_per_token": 34.8, "fixed_cache_gb": 0.15, "max_ctx": 262144, "tok_s_9070xt": 40, "mtp": False,
|
|
22
|
+
"scores": {"en/global": 74.2, "zh/global": 67.8, "zh/regional": 67.8, "es/global": 70.8, "es/regional": 69.2, "hi/global": 56.2, "hi/regional": 55.5, "ar/global": 60.8, "ar/regional": 57.2, "ja/global": 65.8, "ja/regional": 77.9, "th/regional": 54.2},
|
|
23
|
+
"note": "for 10-12 GB cards only: 2-bit costs 8-13 points, most in Hindi, Arabic, Thai",
|
|
24
|
+
},
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def score(key: str, lang: str | None = None) -> float:
|
|
29
|
+
"""Mean accuracy over the user's language tests if we have them, else over everything measured."""
|
|
30
|
+
s = MODELS[key]["scores"]
|
|
31
|
+
mine = [v for k, v in s.items() if lang and k.startswith(lang + "/")]
|
|
32
|
+
pool = mine or list(s.values())
|
|
33
|
+
return sum(pool) / len(pool) if pool else 0.0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
MIN_CTX = 8192 # a model only "fits" if it also leaves room for an 8k-token conversation
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def fits(key: str, vram_gb: float) -> bool:
|
|
40
|
+
from .sizing import context_tokens
|
|
41
|
+
return context_tokens(vram_gb, MODELS[key]) >= MIN_CTX
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
TIE_POINTS = 2.0 # accuracy gaps this small are inside the benchmark's margin: prefer the faster model
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def pick(vram_gb: float, lang: str | None = None) -> str | None:
|
|
48
|
+
"""Most accurate model (for `lang` when measured) that fits the card whole with an 8k context; near-ties go to speed."""
|
|
49
|
+
ok = [k for k in MODELS if fits(k, vram_gb)]
|
|
50
|
+
if not ok:
|
|
51
|
+
return None
|
|
52
|
+
best = max(score(k, lang) for k in ok)
|
|
53
|
+
return max((k for k in ok if score(k, lang) >= best - TIE_POINTS), key=lambda k: MODELS[k]["tok_s_9070xt"])
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""make-localllm-easier: one command, the best local AI your computer can run.
|
|
2
|
+
|
|
3
|
+
localllm check this PC, pick the best model, download, start, open the chat page
|
|
4
|
+
localllm doctor what GPU/RAM you have and which model fits
|
|
5
|
+
localllm list every model we have measured
|
|
6
|
+
localllm serve [MODEL] start an OpenAI-compatible server only (http://127.0.0.1:8080/v1)
|
|
7
|
+
localllm eval score a running server in English + your language (global and local exams)
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import os
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
import time
|
|
16
|
+
import urllib.request
|
|
17
|
+
import webbrowser
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from . import __version__, catalog, runtime
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _say(msg: str) -> None:
|
|
24
|
+
print(f"[localllm] {msg}", flush=True)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _download(url: str, dest: Path, label: str) -> None:
|
|
28
|
+
"""Resumable download with a one-line progress bar."""
|
|
29
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
30
|
+
tmp = dest.with_suffix(dest.suffix + ".part")
|
|
31
|
+
done = tmp.stat().st_size if tmp.exists() else 0
|
|
32
|
+
req = urllib.request.Request(url, headers={"Range": f"bytes={done}-"} if done else {})
|
|
33
|
+
with urllib.request.urlopen(req) as r, open(tmp, "ab") as f:
|
|
34
|
+
total = done + int(r.headers.get("Content-Length") or 0)
|
|
35
|
+
t0, got = time.time(), 0
|
|
36
|
+
while chunk := r.read(1 << 22):
|
|
37
|
+
f.write(chunk); got += len(chunk)
|
|
38
|
+
if total:
|
|
39
|
+
pct = 100 * (done + got) / total
|
|
40
|
+
speed = got / max(time.time() - t0, 1e-3) / 2**20
|
|
41
|
+
print(f"\r {label}: {pct:5.1f}% of {total / 2**30:.1f} GB ({speed:.0f} MB/s) ", end="", flush=True)
|
|
42
|
+
print()
|
|
43
|
+
tmp.replace(dest)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _model_path(key: str) -> Path:
|
|
47
|
+
m = catalog.MODELS[key]
|
|
48
|
+
for d in filter(None, os.environ.get("LOCALLLM_MODELS", "").split(os.pathsep)):
|
|
49
|
+
if (Path(d) / m["file"]).exists():
|
|
50
|
+
return Path(d) / m["file"]
|
|
51
|
+
dest = runtime.HOME / "models" / m["file"]
|
|
52
|
+
if not dest.exists():
|
|
53
|
+
_say(f"downloading {key} ({m['gb']} GB, one time) ...")
|
|
54
|
+
_download(f"https://huggingface.co/{m['repo']}/resolve/main/{m['file']}", dest, m["file"])
|
|
55
|
+
return dest
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _machine():
|
|
59
|
+
server = runtime.find_server()
|
|
60
|
+
devs = runtime.devices(server)
|
|
61
|
+
return server, devs, runtime.best_device(devs), runtime.ram_gb()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _start(key: str | None, port: int, ctx: int) -> tuple[subprocess.Popen, str]:
|
|
65
|
+
server, _devs, dev, _ram = _machine()
|
|
66
|
+
from .bench import system_language
|
|
67
|
+
key = key or (catalog.pick(dev["total_gb"], system_language()) if dev else None)
|
|
68
|
+
if not key:
|
|
69
|
+
sys.exit("[localllm] no measured model fits this GPU yet (need >= 10 GB VRAM). See `localllm list`.")
|
|
70
|
+
model = _model_path(key)
|
|
71
|
+
args = runtime.server_args(model, dev["id"] if dev else None, port, ctx, catalog.MODELS[key]["mtp"])
|
|
72
|
+
runtime.HOME.mkdir(parents=True, exist_ok=True)
|
|
73
|
+
log = open(runtime.HOME / "llama-server.log", "ab")
|
|
74
|
+
proc = subprocess.Popen([str(server), *args], env=runtime.server_env(), stdout=log, stderr=subprocess.STDOUT)
|
|
75
|
+
_say(f"loading {key} on {dev['name'] if dev else 'CPU'} ...")
|
|
76
|
+
url = f"http://127.0.0.1:{port}"
|
|
77
|
+
for _ in range(1000):
|
|
78
|
+
if proc.poll() is not None:
|
|
79
|
+
sys.exit(f"[localllm] llama-server stopped (exit {proc.returncode}); log: {runtime.HOME / 'llama-server.log'}")
|
|
80
|
+
try:
|
|
81
|
+
if b'"ok"' in urllib.request.urlopen(url + "/health", timeout=2).read():
|
|
82
|
+
return proc, url
|
|
83
|
+
except OSError:
|
|
84
|
+
pass
|
|
85
|
+
time.sleep(0.3)
|
|
86
|
+
proc.kill()
|
|
87
|
+
sys.exit("[localllm] model did not load within 5 minutes")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cmd_run(a) -> None:
|
|
91
|
+
proc, url = _start(a.model, a.port, a.ctx)
|
|
92
|
+
_say(f"ready. chat: {url} API (OpenAI-compatible): {url}/v1 Ctrl+C to stop")
|
|
93
|
+
if not a.no_browser:
|
|
94
|
+
webbrowser.open(url)
|
|
95
|
+
try:
|
|
96
|
+
proc.wait()
|
|
97
|
+
except KeyboardInterrupt:
|
|
98
|
+
proc.terminate()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def cmd_serve(a) -> None:
|
|
102
|
+
a.no_browser = True
|
|
103
|
+
cmd_run(a)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
ICON = {"fits": "OK ", "low-bits": "WARN", "offload-moe": "SLOW", "offload-dense": "SLOW", "too-big": "NO "}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def cmd_doctor(_a) -> None:
|
|
110
|
+
from . import sizing
|
|
111
|
+
from .bench import system_language
|
|
112
|
+
server, devs, dev, ram = _machine()
|
|
113
|
+
lang = system_language()
|
|
114
|
+
gpu = dev["name"] if dev else "no GPU found"
|
|
115
|
+
vram = dev["total_gb"] if dev else 0.0
|
|
116
|
+
bw = sizing.bandwidth(gpu)
|
|
117
|
+
print(f"GPU {gpu} {vram:.1f} GB" + (f" ({bw} GB/s)" if bw else "") + f" RAM {ram:.0f} GB language: {lang}")
|
|
118
|
+
print("\nModel sizes for this PC (whole model on the GPU = fast):")
|
|
119
|
+
for r in sizing.tiers(vram, ram, gpu):
|
|
120
|
+
speed = f"~{r['tok_s']} tok/s (est.)" if r["tok_s"] else ""
|
|
121
|
+
what = {"fits": f"{r['quant']} {r['gb']} GB {speed}",
|
|
122
|
+
"low-bits": f"only at {r['quant']} ({r['gb']} GB) - fits, but quality drops sharply below 3 bits",
|
|
123
|
+
"offload-moe": f"{r['quant']} {r['gb']} GB with experts in RAM - works, ~10-25 tok/s",
|
|
124
|
+
"offload-dense": f"{r['quant']} {r['gb']} GB with layers in RAM - very slow (< 5 tok/s)",
|
|
125
|
+
"too-big": f"needs ~{r['gb']} GB - too big for this PC"}[r["status"]]
|
|
126
|
+
print(f" [{ICON[r['status']]}] {r['shape']:24} {what}")
|
|
127
|
+
key = catalog.pick(vram, lang) if dev else None
|
|
128
|
+
if not key:
|
|
129
|
+
print("\nNo measured model fits this GPU yet. Run `localllm list`, or help by measuring one (`localllm eval`).")
|
|
130
|
+
return
|
|
131
|
+
m = catalog.MODELS[key]
|
|
132
|
+
ctx = sizing.context_tokens(vram, m)
|
|
133
|
+
print(f"\nBest measured model for you: {key} ({m['note']})")
|
|
134
|
+
print("What it can do here:")
|
|
135
|
+
shown = [t for t in sorted(m["scores"]) if t.split("/")[0] in (lang, "en")]
|
|
136
|
+
others = sorted({t.split("/")[0] for t in m["scores"]} - {lang, "en"})
|
|
137
|
+
for test in shown:
|
|
138
|
+
acc = m["scores"][test]
|
|
139
|
+
tl, suite = test.split("/")
|
|
140
|
+
kind = "translated world-knowledge exam" if suite == "global" else "real local school/licence exams"
|
|
141
|
+
mark = " <- your language" if tl == lang else ""
|
|
142
|
+
print(f" {tl.upper():3} {kind:32} {acc:5.1f}% correct{mark}")
|
|
143
|
+
if others:
|
|
144
|
+
print(f" also measured in {', '.join(others)} (`localllm list`)")
|
|
145
|
+
print(f" holds ~{ctx // 1000}k tokens at once (~{ctx // sizing.TOKENS_PER_PAGE} pages of text) next to the model")
|
|
146
|
+
same = bw == sizing.BANDWIDTH["rx 9070 xt"]
|
|
147
|
+
est = m["tok_s_9070xt"] if same else (int(m["tok_s_9070xt"] * bw / 640) if bw else None)
|
|
148
|
+
if est:
|
|
149
|
+
print(f" answers at ~{est} tok/s" + ("" if same else " (estimated from memory bandwidth)"))
|
|
150
|
+
print(f"\nRun it: localllm (llama.cpp: {server})")
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def cmd_list(_a) -> None:
|
|
154
|
+
print(f"{'model':22} {'weights':>8} {'tok/s*':>7} scores")
|
|
155
|
+
for k, m in sorted(catalog.MODELS.items(), key=lambda kv: -catalog.score(kv[0])):
|
|
156
|
+
sc = " ".join(f"{t} {v:.1f}" for t, v in m["scores"].items())
|
|
157
|
+
print(f"{k:22} {m['gb']:6.1f}GB {m['tok_s_9070xt']:7} {sc}")
|
|
158
|
+
print("* decode speed on an RX 9070 XT 16 GB. scores: accuracy % from `localllm eval` (lang/suite)")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def cmd_eval(a) -> None:
|
|
162
|
+
from . import bench
|
|
163
|
+
langs = a.langs.split(",") if a.langs else sorted({"en", bench.system_language()})
|
|
164
|
+
_say(f"benchmarking {a.url} in: {', '.join(langs)} (pick others with --langs ja,de,...)")
|
|
165
|
+
bench.run(a.url, a.name, langs, a.limit)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def main() -> None:
|
|
169
|
+
ap = argparse.ArgumentParser(prog="localllm", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
170
|
+
ap.add_argument("--version", action="version", version=__version__)
|
|
171
|
+
ap.add_argument("--model", choices=list(catalog.MODELS), help="override the automatic pick")
|
|
172
|
+
ap.add_argument("--port", type=int, default=8080)
|
|
173
|
+
ap.add_argument("--ctx", type=int, default=8192, help="context length in tokens")
|
|
174
|
+
ap.add_argument("--no-browser", action="store_true")
|
|
175
|
+
ap.set_defaults(fn=cmd_run)
|
|
176
|
+
sub = ap.add_subparsers(dest="cmd")
|
|
177
|
+
sub.add_parser("doctor").set_defaults(fn=cmd_doctor)
|
|
178
|
+
sub.add_parser("list").set_defaults(fn=cmd_list)
|
|
179
|
+
s = sub.add_parser("serve"); s.add_argument("model", nargs="?", choices=list(catalog.MODELS))
|
|
180
|
+
s.add_argument("--port", type=int, default=8080); s.add_argument("--ctx", type=int, default=8192)
|
|
181
|
+
s.set_defaults(fn=cmd_serve)
|
|
182
|
+
e = sub.add_parser("eval"); e.add_argument("--url", default="http://127.0.0.1:8080")
|
|
183
|
+
e.add_argument("--name", default="model"); e.add_argument("--limit", type=int, default=0)
|
|
184
|
+
e.add_argument("--langs", help="comma-separated ISO codes, default: en + this PC's language")
|
|
185
|
+
e.set_defaults(fn=cmd_eval)
|
|
186
|
+
a = ap.parse_args()
|
|
187
|
+
if a.fn is cmd_serve:
|
|
188
|
+
a.model = a.model or None
|
|
189
|
+
a.fn(a)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
if __name__ == "__main__":
|
|
193
|
+
main()
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Fetch an upstream llama.cpp build, list GPUs, and launch llama-server with the settings that measured fastest.
|
|
2
|
+
|
|
3
|
+
Tuning (each measured on an RX 9070 XT, see README):
|
|
4
|
+
GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM=1 1.7x decode when Resizable BAR is off (llama.cpp#27097); harmless when on
|
|
5
|
+
-np 1 -kvu single user: one slot, unified KV
|
|
6
|
+
--spec-type draft-mtp --spec-draft-n-max 2 models that ship an MTP head (Qwen3.8): +40% decode
|
|
7
|
+
-fit off, --load-mode none skip the fit dry-run (~0.6 s) and read weights straight into VRAM
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import io
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import platform
|
|
15
|
+
import re
|
|
16
|
+
import subprocess
|
|
17
|
+
import tarfile
|
|
18
|
+
import urllib.request
|
|
19
|
+
import zipfile
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
HOME = Path(os.environ.get("LOCALLLM_HOME", Path.home() / ".localllm"))
|
|
23
|
+
EXE = "llama-server.exe" if os.name == "nt" else "llama-server"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _asset() -> str:
|
|
27
|
+
system = {"Windows": "win", "Linux": "ubuntu", "Darwin": "macos"}[platform.system()]
|
|
28
|
+
arch = "arm64" if platform.machine().lower() in ("arm64", "aarch64") else "x64"
|
|
29
|
+
if system == "macos":
|
|
30
|
+
return f"bin-macos-{arch}"
|
|
31
|
+
return f"bin-{system}-vulkan-{arch}"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def find_server() -> Path:
|
|
35
|
+
if os.environ.get("LOCALLLM_LLAMA_SERVER"):
|
|
36
|
+
return Path(os.environ["LOCALLLM_LLAMA_SERVER"])
|
|
37
|
+
found = sorted((HOME / "llama.cpp").rglob(EXE)) if (HOME / "llama.cpp").exists() else []
|
|
38
|
+
return found[-1] if found else fetch_server()
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def fetch_server() -> Path:
|
|
42
|
+
# "latest" can be a source-only tag (v0.6.0 had no binaries); the binaries ride on the per-build bNNNNN releases
|
|
43
|
+
rels = json.load(urllib.request.urlopen("https://api.github.com/repos/ggml-org/llama.cpp/releases?per_page=10"))
|
|
44
|
+
want = _asset()
|
|
45
|
+
rel, asset = next((r, a) for r in rels for a in r["assets"] if want in a["name"] and a["name"].startswith("llama-"))
|
|
46
|
+
dest = HOME / "llama.cpp" / rel["tag_name"]
|
|
47
|
+
print(f"downloading {asset['name']} ({asset['size'] / 2**20:.0f} MB) ...")
|
|
48
|
+
data = urllib.request.urlopen(asset["browser_download_url"]).read()
|
|
49
|
+
if asset["name"].endswith(".zip"):
|
|
50
|
+
zipfile.ZipFile(io.BytesIO(data)).extractall(dest)
|
|
51
|
+
else:
|
|
52
|
+
tarfile.open(fileobj=io.BytesIO(data)).extractall(dest, filter="data")
|
|
53
|
+
exe = next(dest.rglob(EXE))
|
|
54
|
+
if os.name != "nt":
|
|
55
|
+
exe.chmod(0o755)
|
|
56
|
+
return exe
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def devices(server: Path) -> list[dict]:
|
|
60
|
+
"""[{'id': 'Vulkan0', 'name': ..., 'total_gb': .., 'free_gb': ..}] from `llama-server --list-devices`."""
|
|
61
|
+
out = subprocess.run([str(server), "--list-devices"], capture_output=True, text=True, errors="replace")
|
|
62
|
+
devs = []
|
|
63
|
+
for m in re.finditer(r"^\s*(\w+\d+): (.+?) \((\d+) MiB, (\d+) MiB free\)", out.stdout + out.stderr, re.M):
|
|
64
|
+
devs.append({"id": m[1], "name": m[2], "total_gb": int(m[3]) / 1024, "free_gb": int(m[4]) / 1024})
|
|
65
|
+
return devs
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def best_device(devs: list[dict]) -> dict | None:
|
|
69
|
+
"""Largest-memory device; integrated GPUs report shared RAM, so prefer names that aren't iGPUs."""
|
|
70
|
+
igpu = re.compile(r"UHD|Iris|Radeon\(TM\) Graphics|Radeon Graphics|890M|780M|Apple", re.I)
|
|
71
|
+
pool = [d for d in devs if not igpu.search(d["name"])] or devs
|
|
72
|
+
return max(pool, key=lambda d: d["total_gb"]) if pool else None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def ram_gb() -> float:
|
|
76
|
+
if os.name == "nt":
|
|
77
|
+
import ctypes
|
|
78
|
+
|
|
79
|
+
class MS(ctypes.Structure):
|
|
80
|
+
_fields_ = [("len", ctypes.c_ulong), ("load", ctypes.c_ulong), ("total", ctypes.c_ulonglong),
|
|
81
|
+
("avail", ctypes.c_ulonglong), ("a", ctypes.c_ulonglong), ("b", ctypes.c_ulonglong),
|
|
82
|
+
("c", ctypes.c_ulonglong), ("d", ctypes.c_ulonglong), ("e", ctypes.c_ulonglong)]
|
|
83
|
+
s = MS(); s.len = ctypes.sizeof(MS)
|
|
84
|
+
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(s))
|
|
85
|
+
return s.total / 2**30
|
|
86
|
+
if Path("/proc/meminfo").exists():
|
|
87
|
+
return int(re.search(r"MemTotal:\s+(\d+)", Path("/proc/meminfo").read_text())[1]) / 2**20
|
|
88
|
+
return int(subprocess.run(["sysctl", "-n", "hw.memsize"], capture_output=True, text=True).stdout) / 2**30
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def server_args(model: Path, device: str | None, port: int, ctx: int, mtp: bool) -> list[str]:
|
|
92
|
+
args = ["-m", str(model), "--host", "127.0.0.1", "--port", str(port), "-c", str(ctx), "-ngl", "999",
|
|
93
|
+
"-fa", "on", "-ctk", "q8_0", "-ctv", "q8_0", "-np", "1", "-kvu", "-fit", "off", "--load-mode", "none"]
|
|
94
|
+
if device:
|
|
95
|
+
args += ["-dev", device]
|
|
96
|
+
if mtp:
|
|
97
|
+
args += ["--spec-type", "draft-mtp", "--spec-draft-n-max", "2"]
|
|
98
|
+
return args
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def server_env() -> dict:
|
|
102
|
+
return {**os.environ, "GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM": os.environ.get("GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM", "1")}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""What size of model a GPU suits, how fast it should run, and how long a document it can hold.
|
|
2
|
+
|
|
3
|
+
Everything here is arithmetic from the card's memory size and bandwidth, calibrated on measured runs:
|
|
4
|
+
dense, whole model in VRAM decode ≈ 0.6 x bandwidth / bytes read per token (Qwen3.8-27B Q3 on 640 GB/s: 35 tok/s)
|
|
5
|
+
MoE, whole model in VRAM decode ≈ 0.3 x bandwidth / active bytes per token (gemma-4-26B-A4B Q4: 69 tok/s)
|
|
6
|
+
Estimates are labelled as such; measured numbers live in catalog.py.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
# memory bandwidth in GB/s of common consumer GPUs (vendor specs)
|
|
13
|
+
BANDWIDTH = {
|
|
14
|
+
"rtx 5090": 1792, "rtx 5080": 960, "rtx 5070 ti": 896, "rtx 5070": 672, "rtx 5060 ti": 448, "rtx 5060": 448,
|
|
15
|
+
"rtx 4090": 1008, "rtx 4080": 717, "rtx 4070 ti": 504, "rtx 4070": 504, "rtx 4060 ti": 288, "rtx 4060": 272,
|
|
16
|
+
"rtx 3090": 936, "rtx 3080": 760, "rtx 3070": 448, "rtx 3060 ti": 448, "rtx 3060": 360,
|
|
17
|
+
"rx 9070 xt": 640, "rx 9070": 640, "rx 9060 xt": 320, "rx 7900 xtx": 960, "rx 7900 xt": 800, "rx 7800 xt": 624,
|
|
18
|
+
"rx 7700 xt": 432, "rx 7600": 288, "rx 6900 xt": 512, "rx 6800 xt": 512, "rx 6700 xt": 384,
|
|
19
|
+
"arc b580": 456, "arc a770": 560, "arc a750": 512,
|
|
20
|
+
}
|
|
21
|
+
# bits per weight including quantization overhead
|
|
22
|
+
QUANTS = [("Q8", 8.5), ("Q6", 6.6), ("Q4", 4.8), ("Q3", 3.9), ("Q2", 2.9)]
|
|
23
|
+
LOW_BITS = 3.0 # below this, measured accuracy drops sharply (Qwen3.8-27B: -17 ThaiExam points at 2.9 bpw)
|
|
24
|
+
DESKTOP_GB = 1.0 # what the OS/desktop usually keeps on the card
|
|
25
|
+
OVERHEAD_GB = 0.7 # compute buffers + small fixed caches
|
|
26
|
+
TOKENS_PER_PAGE = 600 # ~450 English words
|
|
27
|
+
|
|
28
|
+
# reference shapes people actually download: (label, total params B, active params B)
|
|
29
|
+
SHAPES = [("4B", 4, 4), ("8B", 8, 8), ("14B", 14, 14), ("24-32B", 27, 27), ("30B MoE (3B active)", 30, 3),
|
|
30
|
+
("70B", 70, 70), ("120B MoE (10B active)", 120, 10)]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def bandwidth(gpu_name: str) -> int | None:
|
|
34
|
+
n = gpu_name.lower()
|
|
35
|
+
for key in sorted(BANDWIDTH, key=len, reverse=True):
|
|
36
|
+
if re.search(r"\b" + re.escape(key) + r"\b", n):
|
|
37
|
+
return BANDWIDTH[key]
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def size_gb(params_b: float, bpw: float) -> float:
|
|
42
|
+
return params_b * bpw / 8
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def tok_s(total_b: float, active_b: float, bpw: float, bw: int | None) -> int | None:
|
|
46
|
+
if not bw:
|
|
47
|
+
return None
|
|
48
|
+
eff = 0.6 if active_b == total_b else 0.3
|
|
49
|
+
return int(eff * bw / size_gb(active_b, bpw))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def tiers(vram_gb: float, ram_gb: float, gpu_name: str) -> list[dict]:
|
|
53
|
+
"""One row per reference shape: best quant that fits the card whole, else whether RAM offload works."""
|
|
54
|
+
usable, bw, rows = vram_gb - DESKTOP_GB - OVERHEAD_GB, bandwidth(gpu_name), []
|
|
55
|
+
for label, total, active in SHAPES:
|
|
56
|
+
fit = next(((q, b) for q, b in QUANTS if size_gb(total, b) <= usable), None)
|
|
57
|
+
moe = active < total
|
|
58
|
+
if fit and fit[1] < LOW_BITS and moe and size_gb(total, 4.8) <= usable + 0.7 * ram_gb:
|
|
59
|
+
fit = None # a MoE keeps its quality better at Q4 with some experts in RAM than at 2 bits on the GPU
|
|
60
|
+
if fit:
|
|
61
|
+
q, b = fit
|
|
62
|
+
rows.append({"shape": label, "status": "low-bits" if b < LOW_BITS else "fits", "quant": q,
|
|
63
|
+
"gb": round(size_gb(total, b), 1), "tok_s": tok_s(total, active, b, bw)})
|
|
64
|
+
elif size_gb(total, 4.8) <= usable + 0.7 * ram_gb:
|
|
65
|
+
rows.append({"shape": label, "status": "offload-moe" if active < total else "offload-dense", "quant": "Q4",
|
|
66
|
+
"gb": round(size_gb(total, 4.8), 1), "tok_s": None})
|
|
67
|
+
else:
|
|
68
|
+
rows.append({"shape": label, "status": "too-big", "quant": None, "gb": round(size_gb(total, 4.8), 1),
|
|
69
|
+
"tok_s": None})
|
|
70
|
+
return rows
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def context_tokens(vram_gb: float, model: dict) -> int:
|
|
74
|
+
"""How many tokens of conversation/document fit next to the weights (KV cache at q8)."""
|
|
75
|
+
free = vram_gb - DESKTOP_GB - OVERHEAD_GB - model["gb"] - model.get("fixed_cache_gb", 0)
|
|
76
|
+
return max(0, min(model.get("max_ctx", 131072), int(free * 2**20 / model["kv_kb_per_token"])))
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: make-localllm-easier
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: One command runs the best local AI your PC can handle: measured model picks + a tuned llama.cpp (AMD, NVIDIA, Intel, Apple)
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/phonology024/make-localllm-easier
|
|
7
|
+
Keywords: llm,local-llm,llama.cpp,vulkan,amd,gguf,thai,offline-ai
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
12
|
+
Requires-Python: >=3.9
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# make-localllm-easier — run the best local LLM your GPU can handle, in one command
|
|
18
|
+
|
|
19
|
+
**`localllm` picks, downloads and runs the most accurate local AI model for your PC and your language, chosen from real
|
|
20
|
+
benchmark measurements, with llama.cpp tuned for AMD, NVIDIA, Intel and Apple GPUs.**
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
pip install make-localllm-easier
|
|
24
|
+
localllm
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
That's it. `localllm` checks your GPU and RAM, picks the most accurate model we have *measured* for your language that
|
|
28
|
+
fits your card, downloads llama.cpp and the model, starts it with settings profiled op by op, and opens the chat page.
|
|
29
|
+
You also get an OpenAI-compatible API at `http://127.0.0.1:8080/v1` for any app that speaks it. Offline, private, free.
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
localllm doctor # what this GPU is good for: model sizes, speed, how much text it can hold
|
|
33
|
+
localllm list # every model we have measured, with scores per language
|
|
34
|
+
localllm serve # API only, no browser
|
|
35
|
+
localllm eval # score any running server in English + your language
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## FAQ
|
|
39
|
+
|
|
40
|
+
**Which local LLM should I run on my GPU?** Run `localllm doctor`. It lists which model sizes fit your card (4B up to
|
|
41
|
+
120B MoE), at which quantization, how fast they should run, and the most accurate measured model for your language.
|
|
42
|
+
|
|
43
|
+
**Can a 16 GB GPU run a 27B model?** Yes. Qwen3.8-27B at ~3.5 bits (12.2 GB) runs at ~50 tok/s on an RX 9070 XT and
|
|
44
|
+
keeps 81.8% on English Global-MMLU-Lite. gemma-4-26B-A4B (13.3 GB) runs at ~69 tok/s with similar accuracy.
|
|
45
|
+
|
|
46
|
+
**Is a 2-bit quantized model good enough?** Usually not for non-English use: 2-bit costs 8-13 accuracy points, and
|
|
47
|
+
Hindi, Arabic and Thai lose the most (13 points).
|
|
48
|
+
|
|
49
|
+
**Why is llama.cpp slow on my AMD (or Intel) GPU on Windows?** If Resizable BAR is off, llama.cpp's Vulkan backend puts
|
|
50
|
+
buffers in a 256 MB host-visible heap backed by system RAM and decode drops up to 1.7x. `localllm` sets
|
|
51
|
+
`GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM=1` for you ([llama.cpp#27097](https://github.com/ggml-org/llama.cpp/issues/27097)).
|
|
52
|
+
|
|
53
|
+
**Does it work offline?** After the first download, yes. Nothing leaves your PC.
|
|
54
|
+
|
|
55
|
+
**Which languages are measured?** 23 languages on Global-MMLU-Lite, 44 countries' own exams on INCLUDE, plus Thai
|
|
56
|
+
(ThaiExam). `localllm eval --langs ...` measures any of them on your hardware.
|
|
57
|
+
|
|
58
|
+
## What `localllm doctor` tells you
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
GPU AMD Radeon RX 9070 XT 15.9 GB (640 GB/s) RAM 32 GB language: th
|
|
62
|
+
|
|
63
|
+
Model sizes for this PC (whole model on the GPU = fast):
|
|
64
|
+
[OK ] 4B Q8 4.2 GB ~90 tok/s (est.)
|
|
65
|
+
[OK ] 8B Q8 8.5 GB ~45 tok/s (est.)
|
|
66
|
+
[OK ] 14B Q6 11.5 GB ~33 tok/s (est.)
|
|
67
|
+
[OK ] 24-32B Q3 13.2 GB ~29 tok/s (est.)
|
|
68
|
+
[SLOW] 30B MoE (3B active) Q4 18.0 GB with experts in RAM - works, ~10-25 tok/s
|
|
69
|
+
[NO ] 70B needs ~42.0 GB - too big for this PC
|
|
70
|
+
|
|
71
|
+
Best measured model for you: gemma4-26b-a4b-qat (MoE with ~4B active params: fastest)
|
|
72
|
+
What it can do here:
|
|
73
|
+
TH real local school/licence exams 65.7% correct <- your language
|
|
74
|
+
holds ~78k tokens at once (~130 pages of text) next to the model
|
|
75
|
+
answers at ~69 tok/s
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Speeds marked *est.* come from your card's memory bandwidth, calibrated on measured runs. Everything else is measured.
|
|
79
|
+
|
|
80
|
+
## Measured results (RX 9070 XT 16 GB, Windows 11, llama.cpp Vulkan)
|
|
81
|
+
|
|
82
|
+
Accuracy (%) on multiple-choice exams, zero-shot. **global** = Global-MMLU-Lite: the same 400 questions translated, so
|
|
83
|
+
languages compare like for like. **regional** = INCLUDE: real exams written in each country (ThaiExam for Thai).
|
|
84
|
+
|
|
85
|
+
| | Qwen3.8-27B Q3 (12.2 GB) | gemma-4-26B-A4B QAT Q4 (13.3 GB) | Qwen3.8-27B 2-bit (7.8 GB) |
|
|
86
|
+
|---|---|---|---|
|
|
87
|
+
| English | 81.8 | 82.2 | 74.2 |
|
|
88
|
+
| Chinese | 76.2 / 74.7 | 73.5 / 66.5 | 67.8 / 67.8 |
|
|
89
|
+
| Spanish | 80.2 / 76.8 | 74.5 / 75.2 | 70.8 / 69.2 |
|
|
90
|
+
| Japanese | 73.5 / 87.6 | 74.5 / 81.9 | 65.8 / 77.9 |
|
|
91
|
+
| Arabic | 70.8 / 71.2 | 71.5 / 73.6 | 60.8 / 57.2 |
|
|
92
|
+
| Hindi | 69.0 / 74.3 | 69.5 / 71.0 | 56.2 / 55.5 |
|
|
93
|
+
| Thai | – / 67.1 | – / 65.7 | – / 54.2 |
|
|
94
|
+
| **decode speed** | **50 tok/s** (MTP) | **69 tok/s** | 40 tok/s |
|
|
95
|
+
|
|
96
|
+
Cells are global / regional. Margins are about ±4 (global) and ±5 (regional) points at 95%, so `localllm` treats gaps
|
|
97
|
+
under 2 points as a tie and picks the faster model.
|
|
98
|
+
|
|
99
|
+
## Findings worth knowing
|
|
100
|
+
|
|
101
|
+
1. **2-bit costs 8-13 points, and lower-resource languages pay the most.** Hindi, Arabic and Thai lose 13; English,
|
|
102
|
+
Chinese and Spanish about 8-9. A 177B MoE squeezed to 1.6 bits scored *below* a 27B at 3 bits.
|
|
103
|
+
2. **Calibrating the quantization on your language doesn't help at ~3.5 bits.** A Thai-text importance matrix scored the
|
|
104
|
+
same as the stock one in Thai, English and Chinese (64.8 vs 64.6 Thai). At this level the number of bits matters,
|
|
105
|
+
the calibration text doesn't.
|
|
106
|
+
3. **AMD/Intel cards without Resizable BAR lose up to 1.7x decode speed** in llama.cpp's Vulkan backend. Hybrid DeltaNet
|
|
107
|
+
models (Qwen3.5/3.8) suffer most: they rewrite a 3 MB state per layer per token.
|
|
108
|
+
4. **Qwen3.8 GGUFs ship a multi-token-prediction head.** Drafting 2 tokens with it adds ~40% decode speed for free;
|
|
109
|
+
drafting 3 is slower.
|
|
110
|
+
5. **The first run of a new llama.cpp build is slow** while the GPU driver compiles its shaders once (~15 s).
|
|
111
|
+
|
|
112
|
+
## How the benchmark works
|
|
113
|
+
|
|
114
|
+
`localllm eval` asks each question with thinking off and reads the log-probability of every answer letter from the first
|
|
115
|
+
generated token, then takes the most likely one. It's prompt processing only, so a language takes a few minutes, and the
|
|
116
|
+
result is deterministic. Data is downloaded at eval time from the original Apache-2.0 datasets
|
|
117
|
+
([Global-MMLU-Lite](https://huggingface.co/datasets/CohereLabs/Global-MMLU-Lite),
|
|
118
|
+
[INCLUDE](https://huggingface.co/datasets/CohereLabs/include-lite-44),
|
|
119
|
+
[ThaiExam](https://huggingface.co/datasets/typhoon-ai/thai_exam)) and never redistributed. It measures knowledge and
|
|
120
|
+
reasoning in multiple choice, not writing quality.
|
|
121
|
+
|
|
122
|
+
## Contributing
|
|
123
|
+
|
|
124
|
+
The catalog only grows with measurements. Run `localllm eval --langs en,<yours>` on your GPU and open a PR with
|
|
125
|
+
`~/.localllm/results.json` and your GPU name. Other languages' local exams are very welcome. See [ROADMAP.md](ROADMAP.md)
|
|
126
|
+
for what's next: using less system RAM (0.2), working alongside cloud provider APIs (0.3), and a speed-only release (0.4).
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
MIT. Models keep their own licenses; benchmark data keeps its own (Apache-2.0).
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
localllm/__init__.py
|
|
5
|
+
localllm/__main__.py
|
|
6
|
+
localllm/bench.py
|
|
7
|
+
localllm/catalog.py
|
|
8
|
+
localllm/cli.py
|
|
9
|
+
localllm/runtime.py
|
|
10
|
+
localllm/sizing.py
|
|
11
|
+
make_localllm_easier.egg-info/PKG-INFO
|
|
12
|
+
make_localllm_easier.egg-info/SOURCES.txt
|
|
13
|
+
make_localllm_easier.egg-info/dependency_links.txt
|
|
14
|
+
make_localllm_easier.egg-info/entry_points.txt
|
|
15
|
+
make_localllm_easier.egg-info/top_level.txt
|
|
16
|
+
tests/test_core.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
localllm
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "make-localllm-easier"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "One command runs the best local AI your PC can handle: measured model picks + a tuned llama.cpp (AMD, NVIDIA, Intel, Apple)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
dependencies = []
|
|
13
|
+
keywords = ["llm", "local-llm", "llama.cpp", "vulkan", "amd", "gguf", "thai", "offline-ai"]
|
|
14
|
+
classifiers = ["Programming Language :: Python :: 3", "License :: OSI Approved :: MIT License", "Operating System :: OS Independent", "Topic :: Scientific/Engineering :: Artificial Intelligence"]
|
|
15
|
+
|
|
16
|
+
[project.urls]
|
|
17
|
+
Homepage = "https://github.com/phonology024/make-localllm-easier"
|
|
18
|
+
|
|
19
|
+
[project.scripts]
|
|
20
|
+
localllm = "localllm.cli:main"
|
|
21
|
+
make-localllm-easier = "localllm.cli:main"
|
|
22
|
+
|
|
23
|
+
[tool.setuptools]
|
|
24
|
+
packages = ["localllm"]
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from localllm import catalog, runtime
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def test_pick_by_vram():
|
|
5
|
+
assert catalog.pick(8) is None
|
|
6
|
+
assert catalog.pick(12) == "qwen3.8-27b-iq2"
|
|
7
|
+
assert catalog.pick(16, "th") == "gemma4-26b-a4b-qat"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def test_server_args_mtp_and_device():
|
|
11
|
+
a = runtime.server_args(runtime.Path("m.gguf"), "Vulkan0", 8080, 4096, mtp=True)
|
|
12
|
+
assert a[a.index("-dev") + 1] == "Vulkan0" and "draft-mtp" in a and a[a.index("--spec-draft-n-max") + 1] == "2"
|
|
13
|
+
assert "draft-mtp" not in runtime.server_args(runtime.Path("m.gguf"), None, 8080, 4096, mtp=False)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_env_disables_host_visible_vidmem_unless_set(monkeypatch):
|
|
17
|
+
monkeypatch.delenv("GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM", raising=False)
|
|
18
|
+
assert runtime.server_env()["GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM"] == "1"
|
|
19
|
+
monkeypatch.setenv("GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM", "0")
|
|
20
|
+
assert runtime.server_env()["GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM"] == "0"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_best_device_skips_igpu():
|
|
24
|
+
devs = [{"id": "Vulkan0", "name": "AMD Radeon RX 9070 XT", "total_gb": 15.9},
|
|
25
|
+
{"id": "Vulkan1", "name": "Intel(R) UHD Graphics 770", "total_gb": 15.9}]
|
|
26
|
+
assert runtime.best_device(devs)["id"] == "Vulkan0"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_bench_coverage():
|
|
30
|
+
from localllm import bench
|
|
31
|
+
assert bench.available("en") == ["global"]
|
|
32
|
+
assert bench.available("th") == ["regional"]
|
|
33
|
+
assert bench.available("ja") == ["global", "regional"]
|
|
34
|
+
assert bench.available("xx") == []
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_sizing_tiers_16gb():
|
|
38
|
+
from localllm import sizing
|
|
39
|
+
rows = {r["shape"]: r for r in sizing.tiers(15.9, 32, "AMD Radeon RX 9070 XT")}
|
|
40
|
+
assert rows["8B"]["status"] == "fits" and rows["24-32B"]["quant"] == "Q3"
|
|
41
|
+
assert rows["30B MoE (3B active)"]["status"] == "offload-moe"
|
|
42
|
+
assert rows["70B"]["status"] == "too-big"
|
|
43
|
+
assert sizing.bandwidth("NVIDIA GeForce RTX 4060 Ti") == 288 and sizing.bandwidth("Mystery GPU") is None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_pick_prefers_accuracy_then_speed_on_ties():
|
|
47
|
+
assert catalog.pick(15.9, "zh") == "qwen3.8-27b-q3" # 5.5 points better in Chinese
|
|
48
|
+
assert catalog.pick(15.9, "en") == "gemma4-26b-a4b-qat" # tie on accuracy, faster
|