fituna 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fituna-0.1.0/LICENSE +21 -0
- fituna-0.1.0/PKG-INFO +319 -0
- fituna-0.1.0/README.md +289 -0
- fituna-0.1.0/fituna/__init__.py +30 -0
- fituna-0.1.0/fituna/__main__.py +9 -0
- fituna-0.1.0/fituna/bench.py +216 -0
- fituna-0.1.0/fituna/binaries.py +263 -0
- fituna-0.1.0/fituna/cache.py +321 -0
- fituna-0.1.0/fituna/cli.py +683 -0
- fituna-0.1.0/fituna/config.py +262 -0
- fituna-0.1.0/fituna/corpus.py +420 -0
- fituna-0.1.0/fituna/doctor.py +452 -0
- fituna-0.1.0/fituna/errors.py +87 -0
- fituna-0.1.0/fituna/hardware.py +383 -0
- fituna-0.1.0/fituna/mcp_server.py +305 -0
- fituna-0.1.0/fituna/model_info.py +395 -0
- fituna-0.1.0/fituna/py.typed +0 -0
- fituna-0.1.0/fituna/quality.py +185 -0
- fituna-0.1.0/fituna/quantize.py +209 -0
- fituna-0.1.0/fituna/quickstart.py +1132 -0
- fituna-0.1.0/fituna/report.py +413 -0
- fituna-0.1.0/fituna/search.py +592 -0
- fituna-0.1.0/fituna.egg-info/PKG-INFO +319 -0
- fituna-0.1.0/fituna.egg-info/SOURCES.txt +37 -0
- fituna-0.1.0/fituna.egg-info/dependency_links.txt +1 -0
- fituna-0.1.0/fituna.egg-info/entry_points.txt +3 -0
- fituna-0.1.0/fituna.egg-info/requires.txt +3 -0
- fituna-0.1.0/fituna.egg-info/top_level.txt +1 -0
- fituna-0.1.0/pyproject.toml +51 -0
- fituna-0.1.0/setup.cfg +4 -0
- fituna-0.1.0/tests/test_cache.py +343 -0
- fituna-0.1.0/tests/test_cli.py +254 -0
- fituna-0.1.0/tests/test_config.py +301 -0
- fituna-0.1.0/tests/test_corpus.py +534 -0
- fituna-0.1.0/tests/test_doctor.py +575 -0
- fituna-0.1.0/tests/test_hardware.py +370 -0
- fituna-0.1.0/tests/test_quickstart.py +778 -0
- fituna-0.1.0/tests/test_report.py +304 -0
- fituna-0.1.0/tests/test_search.py +443 -0
fituna-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 FiTuna contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
fituna-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fituna
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Hardware-aware auto-tuner that finds the smallest llama.cpp GGUF quantization + runtime config meeting a target throughput and quality-loss budget.
|
|
5
|
+
Author: FiTuna contributors
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/leeyunseokarchive/fituna
|
|
8
|
+
Project-URL: Repository, https://github.com/leeyunseokarchive/fituna
|
|
9
|
+
Project-URL: Issues, https://github.com/leeyunseokarchive/fituna/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/leeyunseokarchive/fituna/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: Documentation, https://github.com/leeyunseokarchive/fituna/blob/main/docs/ARCHITECTURE.md
|
|
12
|
+
Keywords: llama.cpp,gguf,quantization,llm,inference,benchmarking
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Software Development :: Build Tools
|
|
21
|
+
Classifier: Topic :: System :: Hardware
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Environment :: Console
|
|
24
|
+
Requires-Python: >=3.11
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest; extra == "dev"
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
<div align="center">
|
|
32
|
+
|
|
33
|
+
**English** | [νκ΅μ΄](README.ko.md)
|
|
34
|
+
|
|
35
|
+
# π― FiTuna
|
|
36
|
+
|
|
37
|
+
**Stop guessing your llama.cpp config. Measure it.**
|
|
38
|
+
|
|
39
|
+
Hardware-benchmark-driven auto-tuning for local LLMs β give it a model, a
|
|
40
|
+
target speed and a quality budget; get back the smallest llama.cpp config that
|
|
41
|
+
actually hits those numbers on **your** machine.
|
|
42
|
+
|
|
43
|
+
**API subscriptions add up. Going local means guessing which model your
|
|
44
|
+
machine can actually run.** Don't guess β measure it, and run your own.
|
|
45
|
+
|
|
46
|
+
[](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml)
|
|
47
|
+
[](LICENSE)
|
|
48
|
+
[](https://www.python.org/downloads/)
|
|
49
|
+
[](docs/SBOM.md)
|
|
50
|
+
[](CONTRIBUTING.md)
|
|
51
|
+
|
|
52
|
+
**μ¬μ¬μμ Β· κ²μ¦κΈ°κ΄μ© νκ΅μ΄ μ¬ν κ°μ΄λ β [REVIEWERS.md](REVIEWERS.md)** *(Korean reproduction guide for competition judges & verification agency)*
|
|
53
|
+
|
|
54
|
+
</div>
|
|
55
|
+
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
$ fituna run --model Qwen3-4B-Instruct-2507-F16.gguf \
|
|
60
|
+
--target-tps 30 --max-quality-loss 5 --ctx 4096 --wikitext wiki.txt --out ./out
|
|
61
|
+
|
|
62
|
+
[Q6_K] full-offload 28.48 tok/s < target 30.00, skipping (early-exit B)
|
|
63
|
+
[Q8_0] full-offload 24.22 tok/s < target 30.00, skipping (early-exit B)
|
|
64
|
+
[Q5_K_M] full-offload 29.59 tok/s < target 30.00, skipping (early-exit B)
|
|
65
|
+
[Q4_K_M] found ngl=33 meeting target -- done
|
|
66
|
+
|
|
67
|
+
FiTuna result: MEETS TARGET
|
|
68
|
+
quant : Q4_K_M ngl : 33 ctx : 4096
|
|
69
|
+
gen tok/s : 30.81 quality loss : 1.73%
|
|
70
|
+
|
|
71
|
+
artifact: out/Qwen3-4B-Instruct-2507-...-Q4_K_M.gguf (2.3 GB -- already produced during the search)
|
|
72
|
+
|
|
73
|
+
1) local API server (OpenAI-compatible):
|
|
74
|
+
llama-server -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096 --port 8080
|
|
75
|
+
2) import into Ollama: re-run with --export-ollama to write a Modelfile beside the artifact
|
|
76
|
+
3) terminal chat (interactive check):
|
|
77
|
+
llama-cli -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
*(Output formatting above is reconstructed against the current version; the
|
|
81
|
+
numbers are the Run 2 measurements.)*
|
|
82
|
+
|
|
83
|
+
A real run on an Apple M3 Pro. The "obviously best" Q8_0 **failed** the speed
|
|
84
|
+
target, Q5_K_M missed by **0.41 tok/s**, and the answer wasn't a quant alone β
|
|
85
|
+
it was a quant *plus* the minimal GPU offload (`-ngl 33`, not the full 36).
|
|
86
|
+
None of that is predictable from a spec sheet ([full logs](docs/RESULTS.md)).
|
|
87
|
+
|
|
88
|
+
## Install
|
|
89
|
+
|
|
90
|
+
Not on PyPI yet β install from git, into a virtualenv built with Python 3.11+
|
|
91
|
+
(macOS's system `python3` is 3.9):
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
git clone https://github.com/leeyunseokarchive/fituna
|
|
95
|
+
python3.13 -m venv .venv && source .venv/bin/activate
|
|
96
|
+
pip install -e fituna
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
You also need llama.cpp, which provides the quantize/bench/perplexity engines
|
|
100
|
+
FiTuna orchestrates:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
brew install llama.cpp # macOS/Linux Homebrew β ships all needed binaries
|
|
104
|
+
# or build from source (any platform):
|
|
105
|
+
git clone https://github.com/ggml-org/llama.cpp
|
|
106
|
+
cmake -S llama.cpp -B llama.cpp/build && cmake --build llama.cpp/build --config Release
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Quickstart
|
|
110
|
+
|
|
111
|
+
Three ways in: a person runs the wizard; a script or CI job calls
|
|
112
|
+
`fituna run --json`; an AI agent talks to
|
|
113
|
+
[`fituna-mcp`](#mcp-server--measured-answers-for-ai-agents).
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
fituna quickstart
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Six steps β environment check, targets, license requirements, model, quality
|
|
120
|
+
corpus, search β and it prints the assembled `fituna run ...` command
|
|
121
|
+
**before** executing it, so the next run is a one-liner you already have. It
|
|
122
|
+
needs a terminal; in CI or a pipe use `fituna run` directly, since every
|
|
123
|
+
search parameter maps to a public `run` flag (proven by an argv-equality
|
|
124
|
+
test). Model download (a curated shortlist) and HuggingFace search are
|
|
125
|
+
wizard-only conveniences; `run --model` expects a `.gguf` on disk.
|
|
126
|
+
|
|
127
|
+
It never predicts throughput: memory fit is arithmetic (published file size vs
|
|
128
|
+
detected VRAM/RAM, assumed margin stated), speed is measured, and curated or
|
|
129
|
+
HuggingFace-search candidates show their license β local-scan and manual-path
|
|
130
|
+
options cannot, since a `.gguf` carries no license metadata. Any
|
|
131
|
+
[`docs/RESULTS.md`](docs/RESULTS.md) figure it cites is a record of what was
|
|
132
|
+
measured on named hardware, never a prediction.
|
|
133
|
+
|
|
134
|
+
### The script path (what the wizard assembles for you)
|
|
135
|
+
|
|
136
|
+
Quality loss is perplexity increase over a plain-text corpus, so it is only
|
|
137
|
+
meaningful on text resembling your workload. Any UTF-8 file works
|
|
138
|
+
(`--quality-corpus`); both presets are one command away:
|
|
139
|
+
|
|
140
|
+
```bash
|
|
141
|
+
fituna fetch-corpus --lang en --out wikitext-2-raw-test.txt # wikitext-2 test split
|
|
142
|
+
fituna fetch-corpus --lang ko --out kowiki-corpus.txt --rows 500 # Korean Wikipedia
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Measure the language you'll run: the same quant can measure 2β3Γ different
|
|
146
|
+
loss on the two corpora, and in Run 3 that was enough to change the verdict
|
|
147
|
+
the tool returned ([measurement and
|
|
148
|
+
caveats](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
|
|
149
|
+
Both presets are CC BY-SA 3.0 and `fetch-corpus` prints the license notice
|
|
150
|
+
and source URL when it finishes; `--dataset/--config/--split` override the preset
|
|
151
|
+
([provenance and licensing](docs/OPEN_SOURCE_USAGE.md)).
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
fituna doctor # confirm the environment is ready
|
|
155
|
+
fituna detect-hw # see what FiTuna detects
|
|
156
|
+
fituna run --model your-model-F16.gguf \
|
|
157
|
+
--target-tps 30 --max-quality-loss 5 \
|
|
158
|
+
--ctx 4096 --wikitext wikitext-2-raw-test.txt --out ./out --resume
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Pass an F16/BF16 `.gguf` directly (many models publish one), or an HF-format
|
|
162
|
+
directory if `convert_hf_to_gguf.py` is available (source checkout +
|
|
163
|
+
`pip install torch transformers`; package-manager builds don't ship it).
|
|
164
|
+
|
|
165
|
+
> **Disk usage:** the search quantizes every candidate reaching the quality
|
|
166
|
+
> stage β ~12 GB for four candidates of a 4B model. Files are reused across
|
|
167
|
+
> runs; narrow `--quant` to bound this.
|
|
168
|
+
|
|
169
|
+
## Why
|
|
170
|
+
|
|
171
|
+
Running a local LLM means picking a quantization level (Q2βQ8), a GPU offload
|
|
172
|
+
layer count (`-ngl`) and a context length β a search space people navigate
|
|
173
|
+
today by trial and error:
|
|
174
|
+
|
|
175
|
+
- **Ollama / LM Studio** apply fixed per-model presets; a request for finer
|
|
176
|
+
quantization control was [closed as not planned](https://github.com/ollama/ollama/issues/14674).
|
|
177
|
+
- **NVIDIA Model Optimizer**'s AutoQuantize is CUDA-only.
|
|
178
|
+
- **VRAM calculators & chatbot advice** estimate from specs β and specs don't
|
|
179
|
+
know your thermals, memory bandwidth, or llama.cpp build flags.
|
|
180
|
+
|
|
181
|
+
FiTuna replaces the guesswork with a measured search over the llama.cpp
|
|
182
|
+
binaries you already have β verified on your hardware, reproducible from cache.
|
|
183
|
+
|
|
184
|
+
## Features
|
|
185
|
+
|
|
186
|
+
- π **Target-driven search** β in: model + target tok/s + max quality loss %.
|
|
187
|
+
Out: quant Γ `-ngl` Γ ctx config + a ready-to-run command.
|
|
188
|
+
- π **Measured, not assumed** β candidates walked in *measured* perplexity
|
|
189
|
+
order (in our runs Q6_K beat Q8_0 β [data](docs/RESULTS.md)), with a binary
|
|
190
|
+
search for the minimal GPU offload.
|
|
191
|
+
- β‘ **Aggressive early exits** β quality-gate failures and hopeless quants are
|
|
192
|
+
skipped without wasting benches; a bench that can't finish in time counts as
|
|
193
|
+
"too slow", not a crash.
|
|
194
|
+
- ποΈ **Reproducible cache** β sqlite3, keyed by model fingerprint Γ hardware Γ
|
|
195
|
+
llama.cpp build version; `--resume` re-answers in <1s and survives
|
|
196
|
+
interruptions.
|
|
197
|
+
- π₯οΈ **Hardware auto-detection** β NVIDIA (`nvidia-smi`), AMD (`rocm-smi`),
|
|
198
|
+
Apple Silicon unified memory (`system_profiler`), with manual override.
|
|
199
|
+
- πͺΆ **Zero runtime dependencies** β pure Python 3.11+ stdlib.
|
|
200
|
+
|
|
201
|
+
## Measured results
|
|
202
|
+
|
|
203
|
+
| Model | Target | What the "obvious" pick did | What FiTuna found |
|
|
204
|
+
|---|---|---|---|
|
|
205
|
+
| Qwen3-4B-Instruct (Apache 2.0) | 30 tok/s, β€5% loss | Q8_0: 24.22 tok/s β (and measured *worse* quality than Q6_K) | **Q4_K_M @ ngl=33 β 30.81 tok/s, 1.73% loss** β
|
|
|
206
|
+
| SmolLM2-135M (Apache 2.0) | 240 tok/s, β€5% loss | Q8_0: 205.91 tok/s β | **Q6_K β 249.50 tok/s, 0.53% loss** β
(and Q4_K_M measured *slower* than Q6_K) |
|
|
207
|
+
| Midm-2.0-Mini-Instruct, Korean (MIT) | 40 tok/s, β€5% loss | Q8_0: 34.26 tok/s β | **Q4_K_M @ ngl=48 β 44.62 tok/s, 2.58% loss** β
(the two corpora report different mid-table orders, but the per-chunk trace shows that reorder is [not something we could establish](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)) |
|
|
208
|
+
|
|
209
|
+
Apple M3 Pro, llama.cpp build 9960. Full logs, timings and run-to-run variance
|
|
210
|
+
analysis (including a thermal-throttle outlier we caught and documented):
|
|
211
|
+
**[docs/RESULTS.md](docs/RESULTS.md)** Β· Scenarios:
|
|
212
|
+
[docs/USE_CASES.md](docs/USE_CASES.md) Β· Reproduce on NVIDIA/Linux with the
|
|
213
|
+
one-click Colab notebook (free T4 tier):
|
|
214
|
+
[notebooks/colab_nvidia_verification.ipynb](notebooks/colab_nvidia_verification.ipynb)
|
|
215
|
+
[](https://colab.research.google.com/github/leeyunseokarchive/fituna/blob/main/notebooks/colab_nvidia_verification.ipynb)
|
|
216
|
+
|
|
217
|
+
## How it works
|
|
218
|
+
|
|
219
|
+
**Stage 1** measures perplexity loss for *every* candidate, because **Stage 2**
|
|
220
|
+
walks them in *measured* quality order and you can't sort by a number you
|
|
221
|
+
haven't measured. Stage 2 early-exits hard: a quant missing the target at full
|
|
222
|
+
offload is dropped without further benches, and the first quant that passes
|
|
223
|
+
wins. Results cache to sqlite3 keyed by model fingerprint, hardware profile
|
|
224
|
+
**and llama.cpp build version**, so `--resume` never serves numbers from a
|
|
225
|
+
different backend build. Diagrams, module map and full algorithm:
|
|
226
|
+
[docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) Β· Contract:
|
|
227
|
+
[`fituna/config.py`](fituna/config.py)
|
|
228
|
+
|
|
229
|
+
## Use as a library
|
|
230
|
+
|
|
231
|
+
Zero runtime dependencies means the modules import directly:
|
|
232
|
+
|
|
233
|
+
```python
|
|
234
|
+
from fituna.hardware import detect_hardware
|
|
235
|
+
|
|
236
|
+
hw = detect_hardware()
|
|
237
|
+
print(f"{hw.gpu_vendor.value}: {hw.gpu_name}, {hw.vram_mb} MB VRAM, {hw.ram_mb} MB RAM")
|
|
238
|
+
# apple: Apple M3 Pro, 18432 MB VRAM, 18432 MB RAM
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
(Real output from a `python3.13 -c` run on the same M3 Pro as above.) Driving
|
|
242
|
+
the search programmatically means calling `fituna.search.search()`, which also
|
|
243
|
+
needs a `ModelInfo`, resolved `BinaryPaths`, a work directory and a corpus
|
|
244
|
+
path β exactly what `fituna run`/`quickstart` assemble for you
|
|
245
|
+
([`search.py`](fituna/search.py), [`config.py`](fituna/config.py)).
|
|
246
|
+
|
|
247
|
+
## MCP server β measured answers for AI agents
|
|
248
|
+
|
|
249
|
+
Ask a chatbot "which local model config fits my machine?" and it guesses from
|
|
250
|
+
specs. Point it at FiTuna's MCP server and it *measures*:
|
|
251
|
+
|
|
252
|
+
```bash
|
|
253
|
+
claude mcp add fituna -- fituna-mcp # or any MCP client, stdio transport
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
| Tool | What it does |
|
|
257
|
+
|---|---|
|
|
258
|
+
| `fituna_detect_hardware` | GPU vendor/name, VRAM, CPU cores, RAM as FiTuna sees them |
|
|
259
|
+
| `fituna_recommend` | Runs the measured search for a target spec; returns the winning config, measured tok/s, measured quality loss, and a ready-to-run command. Slow once, ~1 s on repeat (cache). |
|
|
260
|
+
|
|
261
|
+
Stdlib-only like the rest of FiTuna β MCP stdio is newline-delimited JSON-RPC
|
|
262
|
+
2.0, no SDK required ([`fituna/mcp_server.py`](fituna/mcp_server.py)).
|
|
263
|
+
|
|
264
|
+
## Scope
|
|
265
|
+
|
|
266
|
+
FiTuna recommends; it doesn't execute or serve. The output is the quantized
|
|
267
|
+
`.gguf` plus `llama-server` / `llama-cli` commands you copy and run (and, with
|
|
268
|
+
`--export-ollama`, an Ollama `Modelfile` beside it) β FiTuna launches none of
|
|
269
|
+
them. That's a deliberate boundary: serving inference is llama.cpp's job, and
|
|
270
|
+
duplicating it would add no differentiated value
|
|
271
|
+
([rationale](docs/ARCHITECTURE.md#why-this-shape)). The two extensions that
|
|
272
|
+
stay inside it β `--launch` and an LM Studio preset export β are tracked in
|
|
273
|
+
[#19](https://github.com/leeyunseokarchive/fituna/issues/19).
|
|
274
|
+
|
|
275
|
+
## Known limitations
|
|
276
|
+
|
|
277
|
+
- **Single GPU only** β first GPU reported by `nvidia-smi`/`rocm-smi`; no
|
|
278
|
+
`--tensor-split` ([#11](https://github.com/leeyunseokarchive/fituna/issues/11),
|
|
279
|
+
help wanted: we have no multi-GPU machine to measure on).
|
|
280
|
+
- **Windows AMD auto-detection** β `rocm-smi` has no mainstream Windows
|
|
281
|
+
distribution; use `--gpu amd --vram-mb <N>`.
|
|
282
|
+
- **Quality = perplexity on a corpus you choose** β a proxy, not a guarantee
|
|
283
|
+
of domain quality. Gate on text resembling your workload
|
|
284
|
+
(`--quality-corpus`; [measured EN-vs-KO
|
|
285
|
+
comparison](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
|
|
286
|
+
- **The quality verdict depends on `--ppl-chunks`** β loss is an estimate over
|
|
287
|
+
`chunks Γ 512` tokens whose absolute value grows with the chunk count, so
|
|
288
|
+
re-measure a candidate close to your budget before trusting the PASS
|
|
289
|
+
([measured effect](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)).
|
|
290
|
+
`quality.py` still parses the PPL and discards llama-perplexity's error bar
|
|
291
|
+
([#8](https://github.com/leeyunseokarchive/fituna/issues/8)) β which is how
|
|
292
|
+
Run 5 came to publish a claim it later had to withdraw.
|
|
293
|
+
- **Benchmarks are thermally sensitive** β verdicts within a few tok/s of the
|
|
294
|
+
target are marginal ([variance
|
|
295
|
+
analysis](docs/RESULTS.md#run-to-run-variance-measured-not-hidden)).
|
|
296
|
+
- **Real-hardware E2E covers macOS and Linux only** β Apple Silicon/Metal and
|
|
297
|
+
NVIDIA T4/CUDA. Windows paths are unit-tested and CI-run, but not yet
|
|
298
|
+
integration-run against real binaries
|
|
299
|
+
([#12](https://github.com/leeyunseokarchive/fituna/issues/12)).
|
|
300
|
+
|
|
301
|
+
## Contributing
|
|
302
|
+
|
|
303
|
+
Contributions welcome β the codebase is small, dependency-free and
|
|
304
|
+
contract-first (start at [`fituna/config.py`](fituna/config.py)); 246 unit
|
|
305
|
+
tests, per-module self-checks and a 3-OS Γ 2-Python CI matrix guard it.
|
|
306
|
+
The roadmap lives in the
|
|
307
|
+
[v0.2.0 milestone](https://github.com/leeyunseokarchive/fituna/milestone/1),
|
|
308
|
+
including [#10](https://github.com/leeyunseokarchive/fituna/issues/10) (parser
|
|
309
|
+
test coverage, good first issue). See [CONTRIBUTING.md](CONTRIBUTING.md) Β·
|
|
310
|
+
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) Β· [CHANGELOG.md](CHANGELOG.md) Β·
|
|
311
|
+
[SECURITY.md](SECURITY.md).
|
|
312
|
+
|
|
313
|
+
## License
|
|
314
|
+
|
|
315
|
+
[MIT](LICENSE) Β© FiTuna contributors. Third-party notices (llama.cpp and
|
|
316
|
+
subprocess-invoked tools): [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md) Β·
|
|
317
|
+
SBOM: [docs/SBOM.md](docs/SBOM.md) Β· Open-source usage:
|
|
318
|
+
[docs/OPEN_SOURCE_USAGE.md](docs/OPEN_SOURCE_USAGE.md) Β· AI-assisted
|
|
319
|
+
development disclosure: [docs/AI_MODEL_USAGE.md](docs/AI_MODEL_USAGE.md)
|
fituna-0.1.0/README.md
ADDED
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
**English** | [νκ΅μ΄](README.ko.md)
|
|
4
|
+
|
|
5
|
+
# π― FiTuna
|
|
6
|
+
|
|
7
|
+
**Stop guessing your llama.cpp config. Measure it.**
|
|
8
|
+
|
|
9
|
+
Hardware-benchmark-driven auto-tuning for local LLMs β give it a model, a
|
|
10
|
+
target speed and a quality budget; get back the smallest llama.cpp config that
|
|
11
|
+
actually hits those numbers on **your** machine.
|
|
12
|
+
|
|
13
|
+
**API subscriptions add up. Going local means guessing which model your
|
|
14
|
+
machine can actually run.** Don't guess β measure it, and run your own.
|
|
15
|
+
|
|
16
|
+
[](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml)
|
|
17
|
+
[](LICENSE)
|
|
18
|
+
[](https://www.python.org/downloads/)
|
|
19
|
+
[](docs/SBOM.md)
|
|
20
|
+
[](CONTRIBUTING.md)
|
|
21
|
+
|
|
22
|
+
**μ¬μ¬μμ Β· κ²μ¦κΈ°κ΄μ© νκ΅μ΄ μ¬ν κ°μ΄λ β [REVIEWERS.md](REVIEWERS.md)** *(Korean reproduction guide for competition judges & verification agency)*
|
|
23
|
+
|
|
24
|
+
</div>
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
$ fituna run --model Qwen3-4B-Instruct-2507-F16.gguf \
|
|
30
|
+
--target-tps 30 --max-quality-loss 5 --ctx 4096 --wikitext wiki.txt --out ./out
|
|
31
|
+
|
|
32
|
+
[Q6_K] full-offload 28.48 tok/s < target 30.00, skipping (early-exit B)
|
|
33
|
+
[Q8_0] full-offload 24.22 tok/s < target 30.00, skipping (early-exit B)
|
|
34
|
+
[Q5_K_M] full-offload 29.59 tok/s < target 30.00, skipping (early-exit B)
|
|
35
|
+
[Q4_K_M] found ngl=33 meeting target -- done
|
|
36
|
+
|
|
37
|
+
FiTuna result: MEETS TARGET
|
|
38
|
+
quant : Q4_K_M ngl : 33 ctx : 4096
|
|
39
|
+
gen tok/s : 30.81 quality loss : 1.73%
|
|
40
|
+
|
|
41
|
+
artifact: out/Qwen3-4B-Instruct-2507-...-Q4_K_M.gguf (2.3 GB -- already produced during the search)
|
|
42
|
+
|
|
43
|
+
1) local API server (OpenAI-compatible):
|
|
44
|
+
llama-server -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096 --port 8080
|
|
45
|
+
2) import into Ollama: re-run with --export-ollama to write a Modelfile beside the artifact
|
|
46
|
+
3) terminal chat (interactive check):
|
|
47
|
+
llama-cli -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
*(Output formatting above is reconstructed against the current version; the
|
|
51
|
+
numbers are the Run 2 measurements.)*
|
|
52
|
+
|
|
53
|
+
A real run on an Apple M3 Pro. The "obviously best" Q8_0 **failed** the speed
|
|
54
|
+
target, Q5_K_M missed by **0.41 tok/s**, and the answer wasn't a quant alone β
|
|
55
|
+
it was a quant *plus* the minimal GPU offload (`-ngl 33`, not the full 36).
|
|
56
|
+
None of that is predictable from a spec sheet ([full logs](docs/RESULTS.md)).
|
|
57
|
+
|
|
58
|
+
## Install
|
|
59
|
+
|
|
60
|
+
Not on PyPI yet β install from git, into a virtualenv built with Python 3.11+
|
|
61
|
+
(macOS's system `python3` is 3.9):
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
git clone https://github.com/leeyunseokarchive/fituna
|
|
65
|
+
python3.13 -m venv .venv && source .venv/bin/activate
|
|
66
|
+
pip install -e fituna
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
You also need llama.cpp, which provides the quantize/bench/perplexity engines
|
|
70
|
+
FiTuna orchestrates:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
brew install llama.cpp # macOS/Linux Homebrew β ships all needed binaries
|
|
74
|
+
# or build from source (any platform):
|
|
75
|
+
git clone https://github.com/ggml-org/llama.cpp
|
|
76
|
+
cmake -S llama.cpp -B llama.cpp/build && cmake --build llama.cpp/build --config Release
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
## Quickstart
|
|
80
|
+
|
|
81
|
+
Three ways in: a person runs the wizard; a script or CI job calls
|
|
82
|
+
`fituna run --json`; an AI agent talks to
|
|
83
|
+
[`fituna-mcp`](#mcp-server--measured-answers-for-ai-agents).
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
fituna quickstart
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Six steps β environment check, targets, license requirements, model, quality
|
|
90
|
+
corpus, search β and it prints the assembled `fituna run ...` command
|
|
91
|
+
**before** executing it, so the next run is a one-liner you already have. It
|
|
92
|
+
needs a terminal; in CI or a pipe use `fituna run` directly, since every
|
|
93
|
+
search parameter maps to a public `run` flag (proven by an argv-equality
|
|
94
|
+
test). Model download (a curated shortlist) and HuggingFace search are
|
|
95
|
+
wizard-only conveniences; `run --model` expects a `.gguf` on disk.
|
|
96
|
+
|
|
97
|
+
It never predicts throughput: memory fit is arithmetic (published file size vs
|
|
98
|
+
detected VRAM/RAM, assumed margin stated), speed is measured, and curated or
|
|
99
|
+
HuggingFace-search candidates show their license β local-scan and manual-path
|
|
100
|
+
options cannot, since a `.gguf` carries no license metadata. Any
|
|
101
|
+
[`docs/RESULTS.md`](docs/RESULTS.md) figure it cites is a record of what was
|
|
102
|
+
measured on named hardware, never a prediction.
|
|
103
|
+
|
|
104
|
+
### The script path (what the wizard assembles for you)
|
|
105
|
+
|
|
106
|
+
Quality loss is perplexity increase over a plain-text corpus, so it is only
|
|
107
|
+
meaningful on text resembling your workload. Any UTF-8 file works
|
|
108
|
+
(`--quality-corpus`); both presets are one command away:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
fituna fetch-corpus --lang en --out wikitext-2-raw-test.txt # wikitext-2 test split
|
|
112
|
+
fituna fetch-corpus --lang ko --out kowiki-corpus.txt --rows 500 # Korean Wikipedia
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Measure the language you'll run: the same quant can measure 2β3Γ different
|
|
116
|
+
loss on the two corpora, and in Run 3 that was enough to change the verdict
|
|
117
|
+
the tool returned ([measurement and
|
|
118
|
+
caveats](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
|
|
119
|
+
Both presets are CC BY-SA 3.0 and `fetch-corpus` prints the license notice
|
|
120
|
+
and source URL when it finishes; `--dataset/--config/--split` override the preset
|
|
121
|
+
([provenance and licensing](docs/OPEN_SOURCE_USAGE.md)).
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
fituna doctor # confirm the environment is ready
|
|
125
|
+
fituna detect-hw # see what FiTuna detects
|
|
126
|
+
fituna run --model your-model-F16.gguf \
|
|
127
|
+
--target-tps 30 --max-quality-loss 5 \
|
|
128
|
+
--ctx 4096 --wikitext wikitext-2-raw-test.txt --out ./out --resume
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Pass an F16/BF16 `.gguf` directly (many models publish one), or an HF-format
|
|
132
|
+
directory if `convert_hf_to_gguf.py` is available (source checkout +
|
|
133
|
+
`pip install torch transformers`; package-manager builds don't ship it).
|
|
134
|
+
|
|
135
|
+
> **Disk usage:** the search quantizes every candidate reaching the quality
|
|
136
|
+
> stage β ~12 GB for four candidates of a 4B model. Files are reused across
|
|
137
|
+
> runs; narrow `--quant` to bound this.
|
|
138
|
+
|
|
139
|
+
## Why
|
|
140
|
+
|
|
141
|
+
Running a local LLM means picking a quantization level (Q2βQ8), a GPU offload
|
|
142
|
+
layer count (`-ngl`) and a context length β a search space people navigate
|
|
143
|
+
today by trial and error:
|
|
144
|
+
|
|
145
|
+
- **Ollama / LM Studio** apply fixed per-model presets; a request for finer
|
|
146
|
+
quantization control was [closed as not planned](https://github.com/ollama/ollama/issues/14674).
|
|
147
|
+
- **NVIDIA Model Optimizer**'s AutoQuantize is CUDA-only.
|
|
148
|
+
- **VRAM calculators & chatbot advice** estimate from specs β and specs don't
|
|
149
|
+
know your thermals, memory bandwidth, or llama.cpp build flags.
|
|
150
|
+
|
|
151
|
+
FiTuna replaces the guesswork with a measured search over the llama.cpp
|
|
152
|
+
binaries you already have β verified on your hardware, reproducible from cache.
|
|
153
|
+
|
|
154
|
+
## Features
|
|
155
|
+
|
|
156
|
+
- π **Target-driven search** β in: model + target tok/s + max quality loss %.
|
|
157
|
+
Out: quant Γ `-ngl` Γ ctx config + a ready-to-run command.
|
|
158
|
+
- π **Measured, not assumed** β candidates walked in *measured* perplexity
|
|
159
|
+
order (in our runs Q6_K beat Q8_0 β [data](docs/RESULTS.md)), with a binary
|
|
160
|
+
search for the minimal GPU offload.
|
|
161
|
+
- β‘ **Aggressive early exits** β quality-gate failures and hopeless quants are
|
|
162
|
+
skipped without wasting benches; a bench that can't finish in time counts as
|
|
163
|
+
"too slow", not a crash.
|
|
164
|
+
- ποΈ **Reproducible cache** β sqlite3, keyed by model fingerprint Γ hardware Γ
|
|
165
|
+
llama.cpp build version; `--resume` re-answers in <1s and survives
|
|
166
|
+
interruptions.
|
|
167
|
+
- π₯οΈ **Hardware auto-detection** β NVIDIA (`nvidia-smi`), AMD (`rocm-smi`),
|
|
168
|
+
Apple Silicon unified memory (`system_profiler`), with manual override.
|
|
169
|
+
- πͺΆ **Zero runtime dependencies** β pure Python 3.11+ stdlib.
|
|
170
|
+
|
|
171
|
+
## Measured results
|
|
172
|
+
|
|
173
|
+
| Model | Target | What the "obvious" pick did | What FiTuna found |
|
|
174
|
+
|---|---|---|---|
|
|
175
|
+
| Qwen3-4B-Instruct (Apache 2.0) | 30 tok/s, β€5% loss | Q8_0: 24.22 tok/s β (and measured *worse* quality than Q6_K) | **Q4_K_M @ ngl=33 β 30.81 tok/s, 1.73% loss** β
|
|
|
176
|
+
| SmolLM2-135M (Apache 2.0) | 240 tok/s, β€5% loss | Q8_0: 205.91 tok/s β | **Q6_K β 249.50 tok/s, 0.53% loss** β
(and Q4_K_M measured *slower* than Q6_K) |
|
|
177
|
+
| Midm-2.0-Mini-Instruct, Korean (MIT) | 40 tok/s, β€5% loss | Q8_0: 34.26 tok/s β | **Q4_K_M @ ngl=48 β 44.62 tok/s, 2.58% loss** β
(the two corpora report different mid-table orders, but the per-chunk trace shows that reorder is [not something we could establish](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)) |
|
|
178
|
+
|
|
179
|
+
Apple M3 Pro, llama.cpp build 9960. Full logs, timings and run-to-run variance
|
|
180
|
+
analysis (including a thermal-throttle outlier we caught and documented):
|
|
181
|
+
**[docs/RESULTS.md](docs/RESULTS.md)** Β· Scenarios:
|
|
182
|
+
[docs/USE_CASES.md](docs/USE_CASES.md) Β· Reproduce on NVIDIA/Linux with the
|
|
183
|
+
one-click Colab notebook (free T4 tier):
|
|
184
|
+
[notebooks/colab_nvidia_verification.ipynb](notebooks/colab_nvidia_verification.ipynb)
|
|
185
|
+
[](https://colab.research.google.com/github/leeyunseokarchive/fituna/blob/main/notebooks/colab_nvidia_verification.ipynb)
|
|
186
|
+
|
|
187
|
+
## How it works
|
|
188
|
+
|
|
189
|
+
**Stage 1** measures perplexity loss for *every* candidate, because **Stage 2**
|
|
190
|
+
walks them in *measured* quality order and you can't sort by a number you
|
|
191
|
+
haven't measured. Stage 2 early-exits hard: a quant missing the target at full
|
|
192
|
+
offload is dropped without further benches, and the first quant that passes
|
|
193
|
+
wins. Results cache to sqlite3 keyed by model fingerprint, hardware profile
|
|
194
|
+
**and llama.cpp build version**, so `--resume` never serves numbers from a
|
|
195
|
+
different backend build. Diagrams, module map and full algorithm:
|
|
196
|
+
[docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) Β· Contract:
|
|
197
|
+
[`fituna/config.py`](fituna/config.py)
|
|
198
|
+
|
|
199
|
+
## Use as a library
|
|
200
|
+
|
|
201
|
+
Zero runtime dependencies means the modules import directly:
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
from fituna.hardware import detect_hardware
|
|
205
|
+
|
|
206
|
+
hw = detect_hardware()
|
|
207
|
+
print(f"{hw.gpu_vendor.value}: {hw.gpu_name}, {hw.vram_mb} MB VRAM, {hw.ram_mb} MB RAM")
|
|
208
|
+
# apple: Apple M3 Pro, 18432 MB VRAM, 18432 MB RAM
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
(Real output from a `python3.13 -c` run on the same M3 Pro as above.) Driving
|
|
212
|
+
the search programmatically means calling `fituna.search.search()`, which also
|
|
213
|
+
needs a `ModelInfo`, resolved `BinaryPaths`, a work directory and a corpus
|
|
214
|
+
path β exactly what `fituna run`/`quickstart` assemble for you
|
|
215
|
+
([`search.py`](fituna/search.py), [`config.py`](fituna/config.py)).
|
|
216
|
+
|
|
217
|
+
## MCP server β measured answers for AI agents
|
|
218
|
+
|
|
219
|
+
Ask a chatbot "which local model config fits my machine?" and it guesses from
|
|
220
|
+
specs. Point it at FiTuna's MCP server and it *measures*:
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
claude mcp add fituna -- fituna-mcp # or any MCP client, stdio transport
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
| Tool | What it does |
|
|
227
|
+
|---|---|
|
|
228
|
+
| `fituna_detect_hardware` | GPU vendor/name, VRAM, CPU cores, RAM as FiTuna sees them |
|
|
229
|
+
| `fituna_recommend` | Runs the measured search for a target spec; returns the winning config, measured tok/s, measured quality loss, and a ready-to-run command. Slow once, ~1 s on repeat (cache). |
|
|
230
|
+
|
|
231
|
+
Stdlib-only like the rest of FiTuna β MCP stdio is newline-delimited JSON-RPC
|
|
232
|
+
2.0, no SDK required ([`fituna/mcp_server.py`](fituna/mcp_server.py)).
|
|
233
|
+
|
|
234
|
+
## Scope
|
|
235
|
+
|
|
236
|
+
FiTuna recommends; it doesn't execute or serve. The output is the quantized
|
|
237
|
+
`.gguf` plus `llama-server` / `llama-cli` commands you copy and run (and, with
|
|
238
|
+
`--export-ollama`, an Ollama `Modelfile` beside it) β FiTuna launches none of
|
|
239
|
+
them. That's a deliberate boundary: serving inference is llama.cpp's job, and
|
|
240
|
+
duplicating it would add no differentiated value
|
|
241
|
+
([rationale](docs/ARCHITECTURE.md#why-this-shape)). The two extensions that
|
|
242
|
+
stay inside it β `--launch` and an LM Studio preset export β are tracked in
|
|
243
|
+
[#19](https://github.com/leeyunseokarchive/fituna/issues/19).
|
|
244
|
+
|
|
245
|
+
## Known limitations
|
|
246
|
+
|
|
247
|
+
- **Single GPU only** β first GPU reported by `nvidia-smi`/`rocm-smi`; no
|
|
248
|
+
`--tensor-split` ([#11](https://github.com/leeyunseokarchive/fituna/issues/11),
|
|
249
|
+
help wanted: we have no multi-GPU machine to measure on).
|
|
250
|
+
- **Windows AMD auto-detection** β `rocm-smi` has no mainstream Windows
|
|
251
|
+
distribution; use `--gpu amd --vram-mb <N>`.
|
|
252
|
+
- **Quality = perplexity on a corpus you choose** β a proxy, not a guarantee
|
|
253
|
+
of domain quality. Gate on text resembling your workload
|
|
254
|
+
(`--quality-corpus`; [measured EN-vs-KO
|
|
255
|
+
comparison](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
|
|
256
|
+
- **The quality verdict depends on `--ppl-chunks`** β loss is an estimate over
|
|
257
|
+
`chunks Γ 512` tokens whose absolute value grows with the chunk count, so
|
|
258
|
+
re-measure a candidate close to your budget before trusting the PASS
|
|
259
|
+
([measured effect](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)).
|
|
260
|
+
`quality.py` still parses the PPL and discards llama-perplexity's error bar
|
|
261
|
+
([#8](https://github.com/leeyunseokarchive/fituna/issues/8)) β which is how
|
|
262
|
+
Run 5 came to publish a claim it later had to withdraw.
|
|
263
|
+
- **Benchmarks are thermally sensitive** β verdicts within a few tok/s of the
|
|
264
|
+
target are marginal ([variance
|
|
265
|
+
analysis](docs/RESULTS.md#run-to-run-variance-measured-not-hidden)).
|
|
266
|
+
- **Real-hardware E2E covers macOS and Linux only** β Apple Silicon/Metal and
|
|
267
|
+
NVIDIA T4/CUDA. Windows paths are unit-tested and CI-run, but not yet
|
|
268
|
+
integration-run against real binaries
|
|
269
|
+
([#12](https://github.com/leeyunseokarchive/fituna/issues/12)).
|
|
270
|
+
|
|
271
|
+
## Contributing
|
|
272
|
+
|
|
273
|
+
Contributions welcome β the codebase is small, dependency-free and
|
|
274
|
+
contract-first (start at [`fituna/config.py`](fituna/config.py)); 246 unit
|
|
275
|
+
tests, per-module self-checks and a 3-OS Γ 2-Python CI matrix guard it.
|
|
276
|
+
The roadmap lives in the
|
|
277
|
+
[v0.2.0 milestone](https://github.com/leeyunseokarchive/fituna/milestone/1),
|
|
278
|
+
including [#10](https://github.com/leeyunseokarchive/fituna/issues/10) (parser
|
|
279
|
+
test coverage, good first issue). See [CONTRIBUTING.md](CONTRIBUTING.md) Β·
|
|
280
|
+
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) Β· [CHANGELOG.md](CHANGELOG.md) Β·
|
|
281
|
+
[SECURITY.md](SECURITY.md).
|
|
282
|
+
|
|
283
|
+
## License
|
|
284
|
+
|
|
285
|
+
[MIT](LICENSE) Β© FiTuna contributors. Third-party notices (llama.cpp and
|
|
286
|
+
subprocess-invoked tools): [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md) Β·
|
|
287
|
+
SBOM: [docs/SBOM.md](docs/SBOM.md) Β· Open-source usage:
|
|
288
|
+
[docs/OPEN_SOURCE_USAGE.md](docs/OPEN_SOURCE_USAGE.md) Β· AI-assisted
|
|
289
|
+
development disclosure: [docs/AI_MODEL_USAGE.md](docs/AI_MODEL_USAGE.md)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""FiTuna: hardware-aware auto-tuner for llama.cpp GGUF quantization + runtime configs."""
|
|
3
|
+
|
|
4
|
+
__version__ = "0.1.0"
|
|
5
|
+
|
|
6
|
+
__all__ = ["__version__"]
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _self_check() -> None:
|
|
10
|
+
"""Guard against __version__ drifting from pyproject.toml's [project.version]."""
|
|
11
|
+
import tomllib
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
pyproject = Path(__file__).resolve().parent.parent / "pyproject.toml"
|
|
15
|
+
if not pyproject.exists():
|
|
16
|
+
return # installed outside the source tree (e.g. site-packages) β nothing to compare against
|
|
17
|
+
with pyproject.open("rb") as f:
|
|
18
|
+
data = tomllib.load(f)
|
|
19
|
+
pyproject_version = data["project"]["version"]
|
|
20
|
+
assert __version__ == pyproject_version, (
|
|
21
|
+
f"__version__ ({__version__}) drifted from pyproject.toml ({pyproject_version})"
|
|
22
|
+
)
|
|
23
|
+
assert isinstance(__version__, str) and __version__.count(".") == 2, (
|
|
24
|
+
f"__version__ ({__version__!r}) is not a MAJOR.MINOR.PATCH string"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
if __name__ == "__main__":
|
|
29
|
+
_self_check()
|
|
30
|
+
print("fituna.__init__ self-check passed:", __version__)
|