fituna 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. fituna-0.1.0/LICENSE +21 -0
  2. fituna-0.1.0/PKG-INFO +319 -0
  3. fituna-0.1.0/README.md +289 -0
  4. fituna-0.1.0/fituna/__init__.py +30 -0
  5. fituna-0.1.0/fituna/__main__.py +9 -0
  6. fituna-0.1.0/fituna/bench.py +216 -0
  7. fituna-0.1.0/fituna/binaries.py +263 -0
  8. fituna-0.1.0/fituna/cache.py +321 -0
  9. fituna-0.1.0/fituna/cli.py +683 -0
  10. fituna-0.1.0/fituna/config.py +262 -0
  11. fituna-0.1.0/fituna/corpus.py +420 -0
  12. fituna-0.1.0/fituna/doctor.py +452 -0
  13. fituna-0.1.0/fituna/errors.py +87 -0
  14. fituna-0.1.0/fituna/hardware.py +383 -0
  15. fituna-0.1.0/fituna/mcp_server.py +305 -0
  16. fituna-0.1.0/fituna/model_info.py +395 -0
  17. fituna-0.1.0/fituna/py.typed +0 -0
  18. fituna-0.1.0/fituna/quality.py +185 -0
  19. fituna-0.1.0/fituna/quantize.py +209 -0
  20. fituna-0.1.0/fituna/quickstart.py +1132 -0
  21. fituna-0.1.0/fituna/report.py +413 -0
  22. fituna-0.1.0/fituna/search.py +592 -0
  23. fituna-0.1.0/fituna.egg-info/PKG-INFO +319 -0
  24. fituna-0.1.0/fituna.egg-info/SOURCES.txt +37 -0
  25. fituna-0.1.0/fituna.egg-info/dependency_links.txt +1 -0
  26. fituna-0.1.0/fituna.egg-info/entry_points.txt +3 -0
  27. fituna-0.1.0/fituna.egg-info/requires.txt +3 -0
  28. fituna-0.1.0/fituna.egg-info/top_level.txt +1 -0
  29. fituna-0.1.0/pyproject.toml +51 -0
  30. fituna-0.1.0/setup.cfg +4 -0
  31. fituna-0.1.0/tests/test_cache.py +343 -0
  32. fituna-0.1.0/tests/test_cli.py +254 -0
  33. fituna-0.1.0/tests/test_config.py +301 -0
  34. fituna-0.1.0/tests/test_corpus.py +534 -0
  35. fituna-0.1.0/tests/test_doctor.py +575 -0
  36. fituna-0.1.0/tests/test_hardware.py +370 -0
  37. fituna-0.1.0/tests/test_quickstart.py +778 -0
  38. fituna-0.1.0/tests/test_report.py +304 -0
  39. fituna-0.1.0/tests/test_search.py +443 -0
fituna-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 FiTuna contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
fituna-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,319 @@
1
+ Metadata-Version: 2.4
2
+ Name: fituna
3
+ Version: 0.1.0
4
+ Summary: Hardware-aware auto-tuner that finds the smallest llama.cpp GGUF quantization + runtime config meeting a target throughput and quality-loss budget.
5
+ Author: FiTuna contributors
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/leeyunseokarchive/fituna
8
+ Project-URL: Repository, https://github.com/leeyunseokarchive/fituna
9
+ Project-URL: Issues, https://github.com/leeyunseokarchive/fituna/issues
10
+ Project-URL: Changelog, https://github.com/leeyunseokarchive/fituna/blob/main/CHANGELOG.md
11
+ Project-URL: Documentation, https://github.com/leeyunseokarchive/fituna/blob/main/docs/ARCHITECTURE.md
12
+ Keywords: llama.cpp,gguf,quantization,llm,inference,benchmarking
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Software Development :: Build Tools
21
+ Classifier: Topic :: System :: Hardware
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Environment :: Console
24
+ Requires-Python: >=3.11
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Provides-Extra: dev
28
+ Requires-Dist: pytest; extra == "dev"
29
+ Dynamic: license-file
30
+
31
+ <div align="center">
32
+
33
+ **English** | [ν•œκ΅­μ–΄](README.ko.md)
34
+
35
+ # 🎯 FiTuna
36
+
37
+ **Stop guessing your llama.cpp config. Measure it.**
38
+
39
+ Hardware-benchmark-driven auto-tuning for local LLMs β€” give it a model, a
40
+ target speed and a quality budget; get back the smallest llama.cpp config that
41
+ actually hits those numbers on **your** machine.
42
+
43
+ **API subscriptions add up. Going local means guessing which model your
44
+ machine can actually run.** Don't guess β€” measure it, and run your own.
45
+
46
+ [![CI](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml/badge.svg)](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml)
47
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
48
+ [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://www.python.org/downloads/)
49
+ [![Zero dependencies](https://img.shields.io/badge/runtime%20deps-0-brightgreen.svg)](docs/SBOM.md)
50
+ [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](CONTRIBUTING.md)
51
+
52
+ **μ‹¬μ‚¬μœ„μ› Β· κ²€μ¦κΈ°κ΄€μš© ν•œκ΅­μ–΄ μž¬ν˜„ κ°€μ΄λ“œ β†’ [REVIEWERS.md](REVIEWERS.md)** *(Korean reproduction guide for competition judges & verification agency)*
53
+
54
+ </div>
55
+
56
+ ---
57
+
58
+ ```bash
59
+ $ fituna run --model Qwen3-4B-Instruct-2507-F16.gguf \
60
+ --target-tps 30 --max-quality-loss 5 --ctx 4096 --wikitext wiki.txt --out ./out
61
+
62
+ [Q6_K] full-offload 28.48 tok/s < target 30.00, skipping (early-exit B)
63
+ [Q8_0] full-offload 24.22 tok/s < target 30.00, skipping (early-exit B)
64
+ [Q5_K_M] full-offload 29.59 tok/s < target 30.00, skipping (early-exit B)
65
+ [Q4_K_M] found ngl=33 meeting target -- done
66
+
67
+ FiTuna result: MEETS TARGET
68
+ quant : Q4_K_M ngl : 33 ctx : 4096
69
+ gen tok/s : 30.81 quality loss : 1.73%
70
+
71
+ artifact: out/Qwen3-4B-Instruct-2507-...-Q4_K_M.gguf (2.3 GB -- already produced during the search)
72
+
73
+ 1) local API server (OpenAI-compatible):
74
+ llama-server -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096 --port 8080
75
+ 2) import into Ollama: re-run with --export-ollama to write a Modelfile beside the artifact
76
+ 3) terminal chat (interactive check):
77
+ llama-cli -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096
78
+ ```
79
+
80
+ *(Output formatting above is reconstructed against the current version; the
81
+ numbers are the Run 2 measurements.)*
82
+
83
+ A real run on an Apple M3 Pro. The "obviously best" Q8_0 **failed** the speed
84
+ target, Q5_K_M missed by **0.41 tok/s**, and the answer wasn't a quant alone β€”
85
+ it was a quant *plus* the minimal GPU offload (`-ngl 33`, not the full 36).
86
+ None of that is predictable from a spec sheet ([full logs](docs/RESULTS.md)).
87
+
88
+ ## Install
89
+
90
+ Not on PyPI yet β€” install from git, into a virtualenv built with Python 3.11+
91
+ (macOS's system `python3` is 3.9):
92
+
93
+ ```bash
94
+ git clone https://github.com/leeyunseokarchive/fituna
95
+ python3.13 -m venv .venv && source .venv/bin/activate
96
+ pip install -e fituna
97
+ ```
98
+
99
+ You also need llama.cpp, which provides the quantize/bench/perplexity engines
100
+ FiTuna orchestrates:
101
+
102
+ ```bash
103
+ brew install llama.cpp # macOS/Linux Homebrew β€” ships all needed binaries
104
+ # or build from source (any platform):
105
+ git clone https://github.com/ggml-org/llama.cpp
106
+ cmake -S llama.cpp -B llama.cpp/build && cmake --build llama.cpp/build --config Release
107
+ ```
108
+
109
+ ## Quickstart
110
+
111
+ Three ways in: a person runs the wizard; a script or CI job calls
112
+ `fituna run --json`; an AI agent talks to
113
+ [`fituna-mcp`](#mcp-server--measured-answers-for-ai-agents).
114
+
115
+ ```bash
116
+ fituna quickstart
117
+ ```
118
+
119
+ Six steps β€” environment check, targets, license requirements, model, quality
120
+ corpus, search β€” and it prints the assembled `fituna run ...` command
121
+ **before** executing it, so the next run is a one-liner you already have. It
122
+ needs a terminal; in CI or a pipe use `fituna run` directly, since every
123
+ search parameter maps to a public `run` flag (proven by an argv-equality
124
+ test). Model download (a curated shortlist) and HuggingFace search are
125
+ wizard-only conveniences; `run --model` expects a `.gguf` on disk.
126
+
127
+ It never predicts throughput: memory fit is arithmetic (published file size vs
128
+ detected VRAM/RAM, assumed margin stated), speed is measured, and curated or
129
+ HuggingFace-search candidates show their license β€” local-scan and manual-path
130
+ options cannot, since a `.gguf` carries no license metadata. Any
131
+ [`docs/RESULTS.md`](docs/RESULTS.md) figure it cites is a record of what was
132
+ measured on named hardware, never a prediction.
133
+
134
+ ### The script path (what the wizard assembles for you)
135
+
136
+ Quality loss is perplexity increase over a plain-text corpus, so it is only
137
+ meaningful on text resembling your workload. Any UTF-8 file works
138
+ (`--quality-corpus`); both presets are one command away:
139
+
140
+ ```bash
141
+ fituna fetch-corpus --lang en --out wikitext-2-raw-test.txt # wikitext-2 test split
142
+ fituna fetch-corpus --lang ko --out kowiki-corpus.txt --rows 500 # Korean Wikipedia
143
+ ```
144
+
145
+ Measure the language you'll run: the same quant can measure 2–3Γ— different
146
+ loss on the two corpora, and in Run 3 that was enough to change the verdict
147
+ the tool returned ([measurement and
148
+ caveats](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
149
+ Both presets are CC BY-SA 3.0 and `fetch-corpus` prints the license notice
150
+ and source URL when it finishes; `--dataset/--config/--split` override the preset
151
+ ([provenance and licensing](docs/OPEN_SOURCE_USAGE.md)).
152
+
153
+ ```bash
154
+ fituna doctor # confirm the environment is ready
155
+ fituna detect-hw # see what FiTuna detects
156
+ fituna run --model your-model-F16.gguf \
157
+ --target-tps 30 --max-quality-loss 5 \
158
+ --ctx 4096 --wikitext wikitext-2-raw-test.txt --out ./out --resume
159
+ ```
160
+
161
+ Pass an F16/BF16 `.gguf` directly (many models publish one), or an HF-format
162
+ directory if `convert_hf_to_gguf.py` is available (source checkout +
163
+ `pip install torch transformers`; package-manager builds don't ship it).
164
+
165
+ > **Disk usage:** the search quantizes every candidate reaching the quality
166
+ > stage β€” ~12 GB for four candidates of a 4B model. Files are reused across
167
+ > runs; narrow `--quant` to bound this.
168
+
169
+ ## Why
170
+
171
+ Running a local LLM means picking a quantization level (Q2–Q8), a GPU offload
172
+ layer count (`-ngl`) and a context length β€” a search space people navigate
173
+ today by trial and error:
174
+
175
+ - **Ollama / LM Studio** apply fixed per-model presets; a request for finer
176
+ quantization control was [closed as not planned](https://github.com/ollama/ollama/issues/14674).
177
+ - **NVIDIA Model Optimizer**'s AutoQuantize is CUDA-only.
178
+ - **VRAM calculators & chatbot advice** estimate from specs β€” and specs don't
179
+ know your thermals, memory bandwidth, or llama.cpp build flags.
180
+
181
+ FiTuna replaces the guesswork with a measured search over the llama.cpp
182
+ binaries you already have β€” verified on your hardware, reproducible from cache.
183
+
184
+ ## Features
185
+
186
+ - πŸ” **Target-driven search** β€” in: model + target tok/s + max quality loss %.
187
+ Out: quant Γ— `-ngl` Γ— ctx config + a ready-to-run command.
188
+ - πŸ“ **Measured, not assumed** β€” candidates walked in *measured* perplexity
189
+ order (in our runs Q6_K beat Q8_0 β€” [data](docs/RESULTS.md)), with a binary
190
+ search for the minimal GPU offload.
191
+ - ⚑ **Aggressive early exits** β€” quality-gate failures and hopeless quants are
192
+ skipped without wasting benches; a bench that can't finish in time counts as
193
+ "too slow", not a crash.
194
+ - πŸ—ƒοΈ **Reproducible cache** β€” sqlite3, keyed by model fingerprint Γ— hardware Γ—
195
+ llama.cpp build version; `--resume` re-answers in <1s and survives
196
+ interruptions.
197
+ - πŸ–₯️ **Hardware auto-detection** β€” NVIDIA (`nvidia-smi`), AMD (`rocm-smi`),
198
+ Apple Silicon unified memory (`system_profiler`), with manual override.
199
+ - πŸͺΆ **Zero runtime dependencies** β€” pure Python 3.11+ stdlib.
200
+
201
+ ## Measured results
202
+
203
+ | Model | Target | What the "obvious" pick did | What FiTuna found |
204
+ |---|---|---|---|
205
+ | Qwen3-4B-Instruct (Apache 2.0) | 30 tok/s, ≀5% loss | Q8_0: 24.22 tok/s ❌ (and measured *worse* quality than Q6_K) | **Q4_K_M @ ngl=33 β†’ 30.81 tok/s, 1.73% loss** βœ… |
206
+ | SmolLM2-135M (Apache 2.0) | 240 tok/s, ≀5% loss | Q8_0: 205.91 tok/s ❌ | **Q6_K β†’ 249.50 tok/s, 0.53% loss** βœ… (and Q4_K_M measured *slower* than Q6_K) |
207
+ | Midm-2.0-Mini-Instruct, Korean (MIT) | 40 tok/s, ≀5% loss | Q8_0: 34.26 tok/s ❌ | **Q4_K_M @ ngl=48 β†’ 44.62 tok/s, 2.58% loss** βœ… (the two corpora report different mid-table orders, but the per-chunk trace shows that reorder is [not something we could establish](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)) |
208
+
209
+ Apple M3 Pro, llama.cpp build 9960. Full logs, timings and run-to-run variance
210
+ analysis (including a thermal-throttle outlier we caught and documented):
211
+ **[docs/RESULTS.md](docs/RESULTS.md)** Β· Scenarios:
212
+ [docs/USE_CASES.md](docs/USE_CASES.md) Β· Reproduce on NVIDIA/Linux with the
213
+ one-click Colab notebook (free T4 tier):
214
+ [notebooks/colab_nvidia_verification.ipynb](notebooks/colab_nvidia_verification.ipynb)
215
+ [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/leeyunseokarchive/fituna/blob/main/notebooks/colab_nvidia_verification.ipynb)
216
+
217
+ ## How it works
218
+
219
+ **Stage 1** measures perplexity loss for *every* candidate, because **Stage 2**
220
+ walks them in *measured* quality order and you can't sort by a number you
221
+ haven't measured. Stage 2 early-exits hard: a quant missing the target at full
222
+ offload is dropped without further benches, and the first quant that passes
223
+ wins. Results cache to sqlite3 keyed by model fingerprint, hardware profile
224
+ **and llama.cpp build version**, so `--resume` never serves numbers from a
225
+ different backend build. Diagrams, module map and full algorithm:
226
+ [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) Β· Contract:
227
+ [`fituna/config.py`](fituna/config.py)
228
+
229
+ ## Use as a library
230
+
231
+ Zero runtime dependencies means the modules import directly:
232
+
233
+ ```python
234
+ from fituna.hardware import detect_hardware
235
+
236
+ hw = detect_hardware()
237
+ print(f"{hw.gpu_vendor.value}: {hw.gpu_name}, {hw.vram_mb} MB VRAM, {hw.ram_mb} MB RAM")
238
+ # apple: Apple M3 Pro, 18432 MB VRAM, 18432 MB RAM
239
+ ```
240
+
241
+ (Real output from a `python3.13 -c` run on the same M3 Pro as above.) Driving
242
+ the search programmatically means calling `fituna.search.search()`, which also
243
+ needs a `ModelInfo`, resolved `BinaryPaths`, a work directory and a corpus
244
+ path β€” exactly what `fituna run`/`quickstart` assemble for you
245
+ ([`search.py`](fituna/search.py), [`config.py`](fituna/config.py)).
246
+
247
+ ## MCP server β€” measured answers for AI agents
248
+
249
+ Ask a chatbot "which local model config fits my machine?" and it guesses from
250
+ specs. Point it at FiTuna's MCP server and it *measures*:
251
+
252
+ ```bash
253
+ claude mcp add fituna -- fituna-mcp # or any MCP client, stdio transport
254
+ ```
255
+
256
+ | Tool | What it does |
257
+ |---|---|
258
+ | `fituna_detect_hardware` | GPU vendor/name, VRAM, CPU cores, RAM as FiTuna sees them |
259
+ | `fituna_recommend` | Runs the measured search for a target spec; returns the winning config, measured tok/s, measured quality loss, and a ready-to-run command. Slow once, ~1 s on repeat (cache). |
260
+
261
+ Stdlib-only like the rest of FiTuna β€” MCP stdio is newline-delimited JSON-RPC
262
+ 2.0, no SDK required ([`fituna/mcp_server.py`](fituna/mcp_server.py)).
263
+
264
+ ## Scope
265
+
266
+ FiTuna recommends; it doesn't execute or serve. The output is the quantized
267
+ `.gguf` plus `llama-server` / `llama-cli` commands you copy and run (and, with
268
+ `--export-ollama`, an Ollama `Modelfile` beside it) β€” FiTuna launches none of
269
+ them. That's a deliberate boundary: serving inference is llama.cpp's job, and
270
+ duplicating it would add no differentiated value
271
+ ([rationale](docs/ARCHITECTURE.md#why-this-shape)). The two extensions that
272
+ stay inside it β€” `--launch` and an LM Studio preset export β€” are tracked in
273
+ [#19](https://github.com/leeyunseokarchive/fituna/issues/19).
274
+
275
+ ## Known limitations
276
+
277
+ - **Single GPU only** β€” first GPU reported by `nvidia-smi`/`rocm-smi`; no
278
+ `--tensor-split` ([#11](https://github.com/leeyunseokarchive/fituna/issues/11),
279
+ help wanted: we have no multi-GPU machine to measure on).
280
+ - **Windows AMD auto-detection** β€” `rocm-smi` has no mainstream Windows
281
+ distribution; use `--gpu amd --vram-mb <N>`.
282
+ - **Quality = perplexity on a corpus you choose** β€” a proxy, not a guarantee
283
+ of domain quality. Gate on text resembling your workload
284
+ (`--quality-corpus`; [measured EN-vs-KO
285
+ comparison](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
286
+ - **The quality verdict depends on `--ppl-chunks`** β€” loss is an estimate over
287
+ `chunks Γ— 512` tokens whose absolute value grows with the chunk count, so
288
+ re-measure a candidate close to your budget before trusting the PASS
289
+ ([measured effect](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)).
290
+ `quality.py` still parses the PPL and discards llama-perplexity's error bar
291
+ ([#8](https://github.com/leeyunseokarchive/fituna/issues/8)) β€” which is how
292
+ Run 5 came to publish a claim it later had to withdraw.
293
+ - **Benchmarks are thermally sensitive** β€” verdicts within a few tok/s of the
294
+ target are marginal ([variance
295
+ analysis](docs/RESULTS.md#run-to-run-variance-measured-not-hidden)).
296
+ - **Real-hardware E2E covers macOS and Linux only** β€” Apple Silicon/Metal and
297
+ NVIDIA T4/CUDA. Windows paths are unit-tested and CI-run, but not yet
298
+ integration-run against real binaries
299
+ ([#12](https://github.com/leeyunseokarchive/fituna/issues/12)).
300
+
301
+ ## Contributing
302
+
303
+ Contributions welcome β€” the codebase is small, dependency-free and
304
+ contract-first (start at [`fituna/config.py`](fituna/config.py)); 246 unit
305
+ tests, per-module self-checks and a 3-OS Γ— 2-Python CI matrix guard it.
306
+ The roadmap lives in the
307
+ [v0.2.0 milestone](https://github.com/leeyunseokarchive/fituna/milestone/1),
308
+ including [#10](https://github.com/leeyunseokarchive/fituna/issues/10) (parser
309
+ test coverage, good first issue). See [CONTRIBUTING.md](CONTRIBUTING.md) Β·
310
+ [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) Β· [CHANGELOG.md](CHANGELOG.md) Β·
311
+ [SECURITY.md](SECURITY.md).
312
+
313
+ ## License
314
+
315
+ [MIT](LICENSE) Β© FiTuna contributors. Third-party notices (llama.cpp and
316
+ subprocess-invoked tools): [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md) Β·
317
+ SBOM: [docs/SBOM.md](docs/SBOM.md) Β· Open-source usage:
318
+ [docs/OPEN_SOURCE_USAGE.md](docs/OPEN_SOURCE_USAGE.md) Β· AI-assisted
319
+ development disclosure: [docs/AI_MODEL_USAGE.md](docs/AI_MODEL_USAGE.md)
fituna-0.1.0/README.md ADDED
@@ -0,0 +1,289 @@
1
+ <div align="center">
2
+
3
+ **English** | [ν•œκ΅­μ–΄](README.ko.md)
4
+
5
+ # 🎯 FiTuna
6
+
7
+ **Stop guessing your llama.cpp config. Measure it.**
8
+
9
+ Hardware-benchmark-driven auto-tuning for local LLMs β€” give it a model, a
10
+ target speed and a quality budget; get back the smallest llama.cpp config that
11
+ actually hits those numbers on **your** machine.
12
+
13
+ **API subscriptions add up. Going local means guessing which model your
14
+ machine can actually run.** Don't guess β€” measure it, and run your own.
15
+
16
+ [![CI](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml/badge.svg)](https://github.com/leeyunseokarchive/fituna/actions/workflows/ci.yml)
17
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
18
+ [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://www.python.org/downloads/)
19
+ [![Zero dependencies](https://img.shields.io/badge/runtime%20deps-0-brightgreen.svg)](docs/SBOM.md)
20
+ [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](CONTRIBUTING.md)
21
+
22
+ **μ‹¬μ‚¬μœ„μ› Β· κ²€μ¦κΈ°κ΄€μš© ν•œκ΅­μ–΄ μž¬ν˜„ κ°€μ΄λ“œ β†’ [REVIEWERS.md](REVIEWERS.md)** *(Korean reproduction guide for competition judges & verification agency)*
23
+
24
+ </div>
25
+
26
+ ---
27
+
28
+ ```bash
29
+ $ fituna run --model Qwen3-4B-Instruct-2507-F16.gguf \
30
+ --target-tps 30 --max-quality-loss 5 --ctx 4096 --wikitext wiki.txt --out ./out
31
+
32
+ [Q6_K] full-offload 28.48 tok/s < target 30.00, skipping (early-exit B)
33
+ [Q8_0] full-offload 24.22 tok/s < target 30.00, skipping (early-exit B)
34
+ [Q5_K_M] full-offload 29.59 tok/s < target 30.00, skipping (early-exit B)
35
+ [Q4_K_M] found ngl=33 meeting target -- done
36
+
37
+ FiTuna result: MEETS TARGET
38
+ quant : Q4_K_M ngl : 33 ctx : 4096
39
+ gen tok/s : 30.81 quality loss : 1.73%
40
+
41
+ artifact: out/Qwen3-4B-Instruct-2507-...-Q4_K_M.gguf (2.3 GB -- already produced during the search)
42
+
43
+ 1) local API server (OpenAI-compatible):
44
+ llama-server -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096 --port 8080
45
+ 2) import into Ollama: re-run with --export-ollama to write a Modelfile beside the artifact
46
+ 3) terminal chat (interactive check):
47
+ llama-cli -m out/Qwen3-...-Q4_K_M.gguf -ngl 33 -c 4096
48
+ ```
49
+
50
+ *(Output formatting above is reconstructed against the current version; the
51
+ numbers are the Run 2 measurements.)*
52
+
53
+ A real run on an Apple M3 Pro. The "obviously best" Q8_0 **failed** the speed
54
+ target, Q5_K_M missed by **0.41 tok/s**, and the answer wasn't a quant alone β€”
55
+ it was a quant *plus* the minimal GPU offload (`-ngl 33`, not the full 36).
56
+ None of that is predictable from a spec sheet ([full logs](docs/RESULTS.md)).
57
+
58
+ ## Install
59
+
60
+ Not on PyPI yet β€” install from git, into a virtualenv built with Python 3.11+
61
+ (macOS's system `python3` is 3.9):
62
+
63
+ ```bash
64
+ git clone https://github.com/leeyunseokarchive/fituna
65
+ python3.13 -m venv .venv && source .venv/bin/activate
66
+ pip install -e fituna
67
+ ```
68
+
69
+ You also need llama.cpp, which provides the quantize/bench/perplexity engines
70
+ FiTuna orchestrates:
71
+
72
+ ```bash
73
+ brew install llama.cpp # macOS/Linux Homebrew β€” ships all needed binaries
74
+ # or build from source (any platform):
75
+ git clone https://github.com/ggml-org/llama.cpp
76
+ cmake -S llama.cpp -B llama.cpp/build && cmake --build llama.cpp/build --config Release
77
+ ```
78
+
79
+ ## Quickstart
80
+
81
+ Three ways in: a person runs the wizard; a script or CI job calls
82
+ `fituna run --json`; an AI agent talks to
83
+ [`fituna-mcp`](#mcp-server--measured-answers-for-ai-agents).
84
+
85
+ ```bash
86
+ fituna quickstart
87
+ ```
88
+
89
+ Six steps β€” environment check, targets, license requirements, model, quality
90
+ corpus, search β€” and it prints the assembled `fituna run ...` command
91
+ **before** executing it, so the next run is a one-liner you already have. It
92
+ needs a terminal; in CI or a pipe use `fituna run` directly, since every
93
+ search parameter maps to a public `run` flag (proven by an argv-equality
94
+ test). Model download (a curated shortlist) and HuggingFace search are
95
+ wizard-only conveniences; `run --model` expects a `.gguf` on disk.
96
+
97
+ It never predicts throughput: memory fit is arithmetic (published file size vs
98
+ detected VRAM/RAM, assumed margin stated), speed is measured, and curated or
99
+ HuggingFace-search candidates show their license β€” local-scan and manual-path
100
+ options cannot, since a `.gguf` carries no license metadata. Any
101
+ [`docs/RESULTS.md`](docs/RESULTS.md) figure it cites is a record of what was
102
+ measured on named hardware, never a prediction.
103
+
104
+ ### The script path (what the wizard assembles for you)
105
+
106
+ Quality loss is perplexity increase over a plain-text corpus, so it is only
107
+ meaningful on text resembling your workload. Any UTF-8 file works
108
+ (`--quality-corpus`); both presets are one command away:
109
+
110
+ ```bash
111
+ fituna fetch-corpus --lang en --out wikitext-2-raw-test.txt # wikitext-2 test split
112
+ fituna fetch-corpus --lang ko --out kowiki-corpus.txt --rows 500 # Korean Wikipedia
113
+ ```
114
+
115
+ Measure the language you'll run: the same quant can measure 2–3Γ— different
116
+ loss on the two corpora, and in Run 3 that was enough to change the verdict
117
+ the tool returned ([measurement and
118
+ caveats](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
119
+ Both presets are CC BY-SA 3.0 and `fetch-corpus` prints the license notice
120
+ and source URL when it finishes; `--dataset/--config/--split` override the preset
121
+ ([provenance and licensing](docs/OPEN_SOURCE_USAGE.md)).
122
+
123
+ ```bash
124
+ fituna doctor # confirm the environment is ready
125
+ fituna detect-hw # see what FiTuna detects
126
+ fituna run --model your-model-F16.gguf \
127
+ --target-tps 30 --max-quality-loss 5 \
128
+ --ctx 4096 --wikitext wikitext-2-raw-test.txt --out ./out --resume
129
+ ```
130
+
131
+ Pass an F16/BF16 `.gguf` directly (many models publish one), or an HF-format
132
+ directory if `convert_hf_to_gguf.py` is available (source checkout +
133
+ `pip install torch transformers`; package-manager builds don't ship it).
134
+
135
+ > **Disk usage:** the search quantizes every candidate reaching the quality
136
+ > stage β€” ~12 GB for four candidates of a 4B model. Files are reused across
137
+ > runs; narrow `--quant` to bound this.
138
+
139
+ ## Why
140
+
141
+ Running a local LLM means picking a quantization level (Q2–Q8), a GPU offload
142
+ layer count (`-ngl`) and a context length β€” a search space people navigate
143
+ today by trial and error:
144
+
145
+ - **Ollama / LM Studio** apply fixed per-model presets; a request for finer
146
+ quantization control was [closed as not planned](https://github.com/ollama/ollama/issues/14674).
147
+ - **NVIDIA Model Optimizer**'s AutoQuantize is CUDA-only.
148
+ - **VRAM calculators & chatbot advice** estimate from specs β€” and specs don't
149
+ know your thermals, memory bandwidth, or llama.cpp build flags.
150
+
151
+ FiTuna replaces the guesswork with a measured search over the llama.cpp
152
+ binaries you already have β€” verified on your hardware, reproducible from cache.
153
+
154
+ ## Features
155
+
156
+ - πŸ” **Target-driven search** β€” in: model + target tok/s + max quality loss %.
157
+ Out: quant Γ— `-ngl` Γ— ctx config + a ready-to-run command.
158
+ - πŸ“ **Measured, not assumed** β€” candidates walked in *measured* perplexity
159
+ order (in our runs Q6_K beat Q8_0 β€” [data](docs/RESULTS.md)), with a binary
160
+ search for the minimal GPU offload.
161
+ - ⚑ **Aggressive early exits** β€” quality-gate failures and hopeless quants are
162
+ skipped without wasting benches; a bench that can't finish in time counts as
163
+ "too slow", not a crash.
164
+ - πŸ—ƒοΈ **Reproducible cache** β€” sqlite3, keyed by model fingerprint Γ— hardware Γ—
165
+ llama.cpp build version; `--resume` re-answers in <1s and survives
166
+ interruptions.
167
+ - πŸ–₯️ **Hardware auto-detection** β€” NVIDIA (`nvidia-smi`), AMD (`rocm-smi`),
168
+ Apple Silicon unified memory (`system_profiler`), with manual override.
169
+ - πŸͺΆ **Zero runtime dependencies** β€” pure Python 3.11+ stdlib.
170
+
171
+ ## Measured results
172
+
173
+ | Model | Target | What the "obvious" pick did | What FiTuna found |
174
+ |---|---|---|---|
175
+ | Qwen3-4B-Instruct (Apache 2.0) | 30 tok/s, ≀5% loss | Q8_0: 24.22 tok/s ❌ (and measured *worse* quality than Q6_K) | **Q4_K_M @ ngl=33 β†’ 30.81 tok/s, 1.73% loss** βœ… |
176
+ | SmolLM2-135M (Apache 2.0) | 240 tok/s, ≀5% loss | Q8_0: 205.91 tok/s ❌ | **Q6_K β†’ 249.50 tok/s, 0.53% loss** βœ… (and Q4_K_M measured *slower* than Q6_K) |
177
+ | Midm-2.0-Mini-Instruct, Korean (MIT) | 40 tok/s, ≀5% loss | Q8_0: 34.26 tok/s ❌ | **Q4_K_M @ ngl=48 β†’ 44.62 tok/s, 2.58% loss** βœ… (the two corpora report different mid-table orders, but the per-chunk trace shows that reorder is [not something we could establish](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)) |
178
+
179
+ Apple M3 Pro, llama.cpp build 9960. Full logs, timings and run-to-run variance
180
+ analysis (including a thermal-throttle outlier we caught and documented):
181
+ **[docs/RESULTS.md](docs/RESULTS.md)** Β· Scenarios:
182
+ [docs/USE_CASES.md](docs/USE_CASES.md) Β· Reproduce on NVIDIA/Linux with the
183
+ one-click Colab notebook (free T4 tier):
184
+ [notebooks/colab_nvidia_verification.ipynb](notebooks/colab_nvidia_verification.ipynb)
185
+ [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/leeyunseokarchive/fituna/blob/main/notebooks/colab_nvidia_verification.ipynb)
186
+
187
+ ## How it works
188
+
189
+ **Stage 1** measures perplexity loss for *every* candidate, because **Stage 2**
190
+ walks them in *measured* quality order and you can't sort by a number you
191
+ haven't measured. Stage 2 early-exits hard: a quant missing the target at full
192
+ offload is dropped without further benches, and the first quant that passes
193
+ wins. Results cache to sqlite3 keyed by model fingerprint, hardware profile
194
+ **and llama.cpp build version**, so `--resume` never serves numbers from a
195
+ different backend build. Diagrams, module map and full algorithm:
196
+ [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) Β· Contract:
197
+ [`fituna/config.py`](fituna/config.py)
198
+
199
+ ## Use as a library
200
+
201
+ Zero runtime dependencies means the modules import directly:
202
+
203
+ ```python
204
+ from fituna.hardware import detect_hardware
205
+
206
+ hw = detect_hardware()
207
+ print(f"{hw.gpu_vendor.value}: {hw.gpu_name}, {hw.vram_mb} MB VRAM, {hw.ram_mb} MB RAM")
208
+ # apple: Apple M3 Pro, 18432 MB VRAM, 18432 MB RAM
209
+ ```
210
+
211
+ (Real output from a `python3.13 -c` run on the same M3 Pro as above.) Driving
212
+ the search programmatically means calling `fituna.search.search()`, which also
213
+ needs a `ModelInfo`, resolved `BinaryPaths`, a work directory and a corpus
214
+ path β€” exactly what `fituna run`/`quickstart` assemble for you
215
+ ([`search.py`](fituna/search.py), [`config.py`](fituna/config.py)).
216
+
217
+ ## MCP server β€” measured answers for AI agents
218
+
219
+ Ask a chatbot "which local model config fits my machine?" and it guesses from
220
+ specs. Point it at FiTuna's MCP server and it *measures*:
221
+
222
+ ```bash
223
+ claude mcp add fituna -- fituna-mcp # or any MCP client, stdio transport
224
+ ```
225
+
226
+ | Tool | What it does |
227
+ |---|---|
228
+ | `fituna_detect_hardware` | GPU vendor/name, VRAM, CPU cores, RAM as FiTuna sees them |
229
+ | `fituna_recommend` | Runs the measured search for a target spec; returns the winning config, measured tok/s, measured quality loss, and a ready-to-run command. Slow once, ~1 s on repeat (cache). |
230
+
231
+ Stdlib-only like the rest of FiTuna β€” MCP stdio is newline-delimited JSON-RPC
232
+ 2.0, no SDK required ([`fituna/mcp_server.py`](fituna/mcp_server.py)).
233
+
234
+ ## Scope
235
+
236
+ FiTuna recommends; it doesn't execute or serve. The output is the quantized
237
+ `.gguf` plus `llama-server` / `llama-cli` commands you copy and run (and, with
238
+ `--export-ollama`, an Ollama `Modelfile` beside it) β€” FiTuna launches none of
239
+ them. That's a deliberate boundary: serving inference is llama.cpp's job, and
240
+ duplicating it would add no differentiated value
241
+ ([rationale](docs/ARCHITECTURE.md#why-this-shape)). The two extensions that
242
+ stay inside it β€” `--launch` and an LM Studio preset export β€” are tracked in
243
+ [#19](https://github.com/leeyunseokarchive/fituna/issues/19).
244
+
245
+ ## Known limitations
246
+
247
+ - **Single GPU only** β€” first GPU reported by `nvidia-smi`/`rocm-smi`; no
248
+ `--tensor-split` ([#11](https://github.com/leeyunseokarchive/fituna/issues/11),
249
+ help wanted: we have no multi-GPU machine to measure on).
250
+ - **Windows AMD auto-detection** β€” `rocm-smi` has no mainstream Windows
251
+ distribution; use `--gpu amd --vram-mb <N>`.
252
+ - **Quality = perplexity on a corpus you choose** β€” a proxy, not a guarantee
253
+ of domain quality. Gate on text resembling your workload
254
+ (`--quality-corpus`; [measured EN-vs-KO
255
+ comparison](docs/RESULTS.md#run-3--english-vs-korean-quality-corpus-same-model-same-quants)).
256
+ - **The quality verdict depends on `--ppl-chunks`** β€” loss is an estimate over
257
+ `chunks Γ— 512` tokens whose absolute value grows with the chunk count, so
258
+ re-measure a candidate close to your budget before trusting the PASS
259
+ ([measured effect](docs/RESULTS.md#how-big-is-a-perplexity-gap-the-error-bar-we-had-been-discarding)).
260
+ `quality.py` still parses the PPL and discards llama-perplexity's error bar
261
+ ([#8](https://github.com/leeyunseokarchive/fituna/issues/8)) β€” which is how
262
+ Run 5 came to publish a claim it later had to withdraw.
263
+ - **Benchmarks are thermally sensitive** β€” verdicts within a few tok/s of the
264
+ target are marginal ([variance
265
+ analysis](docs/RESULTS.md#run-to-run-variance-measured-not-hidden)).
266
+ - **Real-hardware E2E covers macOS and Linux only** β€” Apple Silicon/Metal and
267
+ NVIDIA T4/CUDA. Windows paths are unit-tested and CI-run, but not yet
268
+ integration-run against real binaries
269
+ ([#12](https://github.com/leeyunseokarchive/fituna/issues/12)).
270
+
271
+ ## Contributing
272
+
273
+ Contributions welcome β€” the codebase is small, dependency-free and
274
+ contract-first (start at [`fituna/config.py`](fituna/config.py)); 246 unit
275
+ tests, per-module self-checks and a 3-OS Γ— 2-Python CI matrix guard it.
276
+ The roadmap lives in the
277
+ [v0.2.0 milestone](https://github.com/leeyunseokarchive/fituna/milestone/1),
278
+ including [#10](https://github.com/leeyunseokarchive/fituna/issues/10) (parser
279
+ test coverage, good first issue). See [CONTRIBUTING.md](CONTRIBUTING.md) Β·
280
+ [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) Β· [CHANGELOG.md](CHANGELOG.md) Β·
281
+ [SECURITY.md](SECURITY.md).
282
+
283
+ ## License
284
+
285
+ [MIT](LICENSE) Β© FiTuna contributors. Third-party notices (llama.cpp and
286
+ subprocess-invoked tools): [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md) Β·
287
+ SBOM: [docs/SBOM.md](docs/SBOM.md) Β· Open-source usage:
288
+ [docs/OPEN_SOURCE_USAGE.md](docs/OPEN_SOURCE_USAGE.md) Β· AI-assisted
289
+ development disclosure: [docs/AI_MODEL_USAGE.md](docs/AI_MODEL_USAGE.md)
@@ -0,0 +1,30 @@
1
+ # SPDX-License-Identifier: MIT
2
+ """FiTuna: hardware-aware auto-tuner for llama.cpp GGUF quantization + runtime configs."""
3
+
4
+ __version__ = "0.1.0"
5
+
6
+ __all__ = ["__version__"]
7
+
8
+
9
+ def _self_check() -> None:
10
+ """Guard against __version__ drifting from pyproject.toml's [project.version]."""
11
+ import tomllib
12
+ from pathlib import Path
13
+
14
+ pyproject = Path(__file__).resolve().parent.parent / "pyproject.toml"
15
+ if not pyproject.exists():
16
+ return # installed outside the source tree (e.g. site-packages) β€” nothing to compare against
17
+ with pyproject.open("rb") as f:
18
+ data = tomllib.load(f)
19
+ pyproject_version = data["project"]["version"]
20
+ assert __version__ == pyproject_version, (
21
+ f"__version__ ({__version__}) drifted from pyproject.toml ({pyproject_version})"
22
+ )
23
+ assert isinstance(__version__, str) and __version__.count(".") == 2, (
24
+ f"__version__ ({__version__!r}) is not a MAJOR.MINOR.PATCH string"
25
+ )
26
+
27
+
28
+ if __name__ == "__main__":
29
+ _self_check()
30
+ print("fituna.__init__ self-check passed:", __version__)
@@ -0,0 +1,9 @@
1
+ # SPDX-License-Identifier: MIT
2
+ """``python -m fituna`` entry point."""
3
+
4
+ import sys
5
+
6
+ from fituna.cli import main
7
+
8
+ if __name__ == "__main__":
9
+ raise SystemExit(main(sys.argv[1:]))