tuieval 0.2.0.dev4__tar.gz → 0.2.0.dev5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/CHANGELOG.md +1 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/PKG-INFO +1 -1
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/models.md +1 -1
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/_version.py +2 -2
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/tune.py +17 -4
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/tests/test_tuieval.py +26 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/.gitignore +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/LICENSE +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/README.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/RELEASING.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/README.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/favicon.ico +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/readme-header.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/social-preview.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-1024.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-128.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-256.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-32.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-48.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-512.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-64.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-animated.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/tui-setup.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/writing-packs.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/pyproject.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/__init__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/__main__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/cli.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/client.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/compare.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/engine.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/export.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/__init__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/answer.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/code.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/rag.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/reply.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/tool_call.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/machines.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/packs.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/profiles.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/remove.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/run_evals.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/scaffold.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/selftest.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/models.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/tui.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/verdict.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/watch_proxy.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/workspace.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/yamlout.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/tests/mock_server.py +0 -0
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
- `tuieval tune` no longer rejects settings because the model's loading pushed other apps to swap (e.g. a large model locked in RAM with `--load-mode mlock`); only swapping while the loaded server works counts. Swap at load is noted in the output and the profile (`server_load_swapped_mb`).
|
|
5
6
|
- Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
|
|
6
7
|
- The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
|
|
7
8
|
- `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tuieval
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev5
|
|
4
4
|
Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ashe-wb/tuieval
|
|
6
6
|
Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
|
|
@@ -78,7 +78,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
|
|
|
78
78
|
|
|
79
79
|
1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
|
|
80
80
|
2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
|
|
81
|
-
3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates
|
|
81
|
+
3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates under which macOS swaps while the server works are rejected; swapping while the model loads (e.g. other apps making room for a model locked in RAM) is only noted.
|
|
82
82
|
|
|
83
83
|
Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
|
|
84
84
|
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2, 0, '
|
|
21
|
+
__version__ = version = '0.2.0.dev5'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -26,7 +26,8 @@ An output guard compares greedy answers with the default flags; an option that c
|
|
|
26
26
|
noise is rejected, because speed flags must not change answers. Servers whose answers depend on
|
|
27
27
|
the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
|
|
28
28
|
become part of the results fingerprint instead, so retuning marks that machine's results
|
|
29
|
-
outdated. Any candidate
|
|
29
|
+
outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
|
|
30
|
+
while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
|
|
30
31
|
measured.
|
|
31
32
|
|
|
32
33
|
The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
|
|
@@ -474,12 +475,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
474
475
|
emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
|
|
475
476
|
start=starts[0], max_starts=max_starts)
|
|
476
477
|
swap0 = machines.swapped_out_bytes()
|
|
478
|
+
swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
|
|
477
479
|
try:
|
|
478
480
|
with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
|
|
481
|
+
# Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
|
|
482
|
+
# that's recorded, not held against the settings. Swapping while serving is.
|
|
483
|
+
loaded = machines.swapped_out_bytes()
|
|
479
484
|
r = measure(eng, m, url, work, gen_tokens, passes, emit,
|
|
480
485
|
settings.get("request_timeout_s", REQUEST_LIMIT_S))
|
|
486
|
+
end = machines.swapped_out_bytes()
|
|
487
|
+
if None not in (swap0, loaded, end):
|
|
488
|
+
swapped = (end - loaded) / 2**20
|
|
489
|
+
at_load = (loaded - swap0) / 2**20
|
|
490
|
+
if at_load > SWAP_LIMIT / 2**20:
|
|
491
|
+
emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
|
|
492
|
+
"(close apps for more headroom)")
|
|
481
493
|
r.load_s = round(info["load_s"], 1) if info["load_s"] else None
|
|
482
494
|
r.facts = dict(info["facts"])
|
|
495
|
+
if swapped is not None and at_load > SWAP_LIMIT / 2**20:
|
|
496
|
+
r.facts["load_swapped_mb"] = round(at_load)
|
|
483
497
|
# Settled allocation after the timed pass (brief peaks while processing a prompt are
|
|
484
498
|
# harmless; sustained allocation over the limit is what makes the driver churn).
|
|
485
499
|
settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
|
|
@@ -497,9 +511,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
497
511
|
"close other apps for faster and fairer results")
|
|
498
512
|
except engine_mod.ModelFailed as e:
|
|
499
513
|
r = Measure(error=str(e).splitlines()[0])
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
r.error = f"macOS swapped {(swap1 - swap0) / 2**20:.0f} MB (memory too tight)"
|
|
514
|
+
if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
|
|
515
|
+
r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
|
|
503
516
|
cache[k] = r
|
|
504
517
|
emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
|
|
505
518
|
return r
|
|
@@ -324,6 +324,32 @@ class WarmTune(unittest.TestCase):
|
|
|
324
324
|
self.tune.measure = self.orig
|
|
325
325
|
self.tmp.cleanup()
|
|
326
326
|
|
|
327
|
+
def _swap(self, at_load_mb, while_serving_mb):
|
|
328
|
+
"""Fake macOS swap counter: each server start swaps at_load_mb while loading (between the
|
|
329
|
+
reading before the start and the one once it serves) and while_serving_mb during the work."""
|
|
330
|
+
from tuieval import machines
|
|
331
|
+
state = {"n": 0, "total": 0}
|
|
332
|
+
|
|
333
|
+
def swapped_out_bytes():
|
|
334
|
+
step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
|
|
335
|
+
state["total"] += {0: 0, 1: at_load_mb, 2: while_serving_mb}[step] * 2**20
|
|
336
|
+
state["n"] += 1
|
|
337
|
+
return state["total"]
|
|
338
|
+
orig = machines.swapped_out_bytes
|
|
339
|
+
machines.swapped_out_bytes = swapped_out_bytes
|
|
340
|
+
self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
|
|
341
|
+
|
|
342
|
+
def test_swapping_while_loading_is_only_noted(self):
|
|
343
|
+
self._swap(at_load_mb=900, while_serving_mb=0)
|
|
344
|
+
p = self.tune.tune(self.eng, "new", use_bench=False)
|
|
345
|
+
self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
|
|
346
|
+
|
|
347
|
+
def test_swapping_while_serving_fails_the_settings(self):
|
|
348
|
+
self._swap(at_load_mb=0, while_serving_mb=500)
|
|
349
|
+
with self.assertRaises(self.tune.engine_mod.ModelFailed) as cm:
|
|
350
|
+
self.tune.tune(self.eng, "new", use_bench=False)
|
|
351
|
+
self.assertIn("swapped 500 MB while the server worked", str(cm.exception))
|
|
352
|
+
|
|
327
353
|
def test_family_ignores_the_mtp_layer(self):
|
|
328
354
|
# one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
|
|
329
355
|
d = self.tmp.name
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|