tuieval 0.2.0.dev2__tar.gz → 0.2.0.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/CHANGELOG.md +1 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/PKG-INFO +1 -1
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/models.md +3 -2
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/_version.py +2 -2
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/engine.py +18 -6
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/tests/test_tuieval.py +25 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/.gitignore +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/LICENSE +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/README.md +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/RELEASING.md +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/README.md +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/favicon.ico +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/readme-header.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/social-preview.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-1024.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-128.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-16.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-16.svg +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-256.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-32.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-48.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-512.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-64.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-animated.svg +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon.svg +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/images/tui-setup.png +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/docs/writing-packs.md +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/pyproject.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/__init__.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/__main__.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/cli.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/client.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/compare.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/export.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/__init__.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/answer.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/code.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/rag.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/reply.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/graders/tool_call.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/machines.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/packs.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/profiles.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/remove.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/run_evals.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/scaffold.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/selftest.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/models.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/pack.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/system.txt +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/pack.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/system.txt +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/tests.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/pack.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/system.txt +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/pack.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/system.txt +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/tui.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/tune.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/verdict.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/watch_proxy.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/workspace.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/src/tuieval/yamlout.py +0 -0
- {tuieval-0.2.0.dev2 → tuieval-0.2.0.dev3}/tests/mock_server.py +0 -0
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
- The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
|
|
5
6
|
- `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
|
|
6
7
|
- A new README screenshot from a demo workspace.
|
|
7
8
|
- Versions come from git tags: a tag `vX.Y.Z` is that release, and every commit on `main` after it is published automatically as a dev build `X.(Y+1).0.devN` (N = commits since the release). `pip install tuieval` keeps installing releases only; `pip install --pre tuieval` gets the latest build.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tuieval
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev3
|
|
4
4
|
Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ashe-wb/tuieval
|
|
6
6
|
Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
|
|
@@ -69,7 +69,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
|
|
|
69
69
|
## Machines
|
|
70
70
|
|
|
71
71
|
- **Machine id** is detected automatically (e.g. `m3max-64gb`, `m2-16gb`; set `EVALS_MACHINE` to rename it). Every result records the machine, the server version and the exact speed flags used. `tuieval machines` lists this machine and every other machine that has run the evals (they record themselves in `tuning/`).
|
|
72
|
-
- **Fit check (Apple Silicon, llama):** before starting a GGUF, tuieval reads its header (layers, KV heads, hybrid attention layers) and picks the largest context that fits this machine's GPU memory, capped at `max_ctx`. A model that can't fit at 8k context is skipped with *doesn't fit on <machine>* instead of swapping. Packs that need more context than fits are skipped. It never quantizes the KV cache on its own, since that changes answers.
|
|
72
|
+
- **Fit check (Apple Silicon, llama):** before starting a GGUF, tuieval reads its header (layers, KV heads, hybrid attention layers) and picks the largest context that fits this machine's GPU memory, capped at `max_ctx`. A model that can't fit at 8k context is skipped with *doesn't fit on <machine>* instead of swapping. Packs that need more context than fits are skipped. It never quantizes the KV cache on its own, since that changes answers. Servers that size their own memory (their `cmd` doesn't take `{ctx}`) are left alone: their context is the model's `max_context`; `fit_check` on the server changes that.
|
|
73
73
|
- **Per-machine settings** go under `[machines.<id>]`: `memory_headroom_gb` (GPU memory kept free for macOS, default 4) and `gpu_residency_gb` (see the stall guard).
|
|
74
74
|
|
|
75
75
|
## Tuning
|
|
@@ -120,7 +120,8 @@ On Apple Silicon Macs, once the system's GPU allocations pass about half of RAM,
|
|
|
120
120
|
| `url` | an already-running server's base URL |
|
|
121
121
|
| `port` | the port `cmd` serves on |
|
|
122
122
|
| `cwd` | folder to start `cmd` in |
|
|
123
|
-
| `model_is_path` | `model` is a file or folder: check it exists
|
|
123
|
+
| `model_is_path` | `model` is a file or folder: check it exists before starting |
|
|
124
|
+
| `fit_check` | size the context from the GGUF header and skip models that don't fit (the fit check; it assumes llama.cpp's memory use). Default: on for servers whose `cmd` takes `{ctx}`, off for others, whose context is the model's `max_context` |
|
|
124
125
|
| `vision_args` | flags appended for models with `mmproj` |
|
|
125
126
|
| `kv_type`, `max_ctx` | llama defaults: KV-cache type, upper bound on context |
|
|
126
127
|
| `ctx_flag` | the server's context flag, so long packs are skipped if a configured context is too small |
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2, 0, '
|
|
21
|
+
__version__ = version = '0.2.0.dev3'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0, 'dev3')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -96,6 +96,16 @@ def load_config(path):
|
|
|
96
96
|
return cfg
|
|
97
97
|
|
|
98
98
|
|
|
99
|
+
def fit_check(server):
|
|
100
|
+
"""Whether tuieval sizes the context for this server's models from GGUF headers (machines.fit,
|
|
101
|
+
which assumes llama.cpp's memory use). By default only servers whose command takes {ctx}, i.e.
|
|
102
|
+
where tuieval chooses the context; others size their own memory, and their context is the
|
|
103
|
+
model's max_context. models.toml `fit_check = true|false` on a server overrides it."""
|
|
104
|
+
if "fit_check" in server:
|
|
105
|
+
return bool(server["fit_check"])
|
|
106
|
+
return any("{ctx}" in str(a) for a in server.get("cmd", []))
|
|
107
|
+
|
|
108
|
+
|
|
99
109
|
def effective_sampling(cfg, m):
|
|
100
110
|
"""The request settings this model runs with: [sampling], then the model's own overrides."""
|
|
101
111
|
s = dict(cfg["sampling"])
|
|
@@ -1018,13 +1028,15 @@ class Engine:
|
|
|
1018
1028
|
path = expand(m["model"])
|
|
1019
1029
|
if server.get("model_is_path") and os.path.isfile(path):
|
|
1020
1030
|
try:
|
|
1021
|
-
mmproj = expand(m.get("mmproj", ""))
|
|
1022
|
-
extra = os.path.getsize(mmproj) if mmproj and os.path.isfile(mmproj) else 0
|
|
1023
|
-
headroom = self.machine_settings(mc.id).get("memory_headroom_gb", 4.0)
|
|
1024
|
-
f = machines.fit(path, mc, kv_type=kv or "f16", headroom_gb=headroom, extra_bytes=extra,
|
|
1025
|
-
want_ctx=want)
|
|
1026
1031
|
info = machines.read_gguf(path)
|
|
1027
|
-
|
|
1032
|
+
mtp, size = info["mtp_layers"], info["bytes"]
|
|
1033
|
+
if fit_check(server):
|
|
1034
|
+
mmproj = expand(m.get("mmproj", ""))
|
|
1035
|
+
extra = os.path.getsize(mmproj) if mmproj and os.path.isfile(mmproj) else 0
|
|
1036
|
+
headroom = self.machine_settings(mc.id).get("memory_headroom_gb", 4.0)
|
|
1037
|
+
f = machines.fit(path, mc, kv_type=kv or "f16", headroom_gb=headroom, extra_bytes=extra,
|
|
1038
|
+
want_ctx=want)
|
|
1039
|
+
ctx, fits, note = f.max_ctx, f.fits, f.note
|
|
1028
1040
|
except (OSError, ValueError) as e:
|
|
1029
1041
|
note = f"couldn't read the GGUF header ({e}); context not checked"
|
|
1030
1042
|
profile, perf, source = None, [], "n/a"
|
|
@@ -240,6 +240,31 @@ def fake_gguf(path, embedding=5120, size=0):
|
|
|
240
240
|
f.write(out + b"\0" * max(0, size - len(out)))
|
|
241
241
|
|
|
242
242
|
|
|
243
|
+
class FitCheck(unittest.TestCase):
|
|
244
|
+
def test_default_follows_the_ctx_placeholder(self):
|
|
245
|
+
from tuieval import engine
|
|
246
|
+
self.assertTrue(engine.fit_check({"cmd": ["llama-server", "-c", "{ctx}"]}))
|
|
247
|
+
self.assertFalse(engine.fit_check({"cmd": ["other-server", "--model", "{model}"]}))
|
|
248
|
+
self.assertFalse(engine.fit_check({"url": "http://x"}))
|
|
249
|
+
self.assertTrue(engine.fit_check({"cmd": ["other"], "fit_check": True}))
|
|
250
|
+
self.assertFalse(engine.fit_check({"cmd": ["llama", "-c", "{ctx}"], "fit_check": False}))
|
|
251
|
+
|
|
252
|
+
def test_own_memory_servers_keep_their_context(self):
|
|
253
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
254
|
+
ws = os.path.join(tmp, "ws")
|
|
255
|
+
tuieval(tmp, "init", ws)
|
|
256
|
+
fake_gguf(os.path.join(tmp, "m.gguf"), size=1000)
|
|
257
|
+
with open(os.path.join(ws, "models.toml"), "a") as f:
|
|
258
|
+
f.write(f'\n[servers.own]\ncmd = ["own-server", "--model", "{{model}}"]\nport = 18500\nmodel_is_path = true\n'
|
|
259
|
+
f'\n[[models]]\nlabel = "sized"\nserver = "llama"\nmodel = "{tmp}/m.gguf"\n'
|
|
260
|
+
f'\n[[models]]\nlabel = "own"\nserver = "own"\nmodel = "{tmp}/m.gguf"\nmax_context = 131072\n')
|
|
261
|
+
code = ("from tuieval import engine; e = engine.Engine(); "
|
|
262
|
+
"print(e.serving(e.model('own')).ctx, e.serving(e.model('sized')).fit_note != '')")
|
|
263
|
+
out = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True,
|
|
264
|
+
env=dict(os.environ, TUIEVAL_HOME=ws)).stdout.split()
|
|
265
|
+
self.assertEqual(out, ["131072", "True"]) # own server: its max_context; llama: fit-checked
|
|
266
|
+
|
|
267
|
+
|
|
243
268
|
class WarmTune(unittest.TestCase):
|
|
244
269
|
"""tune.tune starting from a tuned model of the same family, on a fake engine whose speed is a
|
|
245
270
|
function of the flags."""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|