tuieval 0.2.0.dev1__py3-none-any.whl → 0.2.0.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tuieval/_version.py +2 -2
- tuieval/compare.py +1 -1
- tuieval/export.py +10 -7
- tuieval/machines.py +2 -3
- tuieval/run_evals.py +1 -1
- tuieval/templates/models.toml +1 -1
- {tuieval-0.2.0.dev1.dist-info → tuieval-0.2.0.dev2.dist-info}/METADATA +2 -2
- {tuieval-0.2.0.dev1.dist-info → tuieval-0.2.0.dev2.dist-info}/RECORD +11 -11
- {tuieval-0.2.0.dev1.dist-info → tuieval-0.2.0.dev2.dist-info}/WHEEL +0 -0
- {tuieval-0.2.0.dev1.dist-info → tuieval-0.2.0.dev2.dist-info}/entry_points.txt +0 -0
- {tuieval-0.2.0.dev1.dist-info → tuieval-0.2.0.dev2.dist-info}/licenses/LICENSE +0 -0
tuieval/_version.py
CHANGED
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2, 0, '
|
|
21
|
+
__version__ = version = '0.2.0.dev2'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0, 'dev2')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
tuieval/compare.py
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Usage:
|
|
4
4
|
tuieval compare # everything in results/<model>/<pack>.json
|
|
5
5
|
tuieval compare --speed # speed & tokens table (TTFT, tok/s, tokens, memory)
|
|
6
|
-
tuieval compare --machine
|
|
6
|
+
tuieval compare --machine m3max-64gb # speed on another machine (measured there, or projected)
|
|
7
7
|
tuieval compare --pairwise --failures # is each difference real? which tests failed?
|
|
8
8
|
tuieval compare results/some-model/*.json # specific files
|
|
9
9
|
"""
|
tuieval/export.py
CHANGED
|
@@ -1,17 +1,18 @@
|
|
|
1
1
|
"""Export a model's serving settings to the pi coding agent, so pi serves it exactly as the evals did:
|
|
2
2
|
the same model file, output-affecting flags and context, plus the speed flags tuned on this machine.
|
|
3
3
|
|
|
4
|
-
a section in llama
|
|
5
|
-
left out) and an entry under pi's `llama` provider modelOverrides.
|
|
4
|
+
a section in a llama.cpp router's model presets file (llama-server --models-preset; keys that
|
|
5
|
+
equal its [*] section are left out) and an entry under pi's `llama` provider modelOverrides.
|
|
6
|
+
Models on llama servers only.
|
|
6
7
|
|
|
7
8
|
Every file is backed up next to itself (<file>.bak-tuieval-<time>) before it changes, and the diff
|
|
8
9
|
is shown first. Paths come from models.toml [export.pi]:
|
|
9
10
|
|
|
10
11
|
[export.pi]
|
|
11
|
-
presets = "
|
|
12
|
+
presets = "/path/to/presets.ini" # required: the router's --models-preset file
|
|
12
13
|
pi_models = "~/.pi/agent/models.json"
|
|
13
14
|
llama_provider = "llama" # pi's provider name for the router
|
|
14
|
-
servers = ["llama"] # models.toml servers
|
|
15
|
+
servers = ["llama"] # models.toml servers the router can serve
|
|
15
16
|
"""
|
|
16
17
|
import configparser
|
|
17
18
|
import difflib
|
|
@@ -23,7 +24,7 @@ import time
|
|
|
23
24
|
|
|
24
25
|
from . import engine as engine_mod
|
|
25
26
|
|
|
26
|
-
DEFAULTS = {"presets":
|
|
27
|
+
DEFAULTS = {"presets": None, "pi_models": "~/.pi/agent/models.json",
|
|
27
28
|
"llama_provider": "llama", "servers": ["llama"]}
|
|
28
29
|
# llama.cpp short flags -> the long names presets use as keys
|
|
29
30
|
SHORT = {"-m": "model", "-c": "ctx-size", "-t": "threads", "-tb": "threads-batch", "-ub": "ubatch-size",
|
|
@@ -127,7 +128,7 @@ def plan(eng, label, model_id=None, name=None):
|
|
|
127
128
|
m = eng.model(label)
|
|
128
129
|
s = settings(eng)
|
|
129
130
|
if m["server"] not in s["servers"]:
|
|
130
|
-
raise ExportError(f"{label} runs on the {m['server']} server; pi export writes llama
|
|
131
|
+
raise ExportError(f"{label} runs on the {m['server']} server; pi export writes llama.cpp router presets, "
|
|
131
132
|
"for llama servers ([export.pi] servers lists them)")
|
|
132
133
|
model_id = model_id or m.get("pi_id") or default_id(m)
|
|
133
134
|
name = name or m.get("pi_name") or model_id
|
|
@@ -157,6 +158,8 @@ def plan(eng, label, model_id=None, name=None):
|
|
|
157
158
|
for k, key in SAMPLING_KEYS.items():
|
|
158
159
|
if sampling.get(k) is not None:
|
|
159
160
|
flags[key] = str(sampling[k])
|
|
161
|
+
if not s.get("presets"):
|
|
162
|
+
raise ExportError("set [export.pi] presets in models.toml to your llama.cpp router's --models-preset file")
|
|
160
163
|
path = engine_mod.expand(s["presets"])
|
|
161
164
|
old = _read(path)
|
|
162
165
|
if not old:
|
|
@@ -183,7 +186,7 @@ def plan(eng, label, model_id=None, name=None):
|
|
|
183
186
|
"maxTokens": entry.get("maxTokens", sampling.get("max_tokens")),
|
|
184
187
|
"reasoning": bool(sampling.get("enable_thinking", True)),
|
|
185
188
|
"input": ["text", "image"] if vision else ["text"]})
|
|
186
|
-
notes.append("restart llama
|
|
189
|
+
notes.append("restart the llama.cpp router so it reads the new presets")
|
|
187
190
|
if sampling.get("reasoning_effort"):
|
|
188
191
|
notes.append(f"the evals ran reasoning_effort {sampling['reasoning_effort']}: pick that thinking level in pi")
|
|
189
192
|
pi_new = json.dumps(pi, indent=2, ensure_ascii=False) + "\n"
|
tuieval/machines.py
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Which machine this is, what a model needs, and whether it fits.
|
|
2
2
|
|
|
3
|
-
detect() -> Machine (chip, cores, RAM, GPU memory limit, id like "
|
|
3
|
+
detect() -> Machine (chip, cores, RAM, GPU memory limit, id like "m3max-64gb")
|
|
4
4
|
read_gguf(path) -> GGUF header facts (architecture, layers, KV heads, context length, …)
|
|
5
5
|
fit(model_path, …) -> Fit (largest context that fits this machine, estimated memory)
|
|
6
6
|
|
|
@@ -219,8 +219,7 @@ def swapped_out_bytes():
|
|
|
219
219
|
|
|
220
220
|
|
|
221
221
|
# ---------------------------------------------------------------- GPU residency
|
|
222
|
-
#
|
|
223
|
-
# of RAM, the GPU driver evicts and re-maps memory on every command submission (70-95% of the
|
|
222
|
+
# On Apple Silicon Macs, once the system's GPU allocations exceed about half of RAM, the GPU driver evicts and re-maps memory on every command submission (70-95% of the
|
|
224
223
|
# server's CPU in the kernel, GPU idle), whatever iogpu.wired_limit_mb says. Servers that submit
|
|
225
224
|
# many small GPU jobs collapse; see the stall guard in docs/models.md.
|
|
226
225
|
RESIDENCY_FRACTION = 0.5
|
tuieval/run_evals.py
CHANGED
|
@@ -576,7 +576,7 @@ def cmd_export(argv):
|
|
|
576
576
|
description="Write a model's serving settings (model file, output-affecting flags, "
|
|
577
577
|
"context and the speed flags tuned on this machine) to another tool. "
|
|
578
578
|
"Shows the diff and asks first; every file is backed up.")
|
|
579
|
-
p.add_argument("target", choices=["pi"], help="pi: the pi coding agent (llama
|
|
579
|
+
p.add_argument("target", choices=["pi"], help="pi: the pi coding agent (llama.cpp router presets and pi's "
|
|
580
580
|
"models.json)")
|
|
581
581
|
p.add_argument("labels", nargs="+", help="models to export")
|
|
582
582
|
p.add_argument("--id", help="the model id pi sees (default: models.toml pi_id, else the GGUF's name)")
|
tuieval/templates/models.toml
CHANGED
|
@@ -107,7 +107,7 @@ headers = { "X-Title" = "tuieval" }
|
|
|
107
107
|
# cmd = ["~/llama.cpp/build/bin/llama-bench"]
|
|
108
108
|
|
|
109
109
|
# Optional per-machine settings, by machine id (tuieval machines shows the ids).
|
|
110
|
-
# [machines.
|
|
110
|
+
# [machines.m3max-64gb]
|
|
111
111
|
# memory_headroom_gb = 3 # GPU memory kept free for macOS and other apps (default 4)
|
|
112
112
|
# gpu_residency_gb = 12 # GPU memory the driver keeps resident before churning (default: half of RAM)
|
|
113
113
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tuieval
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev2
|
|
4
4
|
Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ashe-wb/tuieval
|
|
6
6
|
Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
|
|
@@ -47,7 +47,7 @@ tuieval # open the TUI
|
|
|
47
47
|
|
|
48
48
|
Once you've added your own packs and models, the setup screen looks like this (packs on the left, models with a verdict code per use case on the right, the highlighted model's details below):
|
|
49
49
|
|
|
50
|
-

|
|
51
51
|
|
|
52
52
|
## What you get
|
|
53
53
|
|
|
@@ -1,16 +1,16 @@
|
|
|
1
1
|
tuieval/__init__.py,sha256=myVXVX7htbwezW6k3x_FuKgn1RSZ2WcTaDEMByGn6Wg,312
|
|
2
2
|
tuieval/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
|
|
3
|
-
tuieval/_version.py,sha256=
|
|
3
|
+
tuieval/_version.py,sha256=EOZVyFNK9VEmUe6471HI7YHfwqJ5uDi2cXJHVylN6eI,533
|
|
4
4
|
tuieval/cli.py,sha256=ZaLoWrxyCABu4SJzx5U2gVUUmwcwRgpZVqdqnxyFMOk,4167
|
|
5
5
|
tuieval/client.py,sha256=Vui7ucVv_GiKObGoVS_EWSZFUKPou9PWmutqMTrahXk,9313
|
|
6
|
-
tuieval/compare.py,sha256=
|
|
6
|
+
tuieval/compare.py,sha256=b-rVcMwq0Tos3oNKcpUP1HXULKhAQNAS4rlXYSY7t6k,25460
|
|
7
7
|
tuieval/engine.py,sha256=Amft1EL0iQVsiutfN7vQpwRXMCq_TYryDj0pMtY2suM,94195
|
|
8
|
-
tuieval/export.py,sha256=
|
|
9
|
-
tuieval/machines.py,sha256=
|
|
8
|
+
tuieval/export.py,sha256=dbHQgL_ubWtIPM1sLzAew_gETnrDJyd0zgqVA5vkpXM,9657
|
|
9
|
+
tuieval/machines.py,sha256=CeestY-_6h8ZKhJbBA_xyf0prqBvjvdgs2XJtCJYnNg,12974
|
|
10
10
|
tuieval/packs.py,sha256=Go-kTSPvYZriXMo4DvmqNNL-iD0J92MKLiNCnP2chJ8,10305
|
|
11
11
|
tuieval/profiles.py,sha256=soeLxjdB-l-OxSY8jZ4yjwfdSnsBV-QICcwKxRJS7nQ,3663
|
|
12
12
|
tuieval/remove.py,sha256=Vwh43nm9lVb9U9aN-h6CMYbkTwdR3UWfZObwweF72n0,5157
|
|
13
|
-
tuieval/run_evals.py,sha256=
|
|
13
|
+
tuieval/run_evals.py,sha256=uPzidbRPUduBwP8sKyzg-U8cka0p-RsbZlVis6qQTlk,37526
|
|
14
14
|
tuieval/scaffold.py,sha256=-AE5vKXeTtMZJ8OMk4UiIstoUWozTKjThKxZ87j0f9E,4720
|
|
15
15
|
tuieval/selftest.py,sha256=xKfR73wMTYcDRSCBYQTNyKeR9bYU5HIYUSg7oZvNCyI,9330
|
|
16
16
|
tuieval/tui.py,sha256=8diKtYE31ZUKq62rfMGdMkEbHgK6arRYpjFu-wcMVZs,131997
|
|
@@ -25,7 +25,7 @@ tuieval/graders/code.py,sha256=mNoPw-C_lNaXTTU57TJjhcVPTvdR400PxoE9Ow-tgHg,3703
|
|
|
25
25
|
tuieval/graders/rag.py,sha256=j_gEjmaXmY_hgKu7-7pVQpQ8Cwb49gadnwFurih-1Hw,1447
|
|
26
26
|
tuieval/graders/reply.py,sha256=BdkL7EWSSMoyjwCOFa97TTQMR1QyrpMH-ntcgjzDPEg,2199
|
|
27
27
|
tuieval/graders/tool_call.py,sha256=_QXURtOk4ZKBnxXVMzFL1c9jWLWksRlLZ4VqOCL0t90,3578
|
|
28
|
-
tuieval/templates/models.toml,sha256=
|
|
28
|
+
tuieval/templates/models.toml,sha256=bZRpQgj2LFeEQhnqkf0YUhrGybrugCilyWTg19_W4g4,7879
|
|
29
29
|
tuieval/templates/packs/answer/pack.toml,sha256=8bjMBxG1Zz3yxEnYCqRTNBJ-JGYlquqiK19p8URWwTc,1167
|
|
30
30
|
tuieval/templates/packs/answer/system.txt,sha256=m4V6X1i49Bas2HowZMWgfdDSPDHTObb_WxHXt0xw1Yo,248
|
|
31
31
|
tuieval/templates/packs/answer/tests.yaml,sha256=WiYPOknPcO9E6f1wXhLcfA0W5v-oK8MgZV956uWtcUQ,2660
|
|
@@ -42,8 +42,8 @@ tuieval/templates/packs/tool_call/pack.toml,sha256=n4bx0lGJHSmW41uVqhfIXJ2BWZpmT
|
|
|
42
42
|
tuieval/templates/packs/tool_call/system.txt,sha256=mKxu-lB-0YjLHXdgeFJa8x2C180OvcNz5cHv2N5Bg3U,254
|
|
43
43
|
tuieval/templates/packs/tool_call/tests.yaml,sha256=Nv7c4syrzJ9cU8EBKSK_1ZSFx9V02aCjW4-2TBCPT7s,1859
|
|
44
44
|
tuieval/templates/packs/tool_call/tools.yaml,sha256=HgxCkh0On2VwbHSsjAGM8tNCK7ZtKLGrs7lMCn3Fprs,1048
|
|
45
|
-
tuieval-0.2.0.
|
|
46
|
-
tuieval-0.2.0.
|
|
47
|
-
tuieval-0.2.0.
|
|
48
|
-
tuieval-0.2.0.
|
|
49
|
-
tuieval-0.2.0.
|
|
45
|
+
tuieval-0.2.0.dev2.dist-info/METADATA,sha256=ukHZagdns1Pj_mnQQsl9dxLIl40A-E8BtYIPx57I-cw,12079
|
|
46
|
+
tuieval-0.2.0.dev2.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
47
|
+
tuieval-0.2.0.dev2.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
|
|
48
|
+
tuieval-0.2.0.dev2.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
|
|
49
|
+
tuieval-0.2.0.dev2.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|