tuieval 0.2.0.dev6__tar.gz → 0.2.0.dev8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/CHANGELOG.md +2 -1
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/PKG-INFO +1 -1
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/models.md +3 -3
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/_version.py +2 -2
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/machines.py +11 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/run_evals.py +3 -1
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/models.toml +2 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/tune.py +92 -27
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/tests/test_tuieval.py +39 -5
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/.gitignore +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/LICENSE +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/README.md +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/RELEASING.md +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/README.md +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/favicon.ico +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/readme-header.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/social-preview.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-1024.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-128.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.svg +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-256.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-32.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-48.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-512.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-64.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-animated.svg +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon.svg +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/tui-setup.png +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/writing-packs.md +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/pyproject.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/__init__.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/__main__.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/cli.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/client.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/compare.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/engine.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/export.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/__init__.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/answer.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/code.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/rag.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/reply.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/tool_call.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/packs.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/profiles.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/remove.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/scaffold.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/selftest.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/pack.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/system.txt +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/pack.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/system.txt +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/tests.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/pack.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/system.txt +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/pack.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/system.txt +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/tui.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/verdict.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/watch_proxy.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/workspace.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/yamlout.py +0 -0
- {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/tests/mock_server.py +0 -0
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
-
- `tuieval tune
|
|
5
|
+
- `tuieval tune` scores options on projected full-length answers (`[tune] answer_tokens`, default 1024), so generation speed counts as in real runs.
|
|
6
|
+
- `tuieval tune` reports an option's memory cost instead of rejecting it; it rejects only on critical memory pressure or over `[tune] swap_limit_mb`.
|
|
6
7
|
- Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
|
|
7
8
|
- The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
|
|
8
9
|
- `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tuieval
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev8
|
|
4
4
|
Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ashe-wb/tuieval
|
|
6
6
|
Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
|
|
@@ -77,8 +77,8 @@ The best server flags differ per model and per machine, so tuieval splits them b
|
|
|
77
77
|
`tuieval tune <model>` (or `t` on the setup screen, for the ticked models) finds the fastest speed flags for a model on this machine:
|
|
78
78
|
|
|
79
79
|
1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
|
|
80
|
-
2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
|
|
81
|
-
3. An **output guard** rejects any option that changes greedy answers beyond noise.
|
|
80
|
+
2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces. Options are scored on the **projected** time: each prompt's measured reading time plus a full-length answer (`[tune] answer_tokens`, default 1024) at its measured generation speed. The workload itself generates 128 tokens per prompt, but eval answers run to thousands, so generation speed counts as much as in real runs.
|
|
81
|
+
3. An **output guard** rejects any option that changes greedy answers beyond noise. Memory is a cost, not a reason to reject: an option that makes macOS push other apps' idle memory to swap can still win, and the result warns about it. An option is rejected only if macOS memory pressure turns critical while it runs, or, with `[tune] swap_limit_mb` set, if it swaps more than that beyond the defaults.
|
|
82
82
|
|
|
83
83
|
Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
|
|
84
84
|
|
|
@@ -128,7 +128,7 @@ On Apple Silicon Macs, once the system's GPU allocations pass about half of RAM,
|
|
|
128
128
|
| `request` | fields added to every request (e.g. `{ cache_prompt = false }`) |
|
|
129
129
|
| `perf` | speed-only flags always applied |
|
|
130
130
|
| `tune` | speed-only knobs for `tuieval tune` |
|
|
131
|
-
| `tune_objective` | `
|
|
131
|
+
| `tune_objective` | `projected` (default: workload time with full-length answers), `total` (workload time as run) or `decode` (tokens/s after a warm-up pass) |
|
|
132
132
|
| `version_cmd` | prints the server version, recorded with every result |
|
|
133
133
|
| `health` | readiness path for servers without `/v1/models` (must return JSON with `model`) |
|
|
134
134
|
| `before_start` | a command run before starting the server (e.g. to free memory another process holds) |
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2, 0, '
|
|
21
|
+
__version__ = version = '0.2.0.dev8'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0, 'dev8')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -225,6 +225,17 @@ def swapped_out_bytes():
|
|
|
225
225
|
RESIDENCY_FRACTION = 0.5
|
|
226
226
|
|
|
227
227
|
|
|
228
|
+
def memory_pressure_level():
|
|
229
|
+
"""macOS memory pressure: 1 normal, 2 warning, 4 critical (the system starts ending processes).
|
|
230
|
+
None elsewhere."""
|
|
231
|
+
try:
|
|
232
|
+
out = subprocess.run(["sysctl", "-n", "kern.memorystatus_vm_pressure_level"], capture_output=True,
|
|
233
|
+
text=True, timeout=5).stdout
|
|
234
|
+
return int(out.strip())
|
|
235
|
+
except (OSError, subprocess.TimeoutExpired, ValueError):
|
|
236
|
+
return None
|
|
237
|
+
|
|
238
|
+
|
|
228
239
|
def gpu_residency_gb(machine, override=None):
|
|
229
240
|
"""GPU memory the driver keeps resident without churning (models.toml
|
|
230
241
|
[machines.<id>] gpu_residency_gb overrides the half-of-RAM default)."""
|
|
@@ -523,7 +523,9 @@ def cmd_tune(argv):
|
|
|
523
523
|
pr = result["profile"]
|
|
524
524
|
ms = pr["measured"]
|
|
525
525
|
gain = ", " + tune.gain_text(ms) if tune.gain_text(ms) else ""
|
|
526
|
-
result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else
|
|
526
|
+
result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else \
|
|
527
|
+
f"{ms['projected_s']}s projected for the workload" if ms.get("objective") == "projected" and ms.get("projected_s") \
|
|
528
|
+
else f"{ms['total_s']}s for the workload"
|
|
527
529
|
say(f"{label}: {' '.join(pr['args']) or '(no knobs)'} -> {result}{gain}", GREEN)
|
|
528
530
|
if pr["meta"].get("warm_start"):
|
|
529
531
|
ws = pr["meta"]["warm_start"]
|
|
@@ -101,6 +101,8 @@ headers = { "X-Title" = "tuieval" }
|
|
|
101
101
|
# request_timeout_s = 600 # a tuning request that takes longer fails that candidate
|
|
102
102
|
# warm_start = true # start from a tuned model with the same architecture and shapes (a fine-tune)
|
|
103
103
|
# warm_retest = ["spec", "ubatch"] # the knobs a warm start re-tries; the rest are inherited
|
|
104
|
+
# answer_tokens = 1024 # answer length options are scored for (projected objective)
|
|
105
|
+
# swap_limit_mb = 500 # reject options that swap more than this beyond the defaults (default: report only)
|
|
104
106
|
|
|
105
107
|
# Optional: llama-bench for the tuner's fast first stage (found on PATH as llama-bench otherwise).
|
|
106
108
|
# [bench]
|
|
@@ -18,18 +18,24 @@ retrained or dropped its MTP layers, and its quant mix shifts the best micro-bat
|
|
|
18
18
|
it tunes in full instead. `tuieval tune --cold` (or [tune] warm_start = false) always tunes in full.
|
|
19
19
|
|
|
20
20
|
One objective per server (models.toml `tune_objective`):
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
21
|
+
projected (default) seconds the workload would take with full-length answers: each prompt's
|
|
22
|
+
measured reading time plus [tune] answer_tokens (default 1024) at its measured
|
|
23
|
+
generation speed. The workload generates 128 tokens per prompt, but eval answers run
|
|
24
|
+
to thousands, so generation speed counts as much as it does in real runs.
|
|
25
|
+
total wall-clock seconds for the workload as run (128-token answers)
|
|
26
|
+
decode generated tokens per second, measured on a second pass over a few prompts after a
|
|
27
|
+
warm-up pass (for servers whose decode speed grows as their caches warm)
|
|
25
28
|
An output guard compares greedy answers with the default flags; an option that changes them beyond
|
|
26
29
|
noise is rejected, because speed flags must not change answers. Servers whose answers depend on
|
|
27
30
|
the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
|
|
28
31
|
become part of the results fingerprint instead, so retuning marks that machine's results
|
|
29
|
-
outdated.
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
32
|
+
outdated.
|
|
33
|
+
|
|
34
|
+
Memory: a candidate is rejected if macOS memory pressure turns critical while it works, or, with
|
|
35
|
+
[tune] swap_limit_mb set, if it swaps more than that beyond the defaults. Otherwise memory is a
|
|
36
|
+
reported cost, not a reason to reject: a flag that pushes other apps' idle memory to swap can still
|
|
37
|
+
win, and the result warns about it. Swapping the defaults already cause is a warning too, and
|
|
38
|
+
swapping while the model loads is only noted. Servers without tune knobs are only measured.
|
|
33
39
|
|
|
34
40
|
The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
|
|
35
41
|
{p_minus_2} and {all} are this machine's core counts.
|
|
@@ -50,7 +56,10 @@ from . import profiles
|
|
|
50
56
|
|
|
51
57
|
GEN_TOKENS = 128 # generated per workload prompt
|
|
52
58
|
DECODE_TOKENS = 256 # generated per prompt for the decode objective
|
|
53
|
-
|
|
59
|
+
SWAP_NOTABLE = 256 * 2**20 # swap worth a warning (bytes); rejecting for swap needs [tune] swap_limit_mb
|
|
60
|
+
ANSWER_TOKENS = 1024 # answer length the projected objective scores ([tune] answer_tokens)
|
|
61
|
+
PRESSURE_CRITICAL = 4 # macOS memory pressure level that rejects a candidate
|
|
62
|
+
PRESSURE_EVERY_S = 2
|
|
54
63
|
GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
|
|
55
64
|
# is rejected even before it stalls: servers grow as requests arrive
|
|
56
65
|
REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
|
|
@@ -72,16 +81,21 @@ class Measure:
|
|
|
72
81
|
error: str = ""
|
|
73
82
|
facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
|
|
74
83
|
swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
|
|
84
|
+
projected_s: float | None = None # the workload with full-length answers (projected objective)
|
|
85
|
+
pressure: int | None = None # highest macOS memory pressure level while it worked
|
|
75
86
|
|
|
76
87
|
def score(self, objective):
|
|
77
88
|
"""Lower is better."""
|
|
78
89
|
if objective == "decode":
|
|
79
90
|
return 1 / self.tg_tps if self.tg_tps else float("inf")
|
|
91
|
+
if objective == "projected" and self.projected_s is not None:
|
|
92
|
+
return self.projected_s
|
|
80
93
|
return self.total_s
|
|
81
94
|
|
|
82
95
|
def summary(self):
|
|
83
96
|
facts = "".join(f" · {k} {v}" for k, v in self.facts.items())
|
|
84
|
-
|
|
97
|
+
projected = f"{self.projected_s:.0f}s projected · " if self.projected_s is not None else ""
|
|
98
|
+
return (f"{projected}{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
|
|
85
99
|
f" · ttft {self.ttft_s}s{facts}")
|
|
86
100
|
|
|
87
101
|
|
|
@@ -194,20 +208,21 @@ def _request(eng, base_url, body, name, emit, limit):
|
|
|
194
208
|
|
|
195
209
|
|
|
196
210
|
def measure(eng, m, base_url, work, gen_tokens=GEN_TOKENS, passes=1, emit=lambda *a, **k: None,
|
|
197
|
-
limit=REQUEST_LIMIT_S):
|
|
211
|
+
limit=REQUEST_LIMIT_S, answer_tokens=ANSWER_TOKENS):
|
|
198
212
|
"""Time the workload (after a warm-up request). With passes > 1 the earlier passes warm the
|
|
199
|
-
server's caches and only the last pass is reported.
|
|
213
|
+
server's caches and only the last pass is reported. answer_tokens: the answer length the
|
|
214
|
+
projected time is for."""
|
|
200
215
|
out = Measure()
|
|
201
216
|
for p in range(passes):
|
|
202
217
|
if passes > 1:
|
|
203
218
|
emit("tune_progress", message=f" pass {p + 1} of {passes}" + (" (warm-up)" if p < passes - 1 else " (timed)"))
|
|
204
|
-
out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit)
|
|
219
|
+
out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens)
|
|
205
220
|
if out.error:
|
|
206
221
|
break
|
|
207
222
|
return out
|
|
208
223
|
|
|
209
224
|
|
|
210
|
-
def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
|
|
225
|
+
def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens=ANSWER_TOKENS):
|
|
211
226
|
sampling = engine_mod.effective_sampling(eng.cfg, m)
|
|
212
227
|
request = eng.cfg["servers"][m["server"]].get("request", {})
|
|
213
228
|
timeout = eng.cfg["defaults"]["request_timeout_ms"] / 1000
|
|
@@ -218,7 +233,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
|
|
|
218
233
|
if warm["error"]:
|
|
219
234
|
out.error = warm["error"]
|
|
220
235
|
return out
|
|
221
|
-
ttfts, p_tok, p_s, g_tok, g_s = [], 0, 0.0, 0, 0.0
|
|
236
|
+
ttfts, p_tok, p_s, g_tok, g_s, projected = [], 0, 0.0, 0, 0.0, 0.0
|
|
222
237
|
for name, msgs in work:
|
|
223
238
|
eng._check()
|
|
224
239
|
emit("tune_progress", message=f" {name} ({gen_tokens} tokens)")
|
|
@@ -230,6 +245,13 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
|
|
|
230
245
|
emit("tune_progress", message=f" done: {met['completion_tokens']} tokens in {res['total_s']:.1f}s"
|
|
231
246
|
+ (f", {met['gen_tps']} tok/s" if met["gen_tps"] else ""))
|
|
232
247
|
out.total_s += res["total_s"]
|
|
248
|
+
# The same request with a full-length answer: its reading time plus answer_tokens at the
|
|
249
|
+
# generation speed measured here (as run if the server doesn't report that speed).
|
|
250
|
+
if met["completion_tokens"] and met["gen_tps"]:
|
|
251
|
+
reading = max(res["total_s"] - met["completion_tokens"] / met["gen_tps"], 0.0)
|
|
252
|
+
projected += reading + answer_tokens / met["gen_tps"]
|
|
253
|
+
else:
|
|
254
|
+
projected += res["total_s"]
|
|
233
255
|
out.texts.append((res["reasoning"] or "") + (res["answer"] or ""))
|
|
234
256
|
if name != "long" and res.get("ttft_s") is not None:
|
|
235
257
|
ttfts.append(res["ttft_s"])
|
|
@@ -238,6 +260,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
|
|
|
238
260
|
if met["completion_tokens"] and met["gen_tps"]:
|
|
239
261
|
g_tok, g_s = g_tok + met["completion_tokens"], g_s + met["completion_tokens"] / met["gen_tps"]
|
|
240
262
|
out.total_s = round(out.total_s, 2)
|
|
263
|
+
out.projected_s = round(projected, 1)
|
|
241
264
|
out.ttft_s = round(sum(ttfts) / len(ttfts), 3) if ttfts else None
|
|
242
265
|
out.pp_tps = round(p_tok / p_s, 1) if p_s else None
|
|
243
266
|
out.tg_tps = round(g_tok / g_s, 2) if g_s else None
|
|
@@ -246,12 +269,37 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
|
|
|
246
269
|
|
|
247
270
|
def gain_pct(objective, base, best):
|
|
248
271
|
"""How much better the chosen settings are than the defaults: % more decode tok/s for the
|
|
249
|
-
decode objective, % less total time otherwise."""
|
|
272
|
+
decode objective, % less (projected or total) time otherwise."""
|
|
250
273
|
if base is None:
|
|
251
274
|
return None
|
|
252
275
|
if objective == "decode":
|
|
253
276
|
return round(100 * (best.tg_tps / base.tg_tps - 1)) if base.tg_tps and best.tg_tps else None
|
|
254
|
-
|
|
277
|
+
b, n = base.score(objective), best.score(objective)
|
|
278
|
+
return round(100 * (1 - n / b)) if b else None
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
class PressureWatch:
|
|
282
|
+
"""The highest macOS memory pressure level seen while a block runs (None where unknown)."""
|
|
283
|
+
|
|
284
|
+
def __enter__(self):
|
|
285
|
+
self.level, self._stop = machines.memory_pressure_level(), threading.Event()
|
|
286
|
+
self._thread = threading.Thread(target=self._run, daemon=True)
|
|
287
|
+
self._thread.start()
|
|
288
|
+
return self
|
|
289
|
+
|
|
290
|
+
def _sample(self):
|
|
291
|
+
lvl = machines.memory_pressure_level()
|
|
292
|
+
if lvl is not None:
|
|
293
|
+
self.level = max(self.level or 0, lvl)
|
|
294
|
+
|
|
295
|
+
def _run(self):
|
|
296
|
+
while not self._stop.wait(PRESSURE_EVERY_S):
|
|
297
|
+
self._sample()
|
|
298
|
+
|
|
299
|
+
def __exit__(self, *exc):
|
|
300
|
+
self._stop.set()
|
|
301
|
+
self._thread.join(PRESSURE_EVERY_S + 1)
|
|
302
|
+
self._sample()
|
|
255
303
|
|
|
256
304
|
|
|
257
305
|
def gain_text(measured):
|
|
@@ -453,12 +501,14 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
453
501
|
sv = eng.serving(m)
|
|
454
502
|
if not sv.fits:
|
|
455
503
|
raise engine_mod.ModelFailed(sv.fit_note)
|
|
456
|
-
objective = server.get("tune_objective", "
|
|
504
|
+
objective = server.get("tune_objective", "projected")
|
|
457
505
|
if objective == "decode":
|
|
458
506
|
work, gen_tokens, passes = decode_workload(eng), DECODE_TOKENS, 2
|
|
459
507
|
else:
|
|
460
508
|
work, gen_tokens, passes = workload(eng, sv.ctx), GEN_TOKENS, 1
|
|
461
509
|
settings = eng.cfg.get("tune", {})
|
|
510
|
+
answer_tokens = settings.get("answer_tokens", ANSWER_TOKENS)
|
|
511
|
+
swap_limit = settings.get("swap_limit_mb") # MB beyond the defaults' swap that rejects; None: report only
|
|
462
512
|
threshold = settings.get("guard_similarity", 0.6)
|
|
463
513
|
guard = not server.get("outputs_depend_on_machine")
|
|
464
514
|
knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
|
|
@@ -484,18 +534,20 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
484
534
|
# Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
|
|
485
535
|
# that's recorded, not held against the settings. Swapping while serving is.
|
|
486
536
|
loaded = machines.swapped_out_bytes()
|
|
487
|
-
|
|
488
|
-
|
|
537
|
+
with PressureWatch() as pressure:
|
|
538
|
+
r = measure(eng, m, url, work, gen_tokens, passes, emit,
|
|
539
|
+
settings.get("request_timeout_s", REQUEST_LIMIT_S), answer_tokens=answer_tokens)
|
|
489
540
|
end = machines.swapped_out_bytes()
|
|
541
|
+
r.pressure = pressure.level
|
|
490
542
|
if None not in (swap0, loaded, end):
|
|
491
543
|
swapped = (end - loaded) / 2**20
|
|
492
544
|
at_load = (loaded - swap0) / 2**20
|
|
493
|
-
if at_load >
|
|
545
|
+
if at_load > SWAP_NOTABLE / 2**20:
|
|
494
546
|
emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
|
|
495
547
|
"(close apps for more headroom)")
|
|
496
548
|
r.load_s = round(info["load_s"], 1) if info["load_s"] else None
|
|
497
549
|
r.facts = dict(info["facts"])
|
|
498
|
-
if swapped is not None and at_load >
|
|
550
|
+
if swapped is not None and at_load > SWAP_NOTABLE / 2**20:
|
|
499
551
|
r.facts["load_swapped_mb"] = round(at_load)
|
|
500
552
|
# Settled allocation after the timed pass (brief peaks while processing a prompt are
|
|
501
553
|
# harmless; sustained allocation over the limit is what makes the driver churn).
|
|
@@ -516,10 +568,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
516
568
|
r = Measure(error=str(e).splitlines()[0])
|
|
517
569
|
r.swapped_mb = swapped
|
|
518
570
|
# Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
|
|
519
|
-
# a flag; only swapping beyond the defaults' does
|
|
571
|
+
# a flag; only swapping beyond the defaults' does, and it's a cost to report unless it harms:
|
|
572
|
+
# critical memory pressure, or more than the user's swap_limit_mb.
|
|
520
573
|
extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
|
|
521
|
-
if
|
|
522
|
-
r.
|
|
574
|
+
if extra is not None and extra > SWAP_NOTABLE / 2**20:
|
|
575
|
+
r.facts["extra_swap_mb"] = round(extra)
|
|
576
|
+
if not r.error and r.pressure is not None and r.pressure >= PRESSURE_CRITICAL:
|
|
577
|
+
r.error = "macOS memory pressure turned critical (memory too tight)"
|
|
578
|
+
elif not r.error and swap_limit is not None and extra is not None and extra > swap_limit:
|
|
579
|
+
r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (over swap_limit_mb = {swap_limit})"
|
|
523
580
|
cache[k] = r
|
|
524
581
|
emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
|
|
525
582
|
return r
|
|
@@ -527,7 +584,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
527
584
|
def set_baseline(r):
|
|
528
585
|
"""The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
|
|
529
586
|
swap_ref[0] = r.swapped_mb or 0
|
|
530
|
-
if swap_ref[0] >
|
|
587
|
+
if swap_ref[0] > SWAP_NOTABLE / 2**20:
|
|
531
588
|
warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
|
|
532
589
|
"this machine; close other apps or lower this model's context for more headroom")
|
|
533
590
|
emit("tune_step", message=" warning: " + warnings[-1])
|
|
@@ -615,6 +672,11 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
615
672
|
if warm_info:
|
|
616
673
|
notes.insert(0, f"warm start from {warm_info['from']}: inherited "
|
|
617
674
|
f"{', '.join(warm_info['inherited']) or 'nothing'}")
|
|
675
|
+
if best_r.facts.get("extra_swap_mb") and best_r is not base:
|
|
676
|
+
warnings.append(f"the chosen flags use more memory than the defaults: macOS swapped "
|
|
677
|
+
f"~{best_r.facts['extra_swap_mb']} MB more of other apps' memory; close apps during runs, "
|
|
678
|
+
"or set [tune] swap_limit_mb to rule such flags out")
|
|
679
|
+
emit("tune_step", message=" warning: " + warnings[-1])
|
|
618
680
|
size = sv.identity.get("model_bytes")
|
|
619
681
|
profile = {
|
|
620
682
|
"args": resolve(knob_opts, best),
|
|
@@ -623,12 +685,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
623
685
|
"tg_tps": best_r.tg_tps, "load_s": best_r.load_s,
|
|
624
686
|
"default_total_s": base.total_s if knob_opts else None,
|
|
625
687
|
"default_tg_tps": base.tg_tps if knob_opts else None, "objective": objective,
|
|
688
|
+
"projected_s": best_r.projected_s,
|
|
689
|
+
"default_projected_s": base.projected_s if knob_opts else None,
|
|
626
690
|
**{f"server_{k}": v for k, v in best_r.facts.items()},
|
|
627
691
|
"gain_pct": gain_pct(objective, base, best_r)},
|
|
628
692
|
"meta": {"method": method, "date": profiles.now(), "machine": sv.machine.summary,
|
|
629
693
|
"server_version": eng.server_version(m), "model_file": os.path.basename(m["model"]),
|
|
630
694
|
"model_bytes": size, "ctx": sv.ctx, "server_starts": starts[0],
|
|
631
|
-
"workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else "")
|
|
695
|
+
"workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else "")
|
|
696
|
+
+ (f", scored for {answer_tokens}-token answers" if objective == "projected" else ""),
|
|
632
697
|
"llama_bench": bool(bench_info), "answer_guard": guard,
|
|
633
698
|
**({"warm_start": warm_info} if knob_opts and warm_info else {}),
|
|
634
699
|
"rejected": rejected, "notes": notes[:10], "warnings": warnings},
|
|
@@ -308,12 +308,17 @@ class WarmTune(unittest.TestCase):
|
|
|
308
308
|
yield "http://x", {"load_s": 1, "facts": {}}
|
|
309
309
|
eng.serve = serve
|
|
310
310
|
|
|
311
|
-
def measure(eng_, m, url, work, *a):
|
|
311
|
+
def measure(eng_, m, url, work, *a, **k):
|
|
312
312
|
args = eng.starts[-1]
|
|
313
313
|
fast = {"6": 1, "256": 1, "draft-mtp": 3}
|
|
314
314
|
return tune.Measure(total_s=10 - sum(v for k, v in fast.items() if k in args), texts=["same"])
|
|
315
315
|
self.orig = tune.measure
|
|
316
316
|
tune.measure = measure
|
|
317
|
+
from tuieval import machines
|
|
318
|
+
self.pressure = lambda args: 1 # macOS memory pressure per start's flags: normal
|
|
319
|
+
orig_pressure = machines.memory_pressure_level
|
|
320
|
+
machines.memory_pressure_level = lambda: self.pressure(eng.starts[-1] if eng.starts else [])
|
|
321
|
+
self.addCleanup(setattr, machines, "memory_pressure_level", orig_pressure)
|
|
317
322
|
self.eng, self.tune, self.profiles = eng, tune, profiles
|
|
318
323
|
for label, args in (("sib", ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]),
|
|
319
324
|
("other", ["-t", "8"])):
|
|
@@ -352,12 +357,41 @@ class WarmTune(unittest.TestCase):
|
|
|
352
357
|
self.assertEqual(p["args"], ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]) # tuned as usual
|
|
353
358
|
self.assertIn("800 MB", p["meta"]["warnings"][0])
|
|
354
359
|
|
|
355
|
-
def
|
|
356
|
-
# -ub 256 is the fastest micro-batch here,
|
|
360
|
+
def test_a_flag_that_costs_memory_still_wins_with_a_warning(self):
|
|
361
|
+
# -ub 256 is the fastest micro-batch here, and costs 600 MB of swap the defaults don't
|
|
362
|
+
self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
|
|
363
|
+
p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
|
|
364
|
+
self.assertIn("256", p["args"])
|
|
365
|
+
self.assertTrue(any("~500 MB more" in w for w in p["meta"]["warnings"]))
|
|
366
|
+
|
|
367
|
+
def test_swap_limit_rules_out_flags_that_cost_memory(self):
|
|
368
|
+
self.eng.cfg["tune"] = {"swap_limit_mb": 256}
|
|
357
369
|
self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
|
|
358
370
|
p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
|
|
359
371
|
self.assertNotIn("256", p["args"])
|
|
360
|
-
self.assertTrue(any("
|
|
372
|
+
self.assertTrue(any("over swap_limit_mb = 256" in n for n in p["meta"]["notes"]))
|
|
373
|
+
|
|
374
|
+
def test_critical_memory_pressure_rejects_a_flag(self):
|
|
375
|
+
self.pressure = lambda args: 4 if "draft-mtp" in args else 1
|
|
376
|
+
p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
|
|
377
|
+
self.assertNotIn("draft-mtp", p["args"])
|
|
378
|
+
self.assertTrue(any("memory pressure turned critical" in n for n in p["meta"]["notes"]))
|
|
379
|
+
|
|
380
|
+
def test_projected_time_scores_full_length_answers(self):
|
|
381
|
+
# 10 s for 128 tokens at 16 tok/s: 2 s reading the prompt + 8 s generating
|
|
382
|
+
res = {"error": None, "total_s": 10.0, "ttft_s": 2.0, "reasoning": "", "answer": "x",
|
|
383
|
+
"usage": {"completion_tokens": 128, "prompt_tokens": 50},
|
|
384
|
+
"timings": {"predicted_per_second": 16.0, "prompt_per_second": 25.0}}
|
|
385
|
+
self.eng.cfg["sampling"] = {"temperature": 0}
|
|
386
|
+
self.eng._check = lambda: None
|
|
387
|
+
orig = self.tune._request
|
|
388
|
+
self.tune._request = lambda *a, **k: dict(res)
|
|
389
|
+
self.addCleanup(setattr, self.tune, "_request", orig)
|
|
390
|
+
out = self.tune._measure_once(self.eng, self.eng.model("new"), "http://x", [("p", [])], 128,
|
|
391
|
+
lambda *a, **k: None, 60, answer_tokens=1024)
|
|
392
|
+
self.assertEqual((out.total_s, out.projected_s), (10.0, 66.0)) # 2 s + 1024 / 16
|
|
393
|
+
self.assertEqual(out.score("projected"), 66.0)
|
|
394
|
+
self.assertEqual(out.score("total"), 10.0)
|
|
361
395
|
|
|
362
396
|
def test_family_ignores_the_mtp_layer(self):
|
|
363
397
|
# one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
|
|
@@ -397,7 +431,7 @@ class WarmTune(unittest.TestCase):
|
|
|
397
431
|
self.profiles.save("mac", "sib", {"args": ["-t", "8", "-ub", "1024"], "meta": {"method": "tuned",
|
|
398
432
|
"server_version": "v1"}}, self.eng.tuning_dir)
|
|
399
433
|
orig = self.tune.measure
|
|
400
|
-
self.tune.measure = lambda *a: self.tune.Measure(
|
|
434
|
+
self.tune.measure = lambda *a, **k: self.tune.Measure(
|
|
401
435
|
total_s=99 if "1024" in self.eng.starts[-1] else orig(*a).total_s, texts=["same"])
|
|
402
436
|
p = self.tune.tune(self.eng, "new", use_bench=False)
|
|
403
437
|
self.assertNotIn("warm_start", p["meta"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|