tuieval 0.2.0.dev6__tar.gz → 0.2.0.dev8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/CHANGELOG.md +2 -1
  2. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/PKG-INFO +1 -1
  3. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/models.md +3 -3
  4. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/_version.py +2 -2
  5. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/machines.py +11 -0
  6. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/run_evals.py +3 -1
  7. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/models.toml +2 -0
  8. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/tune.py +92 -27
  9. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/tests/test_tuieval.py +39 -5
  10. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/.gitignore +0 -0
  11. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/LICENSE +0 -0
  12. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/README.md +0 -0
  13. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/RELEASING.md +0 -0
  14. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/README.md +0 -0
  15. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/favicon.ico +0 -0
  16. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/readme-header.png +0 -0
  17. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/social-preview.png +0 -0
  18. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-1024.png +0 -0
  19. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-128.png +0 -0
  20. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.png +0 -0
  21. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.svg +0 -0
  22. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-256.png +0 -0
  23. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-32.png +0 -0
  24. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-48.png +0 -0
  25. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-512.png +0 -0
  26. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-64.png +0 -0
  27. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  28. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon.svg +0 -0
  29. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/images/tui-setup.png +0 -0
  30. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/docs/writing-packs.md +0 -0
  31. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/pyproject.toml +0 -0
  32. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/__init__.py +0 -0
  33. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/__main__.py +0 -0
  34. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/cli.py +0 -0
  35. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/client.py +0 -0
  36. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/compare.py +0 -0
  37. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/engine.py +0 -0
  38. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/export.py +0 -0
  39. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/__init__.py +0 -0
  40. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/answer.py +0 -0
  41. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/code.py +0 -0
  42. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/rag.py +0 -0
  43. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/reply.py +0 -0
  44. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/graders/tool_call.py +0 -0
  45. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/packs.py +0 -0
  46. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/profiles.py +0 -0
  47. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/remove.py +0 -0
  48. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/scaffold.py +0 -0
  49. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/selftest.py +0 -0
  50. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  51. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/system.txt +0 -0
  52. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  53. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/pack.toml +0 -0
  54. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/system.txt +0 -0
  55. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  56. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  57. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/system.txt +0 -0
  58. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  59. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  60. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/system.txt +0 -0
  61. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  62. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  63. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  64. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  65. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  66. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/tui.py +0 -0
  67. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/verdict.py +0 -0
  68. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/watch_proxy.py +0 -0
  69. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/workspace.py +0 -0
  70. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/src/tuieval/yamlout.py +0 -0
  71. {tuieval-0.2.0.dev6 → tuieval-0.2.0.dev8}/tests/mock_server.py +0 -0
@@ -2,7 +2,8 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
- - `tuieval tune`: swapping the defaults already cause is a warning, not a failure; an option is rejected only if it swaps more than the defaults.
5
+ - `tuieval tune` scores options on projected full-length answers (`[tune] answer_tokens`, default 1024), so generation speed counts as in real runs.
6
+ - `tuieval tune` reports an option's memory cost instead of rejecting it; it rejects only on critical memory pressure or over `[tune] swap_limit_mb`.
6
7
  - Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
7
8
  - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
8
9
  - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev6
3
+ Version: 0.2.0.dev8
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -77,8 +77,8 @@ The best server flags differ per model and per machine, so tuieval splits them b
77
77
  `tuieval tune <model>` (or `t` on the setup screen, for the ticked models) finds the fastest speed flags for a model on this machine:
78
78
 
79
79
  1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
80
- 2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
81
- 3. An **output guard** rejects any option that changes greedy answers beyond noise. An option under which macOS swaps more than with the defaults is rejected (it costs memory). If the defaults already push other apps to swap, the tune warns and carries on.
80
+ 2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces. Options are scored on the **projected** time: each prompt's measured reading time plus a full-length answer (`[tune] answer_tokens`, default 1024) at its measured generation speed. The workload itself generates 128 tokens per prompt, but eval answers run to thousands, so generation speed counts as much as in real runs.
81
+ 3. An **output guard** rejects any option that changes greedy answers beyond noise. Memory is a cost, not a reason to reject: an option that makes macOS push other apps' idle memory to swap can still win, and the result warns about it. An option is rejected only if macOS memory pressure turns critical while it runs, or, with `[tune] swap_limit_mb` set, if it swaps more than that beyond the defaults.
82
82
 
83
83
  Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
84
84
 
@@ -128,7 +128,7 @@ On Apple Silicon Macs, once the system's GPU allocations pass about half of RAM,
128
128
  | `request` | fields added to every request (e.g. `{ cache_prompt = false }`) |
129
129
  | `perf` | speed-only flags always applied |
130
130
  | `tune` | speed-only knobs for `tuieval tune` |
131
- | `tune_objective` | `total` (default: workload time) or `decode` (tokens/s after a warm-up pass) |
131
+ | `tune_objective` | `projected` (default: workload time with full-length answers), `total` (workload time as run) or `decode` (tokens/s after a warm-up pass) |
132
132
  | `version_cmd` | prints the server version, recorded with every result |
133
133
  | `health` | readiness path for servers without `/v1/models` (must return JSON with `model`) |
134
134
  | `before_start` | a command run before starting the server (e.g. to free memory another process holds) |
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev6'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev6')
21
+ __version__ = version = '0.2.0.dev8'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev8')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -225,6 +225,17 @@ def swapped_out_bytes():
225
225
  RESIDENCY_FRACTION = 0.5
226
226
 
227
227
 
228
+ def memory_pressure_level():
229
+ """macOS memory pressure: 1 normal, 2 warning, 4 critical (the system starts ending processes).
230
+ None elsewhere."""
231
+ try:
232
+ out = subprocess.run(["sysctl", "-n", "kern.memorystatus_vm_pressure_level"], capture_output=True,
233
+ text=True, timeout=5).stdout
234
+ return int(out.strip())
235
+ except (OSError, subprocess.TimeoutExpired, ValueError):
236
+ return None
237
+
238
+
228
239
  def gpu_residency_gb(machine, override=None):
229
240
  """GPU memory the driver keeps resident without churning (models.toml
230
241
  [machines.<id>] gpu_residency_gb overrides the half-of-RAM default)."""
@@ -523,7 +523,9 @@ def cmd_tune(argv):
523
523
  pr = result["profile"]
524
524
  ms = pr["measured"]
525
525
  gain = ", " + tune.gain_text(ms) if tune.gain_text(ms) else ""
526
- result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else f"{ms['total_s']}s for the workload"
526
+ result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else \
527
+ f"{ms['projected_s']}s projected for the workload" if ms.get("objective") == "projected" and ms.get("projected_s") \
528
+ else f"{ms['total_s']}s for the workload"
527
529
  say(f"{label}: {' '.join(pr['args']) or '(no knobs)'} -> {result}{gain}", GREEN)
528
530
  if pr["meta"].get("warm_start"):
529
531
  ws = pr["meta"]["warm_start"]
@@ -101,6 +101,8 @@ headers = { "X-Title" = "tuieval" }
101
101
  # request_timeout_s = 600 # a tuning request that takes longer fails that candidate
102
102
  # warm_start = true # start from a tuned model with the same architecture and shapes (a fine-tune)
103
103
  # warm_retest = ["spec", "ubatch"] # the knobs a warm start re-tries; the rest are inherited
104
+ # answer_tokens = 1024 # answer length options are scored for (projected objective)
105
+ # swap_limit_mb = 500 # reject options that swap more than this beyond the defaults (default: report only)
104
106
 
105
107
  # Optional: llama-bench for the tuner's fast first stage (found on PATH as llama-bench otherwise).
106
108
  # [bench]
@@ -18,18 +18,24 @@ retrained or dropped its MTP layers, and its quant mix shifts the best micro-bat
18
18
  it tunes in full instead. `tuieval tune --cold` (or [tune] warm_start = false) always tunes in full.
19
19
 
20
20
  One objective per server (models.toml `tune_objective`):
21
- total (default) total seconds for the workload: short prompts, medium ones and one long
22
- (~8k-token) prompt, fixed generation length
23
- decode generated tokens per second, measured on a second pass over a few prompts after a
24
- warm-up pass (for servers whose decode speed grows as their caches warm)
21
+ projected (default) seconds the workload would take with full-length answers: each prompt's
22
+ measured reading time plus [tune] answer_tokens (default 1024) at its measured
23
+ generation speed. The workload generates 128 tokens per prompt, but eval answers run
24
+ to thousands, so generation speed counts as much as it does in real runs.
25
+ total wall-clock seconds for the workload as run (128-token answers)
26
+ decode generated tokens per second, measured on a second pass over a few prompts after a
27
+ warm-up pass (for servers whose decode speed grows as their caches warm)
25
28
  An output guard compares greedy answers with the default flags; an option that changes them beyond
26
29
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
30
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
31
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. A candidate under which macOS swaps more than with the defaults is rejected (it costs
30
- memory). Swapping the defaults already cause (the model, its context, other apps) is a warning, and
31
- swapping while the model loads is only noted. Servers without tune knobs are only
32
- measured.
32
+ outdated.
33
+
34
+ Memory: a candidate is rejected if macOS memory pressure turns critical while it works, or, with
35
+ [tune] swap_limit_mb set, if it swaps more than that beyond the defaults. Otherwise memory is a
36
+ reported cost, not a reason to reject: a flag that pushes other apps' idle memory to swap can still
37
+ win, and the result warns about it. Swapping the defaults already cause is a warning too, and
38
+ swapping while the model loads is only noted. Servers without tune knobs are only measured.
33
39
 
34
40
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
35
41
  {p_minus_2} and {all} are this machine's core counts.
@@ -50,7 +56,10 @@ from . import profiles
50
56
 
51
57
  GEN_TOKENS = 128 # generated per workload prompt
52
58
  DECODE_TOKENS = 256 # generated per prompt for the decode objective
53
- SWAP_LIMIT = 256 * 2**20 # a candidate that swaps this much more than the defaults is rejected
59
+ SWAP_NOTABLE = 256 * 2**20 # swap worth a warning (bytes); rejecting for swap needs [tune] swap_limit_mb
60
+ ANSWER_TOKENS = 1024 # answer length the projected objective scores ([tune] answer_tokens)
61
+ PRESSURE_CRITICAL = 4 # macOS memory pressure level that rejects a candidate
62
+ PRESSURE_EVERY_S = 2
54
63
  GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
55
64
  # is rejected even before it stalls: servers grow as requests arrive
56
65
  REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
@@ -72,16 +81,21 @@ class Measure:
72
81
  error: str = ""
73
82
  facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
74
83
  swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
84
+ projected_s: float | None = None # the workload with full-length answers (projected objective)
85
+ pressure: int | None = None # highest macOS memory pressure level while it worked
75
86
 
76
87
  def score(self, objective):
77
88
  """Lower is better."""
78
89
  if objective == "decode":
79
90
  return 1 / self.tg_tps if self.tg_tps else float("inf")
91
+ if objective == "projected" and self.projected_s is not None:
92
+ return self.projected_s
80
93
  return self.total_s
81
94
 
82
95
  def summary(self):
83
96
  facts = "".join(f" · {k} {v}" for k, v in self.facts.items())
84
- return (f"{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
97
+ projected = f"{self.projected_s:.0f}s projected · " if self.projected_s is not None else ""
98
+ return (f"{projected}{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
85
99
  f" · ttft {self.ttft_s}s{facts}")
86
100
 
87
101
 
@@ -194,20 +208,21 @@ def _request(eng, base_url, body, name, emit, limit):
194
208
 
195
209
 
196
210
  def measure(eng, m, base_url, work, gen_tokens=GEN_TOKENS, passes=1, emit=lambda *a, **k: None,
197
- limit=REQUEST_LIMIT_S):
211
+ limit=REQUEST_LIMIT_S, answer_tokens=ANSWER_TOKENS):
198
212
  """Time the workload (after a warm-up request). With passes > 1 the earlier passes warm the
199
- server's caches and only the last pass is reported."""
213
+ server's caches and only the last pass is reported. answer_tokens: the answer length the
214
+ projected time is for."""
200
215
  out = Measure()
201
216
  for p in range(passes):
202
217
  if passes > 1:
203
218
  emit("tune_progress", message=f" pass {p + 1} of {passes}" + (" (warm-up)" if p < passes - 1 else " (timed)"))
204
- out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit)
219
+ out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens)
205
220
  if out.error:
206
221
  break
207
222
  return out
208
223
 
209
224
 
210
- def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
225
+ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens=ANSWER_TOKENS):
211
226
  sampling = engine_mod.effective_sampling(eng.cfg, m)
212
227
  request = eng.cfg["servers"][m["server"]].get("request", {})
213
228
  timeout = eng.cfg["defaults"]["request_timeout_ms"] / 1000
@@ -218,7 +233,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
218
233
  if warm["error"]:
219
234
  out.error = warm["error"]
220
235
  return out
221
- ttfts, p_tok, p_s, g_tok, g_s = [], 0, 0.0, 0, 0.0
236
+ ttfts, p_tok, p_s, g_tok, g_s, projected = [], 0, 0.0, 0, 0.0, 0.0
222
237
  for name, msgs in work:
223
238
  eng._check()
224
239
  emit("tune_progress", message=f" {name} ({gen_tokens} tokens)")
@@ -230,6 +245,13 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
230
245
  emit("tune_progress", message=f" done: {met['completion_tokens']} tokens in {res['total_s']:.1f}s"
231
246
  + (f", {met['gen_tps']} tok/s" if met["gen_tps"] else ""))
232
247
  out.total_s += res["total_s"]
248
+ # The same request with a full-length answer: its reading time plus answer_tokens at the
249
+ # generation speed measured here (as run if the server doesn't report that speed).
250
+ if met["completion_tokens"] and met["gen_tps"]:
251
+ reading = max(res["total_s"] - met["completion_tokens"] / met["gen_tps"], 0.0)
252
+ projected += reading + answer_tokens / met["gen_tps"]
253
+ else:
254
+ projected += res["total_s"]
233
255
  out.texts.append((res["reasoning"] or "") + (res["answer"] or ""))
234
256
  if name != "long" and res.get("ttft_s") is not None:
235
257
  ttfts.append(res["ttft_s"])
@@ -238,6 +260,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
238
260
  if met["completion_tokens"] and met["gen_tps"]:
239
261
  g_tok, g_s = g_tok + met["completion_tokens"], g_s + met["completion_tokens"] / met["gen_tps"]
240
262
  out.total_s = round(out.total_s, 2)
263
+ out.projected_s = round(projected, 1)
241
264
  out.ttft_s = round(sum(ttfts) / len(ttfts), 3) if ttfts else None
242
265
  out.pp_tps = round(p_tok / p_s, 1) if p_s else None
243
266
  out.tg_tps = round(g_tok / g_s, 2) if g_s else None
@@ -246,12 +269,37 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
246
269
 
247
270
  def gain_pct(objective, base, best):
248
271
  """How much better the chosen settings are than the defaults: % more decode tok/s for the
249
- decode objective, % less total time otherwise."""
272
+ decode objective, % less (projected or total) time otherwise."""
250
273
  if base is None:
251
274
  return None
252
275
  if objective == "decode":
253
276
  return round(100 * (best.tg_tps / base.tg_tps - 1)) if base.tg_tps and best.tg_tps else None
254
- return round(100 * (1 - best.total_s / base.total_s)) if base.total_s else None
277
+ b, n = base.score(objective), best.score(objective)
278
+ return round(100 * (1 - n / b)) if b else None
279
+
280
+
281
+ class PressureWatch:
282
+ """The highest macOS memory pressure level seen while a block runs (None where unknown)."""
283
+
284
+ def __enter__(self):
285
+ self.level, self._stop = machines.memory_pressure_level(), threading.Event()
286
+ self._thread = threading.Thread(target=self._run, daemon=True)
287
+ self._thread.start()
288
+ return self
289
+
290
+ def _sample(self):
291
+ lvl = machines.memory_pressure_level()
292
+ if lvl is not None:
293
+ self.level = max(self.level or 0, lvl)
294
+
295
+ def _run(self):
296
+ while not self._stop.wait(PRESSURE_EVERY_S):
297
+ self._sample()
298
+
299
+ def __exit__(self, *exc):
300
+ self._stop.set()
301
+ self._thread.join(PRESSURE_EVERY_S + 1)
302
+ self._sample()
255
303
 
256
304
 
257
305
  def gain_text(measured):
@@ -453,12 +501,14 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
453
501
  sv = eng.serving(m)
454
502
  if not sv.fits:
455
503
  raise engine_mod.ModelFailed(sv.fit_note)
456
- objective = server.get("tune_objective", "total")
504
+ objective = server.get("tune_objective", "projected")
457
505
  if objective == "decode":
458
506
  work, gen_tokens, passes = decode_workload(eng), DECODE_TOKENS, 2
459
507
  else:
460
508
  work, gen_tokens, passes = workload(eng, sv.ctx), GEN_TOKENS, 1
461
509
  settings = eng.cfg.get("tune", {})
510
+ answer_tokens = settings.get("answer_tokens", ANSWER_TOKENS)
511
+ swap_limit = settings.get("swap_limit_mb") # MB beyond the defaults' swap that rejects; None: report only
462
512
  threshold = settings.get("guard_similarity", 0.6)
463
513
  guard = not server.get("outputs_depend_on_machine")
464
514
  knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
@@ -484,18 +534,20 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
484
534
  # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
485
535
  # that's recorded, not held against the settings. Swapping while serving is.
486
536
  loaded = machines.swapped_out_bytes()
487
- r = measure(eng, m, url, work, gen_tokens, passes, emit,
488
- settings.get("request_timeout_s", REQUEST_LIMIT_S))
537
+ with PressureWatch() as pressure:
538
+ r = measure(eng, m, url, work, gen_tokens, passes, emit,
539
+ settings.get("request_timeout_s", REQUEST_LIMIT_S), answer_tokens=answer_tokens)
489
540
  end = machines.swapped_out_bytes()
541
+ r.pressure = pressure.level
490
542
  if None not in (swap0, loaded, end):
491
543
  swapped = (end - loaded) / 2**20
492
544
  at_load = (loaded - swap0) / 2**20
493
- if at_load > SWAP_LIMIT / 2**20:
545
+ if at_load > SWAP_NOTABLE / 2**20:
494
546
  emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
495
547
  "(close apps for more headroom)")
496
548
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
497
549
  r.facts = dict(info["facts"])
498
- if swapped is not None and at_load > SWAP_LIMIT / 2**20:
550
+ if swapped is not None and at_load > SWAP_NOTABLE / 2**20:
499
551
  r.facts["load_swapped_mb"] = round(at_load)
500
552
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
501
553
  # harmless; sustained allocation over the limit is what makes the driver churn).
@@ -516,10 +568,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
516
568
  r = Measure(error=str(e).splitlines()[0])
517
569
  r.swapped_mb = swapped
518
570
  # Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
519
- # a flag; only swapping beyond the defaults' does.
571
+ # a flag; only swapping beyond the defaults' does, and it's a cost to report unless it harms:
572
+ # critical memory pressure, or more than the user's swap_limit_mb.
520
573
  extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
521
- if not r.error and extra is not None and extra > SWAP_LIMIT / 2**20:
522
- r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (uses too much memory)"
574
+ if extra is not None and extra > SWAP_NOTABLE / 2**20:
575
+ r.facts["extra_swap_mb"] = round(extra)
576
+ if not r.error and r.pressure is not None and r.pressure >= PRESSURE_CRITICAL:
577
+ r.error = "macOS memory pressure turned critical (memory too tight)"
578
+ elif not r.error and swap_limit is not None and extra is not None and extra > swap_limit:
579
+ r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (over swap_limit_mb = {swap_limit})"
523
580
  cache[k] = r
524
581
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
525
582
  return r
@@ -527,7 +584,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
527
584
  def set_baseline(r):
528
585
  """The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
529
586
  swap_ref[0] = r.swapped_mb or 0
530
- if swap_ref[0] > SWAP_LIMIT / 2**20:
587
+ if swap_ref[0] > SWAP_NOTABLE / 2**20:
531
588
  warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
532
589
  "this machine; close other apps or lower this model's context for more headroom")
533
590
  emit("tune_step", message=" warning: " + warnings[-1])
@@ -615,6 +672,11 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
615
672
  if warm_info:
616
673
  notes.insert(0, f"warm start from {warm_info['from']}: inherited "
617
674
  f"{', '.join(warm_info['inherited']) or 'nothing'}")
675
+ if best_r.facts.get("extra_swap_mb") and best_r is not base:
676
+ warnings.append(f"the chosen flags use more memory than the defaults: macOS swapped "
677
+ f"~{best_r.facts['extra_swap_mb']} MB more of other apps' memory; close apps during runs, "
678
+ "or set [tune] swap_limit_mb to rule such flags out")
679
+ emit("tune_step", message=" warning: " + warnings[-1])
618
680
  size = sv.identity.get("model_bytes")
619
681
  profile = {
620
682
  "args": resolve(knob_opts, best),
@@ -623,12 +685,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
623
685
  "tg_tps": best_r.tg_tps, "load_s": best_r.load_s,
624
686
  "default_total_s": base.total_s if knob_opts else None,
625
687
  "default_tg_tps": base.tg_tps if knob_opts else None, "objective": objective,
688
+ "projected_s": best_r.projected_s,
689
+ "default_projected_s": base.projected_s if knob_opts else None,
626
690
  **{f"server_{k}": v for k, v in best_r.facts.items()},
627
691
  "gain_pct": gain_pct(objective, base, best_r)},
628
692
  "meta": {"method": method, "date": profiles.now(), "machine": sv.machine.summary,
629
693
  "server_version": eng.server_version(m), "model_file": os.path.basename(m["model"]),
630
694
  "model_bytes": size, "ctx": sv.ctx, "server_starts": starts[0],
631
- "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
695
+ "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else "")
696
+ + (f", scored for {answer_tokens}-token answers" if objective == "projected" else ""),
632
697
  "llama_bench": bool(bench_info), "answer_guard": guard,
633
698
  **({"warm_start": warm_info} if knob_opts and warm_info else {}),
634
699
  "rejected": rejected, "notes": notes[:10], "warnings": warnings},
@@ -308,12 +308,17 @@ class WarmTune(unittest.TestCase):
308
308
  yield "http://x", {"load_s": 1, "facts": {}}
309
309
  eng.serve = serve
310
310
 
311
- def measure(eng_, m, url, work, *a):
311
+ def measure(eng_, m, url, work, *a, **k):
312
312
  args = eng.starts[-1]
313
313
  fast = {"6": 1, "256": 1, "draft-mtp": 3}
314
314
  return tune.Measure(total_s=10 - sum(v for k, v in fast.items() if k in args), texts=["same"])
315
315
  self.orig = tune.measure
316
316
  tune.measure = measure
317
+ from tuieval import machines
318
+ self.pressure = lambda args: 1 # macOS memory pressure per start's flags: normal
319
+ orig_pressure = machines.memory_pressure_level
320
+ machines.memory_pressure_level = lambda: self.pressure(eng.starts[-1] if eng.starts else [])
321
+ self.addCleanup(setattr, machines, "memory_pressure_level", orig_pressure)
317
322
  self.eng, self.tune, self.profiles = eng, tune, profiles
318
323
  for label, args in (("sib", ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]),
319
324
  ("other", ["-t", "8"])):
@@ -352,12 +357,41 @@ class WarmTune(unittest.TestCase):
352
357
  self.assertEqual(p["args"], ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]) # tuned as usual
353
358
  self.assertIn("800 MB", p["meta"]["warnings"][0])
354
359
 
355
- def test_a_flag_that_swaps_more_than_the_defaults_is_rejected(self):
356
- # -ub 256 is the fastest micro-batch here, but it costs 600 MB of swap the defaults don't
360
+ def test_a_flag_that_costs_memory_still_wins_with_a_warning(self):
361
+ # -ub 256 is the fastest micro-batch here, and costs 600 MB of swap the defaults don't
362
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
363
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
364
+ self.assertIn("256", p["args"])
365
+ self.assertTrue(any("~500 MB more" in w for w in p["meta"]["warnings"]))
366
+
367
+ def test_swap_limit_rules_out_flags_that_cost_memory(self):
368
+ self.eng.cfg["tune"] = {"swap_limit_mb": 256}
357
369
  self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
358
370
  p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
359
371
  self.assertNotIn("256", p["args"])
360
- self.assertTrue(any("500 MB more than with the defaults" in n for n in p["meta"]["notes"]))
372
+ self.assertTrue(any("over swap_limit_mb = 256" in n for n in p["meta"]["notes"]))
373
+
374
+ def test_critical_memory_pressure_rejects_a_flag(self):
375
+ self.pressure = lambda args: 4 if "draft-mtp" in args else 1
376
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
377
+ self.assertNotIn("draft-mtp", p["args"])
378
+ self.assertTrue(any("memory pressure turned critical" in n for n in p["meta"]["notes"]))
379
+
380
+ def test_projected_time_scores_full_length_answers(self):
381
+ # 10 s for 128 tokens at 16 tok/s: 2 s reading the prompt + 8 s generating
382
+ res = {"error": None, "total_s": 10.0, "ttft_s": 2.0, "reasoning": "", "answer": "x",
383
+ "usage": {"completion_tokens": 128, "prompt_tokens": 50},
384
+ "timings": {"predicted_per_second": 16.0, "prompt_per_second": 25.0}}
385
+ self.eng.cfg["sampling"] = {"temperature": 0}
386
+ self.eng._check = lambda: None
387
+ orig = self.tune._request
388
+ self.tune._request = lambda *a, **k: dict(res)
389
+ self.addCleanup(setattr, self.tune, "_request", orig)
390
+ out = self.tune._measure_once(self.eng, self.eng.model("new"), "http://x", [("p", [])], 128,
391
+ lambda *a, **k: None, 60, answer_tokens=1024)
392
+ self.assertEqual((out.total_s, out.projected_s), (10.0, 66.0)) # 2 s + 1024 / 16
393
+ self.assertEqual(out.score("projected"), 66.0)
394
+ self.assertEqual(out.score("total"), 10.0)
361
395
 
362
396
  def test_family_ignores_the_mtp_layer(self):
363
397
  # one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
@@ -397,7 +431,7 @@ class WarmTune(unittest.TestCase):
397
431
  self.profiles.save("mac", "sib", {"args": ["-t", "8", "-ub", "1024"], "meta": {"method": "tuned",
398
432
  "server_version": "v1"}}, self.eng.tuning_dir)
399
433
  orig = self.tune.measure
400
- self.tune.measure = lambda *a: self.tune.Measure(
434
+ self.tune.measure = lambda *a, **k: self.tune.Measure(
401
435
  total_s=99 if "1024" in self.eng.starts[-1] else orig(*a).total_s, texts=["same"])
402
436
  p = self.tune.tune(self.eng, "new", use_bench=False)
403
437
  self.assertNotIn("warm_start", p["meta"])
File without changes
File without changes
File without changes
File without changes