tuieval 0.2.0.dev5__tar.gz → 0.2.0.dev8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/CHANGELOG.md +2 -1
  2. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/PKG-INFO +1 -1
  3. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/models.md +3 -3
  4. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/_version.py +2 -2
  5. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/machines.py +11 -0
  6. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/run_evals.py +5 -1
  7. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/models.toml +2 -0
  8. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/tui.py +2 -0
  9. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/tune.py +108 -26
  10. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/tests/test_tuieval.py +54 -11
  11. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/.gitignore +0 -0
  12. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/LICENSE +0 -0
  13. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/README.md +0 -0
  14. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/RELEASING.md +0 -0
  15. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/README.md +0 -0
  16. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/favicon.ico +0 -0
  17. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/readme-header.png +0 -0
  18. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/social-preview.png +0 -0
  19. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-1024.png +0 -0
  20. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-128.png +0 -0
  21. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.png +0 -0
  22. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-16.svg +0 -0
  23. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-256.png +0 -0
  24. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-32.png +0 -0
  25. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-48.png +0 -0
  26. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-512.png +0 -0
  27. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-64.png +0 -0
  28. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  29. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/brand/tuieval-icon.svg +0 -0
  30. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/images/tui-setup.png +0 -0
  31. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/docs/writing-packs.md +0 -0
  32. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/pyproject.toml +0 -0
  33. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/__init__.py +0 -0
  34. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/__main__.py +0 -0
  35. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/cli.py +0 -0
  36. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/client.py +0 -0
  37. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/compare.py +0 -0
  38. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/engine.py +0 -0
  39. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/export.py +0 -0
  40. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/__init__.py +0 -0
  41. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/answer.py +0 -0
  42. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/code.py +0 -0
  43. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/rag.py +0 -0
  44. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/reply.py +0 -0
  45. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/graders/tool_call.py +0 -0
  46. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/packs.py +0 -0
  47. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/profiles.py +0 -0
  48. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/remove.py +0 -0
  49. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/scaffold.py +0 -0
  50. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/selftest.py +0 -0
  51. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  52. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/system.txt +0 -0
  53. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  54. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/pack.toml +0 -0
  55. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/system.txt +0 -0
  56. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  57. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  58. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/system.txt +0 -0
  59. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  60. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  61. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/system.txt +0 -0
  62. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  63. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  64. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  65. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  66. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  67. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/verdict.py +0 -0
  68. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/watch_proxy.py +0 -0
  69. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/workspace.py +0 -0
  70. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/src/tuieval/yamlout.py +0 -0
  71. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev8}/tests/mock_server.py +0 -0
@@ -2,7 +2,8 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
- - `tuieval tune` no longer rejects settings because the model's loading pushed other apps to swap (e.g. a large model locked in RAM with `--load-mode mlock`); only swapping while the loaded server works counts. Swap at load is noted in the output and the profile (`server_load_swapped_mb`).
5
+ - `tuieval tune` scores options on projected full-length answers (`[tune] answer_tokens`, default 1024), so generation speed counts as in real runs.
6
+ - `tuieval tune` reports an option's memory cost instead of rejecting it; it rejects only on critical memory pressure or over `[tune] swap_limit_mb`.
6
7
  - Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
7
8
  - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
8
9
  - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev5
3
+ Version: 0.2.0.dev8
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -77,8 +77,8 @@ The best server flags differ per model and per machine, so tuieval splits them b
77
77
  `tuieval tune <model>` (or `t` on the setup screen, for the ticked models) finds the fastest speed flags for a model on this machine:
78
78
 
79
79
  1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
80
- 2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
81
- 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates under which macOS swaps while the server works are rejected; swapping while the model loads (e.g. other apps making room for a model locked in RAM) is only noted.
80
+ 2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces. Options are scored on the **projected** time: each prompt's measured reading time plus a full-length answer (`[tune] answer_tokens`, default 1024) at its measured generation speed. The workload itself generates 128 tokens per prompt, but eval answers run to thousands, so generation speed counts as much as in real runs.
81
+ 3. An **output guard** rejects any option that changes greedy answers beyond noise. Memory is a cost, not a reason to reject: an option that makes macOS push other apps' idle memory to swap can still win, and the result warns about it. An option is rejected only if macOS memory pressure turns critical while it runs, or, with `[tune] swap_limit_mb` set, if it swaps more than that beyond the defaults.
82
82
 
83
83
  Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
84
84
 
@@ -128,7 +128,7 @@ On Apple Silicon Macs, once the system's GPU allocations pass about half of RAM,
128
128
  | `request` | fields added to every request (e.g. `{ cache_prompt = false }`) |
129
129
  | `perf` | speed-only flags always applied |
130
130
  | `tune` | speed-only knobs for `tuieval tune` |
131
- | `tune_objective` | `total` (default: workload time) or `decode` (tokens/s after a warm-up pass) |
131
+ | `tune_objective` | `projected` (default: workload time with full-length answers), `total` (workload time as run) or `decode` (tokens/s after a warm-up pass) |
132
132
  | `version_cmd` | prints the server version, recorded with every result |
133
133
  | `health` | readiness path for servers without `/v1/models` (must return JSON with `model`) |
134
134
  | `before_start` | a command run before starting the server (e.g. to free memory another process holds) |
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev5'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
21
+ __version__ = version = '0.2.0.dev8'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev8')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -225,6 +225,17 @@ def swapped_out_bytes():
225
225
  RESIDENCY_FRACTION = 0.5
226
226
 
227
227
 
228
+ def memory_pressure_level():
229
+ """macOS memory pressure: 1 normal, 2 warning, 4 critical (the system starts ending processes).
230
+ None elsewhere."""
231
+ try:
232
+ out = subprocess.run(["sysctl", "-n", "kern.memorystatus_vm_pressure_level"], capture_output=True,
233
+ text=True, timeout=5).stdout
234
+ return int(out.strip())
235
+ except (OSError, subprocess.TimeoutExpired, ValueError):
236
+ return None
237
+
238
+
228
239
  def gpu_residency_gb(machine, override=None):
229
240
  """GPU memory the driver keeps resident without churning (models.toml
230
241
  [machines.<id>] gpu_residency_gb overrides the half-of-RAM default)."""
@@ -523,13 +523,17 @@ def cmd_tune(argv):
523
523
  pr = result["profile"]
524
524
  ms = pr["measured"]
525
525
  gain = ", " + tune.gain_text(ms) if tune.gain_text(ms) else ""
526
- result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else f"{ms['total_s']}s for the workload"
526
+ result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else \
527
+ f"{ms['projected_s']}s projected for the workload" if ms.get("objective") == "projected" and ms.get("projected_s") \
528
+ else f"{ms['total_s']}s for the workload"
527
529
  say(f"{label}: {' '.join(pr['args']) or '(no knobs)'} -> {result}{gain}", GREEN)
528
530
  if pr["meta"].get("warm_start"):
529
531
  ws = pr["meta"]["warm_start"]
530
532
  say(f"{label}: started from {ws['from']}'s flags; re-tried {', '.join(ws['retested']) or 'nothing'}")
531
533
  if not pr["meta"].get("answer_guard", True):
532
534
  say(f"{label}: its answers depend on these settings; rerun its evals on this machine", "\033[33m")
535
+ for w in pr["meta"].get("warnings", []):
536
+ say(f"{label}: warning: {w}", YELLOW)
533
537
  if pr["meta"].get("rejected"):
534
538
  say(f"{label}: rejected because answers changed: {'; '.join(pr['meta']['rejected'])}")
535
539
  if a.export_pi and not export_pi(e, label):
@@ -101,6 +101,8 @@ headers = { "X-Title" = "tuieval" }
101
101
  # request_timeout_s = 600 # a tuning request that takes longer fails that candidate
102
102
  # warm_start = true # start from a tuned model with the same architecture and shapes (a fine-tune)
103
103
  # warm_retest = ["spec", "ubatch"] # the knobs a warm start re-tries; the rest are inherited
104
+ # answer_tokens = 1024 # answer length options are scored for (projected objective)
105
+ # swap_limit_mb = 500 # reject options that swap more than this beyond the defaults (default: report only)
104
106
 
105
107
  # Optional: llama-bench for the tuner's fast first stage (found on PATH as llama-bench otherwise).
106
108
  # [bench]
@@ -2565,6 +2565,8 @@ class EvalsApp(App):
2565
2565
  p = tune.tune(e, label, lambda kind, **d: on_event(kind, label=label, **d))
2566
2566
  ms = p["measured"]
2567
2567
  gain = f" ({tune.gain_text(ms)})" if tune.gain_text(ms) else ""
2568
+ if p["meta"].get("warnings"):
2569
+ gain += " (warning: it pushes other apps' memory to swap; see the log)"
2568
2570
  if not p["meta"].get("answer_guard", True):
2569
2571
  gain += " (answers depend on these settings: rerun its evals here)"
2570
2572
  done.append(f"{label}{gain}")
@@ -18,17 +18,24 @@ retrained or dropped its MTP layers, and its quant mix shifts the best micro-bat
18
18
  it tunes in full instead. `tuieval tune --cold` (or [tune] warm_start = false) always tunes in full.
19
19
 
20
20
  One objective per server (models.toml `tune_objective`):
21
- total (default) total seconds for the workload: short prompts, medium ones and one long
22
- (~8k-token) prompt, fixed generation length
23
- decode generated tokens per second, measured on a second pass over a few prompts after a
24
- warm-up pass (for servers whose decode speed grows as their caches warm)
21
+ projected (default) seconds the workload would take with full-length answers: each prompt's
22
+ measured reading time plus [tune] answer_tokens (default 1024) at its measured
23
+ generation speed. The workload generates 128 tokens per prompt, but eval answers run
24
+ to thousands, so generation speed counts as much as it does in real runs.
25
+ total wall-clock seconds for the workload as run (128-token answers)
26
+ decode generated tokens per second, measured on a second pass over a few prompts after a
27
+ warm-up pass (for servers whose decode speed grows as their caches warm)
25
28
  An output guard compares greedy answers with the default flags; an option that changes them beyond
26
29
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
30
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
31
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
30
- while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
31
- measured.
32
+ outdated.
33
+
34
+ Memory: a candidate is rejected if macOS memory pressure turns critical while it works, or, with
35
+ [tune] swap_limit_mb set, if it swaps more than that beyond the defaults. Otherwise memory is a
36
+ reported cost, not a reason to reject: a flag that pushes other apps' idle memory to swap can still
37
+ win, and the result warns about it. Swapping the defaults already cause is a warning too, and
38
+ swapping while the model loads is only noted. Servers without tune knobs are only measured.
32
39
 
33
40
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
34
41
  {p_minus_2} and {all} are this machine's core counts.
@@ -49,7 +56,10 @@ from . import profiles
49
56
 
50
57
  GEN_TOKENS = 128 # generated per workload prompt
51
58
  DECODE_TOKENS = 256 # generated per prompt for the decode objective
52
- SWAP_LIMIT = 256 * 2**20 # a candidate that swaps out more than this is rejected
59
+ SWAP_NOTABLE = 256 * 2**20 # swap worth a warning (bytes); rejecting for swap needs [tune] swap_limit_mb
60
+ ANSWER_TOKENS = 1024 # answer length the projected objective scores ([tune] answer_tokens)
61
+ PRESSURE_CRITICAL = 4 # macOS memory pressure level that rejects a candidate
62
+ PRESSURE_EVERY_S = 2
53
63
  GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
54
64
  # is rejected even before it stalls: servers grow as requests arrive
55
65
  REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
@@ -70,16 +80,22 @@ class Measure:
70
80
  load_s: float | None = None
71
81
  error: str = ""
72
82
  facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
83
+ swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
84
+ projected_s: float | None = None # the workload with full-length answers (projected objective)
85
+ pressure: int | None = None # highest macOS memory pressure level while it worked
73
86
 
74
87
  def score(self, objective):
75
88
  """Lower is better."""
76
89
  if objective == "decode":
77
90
  return 1 / self.tg_tps if self.tg_tps else float("inf")
91
+ if objective == "projected" and self.projected_s is not None:
92
+ return self.projected_s
78
93
  return self.total_s
79
94
 
80
95
  def summary(self):
81
96
  facts = "".join(f" · {k} {v}" for k, v in self.facts.items())
82
- return (f"{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
97
+ projected = f"{self.projected_s:.0f}s projected · " if self.projected_s is not None else ""
98
+ return (f"{projected}{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
83
99
  f" · ttft {self.ttft_s}s{facts}")
84
100
 
85
101
 
@@ -192,20 +208,21 @@ def _request(eng, base_url, body, name, emit, limit):
192
208
 
193
209
 
194
210
  def measure(eng, m, base_url, work, gen_tokens=GEN_TOKENS, passes=1, emit=lambda *a, **k: None,
195
- limit=REQUEST_LIMIT_S):
211
+ limit=REQUEST_LIMIT_S, answer_tokens=ANSWER_TOKENS):
196
212
  """Time the workload (after a warm-up request). With passes > 1 the earlier passes warm the
197
- server's caches and only the last pass is reported."""
213
+ server's caches and only the last pass is reported. answer_tokens: the answer length the
214
+ projected time is for."""
198
215
  out = Measure()
199
216
  for p in range(passes):
200
217
  if passes > 1:
201
218
  emit("tune_progress", message=f" pass {p + 1} of {passes}" + (" (warm-up)" if p < passes - 1 else " (timed)"))
202
- out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit)
219
+ out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens)
203
220
  if out.error:
204
221
  break
205
222
  return out
206
223
 
207
224
 
208
- def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
225
+ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens=ANSWER_TOKENS):
209
226
  sampling = engine_mod.effective_sampling(eng.cfg, m)
210
227
  request = eng.cfg["servers"][m["server"]].get("request", {})
211
228
  timeout = eng.cfg["defaults"]["request_timeout_ms"] / 1000
@@ -216,7 +233,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
216
233
  if warm["error"]:
217
234
  out.error = warm["error"]
218
235
  return out
219
- ttfts, p_tok, p_s, g_tok, g_s = [], 0, 0.0, 0, 0.0
236
+ ttfts, p_tok, p_s, g_tok, g_s, projected = [], 0, 0.0, 0, 0.0, 0.0
220
237
  for name, msgs in work:
221
238
  eng._check()
222
239
  emit("tune_progress", message=f" {name} ({gen_tokens} tokens)")
@@ -228,6 +245,13 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
228
245
  emit("tune_progress", message=f" done: {met['completion_tokens']} tokens in {res['total_s']:.1f}s"
229
246
  + (f", {met['gen_tps']} tok/s" if met["gen_tps"] else ""))
230
247
  out.total_s += res["total_s"]
248
+ # The same request with a full-length answer: its reading time plus answer_tokens at the
249
+ # generation speed measured here (as run if the server doesn't report that speed).
250
+ if met["completion_tokens"] and met["gen_tps"]:
251
+ reading = max(res["total_s"] - met["completion_tokens"] / met["gen_tps"], 0.0)
252
+ projected += reading + answer_tokens / met["gen_tps"]
253
+ else:
254
+ projected += res["total_s"]
231
255
  out.texts.append((res["reasoning"] or "") + (res["answer"] or ""))
232
256
  if name != "long" and res.get("ttft_s") is not None:
233
257
  ttfts.append(res["ttft_s"])
@@ -236,6 +260,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
236
260
  if met["completion_tokens"] and met["gen_tps"]:
237
261
  g_tok, g_s = g_tok + met["completion_tokens"], g_s + met["completion_tokens"] / met["gen_tps"]
238
262
  out.total_s = round(out.total_s, 2)
263
+ out.projected_s = round(projected, 1)
239
264
  out.ttft_s = round(sum(ttfts) / len(ttfts), 3) if ttfts else None
240
265
  out.pp_tps = round(p_tok / p_s, 1) if p_s else None
241
266
  out.tg_tps = round(g_tok / g_s, 2) if g_s else None
@@ -244,12 +269,37 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
244
269
 
245
270
  def gain_pct(objective, base, best):
246
271
  """How much better the chosen settings are than the defaults: % more decode tok/s for the
247
- decode objective, % less total time otherwise."""
272
+ decode objective, % less (projected or total) time otherwise."""
248
273
  if base is None:
249
274
  return None
250
275
  if objective == "decode":
251
276
  return round(100 * (best.tg_tps / base.tg_tps - 1)) if base.tg_tps and best.tg_tps else None
252
- return round(100 * (1 - best.total_s / base.total_s)) if base.total_s else None
277
+ b, n = base.score(objective), best.score(objective)
278
+ return round(100 * (1 - n / b)) if b else None
279
+
280
+
281
+ class PressureWatch:
282
+ """The highest macOS memory pressure level seen while a block runs (None where unknown)."""
283
+
284
+ def __enter__(self):
285
+ self.level, self._stop = machines.memory_pressure_level(), threading.Event()
286
+ self._thread = threading.Thread(target=self._run, daemon=True)
287
+ self._thread.start()
288
+ return self
289
+
290
+ def _sample(self):
291
+ lvl = machines.memory_pressure_level()
292
+ if lvl is not None:
293
+ self.level = max(self.level or 0, lvl)
294
+
295
+ def _run(self):
296
+ while not self._stop.wait(PRESSURE_EVERY_S):
297
+ self._sample()
298
+
299
+ def __exit__(self, *exc):
300
+ self._stop.set()
301
+ self._thread.join(PRESSURE_EVERY_S + 1)
302
+ self._sample()
253
303
 
254
304
 
255
305
  def gain_text(measured):
@@ -451,18 +501,21 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
451
501
  sv = eng.serving(m)
452
502
  if not sv.fits:
453
503
  raise engine_mod.ModelFailed(sv.fit_note)
454
- objective = server.get("tune_objective", "total")
504
+ objective = server.get("tune_objective", "projected")
455
505
  if objective == "decode":
456
506
  work, gen_tokens, passes = decode_workload(eng), DECODE_TOKENS, 2
457
507
  else:
458
508
  work, gen_tokens, passes = workload(eng, sv.ctx), GEN_TOKENS, 1
459
509
  settings = eng.cfg.get("tune", {})
510
+ answer_tokens = settings.get("answer_tokens", ANSWER_TOKENS)
511
+ swap_limit = settings.get("swap_limit_mb") # MB beyond the defaults' swap that rejects; None: report only
460
512
  threshold = settings.get("guard_similarity", 0.6)
461
513
  guard = not server.get("outputs_depend_on_machine")
462
514
  knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
463
515
  fixed = list(server.get("perf", [])) if server.get("cmd") else []
464
516
  cache, starts, notes, rejected, bad = {}, [0], [], [], set() # bad: (knob, option) that failed
465
- warned = []
517
+ warned, warnings = [], []
518
+ swap_ref = [None] # MB the defaults swapped while serving: the machine's baseline
466
519
 
467
520
  def evaluate(choices, why):
468
521
  k = tuple(sorted(choices.items()))
@@ -481,18 +534,20 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
481
534
  # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
482
535
  # that's recorded, not held against the settings. Swapping while serving is.
483
536
  loaded = machines.swapped_out_bytes()
484
- r = measure(eng, m, url, work, gen_tokens, passes, emit,
485
- settings.get("request_timeout_s", REQUEST_LIMIT_S))
537
+ with PressureWatch() as pressure:
538
+ r = measure(eng, m, url, work, gen_tokens, passes, emit,
539
+ settings.get("request_timeout_s", REQUEST_LIMIT_S), answer_tokens=answer_tokens)
486
540
  end = machines.swapped_out_bytes()
541
+ r.pressure = pressure.level
487
542
  if None not in (swap0, loaded, end):
488
543
  swapped = (end - loaded) / 2**20
489
544
  at_load = (loaded - swap0) / 2**20
490
- if at_load > SWAP_LIMIT / 2**20:
545
+ if at_load > SWAP_NOTABLE / 2**20:
491
546
  emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
492
547
  "(close apps for more headroom)")
493
548
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
494
549
  r.facts = dict(info["facts"])
495
- if swapped is not None and at_load > SWAP_LIMIT / 2**20:
550
+ if swapped is not None and at_load > SWAP_NOTABLE / 2**20:
496
551
  r.facts["load_swapped_mb"] = round(at_load)
497
552
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
498
553
  # harmless; sustained allocation over the limit is what makes the driver churn).
@@ -511,17 +566,35 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
511
566
  "close other apps for faster and fairer results")
512
567
  except engine_mod.ModelFailed as e:
513
568
  r = Measure(error=str(e).splitlines()[0])
514
- if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
515
- r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
569
+ r.swapped_mb = swapped
570
+ # Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
571
+ # a flag; only swapping beyond the defaults' does, and it's a cost to report unless it harms:
572
+ # critical memory pressure, or more than the user's swap_limit_mb.
573
+ extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
574
+ if extra is not None and extra > SWAP_NOTABLE / 2**20:
575
+ r.facts["extra_swap_mb"] = round(extra)
576
+ if not r.error and r.pressure is not None and r.pressure >= PRESSURE_CRITICAL:
577
+ r.error = "macOS memory pressure turned critical (memory too tight)"
578
+ elif not r.error and swap_limit is not None and extra is not None and extra > swap_limit:
579
+ r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (over swap_limit_mb = {swap_limit})"
516
580
  cache[k] = r
517
581
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
518
582
  return r
519
583
 
584
+ def set_baseline(r):
585
+ """The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
586
+ swap_ref[0] = r.swapped_mb or 0
587
+ if swap_ref[0] > SWAP_NOTABLE / 2**20:
588
+ warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
589
+ "this machine; close other apps or lower this model's context for more headroom")
590
+ emit("tune_step", message=" warning: " + warnings[-1])
591
+
520
592
  base = None
521
593
  if not server.get("cmd") or not knob_opts:
522
594
  r = evaluate({}, "measuring (no speed knobs to tune)")
523
595
  if r.error:
524
596
  raise engine_mod.ModelFailed(r.error)
597
+ set_baseline(r)
525
598
  best, best_r = {}, r
526
599
  method, bench_info = "measured", None
527
600
  else:
@@ -530,6 +603,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
530
603
  if base is None or base.error:
531
604
  raise engine_mod.ModelFailed(f"the server doesn't start with the default flags: "
532
605
  f"{base.error if base else 'no starts left'}")
606
+ set_baseline(base)
533
607
  settled, bench_info, warm_info = {}, None, None
534
608
  best, best_r, search = dict(default), base, list(knob_opts)
535
609
  sib = find_sibling(eng, m, knob_opts, sv.machine.id) \
@@ -598,6 +672,11 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
598
672
  if warm_info:
599
673
  notes.insert(0, f"warm start from {warm_info['from']}: inherited "
600
674
  f"{', '.join(warm_info['inherited']) or 'nothing'}")
675
+ if best_r.facts.get("extra_swap_mb") and best_r is not base:
676
+ warnings.append(f"the chosen flags use more memory than the defaults: macOS swapped "
677
+ f"~{best_r.facts['extra_swap_mb']} MB more of other apps' memory; close apps during runs, "
678
+ "or set [tune] swap_limit_mb to rule such flags out")
679
+ emit("tune_step", message=" warning: " + warnings[-1])
601
680
  size = sv.identity.get("model_bytes")
602
681
  profile = {
603
682
  "args": resolve(knob_opts, best),
@@ -606,15 +685,18 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
606
685
  "tg_tps": best_r.tg_tps, "load_s": best_r.load_s,
607
686
  "default_total_s": base.total_s if knob_opts else None,
608
687
  "default_tg_tps": base.tg_tps if knob_opts else None, "objective": objective,
688
+ "projected_s": best_r.projected_s,
689
+ "default_projected_s": base.projected_s if knob_opts else None,
609
690
  **{f"server_{k}": v for k, v in best_r.facts.items()},
610
691
  "gain_pct": gain_pct(objective, base, best_r)},
611
692
  "meta": {"method": method, "date": profiles.now(), "machine": sv.machine.summary,
612
693
  "server_version": eng.server_version(m), "model_file": os.path.basename(m["model"]),
613
694
  "model_bytes": size, "ctx": sv.ctx, "server_starts": starts[0],
614
- "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
695
+ "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else "")
696
+ + (f", scored for {answer_tokens}-token answers" if objective == "projected" else ""),
615
697
  "llama_bench": bool(bench_info), "answer_guard": guard,
616
698
  **({"warm_start": warm_info} if knob_opts and warm_info else {}),
617
- "rejected": rejected, "notes": notes[:10]},
699
+ "rejected": rejected, "notes": notes[:10], "warnings": warnings},
618
700
  }
619
701
  path = profiles.save(sv.machine.id, label, profile, eng.tuning_dir)
620
702
  eng._serving.clear()
@@ -308,12 +308,17 @@ class WarmTune(unittest.TestCase):
308
308
  yield "http://x", {"load_s": 1, "facts": {}}
309
309
  eng.serve = serve
310
310
 
311
- def measure(eng_, m, url, work, *a):
311
+ def measure(eng_, m, url, work, *a, **k):
312
312
  args = eng.starts[-1]
313
313
  fast = {"6": 1, "256": 1, "draft-mtp": 3}
314
314
  return tune.Measure(total_s=10 - sum(v for k, v in fast.items() if k in args), texts=["same"])
315
315
  self.orig = tune.measure
316
316
  tune.measure = measure
317
+ from tuieval import machines
318
+ self.pressure = lambda args: 1 # macOS memory pressure per start's flags: normal
319
+ orig_pressure = machines.memory_pressure_level
320
+ machines.memory_pressure_level = lambda: self.pressure(eng.starts[-1] if eng.starts else [])
321
+ self.addCleanup(setattr, machines, "memory_pressure_level", orig_pressure)
317
322
  self.eng, self.tune, self.profiles = eng, tune, profiles
318
323
  for label, args in (("sib", ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]),
319
324
  ("other", ["-t", "8"])):
@@ -325,14 +330,15 @@ class WarmTune(unittest.TestCase):
325
330
  self.tmp.cleanup()
326
331
 
327
332
  def _swap(self, at_load_mb, while_serving_mb):
328
- """Fake macOS swap counter: each server start swaps at_load_mb while loading (between the
329
- reading before the start and the one once it serves) and while_serving_mb during the work."""
333
+ """Fake macOS swap counter: each server start swaps at_load_mb while loading and
334
+ while_serving_mb(args) during the work (args: that start's flags)."""
330
335
  from tuieval import machines
331
336
  state = {"n": 0, "total": 0}
332
337
 
333
338
  def swapped_out_bytes():
334
339
  step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
335
- state["total"] += {0: 0, 1: at_load_mb, 2: while_serving_mb}[step] * 2**20
340
+ mb = {0: 0, 1: at_load_mb, 2: while_serving_mb(self.eng.starts[-1]) if step == 2 else 0}[step]
341
+ state["total"] += mb * 2**20
336
342
  state["n"] += 1
337
343
  return state["total"]
338
344
  orig = machines.swapped_out_bytes
@@ -340,15 +346,52 @@ class WarmTune(unittest.TestCase):
340
346
  self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
341
347
 
342
348
  def test_swapping_while_loading_is_only_noted(self):
343
- self._swap(at_load_mb=900, while_serving_mb=0)
349
+ self._swap(at_load_mb=900, while_serving_mb=lambda args: 0)
344
350
  p = self.tune.tune(self.eng, "new", use_bench=False)
345
351
  self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
352
+ self.assertEqual(p["meta"]["warnings"], [])
353
+
354
+ def test_swapping_every_candidate_shares_is_a_warning(self):
355
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 800)
356
+ p = self.tune.tune(self.eng, "new", use_bench=False)
357
+ self.assertEqual(p["args"], ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]) # tuned as usual
358
+ self.assertIn("800 MB", p["meta"]["warnings"][0])
359
+
360
+ def test_a_flag_that_costs_memory_still_wins_with_a_warning(self):
361
+ # -ub 256 is the fastest micro-batch here, and costs 600 MB of swap the defaults don't
362
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
363
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
364
+ self.assertIn("256", p["args"])
365
+ self.assertTrue(any("~500 MB more" in w for w in p["meta"]["warnings"]))
366
+
367
+ def test_swap_limit_rules_out_flags_that_cost_memory(self):
368
+ self.eng.cfg["tune"] = {"swap_limit_mb": 256}
369
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
370
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
371
+ self.assertNotIn("256", p["args"])
372
+ self.assertTrue(any("over swap_limit_mb = 256" in n for n in p["meta"]["notes"]))
346
373
 
347
- def test_swapping_while_serving_fails_the_settings(self):
348
- self._swap(at_load_mb=0, while_serving_mb=500)
349
- with self.assertRaises(self.tune.engine_mod.ModelFailed) as cm:
350
- self.tune.tune(self.eng, "new", use_bench=False)
351
- self.assertIn("swapped 500 MB while the server worked", str(cm.exception))
374
+ def test_critical_memory_pressure_rejects_a_flag(self):
375
+ self.pressure = lambda args: 4 if "draft-mtp" in args else 1
376
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
377
+ self.assertNotIn("draft-mtp", p["args"])
378
+ self.assertTrue(any("memory pressure turned critical" in n for n in p["meta"]["notes"]))
379
+
380
+ def test_projected_time_scores_full_length_answers(self):
381
+ # 10 s for 128 tokens at 16 tok/s: 2 s reading the prompt + 8 s generating
382
+ res = {"error": None, "total_s": 10.0, "ttft_s": 2.0, "reasoning": "", "answer": "x",
383
+ "usage": {"completion_tokens": 128, "prompt_tokens": 50},
384
+ "timings": {"predicted_per_second": 16.0, "prompt_per_second": 25.0}}
385
+ self.eng.cfg["sampling"] = {"temperature": 0}
386
+ self.eng._check = lambda: None
387
+ orig = self.tune._request
388
+ self.tune._request = lambda *a, **k: dict(res)
389
+ self.addCleanup(setattr, self.tune, "_request", orig)
390
+ out = self.tune._measure_once(self.eng, self.eng.model("new"), "http://x", [("p", [])], 128,
391
+ lambda *a, **k: None, 60, answer_tokens=1024)
392
+ self.assertEqual((out.total_s, out.projected_s), (10.0, 66.0)) # 2 s + 1024 / 16
393
+ self.assertEqual(out.score("projected"), 66.0)
394
+ self.assertEqual(out.score("total"), 10.0)
352
395
 
353
396
  def test_family_ignores_the_mtp_layer(self):
354
397
  # one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
@@ -388,7 +431,7 @@ class WarmTune(unittest.TestCase):
388
431
  self.profiles.save("mac", "sib", {"args": ["-t", "8", "-ub", "1024"], "meta": {"method": "tuned",
389
432
  "server_version": "v1"}}, self.eng.tuning_dir)
390
433
  orig = self.tune.measure
391
- self.tune.measure = lambda *a: self.tune.Measure(
434
+ self.tune.measure = lambda *a, **k: self.tune.Measure(
392
435
  total_s=99 if "1024" in self.eng.starts[-1] else orig(*a).total_s, texts=["same"])
393
436
  p = self.tune.tune(self.eng, "new", use_bench=False)
394
437
  self.assertNotIn("warm_start", p["meta"])
File without changes
File without changes
File without changes
File without changes