tuieval 0.2.0.dev3__tar.gz → 0.2.0.dev5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/CHANGELOG.md +2 -0
  2. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/PKG-INFO +1 -1
  3. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/models.md +1 -1
  4. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/_version.py +2 -2
  5. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/tune.py +21 -5
  6. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/tests/test_tuieval.py +47 -4
  7. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/.gitignore +0 -0
  8. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/LICENSE +0 -0
  9. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/README.md +0 -0
  10. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/RELEASING.md +0 -0
  11. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/README.md +0 -0
  12. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/favicon.ico +0 -0
  13. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/readme-header.png +0 -0
  14. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/social-preview.png +0 -0
  15. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-1024.png +0 -0
  16. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-128.png +0 -0
  17. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.png +0 -0
  18. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.svg +0 -0
  19. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-256.png +0 -0
  20. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-32.png +0 -0
  21. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-48.png +0 -0
  22. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-512.png +0 -0
  23. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-64.png +0 -0
  24. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  25. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon.svg +0 -0
  26. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/images/tui-setup.png +0 -0
  27. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/docs/writing-packs.md +0 -0
  28. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/pyproject.toml +0 -0
  29. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/__init__.py +0 -0
  30. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/__main__.py +0 -0
  31. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/cli.py +0 -0
  32. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/client.py +0 -0
  33. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/compare.py +0 -0
  34. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/engine.py +0 -0
  35. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/export.py +0 -0
  36. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/__init__.py +0 -0
  37. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/answer.py +0 -0
  38. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/code.py +0 -0
  39. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/rag.py +0 -0
  40. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/reply.py +0 -0
  41. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/graders/tool_call.py +0 -0
  42. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/machines.py +0 -0
  43. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/packs.py +0 -0
  44. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/profiles.py +0 -0
  45. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/remove.py +0 -0
  46. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/run_evals.py +0 -0
  47. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/scaffold.py +0 -0
  48. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/selftest.py +0 -0
  49. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/models.toml +0 -0
  50. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  51. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/system.txt +0 -0
  52. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  53. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/pack.toml +0 -0
  54. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/system.txt +0 -0
  55. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  56. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  57. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/system.txt +0 -0
  58. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  59. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  60. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/system.txt +0 -0
  61. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  62. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  63. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  64. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  65. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  66. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/tui.py +0 -0
  67. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/verdict.py +0 -0
  68. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/watch_proxy.py +0 -0
  69. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/workspace.py +0 -0
  70. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/src/tuieval/yamlout.py +0 -0
  71. {tuieval-0.2.0.dev3 → tuieval-0.2.0.dev5}/tests/mock_server.py +0 -0
@@ -2,6 +2,8 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ - `tuieval tune` no longer rejects settings because the model's loading pushed other apps to swap (e.g. a large model locked in RAM with `--load-mode mlock`); only swapping while the loaded server works counts. Swap at load is noted in the output and the profile (`server_load_swapped_mb`).
6
+ - Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
5
7
  - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
6
8
  - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
7
9
  - A new README screenshot from a demo workspace.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev3
3
+ Version: 0.2.0.dev5
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -78,7 +78,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
78
78
 
79
79
  1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
80
80
  2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
81
- 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates that make macOS swap are rejected.
81
+ 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates under which macOS swaps while the server works are rejected; swapping while the model loads (e.g. other apps making room for a model locked in RAM) is only noted.
82
82
 
83
83
  Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
84
84
 
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev3'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev3')
21
+ __version__ = version = '0.2.0.dev5'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -26,7 +26,8 @@ An output guard compares greedy answers with the default flags; an option that c
26
26
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
27
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
28
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate that makes macOS swap is rejected. Servers without tune knobs are only
29
+ outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
30
+ while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
30
31
  measured.
31
32
 
32
33
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
@@ -315,8 +316,11 @@ def family(path):
315
316
  i = machines.read_gguf(engine_mod.expand(path))
316
317
  except (OSError, ValueError):
317
318
  return None
319
+ # The MTP (nextn) draft layers at the end are left out: GGUFs of the same model describe them
320
+ # differently, and whether draft-mtp is offered is decided per model (knobs) anyway.
321
+ heads = i["kv_heads_per_layer"][:len(i["kv_heads_per_layer"]) - i["mtp_layers"]]
318
322
  return (i["architecture"], i["layers"], i["embedding"], i["experts"], i["experts_used"],
319
- i["head_dim_k"], i["head_dim_v"], tuple(i["kv_heads_per_layer"]))
323
+ i["head_dim_k"], i["head_dim_v"], tuple(heads))
320
324
 
321
325
 
322
326
  @dataclasses.dataclass
@@ -471,12 +475,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
471
475
  emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
472
476
  start=starts[0], max_starts=max_starts)
473
477
  swap0 = machines.swapped_out_bytes()
478
+ swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
474
479
  try:
475
480
  with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
481
+ # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
482
+ # that's recorded, not held against the settings. Swapping while serving is.
483
+ loaded = machines.swapped_out_bytes()
476
484
  r = measure(eng, m, url, work, gen_tokens, passes, emit,
477
485
  settings.get("request_timeout_s", REQUEST_LIMIT_S))
486
+ end = machines.swapped_out_bytes()
487
+ if None not in (swap0, loaded, end):
488
+ swapped = (end - loaded) / 2**20
489
+ at_load = (loaded - swap0) / 2**20
490
+ if at_load > SWAP_LIMIT / 2**20:
491
+ emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
492
+ "(close apps for more headroom)")
478
493
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
479
494
  r.facts = dict(info["facts"])
495
+ if swapped is not None and at_load > SWAP_LIMIT / 2**20:
496
+ r.facts["load_swapped_mb"] = round(at_load)
480
497
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
481
498
  # harmless; sustained allocation over the limit is what makes the driver churn).
482
499
  settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
@@ -494,9 +511,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
494
511
  "close other apps for faster and fairer results")
495
512
  except engine_mod.ModelFailed as e:
496
513
  r = Measure(error=str(e).splitlines()[0])
497
- swap1 = machines.swapped_out_bytes()
498
- if not r.error and swap0 is not None and swap1 is not None and swap1 - swap0 > SWAP_LIMIT:
499
- r.error = f"macOS swapped {(swap1 - swap0) / 2**20:.0f} MB (memory too tight)"
514
+ if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
515
+ r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
500
516
  cache[k] = r
501
517
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
502
518
  return r
@@ -225,17 +225,25 @@ class Units(unittest.TestCase):
225
225
  self.assertEqual(p.modules, ["some_missing_module"])
226
226
 
227
227
 
228
- def fake_gguf(path, embedding=5120, size=0):
229
- """A GGUF header with just the keys machines.read_gguf reads, padded to size bytes."""
228
+ def fake_gguf(path, embedding=5120, size=0, kv_heads=4):
229
+ """A GGUF header with just the keys machines.read_gguf reads, padded to size bytes. kv_heads: a
230
+ number, or one per layer (a list, as some GGUFs store it)."""
230
231
  import struct
231
232
  s = lambda t: struct.pack("<Q", len(t)) + t.encode() # noqa: E731
232
233
  kv = [("general.architecture", 8, "qwen35"), ("qwen35.block_count", 4, 64),
233
234
  ("qwen35.embedding_length", 4, embedding), ("qwen35.attention.head_count", 4, 24),
234
- ("qwen35.attention.head_count_kv", 4, 4), ("qwen35.attention.key_length", 4, 256),
235
+ ("qwen35.attention.head_count_kv", 9 if isinstance(kv_heads, list) else 4, kv_heads),
236
+ ("qwen35.attention.key_length", 4, 256),
235
237
  ("qwen35.full_attention_interval", 4, 4), ("qwen35.nextn_predict_layers", 4, 1)]
236
238
  out = b"GGUF" + struct.pack("<IQQ", 3, 0, len(kv))
237
239
  for key, t, v in kv:
238
- out += s(key) + struct.pack("<I", t) + (s(v) if t == 8 else struct.pack("<I", v))
240
+ out += s(key) + struct.pack("<I", t)
241
+ if t == 8:
242
+ out += s(v)
243
+ elif t == 9: # an array of uint32
244
+ out += struct.pack("<IQ", 4, len(v)) + b"".join(struct.pack("<I", x) for x in v)
245
+ else:
246
+ out += struct.pack("<I", v)
239
247
  with open(path, "wb") as f:
240
248
  f.write(out + b"\0" * max(0, size - len(out)))
241
249
 
@@ -316,6 +324,41 @@ class WarmTune(unittest.TestCase):
316
324
  self.tune.measure = self.orig
317
325
  self.tmp.cleanup()
318
326
 
327
+ def _swap(self, at_load_mb, while_serving_mb):
328
+ """Fake macOS swap counter: each server start swaps at_load_mb while loading (between the
329
+ reading before the start and the one once it serves) and while_serving_mb during the work."""
330
+ from tuieval import machines
331
+ state = {"n": 0, "total": 0}
332
+
333
+ def swapped_out_bytes():
334
+ step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
335
+ state["total"] += {0: 0, 1: at_load_mb, 2: while_serving_mb}[step] * 2**20
336
+ state["n"] += 1
337
+ return state["total"]
338
+ orig = machines.swapped_out_bytes
339
+ machines.swapped_out_bytes = swapped_out_bytes
340
+ self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
341
+
342
+ def test_swapping_while_loading_is_only_noted(self):
343
+ self._swap(at_load_mb=900, while_serving_mb=0)
344
+ p = self.tune.tune(self.eng, "new", use_bench=False)
345
+ self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
346
+
347
+ def test_swapping_while_serving_fails_the_settings(self):
348
+ self._swap(at_load_mb=0, while_serving_mb=500)
349
+ with self.assertRaises(self.tune.engine_mod.ModelFailed) as cm:
350
+ self.tune.tune(self.eng, "new", use_bench=False)
351
+ self.assertIn("swapped 500 MB while the server worked", str(cm.exception))
352
+
353
+ def test_family_ignores_the_mtp_layer(self):
354
+ # one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
355
+ d = self.tmp.name
356
+ per_layer = [4 if (i + 1) % 4 == 0 else 0 for i in range(63)] + [0]
357
+ fake_gguf(f"{d}/listed.gguf", kv_heads=per_layer)
358
+ self.assertEqual(self.tune.family(f"{d}/listed.gguf"), self.tune.family(f"{d}/new.gguf"))
359
+ fake_gguf(f"{d}/other-shape.gguf", kv_heads=[8 if (i + 1) % 4 == 0 else 0 for i in range(64)])
360
+ self.assertNotEqual(self.tune.family(f"{d}/other-shape.gguf"), self.tune.family(f"{d}/new.gguf"))
361
+
319
362
  def test_option_index(self):
320
363
  opts = self.KNOBS["spec"]
321
364
  self.assertEqual(self.tune.option_index(opts, ["-t", "6", "--spec-type", "draft-mtp"]), 2)
File without changes
File without changes
File without changes
File without changes