tuieval 0.2.0.dev5__tar.gz → 0.2.0.dev6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/CHANGELOG.md +1 -1
  2. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/PKG-INFO +1 -1
  3. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/models.md +1 -1
  4. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/_version.py +2 -2
  5. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/run_evals.py +2 -0
  6. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/tui.py +2 -0
  7. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/tune.py +24 -7
  8. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/tests/test_tuieval.py +18 -9
  9. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/.gitignore +0 -0
  10. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/LICENSE +0 -0
  11. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/README.md +0 -0
  12. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/RELEASING.md +0 -0
  13. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/README.md +0 -0
  14. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/favicon.ico +0 -0
  15. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/readme-header.png +0 -0
  16. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/social-preview.png +0 -0
  17. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-1024.png +0 -0
  18. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-128.png +0 -0
  19. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-16.png +0 -0
  20. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-16.svg +0 -0
  21. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-256.png +0 -0
  22. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-32.png +0 -0
  23. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-48.png +0 -0
  24. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-512.png +0 -0
  25. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-64.png +0 -0
  26. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  27. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon.svg +0 -0
  28. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/images/tui-setup.png +0 -0
  29. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/docs/writing-packs.md +0 -0
  30. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/pyproject.toml +0 -0
  31. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/__init__.py +0 -0
  32. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/__main__.py +0 -0
  33. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/cli.py +0 -0
  34. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/client.py +0 -0
  35. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/compare.py +0 -0
  36. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/engine.py +0 -0
  37. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/export.py +0 -0
  38. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/__init__.py +0 -0
  39. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/answer.py +0 -0
  40. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/code.py +0 -0
  41. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/rag.py +0 -0
  42. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/reply.py +0 -0
  43. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/graders/tool_call.py +0 -0
  44. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/machines.py +0 -0
  45. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/packs.py +0 -0
  46. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/profiles.py +0 -0
  47. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/remove.py +0 -0
  48. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/scaffold.py +0 -0
  49. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/selftest.py +0 -0
  50. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/models.toml +0 -0
  51. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  52. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/system.txt +0 -0
  53. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  54. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/pack.toml +0 -0
  55. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/system.txt +0 -0
  56. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  57. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  58. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/system.txt +0 -0
  59. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  60. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  61. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/system.txt +0 -0
  62. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  63. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  64. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  65. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  66. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  67. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/verdict.py +0 -0
  68. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/watch_proxy.py +0 -0
  69. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/workspace.py +0 -0
  70. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/src/tuieval/yamlout.py +0 -0
  71. {tuieval-0.2.0.dev5 → tuieval-0.2.0.dev6}/tests/mock_server.py +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
- - `tuieval tune` no longer rejects settings because the model's loading pushed other apps to swap (e.g. a large model locked in RAM with `--load-mode mlock`); only swapping while the loaded server works counts. Swap at load is noted in the output and the profile (`server_load_swapped_mb`).
5
+ - `tuieval tune`: swapping the defaults already cause is a warning, not a failure; an option is rejected only if it swaps more than the defaults.
6
6
  - Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
7
7
  - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
8
8
  - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev5
3
+ Version: 0.2.0.dev6
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -78,7 +78,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
78
78
 
79
79
  1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
80
80
  2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
81
- 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates under which macOS swaps while the server works are rejected; swapping while the model loads (e.g. other apps making room for a model locked in RAM) is only noted.
81
+ 3. An **output guard** rejects any option that changes greedy answers beyond noise. An option under which macOS swaps more than with the defaults is rejected (it costs memory). If the defaults already push other apps to swap, the tune warns and carries on.
82
82
 
83
83
  Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
84
84
 
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev5'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
21
+ __version__ = version = '0.2.0.dev6'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev6')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -530,6 +530,8 @@ def cmd_tune(argv):
530
530
  say(f"{label}: started from {ws['from']}'s flags; re-tried {', '.join(ws['retested']) or 'nothing'}")
531
531
  if not pr["meta"].get("answer_guard", True):
532
532
  say(f"{label}: its answers depend on these settings; rerun its evals on this machine", "\033[33m")
533
+ for w in pr["meta"].get("warnings", []):
534
+ say(f"{label}: warning: {w}", YELLOW)
533
535
  if pr["meta"].get("rejected"):
534
536
  say(f"{label}: rejected because answers changed: {'; '.join(pr['meta']['rejected'])}")
535
537
  if a.export_pi and not export_pi(e, label):
@@ -2565,6 +2565,8 @@ class EvalsApp(App):
2565
2565
  p = tune.tune(e, label, lambda kind, **d: on_event(kind, label=label, **d))
2566
2566
  ms = p["measured"]
2567
2567
  gain = f" ({tune.gain_text(ms)})" if tune.gain_text(ms) else ""
2568
+ if p["meta"].get("warnings"):
2569
+ gain += " (warning: it pushes other apps' memory to swap; see the log)"
2568
2570
  if not p["meta"].get("answer_guard", True):
2569
2571
  gain += " (answers depend on these settings: rerun its evals here)"
2570
2572
  done.append(f"{label}{gain}")
@@ -26,8 +26,9 @@ An output guard compares greedy answers with the default flags; an option that c
26
26
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
27
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
28
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
30
- while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
29
+ outdated. A candidate under which macOS swaps more than with the defaults is rejected (it costs
30
+ memory). Swapping the defaults already cause (the model, its context, other apps) is a warning, and
31
+ swapping while the model loads is only noted. Servers without tune knobs are only
31
32
  measured.
32
33
 
33
34
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
@@ -49,7 +50,7 @@ from . import profiles
49
50
 
50
51
  GEN_TOKENS = 128 # generated per workload prompt
51
52
  DECODE_TOKENS = 256 # generated per prompt for the decode objective
52
- SWAP_LIMIT = 256 * 2**20 # a candidate that swaps out more than this is rejected
53
+ SWAP_LIMIT = 256 * 2**20 # a candidate that swaps this much more than the defaults is rejected
53
54
  GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
54
55
  # is rejected even before it stalls: servers grow as requests arrive
55
56
  REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
@@ -70,6 +71,7 @@ class Measure:
70
71
  load_s: float | None = None
71
72
  error: str = ""
72
73
  facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
74
+ swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
73
75
 
74
76
  def score(self, objective):
75
77
  """Lower is better."""
@@ -462,7 +464,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
462
464
  knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
463
465
  fixed = list(server.get("perf", [])) if server.get("cmd") else []
464
466
  cache, starts, notes, rejected, bad = {}, [0], [], [], set() # bad: (knob, option) that failed
465
- warned = []
467
+ warned, warnings = [], []
468
+ swap_ref = [None] # MB the defaults swapped while serving: the machine's baseline
466
469
 
467
470
  def evaluate(choices, why):
468
471
  k = tuple(sorted(choices.items()))
@@ -511,17 +514,30 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
511
514
  "close other apps for faster and fairer results")
512
515
  except engine_mod.ModelFailed as e:
513
516
  r = Measure(error=str(e).splitlines()[0])
514
- if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
515
- r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
517
+ r.swapped_mb = swapped
518
+ # Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
519
+ # a flag; only swapping beyond the defaults' does.
520
+ extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
521
+ if not r.error and extra is not None and extra > SWAP_LIMIT / 2**20:
522
+ r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (uses too much memory)"
516
523
  cache[k] = r
517
524
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
518
525
  return r
519
526
 
527
+ def set_baseline(r):
528
+ """The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
529
+ swap_ref[0] = r.swapped_mb or 0
530
+ if swap_ref[0] > SWAP_LIMIT / 2**20:
531
+ warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
532
+ "this machine; close other apps or lower this model's context for more headroom")
533
+ emit("tune_step", message=" warning: " + warnings[-1])
534
+
520
535
  base = None
521
536
  if not server.get("cmd") or not knob_opts:
522
537
  r = evaluate({}, "measuring (no speed knobs to tune)")
523
538
  if r.error:
524
539
  raise engine_mod.ModelFailed(r.error)
540
+ set_baseline(r)
525
541
  best, best_r = {}, r
526
542
  method, bench_info = "measured", None
527
543
  else:
@@ -530,6 +546,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
530
546
  if base is None or base.error:
531
547
  raise engine_mod.ModelFailed(f"the server doesn't start with the default flags: "
532
548
  f"{base.error if base else 'no starts left'}")
549
+ set_baseline(base)
533
550
  settled, bench_info, warm_info = {}, None, None
534
551
  best, best_r, search = dict(default), base, list(knob_opts)
535
552
  sib = find_sibling(eng, m, knob_opts, sv.machine.id) \
@@ -614,7 +631,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
614
631
  "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
615
632
  "llama_bench": bool(bench_info), "answer_guard": guard,
616
633
  **({"warm_start": warm_info} if knob_opts and warm_info else {}),
617
- "rejected": rejected, "notes": notes[:10]},
634
+ "rejected": rejected, "notes": notes[:10], "warnings": warnings},
618
635
  }
619
636
  path = profiles.save(sv.machine.id, label, profile, eng.tuning_dir)
620
637
  eng._serving.clear()
@@ -325,14 +325,15 @@ class WarmTune(unittest.TestCase):
325
325
  self.tmp.cleanup()
326
326
 
327
327
  def _swap(self, at_load_mb, while_serving_mb):
328
- """Fake macOS swap counter: each server start swaps at_load_mb while loading (between the
329
- reading before the start and the one once it serves) and while_serving_mb during the work."""
328
+ """Fake macOS swap counter: each server start swaps at_load_mb while loading and
329
+ while_serving_mb(args) during the work (args: that start's flags)."""
330
330
  from tuieval import machines
331
331
  state = {"n": 0, "total": 0}
332
332
 
333
333
  def swapped_out_bytes():
334
334
  step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
335
- state["total"] += {0: 0, 1: at_load_mb, 2: while_serving_mb}[step] * 2**20
335
+ mb = {0: 0, 1: at_load_mb, 2: while_serving_mb(self.eng.starts[-1]) if step == 2 else 0}[step]
336
+ state["total"] += mb * 2**20
336
337
  state["n"] += 1
337
338
  return state["total"]
338
339
  orig = machines.swapped_out_bytes
@@ -340,15 +341,23 @@ class WarmTune(unittest.TestCase):
340
341
  self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
341
342
 
342
343
  def test_swapping_while_loading_is_only_noted(self):
343
- self._swap(at_load_mb=900, while_serving_mb=0)
344
+ self._swap(at_load_mb=900, while_serving_mb=lambda args: 0)
344
345
  p = self.tune.tune(self.eng, "new", use_bench=False)
345
346
  self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
347
+ self.assertEqual(p["meta"]["warnings"], [])
346
348
 
347
- def test_swapping_while_serving_fails_the_settings(self):
348
- self._swap(at_load_mb=0, while_serving_mb=500)
349
- with self.assertRaises(self.tune.engine_mod.ModelFailed) as cm:
350
- self.tune.tune(self.eng, "new", use_bench=False)
351
- self.assertIn("swapped 500 MB while the server worked", str(cm.exception))
349
+ def test_swapping_every_candidate_shares_is_a_warning(self):
350
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 800)
351
+ p = self.tune.tune(self.eng, "new", use_bench=False)
352
+ self.assertEqual(p["args"], ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]) # tuned as usual
353
+ self.assertIn("800 MB", p["meta"]["warnings"][0])
354
+
355
+ def test_a_flag_that_swaps_more_than_the_defaults_is_rejected(self):
356
+ # -ub 256 is the fastest micro-batch here, but it costs 600 MB of swap the defaults don't
357
+ self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
358
+ p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
359
+ self.assertNotIn("256", p["args"])
360
+ self.assertTrue(any("500 MB more than with the defaults" in n for n in p["meta"]["notes"]))
352
361
 
353
362
  def test_family_ignores_the_mtp_layer(self):
354
363
  # one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
File without changes
File without changes
File without changes
File without changes