tuieval 0.2.0.dev4__tar.gz → 0.2.0.dev5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/CHANGELOG.md +1 -0
  2. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/PKG-INFO +1 -1
  3. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/models.md +1 -1
  4. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/_version.py +2 -2
  5. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/tune.py +17 -4
  6. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/tests/test_tuieval.py +26 -0
  7. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/.gitignore +0 -0
  8. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/LICENSE +0 -0
  9. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/README.md +0 -0
  10. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/RELEASING.md +0 -0
  11. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/README.md +0 -0
  12. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/favicon.ico +0 -0
  13. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/readme-header.png +0 -0
  14. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/social-preview.png +0 -0
  15. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-1024.png +0 -0
  16. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-128.png +0 -0
  17. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.png +0 -0
  18. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-16.svg +0 -0
  19. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-256.png +0 -0
  20. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-32.png +0 -0
  21. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-48.png +0 -0
  22. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-512.png +0 -0
  23. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-64.png +0 -0
  24. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  25. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/brand/tuieval-icon.svg +0 -0
  26. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/images/tui-setup.png +0 -0
  27. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/docs/writing-packs.md +0 -0
  28. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/pyproject.toml +0 -0
  29. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/__init__.py +0 -0
  30. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/__main__.py +0 -0
  31. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/cli.py +0 -0
  32. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/client.py +0 -0
  33. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/compare.py +0 -0
  34. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/engine.py +0 -0
  35. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/export.py +0 -0
  36. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/__init__.py +0 -0
  37. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/answer.py +0 -0
  38. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/code.py +0 -0
  39. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/rag.py +0 -0
  40. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/reply.py +0 -0
  41. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/graders/tool_call.py +0 -0
  42. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/machines.py +0 -0
  43. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/packs.py +0 -0
  44. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/profiles.py +0 -0
  45. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/remove.py +0 -0
  46. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/run_evals.py +0 -0
  47. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/scaffold.py +0 -0
  48. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/selftest.py +0 -0
  49. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/models.toml +0 -0
  50. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  51. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/system.txt +0 -0
  52. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  53. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/pack.toml +0 -0
  54. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/system.txt +0 -0
  55. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  56. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  57. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/system.txt +0 -0
  58. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  59. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  60. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/system.txt +0 -0
  61. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  62. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  63. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  64. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  65. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  66. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/tui.py +0 -0
  67. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/verdict.py +0 -0
  68. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/watch_proxy.py +0 -0
  69. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/workspace.py +0 -0
  70. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/src/tuieval/yamlout.py +0 -0
  71. {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev5}/tests/mock_server.py +0 -0
@@ -2,6 +2,7 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ - `tuieval tune` no longer rejects settings because the model's loading pushed other apps to swap (e.g. a large model locked in RAM with `--load-mode mlock`); only swapping while the loaded server works counts. Swap at load is noted in the output and the profile (`server_load_swapped_mb`).
5
6
  - Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
6
7
  - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
7
8
  - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev4
3
+ Version: 0.2.0.dev5
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -78,7 +78,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
78
78
 
79
79
  1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
80
80
  2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
81
- 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates that make macOS swap are rejected.
81
+ 3. An **output guard** rejects any option that changes greedy answers beyond noise. Candidates under which macOS swaps while the server works are rejected; swapping while the model loads (e.g. other apps making room for a model locked in RAM) is only noted.
82
82
 
83
83
  Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
84
84
 
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev4'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev4')
21
+ __version__ = version = '0.2.0.dev5'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -26,7 +26,8 @@ An output guard compares greedy answers with the default flags; an option that c
26
26
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
27
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
28
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate that makes macOS swap is rejected. Servers without tune knobs are only
29
+ outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
30
+ while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
30
31
  measured.
31
32
 
32
33
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
@@ -474,12 +475,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
474
475
  emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
475
476
  start=starts[0], max_starts=max_starts)
476
477
  swap0 = machines.swapped_out_bytes()
478
+ swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
477
479
  try:
478
480
  with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
481
+ # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
482
+ # that's recorded, not held against the settings. Swapping while serving is.
483
+ loaded = machines.swapped_out_bytes()
479
484
  r = measure(eng, m, url, work, gen_tokens, passes, emit,
480
485
  settings.get("request_timeout_s", REQUEST_LIMIT_S))
486
+ end = machines.swapped_out_bytes()
487
+ if None not in (swap0, loaded, end):
488
+ swapped = (end - loaded) / 2**20
489
+ at_load = (loaded - swap0) / 2**20
490
+ if at_load > SWAP_LIMIT / 2**20:
491
+ emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
492
+ "(close apps for more headroom)")
481
493
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
482
494
  r.facts = dict(info["facts"])
495
+ if swapped is not None and at_load > SWAP_LIMIT / 2**20:
496
+ r.facts["load_swapped_mb"] = round(at_load)
483
497
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
484
498
  # harmless; sustained allocation over the limit is what makes the driver churn).
485
499
  settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
@@ -497,9 +511,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
497
511
  "close other apps for faster and fairer results")
498
512
  except engine_mod.ModelFailed as e:
499
513
  r = Measure(error=str(e).splitlines()[0])
500
- swap1 = machines.swapped_out_bytes()
501
- if not r.error and swap0 is not None and swap1 is not None and swap1 - swap0 > SWAP_LIMIT:
502
- r.error = f"macOS swapped {(swap1 - swap0) / 2**20:.0f} MB (memory too tight)"
514
+ if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
515
+ r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
503
516
  cache[k] = r
504
517
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
505
518
  return r
@@ -324,6 +324,32 @@ class WarmTune(unittest.TestCase):
324
324
  self.tune.measure = self.orig
325
325
  self.tmp.cleanup()
326
326
 
327
+ def _swap(self, at_load_mb, while_serving_mb):
328
+ """Fake macOS swap counter: each server start swaps at_load_mb while loading (between the
329
+ reading before the start and the one once it serves) and while_serving_mb during the work."""
330
+ from tuieval import machines
331
+ state = {"n": 0, "total": 0}
332
+
333
+ def swapped_out_bytes():
334
+ step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
335
+ state["total"] += {0: 0, 1: at_load_mb, 2: while_serving_mb}[step] * 2**20
336
+ state["n"] += 1
337
+ return state["total"]
338
+ orig = machines.swapped_out_bytes
339
+ machines.swapped_out_bytes = swapped_out_bytes
340
+ self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
341
+
342
+ def test_swapping_while_loading_is_only_noted(self):
343
+ self._swap(at_load_mb=900, while_serving_mb=0)
344
+ p = self.tune.tune(self.eng, "new", use_bench=False)
345
+ self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
346
+
347
+ def test_swapping_while_serving_fails_the_settings(self):
348
+ self._swap(at_load_mb=0, while_serving_mb=500)
349
+ with self.assertRaises(self.tune.engine_mod.ModelFailed) as cm:
350
+ self.tune.tune(self.eng, "new", use_bench=False)
351
+ self.assertIn("swapped 500 MB while the server worked", str(cm.exception))
352
+
327
353
  def test_family_ignores_the_mtp_layer(self):
328
354
  # one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
329
355
  d = self.tmp.name
File without changes
File without changes
File without changes
File without changes