tuieval 0.2.0.dev4__py3-none-any.whl → 0.2.0.dev6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tuieval/_version.py CHANGED
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev4'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev4')
21
+ __version__ = version = '0.2.0.dev6'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev6')
23
23
 
24
24
  __commit_id__ = commit_id = None
tuieval/run_evals.py CHANGED
@@ -530,6 +530,8 @@ def cmd_tune(argv):
530
530
  say(f"{label}: started from {ws['from']}'s flags; re-tried {', '.join(ws['retested']) or 'nothing'}")
531
531
  if not pr["meta"].get("answer_guard", True):
532
532
  say(f"{label}: its answers depend on these settings; rerun its evals on this machine", "\033[33m")
533
+ for w in pr["meta"].get("warnings", []):
534
+ say(f"{label}: warning: {w}", YELLOW)
533
535
  if pr["meta"].get("rejected"):
534
536
  say(f"{label}: rejected because answers changed: {'; '.join(pr['meta']['rejected'])}")
535
537
  if a.export_pi and not export_pi(e, label):
tuieval/tui.py CHANGED
@@ -2565,6 +2565,8 @@ class EvalsApp(App):
2565
2565
  p = tune.tune(e, label, lambda kind, **d: on_event(kind, label=label, **d))
2566
2566
  ms = p["measured"]
2567
2567
  gain = f" ({tune.gain_text(ms)})" if tune.gain_text(ms) else ""
2568
+ if p["meta"].get("warnings"):
2569
+ gain += " (warning: it pushes other apps' memory to swap; see the log)"
2568
2570
  if not p["meta"].get("answer_guard", True):
2569
2571
  gain += " (answers depend on these settings: rerun its evals here)"
2570
2572
  done.append(f"{label}{gain}")
tuieval/tune.py CHANGED
@@ -26,7 +26,9 @@ An output guard compares greedy answers with the default flags; an option that c
26
26
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
27
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
28
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate that makes macOS swap is rejected. Servers without tune knobs are only
29
+ outdated. A candidate under which macOS swaps more than with the defaults is rejected (it costs
30
+ memory). Swapping the defaults already cause (the model, its context, other apps) is a warning, and
31
+ swapping while the model loads is only noted. Servers without tune knobs are only
30
32
  measured.
31
33
 
32
34
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
@@ -48,7 +50,7 @@ from . import profiles
48
50
 
49
51
  GEN_TOKENS = 128 # generated per workload prompt
50
52
  DECODE_TOKENS = 256 # generated per prompt for the decode objective
51
- SWAP_LIMIT = 256 * 2**20 # a candidate that swaps out more than this is rejected
53
+ SWAP_LIMIT = 256 * 2**20 # a candidate that swaps this much more than the defaults is rejected
52
54
  GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
53
55
  # is rejected even before it stalls: servers grow as requests arrive
54
56
  REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
@@ -69,6 +71,7 @@ class Measure:
69
71
  load_s: float | None = None
70
72
  error: str = ""
71
73
  facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
74
+ swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
72
75
 
73
76
  def score(self, objective):
74
77
  """Lower is better."""
@@ -461,7 +464,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
461
464
  knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
462
465
  fixed = list(server.get("perf", [])) if server.get("cmd") else []
463
466
  cache, starts, notes, rejected, bad = {}, [0], [], [], set() # bad: (knob, option) that failed
464
- warned = []
467
+ warned, warnings = [], []
468
+ swap_ref = [None] # MB the defaults swapped while serving: the machine's baseline
465
469
 
466
470
  def evaluate(choices, why):
467
471
  k = tuple(sorted(choices.items()))
@@ -474,12 +478,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
474
478
  emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
475
479
  start=starts[0], max_starts=max_starts)
476
480
  swap0 = machines.swapped_out_bytes()
481
+ swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
477
482
  try:
478
483
  with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
484
+ # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
485
+ # that's recorded, not held against the settings. Swapping while serving is.
486
+ loaded = machines.swapped_out_bytes()
479
487
  r = measure(eng, m, url, work, gen_tokens, passes, emit,
480
488
  settings.get("request_timeout_s", REQUEST_LIMIT_S))
489
+ end = machines.swapped_out_bytes()
490
+ if None not in (swap0, loaded, end):
491
+ swapped = (end - loaded) / 2**20
492
+ at_load = (loaded - swap0) / 2**20
493
+ if at_load > SWAP_LIMIT / 2**20:
494
+ emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
495
+ "(close apps for more headroom)")
481
496
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
482
497
  r.facts = dict(info["facts"])
498
+ if swapped is not None and at_load > SWAP_LIMIT / 2**20:
499
+ r.facts["load_swapped_mb"] = round(at_load)
483
500
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
484
501
  # harmless; sustained allocation over the limit is what makes the driver churn).
485
502
  settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
@@ -497,18 +514,30 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
497
514
  "close other apps for faster and fairer results")
498
515
  except engine_mod.ModelFailed as e:
499
516
  r = Measure(error=str(e).splitlines()[0])
500
- swap1 = machines.swapped_out_bytes()
501
- if not r.error and swap0 is not None and swap1 is not None and swap1 - swap0 > SWAP_LIMIT:
502
- r.error = f"macOS swapped {(swap1 - swap0) / 2**20:.0f} MB (memory too tight)"
517
+ r.swapped_mb = swapped
518
+ # Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
519
+ # a flag; only swapping beyond the defaults' does.
520
+ extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
521
+ if not r.error and extra is not None and extra > SWAP_LIMIT / 2**20:
522
+ r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (uses too much memory)"
503
523
  cache[k] = r
504
524
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
505
525
  return r
506
526
 
527
+ def set_baseline(r):
528
+ """The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
529
+ swap_ref[0] = r.swapped_mb or 0
530
+ if swap_ref[0] > SWAP_LIMIT / 2**20:
531
+ warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
532
+ "this machine; close other apps or lower this model's context for more headroom")
533
+ emit("tune_step", message=" warning: " + warnings[-1])
534
+
507
535
  base = None
508
536
  if not server.get("cmd") or not knob_opts:
509
537
  r = evaluate({}, "measuring (no speed knobs to tune)")
510
538
  if r.error:
511
539
  raise engine_mod.ModelFailed(r.error)
540
+ set_baseline(r)
512
541
  best, best_r = {}, r
513
542
  method, bench_info = "measured", None
514
543
  else:
@@ -517,6 +546,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
517
546
  if base is None or base.error:
518
547
  raise engine_mod.ModelFailed(f"the server doesn't start with the default flags: "
519
548
  f"{base.error if base else 'no starts left'}")
549
+ set_baseline(base)
520
550
  settled, bench_info, warm_info = {}, None, None
521
551
  best, best_r, search = dict(default), base, list(knob_opts)
522
552
  sib = find_sibling(eng, m, knob_opts, sv.machine.id) \
@@ -601,7 +631,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
601
631
  "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
602
632
  "llama_bench": bool(bench_info), "answer_guard": guard,
603
633
  **({"warm_start": warm_info} if knob_opts and warm_info else {}),
604
- "rejected": rejected, "notes": notes[:10]},
634
+ "rejected": rejected, "notes": notes[:10], "warnings": warnings},
605
635
  }
606
636
  path = profiles.save(sv.machine.id, label, profile, eng.tuning_dir)
607
637
  eng._serving.clear()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev4
3
+ Version: 0.2.0.dev6
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -1,6 +1,6 @@
1
1
  tuieval/__init__.py,sha256=myVXVX7htbwezW6k3x_FuKgn1RSZ2WcTaDEMByGn6Wg,312
2
2
  tuieval/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
3
- tuieval/_version.py,sha256=7rQfE9HWjmke7Ju8Nk_aqQp4chEDuGAHhvH3RPUbSyc,533
3
+ tuieval/_version.py,sha256=9gdVeCkadGCQ8txcnOoyULu4u3PP7cNd2WZB7FPQHh8,533
4
4
  tuieval/cli.py,sha256=ZaLoWrxyCABu4SJzx5U2gVUUmwcwRgpZVqdqnxyFMOk,4167
5
5
  tuieval/client.py,sha256=Vui7ucVv_GiKObGoVS_EWSZFUKPou9PWmutqMTrahXk,9313
6
6
  tuieval/compare.py,sha256=b-rVcMwq0Tos3oNKcpUP1HXULKhAQNAS4rlXYSY7t6k,25460
@@ -10,11 +10,11 @@ tuieval/machines.py,sha256=CeestY-_6h8ZKhJbBA_xyf0prqBvjvdgs2XJtCJYnNg,12974
10
10
  tuieval/packs.py,sha256=Go-kTSPvYZriXMo4DvmqNNL-iD0J92MKLiNCnP2chJ8,10305
11
11
  tuieval/profiles.py,sha256=soeLxjdB-l-OxSY8jZ4yjwfdSnsBV-QICcwKxRJS7nQ,3663
12
12
  tuieval/remove.py,sha256=Vwh43nm9lVb9U9aN-h6CMYbkTwdR3UWfZObwweF72n0,5157
13
- tuieval/run_evals.py,sha256=uPzidbRPUduBwP8sKyzg-U8cka0p-RsbZlVis6qQTlk,37526
13
+ tuieval/run_evals.py,sha256=0dtPqhXxFqRLj_11I7hs2IlNkrxxilm6GOlEPfef3Fw,37625
14
14
  tuieval/scaffold.py,sha256=-AE5vKXeTtMZJ8OMk4UiIstoUWozTKjThKxZ87j0f9E,4720
15
15
  tuieval/selftest.py,sha256=xKfR73wMTYcDRSCBYQTNyKeR9bYU5HIYUSg7oZvNCyI,9330
16
- tuieval/tui.py,sha256=8diKtYE31ZUKq62rfMGdMkEbHgK6arRYpjFu-wcMVZs,131997
17
- tuieval/tune.py,sha256=Kf5angorAFNPLqeIuNdcFphMCL53BffYqjPi_XPW77Q,32879
16
+ tuieval/tui.py,sha256=52hK8VSOVA0iHW8AyVFEyEXB32rE3U_hezJEQ5wc0TI,132159
17
+ tuieval/tune.py,sha256=3vUll2EQkPIqRuZpAWV2-r-sk7FMPUkGeWmo63eTmHA,35031
18
18
  tuieval/verdict.py,sha256=oLKpRIruk7W8eF1YdmITFx9BgBK3kPKYViKL20Nzyu0,16306
19
19
  tuieval/watch_proxy.py,sha256=z916NRhEur04a9RnLm6ZpIEY1SnnUGPiHYbmlD9ktQ0,11035
20
20
  tuieval/workspace.py,sha256=V_HhkaN1odr0OpjNsPYradRWLpzDldM7SwpvxohBL4U,895
@@ -42,8 +42,8 @@ tuieval/templates/packs/tool_call/pack.toml,sha256=n4bx0lGJHSmW41uVqhfIXJ2BWZpmT
42
42
  tuieval/templates/packs/tool_call/system.txt,sha256=mKxu-lB-0YjLHXdgeFJa8x2C180OvcNz5cHv2N5Bg3U,254
43
43
  tuieval/templates/packs/tool_call/tests.yaml,sha256=Nv7c4syrzJ9cU8EBKSK_1ZSFx9V02aCjW4-2TBCPT7s,1859
44
44
  tuieval/templates/packs/tool_call/tools.yaml,sha256=HgxCkh0On2VwbHSsjAGM8tNCK7ZtKLGrs7lMCn3Fprs,1048
45
- tuieval-0.2.0.dev4.dist-info/METADATA,sha256=c653KXH4TRW-6b59HkXhbb0HW9jKyabSHWoD30aLI3w,12079
46
- tuieval-0.2.0.dev4.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
- tuieval-0.2.0.dev4.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
- tuieval-0.2.0.dev4.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
- tuieval-0.2.0.dev4.dist-info/RECORD,,
45
+ tuieval-0.2.0.dev6.dist-info/METADATA,sha256=3jzFS5nE1qalIrAnoAi5TBvAEZ9cyyVQoRGxbS3kvAs,12079
46
+ tuieval-0.2.0.dev6.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
+ tuieval-0.2.0.dev6.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
+ tuieval-0.2.0.dev6.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
+ tuieval-0.2.0.dev6.dist-info/RECORD,,