tuieval 0.2.0.dev3__py3-none-any.whl → 0.2.0.dev5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tuieval/_version.py CHANGED
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev3'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev3')
21
+ __version__ = version = '0.2.0.dev5'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev5')
23
23
 
24
24
  __commit_id__ = commit_id = None
tuieval/tune.py CHANGED
@@ -26,7 +26,8 @@ An output guard compares greedy answers with the default flags; an option that c
26
26
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
27
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
28
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. Any candidate that makes macOS swap is rejected. Servers without tune knobs are only
29
+ outdated. Any candidate under which macOS swaps while the server works is rejected (swapping
30
+ while the model loads, e.g. to lock it in RAM, is only noted). Servers without tune knobs are only
30
31
  measured.
31
32
 
32
33
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
@@ -315,8 +316,11 @@ def family(path):
315
316
  i = machines.read_gguf(engine_mod.expand(path))
316
317
  except (OSError, ValueError):
317
318
  return None
319
+ # The MTP (nextn) draft layers at the end are left out: GGUFs of the same model describe them
320
+ # differently, and whether draft-mtp is offered is decided per model (knobs) anyway.
321
+ heads = i["kv_heads_per_layer"][:len(i["kv_heads_per_layer"]) - i["mtp_layers"]]
318
322
  return (i["architecture"], i["layers"], i["embedding"], i["experts"], i["experts_used"],
319
- i["head_dim_k"], i["head_dim_v"], tuple(i["kv_heads_per_layer"]))
323
+ i["head_dim_k"], i["head_dim_v"], tuple(heads))
320
324
 
321
325
 
322
326
  @dataclasses.dataclass
@@ -471,12 +475,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
471
475
  emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
472
476
  start=starts[0], max_starts=max_starts)
473
477
  swap0 = machines.swapped_out_bytes()
478
+ swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
474
479
  try:
475
480
  with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
481
+ # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
482
+ # that's recorded, not held against the settings. Swapping while serving is.
483
+ loaded = machines.swapped_out_bytes()
476
484
  r = measure(eng, m, url, work, gen_tokens, passes, emit,
477
485
  settings.get("request_timeout_s", REQUEST_LIMIT_S))
486
+ end = machines.swapped_out_bytes()
487
+ if None not in (swap0, loaded, end):
488
+ swapped = (end - loaded) / 2**20
489
+ at_load = (loaded - swap0) / 2**20
490
+ if at_load > SWAP_LIMIT / 2**20:
491
+ emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
492
+ "(close apps for more headroom)")
478
493
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
479
494
  r.facts = dict(info["facts"])
495
+ if swapped is not None and at_load > SWAP_LIMIT / 2**20:
496
+ r.facts["load_swapped_mb"] = round(at_load)
480
497
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
481
498
  # harmless; sustained allocation over the limit is what makes the driver churn).
482
499
  settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
@@ -494,9 +511,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
494
511
  "close other apps for faster and fairer results")
495
512
  except engine_mod.ModelFailed as e:
496
513
  r = Measure(error=str(e).splitlines()[0])
497
- swap1 = machines.swapped_out_bytes()
498
- if not r.error and swap0 is not None and swap1 is not None and swap1 - swap0 > SWAP_LIMIT:
499
- r.error = f"macOS swapped {(swap1 - swap0) / 2**20:.0f} MB (memory too tight)"
514
+ if not r.error and swapped is not None and swapped > SWAP_LIMIT / 2**20:
515
+ r.error = f"macOS swapped {swapped:.0f} MB while the server worked (memory too tight)"
500
516
  cache[k] = r
501
517
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
502
518
  return r
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev3
3
+ Version: 0.2.0.dev5
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -1,6 +1,6 @@
1
1
  tuieval/__init__.py,sha256=myVXVX7htbwezW6k3x_FuKgn1RSZ2WcTaDEMByGn6Wg,312
2
2
  tuieval/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
3
- tuieval/_version.py,sha256=2ILWlstNdKCO-w6pi6qFsd7f8ZJMAD8pn_6jJ7S_lmY,533
3
+ tuieval/_version.py,sha256=s3bhRhA0S1Yy9V9RoQJoCTBpd5_-bYzuMi3GCnt_F5k,533
4
4
  tuieval/cli.py,sha256=ZaLoWrxyCABu4SJzx5U2gVUUmwcwRgpZVqdqnxyFMOk,4167
5
5
  tuieval/client.py,sha256=Vui7ucVv_GiKObGoVS_EWSZFUKPou9PWmutqMTrahXk,9313
6
6
  tuieval/compare.py,sha256=b-rVcMwq0Tos3oNKcpUP1HXULKhAQNAS4rlXYSY7t6k,25460
@@ -14,7 +14,7 @@ tuieval/run_evals.py,sha256=uPzidbRPUduBwP8sKyzg-U8cka0p-RsbZlVis6qQTlk,37526
14
14
  tuieval/scaffold.py,sha256=-AE5vKXeTtMZJ8OMk4UiIstoUWozTKjThKxZ87j0f9E,4720
15
15
  tuieval/selftest.py,sha256=xKfR73wMTYcDRSCBYQTNyKeR9bYU5HIYUSg7oZvNCyI,9330
16
16
  tuieval/tui.py,sha256=8diKtYE31ZUKq62rfMGdMkEbHgK6arRYpjFu-wcMVZs,131997
17
- tuieval/tune.py,sha256=zOs1cfKS3NhaFenjCY6fDb7oa8cx5lsoxgGaIEas_ys,32625
17
+ tuieval/tune.py,sha256=dPVs_l4xBpq0ymjreAvcMAk1QmPVazf9_tLxooGoY88,33876
18
18
  tuieval/verdict.py,sha256=oLKpRIruk7W8eF1YdmITFx9BgBK3kPKYViKL20Nzyu0,16306
19
19
  tuieval/watch_proxy.py,sha256=z916NRhEur04a9RnLm6ZpIEY1SnnUGPiHYbmlD9ktQ0,11035
20
20
  tuieval/workspace.py,sha256=V_HhkaN1odr0OpjNsPYradRWLpzDldM7SwpvxohBL4U,895
@@ -42,8 +42,8 @@ tuieval/templates/packs/tool_call/pack.toml,sha256=n4bx0lGJHSmW41uVqhfIXJ2BWZpmT
42
42
  tuieval/templates/packs/tool_call/system.txt,sha256=mKxu-lB-0YjLHXdgeFJa8x2C180OvcNz5cHv2N5Bg3U,254
43
43
  tuieval/templates/packs/tool_call/tests.yaml,sha256=Nv7c4syrzJ9cU8EBKSK_1ZSFx9V02aCjW4-2TBCPT7s,1859
44
44
  tuieval/templates/packs/tool_call/tools.yaml,sha256=HgxCkh0On2VwbHSsjAGM8tNCK7ZtKLGrs7lMCn3Fprs,1048
45
- tuieval-0.2.0.dev3.dist-info/METADATA,sha256=xZqfkjw1ZXBPMF-goFno6NYjbKpP379DrEAcKfZC2fg,12079
46
- tuieval-0.2.0.dev3.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
- tuieval-0.2.0.dev3.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
- tuieval-0.2.0.dev3.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
- tuieval-0.2.0.dev3.dist-info/RECORD,,
45
+ tuieval-0.2.0.dev5.dist-info/METADATA,sha256=_-BmKuon4FeyODFRw63PSgzIPfsziCzapadFDb349ps,12079
46
+ tuieval-0.2.0.dev5.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
+ tuieval-0.2.0.dev5.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
+ tuieval-0.2.0.dev5.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
+ tuieval-0.2.0.dev5.dist-info/RECORD,,