tuieval 0.2.0.dev4__tar.gz → 0.2.0.dev6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/CHANGELOG.md +1 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/PKG-INFO +1 -1
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/models.md +1 -1
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/_version.py +2 -2
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/run_evals.py +2 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/tui.py +2 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/tune.py +37 -7
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/tests/test_tuieval.py +35 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/.gitignore +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/LICENSE +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/README.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/RELEASING.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/README.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/favicon.ico +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/readme-header.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/social-preview.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-1024.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-128.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-16.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-16.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-256.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-32.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-48.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-512.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-64.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon-animated.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/brand/tuieval-icon.svg +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/images/tui-setup.png +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/docs/writing-packs.md +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/pyproject.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/__init__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/__main__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/cli.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/client.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/compare.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/engine.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/export.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/__init__.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/answer.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/code.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/rag.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/reply.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/graders/tool_call.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/machines.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/packs.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/profiles.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/remove.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/scaffold.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/selftest.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/models.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/code/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/verdict.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/watch_proxy.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/workspace.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/src/tuieval/yamlout.py +0 -0
- {tuieval-0.2.0.dev4 → tuieval-0.2.0.dev6}/tests/mock_server.py +0 -0
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
- `tuieval tune`: swapping the defaults already cause is a warning, not a failure; an option is rejected only if it swaps more than the defaults.
|
|
5
6
|
- Warm-start tuning recognises fine-tunes whose GGUF describes the MTP draft layer differently (e.g. listing KV heads per layer) as the same model shape, so they start from an already-tuned sibling instead of tuning in full.
|
|
6
7
|
- The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
|
|
7
8
|
- `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tuieval
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev6
|
|
4
4
|
Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ashe-wb/tuieval
|
|
6
6
|
Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
|
|
@@ -78,7 +78,7 @@ The best server flags differ per model and per machine, so tuieval splits them b
|
|
|
78
78
|
|
|
79
79
|
1. If `llama-bench` is installed, it sweeps threads, micro-batch and flash attention first (fast, no server starts).
|
|
80
80
|
2. Then it starts the real server with one knob changed at a time and times a fixed **built-in** workload (three short prompts, three medium ones and one ~8k-token prompt), so tuning needs no packs and speeds are comparable between workspaces.
|
|
81
|
-
3. An **output guard** rejects any option that changes greedy answers beyond noise.
|
|
81
|
+
3. An **output guard** rejects any option that changes greedy answers beyond noise. An option under which macOS swaps more than with the defaults is rejected (it costs memory). If the defaults already push other apps to swap, the tune warns and carries on.
|
|
82
82
|
|
|
83
83
|
Expect 8–15 server starts, about 20–30 minutes for a 27B model, once per model per machine. The result is saved in `tuning/<machine>/<model>.toml` and used by every later run there. Models without a profile run with each knob's first option and show *untuned*. A profile is marked for retuning when the model file or server version changes.
|
|
84
84
|
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2, 0, '
|
|
21
|
+
__version__ = version = '0.2.0.dev6'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0, 'dev6')
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -530,6 +530,8 @@ def cmd_tune(argv):
|
|
|
530
530
|
say(f"{label}: started from {ws['from']}'s flags; re-tried {', '.join(ws['retested']) or 'nothing'}")
|
|
531
531
|
if not pr["meta"].get("answer_guard", True):
|
|
532
532
|
say(f"{label}: its answers depend on these settings; rerun its evals on this machine", "\033[33m")
|
|
533
|
+
for w in pr["meta"].get("warnings", []):
|
|
534
|
+
say(f"{label}: warning: {w}", YELLOW)
|
|
533
535
|
if pr["meta"].get("rejected"):
|
|
534
536
|
say(f"{label}: rejected because answers changed: {'; '.join(pr['meta']['rejected'])}")
|
|
535
537
|
if a.export_pi and not export_pi(e, label):
|
|
@@ -2565,6 +2565,8 @@ class EvalsApp(App):
|
|
|
2565
2565
|
p = tune.tune(e, label, lambda kind, **d: on_event(kind, label=label, **d))
|
|
2566
2566
|
ms = p["measured"]
|
|
2567
2567
|
gain = f" ({tune.gain_text(ms)})" if tune.gain_text(ms) else ""
|
|
2568
|
+
if p["meta"].get("warnings"):
|
|
2569
|
+
gain += " (warning: it pushes other apps' memory to swap; see the log)"
|
|
2568
2570
|
if not p["meta"].get("answer_guard", True):
|
|
2569
2571
|
gain += " (answers depend on these settings: rerun its evals here)"
|
|
2570
2572
|
done.append(f"{label}{gain}")
|
|
@@ -26,7 +26,9 @@ An output guard compares greedy answers with the default flags; an option that c
|
|
|
26
26
|
noise is rejected, because speed flags must not change answers. Servers whose answers depend on
|
|
27
27
|
the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
|
|
28
28
|
become part of the results fingerprint instead, so retuning marks that machine's results
|
|
29
|
-
outdated.
|
|
29
|
+
outdated. A candidate under which macOS swaps more than with the defaults is rejected (it costs
|
|
30
|
+
memory). Swapping the defaults already cause (the model, its context, other apps) is a warning, and
|
|
31
|
+
swapping while the model loads is only noted. Servers without tune knobs are only
|
|
30
32
|
measured.
|
|
31
33
|
|
|
32
34
|
The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
|
|
@@ -48,7 +50,7 @@ from . import profiles
|
|
|
48
50
|
|
|
49
51
|
GEN_TOKENS = 128 # generated per workload prompt
|
|
50
52
|
DECODE_TOKENS = 256 # generated per prompt for the decode objective
|
|
51
|
-
SWAP_LIMIT = 256 * 2**20 # a candidate that swaps
|
|
53
|
+
SWAP_LIMIT = 256 * 2**20 # a candidate that swaps this much more than the defaults is rejected
|
|
52
54
|
GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
|
|
53
55
|
# is rejected even before it stalls: servers grow as requests arrive
|
|
54
56
|
REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
|
|
@@ -69,6 +71,7 @@ class Measure:
|
|
|
69
71
|
load_s: float | None = None
|
|
70
72
|
error: str = ""
|
|
71
73
|
facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
|
|
74
|
+
swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
|
|
72
75
|
|
|
73
76
|
def score(self, objective):
|
|
74
77
|
"""Lower is better."""
|
|
@@ -461,7 +464,8 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
461
464
|
knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
|
|
462
465
|
fixed = list(server.get("perf", [])) if server.get("cmd") else []
|
|
463
466
|
cache, starts, notes, rejected, bad = {}, [0], [], [], set() # bad: (knob, option) that failed
|
|
464
|
-
warned = []
|
|
467
|
+
warned, warnings = [], []
|
|
468
|
+
swap_ref = [None] # MB the defaults swapped while serving: the machine's baseline
|
|
465
469
|
|
|
466
470
|
def evaluate(choices, why):
|
|
467
471
|
k = tuple(sorted(choices.items()))
|
|
@@ -474,12 +478,25 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
474
478
|
emit("tune_step", message=f"[{starts[0]}] {why}: {' '.join(args) or '(server defaults)'}",
|
|
475
479
|
start=starts[0], max_starts=max_starts)
|
|
476
480
|
swap0 = machines.swapped_out_bytes()
|
|
481
|
+
swapped = None # MB swapped out while the loaded server worked (loading itself doesn't count)
|
|
477
482
|
try:
|
|
478
483
|
with eng.serve(m, perf_args=fixed + args, log_name=f"{label}.tune") as (url, info):
|
|
484
|
+
# Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
|
|
485
|
+
# that's recorded, not held against the settings. Swapping while serving is.
|
|
486
|
+
loaded = machines.swapped_out_bytes()
|
|
479
487
|
r = measure(eng, m, url, work, gen_tokens, passes, emit,
|
|
480
488
|
settings.get("request_timeout_s", REQUEST_LIMIT_S))
|
|
489
|
+
end = machines.swapped_out_bytes()
|
|
490
|
+
if None not in (swap0, loaded, end):
|
|
491
|
+
swapped = (end - loaded) / 2**20
|
|
492
|
+
at_load = (loaded - swap0) / 2**20
|
|
493
|
+
if at_load > SWAP_LIMIT / 2**20:
|
|
494
|
+
emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
|
|
495
|
+
"(close apps for more headroom)")
|
|
481
496
|
r.load_s = round(info["load_s"], 1) if info["load_s"] else None
|
|
482
497
|
r.facts = dict(info["facts"])
|
|
498
|
+
if swapped is not None and at_load > SWAP_LIMIT / 2**20:
|
|
499
|
+
r.facts["load_swapped_mb"] = round(at_load)
|
|
483
500
|
# Settled allocation after the timed pass (brief peaks while processing a prompt are
|
|
484
501
|
# harmless; sustained allocation over the limit is what makes the driver churn).
|
|
485
502
|
settled = machines.gpu_allocated_gb() if server.get("stall_guard") else None
|
|
@@ -497,18 +514,30 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
497
514
|
"close other apps for faster and fairer results")
|
|
498
515
|
except engine_mod.ModelFailed as e:
|
|
499
516
|
r = Measure(error=str(e).splitlines()[0])
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
517
|
+
r.swapped_mb = swapped
|
|
518
|
+
# Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
|
|
519
|
+
# a flag; only swapping beyond the defaults' does.
|
|
520
|
+
extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
|
|
521
|
+
if not r.error and extra is not None and extra > SWAP_LIMIT / 2**20:
|
|
522
|
+
r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (uses too much memory)"
|
|
503
523
|
cache[k] = r
|
|
504
524
|
emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
|
|
505
525
|
return r
|
|
506
526
|
|
|
527
|
+
def set_baseline(r):
|
|
528
|
+
"""The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
|
|
529
|
+
swap_ref[0] = r.swapped_mb or 0
|
|
530
|
+
if swap_ref[0] > SWAP_LIMIT / 2**20:
|
|
531
|
+
warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
|
|
532
|
+
"this machine; close other apps or lower this model's context for more headroom")
|
|
533
|
+
emit("tune_step", message=" warning: " + warnings[-1])
|
|
534
|
+
|
|
507
535
|
base = None
|
|
508
536
|
if not server.get("cmd") or not knob_opts:
|
|
509
537
|
r = evaluate({}, "measuring (no speed knobs to tune)")
|
|
510
538
|
if r.error:
|
|
511
539
|
raise engine_mod.ModelFailed(r.error)
|
|
540
|
+
set_baseline(r)
|
|
512
541
|
best, best_r = {}, r
|
|
513
542
|
method, bench_info = "measured", None
|
|
514
543
|
else:
|
|
@@ -517,6 +546,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
517
546
|
if base is None or base.error:
|
|
518
547
|
raise engine_mod.ModelFailed(f"the server doesn't start with the default flags: "
|
|
519
548
|
f"{base.error if base else 'no starts left'}")
|
|
549
|
+
set_baseline(base)
|
|
520
550
|
settled, bench_info, warm_info = {}, None, None
|
|
521
551
|
best, best_r, search = dict(default), base, list(knob_opts)
|
|
522
552
|
sib = find_sibling(eng, m, knob_opts, sv.machine.id) \
|
|
@@ -601,7 +631,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
|
|
|
601
631
|
"workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
|
|
602
632
|
"llama_bench": bool(bench_info), "answer_guard": guard,
|
|
603
633
|
**({"warm_start": warm_info} if knob_opts and warm_info else {}),
|
|
604
|
-
"rejected": rejected, "notes": notes[:10]},
|
|
634
|
+
"rejected": rejected, "notes": notes[:10], "warnings": warnings},
|
|
605
635
|
}
|
|
606
636
|
path = profiles.save(sv.machine.id, label, profile, eng.tuning_dir)
|
|
607
637
|
eng._serving.clear()
|
|
@@ -324,6 +324,41 @@ class WarmTune(unittest.TestCase):
|
|
|
324
324
|
self.tune.measure = self.orig
|
|
325
325
|
self.tmp.cleanup()
|
|
326
326
|
|
|
327
|
+
def _swap(self, at_load_mb, while_serving_mb):
|
|
328
|
+
"""Fake macOS swap counter: each server start swaps at_load_mb while loading and
|
|
329
|
+
while_serving_mb(args) during the work (args: that start's flags)."""
|
|
330
|
+
from tuieval import machines
|
|
331
|
+
state = {"n": 0, "total": 0}
|
|
332
|
+
|
|
333
|
+
def swapped_out_bytes():
|
|
334
|
+
step = state["n"] % 3 # 0: before the start, 1: loaded, 2: after the work
|
|
335
|
+
mb = {0: 0, 1: at_load_mb, 2: while_serving_mb(self.eng.starts[-1]) if step == 2 else 0}[step]
|
|
336
|
+
state["total"] += mb * 2**20
|
|
337
|
+
state["n"] += 1
|
|
338
|
+
return state["total"]
|
|
339
|
+
orig = machines.swapped_out_bytes
|
|
340
|
+
machines.swapped_out_bytes = swapped_out_bytes
|
|
341
|
+
self.addCleanup(setattr, machines, "swapped_out_bytes", orig)
|
|
342
|
+
|
|
343
|
+
def test_swapping_while_loading_is_only_noted(self):
|
|
344
|
+
self._swap(at_load_mb=900, while_serving_mb=lambda args: 0)
|
|
345
|
+
p = self.tune.tune(self.eng, "new", use_bench=False)
|
|
346
|
+
self.assertEqual(p["measured"]["server_load_swapped_mb"], 900)
|
|
347
|
+
self.assertEqual(p["meta"]["warnings"], [])
|
|
348
|
+
|
|
349
|
+
def test_swapping_every_candidate_shares_is_a_warning(self):
|
|
350
|
+
self._swap(at_load_mb=0, while_serving_mb=lambda args: 800)
|
|
351
|
+
p = self.tune.tune(self.eng, "new", use_bench=False)
|
|
352
|
+
self.assertEqual(p["args"], ["-t", "6", "-ub", "256", "--spec-type", "draft-mtp"]) # tuned as usual
|
|
353
|
+
self.assertIn("800 MB", p["meta"]["warnings"][0])
|
|
354
|
+
|
|
355
|
+
def test_a_flag_that_swaps_more_than_the_defaults_is_rejected(self):
|
|
356
|
+
# -ub 256 is the fastest micro-batch here, but it costs 600 MB of swap the defaults don't
|
|
357
|
+
self._swap(at_load_mb=0, while_serving_mb=lambda args: 600 if "256" in args else 100)
|
|
358
|
+
p = self.tune.tune(self.eng, "new", use_bench=False, warm=False)
|
|
359
|
+
self.assertNotIn("256", p["args"])
|
|
360
|
+
self.assertTrue(any("500 MB more than with the defaults" in n for n in p["meta"]["notes"]))
|
|
361
|
+
|
|
327
362
|
def test_family_ignores_the_mtp_layer(self):
|
|
328
363
|
# one GGUF lists KV heads per layer, none on the MTP (last) layer; the other gives one number
|
|
329
364
|
d = self.tmp.name
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|