tuieval 0.2.0.dev1__tar.gz → 0.2.0.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/CHANGELOG.md +4 -1
  2. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/PKG-INFO +2 -2
  3. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/README.md +1 -1
  4. tuieval-0.2.0.dev3/docs/images/tui-setup.png +0 -0
  5. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/models.md +10 -9
  6. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/_version.py +2 -2
  7. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/compare.py +1 -1
  8. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/engine.py +18 -6
  9. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/export.py +10 -7
  10. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/machines.py +2 -3
  11. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/run_evals.py +1 -1
  12. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/models.toml +1 -1
  13. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/tests/test_tuieval.py +30 -0
  14. tuieval-0.2.0.dev1/docs/images/tui-setup.png +0 -0
  15. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/.gitignore +0 -0
  16. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/LICENSE +0 -0
  17. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/RELEASING.md +0 -0
  18. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/README.md +0 -0
  19. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/favicon.ico +0 -0
  20. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/readme-header.png +0 -0
  21. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/social-preview.png +0 -0
  22. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-1024.png +0 -0
  23. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-128.png +0 -0
  24. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-16.png +0 -0
  25. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-16.svg +0 -0
  26. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-256.png +0 -0
  27. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-32.png +0 -0
  28. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-48.png +0 -0
  29. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-512.png +0 -0
  30. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-64.png +0 -0
  31. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon-animated.svg +0 -0
  32. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/images/brand/tuieval-icon.svg +0 -0
  33. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/docs/writing-packs.md +0 -0
  34. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/pyproject.toml +0 -0
  35. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/__init__.py +0 -0
  36. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/__main__.py +0 -0
  37. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/cli.py +0 -0
  38. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/client.py +0 -0
  39. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/__init__.py +0 -0
  40. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/answer.py +0 -0
  41. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/code.py +0 -0
  42. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/rag.py +0 -0
  43. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/reply.py +0 -0
  44. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/graders/tool_call.py +0 -0
  45. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/packs.py +0 -0
  46. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/profiles.py +0 -0
  47. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/remove.py +0 -0
  48. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/scaffold.py +0 -0
  49. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/selftest.py +0 -0
  50. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/pack.toml +0 -0
  51. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/system.txt +0 -0
  52. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/answer/tests.yaml +0 -0
  53. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/pack.toml +0 -0
  54. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/system.txt +0 -0
  55. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/code/tests.yaml +0 -0
  56. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/pack.toml +0 -0
  57. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/system.txt +0 -0
  58. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/rag/tests.yaml +0 -0
  59. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/pack.toml +0 -0
  60. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/system.txt +0 -0
  61. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/reply/tests.yaml +0 -0
  62. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/pack.toml +0 -0
  63. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/system.txt +0 -0
  64. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/tests.yaml +0 -0
  65. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/templates/packs/tool_call/tools.yaml +0 -0
  66. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/tui.py +0 -0
  67. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/tune.py +0 -0
  68. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/verdict.py +0 -0
  69. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/watch_proxy.py +0 -0
  70. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/workspace.py +0 -0
  71. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/src/tuieval/yamlout.py +0 -0
  72. {tuieval-0.2.0.dev1 → tuieval-0.2.0.dev3}/tests/mock_server.py +0 -0
@@ -2,12 +2,15 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ - The fit check (context sized from the GGUF header, llama.cpp's memory use) only applies to servers whose command takes `{ctx}`. Servers that size their own memory keep the model's `max_context` instead of an estimate that didn't apply to them; `fit_check = true|false` on a server overrides it.
6
+ - `tuieval export pi` has no default presets path any more: set `[export.pi] presets` to your llama.cpp router's `--models-preset` file.
7
+ - A new README screenshot from a demo workspace.
5
8
  - Versions come from git tags: a tag `vX.Y.Z` is that release, and every commit on `main` after it is published automatically as a dev build `X.(Y+1).0.devN` (N = commits since the release). `pip install tuieval` keeps installing releases only; `pip install --pre tuieval` gets the latest build.
6
9
 
7
10
  ## 0.1.7 (replaces 0.1.6, withdrawn)
8
11
 
9
12
  - `tuieval tune` starts warm for fine-tunes: when a model with the same architecture and tensor shapes is already tuned on this machine, its flags are the starting point and only speculative decoding and micro-batch are re-tried (~4-6 server starts instead of 8-15). It tunes in full if the inherited flags fail, change answers or are slower than the defaults. `--cold` (or `[tune] warm_start = false`) always tunes in full; `[tune] warm_retest` names the re-tried knobs.
10
- - `tuieval export pi <model>` (and `tuieval tune --export-pi`) writes a model's serving settings to the pi coding agent: a llama-router preset section and pi's models.json. It shows the diff, asks, and backs every file up. See docs/models.md.
13
+ - `tuieval export pi <model>` (and `tuieval tune --export-pi`) writes a model's serving settings to the pi coding agent: a llama.cpp router presets section and pi's models.json. It shows the diff, asks, and backs every file up. See docs/models.md.
11
14
  - `tuieval list` lists the models, with hidden ones separately (`--packs` lists the packs).
12
15
  - `tuieval remove <model>` / `--hidden` removes models: their models.toml entry, results, tuning profiles and hidden mark, archived in removed/.
13
16
  - `tuieval --help` gives every command its own line.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev1
3
+ Version: 0.2.0.dev3
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -47,7 +47,7 @@ tuieval # open the TUI
47
47
 
48
48
  Once you've added your own packs and models, the setup screen looks like this (packs on the left, models with a verdict code per use case on the right, the highlighted model's details below):
49
49
 
50
- ![tuieval's setup screen with five eval packs and a dozen local models](https://raw.githubusercontent.com/ashe-wb/tuieval/main/docs/images/tui-setup.png)
50
+ ![tuieval's setup screen with five eval packs and three models with their verdicts](https://raw.githubusercontent.com/ashe-wb/tuieval/main/docs/images/tui-setup.png)
51
51
 
52
52
  ## What you get
53
53
 
@@ -21,7 +21,7 @@ tuieval # open the TUI
21
21
 
22
22
  Once you've added your own packs and models, the setup screen looks like this (packs on the left, models with a verdict code per use case on the right, the highlighted model's details below):
23
23
 
24
- ![tuieval's setup screen with five eval packs and a dozen local models](https://raw.githubusercontent.com/ashe-wb/tuieval/main/docs/images/tui-setup.png)
24
+ ![tuieval's setup screen with five eval packs and three models with their verdicts](https://raw.githubusercontent.com/ashe-wb/tuieval/main/docs/images/tui-setup.png)
25
25
 
26
26
  ## What you get
27
27
 
@@ -68,8 +68,8 @@ The best server flags differ per model and per machine, so tuieval splits them b
68
68
 
69
69
  ## Machines
70
70
 
71
- - **Machine id** is detected automatically (`m1max-32gb`, `m4pro-24gb`, …; set `EVALS_MACHINE` to rename it). Every result records the machine, the server version and the exact speed flags used. `tuieval machines` lists this machine and every other machine that has run the evals (they record themselves in `tuning/`).
72
- - **Fit check (Apple Silicon, llama):** before starting a GGUF, tuieval reads its header (layers, KV heads, hybrid attention layers) and picks the largest context that fits this machine's GPU memory, capped at `max_ctx`. A model that can't fit at 8k context is skipped with *doesn't fit on <machine>* instead of swapping. Packs that need more context than fits are skipped. It never quantizes the KV cache on its own, since that changes answers.
71
+ - **Machine id** is detected automatically (e.g. `m3max-64gb`, `m2-16gb`; set `EVALS_MACHINE` to rename it). Every result records the machine, the server version and the exact speed flags used. `tuieval machines` lists this machine and every other machine that has run the evals (they record themselves in `tuning/`).
72
+ - **Fit check (Apple Silicon, llama):** before starting a GGUF, tuieval reads its header (layers, KV heads, hybrid attention layers) and picks the largest context that fits this machine's GPU memory, capped at `max_ctx`. A model that can't fit at 8k context is skipped with *doesn't fit on <machine>* instead of swapping. Packs that need more context than fits are skipped. It never quantizes the KV cache on its own, since that changes answers. Servers that size their own memory (their `cmd` doesn't take `{ctx}`) are left alone: their context is the model's `max_context`; `fit_check` on the server changes that.
73
73
  - **Per-machine settings** go under `[machines.<id>]`: `memory_headroom_gb` (GPU memory kept free for macOS, default 4) and `gpu_residency_gb` (see the stall guard).
74
74
 
75
75
  ## Tuning
@@ -90,18 +90,18 @@ The knobs are `[servers.<name>.tune]` in `models.toml`: each knob is a list of o
90
90
 
91
91
  `tuieval export pi <model>` makes pi serve a model the way the evals did: the same file, the output-affecting flags, the context from the fit check, the model's sampling, and the speed flags tuned on this machine. `tuieval tune <model> --export-pi` does it right after tuning.
92
92
 
93
- It writes a `[<id>]` section in llama-router's presets file (keys that equal its `[*]` section are left out) and an entry under pi's `llama` provider `modelOverrides` (name, context window, vision, reasoning). It exports models on llama servers.
93
+ It writes a `[<id>]` section in the model presets file of a llama.cpp router (`llama-server --models-preset <file>`; keys that equal its `[*]` section are left out) and an entry under pi's `llama` provider `modelOverrides` (name, context window, vision, reasoning). It exports models on llama servers.
94
94
 
95
- It shows the diff and the notes first (untuned or outdated tuning, other preset sections that name missing files, a reasoning effort to pick in pi), asks, backs every file up as `<file>.bak-tuieval-<time>`, and refuses to write a file that changed in the meantime. `--dry-run` only shows; `--yes` doesn't ask. Restart the router afterwards (`llama-router restart`).
95
+ It shows the diff and the notes first (untuned or outdated tuning, other preset sections that name missing files, a reasoning effort to pick in pi), asks, backs every file up as `<file>.bak-tuieval-<time>`, and refuses to write a file that changed in the meantime. `--dry-run` only shows; `--yes` doesn't ask. Restart the router afterwards so it reads the new presets.
96
96
 
97
97
  The id pi sees defaults to the GGUF's name; set `pi_id` and `pi_name` on the model (or pass `--id`/`--name`). Paths and provider names come from `[export.pi]` in `models.toml`:
98
98
 
99
99
  ```toml
100
100
  [export.pi]
101
- presets = "~/models/presets.ini"
102
- pi_models = "~/.pi/agent/models.json"
101
+ presets = "/path/to/presets.ini" # required: the router's --models-preset file
102
+ pi_models = "~/.pi/agent/models.json" # pi's default
103
103
  llama_provider = "llama" # pi's provider name for the router
104
- servers = ["llama"] # models.toml servers that llama-router can serve
104
+ servers = ["llama"] # models.toml servers the router can serve
105
105
  ```
106
106
 
107
107
  ## Speed verdicts
@@ -110,7 +110,7 @@ Speed is judged separately, per machine: the same model can be production-grade
110
110
 
111
111
  ## The stall guard (Apple Silicon)
112
112
 
113
- Measured on an M1 Max 32 GB: once the system's GPU allocations pass about half of RAM, the GPU driver evicts and re-maps memory on every GPU job, the server spends 70–95% of its CPU in the kernel while the GPU idles, and servers that submit many small GPU jobs slow to a crawl. For servers with `stall_guard = true`, tuieval samples the server's own vs kernel CPU time and the system's GPU allocation every 5 s during runs and tuning. If more than 70% of its CPU goes to the kernel for a minute, the run stops with an explanation instead of crawling for hours. Finished answers are kept and resume next time. If your machine behaves differently, set `gpu_residency_gb` under `[machines.<id>]`.
113
+ On Apple Silicon Macs, once the system's GPU allocations pass about half of RAM, the GPU driver evicts and re-maps memory on every GPU job, the server spends 70–95% of its CPU in the kernel while the GPU idles, and servers that submit many small GPU jobs slow to a crawl. For servers with `stall_guard = true`, tuieval samples the server's own vs kernel CPU time and the system's GPU allocation every 5 s during runs and tuning. If more than 70% of its CPU goes to the kernel for a minute, the run stops with an explanation instead of crawling for hours. Finished answers are kept and resume next time. If your machine behaves differently, set `gpu_residency_gb` under `[machines.<id>]`.
114
114
 
115
115
  ## Server options
116
116
 
@@ -120,7 +120,8 @@ Measured on an M1 Max 32 GB: once the system's GPU allocations pass about half o
120
120
  | `url` | an already-running server's base URL |
121
121
  | `port` | the port `cmd` serves on |
122
122
  | `cwd` | folder to start `cmd` in |
123
- | `model_is_path` | `model` is a file or folder: check it exists (and, for GGUFs, that it fits) before starting |
123
+ | `model_is_path` | `model` is a file or folder: check it exists before starting |
124
+ | `fit_check` | size the context from the GGUF header and skip models that don't fit (the fit check; it assumes llama.cpp's memory use). Default: on for servers whose `cmd` takes `{ctx}`, off for others, whose context is the model's `max_context` |
124
125
  | `vision_args` | flags appended for models with `mmproj` |
125
126
  | `kv_type`, `max_ctx` | llama defaults: KV-cache type, upper bound on context |
126
127
  | `ctx_flag` | the server's context flag, so long packs are skipped if a configured context is too small |
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev1'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev1')
21
+ __version__ = version = '0.2.0.dev3'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev3')
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -3,7 +3,7 @@
3
3
  Usage:
4
4
  tuieval compare # everything in results/<model>/<pack>.json
5
5
  tuieval compare --speed # speed & tokens table (TTFT, tok/s, tokens, memory)
6
- tuieval compare --machine m4pro-24gb # speed on another machine (measured there, or projected)
6
+ tuieval compare --machine m3max-64gb # speed on another machine (measured there, or projected)
7
7
  tuieval compare --pairwise --failures # is each difference real? which tests failed?
8
8
  tuieval compare results/some-model/*.json # specific files
9
9
  """
@@ -96,6 +96,16 @@ def load_config(path):
96
96
  return cfg
97
97
 
98
98
 
99
+ def fit_check(server):
100
+ """Whether tuieval sizes the context for this server's models from GGUF headers (machines.fit,
101
+ which assumes llama.cpp's memory use). By default only servers whose command takes {ctx}, i.e.
102
+ where tuieval chooses the context; others size their own memory, and their context is the
103
+ model's max_context. models.toml `fit_check = true|false` on a server overrides it."""
104
+ if "fit_check" in server:
105
+ return bool(server["fit_check"])
106
+ return any("{ctx}" in str(a) for a in server.get("cmd", []))
107
+
108
+
99
109
  def effective_sampling(cfg, m):
100
110
  """The request settings this model runs with: [sampling], then the model's own overrides."""
101
111
  s = dict(cfg["sampling"])
@@ -1018,13 +1028,15 @@ class Engine:
1018
1028
  path = expand(m["model"])
1019
1029
  if server.get("model_is_path") and os.path.isfile(path):
1020
1030
  try:
1021
- mmproj = expand(m.get("mmproj", ""))
1022
- extra = os.path.getsize(mmproj) if mmproj and os.path.isfile(mmproj) else 0
1023
- headroom = self.machine_settings(mc.id).get("memory_headroom_gb", 4.0)
1024
- f = machines.fit(path, mc, kv_type=kv or "f16", headroom_gb=headroom, extra_bytes=extra,
1025
- want_ctx=want)
1026
1031
  info = machines.read_gguf(path)
1027
- ctx, fits, note, mtp, size = f.max_ctx, f.fits, f.note, info["mtp_layers"], info["bytes"]
1032
+ mtp, size = info["mtp_layers"], info["bytes"]
1033
+ if fit_check(server):
1034
+ mmproj = expand(m.get("mmproj", ""))
1035
+ extra = os.path.getsize(mmproj) if mmproj and os.path.isfile(mmproj) else 0
1036
+ headroom = self.machine_settings(mc.id).get("memory_headroom_gb", 4.0)
1037
+ f = machines.fit(path, mc, kv_type=kv or "f16", headroom_gb=headroom, extra_bytes=extra,
1038
+ want_ctx=want)
1039
+ ctx, fits, note = f.max_ctx, f.fits, f.note
1028
1040
  except (OSError, ValueError) as e:
1029
1041
  note = f"couldn't read the GGUF header ({e}); context not checked"
1030
1042
  profile, perf, source = None, [], "n/a"
@@ -1,17 +1,18 @@
1
1
  """Export a model's serving settings to the pi coding agent, so pi serves it exactly as the evals did:
2
2
  the same model file, output-affecting flags and context, plus the speed flags tuned on this machine.
3
3
 
4
- a section in llama-router's presets (~/models/presets.ini; keys that equal the [*] section are
5
- left out) and an entry under pi's `llama` provider modelOverrides. Models on llama servers only.
4
+ a section in a llama.cpp router's model presets file (llama-server --models-preset; keys that
5
+ equal its [*] section are left out) and an entry under pi's `llama` provider modelOverrides.
6
+ Models on llama servers only.
6
7
 
7
8
  Every file is backed up next to itself (<file>.bak-tuieval-<time>) before it changes, and the diff
8
9
  is shown first. Paths come from models.toml [export.pi]:
9
10
 
10
11
  [export.pi]
11
- presets = "~/models/presets.ini" # llama-router --models-preset
12
+ presets = "/path/to/presets.ini" # required: the router's --models-preset file
12
13
  pi_models = "~/.pi/agent/models.json"
13
14
  llama_provider = "llama" # pi's provider name for the router
14
- servers = ["llama"] # models.toml servers that llama-router can serve
15
+ servers = ["llama"] # models.toml servers the router can serve
15
16
  """
16
17
  import configparser
17
18
  import difflib
@@ -23,7 +24,7 @@ import time
23
24
 
24
25
  from . import engine as engine_mod
25
26
 
26
- DEFAULTS = {"presets": "~/models/presets.ini", "pi_models": "~/.pi/agent/models.json",
27
+ DEFAULTS = {"presets": None, "pi_models": "~/.pi/agent/models.json",
27
28
  "llama_provider": "llama", "servers": ["llama"]}
28
29
  # llama.cpp short flags -> the long names presets use as keys
29
30
  SHORT = {"-m": "model", "-c": "ctx-size", "-t": "threads", "-tb": "threads-batch", "-ub": "ubatch-size",
@@ -127,7 +128,7 @@ def plan(eng, label, model_id=None, name=None):
127
128
  m = eng.model(label)
128
129
  s = settings(eng)
129
130
  if m["server"] not in s["servers"]:
130
- raise ExportError(f"{label} runs on the {m['server']} server; pi export writes llama-router presets, "
131
+ raise ExportError(f"{label} runs on the {m['server']} server; pi export writes llama.cpp router presets, "
131
132
  "for llama servers ([export.pi] servers lists them)")
132
133
  model_id = model_id or m.get("pi_id") or default_id(m)
133
134
  name = name or m.get("pi_name") or model_id
@@ -157,6 +158,8 @@ def plan(eng, label, model_id=None, name=None):
157
158
  for k, key in SAMPLING_KEYS.items():
158
159
  if sampling.get(k) is not None:
159
160
  flags[key] = str(sampling[k])
161
+ if not s.get("presets"):
162
+ raise ExportError("set [export.pi] presets in models.toml to your llama.cpp router's --models-preset file")
160
163
  path = engine_mod.expand(s["presets"])
161
164
  old = _read(path)
162
165
  if not old:
@@ -183,7 +186,7 @@ def plan(eng, label, model_id=None, name=None):
183
186
  "maxTokens": entry.get("maxTokens", sampling.get("max_tokens")),
184
187
  "reasoning": bool(sampling.get("enable_thinking", True)),
185
188
  "input": ["text", "image"] if vision else ["text"]})
186
- notes.append("restart llama-router to load the new presets (llama-router restart)")
189
+ notes.append("restart the llama.cpp router so it reads the new presets")
187
190
  if sampling.get("reasoning_effort"):
188
191
  notes.append(f"the evals ran reasoning_effort {sampling['reasoning_effort']}: pick that thinking level in pi")
189
192
  pi_new = json.dumps(pi, indent=2, ensure_ascii=False) + "\n"
@@ -1,6 +1,6 @@
1
1
  """Which machine this is, what a model needs, and whether it fits.
2
2
 
3
- detect() -> Machine (chip, cores, RAM, GPU memory limit, id like "m1max-32gb")
3
+ detect() -> Machine (chip, cores, RAM, GPU memory limit, id like "m3max-64gb")
4
4
  read_gguf(path) -> GGUF header facts (architecture, layers, KV heads, context length, …)
5
5
  fit(model_path, …) -> Fit (largest context that fits this machine, estimated memory)
6
6
 
@@ -219,8 +219,7 @@ def swapped_out_bytes():
219
219
 
220
220
 
221
221
  # ---------------------------------------------------------------- GPU residency
222
- # Measured on an M1 Max 32 GB with macOS 27.0: once the system's GPU allocations exceed about half
223
- # of RAM, the GPU driver evicts and re-maps memory on every command submission (70-95% of the
222
+ # On Apple Silicon Macs, once the system's GPU allocations exceed about half of RAM, the GPU driver evicts and re-maps memory on every command submission (70-95% of the
224
223
  # server's CPU in the kernel, GPU idle), whatever iogpu.wired_limit_mb says. Servers that submit
225
224
  # many small GPU jobs collapse; see the stall guard in docs/models.md.
226
225
  RESIDENCY_FRACTION = 0.5
@@ -576,7 +576,7 @@ def cmd_export(argv):
576
576
  description="Write a model's serving settings (model file, output-affecting flags, "
577
577
  "context and the speed flags tuned on this machine) to another tool. "
578
578
  "Shows the diff and asks first; every file is backed up.")
579
- p.add_argument("target", choices=["pi"], help="pi: the pi coding agent (llama-router presets and pi's "
579
+ p.add_argument("target", choices=["pi"], help="pi: the pi coding agent (llama.cpp router presets and pi's "
580
580
  "models.json)")
581
581
  p.add_argument("labels", nargs="+", help="models to export")
582
582
  p.add_argument("--id", help="the model id pi sees (default: models.toml pi_id, else the GGUF's name)")
@@ -107,7 +107,7 @@ headers = { "X-Title" = "tuieval" }
107
107
  # cmd = ["~/llama.cpp/build/bin/llama-bench"]
108
108
 
109
109
  # Optional per-machine settings, by machine id (tuieval machines shows the ids).
110
- # [machines.m4pro-24gb]
110
+ # [machines.m3max-64gb]
111
111
  # memory_headroom_gb = 3 # GPU memory kept free for macOS and other apps (default 4)
112
112
  # gpu_residency_gb = 12 # GPU memory the driver keeps resident before churning (default: half of RAM)
113
113
 
@@ -240,6 +240,31 @@ def fake_gguf(path, embedding=5120, size=0):
240
240
  f.write(out + b"\0" * max(0, size - len(out)))
241
241
 
242
242
 
243
+ class FitCheck(unittest.TestCase):
244
+ def test_default_follows_the_ctx_placeholder(self):
245
+ from tuieval import engine
246
+ self.assertTrue(engine.fit_check({"cmd": ["llama-server", "-c", "{ctx}"]}))
247
+ self.assertFalse(engine.fit_check({"cmd": ["other-server", "--model", "{model}"]}))
248
+ self.assertFalse(engine.fit_check({"url": "http://x"}))
249
+ self.assertTrue(engine.fit_check({"cmd": ["other"], "fit_check": True}))
250
+ self.assertFalse(engine.fit_check({"cmd": ["llama", "-c", "{ctx}"], "fit_check": False}))
251
+
252
+ def test_own_memory_servers_keep_their_context(self):
253
+ with tempfile.TemporaryDirectory() as tmp:
254
+ ws = os.path.join(tmp, "ws")
255
+ tuieval(tmp, "init", ws)
256
+ fake_gguf(os.path.join(tmp, "m.gguf"), size=1000)
257
+ with open(os.path.join(ws, "models.toml"), "a") as f:
258
+ f.write(f'\n[servers.own]\ncmd = ["own-server", "--model", "{{model}}"]\nport = 18500\nmodel_is_path = true\n'
259
+ f'\n[[models]]\nlabel = "sized"\nserver = "llama"\nmodel = "{tmp}/m.gguf"\n'
260
+ f'\n[[models]]\nlabel = "own"\nserver = "own"\nmodel = "{tmp}/m.gguf"\nmax_context = 131072\n')
261
+ code = ("from tuieval import engine; e = engine.Engine(); "
262
+ "print(e.serving(e.model('own')).ctx, e.serving(e.model('sized')).fit_note != '')")
263
+ out = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True,
264
+ env=dict(os.environ, TUIEVAL_HOME=ws)).stdout.split()
265
+ self.assertEqual(out, ["131072", "True"]) # own server: its max_context; llama: fit-checked
266
+
267
+
243
268
  class WarmTune(unittest.TestCase):
244
269
  """tune.tune starting from a tuned model of the same family, on a fake engine whose speed is a
245
270
  function of the flags."""
@@ -398,6 +423,11 @@ class ExportPi(unittest.TestCase):
398
423
  pi = json.loads(next(c for c in changes if c.path.endswith("models.json")).new)
399
424
  self.assertEqual(pi["providers"]["llama"]["modelOverrides"]["Old"]["contextWindow"], 98304)
400
425
 
426
+ def test_presets_path_is_required(self):
427
+ del self.eng.cfg["export"]["pi"]["presets"]
428
+ with self.assertRaises(self.export.ExportError):
429
+ self.export.plan(self.eng, "l")
430
+
401
431
  def test_only_llama_servers(self):
402
432
  with self.assertRaises(self.export.ExportError):
403
433
  self.export.plan(self.eng, "s")
File without changes
File without changes
File without changes