tuieval 0.2.0.dev6__py3-none-any.whl → 0.2.0.dev9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tuieval/_version.py CHANGED
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0.dev6'
22
- __version_tuple__ = version_tuple = (0, 2, 0, 'dev6')
21
+ __version__ = version = '0.2.0.dev9'
22
+ __version_tuple__ = version_tuple = (0, 2, 0, 'dev9')
23
23
 
24
24
  __commit_id__ = commit_id = None
tuieval/machines.py CHANGED
@@ -225,6 +225,17 @@ def swapped_out_bytes():
225
225
  RESIDENCY_FRACTION = 0.5
226
226
 
227
227
 
228
+ def memory_pressure_level():
229
+ """macOS memory pressure: 1 normal, 2 warning, 4 critical (the system starts ending processes).
230
+ None elsewhere."""
231
+ try:
232
+ out = subprocess.run(["sysctl", "-n", "kern.memorystatus_vm_pressure_level"], capture_output=True,
233
+ text=True, timeout=5).stdout
234
+ return int(out.strip())
235
+ except (OSError, subprocess.TimeoutExpired, ValueError):
236
+ return None
237
+
238
+
228
239
  def gpu_residency_gb(machine, override=None):
229
240
  """GPU memory the driver keeps resident without churning (models.toml
230
241
  [machines.<id>] gpu_residency_gb overrides the half-of-RAM default)."""
tuieval/run_evals.py CHANGED
@@ -523,7 +523,9 @@ def cmd_tune(argv):
523
523
  pr = result["profile"]
524
524
  ms = pr["measured"]
525
525
  gain = ", " + tune.gain_text(ms) if tune.gain_text(ms) else ""
526
- result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else f"{ms['total_s']}s for the workload"
526
+ result = f"{ms['tg_tps']} tok/s decode" if ms.get("objective") == "decode" else \
527
+ f"{ms['projected_s']}s projected for the workload" if ms.get("objective") == "projected" and ms.get("projected_s") \
528
+ else f"{ms['total_s']}s for the workload"
527
529
  say(f"{label}: {' '.join(pr['args']) or '(no knobs)'} -> {result}{gain}", GREEN)
528
530
  if pr["meta"].get("warm_start"):
529
531
  ws = pr["meta"]["warm_start"]
@@ -101,6 +101,8 @@ headers = { "X-Title" = "tuieval" }
101
101
  # request_timeout_s = 600 # a tuning request that takes longer fails that candidate
102
102
  # warm_start = true # start from a tuned model with the same architecture and shapes (a fine-tune)
103
103
  # warm_retest = ["spec", "ubatch"] # the knobs a warm start re-tries; the rest are inherited
104
+ # answer_tokens = 1024 # answer length options are scored for (projected objective)
105
+ # swap_limit_mb = 500 # reject options that swap more than this beyond the defaults (default: report only)
104
106
 
105
107
  # Optional: llama-bench for the tuner's fast first stage (found on PATH as llama-bench otherwise).
106
108
  # [bench]
tuieval/tune.py CHANGED
@@ -18,18 +18,24 @@ retrained or dropped its MTP layers, and its quant mix shifts the best micro-bat
18
18
  it tunes in full instead. `tuieval tune --cold` (or [tune] warm_start = false) always tunes in full.
19
19
 
20
20
  One objective per server (models.toml `tune_objective`):
21
- total (default) total seconds for the workload: short prompts, medium ones and one long
22
- (~8k-token) prompt, fixed generation length
23
- decode generated tokens per second, measured on a second pass over a few prompts after a
24
- warm-up pass (for servers whose decode speed grows as their caches warm)
21
+ projected (default) seconds the workload would take with full-length answers: each prompt's
22
+ measured reading time plus [tune] answer_tokens (default 1024) at its measured
23
+ generation speed. The workload generates 128 tokens per prompt, but eval answers run
24
+ to thousands, so generation speed counts as much as it does in real runs.
25
+ total wall-clock seconds for the workload as run (128-token answers)
26
+ decode generated tokens per second, measured on a second pass over a few prompts after a
27
+ warm-up pass (for servers whose decode speed grows as their caches warm)
25
28
  An output guard compares greedy answers with the default flags; an option that changes them beyond
26
29
  noise is rejected, because speed flags must not change answers. Servers whose answers depend on
27
30
  the machine's memory settings (`outputs_depend_on_machine`) skip the guard; their tuned flags
28
31
  become part of the results fingerprint instead, so retuning marks that machine's results
29
- outdated. A candidate under which macOS swaps more than with the defaults is rejected (it costs
30
- memory). Swapping the defaults already cause (the model, its context, other apps) is a warning, and
31
- swapping while the model loads is only noted. Servers without tune knobs are only
32
- measured.
32
+ outdated.
33
+
34
+ Memory: a candidate is rejected if macOS memory pressure turns critical while it works, or, with
35
+ [tune] swap_limit_mb set, if it swaps more than that beyond the defaults. Otherwise memory is a
36
+ reported cost, not a reason to reject: a flag that pushes other apps' idle memory to swap can still
37
+ win, and the result warns about it. Swapping the defaults already cause is a warning too, and
38
+ swapping while the model loads is only noted. Servers without tune knobs are only measured.
33
39
 
34
40
  The knobs and their options come from models.toml [servers.<name>.tune]; placeholders {p},
35
41
  {p_minus_2} and {all} are this machine's core counts.
@@ -50,7 +56,10 @@ from . import profiles
50
56
 
51
57
  GEN_TOKENS = 128 # generated per workload prompt
52
58
  DECODE_TOKENS = 256 # generated per prompt for the decode objective
53
- SWAP_LIMIT = 256 * 2**20 # a candidate that swaps this much more than the defaults is rejected
59
+ SWAP_NOTABLE = 256 * 2**20 # swap worth a warning (bytes); rejecting for swap needs [tune] swap_limit_mb
60
+ ANSWER_TOKENS = 1024 # answer length the projected objective scores ([tune] answer_tokens)
61
+ PRESSURE_CRITICAL = 4 # macOS memory pressure level that rejects a candidate
62
+ PRESSURE_EVERY_S = 2
54
63
  GPU_MARGIN_GB = 0.75 # a candidate whose GPU allocation comes this close to the residency limit
55
64
  # is rejected even before it stalls: servers grow as requests arrive
56
65
  REQUEST_LIMIT_S = 600 # a tuning request taking longer fails the candidate ([tune] request_timeout_s)
@@ -72,16 +81,21 @@ class Measure:
72
81
  error: str = ""
73
82
  facts: dict = dataclasses.field(default_factory=dict) # from the server log, e.g. expert capacity
74
83
  swapped_mb: float | None = None # macOS swap-outs while the loaded server worked
84
+ projected_s: float | None = None # the workload with full-length answers (projected objective)
85
+ pressure: int | None = None # highest macOS memory pressure level while it worked
75
86
 
76
87
  def score(self, objective):
77
88
  """Lower is better."""
78
89
  if objective == "decode":
79
90
  return 1 / self.tg_tps if self.tg_tps else float("inf")
91
+ if objective == "projected" and self.projected_s is not None:
92
+ return self.projected_s
80
93
  return self.total_s
81
94
 
82
95
  def summary(self):
83
96
  facts = "".join(f" · {k} {v}" for k, v in self.facts.items())
84
- return (f"{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
97
+ projected = f"{self.projected_s:.0f}s projected · " if self.projected_s is not None else ""
98
+ return (f"{projected}{self.total_s:.1f}s total · gen {self.tg_tps} tok/s · prompt {self.pp_tps} tok/s"
85
99
  f" · ttft {self.ttft_s}s{facts}")
86
100
 
87
101
 
@@ -194,20 +208,21 @@ def _request(eng, base_url, body, name, emit, limit):
194
208
 
195
209
 
196
210
  def measure(eng, m, base_url, work, gen_tokens=GEN_TOKENS, passes=1, emit=lambda *a, **k: None,
197
- limit=REQUEST_LIMIT_S):
211
+ limit=REQUEST_LIMIT_S, answer_tokens=ANSWER_TOKENS):
198
212
  """Time the workload (after a warm-up request). With passes > 1 the earlier passes warm the
199
- server's caches and only the last pass is reported."""
213
+ server's caches and only the last pass is reported. answer_tokens: the answer length the
214
+ projected time is for."""
200
215
  out = Measure()
201
216
  for p in range(passes):
202
217
  if passes > 1:
203
218
  emit("tune_progress", message=f" pass {p + 1} of {passes}" + (" (warm-up)" if p < passes - 1 else " (timed)"))
204
- out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit)
219
+ out = _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens)
205
220
  if out.error:
206
221
  break
207
222
  return out
208
223
 
209
224
 
210
- def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
225
+ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit, answer_tokens=ANSWER_TOKENS):
211
226
  sampling = engine_mod.effective_sampling(eng.cfg, m)
212
227
  request = eng.cfg["servers"][m["server"]].get("request", {})
213
228
  timeout = eng.cfg["defaults"]["request_timeout_ms"] / 1000
@@ -218,7 +233,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
218
233
  if warm["error"]:
219
234
  out.error = warm["error"]
220
235
  return out
221
- ttfts, p_tok, p_s, g_tok, g_s = [], 0, 0.0, 0, 0.0
236
+ ttfts, p_tok, p_s, g_tok, g_s, projected = [], 0, 0.0, 0, 0.0, 0.0
222
237
  for name, msgs in work:
223
238
  eng._check()
224
239
  emit("tune_progress", message=f" {name} ({gen_tokens} tokens)")
@@ -230,6 +245,13 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
230
245
  emit("tune_progress", message=f" done: {met['completion_tokens']} tokens in {res['total_s']:.1f}s"
231
246
  + (f", {met['gen_tps']} tok/s" if met["gen_tps"] else ""))
232
247
  out.total_s += res["total_s"]
248
+ # The same request with a full-length answer: its reading time plus answer_tokens at the
249
+ # generation speed measured here (as run if the server doesn't report that speed).
250
+ if met["completion_tokens"] and met["gen_tps"]:
251
+ reading = max(res["total_s"] - met["completion_tokens"] / met["gen_tps"], 0.0)
252
+ projected += reading + answer_tokens / met["gen_tps"]
253
+ else:
254
+ projected += res["total_s"]
233
255
  out.texts.append((res["reasoning"] or "") + (res["answer"] or ""))
234
256
  if name != "long" and res.get("ttft_s") is not None:
235
257
  ttfts.append(res["ttft_s"])
@@ -238,6 +260,7 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
238
260
  if met["completion_tokens"] and met["gen_tps"]:
239
261
  g_tok, g_s = g_tok + met["completion_tokens"], g_s + met["completion_tokens"] / met["gen_tps"]
240
262
  out.total_s = round(out.total_s, 2)
263
+ out.projected_s = round(projected, 1)
241
264
  out.ttft_s = round(sum(ttfts) / len(ttfts), 3) if ttfts else None
242
265
  out.pp_tps = round(p_tok / p_s, 1) if p_s else None
243
266
  out.tg_tps = round(g_tok / g_s, 2) if g_s else None
@@ -246,12 +269,37 @@ def _measure_once(eng, m, base_url, work, gen_tokens, emit, limit):
246
269
 
247
270
  def gain_pct(objective, base, best):
248
271
  """How much better the chosen settings are than the defaults: % more decode tok/s for the
249
- decode objective, % less total time otherwise."""
272
+ decode objective, % less (projected or total) time otherwise."""
250
273
  if base is None:
251
274
  return None
252
275
  if objective == "decode":
253
276
  return round(100 * (best.tg_tps / base.tg_tps - 1)) if base.tg_tps and best.tg_tps else None
254
- return round(100 * (1 - best.total_s / base.total_s)) if base.total_s else None
277
+ b, n = base.score(objective), best.score(objective)
278
+ return round(100 * (1 - n / b)) if b else None
279
+
280
+
281
+ class PressureWatch:
282
+ """The highest macOS memory pressure level seen while a block runs (None where unknown)."""
283
+
284
+ def __enter__(self):
285
+ self.level, self._stop = machines.memory_pressure_level(), threading.Event()
286
+ self._thread = threading.Thread(target=self._run, daemon=True)
287
+ self._thread.start()
288
+ return self
289
+
290
+ def _sample(self):
291
+ lvl = machines.memory_pressure_level()
292
+ if lvl is not None:
293
+ self.level = max(self.level or 0, lvl)
294
+
295
+ def _run(self):
296
+ while not self._stop.wait(PRESSURE_EVERY_S):
297
+ self._sample()
298
+
299
+ def __exit__(self, *exc):
300
+ self._stop.set()
301
+ self._thread.join(PRESSURE_EVERY_S + 1)
302
+ self._sample()
255
303
 
256
304
 
257
305
  def gain_text(measured):
@@ -453,12 +501,14 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
453
501
  sv = eng.serving(m)
454
502
  if not sv.fits:
455
503
  raise engine_mod.ModelFailed(sv.fit_note)
456
- objective = server.get("tune_objective", "total")
504
+ objective = server.get("tune_objective", "projected")
457
505
  if objective == "decode":
458
506
  work, gen_tokens, passes = decode_workload(eng), DECODE_TOKENS, 2
459
507
  else:
460
508
  work, gen_tokens, passes = workload(eng, sv.ctx), GEN_TOKENS, 1
461
509
  settings = eng.cfg.get("tune", {})
510
+ answer_tokens = settings.get("answer_tokens", ANSWER_TOKENS)
511
+ swap_limit = settings.get("swap_limit_mb") # MB beyond the defaults' swap that rejects; None: report only
462
512
  threshold = settings.get("guard_similarity", 0.6)
463
513
  guard = not server.get("outputs_depend_on_machine")
464
514
  knob_opts = knobs(eng, m, sv) if server.get("cmd") else {}
@@ -484,18 +534,20 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
484
534
  # Loading may push other apps to swap once (e.g. a model locked in RAM with mlock);
485
535
  # that's recorded, not held against the settings. Swapping while serving is.
486
536
  loaded = machines.swapped_out_bytes()
487
- r = measure(eng, m, url, work, gen_tokens, passes, emit,
488
- settings.get("request_timeout_s", REQUEST_LIMIT_S))
537
+ with PressureWatch() as pressure:
538
+ r = measure(eng, m, url, work, gen_tokens, passes, emit,
539
+ settings.get("request_timeout_s", REQUEST_LIMIT_S), answer_tokens=answer_tokens)
489
540
  end = machines.swapped_out_bytes()
541
+ r.pressure = pressure.level
490
542
  if None not in (swap0, loaded, end):
491
543
  swapped = (end - loaded) / 2**20
492
544
  at_load = (loaded - swap0) / 2**20
493
- if at_load > SWAP_LIMIT / 2**20:
545
+ if at_load > SWAP_NOTABLE / 2**20:
494
546
  emit("tune_step", message=f" note: loading pushed {at_load:.0f} MB of other apps to swap "
495
547
  "(close apps for more headroom)")
496
548
  r.load_s = round(info["load_s"], 1) if info["load_s"] else None
497
549
  r.facts = dict(info["facts"])
498
- if swapped is not None and at_load > SWAP_LIMIT / 2**20:
550
+ if swapped is not None and at_load > SWAP_NOTABLE / 2**20:
499
551
  r.facts["load_swapped_mb"] = round(at_load)
500
552
  # Settled allocation after the timed pass (brief peaks while processing a prompt are
501
553
  # harmless; sustained allocation over the limit is what makes the driver churn).
@@ -516,10 +568,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
516
568
  r = Measure(error=str(e).splitlines()[0])
517
569
  r.swapped_mb = swapped
518
570
  # Swapping every candidate shares (model, context, prompt cache, other apps) says nothing about
519
- # a flag; only swapping beyond the defaults' does.
571
+ # a flag; only swapping beyond the defaults' does, and it's a cost to report unless it harms:
572
+ # critical memory pressure, or more than the user's swap_limit_mb.
520
573
  extra = None if swapped is None or swap_ref[0] is None else swapped - swap_ref[0]
521
- if not r.error and extra is not None and extra > SWAP_LIMIT / 2**20:
522
- r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (uses too much memory)"
574
+ if extra is not None and extra > SWAP_NOTABLE / 2**20:
575
+ r.facts["extra_swap_mb"] = round(extra)
576
+ if not r.error and r.pressure is not None and r.pressure >= PRESSURE_CRITICAL:
577
+ r.error = "macOS memory pressure turned critical (memory too tight)"
578
+ elif not r.error and swap_limit is not None and extra is not None and extra > swap_limit:
579
+ r.error = f"macOS swapped {extra:.0f} MB more than with the defaults (over swap_limit_mb = {swap_limit})"
523
580
  cache[k] = r
524
581
  emit("tune_result", message=f" failed: {r.error}" if r.error else " " + r.summary(), result=r, args=args)
525
582
  return r
@@ -527,7 +584,7 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
527
584
  def set_baseline(r):
528
585
  """The defaults' swapping is the machine's baseline: warned about, never held against a flag."""
529
586
  swap_ref[0] = r.swapped_mb or 0
530
- if swap_ref[0] > SWAP_LIMIT / 2**20:
587
+ if swap_ref[0] > SWAP_NOTABLE / 2**20:
531
588
  warnings.append(f"serving this model pushed {swap_ref[0]:.0f} MB of other apps' memory to swap on "
532
589
  "this machine; close other apps or lower this model's context for more headroom")
533
590
  emit("tune_step", message=" warning: " + warnings[-1])
@@ -615,6 +672,11 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
615
672
  if warm_info:
616
673
  notes.insert(0, f"warm start from {warm_info['from']}: inherited "
617
674
  f"{', '.join(warm_info['inherited']) or 'nothing'}")
675
+ if best_r.facts.get("extra_swap_mb") and best_r is not base:
676
+ warnings.append(f"the chosen flags use more memory than the defaults: macOS swapped "
677
+ f"~{best_r.facts['extra_swap_mb']} MB more of other apps' memory; close apps during runs, "
678
+ "or set [tune] swap_limit_mb to rule such flags out")
679
+ emit("tune_step", message=" warning: " + warnings[-1])
618
680
  size = sv.identity.get("model_bytes")
619
681
  profile = {
620
682
  "args": resolve(knob_opts, best),
@@ -623,12 +685,15 @@ def tune(eng, label, emit=lambda *a, **k: None, max_starts=16, min_gain=0.03, us
623
685
  "tg_tps": best_r.tg_tps, "load_s": best_r.load_s,
624
686
  "default_total_s": base.total_s if knob_opts else None,
625
687
  "default_tg_tps": base.tg_tps if knob_opts else None, "objective": objective,
688
+ "projected_s": best_r.projected_s,
689
+ "default_projected_s": base.projected_s if knob_opts else None,
626
690
  **{f"server_{k}": v for k, v in best_r.facts.items()},
627
691
  "gain_pct": gain_pct(objective, base, best_r)},
628
692
  "meta": {"method": method, "date": profiles.now(), "machine": sv.machine.summary,
629
693
  "server_version": eng.server_version(m), "model_file": os.path.basename(m["model"]),
630
694
  "model_bytes": size, "ctx": sv.ctx, "server_starts": starts[0],
631
- "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else ""),
695
+ "workload": f"{len(work)} prompts x {gen_tokens} tokens" + (f", {passes} passes" if passes > 1 else "")
696
+ + (f", scored for {answer_tokens}-token answers" if objective == "projected" else ""),
632
697
  "llama_bench": bool(bench_info), "answer_guard": guard,
633
698
  **({"warm_start": warm_info} if knob_opts and warm_info else {}),
634
699
  "rejected": rejected, "notes": notes[:10], "warnings": warnings},
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuieval
3
- Version: 0.2.0.dev6
3
+ Version: 0.2.0.dev9
4
4
  Summary: Evaluate local and frontier LLMs on your own questions: accuracy, speed, tokens and PASS/FAIL verdicts, in the terminal.
5
5
  Project-URL: Homepage, https://github.com/ashe-wb/tuieval
6
6
  Project-URL: Issues, https://github.com/ashe-wb/tuieval/issues
@@ -26,7 +26,9 @@ Description-Content-Type: text/markdown
26
26
 
27
27
  <p align="center"><img src="https://raw.githubusercontent.com/ashe-wb/tuieval/main/docs/images/brand/readme-header.png" alt="tuieval" width="100%"></p>
28
28
 
29
- **Find out which model you can trust with your work, on your own questions, in the terminal.** Run the same eval packs against local models (llama.cpp GGUFs, LM Studio, Ollama, vLLM; tiny, dense or MoE) and against any frontier model on OpenRouter, side by side. tuieval measures **accuracy, speed and token use** together, and gives a **PASS / FAIL / INCONCLUSIVE** verdict per use case. Grading is automatic; there's no LLM judge.
29
+ **Find out which model you can trust with your work, on your own questions, in the terminal.**
30
+
31
+ Run the same eval packs against local models (llama.cpp GGUFs, LM Studio, Ollama, vLLM; tiny, dense or MoE) and against any frontier model on OpenRouter, side by side. tuieval measures **accuracy, speed and token use** together, and gives a **PASS / FAIL / INCONCLUSIVE** verdict per use case. Grading is automatic; there's no LLM judge.
30
32
 
31
33
  Public benchmarks leak into training data and rarely look like your work, so tuieval ships with **no built-in benchmark**. Instead you build *eval packs* from what you actually do: questions with checkable answers, in your domain, with your rules. Then you get an answer to the questions that matter:
32
34
 
@@ -1,20 +1,20 @@
1
1
  tuieval/__init__.py,sha256=myVXVX7htbwezW6k3x_FuKgn1RSZ2WcTaDEMByGn6Wg,312
2
2
  tuieval/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
3
- tuieval/_version.py,sha256=9gdVeCkadGCQ8txcnOoyULu4u3PP7cNd2WZB7FPQHh8,533
3
+ tuieval/_version.py,sha256=IHVwmalkoW20mASroKhIhmhNf5irHisjdIrjTxJWRpc,533
4
4
  tuieval/cli.py,sha256=ZaLoWrxyCABu4SJzx5U2gVUUmwcwRgpZVqdqnxyFMOk,4167
5
5
  tuieval/client.py,sha256=Vui7ucVv_GiKObGoVS_EWSZFUKPou9PWmutqMTrahXk,9313
6
6
  tuieval/compare.py,sha256=b-rVcMwq0Tos3oNKcpUP1HXULKhAQNAS4rlXYSY7t6k,25460
7
7
  tuieval/engine.py,sha256=SizirCzV1klc_a2nS14FIjS5jLiHCGbVT23eES4StEc,94817
8
8
  tuieval/export.py,sha256=dbHQgL_ubWtIPM1sLzAew_gETnrDJyd0zgqVA5vkpXM,9657
9
- tuieval/machines.py,sha256=CeestY-_6h8ZKhJbBA_xyf0prqBvjvdgs2XJtCJYnNg,12974
9
+ tuieval/machines.py,sha256=ImEihNj8QlBRdCHLED4QIdjJUvjVtZgo6DiVU5h6Zzs,13415
10
10
  tuieval/packs.py,sha256=Go-kTSPvYZriXMo4DvmqNNL-iD0J92MKLiNCnP2chJ8,10305
11
11
  tuieval/profiles.py,sha256=soeLxjdB-l-OxSY8jZ4yjwfdSnsBV-QICcwKxRJS7nQ,3663
12
12
  tuieval/remove.py,sha256=Vwh43nm9lVb9U9aN-h6CMYbkTwdR3UWfZObwweF72n0,5157
13
- tuieval/run_evals.py,sha256=0dtPqhXxFqRLj_11I7hs2IlNkrxxilm6GOlEPfef3Fw,37625
13
+ tuieval/run_evals.py,sha256=hi-bKDjOHQjmAvWRV9-lxJGcUSvHnP36gUGkHeSFLOc,37773
14
14
  tuieval/scaffold.py,sha256=-AE5vKXeTtMZJ8OMk4UiIstoUWozTKjThKxZ87j0f9E,4720
15
15
  tuieval/selftest.py,sha256=xKfR73wMTYcDRSCBYQTNyKeR9bYU5HIYUSg7oZvNCyI,9330
16
16
  tuieval/tui.py,sha256=52hK8VSOVA0iHW8AyVFEyEXB32rE3U_hezJEQ5wc0TI,132159
17
- tuieval/tune.py,sha256=3vUll2EQkPIqRuZpAWV2-r-sk7FMPUkGeWmo63eTmHA,35031
17
+ tuieval/tune.py,sha256=jXwYuSjfdqbLCE3WonP2xaUHwk_FnhI67twB1BaHggI,38992
18
18
  tuieval/verdict.py,sha256=oLKpRIruk7W8eF1YdmITFx9BgBK3kPKYViKL20Nzyu0,16306
19
19
  tuieval/watch_proxy.py,sha256=z916NRhEur04a9RnLm6ZpIEY1SnnUGPiHYbmlD9ktQ0,11035
20
20
  tuieval/workspace.py,sha256=V_HhkaN1odr0OpjNsPYradRWLpzDldM7SwpvxohBL4U,895
@@ -25,7 +25,7 @@ tuieval/graders/code.py,sha256=mNoPw-C_lNaXTTU57TJjhcVPTvdR400PxoE9Ow-tgHg,3703
25
25
  tuieval/graders/rag.py,sha256=j_gEjmaXmY_hgKu7-7pVQpQ8Cwb49gadnwFurih-1Hw,1447
26
26
  tuieval/graders/reply.py,sha256=BdkL7EWSSMoyjwCOFa97TTQMR1QyrpMH-ntcgjzDPEg,2199
27
27
  tuieval/graders/tool_call.py,sha256=_QXURtOk4ZKBnxXVMzFL1c9jWLWksRlLZ4VqOCL0t90,3578
28
- tuieval/templates/models.toml,sha256=bZRpQgj2LFeEQhnqkf0YUhrGybrugCilyWTg19_W4g4,7879
28
+ tuieval/templates/models.toml,sha256=hwXF1aFGNPyg3rQ-V29-BwVXCnQM2KoJG1RBB-l3_O8,8087
29
29
  tuieval/templates/packs/answer/pack.toml,sha256=8bjMBxG1Zz3yxEnYCqRTNBJ-JGYlquqiK19p8URWwTc,1167
30
30
  tuieval/templates/packs/answer/system.txt,sha256=m4V6X1i49Bas2HowZMWgfdDSPDHTObb_WxHXt0xw1Yo,248
31
31
  tuieval/templates/packs/answer/tests.yaml,sha256=WiYPOknPcO9E6f1wXhLcfA0W5v-oK8MgZV956uWtcUQ,2660
@@ -42,8 +42,8 @@ tuieval/templates/packs/tool_call/pack.toml,sha256=n4bx0lGJHSmW41uVqhfIXJ2BWZpmT
42
42
  tuieval/templates/packs/tool_call/system.txt,sha256=mKxu-lB-0YjLHXdgeFJa8x2C180OvcNz5cHv2N5Bg3U,254
43
43
  tuieval/templates/packs/tool_call/tests.yaml,sha256=Nv7c4syrzJ9cU8EBKSK_1ZSFx9V02aCjW4-2TBCPT7s,1859
44
44
  tuieval/templates/packs/tool_call/tools.yaml,sha256=HgxCkh0On2VwbHSsjAGM8tNCK7ZtKLGrs7lMCn3Fprs,1048
45
- tuieval-0.2.0.dev6.dist-info/METADATA,sha256=3jzFS5nE1qalIrAnoAi5TBvAEZ9cyyVQoRGxbS3kvAs,12079
46
- tuieval-0.2.0.dev6.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
- tuieval-0.2.0.dev6.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
- tuieval-0.2.0.dev6.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
- tuieval-0.2.0.dev6.dist-info/RECORD,,
45
+ tuieval-0.2.0.dev9.dist-info/METADATA,sha256=BuLF9eKFTrAyKzsz2etiwhH3qBrr1Gv_Czy0VGrHx5A,12080
46
+ tuieval-0.2.0.dev9.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
47
+ tuieval-0.2.0.dev9.dist-info/entry_points.txt,sha256=7eIhi1GpFxBP2OLJV02Nn8LwsNuO512eKOGt1IBbj04,45
48
+ tuieval-0.2.0.dev9.dist-info/licenses/LICENSE,sha256=7wfRgGtBGH1aXhZkIn47oUznqKC18kEPMvUVgol7Ffc,1077
49
+ tuieval-0.2.0.dev9.dist-info/RECORD,,