tuieval 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. tuieval/__init__.py +2 -0
  2. tuieval/__main__.py +5 -0
  3. tuieval/cli.py +78 -0
  4. tuieval/client.py +196 -0
  5. tuieval/compare.py +511 -0
  6. tuieval/engine.py +1827 -0
  7. tuieval/graders/__init__.py +138 -0
  8. tuieval/graders/answer.py +52 -0
  9. tuieval/graders/code.py +104 -0
  10. tuieval/graders/rag.py +33 -0
  11. tuieval/graders/reply.py +44 -0
  12. tuieval/graders/tool_call.py +75 -0
  13. tuieval/machines.py +293 -0
  14. tuieval/packs.py +239 -0
  15. tuieval/profiles.py +98 -0
  16. tuieval/run_evals.py +588 -0
  17. tuieval/scaffold.py +106 -0
  18. tuieval/selftest.py +185 -0
  19. tuieval/templates/models.toml +137 -0
  20. tuieval/templates/packs/answer/pack.toml +17 -0
  21. tuieval/templates/packs/answer/system.txt +4 -0
  22. tuieval/templates/packs/answer/tests.yaml +67 -0
  23. tuieval/templates/packs/code/pack.toml +16 -0
  24. tuieval/templates/packs/code/system.txt +3 -0
  25. tuieval/templates/packs/code/tests.yaml +197 -0
  26. tuieval/templates/packs/rag/pack.toml +16 -0
  27. tuieval/templates/packs/rag/system.txt +9 -0
  28. tuieval/templates/packs/rag/tests.yaml +99 -0
  29. tuieval/templates/packs/reply/pack.toml +15 -0
  30. tuieval/templates/packs/reply/system.txt +2 -0
  31. tuieval/templates/packs/reply/tests.yaml +67 -0
  32. tuieval/templates/packs/tool_call/pack.toml +18 -0
  33. tuieval/templates/packs/tool_call/system.txt +2 -0
  34. tuieval/templates/packs/tool_call/tests.yaml +59 -0
  35. tuieval/templates/packs/tool_call/tools.yaml +32 -0
  36. tuieval/tui.py +2558 -0
  37. tuieval/tune.py +518 -0
  38. tuieval/verdict.py +346 -0
  39. tuieval/watch_proxy.py +260 -0
  40. tuieval/workspace.py +27 -0
  41. tuieval/yamlout.py +21 -0
  42. tuieval-0.1.0.dist-info/METADATA +156 -0
  43. tuieval-0.1.0.dist-info/RECORD +46 -0
  44. tuieval-0.1.0.dist-info/WHEEL +4 -0
  45. tuieval-0.1.0.dist-info/entry_points.txt +2 -0
  46. tuieval-0.1.0.dist-info/licenses/LICENSE +21 -0
tuieval/__init__.py ADDED
@@ -0,0 +1,2 @@
1
+ """tuieval: evaluate local models on your own eval packs, in the terminal."""
2
+ __version__ = "0.1.0"
tuieval/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
tuieval/cli.py ADDED
@@ -0,0 +1,78 @@
1
+ """tuieval: evaluate local models on your own eval packs, in the terminal.
2
+
3
+ Usage: tuieval [--workspace DIR] [command] [options]
4
+
5
+ tuieval open the TUI
6
+ tuieval init [DIR] create a workspace (models.toml, packs/, graders/)
7
+ tuieval new-pack NAME a new pack from a starter template (--grader answer|rag|reply|tool_call|code)
8
+ tuieval run [...] run evals from the command line (tuieval run --help)
9
+ tuieval add|scan|list manage models
10
+ tuieval compare [...] scorecards (--speed, --pairwise, --failures)
11
+ tuieval regrade re-score stored answers after changing a grader
12
+ tuieval verdict|report production readiness (PASS/FAIL per model and use case)
13
+ tuieval history every verdict change over time
14
+ tuieval selftest|items check the tests themselves
15
+ tuieval capture LOG --pack turn a real failure into a new test
16
+ tuieval machines|tune fit per machine; find the fastest server flags here
17
+ tuieval watch a proxy that shows the reasoning of any app using your server
18
+ tuieval --version
19
+
20
+ The workspace is --workspace DIR, else $TUIEVAL_HOME, else the current folder.
21
+ """
22
+ import os
23
+ import sys
24
+
25
+ from . import __version__
26
+ from . import workspace
27
+
28
+ NEEDS_NO_WORKSPACE = {"init", "watch", "help", "-h", "--help", "--version", "-V"}
29
+
30
+
31
+ def main(argv=None):
32
+ argv = list(sys.argv[1:] if argv is None else argv)
33
+ if argv and argv[0] in ("--workspace", "-w"):
34
+ if len(argv) < 2:
35
+ sys.exit("--workspace needs a folder")
36
+ os.environ[workspace.ENV] = os.path.abspath(os.path.expanduser(argv[1]))
37
+ argv = argv[2:]
38
+ elif argv and argv[0].startswith("--workspace="):
39
+ os.environ[workspace.ENV] = os.path.abspath(os.path.expanduser(argv[0].split("=", 1)[1]))
40
+ argv = argv[1:]
41
+ cmd, rest = (argv[0], argv[1:]) if argv and not argv[0].startswith("--") else ("", argv)
42
+ if cmd in ("help", "-h", "--help") or (not cmd and rest[:1] in (["-h"], ["--help"])):
43
+ print(__doc__.strip())
44
+ return 0
45
+ if "--version" in argv[:1] or "-V" in argv[:1]:
46
+ print(f"tuieval {__version__}")
47
+ return 0
48
+ if cmd not in NEEDS_NO_WORKSPACE and cmd != "new-pack" and not workspace.is_workspace():
49
+ sys.exit(f"{workspace.root()} isn't a tuieval workspace (no models.toml).\n"
50
+ "Create one here with `tuieval init`, or point to one with --workspace DIR or TUIEVAL_HOME.")
51
+ if cmd == "":
52
+ from . import tui
53
+ return tui.main(rest)
54
+ if cmd == "init":
55
+ from . import scaffold
56
+ return scaffold.cmd_init(rest)
57
+ if cmd == "new-pack":
58
+ from . import scaffold
59
+ if not workspace.is_workspace():
60
+ sys.exit(f"{workspace.root()} isn't a tuieval workspace; run `tuieval init` first")
61
+ return scaffold.cmd_new_pack(rest)
62
+ if cmd == "run":
63
+ from . import run_evals
64
+ return run_evals.main(rest)
65
+ if cmd == "compare":
66
+ from . import compare
67
+ return compare.main(rest)
68
+ if cmd == "watch":
69
+ from . import watch_proxy
70
+ return watch_proxy.main(rest)
71
+ from . import run_evals
72
+ if cmd in run_evals.COMMANDS:
73
+ return run_evals.COMMANDS[cmd](rest)
74
+ sys.exit(f"unknown command: {cmd} (try tuieval help)")
75
+
76
+
77
+ if __name__ == "__main__":
78
+ sys.exit(main())
tuieval/client.py ADDED
@@ -0,0 +1,196 @@
1
+ """Streaming client for OpenAI-compatible chat servers (llama.cpp, vLLM, LM Studio, Ollama, OpenRouter, …).
2
+
3
+ stream_chat() sends one request, reports reasoning/answer tokens as they arrive, and returns the
4
+ answer together with the numbers that matter for choosing a local model:
5
+
6
+ ttft_s time to the first generated token (reasoning or answer)
7
+ total_s wall time for the whole request
8
+ gen_tps generation speed, tokens/s (server-reported when available)
9
+ prompt_tps prompt processing speed, tokens/s (llama.cpp reports it)
10
+ prompt_tokens, completion_tokens, and — when the split can be known — reasoning_tokens and
11
+ answer_tokens (tokens_estimated=True means the split was estimated from text length)
12
+ """
13
+ import http.client
14
+ import json
15
+ import socket
16
+ import time
17
+ import urllib.parse
18
+
19
+ SERVER_ERROR_PREFIXES = ("server returned HTTP 5", "server returned HTTP 429", "server error", "connection error")
20
+
21
+
22
+ def is_server_error(error):
23
+ """True for an error that is the server's fault rather than the model's answer: HTTP 5xx or
24
+ 429, a dropped connection, a mid-stream error. A read timeout is not one: that's the model
25
+ taking too long, and it counts against it."""
26
+ return bool(error) and error.startswith(SERVER_ERROR_PREFIXES) and "timed out" not in error.lower() \
27
+ and "timeout" not in error.lower()
28
+
29
+
30
+ def server_error_row(rec):
31
+ """A stored answer that was really a server failure (never a judgement of the model)."""
32
+ return not rec.get("pass") and is_server_error(rec.get("reason"))
33
+
34
+
35
+ class Cancelled(Exception):
36
+ pass
37
+
38
+
39
+ class Stream:
40
+ """One in-flight request. close() from another thread aborts it."""
41
+
42
+ def __init__(self):
43
+ self.conn = None
44
+ self.sock = None # kept separately: http.client drops conn.sock once a response streams
45
+ self.cancelled = False
46
+
47
+ def close(self):
48
+ self.cancelled = True
49
+ sock = self.sock or (self.conn.sock if self.conn is not None else None)
50
+ if sock is not None:
51
+ try:
52
+ # shutdown() wakes a thread blocked reading this socket; close() alone doesn't on macOS
53
+ sock.shutdown(socket.SHUT_RDWR)
54
+ except OSError:
55
+ pass
56
+ try:
57
+ sock.close()
58
+ except OSError:
59
+ pass
60
+
61
+
62
+ def _conn(base_url, timeout):
63
+ u = urllib.parse.urlparse(base_url)
64
+ cls = http.client.HTTPSConnection if u.scheme == "https" else http.client.HTTPConnection
65
+ return cls(u.hostname, u.port or (443 if u.scheme == "https" else 80), timeout=timeout), u.path.rstrip("/")
66
+
67
+
68
+ def post_json(base_url, path, payload, timeout=30):
69
+ conn, prefix = _conn(base_url, timeout)
70
+ try:
71
+ conn.request("POST", prefix + path, body=json.dumps(payload), headers={"Content-Type": "application/json"})
72
+ r = conn.getresponse()
73
+ data = r.read()
74
+ if r.status != 200:
75
+ raise OSError(f"HTTP {r.status}")
76
+ return json.loads(data)
77
+ finally:
78
+ conn.close()
79
+
80
+
81
+ def count_tokens(base_url, text):
82
+ """Exact token count via llama.cpp's /tokenize; None if the server doesn't support it."""
83
+ if not text:
84
+ return 0
85
+ try:
86
+ root = base_url[:-3] if base_url.rstrip("/").endswith("/v1") else base_url
87
+ return len(post_json(root, "/tokenize", {"content": text}, timeout=10).get("tokens", []))
88
+ except (OSError, ValueError, http.client.HTTPException):
89
+ return None
90
+
91
+
92
+ def stream_chat(base_url, body, on_delta=None, timeout=1800, stream=None, headers=None):
93
+ """POST {base_url}/v1/chat/completions with streaming. Returns a dict (see module docstring).
94
+
95
+ on_delta(kind, text) is called for "reasoning" and "answer" text as it arrives.
96
+ """
97
+ stream = stream or Stream()
98
+ body = dict(body, stream=True, stream_options={"include_usage": True})
99
+ conn, prefix = _conn(base_url, timeout)
100
+ stream.conn = conn
101
+ out = {"answer": "", "reasoning": "", "tool_calls": [], "finish": None, "usage": {}, "timings": {},
102
+ "ttft_s": None, "total_s": None, "error": None}
103
+ reasoning, answer, tools = [], [], {}
104
+ start = time.time()
105
+ try:
106
+ conn.request("POST", prefix + "/v1/chat/completions", body=json.dumps(body),
107
+ headers={"Content-Type": "application/json", "Authorization": "Bearer none", **(headers or {})})
108
+ stream.sock = conn.sock
109
+ if stream.cancelled: # cancelled while connecting
110
+ raise OSError("cancelled")
111
+ r = conn.getresponse()
112
+ if r.status != 200:
113
+ out["error"] = f"server returned HTTP {r.status}: {r.read()[:300].decode('utf-8', 'replace')}"
114
+ return out
115
+ for raw in r:
116
+ line = raw.decode("utf-8", "replace").strip()
117
+ if not line.startswith("data:"):
118
+ continue
119
+ payload = line[5:].strip()
120
+ if payload == "[DONE]":
121
+ break
122
+ chunk = json.loads(payload)
123
+ if chunk.get("error"): # OpenRouter reports mid-stream failures this way
124
+ err = chunk["error"]
125
+ out["error"] = f"server error: {err.get('message', err) if isinstance(err, dict) else err}"
126
+ break
127
+ out["provider"] = chunk.get("provider") or out.get("provider") # OpenRouter: who answered
128
+ out["usage"] = chunk.get("usage") or out["usage"]
129
+ out["timings"] = chunk.get("timings") or out["timings"]
130
+ for ch in chunk.get("choices", []):
131
+ delta = ch.get("delta") or {}
132
+ out["finish"] = ch.get("finish_reason") or out["finish"]
133
+ think = delta.get("reasoning_content") or delta.get("reasoning")
134
+ text = delta.get("content")
135
+ for tc in delta.get("tool_calls") or []:
136
+ slot = tools.setdefault(tc.get("index", len(tools)), {"name": "", "arguments": ""})
137
+ fn = tc.get("function") or {}
138
+ slot["name"] += fn.get("name") or ""
139
+ slot["arguments"] += fn.get("arguments") or ""
140
+ if (think or text or delta.get("tool_calls")) and out["ttft_s"] is None:
141
+ out["ttft_s"] = time.time() - start
142
+ if think:
143
+ reasoning.append(think)
144
+ on_delta and on_delta("reasoning", think)
145
+ if text:
146
+ answer.append(text)
147
+ on_delta and on_delta("answer", text)
148
+ except (OSError, http.client.HTTPException, ValueError) as e:
149
+ if stream.cancelled:
150
+ raise Cancelled() from None
151
+ out["error"] = f"connection error: {e!r}"
152
+ finally:
153
+ conn.close()
154
+ out["total_s"] = time.time() - start
155
+ out["answer"], out["reasoning"] = "".join(answer), "".join(reasoning)
156
+ for slot in (tools[k] for k in sorted(tools)):
157
+ try:
158
+ args = json.loads(slot["arguments"]) if slot["arguments"].strip() else {}
159
+ except ValueError:
160
+ args = {"_raw": slot["arguments"]}
161
+ out["tool_calls"].append({"name": slot["name"], "arguments": args})
162
+ return out
163
+
164
+
165
+ def metrics(res, base_url=None):
166
+ """Token and speed numbers for a finished request (tokenizes via the server when it can)."""
167
+ usage, timings = res.get("usage") or {}, res.get("timings") or {}
168
+ completion = usage.get("completion_tokens") or timings.get("predicted_n")
169
+ prompt = usage.get("prompt_tokens") or timings.get("prompt_n")
170
+ details = usage.get("completion_tokens_details") or {}
171
+ reasoning_tokens, answer_tokens, estimated = details.get("reasoning_tokens"), None, False
172
+ if completion and reasoning_tokens is None and res.get("reasoning"):
173
+ answer_tokens = count_tokens(base_url, res["answer"]) if base_url else None
174
+ if answer_tokens is not None:
175
+ reasoning_tokens = max(0, completion - answer_tokens)
176
+ else: # split by text length
177
+ total_chars = len(res["reasoning"]) + len(res["answer"]) or 1
178
+ reasoning_tokens = round(completion * len(res["reasoning"]) / total_chars)
179
+ estimated = True
180
+ if completion is not None:
181
+ reasoning_tokens = reasoning_tokens or 0
182
+ answer_tokens = completion - reasoning_tokens
183
+ # prompt tokens the server reused from earlier requests instead of computing (should be 0 in evals)
184
+ cached = (usage.get("prompt_tokens_details") or {}).get("cached_tokens") or timings.get("cache_n")
185
+ gen_tps = timings.get("predicted_per_second")
186
+ if not gen_tps and completion and res.get("ttft_s") is not None and res["total_s"] - res["ttft_s"] > 0.05:
187
+ gen_tps = completion / (res["total_s"] - res["ttft_s"])
188
+ hosted = {k: v for k, v in (("provider", res.get("provider")), ("cost", usage.get("cost"))) if v is not None}
189
+ return {
190
+ **hosted,
191
+ "prompt_tokens": prompt, "cached_tokens": cached, "completion_tokens": completion,
192
+ "reasoning_tokens": reasoning_tokens, "answer_tokens": answer_tokens, "tokens_estimated": estimated,
193
+ "ttft_s": res.get("ttft_s"), "total_s": res.get("total_s"),
194
+ "gen_tps": round(gen_tps, 2) if gen_tps else None,
195
+ "prompt_tps": round(timings["prompt_per_second"], 1) if timings.get("prompt_per_second") else None,
196
+ }