tuieval 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tuieval/__init__.py +2 -0
- tuieval/__main__.py +5 -0
- tuieval/cli.py +78 -0
- tuieval/client.py +196 -0
- tuieval/compare.py +511 -0
- tuieval/engine.py +1827 -0
- tuieval/graders/__init__.py +138 -0
- tuieval/graders/answer.py +52 -0
- tuieval/graders/code.py +104 -0
- tuieval/graders/rag.py +33 -0
- tuieval/graders/reply.py +44 -0
- tuieval/graders/tool_call.py +75 -0
- tuieval/machines.py +293 -0
- tuieval/packs.py +239 -0
- tuieval/profiles.py +98 -0
- tuieval/run_evals.py +588 -0
- tuieval/scaffold.py +106 -0
- tuieval/selftest.py +185 -0
- tuieval/templates/models.toml +137 -0
- tuieval/templates/packs/answer/pack.toml +17 -0
- tuieval/templates/packs/answer/system.txt +4 -0
- tuieval/templates/packs/answer/tests.yaml +67 -0
- tuieval/templates/packs/code/pack.toml +16 -0
- tuieval/templates/packs/code/system.txt +3 -0
- tuieval/templates/packs/code/tests.yaml +197 -0
- tuieval/templates/packs/rag/pack.toml +16 -0
- tuieval/templates/packs/rag/system.txt +9 -0
- tuieval/templates/packs/rag/tests.yaml +99 -0
- tuieval/templates/packs/reply/pack.toml +15 -0
- tuieval/templates/packs/reply/system.txt +2 -0
- tuieval/templates/packs/reply/tests.yaml +67 -0
- tuieval/templates/packs/tool_call/pack.toml +18 -0
- tuieval/templates/packs/tool_call/system.txt +2 -0
- tuieval/templates/packs/tool_call/tests.yaml +59 -0
- tuieval/templates/packs/tool_call/tools.yaml +32 -0
- tuieval/tui.py +2558 -0
- tuieval/tune.py +518 -0
- tuieval/verdict.py +346 -0
- tuieval/watch_proxy.py +260 -0
- tuieval/workspace.py +27 -0
- tuieval/yamlout.py +21 -0
- tuieval-0.1.0.dist-info/METADATA +156 -0
- tuieval-0.1.0.dist-info/RECORD +46 -0
- tuieval-0.1.0.dist-info/WHEEL +4 -0
- tuieval-0.1.0.dist-info/entry_points.txt +2 -0
- tuieval-0.1.0.dist-info/licenses/LICENSE +21 -0
tuieval/__init__.py
ADDED
tuieval/__main__.py
ADDED
tuieval/cli.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""tuieval: evaluate local models on your own eval packs, in the terminal.
|
|
2
|
+
|
|
3
|
+
Usage: tuieval [--workspace DIR] [command] [options]
|
|
4
|
+
|
|
5
|
+
tuieval open the TUI
|
|
6
|
+
tuieval init [DIR] create a workspace (models.toml, packs/, graders/)
|
|
7
|
+
tuieval new-pack NAME a new pack from a starter template (--grader answer|rag|reply|tool_call|code)
|
|
8
|
+
tuieval run [...] run evals from the command line (tuieval run --help)
|
|
9
|
+
tuieval add|scan|list manage models
|
|
10
|
+
tuieval compare [...] scorecards (--speed, --pairwise, --failures)
|
|
11
|
+
tuieval regrade re-score stored answers after changing a grader
|
|
12
|
+
tuieval verdict|report production readiness (PASS/FAIL per model and use case)
|
|
13
|
+
tuieval history every verdict change over time
|
|
14
|
+
tuieval selftest|items check the tests themselves
|
|
15
|
+
tuieval capture LOG --pack turn a real failure into a new test
|
|
16
|
+
tuieval machines|tune fit per machine; find the fastest server flags here
|
|
17
|
+
tuieval watch a proxy that shows the reasoning of any app using your server
|
|
18
|
+
tuieval --version
|
|
19
|
+
|
|
20
|
+
The workspace is --workspace DIR, else $TUIEVAL_HOME, else the current folder.
|
|
21
|
+
"""
|
|
22
|
+
import os
|
|
23
|
+
import sys
|
|
24
|
+
|
|
25
|
+
from . import __version__
|
|
26
|
+
from . import workspace
|
|
27
|
+
|
|
28
|
+
NEEDS_NO_WORKSPACE = {"init", "watch", "help", "-h", "--help", "--version", "-V"}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def main(argv=None):
|
|
32
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
33
|
+
if argv and argv[0] in ("--workspace", "-w"):
|
|
34
|
+
if len(argv) < 2:
|
|
35
|
+
sys.exit("--workspace needs a folder")
|
|
36
|
+
os.environ[workspace.ENV] = os.path.abspath(os.path.expanduser(argv[1]))
|
|
37
|
+
argv = argv[2:]
|
|
38
|
+
elif argv and argv[0].startswith("--workspace="):
|
|
39
|
+
os.environ[workspace.ENV] = os.path.abspath(os.path.expanduser(argv[0].split("=", 1)[1]))
|
|
40
|
+
argv = argv[1:]
|
|
41
|
+
cmd, rest = (argv[0], argv[1:]) if argv and not argv[0].startswith("--") else ("", argv)
|
|
42
|
+
if cmd in ("help", "-h", "--help") or (not cmd and rest[:1] in (["-h"], ["--help"])):
|
|
43
|
+
print(__doc__.strip())
|
|
44
|
+
return 0
|
|
45
|
+
if "--version" in argv[:1] or "-V" in argv[:1]:
|
|
46
|
+
print(f"tuieval {__version__}")
|
|
47
|
+
return 0
|
|
48
|
+
if cmd not in NEEDS_NO_WORKSPACE and cmd != "new-pack" and not workspace.is_workspace():
|
|
49
|
+
sys.exit(f"{workspace.root()} isn't a tuieval workspace (no models.toml).\n"
|
|
50
|
+
"Create one here with `tuieval init`, or point to one with --workspace DIR or TUIEVAL_HOME.")
|
|
51
|
+
if cmd == "":
|
|
52
|
+
from . import tui
|
|
53
|
+
return tui.main(rest)
|
|
54
|
+
if cmd == "init":
|
|
55
|
+
from . import scaffold
|
|
56
|
+
return scaffold.cmd_init(rest)
|
|
57
|
+
if cmd == "new-pack":
|
|
58
|
+
from . import scaffold
|
|
59
|
+
if not workspace.is_workspace():
|
|
60
|
+
sys.exit(f"{workspace.root()} isn't a tuieval workspace; run `tuieval init` first")
|
|
61
|
+
return scaffold.cmd_new_pack(rest)
|
|
62
|
+
if cmd == "run":
|
|
63
|
+
from . import run_evals
|
|
64
|
+
return run_evals.main(rest)
|
|
65
|
+
if cmd == "compare":
|
|
66
|
+
from . import compare
|
|
67
|
+
return compare.main(rest)
|
|
68
|
+
if cmd == "watch":
|
|
69
|
+
from . import watch_proxy
|
|
70
|
+
return watch_proxy.main(rest)
|
|
71
|
+
from . import run_evals
|
|
72
|
+
if cmd in run_evals.COMMANDS:
|
|
73
|
+
return run_evals.COMMANDS[cmd](rest)
|
|
74
|
+
sys.exit(f"unknown command: {cmd} (try tuieval help)")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
if __name__ == "__main__":
|
|
78
|
+
sys.exit(main())
|
tuieval/client.py
ADDED
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Streaming client for OpenAI-compatible chat servers (llama.cpp, vLLM, LM Studio, Ollama, OpenRouter, …).
|
|
2
|
+
|
|
3
|
+
stream_chat() sends one request, reports reasoning/answer tokens as they arrive, and returns the
|
|
4
|
+
answer together with the numbers that matter for choosing a local model:
|
|
5
|
+
|
|
6
|
+
ttft_s time to the first generated token (reasoning or answer)
|
|
7
|
+
total_s wall time for the whole request
|
|
8
|
+
gen_tps generation speed, tokens/s (server-reported when available)
|
|
9
|
+
prompt_tps prompt processing speed, tokens/s (llama.cpp reports it)
|
|
10
|
+
prompt_tokens, completion_tokens, and — when the split can be known — reasoning_tokens and
|
|
11
|
+
answer_tokens (tokens_estimated=True means the split was estimated from text length)
|
|
12
|
+
"""
|
|
13
|
+
import http.client
|
|
14
|
+
import json
|
|
15
|
+
import socket
|
|
16
|
+
import time
|
|
17
|
+
import urllib.parse
|
|
18
|
+
|
|
19
|
+
SERVER_ERROR_PREFIXES = ("server returned HTTP 5", "server returned HTTP 429", "server error", "connection error")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_server_error(error):
|
|
23
|
+
"""True for an error that is the server's fault rather than the model's answer: HTTP 5xx or
|
|
24
|
+
429, a dropped connection, a mid-stream error. A read timeout is not one: that's the model
|
|
25
|
+
taking too long, and it counts against it."""
|
|
26
|
+
return bool(error) and error.startswith(SERVER_ERROR_PREFIXES) and "timed out" not in error.lower() \
|
|
27
|
+
and "timeout" not in error.lower()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def server_error_row(rec):
|
|
31
|
+
"""A stored answer that was really a server failure (never a judgement of the model)."""
|
|
32
|
+
return not rec.get("pass") and is_server_error(rec.get("reason"))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Cancelled(Exception):
|
|
36
|
+
pass
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Stream:
|
|
40
|
+
"""One in-flight request. close() from another thread aborts it."""
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
self.conn = None
|
|
44
|
+
self.sock = None # kept separately: http.client drops conn.sock once a response streams
|
|
45
|
+
self.cancelled = False
|
|
46
|
+
|
|
47
|
+
def close(self):
|
|
48
|
+
self.cancelled = True
|
|
49
|
+
sock = self.sock or (self.conn.sock if self.conn is not None else None)
|
|
50
|
+
if sock is not None:
|
|
51
|
+
try:
|
|
52
|
+
# shutdown() wakes a thread blocked reading this socket; close() alone doesn't on macOS
|
|
53
|
+
sock.shutdown(socket.SHUT_RDWR)
|
|
54
|
+
except OSError:
|
|
55
|
+
pass
|
|
56
|
+
try:
|
|
57
|
+
sock.close()
|
|
58
|
+
except OSError:
|
|
59
|
+
pass
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _conn(base_url, timeout):
|
|
63
|
+
u = urllib.parse.urlparse(base_url)
|
|
64
|
+
cls = http.client.HTTPSConnection if u.scheme == "https" else http.client.HTTPConnection
|
|
65
|
+
return cls(u.hostname, u.port or (443 if u.scheme == "https" else 80), timeout=timeout), u.path.rstrip("/")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def post_json(base_url, path, payload, timeout=30):
|
|
69
|
+
conn, prefix = _conn(base_url, timeout)
|
|
70
|
+
try:
|
|
71
|
+
conn.request("POST", prefix + path, body=json.dumps(payload), headers={"Content-Type": "application/json"})
|
|
72
|
+
r = conn.getresponse()
|
|
73
|
+
data = r.read()
|
|
74
|
+
if r.status != 200:
|
|
75
|
+
raise OSError(f"HTTP {r.status}")
|
|
76
|
+
return json.loads(data)
|
|
77
|
+
finally:
|
|
78
|
+
conn.close()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def count_tokens(base_url, text):
|
|
82
|
+
"""Exact token count via llama.cpp's /tokenize; None if the server doesn't support it."""
|
|
83
|
+
if not text:
|
|
84
|
+
return 0
|
|
85
|
+
try:
|
|
86
|
+
root = base_url[:-3] if base_url.rstrip("/").endswith("/v1") else base_url
|
|
87
|
+
return len(post_json(root, "/tokenize", {"content": text}, timeout=10).get("tokens", []))
|
|
88
|
+
except (OSError, ValueError, http.client.HTTPException):
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def stream_chat(base_url, body, on_delta=None, timeout=1800, stream=None, headers=None):
|
|
93
|
+
"""POST {base_url}/v1/chat/completions with streaming. Returns a dict (see module docstring).
|
|
94
|
+
|
|
95
|
+
on_delta(kind, text) is called for "reasoning" and "answer" text as it arrives.
|
|
96
|
+
"""
|
|
97
|
+
stream = stream or Stream()
|
|
98
|
+
body = dict(body, stream=True, stream_options={"include_usage": True})
|
|
99
|
+
conn, prefix = _conn(base_url, timeout)
|
|
100
|
+
stream.conn = conn
|
|
101
|
+
out = {"answer": "", "reasoning": "", "tool_calls": [], "finish": None, "usage": {}, "timings": {},
|
|
102
|
+
"ttft_s": None, "total_s": None, "error": None}
|
|
103
|
+
reasoning, answer, tools = [], [], {}
|
|
104
|
+
start = time.time()
|
|
105
|
+
try:
|
|
106
|
+
conn.request("POST", prefix + "/v1/chat/completions", body=json.dumps(body),
|
|
107
|
+
headers={"Content-Type": "application/json", "Authorization": "Bearer none", **(headers or {})})
|
|
108
|
+
stream.sock = conn.sock
|
|
109
|
+
if stream.cancelled: # cancelled while connecting
|
|
110
|
+
raise OSError("cancelled")
|
|
111
|
+
r = conn.getresponse()
|
|
112
|
+
if r.status != 200:
|
|
113
|
+
out["error"] = f"server returned HTTP {r.status}: {r.read()[:300].decode('utf-8', 'replace')}"
|
|
114
|
+
return out
|
|
115
|
+
for raw in r:
|
|
116
|
+
line = raw.decode("utf-8", "replace").strip()
|
|
117
|
+
if not line.startswith("data:"):
|
|
118
|
+
continue
|
|
119
|
+
payload = line[5:].strip()
|
|
120
|
+
if payload == "[DONE]":
|
|
121
|
+
break
|
|
122
|
+
chunk = json.loads(payload)
|
|
123
|
+
if chunk.get("error"): # OpenRouter reports mid-stream failures this way
|
|
124
|
+
err = chunk["error"]
|
|
125
|
+
out["error"] = f"server error: {err.get('message', err) if isinstance(err, dict) else err}"
|
|
126
|
+
break
|
|
127
|
+
out["provider"] = chunk.get("provider") or out.get("provider") # OpenRouter: who answered
|
|
128
|
+
out["usage"] = chunk.get("usage") or out["usage"]
|
|
129
|
+
out["timings"] = chunk.get("timings") or out["timings"]
|
|
130
|
+
for ch in chunk.get("choices", []):
|
|
131
|
+
delta = ch.get("delta") or {}
|
|
132
|
+
out["finish"] = ch.get("finish_reason") or out["finish"]
|
|
133
|
+
think = delta.get("reasoning_content") or delta.get("reasoning")
|
|
134
|
+
text = delta.get("content")
|
|
135
|
+
for tc in delta.get("tool_calls") or []:
|
|
136
|
+
slot = tools.setdefault(tc.get("index", len(tools)), {"name": "", "arguments": ""})
|
|
137
|
+
fn = tc.get("function") or {}
|
|
138
|
+
slot["name"] += fn.get("name") or ""
|
|
139
|
+
slot["arguments"] += fn.get("arguments") or ""
|
|
140
|
+
if (think or text or delta.get("tool_calls")) and out["ttft_s"] is None:
|
|
141
|
+
out["ttft_s"] = time.time() - start
|
|
142
|
+
if think:
|
|
143
|
+
reasoning.append(think)
|
|
144
|
+
on_delta and on_delta("reasoning", think)
|
|
145
|
+
if text:
|
|
146
|
+
answer.append(text)
|
|
147
|
+
on_delta and on_delta("answer", text)
|
|
148
|
+
except (OSError, http.client.HTTPException, ValueError) as e:
|
|
149
|
+
if stream.cancelled:
|
|
150
|
+
raise Cancelled() from None
|
|
151
|
+
out["error"] = f"connection error: {e!r}"
|
|
152
|
+
finally:
|
|
153
|
+
conn.close()
|
|
154
|
+
out["total_s"] = time.time() - start
|
|
155
|
+
out["answer"], out["reasoning"] = "".join(answer), "".join(reasoning)
|
|
156
|
+
for slot in (tools[k] for k in sorted(tools)):
|
|
157
|
+
try:
|
|
158
|
+
args = json.loads(slot["arguments"]) if slot["arguments"].strip() else {}
|
|
159
|
+
except ValueError:
|
|
160
|
+
args = {"_raw": slot["arguments"]}
|
|
161
|
+
out["tool_calls"].append({"name": slot["name"], "arguments": args})
|
|
162
|
+
return out
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def metrics(res, base_url=None):
|
|
166
|
+
"""Token and speed numbers for a finished request (tokenizes via the server when it can)."""
|
|
167
|
+
usage, timings = res.get("usage") or {}, res.get("timings") or {}
|
|
168
|
+
completion = usage.get("completion_tokens") or timings.get("predicted_n")
|
|
169
|
+
prompt = usage.get("prompt_tokens") or timings.get("prompt_n")
|
|
170
|
+
details = usage.get("completion_tokens_details") or {}
|
|
171
|
+
reasoning_tokens, answer_tokens, estimated = details.get("reasoning_tokens"), None, False
|
|
172
|
+
if completion and reasoning_tokens is None and res.get("reasoning"):
|
|
173
|
+
answer_tokens = count_tokens(base_url, res["answer"]) if base_url else None
|
|
174
|
+
if answer_tokens is not None:
|
|
175
|
+
reasoning_tokens = max(0, completion - answer_tokens)
|
|
176
|
+
else: # split by text length
|
|
177
|
+
total_chars = len(res["reasoning"]) + len(res["answer"]) or 1
|
|
178
|
+
reasoning_tokens = round(completion * len(res["reasoning"]) / total_chars)
|
|
179
|
+
estimated = True
|
|
180
|
+
if completion is not None:
|
|
181
|
+
reasoning_tokens = reasoning_tokens or 0
|
|
182
|
+
answer_tokens = completion - reasoning_tokens
|
|
183
|
+
# prompt tokens the server reused from earlier requests instead of computing (should be 0 in evals)
|
|
184
|
+
cached = (usage.get("prompt_tokens_details") or {}).get("cached_tokens") or timings.get("cache_n")
|
|
185
|
+
gen_tps = timings.get("predicted_per_second")
|
|
186
|
+
if not gen_tps and completion and res.get("ttft_s") is not None and res["total_s"] - res["ttft_s"] > 0.05:
|
|
187
|
+
gen_tps = completion / (res["total_s"] - res["ttft_s"])
|
|
188
|
+
hosted = {k: v for k, v in (("provider", res.get("provider")), ("cost", usage.get("cost"))) if v is not None}
|
|
189
|
+
return {
|
|
190
|
+
**hosted,
|
|
191
|
+
"prompt_tokens": prompt, "cached_tokens": cached, "completion_tokens": completion,
|
|
192
|
+
"reasoning_tokens": reasoning_tokens, "answer_tokens": answer_tokens, "tokens_estimated": estimated,
|
|
193
|
+
"ttft_s": res.get("ttft_s"), "total_s": res.get("total_s"),
|
|
194
|
+
"gen_tps": round(gen_tps, 2) if gen_tps else None,
|
|
195
|
+
"prompt_tps": round(timings["prompt_per_second"], 1) if timings.get("prompt_per_second") else None,
|
|
196
|
+
}
|