hearthwork 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hearthwork/__init__.py +2 -0
- hearthwork/__main__.py +3 -0
- hearthwork/bench.py +237 -0
- hearthwork/check.py +189 -0
- hearthwork/cli.py +133 -0
- hearthwork/harnesses.py +244 -0
- hearthwork/menu.py +118 -0
- hearthwork/model.py +131 -0
- hearthwork/onboard.py +505 -0
- hearthwork/paths.py +33 -0
- hearthwork/server.py +243 -0
- hearthwork-0.2.1.dist-info/METADATA +233 -0
- hearthwork-0.2.1.dist-info/RECORD +16 -0
- hearthwork-0.2.1.dist-info/WHEEL +5 -0
- hearthwork-0.2.1.dist-info/entry_points.txt +3 -0
- hearthwork-0.2.1.dist-info/top_level.txt +1 -0
hearthwork/__init__.py
ADDED
hearthwork/__main__.py
ADDED
hearthwork/bench.py
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Benchmark the running model through a real coding agent, and keep a scoreboard.
|
|
3
|
+
|
|
4
|
+
hearthwork bench [--agent claude|codex] [--show]
|
|
5
|
+
(or menu option "Benchmark the running model")
|
|
6
|
+
|
|
7
|
+
Eight prompts run as one agent conversation in a fresh folder (bench/runs/...): chat, list files, read a file,
|
|
8
|
+
write and run a script, edit it, fix three planted bugs, write unit tests, summarize. Each step is graded by
|
|
9
|
+
checking the files and running the code, not by trusting the agent's reply. Results go to bench/results.jsonl;
|
|
10
|
+
--show prints the scoreboard without running anything.
|
|
11
|
+
"""
|
|
12
|
+
import argparse
|
|
13
|
+
import ast
|
|
14
|
+
import json
|
|
15
|
+
import re
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
import sys
|
|
19
|
+
import time
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from .harnesses import HARNESSES, installed, launch
|
|
24
|
+
from .onboard import CYAN, GREEN, RED, RESET, YELLOW, load_config
|
|
25
|
+
from .paths import BENCH
|
|
26
|
+
from .server import served_model
|
|
27
|
+
|
|
28
|
+
RESULTS = BENCH / "results.jsonl"
|
|
29
|
+
BOLD, DIM = "\033[1m", "\033[2m"
|
|
30
|
+
TIMEOUT = 20 * 60 # per prompt
|
|
31
|
+
|
|
32
|
+
DEMO = "The secret phrase is: copper owl builds a bridge.\n"
|
|
33
|
+
BUGGY = '''def average(numbers):
|
|
34
|
+
total = 0
|
|
35
|
+
for i in range(1, len(numbers)):
|
|
36
|
+
total += numbers[i]
|
|
37
|
+
return total / len(numbers)
|
|
38
|
+
|
|
39
|
+
def top_n(items, n):
|
|
40
|
+
return sorted(items)[:n]
|
|
41
|
+
|
|
42
|
+
print(average([10, 20, 30])) # should print 20.0
|
|
43
|
+
print(top_n([5, 1, 9, 3], 2)) # should print [9, 5]
|
|
44
|
+
print(average([])) # should print 0, not crash
|
|
45
|
+
'''
|
|
46
|
+
FIZZBUZZ = ["FizzBuzz" if i % 15 == 0 else "Fizz" if i % 3 == 0 else "Buzz" if i % 5 == 0 else str(i) for i in range(1, 31)]
|
|
47
|
+
PRIMES = {2, 3, 5, 7, 11, 13, 17, 19, 23, 29}
|
|
48
|
+
|
|
49
|
+
PROMPTS = [
|
|
50
|
+
("chat", "hi"),
|
|
51
|
+
("list files", "List the files in this folder and tell me what each one is for."),
|
|
52
|
+
("read a file", "Read demo.txt and tell me the secret phrase in it."),
|
|
53
|
+
("write + run", "Create a file called fizzbuzz.py that prints FizzBuzz from 1 to 30, then run it with python and "
|
|
54
|
+
"show me the output."),
|
|
55
|
+
("edit + run", "Add a function is_prime(n) with a docstring to fizzbuzz.py, and at the end print all primes below "
|
|
56
|
+
"30. Run it again and show the output."),
|
|
57
|
+
("fix 3 bugs", "Run buggy.py. The comments say what each line should print. Find and fix all the bugs, then run it "
|
|
58
|
+
"again to show it is correct."),
|
|
59
|
+
("write tests", "Write unit tests for the functions in buggy.py in a file called test_buggy.py using unittest. Run "
|
|
60
|
+
"them and make sure they all pass."),
|
|
61
|
+
("summary", "Summarize which files you created or changed in this session and what each one does."),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def run_py(folder, *args):
|
|
66
|
+
try:
|
|
67
|
+
out = subprocess.run([sys.executable, *args], cwd=folder, capture_output=True, text=True, timeout=60)
|
|
68
|
+
return out.returncode, out.stdout + out.stderr
|
|
69
|
+
except subprocess.TimeoutExpired:
|
|
70
|
+
return -1, "timeout"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def grade(step, folder, reply):
|
|
74
|
+
"""(points 0..1, note) for prompt `step` after the agent answered `reply`."""
|
|
75
|
+
reply_l = reply.lower()
|
|
76
|
+
if step == 0:
|
|
77
|
+
return (1, "replied") if reply.strip() else (0, "no reply")
|
|
78
|
+
if step == 1:
|
|
79
|
+
ok = "demo.txt" in reply_l and "buggy.py" in reply_l
|
|
80
|
+
return (1, "named both files") if ok else (0, "did not name demo.txt and buggy.py")
|
|
81
|
+
if step == 2:
|
|
82
|
+
return (1, "found the phrase") if "copper owl builds a bridge" in reply_l else (0, "wrong or no phrase")
|
|
83
|
+
if step == 3:
|
|
84
|
+
if not (folder / "fizzbuzz.py").exists():
|
|
85
|
+
return 0, "fizzbuzz.py not created"
|
|
86
|
+
code, out = run_py(folder, "fizzbuzz.py")
|
|
87
|
+
lines = [line.strip() for line in out.splitlines() if line.strip()]
|
|
88
|
+
return (1, "output correct") if code == 0 and lines[:30] == FIZZBUZZ else (0, "wrong FizzBuzz output")
|
|
89
|
+
if step == 4:
|
|
90
|
+
path = folder / "fizzbuzz.py"
|
|
91
|
+
if not path.exists():
|
|
92
|
+
return 0, "fizzbuzz.py missing"
|
|
93
|
+
try:
|
|
94
|
+
funcs = {n.name: n for n in ast.walk(ast.parse(path.read_text(encoding="utf-8-sig"))) if isinstance(n, ast.FunctionDef)}
|
|
95
|
+
except SyntaxError:
|
|
96
|
+
return 0, "fizzbuzz.py does not parse"
|
|
97
|
+
code, out = run_py(folder, "fizzbuzz.py")
|
|
98
|
+
lines = [line.strip() for line in out.splitlines() if line.strip()]
|
|
99
|
+
tail = " ".join(lines[30:])
|
|
100
|
+
numbers = {int(n) for n in re.findall(r"\b\d+\b", tail)}
|
|
101
|
+
checks = ["is_prime" in funcs, bool(funcs.get("is_prime") and ast.get_docstring(funcs["is_prime"])),
|
|
102
|
+
code == 0 and lines[:30] == FIZZBUZZ, PRIMES <= numbers and not (numbers - PRIMES - {30})]
|
|
103
|
+
notes = ["is_prime", "docstring", "FizzBuzz still right", "primes printed"]
|
|
104
|
+
missing = [n for n, ok in zip(notes, checks) if not ok]
|
|
105
|
+
return (1, "all correct") if not missing else (sum(checks) / 4, "missing: " + ", ".join(missing))
|
|
106
|
+
if step == 5:
|
|
107
|
+
code, out = run_py(folder, "buggy.py")
|
|
108
|
+
lines = [line.strip() for line in out.splitlines() if line.strip()]
|
|
109
|
+
expected = ["20.0", "[9, 5]", "0"]
|
|
110
|
+
right = sum(1 for got, want in zip(lines, expected) if got in (want, "0.0" if want == "0" else want))
|
|
111
|
+
return (1, "all 3 bugs fixed") if code == 0 and right == 3 else (right / 3, f"{right}/3 lines right")
|
|
112
|
+
if step == 6:
|
|
113
|
+
path = folder / "test_buggy.py"
|
|
114
|
+
if not path.exists():
|
|
115
|
+
return 0, "test_buggy.py not created"
|
|
116
|
+
code, out = run_py(folder, "-m", "unittest", "-v", "test_buggy")
|
|
117
|
+
tests = len(re.findall(r" \.\.\. ok", out))
|
|
118
|
+
if code != 0 or not tests:
|
|
119
|
+
return 0, "tests fail or none ran"
|
|
120
|
+
source = path.read_text(encoding="utf-8-sig")
|
|
121
|
+
if not re.search(r"^\s*(from buggy import|import buggy)", source, re.M):
|
|
122
|
+
return 0.5, f"{tests} tests pass, but they test a copy instead of importing buggy.py"
|
|
123
|
+
return 1, f"{tests} tests pass, importing buggy.py"
|
|
124
|
+
named = [f for f in ("fizzbuzz.py", "buggy.py", "test_buggy.py") if f in reply_l]
|
|
125
|
+
return (1, "named all 3 files") if len(named) == 3 else (len(named) / 3, f"named {len(named)}/3 files")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class AgentError(Exception):
|
|
129
|
+
"""The agent itself failed (crashed, refused to start, or reached the wrong model): not a model score."""
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def ask_agent(agent, config, model, folder, prompt, first, session):
|
|
133
|
+
"""One turn. Returns (reply, session id); raises AgentError when the agent itself fails."""
|
|
134
|
+
port, context = config["server"]["port"], config["server"]["context"]
|
|
135
|
+
if agent == "claude":
|
|
136
|
+
# dontAsk: the allowed tools just run. Otherwise a user's default "auto" mode asks the (local, slow) model to
|
|
137
|
+
# classify every command first; with GLM-4.7-Flash those checks timed out and blocked the commands.
|
|
138
|
+
args = ["-p", prompt, "--output-format", "json", "--permission-mode", "dontAsk",
|
|
139
|
+
"--allowedTools", "Read Write Edit Bash Glob Grep"]
|
|
140
|
+
if not first:
|
|
141
|
+
args.append("--continue")
|
|
142
|
+
result = launch("claude", port, model, context, capture=True, cwd=folder, timeout=TIMEOUT, args=args)
|
|
143
|
+
try:
|
|
144
|
+
data = json.loads(result.stdout.strip().splitlines()[-1])
|
|
145
|
+
except (ValueError, IndexError):
|
|
146
|
+
raise AgentError(f"Claude Code exited with {result.returncode}: {(result.stdout + result.stderr)[-400:]}")
|
|
147
|
+
if data.get("is_error") and not data.get("result"):
|
|
148
|
+
raise AgentError(f"Claude Code reported an error: {str(data)[:400]}")
|
|
149
|
+
return data.get("result") or "", session
|
|
150
|
+
last = folder / ".codex-last-message.txt"
|
|
151
|
+
last.unlink(missing_ok=True) # never grade a turn on the previous turn's reply
|
|
152
|
+
# `codex exec resume` has no -s flag; the sandbox is set as config, which both forms accept.
|
|
153
|
+
common = ["--skip-git-repo-check", "-c", 'sandbox_mode="workspace-write"', "-o", str(last)]
|
|
154
|
+
args = ["exec", *common, prompt] if first else ["exec", "resume", *common, session, prompt]
|
|
155
|
+
result = launch("codex", port, model, context, capture=True, cwd=folder, timeout=TIMEOUT, args=args)
|
|
156
|
+
output = result.stdout + result.stderr
|
|
157
|
+
provider = re.search(r"^provider:\s*(\S+)", output, re.M)
|
|
158
|
+
if provider and provider.group(1) != "llamacpp":
|
|
159
|
+
raise AgentError(f"Codex used provider '{provider.group(1)}' instead of the local model; stopped.")
|
|
160
|
+
if first:
|
|
161
|
+
found = re.search(r"session id:\s*([0-9a-f-]{36})", output)
|
|
162
|
+
session = found.group(1) if found else None
|
|
163
|
+
if not session:
|
|
164
|
+
raise AgentError(f"Codex did not start a session (exit {result.returncode}): {output[-400:]}")
|
|
165
|
+
if result.returncode != 0 and not last.exists():
|
|
166
|
+
raise AgentError(f"Codex exited with {result.returncode}: {output[-400:]}")
|
|
167
|
+
reply = last.read_text(encoding="utf-8", errors="replace") if last.exists() else ""
|
|
168
|
+
return reply, session
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def scoreboard():
|
|
172
|
+
if not RESULTS.exists():
|
|
173
|
+
print("No benchmark results yet.")
|
|
174
|
+
return
|
|
175
|
+
runs = [json.loads(line) for line in RESULTS.read_text(encoding="utf-8").splitlines() if line.strip()]
|
|
176
|
+
runs.sort(key=lambda r: (-r["score"], r["seconds"]))
|
|
177
|
+
print(f"\n{BOLD}Scoreboard{RESET} {DIM}({RESULTS}){RESET}")
|
|
178
|
+
print(f" {'model':<48} {'agent':<12} {'score':>7} {'time':>8} date")
|
|
179
|
+
for r in runs:
|
|
180
|
+
color = GREEN if r["score"] >= 7 else YELLOW if r["score"] >= 5 else RED
|
|
181
|
+
print(f" {r['model'][:48]:<48} {r['agent']:<12} {color}{r['score']:>5.1f}/8{RESET} {r['seconds'] / 60:>6.1f} m {r['date']}")
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def main(argv=None):
|
|
185
|
+
parser = argparse.ArgumentParser(prog="hearthwork bench", description="Benchmark the running model through a coding agent.")
|
|
186
|
+
parser.add_argument("--agent", choices=sorted(HARNESSES), default="claude")
|
|
187
|
+
parser.add_argument("--show", action="store_true", help="only print the scoreboard")
|
|
188
|
+
args = parser.parse_args(argv)
|
|
189
|
+
if args.show:
|
|
190
|
+
return scoreboard()
|
|
191
|
+
config = load_config()
|
|
192
|
+
model = served_model(config["server"]["port"])
|
|
193
|
+
if not model:
|
|
194
|
+
sys.exit("No model is running. Start one first (hearthwork menu: Start / switch model).")
|
|
195
|
+
if not installed(args.agent):
|
|
196
|
+
sys.exit(f"{HARNESSES[args.agent]['title']} is not installed.")
|
|
197
|
+
stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
|
|
198
|
+
folder = BENCH / "runs" / f"{stamp}-{args.agent}-{model}"
|
|
199
|
+
folder.mkdir(parents=True)
|
|
200
|
+
(folder / "demo.txt").write_text(DEMO, encoding="utf-8")
|
|
201
|
+
(folder / "buggy.py").write_text(BUGGY, encoding="utf-8")
|
|
202
|
+
title = HARNESSES[args.agent]["title"]
|
|
203
|
+
print(f"\n{BOLD}Benchmark{RESET}: {model} via {title} {DIM}(8 prompts, one conversation; folder {folder}){RESET}\n")
|
|
204
|
+
|
|
205
|
+
steps, session, started = [], None, time.time()
|
|
206
|
+
for i, (name, prompt) in enumerate(PROMPTS):
|
|
207
|
+
print(f" {i + 1}. {name:<12}", end=" ", flush=True)
|
|
208
|
+
t0 = time.time()
|
|
209
|
+
try:
|
|
210
|
+
reply, session = ask_agent(args.agent, config, model, folder, prompt, i == 0, session)
|
|
211
|
+
except subprocess.TimeoutExpired:
|
|
212
|
+
reply = ""
|
|
213
|
+
except AgentError as error:
|
|
214
|
+
print(f"\n\n {RED}{error}{RESET}\n Not recorded: this is an agent problem, not a model score.")
|
|
215
|
+
sys.exit(1)
|
|
216
|
+
seconds = time.time() - t0
|
|
217
|
+
points, note = grade(i, folder, reply)
|
|
218
|
+
color = GREEN if points == 1 else YELLOW if points > 0 else RED
|
|
219
|
+
print(f"{color}{points:>4.2g}{RESET} {seconds:6.0f} s {DIM}{note}{RESET}")
|
|
220
|
+
steps.append({"step": name, "points": points, "seconds": round(seconds, 1), "note": note})
|
|
221
|
+
|
|
222
|
+
total = sum(s["points"] for s in steps)
|
|
223
|
+
seconds = time.time() - started
|
|
224
|
+
record = {"date": datetime.now().strftime("%Y-%m-%d %H:%M"), "model": model, "agent": title,
|
|
225
|
+
"score": round(total, 2), "seconds": round(seconds), "steps": steps, "folder": str(folder)}
|
|
226
|
+
BENCH.mkdir(exist_ok=True)
|
|
227
|
+
with open(RESULTS, "a", encoding="utf-8") as f:
|
|
228
|
+
f.write(json.dumps(record) + "\n")
|
|
229
|
+
print(f"\n {BOLD}Score {total:.1f}/8{RESET} in {seconds / 60:.1f} min")
|
|
230
|
+
scoreboard()
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
if __name__ == "__main__":
|
|
234
|
+
try:
|
|
235
|
+
main()
|
|
236
|
+
except KeyboardInterrupt:
|
|
237
|
+
print("\nStopped.")
|
hearthwork/check.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Hearthwork check: will a local model for coding agents be worth it on this computer, and which one?
|
|
3
|
+
|
|
4
|
+
Run it without installing anything:
|
|
5
|
+
uvx --from git+https://github.com/BillyMRX1/hearthwork hearthwork-check
|
|
6
|
+
or, once installed: hearthwork check (--json for scripts)
|
|
7
|
+
|
|
8
|
+
Read-only: it checks the hardware and installed agents, gives the same verdict Hearthwork's setup gives, and
|
|
9
|
+
looks up live file sizes on Hugging Face to pick the best version of each suggested model that fits.
|
|
10
|
+
"""
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
import platform
|
|
14
|
+
import re
|
|
15
|
+
import shutil
|
|
16
|
+
import subprocess
|
|
17
|
+
import sys
|
|
18
|
+
import urllib.request
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from .onboard import CYAN, GREEN, RED, RESET, WINDOWS, YELLOW, detect, judge, lmstudio_folders, memory_gb
|
|
22
|
+
|
|
23
|
+
BOLD, DIM = "\033[1m", "\033[2m"
|
|
24
|
+
REPO = "https://github.com/BillyMRX1/hearthwork"
|
|
25
|
+
|
|
26
|
+
# Coding models with tool-calling chat templates (checked on Hugging Face, 2026-10). "moe": only a few experts run
|
|
27
|
+
# per token, so the model stays usable when part of it sits in system RAM. Measured on the reference PC
|
|
28
|
+
# (RTX 5060 Ti 16 GB + 22.6 GB RAM, 64K context): MoE Qwen3-Coder-30B partly in RAM 18-31 tokens/s; dense
|
|
29
|
+
# Qwen3.8-27B partly in RAM ~6 tokens/s.
|
|
30
|
+
# "tested": Hearthwork benchmark (bench.py, 8 graded coding tasks) on the reference PC, Q4_K_M, 2026-10-07.
|
|
31
|
+
MODELS = [
|
|
32
|
+
{"repo": "unsloth/Qwen3.5-35B-A3B-GGUF", "name": "Qwen3.5-35B-A3B", "moe": True, "tested": True,
|
|
33
|
+
"note": "best in the benchmark: 8/8 with Codex (2.6 min) and Claude Code (3.6 min)"},
|
|
34
|
+
{"repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF", "name": "Qwen3-Coder-30B-A3B", "moe": True, "tested": True,
|
|
35
|
+
"note": "8/8 with Claude Code (3.9 min) and Codex (4.2 min); a bit smaller"},
|
|
36
|
+
{"repo": "unsloth/GLM-4.7-Flash-GGUF", "name": "GLM-4.7-Flash (30B MoE)", "moe": True, "tested": True,
|
|
37
|
+
"note": "8/8 with Codex (4.5 min), 7.7/8 with Claude Code (missed one bug)"},
|
|
38
|
+
{"repo": "ggml-org/gpt-oss-20b-GGUF", "name": "gpt-oss-20b", "moe": True, "note": "small and fast; good at tool use"},
|
|
39
|
+
{"repo": "unsloth/Devstral-Small-2-24B-Instruct-2512-GGUF", "name": "Devstral-Small-2-24B", "moe": False,
|
|
40
|
+
"note": "coding agent (dense: needs to fit the GPU to be fast)"},
|
|
41
|
+
{"repo": "unsloth/Qwen3.5-27B-GGUF", "name": "Qwen3.5-27B", "moe": False, "note": "dense"},
|
|
42
|
+
{"repo": "unsloth/Qwen3-Coder-Next-GGUF", "name": "Qwen3-Coder-Next (80B MoE)", "moe": True,
|
|
43
|
+
"note": "strongest here; needs a big machine"},
|
|
44
|
+
{"repo": "lmstudio-community/Qwen3.5-9B-GGUF", "name": "Qwen3.5-9B", "moe": False,
|
|
45
|
+
"note": "small machines only; much weaker with agents' tools"},
|
|
46
|
+
]
|
|
47
|
+
# Best quality first. Below Q3, answers get noticeably worse.
|
|
48
|
+
QUANTS = ["Q4_K_M", "UD-Q4_K_XL", "MXFP4", "Q4_K_S", "Q3_K_M", "UD-Q3_K_XL", "Q3_K_S", "UD-Q2_K_XL", "Q2_K"]
|
|
49
|
+
KV_GB_64K = 3.4 # KV cache for a 64K context at q8_0, measured with Qwen3-Coder-30B (varies by model)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def version(binary):
|
|
53
|
+
path = shutil.which(binary)
|
|
54
|
+
if not path:
|
|
55
|
+
return None
|
|
56
|
+
try:
|
|
57
|
+
out = subprocess.run([path, "--version"], capture_output=True, text=True, timeout=20).stdout.strip()
|
|
58
|
+
return out.splitlines()[0] if out else "installed"
|
|
59
|
+
except (OSError, subprocess.SubprocessError):
|
|
60
|
+
return "installed"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def quant_sizes(repo):
|
|
64
|
+
"""{quant: GB} from Hugging Face (multi-part files summed); None when offline."""
|
|
65
|
+
url = f"https://huggingface.co/api/models/{repo}/tree/main?recursive=true"
|
|
66
|
+
try:
|
|
67
|
+
with urllib.request.urlopen(urllib.request.Request(url, headers={"User-Agent": "hearthwork-check"}), timeout=20) as r:
|
|
68
|
+
files = json.load(r)
|
|
69
|
+
except Exception:
|
|
70
|
+
return None
|
|
71
|
+
sizes = {}
|
|
72
|
+
for f in files:
|
|
73
|
+
path = f.get("path", "")
|
|
74
|
+
if not path.endswith(".gguf") or "mmproj" in path.lower():
|
|
75
|
+
continue
|
|
76
|
+
name = re.sub(r"-\d{5}-of-\d{5}", "", Path(path).name)
|
|
77
|
+
for quant in QUANTS:
|
|
78
|
+
if name.upper().endswith(f"-{quant.upper()}.GGUF") or name.upper().endswith(f"_{quant.upper()}.GGUF"):
|
|
79
|
+
sizes[quant] = sizes.get(quant, 0) + (f.get("lfs") or {}).get("size", f.get("size", 0)) / 2**30
|
|
80
|
+
return sizes
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def place(model, sizes, gpu_room, ram_room):
|
|
84
|
+
"""Best quantization for this machine -> (quant, GB, fit) with fit in fast / good / slow, or None if too big."""
|
|
85
|
+
if model["moe"]: # MoE stays fast partly in RAM: best quality that fits at all
|
|
86
|
+
for quant in QUANTS:
|
|
87
|
+
if quant in sizes and sizes[quant] <= gpu_room + ram_room:
|
|
88
|
+
return quant, sizes[quant], "fast" if sizes[quant] <= gpu_room else "good"
|
|
89
|
+
return None
|
|
90
|
+
for quant in QUANTS: # dense: fitting the GPU entirely matters most
|
|
91
|
+
if quant in sizes and sizes[quant] <= gpu_room:
|
|
92
|
+
return quant, sizes[quant], "fast"
|
|
93
|
+
for quant in QUANTS: # then GPU + RAM
|
|
94
|
+
if quant in sizes and sizes[quant] <= gpu_room + ram_room:
|
|
95
|
+
return quant, sizes[quant], "good" if model["moe"] else "slow"
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def main(argv=None):
|
|
100
|
+
parser = argparse.ArgumentParser(prog="hearthwork check", description="Will Hearthwork (local models for coding agents) be worth it here?")
|
|
101
|
+
parser.add_argument("--json", action="store_true", help="machine-readable output")
|
|
102
|
+
args = parser.parse_args(argv)
|
|
103
|
+
|
|
104
|
+
hw = detect()
|
|
105
|
+
verdict, reasons, expect = judge(hw)
|
|
106
|
+
ram_total, ram_free = memory_gb()
|
|
107
|
+
home_disk = shutil.disk_usage(Path.home()).free / 2**30
|
|
108
|
+
agents = {"Claude Code": version("claude"), "Codex": version("codex")}
|
|
109
|
+
vram = hw.get("vramGB") or 0
|
|
110
|
+
if hw["backend"] == "metal":
|
|
111
|
+
gpu_room, ram_room = max(0.0, hw["ramGB"] * 0.75 - 1.5 - KV_GB_64K), 0.0
|
|
112
|
+
else:
|
|
113
|
+
gpu_room = max(0.0, vram - 1.5 - KV_GB_64K) if hw["backend"] in ("cuda", "vulkan") else 0.0
|
|
114
|
+
ram_room = max(0.0, hw["ramGB"] - 6)
|
|
115
|
+
|
|
116
|
+
picks = []
|
|
117
|
+
offline = False
|
|
118
|
+
for model in MODELS:
|
|
119
|
+
sizes = quant_sizes(model["repo"])
|
|
120
|
+
if sizes is None:
|
|
121
|
+
offline = True
|
|
122
|
+
continue
|
|
123
|
+
fit = place(model, sizes, gpu_room, ram_room)
|
|
124
|
+
picks.append({**model, "quant": fit[0] if fit else None, "sizeGB": round(fit[1], 1) if fit else None,
|
|
125
|
+
"fit": fit[2] if fit else "too big"})
|
|
126
|
+
rank = {"fast": 0, "good": 1, "slow": 2, "too big": 3}
|
|
127
|
+
picks.sort(key=lambda p: (p["fit"] == "too big", not p.get("tested"), rank[p["fit"]], MODELS.index(next(m for m in MODELS if m["repo"] == p["repo"]))))
|
|
128
|
+
|
|
129
|
+
if args.json:
|
|
130
|
+
print(json.dumps({"hardware": hw, "ramFreeGB": round(ram_free, 1), "diskFreeGB": round(home_disk, 1),
|
|
131
|
+
"verdict": verdict, "reasons": reasons, "expect": expect, "agents": agents,
|
|
132
|
+
"models": picks, "offline": offline}, indent=2))
|
|
133
|
+
return
|
|
134
|
+
|
|
135
|
+
print(f"\n{BOLD}Hearthwork check{RESET} {DIM}(local models for Claude Code / Codex){RESET}\n")
|
|
136
|
+
print(f"{BOLD}Your computer{RESET}")
|
|
137
|
+
print(f" OS {platform.system()} {platform.release()} ({hw['arch']}) Python {platform.python_version()}")
|
|
138
|
+
print(f" CPU {hw['cpuThreads']} threads")
|
|
139
|
+
print(f" RAM {ram_total:.1f} GB ({ram_free:.1f} GB free now)")
|
|
140
|
+
gpus = ", ".join(hw["gpus"]) or "no usable GPU"
|
|
141
|
+
print(f" GPU {gpus}" + (f" ({vram:.1f} GB)" if hw["backend"] == "vulkan" and vram else "")
|
|
142
|
+
+ f" -> llama.cpp {hw['backend']}" + (f", driver CUDA {hw['cudaDriver']}" if hw.get("cudaDriver") else ""))
|
|
143
|
+
print(f" Disk {home_disk:.0f} GB free in your home drive"
|
|
144
|
+
+ (f"; LM Studio models: {lmstudio_folders()[0]}" if lmstudio_folders() else ""))
|
|
145
|
+
|
|
146
|
+
color = {"recommended": GREEN, "limited": YELLOW}.get(verdict, RED)
|
|
147
|
+
print(f"\n{BOLD}Verdict{RESET} {color}{BOLD}{verdict.upper()}{RESET}")
|
|
148
|
+
for reason in reasons:
|
|
149
|
+
print(f" - {reason}")
|
|
150
|
+
print(f" {expect}")
|
|
151
|
+
|
|
152
|
+
print(f"\n{BOLD}Coding agents{RESET}")
|
|
153
|
+
for name, found in agents.items():
|
|
154
|
+
print(f" {name:12} " + (f"{GREEN}{found}{RESET}" if found else f"{YELLOW}not installed{RESET}"))
|
|
155
|
+
if not any(agents.values()):
|
|
156
|
+
print(f" {YELLOW}Install at least one: Claude Code (https://code.claude.com) or Codex "
|
|
157
|
+
f"(https://developers.openai.com/codex).{RESET}")
|
|
158
|
+
|
|
159
|
+
print(f"\n{BOLD}Models for this computer{RESET} {DIM}(best version that fits; sizes live from Hugging Face){RESET}")
|
|
160
|
+
labels = {"fast": f"{GREEN}fits the GPU: fast{RESET}", "good": f"{GREEN}GPU + RAM: good (MoE){RESET}",
|
|
161
|
+
"slow": f"{YELLOW}GPU + RAM: slow (dense){RESET}", "too big": f"{RED}too big{RESET}"}
|
|
162
|
+
for p in picks:
|
|
163
|
+
size = f"{p['quant']:<10} {p['sizeGB']:5.1f} GB" if p["quant"] else " " * 19
|
|
164
|
+
tested = f" {CYAN}[tested]{RESET}" if p.get("tested") else ""
|
|
165
|
+
print(f" {p['name']:<28} {size} {labels[p['fit']]}{tested}")
|
|
166
|
+
print(f" {DIM}{'':<28} {p['note']}{RESET}")
|
|
167
|
+
if offline:
|
|
168
|
+
print(f" {YELLOW}Some models could not be looked up (offline?).{RESET}")
|
|
169
|
+
print(f" {DIM}Measured on an RTX 5060 Ti 16 GB + 22.6 GB RAM: MoE partly in RAM 18-31 tokens/s, "
|
|
170
|
+
f"dense partly in RAM ~6 tokens/s.{RESET}")
|
|
171
|
+
|
|
172
|
+
best = next((p for p in picks if p["fit"] in ("fast", "good")), None)
|
|
173
|
+
print(f"\n{BOLD}Next step{RESET}")
|
|
174
|
+
if verdict == "not recommended" or not best:
|
|
175
|
+
print(" A local model would be too slow or too weak for coding on this computer. Cloud Claude Code / Codex is the"
|
|
176
|
+
" better choice here.")
|
|
177
|
+
else:
|
|
178
|
+
print(f" 1. Install Hearthwork: {CYAN}uv tool install hearthwork{RESET} (or: pipx install hearthwork)")
|
|
179
|
+
print(f" 2. Download a model: {CYAN}hearthwork model {best['repo']}{RESET} "
|
|
180
|
+
f"(pick {best['quant']}, {best['sizeGB']} GB)")
|
|
181
|
+
print(f" 3. From your project folder, run {CYAN}hearthwork{RESET}: setup takes a minute, then pick your agent.")
|
|
182
|
+
print()
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
if __name__ == "__main__":
|
|
186
|
+
try:
|
|
187
|
+
main()
|
|
188
|
+
except KeyboardInterrupt:
|
|
189
|
+
sys.exit(130)
|
hearthwork/cli.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""The `hearthwork` command.
|
|
2
|
+
|
|
3
|
+
hearthwork menu (run it from the project you want the agent to work on)
|
|
4
|
+
hearthwork claude [args...] Claude Code with the local model (starts one if needed); args go to `claude`
|
|
5
|
+
hearthwork codex [args...] Codex with the local model; args go to `codex`
|
|
6
|
+
hearthwork start [--model X] start the model in the background hearthwork stop
|
|
7
|
+
hearthwork serve [--model X] run the model server in this terminal (Ctrl+C stops it)
|
|
8
|
+
hearthwork status what is running
|
|
9
|
+
hearthwork model <link> download a GGUF model from Hugging Face
|
|
10
|
+
hearthwork bench 8 graded coding tasks through an agent, with a scoreboard
|
|
11
|
+
hearthwork check is this computer suited, and which models fit
|
|
12
|
+
hearthwork setup hardware check, llama.cpp download/update, models folder
|
|
13
|
+
hearthwork update update Hearthwork itself
|
|
14
|
+
"""
|
|
15
|
+
import shutil
|
|
16
|
+
import subprocess
|
|
17
|
+
import sys
|
|
18
|
+
|
|
19
|
+
from . import __version__
|
|
20
|
+
from .harnesses import HARNESSES
|
|
21
|
+
from .onboard import CYAN, GREEN, RESET, load_config, offer_import, onboard, setup_complete
|
|
22
|
+
from .paths import HOME
|
|
23
|
+
|
|
24
|
+
REPO = "git+https://github.com/BillyMRX1/hearthwork"
|
|
25
|
+
USAGE = __doc__.split("\n", 2)[2]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def configured(interactive=True):
|
|
29
|
+
"""config, running first-time setup (or importing an older setup) when needed."""
|
|
30
|
+
config = load_config()
|
|
31
|
+
if setup_complete(config):
|
|
32
|
+
return config
|
|
33
|
+
config = offer_import(config)
|
|
34
|
+
if setup_complete(config):
|
|
35
|
+
return config
|
|
36
|
+
config = onboard(config)
|
|
37
|
+
if interactive:
|
|
38
|
+
input("\nPress Enter to continue...")
|
|
39
|
+
return config
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def model_args(argv, name):
|
|
43
|
+
import argparse
|
|
44
|
+
parser = argparse.ArgumentParser(prog=f"hearthwork {name}")
|
|
45
|
+
parser.add_argument("--model", help="part of a model file name; skips the model menu")
|
|
46
|
+
parser.add_argument("--context", type=int, help="context size (default: chosen by setup)")
|
|
47
|
+
return parser.parse_args(argv)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def update():
|
|
51
|
+
"""Upgrade with whichever tool installed Hearthwork."""
|
|
52
|
+
if shutil.which("uv") and "uv" in sys.executable.replace("\\", "/").split("/"):
|
|
53
|
+
return subprocess.call(["uv", "tool", "upgrade", "hearthwork"])
|
|
54
|
+
if shutil.which("pipx") and "pipx" in sys.executable.replace("\\", "/"):
|
|
55
|
+
return subprocess.call(["pipx", "upgrade", "hearthwork"])
|
|
56
|
+
print(f"Upgrade with the tool you installed Hearthwork with, e.g.\n uv tool upgrade hearthwork\n"
|
|
57
|
+
f" pipx upgrade hearthwork\n pip install --upgrade hearthwork")
|
|
58
|
+
return 0
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def main(argv=None):
|
|
62
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
63
|
+
command, rest = (argv[0], argv[1:]) if argv else ("menu", [])
|
|
64
|
+
try:
|
|
65
|
+
if command in ("-h", "--help", "help"):
|
|
66
|
+
print(USAGE)
|
|
67
|
+
elif command in ("-V", "--version", "version"):
|
|
68
|
+
print(f"hearthwork {__version__} (data: {HOME})")
|
|
69
|
+
elif command == "menu":
|
|
70
|
+
from . import menu
|
|
71
|
+
menu.main(configured())
|
|
72
|
+
elif command in HARNESSES: # every following argument belongs to the agent (e.g. -p "...")
|
|
73
|
+
from .menu import run_agent
|
|
74
|
+
sys.exit(run_agent(configured(), command, rest, back_to_menu=False))
|
|
75
|
+
elif command in ("start", "serve"):
|
|
76
|
+
from .server import choose_and_remember, run_foreground, start_background
|
|
77
|
+
args = model_args(rest, command)
|
|
78
|
+
config = configured()
|
|
79
|
+
model = choose_and_remember(config, args.model)
|
|
80
|
+
if not model:
|
|
81
|
+
sys.exit(1)
|
|
82
|
+
if command == "start":
|
|
83
|
+
ok = start_background(config, model, args.context)
|
|
84
|
+
if ok:
|
|
85
|
+
print(f"{CYAN}Now run `hearthwork claude` or `hearthwork codex` from your project.{RESET}")
|
|
86
|
+
sys.exit(0 if ok else 1)
|
|
87
|
+
print(f"\nStarting {model.name} context: {args.context or config['server']['context']} "
|
|
88
|
+
f"port: {config['server']['port']}")
|
|
89
|
+
print(f"{CYAN}When it says 'listening on', run `hearthwork claude` or `hearthwork codex` "
|
|
90
|
+
f"from your project.{RESET}\n", flush=True)
|
|
91
|
+
run_foreground(config, model, args.context)
|
|
92
|
+
elif command == "stop":
|
|
93
|
+
from .server import stop
|
|
94
|
+
stop(load_config())
|
|
95
|
+
elif command == "status":
|
|
96
|
+
from .server import served_model
|
|
97
|
+
config = load_config()
|
|
98
|
+
port = config.get("server", {}).get("port", 8001)
|
|
99
|
+
running = served_model(port)
|
|
100
|
+
print(f"hearthwork {__version__} data: {HOME}")
|
|
101
|
+
print(f"model: {GREEN + running + RESET if running else 'not running'}" + (f" (port {port})" if running else ""))
|
|
102
|
+
print(f"models folder: {config.get('modelsDir', '-')}")
|
|
103
|
+
elif command == "model":
|
|
104
|
+
from . import model
|
|
105
|
+
configured()
|
|
106
|
+
model.main(rest)
|
|
107
|
+
elif command == "bench":
|
|
108
|
+
from . import bench
|
|
109
|
+
if "--show" not in rest:
|
|
110
|
+
from .menu import ensure_server
|
|
111
|
+
if not ensure_server(configured()):
|
|
112
|
+
sys.exit(1)
|
|
113
|
+
bench.main(rest)
|
|
114
|
+
elif command == "check":
|
|
115
|
+
from . import check
|
|
116
|
+
check.main(rest)
|
|
117
|
+
elif command == "setup":
|
|
118
|
+
from . import onboard as setup
|
|
119
|
+
setup.main(rest)
|
|
120
|
+
elif command == "update":
|
|
121
|
+
sys.exit(update())
|
|
122
|
+
else:
|
|
123
|
+
print(f"Unknown command: {command}\n\n{USAGE}")
|
|
124
|
+
sys.exit(2)
|
|
125
|
+
except KeyboardInterrupt:
|
|
126
|
+
print()
|
|
127
|
+
sys.exit(130)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def check_main():
|
|
131
|
+
"""`hearthwork-check`: the checker on its own, e.g. `uvx --from git+https://github.com/BillyMRX1/hearthwork hearthwork-check`."""
|
|
132
|
+
from . import check
|
|
133
|
+
check.main()
|