hearthwork 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hearthwork/__init__.py ADDED
@@ -0,0 +1,2 @@
1
+ """Hearthwork: run a local llama.cpp model and connect Claude Code or Codex to it, tuned to your own machine."""
2
+ __version__ = "0.2.1"
hearthwork/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
hearthwork/bench.py ADDED
@@ -0,0 +1,237 @@
1
+ #!/usr/bin/env python3
2
+ """Benchmark the running model through a real coding agent, and keep a scoreboard.
3
+
4
+ hearthwork bench [--agent claude|codex] [--show]
5
+ (or menu option "Benchmark the running model")
6
+
7
+ Eight prompts run as one agent conversation in a fresh folder (bench/runs/...): chat, list files, read a file,
8
+ write and run a script, edit it, fix three planted bugs, write unit tests, summarize. Each step is graded by
9
+ checking the files and running the code, not by trusting the agent's reply. Results go to bench/results.jsonl;
10
+ --show prints the scoreboard without running anything.
11
+ """
12
+ import argparse
13
+ import ast
14
+ import json
15
+ import re
16
+ import shutil
17
+ import subprocess
18
+ import sys
19
+ import time
20
+ from datetime import datetime
21
+ from pathlib import Path
22
+
23
+ from .harnesses import HARNESSES, installed, launch
24
+ from .onboard import CYAN, GREEN, RED, RESET, YELLOW, load_config
25
+ from .paths import BENCH
26
+ from .server import served_model
27
+
28
+ RESULTS = BENCH / "results.jsonl"
29
+ BOLD, DIM = "\033[1m", "\033[2m"
30
+ TIMEOUT = 20 * 60 # per prompt
31
+
32
+ DEMO = "The secret phrase is: copper owl builds a bridge.\n"
33
+ BUGGY = '''def average(numbers):
34
+ total = 0
35
+ for i in range(1, len(numbers)):
36
+ total += numbers[i]
37
+ return total / len(numbers)
38
+
39
+ def top_n(items, n):
40
+ return sorted(items)[:n]
41
+
42
+ print(average([10, 20, 30])) # should print 20.0
43
+ print(top_n([5, 1, 9, 3], 2)) # should print [9, 5]
44
+ print(average([])) # should print 0, not crash
45
+ '''
46
+ FIZZBUZZ = ["FizzBuzz" if i % 15 == 0 else "Fizz" if i % 3 == 0 else "Buzz" if i % 5 == 0 else str(i) for i in range(1, 31)]
47
+ PRIMES = {2, 3, 5, 7, 11, 13, 17, 19, 23, 29}
48
+
49
+ PROMPTS = [
50
+ ("chat", "hi"),
51
+ ("list files", "List the files in this folder and tell me what each one is for."),
52
+ ("read a file", "Read demo.txt and tell me the secret phrase in it."),
53
+ ("write + run", "Create a file called fizzbuzz.py that prints FizzBuzz from 1 to 30, then run it with python and "
54
+ "show me the output."),
55
+ ("edit + run", "Add a function is_prime(n) with a docstring to fizzbuzz.py, and at the end print all primes below "
56
+ "30. Run it again and show the output."),
57
+ ("fix 3 bugs", "Run buggy.py. The comments say what each line should print. Find and fix all the bugs, then run it "
58
+ "again to show it is correct."),
59
+ ("write tests", "Write unit tests for the functions in buggy.py in a file called test_buggy.py using unittest. Run "
60
+ "them and make sure they all pass."),
61
+ ("summary", "Summarize which files you created or changed in this session and what each one does."),
62
+ ]
63
+
64
+
65
+ def run_py(folder, *args):
66
+ try:
67
+ out = subprocess.run([sys.executable, *args], cwd=folder, capture_output=True, text=True, timeout=60)
68
+ return out.returncode, out.stdout + out.stderr
69
+ except subprocess.TimeoutExpired:
70
+ return -1, "timeout"
71
+
72
+
73
+ def grade(step, folder, reply):
74
+ """(points 0..1, note) for prompt `step` after the agent answered `reply`."""
75
+ reply_l = reply.lower()
76
+ if step == 0:
77
+ return (1, "replied") if reply.strip() else (0, "no reply")
78
+ if step == 1:
79
+ ok = "demo.txt" in reply_l and "buggy.py" in reply_l
80
+ return (1, "named both files") if ok else (0, "did not name demo.txt and buggy.py")
81
+ if step == 2:
82
+ return (1, "found the phrase") if "copper owl builds a bridge" in reply_l else (0, "wrong or no phrase")
83
+ if step == 3:
84
+ if not (folder / "fizzbuzz.py").exists():
85
+ return 0, "fizzbuzz.py not created"
86
+ code, out = run_py(folder, "fizzbuzz.py")
87
+ lines = [line.strip() for line in out.splitlines() if line.strip()]
88
+ return (1, "output correct") if code == 0 and lines[:30] == FIZZBUZZ else (0, "wrong FizzBuzz output")
89
+ if step == 4:
90
+ path = folder / "fizzbuzz.py"
91
+ if not path.exists():
92
+ return 0, "fizzbuzz.py missing"
93
+ try:
94
+ funcs = {n.name: n for n in ast.walk(ast.parse(path.read_text(encoding="utf-8-sig"))) if isinstance(n, ast.FunctionDef)}
95
+ except SyntaxError:
96
+ return 0, "fizzbuzz.py does not parse"
97
+ code, out = run_py(folder, "fizzbuzz.py")
98
+ lines = [line.strip() for line in out.splitlines() if line.strip()]
99
+ tail = " ".join(lines[30:])
100
+ numbers = {int(n) for n in re.findall(r"\b\d+\b", tail)}
101
+ checks = ["is_prime" in funcs, bool(funcs.get("is_prime") and ast.get_docstring(funcs["is_prime"])),
102
+ code == 0 and lines[:30] == FIZZBUZZ, PRIMES <= numbers and not (numbers - PRIMES - {30})]
103
+ notes = ["is_prime", "docstring", "FizzBuzz still right", "primes printed"]
104
+ missing = [n for n, ok in zip(notes, checks) if not ok]
105
+ return (1, "all correct") if not missing else (sum(checks) / 4, "missing: " + ", ".join(missing))
106
+ if step == 5:
107
+ code, out = run_py(folder, "buggy.py")
108
+ lines = [line.strip() for line in out.splitlines() if line.strip()]
109
+ expected = ["20.0", "[9, 5]", "0"]
110
+ right = sum(1 for got, want in zip(lines, expected) if got in (want, "0.0" if want == "0" else want))
111
+ return (1, "all 3 bugs fixed") if code == 0 and right == 3 else (right / 3, f"{right}/3 lines right")
112
+ if step == 6:
113
+ path = folder / "test_buggy.py"
114
+ if not path.exists():
115
+ return 0, "test_buggy.py not created"
116
+ code, out = run_py(folder, "-m", "unittest", "-v", "test_buggy")
117
+ tests = len(re.findall(r" \.\.\. ok", out))
118
+ if code != 0 or not tests:
119
+ return 0, "tests fail or none ran"
120
+ source = path.read_text(encoding="utf-8-sig")
121
+ if not re.search(r"^\s*(from buggy import|import buggy)", source, re.M):
122
+ return 0.5, f"{tests} tests pass, but they test a copy instead of importing buggy.py"
123
+ return 1, f"{tests} tests pass, importing buggy.py"
124
+ named = [f for f in ("fizzbuzz.py", "buggy.py", "test_buggy.py") if f in reply_l]
125
+ return (1, "named all 3 files") if len(named) == 3 else (len(named) / 3, f"named {len(named)}/3 files")
126
+
127
+
128
+ class AgentError(Exception):
129
+ """The agent itself failed (crashed, refused to start, or reached the wrong model): not a model score."""
130
+
131
+
132
+ def ask_agent(agent, config, model, folder, prompt, first, session):
133
+ """One turn. Returns (reply, session id); raises AgentError when the agent itself fails."""
134
+ port, context = config["server"]["port"], config["server"]["context"]
135
+ if agent == "claude":
136
+ # dontAsk: the allowed tools just run. Otherwise a user's default "auto" mode asks the (local, slow) model to
137
+ # classify every command first; with GLM-4.7-Flash those checks timed out and blocked the commands.
138
+ args = ["-p", prompt, "--output-format", "json", "--permission-mode", "dontAsk",
139
+ "--allowedTools", "Read Write Edit Bash Glob Grep"]
140
+ if not first:
141
+ args.append("--continue")
142
+ result = launch("claude", port, model, context, capture=True, cwd=folder, timeout=TIMEOUT, args=args)
143
+ try:
144
+ data = json.loads(result.stdout.strip().splitlines()[-1])
145
+ except (ValueError, IndexError):
146
+ raise AgentError(f"Claude Code exited with {result.returncode}: {(result.stdout + result.stderr)[-400:]}")
147
+ if data.get("is_error") and not data.get("result"):
148
+ raise AgentError(f"Claude Code reported an error: {str(data)[:400]}")
149
+ return data.get("result") or "", session
150
+ last = folder / ".codex-last-message.txt"
151
+ last.unlink(missing_ok=True) # never grade a turn on the previous turn's reply
152
+ # `codex exec resume` has no -s flag; the sandbox is set as config, which both forms accept.
153
+ common = ["--skip-git-repo-check", "-c", 'sandbox_mode="workspace-write"', "-o", str(last)]
154
+ args = ["exec", *common, prompt] if first else ["exec", "resume", *common, session, prompt]
155
+ result = launch("codex", port, model, context, capture=True, cwd=folder, timeout=TIMEOUT, args=args)
156
+ output = result.stdout + result.stderr
157
+ provider = re.search(r"^provider:\s*(\S+)", output, re.M)
158
+ if provider and provider.group(1) != "llamacpp":
159
+ raise AgentError(f"Codex used provider '{provider.group(1)}' instead of the local model; stopped.")
160
+ if first:
161
+ found = re.search(r"session id:\s*([0-9a-f-]{36})", output)
162
+ session = found.group(1) if found else None
163
+ if not session:
164
+ raise AgentError(f"Codex did not start a session (exit {result.returncode}): {output[-400:]}")
165
+ if result.returncode != 0 and not last.exists():
166
+ raise AgentError(f"Codex exited with {result.returncode}: {output[-400:]}")
167
+ reply = last.read_text(encoding="utf-8", errors="replace") if last.exists() else ""
168
+ return reply, session
169
+
170
+
171
+ def scoreboard():
172
+ if not RESULTS.exists():
173
+ print("No benchmark results yet.")
174
+ return
175
+ runs = [json.loads(line) for line in RESULTS.read_text(encoding="utf-8").splitlines() if line.strip()]
176
+ runs.sort(key=lambda r: (-r["score"], r["seconds"]))
177
+ print(f"\n{BOLD}Scoreboard{RESET} {DIM}({RESULTS}){RESET}")
178
+ print(f" {'model':<48} {'agent':<12} {'score':>7} {'time':>8} date")
179
+ for r in runs:
180
+ color = GREEN if r["score"] >= 7 else YELLOW if r["score"] >= 5 else RED
181
+ print(f" {r['model'][:48]:<48} {r['agent']:<12} {color}{r['score']:>5.1f}/8{RESET} {r['seconds'] / 60:>6.1f} m {r['date']}")
182
+
183
+
184
+ def main(argv=None):
185
+ parser = argparse.ArgumentParser(prog="hearthwork bench", description="Benchmark the running model through a coding agent.")
186
+ parser.add_argument("--agent", choices=sorted(HARNESSES), default="claude")
187
+ parser.add_argument("--show", action="store_true", help="only print the scoreboard")
188
+ args = parser.parse_args(argv)
189
+ if args.show:
190
+ return scoreboard()
191
+ config = load_config()
192
+ model = served_model(config["server"]["port"])
193
+ if not model:
194
+ sys.exit("No model is running. Start one first (hearthwork menu: Start / switch model).")
195
+ if not installed(args.agent):
196
+ sys.exit(f"{HARNESSES[args.agent]['title']} is not installed.")
197
+ stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
198
+ folder = BENCH / "runs" / f"{stamp}-{args.agent}-{model}"
199
+ folder.mkdir(parents=True)
200
+ (folder / "demo.txt").write_text(DEMO, encoding="utf-8")
201
+ (folder / "buggy.py").write_text(BUGGY, encoding="utf-8")
202
+ title = HARNESSES[args.agent]["title"]
203
+ print(f"\n{BOLD}Benchmark{RESET}: {model} via {title} {DIM}(8 prompts, one conversation; folder {folder}){RESET}\n")
204
+
205
+ steps, session, started = [], None, time.time()
206
+ for i, (name, prompt) in enumerate(PROMPTS):
207
+ print(f" {i + 1}. {name:<12}", end=" ", flush=True)
208
+ t0 = time.time()
209
+ try:
210
+ reply, session = ask_agent(args.agent, config, model, folder, prompt, i == 0, session)
211
+ except subprocess.TimeoutExpired:
212
+ reply = ""
213
+ except AgentError as error:
214
+ print(f"\n\n {RED}{error}{RESET}\n Not recorded: this is an agent problem, not a model score.")
215
+ sys.exit(1)
216
+ seconds = time.time() - t0
217
+ points, note = grade(i, folder, reply)
218
+ color = GREEN if points == 1 else YELLOW if points > 0 else RED
219
+ print(f"{color}{points:>4.2g}{RESET} {seconds:6.0f} s {DIM}{note}{RESET}")
220
+ steps.append({"step": name, "points": points, "seconds": round(seconds, 1), "note": note})
221
+
222
+ total = sum(s["points"] for s in steps)
223
+ seconds = time.time() - started
224
+ record = {"date": datetime.now().strftime("%Y-%m-%d %H:%M"), "model": model, "agent": title,
225
+ "score": round(total, 2), "seconds": round(seconds), "steps": steps, "folder": str(folder)}
226
+ BENCH.mkdir(exist_ok=True)
227
+ with open(RESULTS, "a", encoding="utf-8") as f:
228
+ f.write(json.dumps(record) + "\n")
229
+ print(f"\n {BOLD}Score {total:.1f}/8{RESET} in {seconds / 60:.1f} min")
230
+ scoreboard()
231
+
232
+
233
+ if __name__ == "__main__":
234
+ try:
235
+ main()
236
+ except KeyboardInterrupt:
237
+ print("\nStopped.")
hearthwork/check.py ADDED
@@ -0,0 +1,189 @@
1
+ #!/usr/bin/env python3
2
+ """Hearthwork check: will a local model for coding agents be worth it on this computer, and which one?
3
+
4
+ Run it without installing anything:
5
+ uvx --from git+https://github.com/BillyMRX1/hearthwork hearthwork-check
6
+ or, once installed: hearthwork check (--json for scripts)
7
+
8
+ Read-only: it checks the hardware and installed agents, gives the same verdict Hearthwork's setup gives, and
9
+ looks up live file sizes on Hugging Face to pick the best version of each suggested model that fits.
10
+ """
11
+ import argparse
12
+ import json
13
+ import platform
14
+ import re
15
+ import shutil
16
+ import subprocess
17
+ import sys
18
+ import urllib.request
19
+ from pathlib import Path
20
+
21
+ from .onboard import CYAN, GREEN, RED, RESET, WINDOWS, YELLOW, detect, judge, lmstudio_folders, memory_gb
22
+
23
+ BOLD, DIM = "\033[1m", "\033[2m"
24
+ REPO = "https://github.com/BillyMRX1/hearthwork"
25
+
26
+ # Coding models with tool-calling chat templates (checked on Hugging Face, 2026-10). "moe": only a few experts run
27
+ # per token, so the model stays usable when part of it sits in system RAM. Measured on the reference PC
28
+ # (RTX 5060 Ti 16 GB + 22.6 GB RAM, 64K context): MoE Qwen3-Coder-30B partly in RAM 18-31 tokens/s; dense
29
+ # Qwen3.8-27B partly in RAM ~6 tokens/s.
30
+ # "tested": Hearthwork benchmark (bench.py, 8 graded coding tasks) on the reference PC, Q4_K_M, 2026-10-07.
31
+ MODELS = [
32
+ {"repo": "unsloth/Qwen3.5-35B-A3B-GGUF", "name": "Qwen3.5-35B-A3B", "moe": True, "tested": True,
33
+ "note": "best in the benchmark: 8/8 with Codex (2.6 min) and Claude Code (3.6 min)"},
34
+ {"repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF", "name": "Qwen3-Coder-30B-A3B", "moe": True, "tested": True,
35
+ "note": "8/8 with Claude Code (3.9 min) and Codex (4.2 min); a bit smaller"},
36
+ {"repo": "unsloth/GLM-4.7-Flash-GGUF", "name": "GLM-4.7-Flash (30B MoE)", "moe": True, "tested": True,
37
+ "note": "8/8 with Codex (4.5 min), 7.7/8 with Claude Code (missed one bug)"},
38
+ {"repo": "ggml-org/gpt-oss-20b-GGUF", "name": "gpt-oss-20b", "moe": True, "note": "small and fast; good at tool use"},
39
+ {"repo": "unsloth/Devstral-Small-2-24B-Instruct-2512-GGUF", "name": "Devstral-Small-2-24B", "moe": False,
40
+ "note": "coding agent (dense: needs to fit the GPU to be fast)"},
41
+ {"repo": "unsloth/Qwen3.5-27B-GGUF", "name": "Qwen3.5-27B", "moe": False, "note": "dense"},
42
+ {"repo": "unsloth/Qwen3-Coder-Next-GGUF", "name": "Qwen3-Coder-Next (80B MoE)", "moe": True,
43
+ "note": "strongest here; needs a big machine"},
44
+ {"repo": "lmstudio-community/Qwen3.5-9B-GGUF", "name": "Qwen3.5-9B", "moe": False,
45
+ "note": "small machines only; much weaker with agents' tools"},
46
+ ]
47
+ # Best quality first. Below Q3, answers get noticeably worse.
48
+ QUANTS = ["Q4_K_M", "UD-Q4_K_XL", "MXFP4", "Q4_K_S", "Q3_K_M", "UD-Q3_K_XL", "Q3_K_S", "UD-Q2_K_XL", "Q2_K"]
49
+ KV_GB_64K = 3.4 # KV cache for a 64K context at q8_0, measured with Qwen3-Coder-30B (varies by model)
50
+
51
+
52
+ def version(binary):
53
+ path = shutil.which(binary)
54
+ if not path:
55
+ return None
56
+ try:
57
+ out = subprocess.run([path, "--version"], capture_output=True, text=True, timeout=20).stdout.strip()
58
+ return out.splitlines()[0] if out else "installed"
59
+ except (OSError, subprocess.SubprocessError):
60
+ return "installed"
61
+
62
+
63
+ def quant_sizes(repo):
64
+ """{quant: GB} from Hugging Face (multi-part files summed); None when offline."""
65
+ url = f"https://huggingface.co/api/models/{repo}/tree/main?recursive=true"
66
+ try:
67
+ with urllib.request.urlopen(urllib.request.Request(url, headers={"User-Agent": "hearthwork-check"}), timeout=20) as r:
68
+ files = json.load(r)
69
+ except Exception:
70
+ return None
71
+ sizes = {}
72
+ for f in files:
73
+ path = f.get("path", "")
74
+ if not path.endswith(".gguf") or "mmproj" in path.lower():
75
+ continue
76
+ name = re.sub(r"-\d{5}-of-\d{5}", "", Path(path).name)
77
+ for quant in QUANTS:
78
+ if name.upper().endswith(f"-{quant.upper()}.GGUF") or name.upper().endswith(f"_{quant.upper()}.GGUF"):
79
+ sizes[quant] = sizes.get(quant, 0) + (f.get("lfs") or {}).get("size", f.get("size", 0)) / 2**30
80
+ return sizes
81
+
82
+
83
+ def place(model, sizes, gpu_room, ram_room):
84
+ """Best quantization for this machine -> (quant, GB, fit) with fit in fast / good / slow, or None if too big."""
85
+ if model["moe"]: # MoE stays fast partly in RAM: best quality that fits at all
86
+ for quant in QUANTS:
87
+ if quant in sizes and sizes[quant] <= gpu_room + ram_room:
88
+ return quant, sizes[quant], "fast" if sizes[quant] <= gpu_room else "good"
89
+ return None
90
+ for quant in QUANTS: # dense: fitting the GPU entirely matters most
91
+ if quant in sizes and sizes[quant] <= gpu_room:
92
+ return quant, sizes[quant], "fast"
93
+ for quant in QUANTS: # then GPU + RAM
94
+ if quant in sizes and sizes[quant] <= gpu_room + ram_room:
95
+ return quant, sizes[quant], "good" if model["moe"] else "slow"
96
+ return None
97
+
98
+
99
+ def main(argv=None):
100
+ parser = argparse.ArgumentParser(prog="hearthwork check", description="Will Hearthwork (local models for coding agents) be worth it here?")
101
+ parser.add_argument("--json", action="store_true", help="machine-readable output")
102
+ args = parser.parse_args(argv)
103
+
104
+ hw = detect()
105
+ verdict, reasons, expect = judge(hw)
106
+ ram_total, ram_free = memory_gb()
107
+ home_disk = shutil.disk_usage(Path.home()).free / 2**30
108
+ agents = {"Claude Code": version("claude"), "Codex": version("codex")}
109
+ vram = hw.get("vramGB") or 0
110
+ if hw["backend"] == "metal":
111
+ gpu_room, ram_room = max(0.0, hw["ramGB"] * 0.75 - 1.5 - KV_GB_64K), 0.0
112
+ else:
113
+ gpu_room = max(0.0, vram - 1.5 - KV_GB_64K) if hw["backend"] in ("cuda", "vulkan") else 0.0
114
+ ram_room = max(0.0, hw["ramGB"] - 6)
115
+
116
+ picks = []
117
+ offline = False
118
+ for model in MODELS:
119
+ sizes = quant_sizes(model["repo"])
120
+ if sizes is None:
121
+ offline = True
122
+ continue
123
+ fit = place(model, sizes, gpu_room, ram_room)
124
+ picks.append({**model, "quant": fit[0] if fit else None, "sizeGB": round(fit[1], 1) if fit else None,
125
+ "fit": fit[2] if fit else "too big"})
126
+ rank = {"fast": 0, "good": 1, "slow": 2, "too big": 3}
127
+ picks.sort(key=lambda p: (p["fit"] == "too big", not p.get("tested"), rank[p["fit"]], MODELS.index(next(m for m in MODELS if m["repo"] == p["repo"]))))
128
+
129
+ if args.json:
130
+ print(json.dumps({"hardware": hw, "ramFreeGB": round(ram_free, 1), "diskFreeGB": round(home_disk, 1),
131
+ "verdict": verdict, "reasons": reasons, "expect": expect, "agents": agents,
132
+ "models": picks, "offline": offline}, indent=2))
133
+ return
134
+
135
+ print(f"\n{BOLD}Hearthwork check{RESET} {DIM}(local models for Claude Code / Codex){RESET}\n")
136
+ print(f"{BOLD}Your computer{RESET}")
137
+ print(f" OS {platform.system()} {platform.release()} ({hw['arch']}) Python {platform.python_version()}")
138
+ print(f" CPU {hw['cpuThreads']} threads")
139
+ print(f" RAM {ram_total:.1f} GB ({ram_free:.1f} GB free now)")
140
+ gpus = ", ".join(hw["gpus"]) or "no usable GPU"
141
+ print(f" GPU {gpus}" + (f" ({vram:.1f} GB)" if hw["backend"] == "vulkan" and vram else "")
142
+ + f" -> llama.cpp {hw['backend']}" + (f", driver CUDA {hw['cudaDriver']}" if hw.get("cudaDriver") else ""))
143
+ print(f" Disk {home_disk:.0f} GB free in your home drive"
144
+ + (f"; LM Studio models: {lmstudio_folders()[0]}" if lmstudio_folders() else ""))
145
+
146
+ color = {"recommended": GREEN, "limited": YELLOW}.get(verdict, RED)
147
+ print(f"\n{BOLD}Verdict{RESET} {color}{BOLD}{verdict.upper()}{RESET}")
148
+ for reason in reasons:
149
+ print(f" - {reason}")
150
+ print(f" {expect}")
151
+
152
+ print(f"\n{BOLD}Coding agents{RESET}")
153
+ for name, found in agents.items():
154
+ print(f" {name:12} " + (f"{GREEN}{found}{RESET}" if found else f"{YELLOW}not installed{RESET}"))
155
+ if not any(agents.values()):
156
+ print(f" {YELLOW}Install at least one: Claude Code (https://code.claude.com) or Codex "
157
+ f"(https://developers.openai.com/codex).{RESET}")
158
+
159
+ print(f"\n{BOLD}Models for this computer{RESET} {DIM}(best version that fits; sizes live from Hugging Face){RESET}")
160
+ labels = {"fast": f"{GREEN}fits the GPU: fast{RESET}", "good": f"{GREEN}GPU + RAM: good (MoE){RESET}",
161
+ "slow": f"{YELLOW}GPU + RAM: slow (dense){RESET}", "too big": f"{RED}too big{RESET}"}
162
+ for p in picks:
163
+ size = f"{p['quant']:<10} {p['sizeGB']:5.1f} GB" if p["quant"] else " " * 19
164
+ tested = f" {CYAN}[tested]{RESET}" if p.get("tested") else ""
165
+ print(f" {p['name']:<28} {size} {labels[p['fit']]}{tested}")
166
+ print(f" {DIM}{'':<28} {p['note']}{RESET}")
167
+ if offline:
168
+ print(f" {YELLOW}Some models could not be looked up (offline?).{RESET}")
169
+ print(f" {DIM}Measured on an RTX 5060 Ti 16 GB + 22.6 GB RAM: MoE partly in RAM 18-31 tokens/s, "
170
+ f"dense partly in RAM ~6 tokens/s.{RESET}")
171
+
172
+ best = next((p for p in picks if p["fit"] in ("fast", "good")), None)
173
+ print(f"\n{BOLD}Next step{RESET}")
174
+ if verdict == "not recommended" or not best:
175
+ print(" A local model would be too slow or too weak for coding on this computer. Cloud Claude Code / Codex is the"
176
+ " better choice here.")
177
+ else:
178
+ print(f" 1. Install Hearthwork: {CYAN}uv tool install hearthwork{RESET} (or: pipx install hearthwork)")
179
+ print(f" 2. Download a model: {CYAN}hearthwork model {best['repo']}{RESET} "
180
+ f"(pick {best['quant']}, {best['sizeGB']} GB)")
181
+ print(f" 3. From your project folder, run {CYAN}hearthwork{RESET}: setup takes a minute, then pick your agent.")
182
+ print()
183
+
184
+
185
+ if __name__ == "__main__":
186
+ try:
187
+ main()
188
+ except KeyboardInterrupt:
189
+ sys.exit(130)
hearthwork/cli.py ADDED
@@ -0,0 +1,133 @@
1
+ """The `hearthwork` command.
2
+
3
+ hearthwork menu (run it from the project you want the agent to work on)
4
+ hearthwork claude [args...] Claude Code with the local model (starts one if needed); args go to `claude`
5
+ hearthwork codex [args...] Codex with the local model; args go to `codex`
6
+ hearthwork start [--model X] start the model in the background hearthwork stop
7
+ hearthwork serve [--model X] run the model server in this terminal (Ctrl+C stops it)
8
+ hearthwork status what is running
9
+ hearthwork model <link> download a GGUF model from Hugging Face
10
+ hearthwork bench 8 graded coding tasks through an agent, with a scoreboard
11
+ hearthwork check is this computer suited, and which models fit
12
+ hearthwork setup hardware check, llama.cpp download/update, models folder
13
+ hearthwork update update Hearthwork itself
14
+ """
15
+ import shutil
16
+ import subprocess
17
+ import sys
18
+
19
+ from . import __version__
20
+ from .harnesses import HARNESSES
21
+ from .onboard import CYAN, GREEN, RESET, load_config, offer_import, onboard, setup_complete
22
+ from .paths import HOME
23
+
24
+ REPO = "git+https://github.com/BillyMRX1/hearthwork"
25
+ USAGE = __doc__.split("\n", 2)[2]
26
+
27
+
28
+ def configured(interactive=True):
29
+ """config, running first-time setup (or importing an older setup) when needed."""
30
+ config = load_config()
31
+ if setup_complete(config):
32
+ return config
33
+ config = offer_import(config)
34
+ if setup_complete(config):
35
+ return config
36
+ config = onboard(config)
37
+ if interactive:
38
+ input("\nPress Enter to continue...")
39
+ return config
40
+
41
+
42
+ def model_args(argv, name):
43
+ import argparse
44
+ parser = argparse.ArgumentParser(prog=f"hearthwork {name}")
45
+ parser.add_argument("--model", help="part of a model file name; skips the model menu")
46
+ parser.add_argument("--context", type=int, help="context size (default: chosen by setup)")
47
+ return parser.parse_args(argv)
48
+
49
+
50
+ def update():
51
+ """Upgrade with whichever tool installed Hearthwork."""
52
+ if shutil.which("uv") and "uv" in sys.executable.replace("\\", "/").split("/"):
53
+ return subprocess.call(["uv", "tool", "upgrade", "hearthwork"])
54
+ if shutil.which("pipx") and "pipx" in sys.executable.replace("\\", "/"):
55
+ return subprocess.call(["pipx", "upgrade", "hearthwork"])
56
+ print(f"Upgrade with the tool you installed Hearthwork with, e.g.\n uv tool upgrade hearthwork\n"
57
+ f" pipx upgrade hearthwork\n pip install --upgrade hearthwork")
58
+ return 0
59
+
60
+
61
+ def main(argv=None):
62
+ argv = list(sys.argv[1:] if argv is None else argv)
63
+ command, rest = (argv[0], argv[1:]) if argv else ("menu", [])
64
+ try:
65
+ if command in ("-h", "--help", "help"):
66
+ print(USAGE)
67
+ elif command in ("-V", "--version", "version"):
68
+ print(f"hearthwork {__version__} (data: {HOME})")
69
+ elif command == "menu":
70
+ from . import menu
71
+ menu.main(configured())
72
+ elif command in HARNESSES: # every following argument belongs to the agent (e.g. -p "...")
73
+ from .menu import run_agent
74
+ sys.exit(run_agent(configured(), command, rest, back_to_menu=False))
75
+ elif command in ("start", "serve"):
76
+ from .server import choose_and_remember, run_foreground, start_background
77
+ args = model_args(rest, command)
78
+ config = configured()
79
+ model = choose_and_remember(config, args.model)
80
+ if not model:
81
+ sys.exit(1)
82
+ if command == "start":
83
+ ok = start_background(config, model, args.context)
84
+ if ok:
85
+ print(f"{CYAN}Now run `hearthwork claude` or `hearthwork codex` from your project.{RESET}")
86
+ sys.exit(0 if ok else 1)
87
+ print(f"\nStarting {model.name} context: {args.context or config['server']['context']} "
88
+ f"port: {config['server']['port']}")
89
+ print(f"{CYAN}When it says 'listening on', run `hearthwork claude` or `hearthwork codex` "
90
+ f"from your project.{RESET}\n", flush=True)
91
+ run_foreground(config, model, args.context)
92
+ elif command == "stop":
93
+ from .server import stop
94
+ stop(load_config())
95
+ elif command == "status":
96
+ from .server import served_model
97
+ config = load_config()
98
+ port = config.get("server", {}).get("port", 8001)
99
+ running = served_model(port)
100
+ print(f"hearthwork {__version__} data: {HOME}")
101
+ print(f"model: {GREEN + running + RESET if running else 'not running'}" + (f" (port {port})" if running else ""))
102
+ print(f"models folder: {config.get('modelsDir', '-')}")
103
+ elif command == "model":
104
+ from . import model
105
+ configured()
106
+ model.main(rest)
107
+ elif command == "bench":
108
+ from . import bench
109
+ if "--show" not in rest:
110
+ from .menu import ensure_server
111
+ if not ensure_server(configured()):
112
+ sys.exit(1)
113
+ bench.main(rest)
114
+ elif command == "check":
115
+ from . import check
116
+ check.main(rest)
117
+ elif command == "setup":
118
+ from . import onboard as setup
119
+ setup.main(rest)
120
+ elif command == "update":
121
+ sys.exit(update())
122
+ else:
123
+ print(f"Unknown command: {command}\n\n{USAGE}")
124
+ sys.exit(2)
125
+ except KeyboardInterrupt:
126
+ print()
127
+ sys.exit(130)
128
+
129
+
130
+ def check_main():
131
+ """`hearthwork-check`: the checker on its own, e.g. `uvx --from git+https://github.com/BillyMRX1/hearthwork hearthwork-check`."""
132
+ from . import check
133
+ check.main()