lcode-cli 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lcode/__init__.py +3 -0
- lcode/__main__.py +3 -0
- lcode/agent.py +376 -0
- lcode/catalog.py +114 -0
- lcode/cli.py +437 -0
- lcode/config.py +135 -0
- lcode/hardware.py +84 -0
- lcode/models.toml +128 -0
- lcode/ollama.py +163 -0
- lcode/permissions.py +90 -0
- lcode/render.py +43 -0
- lcode/repl.py +281 -0
- lcode/tools.py +533 -0
- lcode_cli-0.1.1.dist-info/METADATA +181 -0
- lcode_cli-0.1.1.dist-info/RECORD +18 -0
- lcode_cli-0.1.1.dist-info/WHEEL +4 -0
- lcode_cli-0.1.1.dist-info/entry_points.txt +2 -0
- lcode_cli-0.1.1.dist-info/licenses/LICENSE +22 -0
lcode/__init__.py
ADDED
lcode/__main__.py
ADDED
lcode/agent.py
ADDED
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
"""The agent loop: stream a model turn, run the tools it calls, repeat until it answers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime as dt
|
|
6
|
+
import json
|
|
7
|
+
import platform
|
|
8
|
+
import re
|
|
9
|
+
import shutil
|
|
10
|
+
import subprocess
|
|
11
|
+
import time
|
|
12
|
+
import uuid
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from rich.console import Console
|
|
17
|
+
from rich.markdown import Markdown
|
|
18
|
+
from rich.panel import Panel
|
|
19
|
+
from rich.text import Text
|
|
20
|
+
|
|
21
|
+
from lcode import catalog
|
|
22
|
+
from lcode.config import STATE_DIR, format_tokens
|
|
23
|
+
from lcode.ollama import Ollama, OllamaError
|
|
24
|
+
from lcode.permissions import Permissions
|
|
25
|
+
from lcode.render import MarkdownStreamer
|
|
26
|
+
from lcode.tools import SCHEMAS, Toolbox, is_binary, parse_text_tool_calls, tree, truncate
|
|
27
|
+
|
|
28
|
+
MAX_STEPS_PER_TURN = 150 # safety cap on tool-call iterations for one request
|
|
29
|
+
AUTO_COMPACT_RATIO = 0.85 # summarize the history when the context is this full
|
|
30
|
+
PROJECT_FILES = ("AGENTS.md", "LCODE.md", "CLAUDE.md")
|
|
31
|
+
|
|
32
|
+
SYSTEM_PROMPT = """You are lcode, an autonomous software-engineering agent running in the user's terminal on their own machine, powered by the local model {model}. You help the user understand codebases, answer questions about code, write scripts (Python by default), fix bugs, refactor, and run commands.
|
|
33
|
+
|
|
34
|
+
# How to work
|
|
35
|
+
- Use your tools to gather facts. Never guess file contents, APIs, or command output — read, search, or run them.
|
|
36
|
+
- To understand a repository: look at the layout, README/docs, entry points and config, then grep and read the relevant files. Cite code as `path:line`.
|
|
37
|
+
- Read a file before editing it. Use edit_file for targeted changes and write_file for new files or full rewrites. old_string must be copied exactly from the file (without the line-number prefixes).
|
|
38
|
+
- After writing or changing code, verify it: run the script, the tests, or at least a syntax check (e.g. `{python} -m py_compile file.py`). Fix what fails.
|
|
39
|
+
- Match the existing code style, naming and structure. Don't add unrequested features.
|
|
40
|
+
- For multi-step tasks, plan with todo_write and keep it updated.
|
|
41
|
+
- Work autonomously until the task is done. Only stop to ask the user when you are genuinely blocked or the decision is theirs.
|
|
42
|
+
- Don't run destructive or irreversible commands (rm -rf, git reset --hard, git push --force, dropping data) unless the user explicitly asked.
|
|
43
|
+
- Never claim something works without having run it. If something fails, say so and show the relevant error.
|
|
44
|
+
- If the user denies a tool call, don't retry the same thing; adjust or ask.
|
|
45
|
+
|
|
46
|
+
# Communication
|
|
47
|
+
- Be concise and direct; no filler. Use GitHub-flavored markdown.
|
|
48
|
+
- When you finish, give a short summary of what you found or changed and anything the user needs to do.
|
|
49
|
+
|
|
50
|
+
# Environment
|
|
51
|
+
- Working directory: {cwd}
|
|
52
|
+
- OS: {os}; shell: bash; Python command: `{python}`
|
|
53
|
+
- Date: {date}
|
|
54
|
+
- Git: {git}
|
|
55
|
+
|
|
56
|
+
# Top-level layout of the working directory
|
|
57
|
+
{tree}
|
|
58
|
+
{memory}"""
|
|
59
|
+
|
|
60
|
+
INIT_PROMPT = """Analyze this repository and create (or improve, if it exists) an AGENTS.md file at its root that will be given to you in future sessions. Explore the codebase first (layout, README, config/build files, entry points, main modules, tests). AGENTS.md should contain:
|
|
61
|
+
1. A short overview of what the project does.
|
|
62
|
+
2. How to set up, build, run, and test it (exact commands).
|
|
63
|
+
3. The architecture: the important directories/modules and how they fit together, with key file paths.
|
|
64
|
+
4. Code conventions and any gotchas you noticed.
|
|
65
|
+
Keep it concise (under ~150 lines) and factual — only include what you verified in the code."""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def git_info(cwd: Path) -> str:
|
|
69
|
+
def git(*args: str) -> subprocess.CompletedProcess:
|
|
70
|
+
return subprocess.run(["git", *args], cwd=cwd, capture_output=True, text=True, timeout=5)
|
|
71
|
+
|
|
72
|
+
try:
|
|
73
|
+
branch = git("rev-parse", "--abbrev-ref", "HEAD")
|
|
74
|
+
if branch.returncode != 0:
|
|
75
|
+
return "not a git repository"
|
|
76
|
+
changed = git("status", "--short").stdout.strip().splitlines()
|
|
77
|
+
log = git("log", "--oneline", "-5").stdout.strip() or "(no commits)"
|
|
78
|
+
return f"branch {branch.stdout.strip()}, {len(changed)} changed file(s)\nRecent commits:\n{log}"
|
|
79
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
80
|
+
return "unknown"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class Settings:
|
|
85
|
+
model: str # Ollama model name to call
|
|
86
|
+
context: int
|
|
87
|
+
num_batch: int | None = None
|
|
88
|
+
keep_alive: str = "30m"
|
|
89
|
+
think: bool = True
|
|
90
|
+
show_thinking: bool = False
|
|
91
|
+
permission_mode: str = "ask"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class Agent:
|
|
95
|
+
def __init__(self, ollama: Ollama, settings: Settings, cwd: Path, console: Console | None = None):
|
|
96
|
+
self.console = console or Console(highlight=False)
|
|
97
|
+
self.ollama = ollama
|
|
98
|
+
self.settings = settings
|
|
99
|
+
self.cwd = cwd.resolve()
|
|
100
|
+
self.perms = Permissions(self.console, settings.permission_mode)
|
|
101
|
+
self.tools = Toolbox(self)
|
|
102
|
+
self.session_id = self.new_session_id()
|
|
103
|
+
self.ctx_used = 0
|
|
104
|
+
self.last_speed = 0.0
|
|
105
|
+
self.messages: list[dict] = []
|
|
106
|
+
self.reset()
|
|
107
|
+
|
|
108
|
+
# -- prompt and sessions
|
|
109
|
+
@staticmethod
|
|
110
|
+
def new_session_id() -> str:
|
|
111
|
+
return dt.datetime.now().strftime("%Y%m%d-%H%M%S-") + uuid.uuid4().hex[:6]
|
|
112
|
+
|
|
113
|
+
def system_prompt(self) -> str:
|
|
114
|
+
memory = ""
|
|
115
|
+
for name in PROJECT_FILES:
|
|
116
|
+
f = self.cwd / name
|
|
117
|
+
if f.is_file():
|
|
118
|
+
memory = f"\n# Project instructions ({name})\n{truncate(f.read_text(errors='replace'), 20_000)}\n"
|
|
119
|
+
break
|
|
120
|
+
return SYSTEM_PROMPT.format(
|
|
121
|
+
model=self.settings.model,
|
|
122
|
+
cwd=self.cwd,
|
|
123
|
+
os=f"{platform.system()} {platform.release()} ({platform.machine()})",
|
|
124
|
+
python="python" if shutil.which("python") else "python3",
|
|
125
|
+
date=dt.date.today().isoformat(),
|
|
126
|
+
git=git_info(self.cwd),
|
|
127
|
+
tree=tree(self.cwd, depth=1, limit=120),
|
|
128
|
+
memory=memory,
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
def reset(self) -> None:
|
|
132
|
+
self.messages = [{"role": "system", "content": self.system_prompt()}]
|
|
133
|
+
self.tools.read_mtimes.clear()
|
|
134
|
+
self.ctx_used = len(self.messages[0]["content"]) // 3
|
|
135
|
+
|
|
136
|
+
def session_file(self) -> Path:
|
|
137
|
+
return STATE_DIR / "sessions" / f"{self.session_id}.json"
|
|
138
|
+
|
|
139
|
+
def save(self) -> None:
|
|
140
|
+
f = self.session_file()
|
|
141
|
+
f.parent.mkdir(parents=True, exist_ok=True)
|
|
142
|
+
f.write_text(json.dumps({"cwd": str(self.cwd), "model": self.settings.model, "messages": self.messages}))
|
|
143
|
+
|
|
144
|
+
def load_latest(self) -> bool:
|
|
145
|
+
for f in sorted((STATE_DIR / "sessions").glob("*.json"), reverse=True):
|
|
146
|
+
try:
|
|
147
|
+
data = json.loads(f.read_text())
|
|
148
|
+
except (OSError, json.JSONDecodeError):
|
|
149
|
+
continue
|
|
150
|
+
if data.get("cwd") == str(self.cwd):
|
|
151
|
+
self.messages = [{"role": "system", "content": self.system_prompt()}, *data["messages"][1:]]
|
|
152
|
+
self.session_id = f.stem
|
|
153
|
+
self.ctx_used = sum(len(json.dumps(m)) for m in self.messages) // 3
|
|
154
|
+
return True
|
|
155
|
+
return False
|
|
156
|
+
|
|
157
|
+
# -- model calls
|
|
158
|
+
def options(self) -> dict:
|
|
159
|
+
opts: dict = {"num_ctx": self.settings.context}
|
|
160
|
+
if self.settings.num_batch:
|
|
161
|
+
opts["num_batch"] = self.settings.num_batch
|
|
162
|
+
return opts
|
|
163
|
+
|
|
164
|
+
def chat(self, messages: list[dict], tools: list | None, think: bool):
|
|
165
|
+
payload = {
|
|
166
|
+
"model": self.settings.model,
|
|
167
|
+
"messages": messages,
|
|
168
|
+
"think": think,
|
|
169
|
+
"keep_alive": self.settings.keep_alive,
|
|
170
|
+
"options": self.options(),
|
|
171
|
+
}
|
|
172
|
+
if tools:
|
|
173
|
+
payload["tools"] = tools
|
|
174
|
+
try:
|
|
175
|
+
yield from self.ollama.chat_stream(payload)
|
|
176
|
+
except OllamaError as e:
|
|
177
|
+
if think and "think" in str(e) and "support" in str(e):
|
|
178
|
+
self.settings.think = False
|
|
179
|
+
self.console.print(f"[dim]{self.settings.model} does not support reasoning; continuing without.[/]")
|
|
180
|
+
yield from self.ollama.chat_stream({**payload, "think": False})
|
|
181
|
+
return
|
|
182
|
+
if "out of memory" in str(e).lower():
|
|
183
|
+
raise OllamaError(
|
|
184
|
+
f"{e}\nThe model ran out of GPU memory. Try a smaller context (/ctx 128k) or a smaller "
|
|
185
|
+
"prompt batch (`lcode config set num_batch 512`)."
|
|
186
|
+
) from e
|
|
187
|
+
raise
|
|
188
|
+
|
|
189
|
+
def assistant_step(self) -> dict:
|
|
190
|
+
"""Stream one model response, rendering thinking/content live. Returns the assistant message."""
|
|
191
|
+
content, thinking, tool_calls, final = "", "", [], {}
|
|
192
|
+
md = MarkdownStreamer(self.console)
|
|
193
|
+
status = self.console.status("[cyan]Thinking…[/]", spinner="dots")
|
|
194
|
+
status.start()
|
|
195
|
+
spinning, printed_thinking, t0 = True, False, time.time()
|
|
196
|
+
|
|
197
|
+
def stop_spinner() -> None:
|
|
198
|
+
nonlocal spinning
|
|
199
|
+
if spinning:
|
|
200
|
+
status.stop()
|
|
201
|
+
spinning = False
|
|
202
|
+
|
|
203
|
+
try:
|
|
204
|
+
for chunk in self.chat(self.messages, SCHEMAS, self.settings.think):
|
|
205
|
+
msg = chunk.get("message", {})
|
|
206
|
+
if msg.get("thinking"):
|
|
207
|
+
thinking += msg["thinking"]
|
|
208
|
+
if self.settings.show_thinking:
|
|
209
|
+
stop_spinner()
|
|
210
|
+
self.console.print(Text(msg["thinking"], style="dim italic"), end="")
|
|
211
|
+
printed_thinking = True
|
|
212
|
+
else:
|
|
213
|
+
status.update(
|
|
214
|
+
f"[cyan]Thinking…[/] [dim]({len(thinking) // 4} tokens, {time.time() - t0:.0f}s)[/]"
|
|
215
|
+
)
|
|
216
|
+
if msg.get("content"):
|
|
217
|
+
stop_spinner()
|
|
218
|
+
if printed_thinking:
|
|
219
|
+
self.console.print("\n")
|
|
220
|
+
printed_thinking = False
|
|
221
|
+
content += msg["content"]
|
|
222
|
+
md.feed(msg["content"])
|
|
223
|
+
if msg.get("tool_calls"):
|
|
224
|
+
tool_calls.extend(msg["tool_calls"])
|
|
225
|
+
if chunk.get("done"):
|
|
226
|
+
final = chunk
|
|
227
|
+
except KeyboardInterrupt:
|
|
228
|
+
stop_spinner()
|
|
229
|
+
md.flush()
|
|
230
|
+
self.console.print("[yellow]⏹ Interrupted[/]")
|
|
231
|
+
self.messages.append({"role": "assistant", "content": content + "\n[interrupted by user]"})
|
|
232
|
+
raise
|
|
233
|
+
finally:
|
|
234
|
+
stop_spinner()
|
|
235
|
+
if printed_thinking:
|
|
236
|
+
self.console.print()
|
|
237
|
+
md.flush()
|
|
238
|
+
if not tool_calls and ("<tool_call>" in content or '"name"' in content):
|
|
239
|
+
tool_calls = parse_text_tool_calls(content)
|
|
240
|
+
message: dict = {"role": "assistant", "content": content}
|
|
241
|
+
if thinking:
|
|
242
|
+
message["thinking"] = thinking
|
|
243
|
+
if tool_calls:
|
|
244
|
+
message["tool_calls"] = tool_calls
|
|
245
|
+
self.messages.append(message)
|
|
246
|
+
if final:
|
|
247
|
+
self.ctx_used = final.get("prompt_eval_count", 0) + final.get("eval_count", 0)
|
|
248
|
+
seconds = final.get("eval_duration", 0) / 1e9
|
|
249
|
+
self.last_speed = final.get("eval_count", 0) / seconds if seconds else 0.0
|
|
250
|
+
return message
|
|
251
|
+
|
|
252
|
+
def describe_call(self, name: str, args: dict) -> str:
|
|
253
|
+
if name in ("read_file", "write_file", "edit_file"):
|
|
254
|
+
path = self.tools.rel(self.tools.resolve(args.get("path", "")))
|
|
255
|
+
offset = args.get("offset")
|
|
256
|
+
return f"{name}({path})" + (
|
|
257
|
+
f" from line {offset}" if name == "read_file" and offset not in (None, 1) else ""
|
|
258
|
+
)
|
|
259
|
+
if name == "grep":
|
|
260
|
+
where = args.get("path", ".") + (f" {args['glob']}" if args.get("glob") else "")
|
|
261
|
+
return f"grep({args.get('pattern', '')!r} in {where})"
|
|
262
|
+
if name == "glob":
|
|
263
|
+
return f"glob({args.get('pattern', '')})"
|
|
264
|
+
if name == "list_dir":
|
|
265
|
+
return f"list_dir({args.get('path', '.')})"
|
|
266
|
+
return name
|
|
267
|
+
|
|
268
|
+
def run_turn(self, user_text: str) -> None:
|
|
269
|
+
self.messages.append({"role": "user", "content": self.expand_mentions(user_text)})
|
|
270
|
+
for _ in range(MAX_STEPS_PER_TURN):
|
|
271
|
+
self.maybe_compact()
|
|
272
|
+
calls = self.assistant_step().get("tool_calls") or []
|
|
273
|
+
if not calls:
|
|
274
|
+
break
|
|
275
|
+
for i, call in enumerate(calls):
|
|
276
|
+
fn = call.get("function", {})
|
|
277
|
+
name, args = fn.get("name", ""), fn.get("arguments") or {}
|
|
278
|
+
if isinstance(args, str):
|
|
279
|
+
try:
|
|
280
|
+
args = json.loads(args)
|
|
281
|
+
except json.JSONDecodeError:
|
|
282
|
+
args = {}
|
|
283
|
+
if name not in ("bash", "todo_write"): # those print their own header
|
|
284
|
+
self.console.print(Text(f"● {self.describe_call(name, args)}", style="bold magenta"))
|
|
285
|
+
try:
|
|
286
|
+
result = self.tools.run(name, args)
|
|
287
|
+
except KeyboardInterrupt:
|
|
288
|
+
for rest in calls[i:]:
|
|
289
|
+
self.messages.append(
|
|
290
|
+
{
|
|
291
|
+
"role": "tool",
|
|
292
|
+
"tool_name": rest.get("function", {}).get("name", ""),
|
|
293
|
+
"content": "Interrupted by the user before completion.",
|
|
294
|
+
}
|
|
295
|
+
)
|
|
296
|
+
self.console.print("[yellow]⏹ Interrupted[/]")
|
|
297
|
+
raise
|
|
298
|
+
if result.startswith("Error:"):
|
|
299
|
+
self.console.print(Text(f" {result[:300]}", style="red"))
|
|
300
|
+
elif name in ("read_file", "grep", "glob", "list_dir"):
|
|
301
|
+
self.console.print(Text(f" ⎿ {result.count(chr(10)) + 1} line(s)", style="dim"))
|
|
302
|
+
tool_msg = {"role": "tool", "tool_name": name, "content": result}
|
|
303
|
+
if call.get("id"):
|
|
304
|
+
tool_msg["tool_call_id"] = call["id"]
|
|
305
|
+
self.messages.append(tool_msg)
|
|
306
|
+
else:
|
|
307
|
+
self.console.print(f"[yellow]Stopped after {MAX_STEPS_PER_TURN} steps.[/]")
|
|
308
|
+
self.print_stats()
|
|
309
|
+
|
|
310
|
+
def print_stats(self) -> None:
|
|
311
|
+
pct = 100 * self.ctx_used / self.settings.context
|
|
312
|
+
self.console.print(
|
|
313
|
+
Text(
|
|
314
|
+
f" ctx {format_tokens(self.ctx_used)}/{format_tokens(self.settings.context)} ({pct:.0f}%) · "
|
|
315
|
+
f"{self.last_speed:.1f} tok/s",
|
|
316
|
+
style="dim",
|
|
317
|
+
)
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
def expand_mentions(self, text: str) -> str:
|
|
321
|
+
"""Inline files referenced as @path in the user's message."""
|
|
322
|
+
attached = []
|
|
323
|
+
for ref in re.findall(r"(?<!\S)@([\w./~\-]+)", text):
|
|
324
|
+
p = self.tools.resolve(ref)
|
|
325
|
+
if p.is_file() and not is_binary(p) and p.stat().st_size < 200_000:
|
|
326
|
+
attached.append(f'<file path="{self.tools.rel(p)}">\n{p.read_text(errors="replace")}\n</file>')
|
|
327
|
+
self.tools.read_mtimes[str(p)] = p.stat().st_mtime
|
|
328
|
+
self.console.print(Text(f" ⎿ attached {self.tools.rel(p)}", style="dim"))
|
|
329
|
+
elif p.is_dir():
|
|
330
|
+
attached.append(f'<directory path="{self.tools.rel(p)}">\n{tree(p, 2)}\n</directory>')
|
|
331
|
+
return text + ("\n\n" + "\n\n".join(attached) if attached else "")
|
|
332
|
+
|
|
333
|
+
# -- context management
|
|
334
|
+
def maybe_compact(self) -> None:
|
|
335
|
+
if self.ctx_used > AUTO_COMPACT_RATIO * self.settings.context:
|
|
336
|
+
self.console.print("[yellow]Context is nearly full — compacting the conversation…[/]")
|
|
337
|
+
self.compact()
|
|
338
|
+
|
|
339
|
+
def compact(self, focus: str = "") -> None:
|
|
340
|
+
if len(self.messages) <= 2:
|
|
341
|
+
return
|
|
342
|
+
instructions = (
|
|
343
|
+
"Summarize this conversation so the work can continue in a fresh context. Include: the user's goals "
|
|
344
|
+
"and requests, key facts learned about the codebase (files, functions, paths with line numbers), "
|
|
345
|
+
"decisions made, every file created or modified and how, commands run and their outcomes, the current "
|
|
346
|
+
"state, and the remaining next steps. Be thorough but compact. Do not call tools."
|
|
347
|
+
)
|
|
348
|
+
if focus:
|
|
349
|
+
instructions += f" Focus especially on: {focus}"
|
|
350
|
+
summary = ""
|
|
351
|
+
with self.console.status("[cyan]Compacting…[/]"):
|
|
352
|
+
for chunk in self.chat([*self.messages, {"role": "user", "content": instructions}], None, False):
|
|
353
|
+
summary += chunk.get("message", {}).get("content", "")
|
|
354
|
+
self.messages = [
|
|
355
|
+
{"role": "system", "content": self.system_prompt()},
|
|
356
|
+
{"role": "user", "content": f"[Summary of our conversation so far]\n\n{summary}"},
|
|
357
|
+
{"role": "assistant", "content": "Got it — I have the context from the summary and will continue."},
|
|
358
|
+
]
|
|
359
|
+
self.ctx_used = sum(len(m["content"]) for m in self.messages) // 3
|
|
360
|
+
self.console.print(Panel(Markdown(summary), title="Compacted summary", border_style="blue"))
|
|
361
|
+
|
|
362
|
+
# -- settings changes
|
|
363
|
+
def set_context(self, tokens: int) -> str:
|
|
364
|
+
spec = catalog.find(self.settings.model)
|
|
365
|
+
limit = None
|
|
366
|
+
try:
|
|
367
|
+
limit = self.ollama.max_context(self.settings.model)
|
|
368
|
+
except OllamaError:
|
|
369
|
+
limit = spec.max_context if spec else None
|
|
370
|
+
if limit and tokens > limit:
|
|
371
|
+
tokens = limit
|
|
372
|
+
note = f" (capped at the model's maximum, {format_tokens(limit)})"
|
|
373
|
+
else:
|
|
374
|
+
note = ""
|
|
375
|
+
self.settings.context = tokens
|
|
376
|
+
return f"Context window set to {format_tokens(tokens)} tokens{note}. The model reloads on the next request."
|
lcode/catalog.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""The model catalog (models.toml) and memory-fit estimates."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from functools import cache
|
|
8
|
+
from importlib import resources
|
|
9
|
+
|
|
10
|
+
from lcode.hardware import GIB, Hardware
|
|
11
|
+
|
|
12
|
+
if sys.version_info >= (3, 11):
|
|
13
|
+
import tomllib
|
|
14
|
+
else: # pragma: no cover
|
|
15
|
+
import tomli as tomllib
|
|
16
|
+
|
|
17
|
+
OVERHEAD_GIB = 1.0 # compute buffers and runtime; measured <1 GiB for qwen3.6-35b and qwen3.5-9b on CUDA
|
|
18
|
+
HEADROOM_GIB = 0.5 # keep a little memory free so estimates at the edge don't run out
|
|
19
|
+
CONTEXT_STEPS = [1048576, 524288, 262144, 131072, 65536, 32768]
|
|
20
|
+
MIN_USEFUL_CONTEXT = 32768
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class ModelSpec:
|
|
25
|
+
key: str
|
|
26
|
+
tag: str
|
|
27
|
+
name: str
|
|
28
|
+
publisher: str
|
|
29
|
+
params: str
|
|
30
|
+
moe: bool
|
|
31
|
+
size_gb: float
|
|
32
|
+
max_context: int
|
|
33
|
+
kv_kib_per_token: float
|
|
34
|
+
tested: bool = False
|
|
35
|
+
kv_estimated: bool = False
|
|
36
|
+
num_batch: int | None = None
|
|
37
|
+
swe_bench_verified: float | None = None
|
|
38
|
+
notes: str = ""
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def local_name(self) -> str:
|
|
42
|
+
"""Name of the text-only variant `lcode setup` creates for models that ship a vision projector."""
|
|
43
|
+
return f"lcode-{self.key}"
|
|
44
|
+
|
|
45
|
+
def memory_gib(self, context: int) -> float:
|
|
46
|
+
return self.size_gb * 1e9 / GIB + self.kv_kib_per_token * 1024 * context / GIB + OVERHEAD_GIB
|
|
47
|
+
|
|
48
|
+
def best_context(self, budget_gib: float) -> int | None:
|
|
49
|
+
"""Largest standard context window whose estimated memory fits the budget."""
|
|
50
|
+
for ctx in CONTEXT_STEPS:
|
|
51
|
+
if ctx <= self.max_context and self.memory_gib(ctx) <= budget_gib - HEADROOM_GIB:
|
|
52
|
+
return ctx
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
def fit(self, hw: Hardware) -> tuple[int | None, str]:
|
|
56
|
+
"""(largest context that fits, speed note) on this machine."""
|
|
57
|
+
if hw.unified:
|
|
58
|
+
ctx = self.best_context(hw.budget_gib)
|
|
59
|
+
return ctx, "fast" if ctx else "too large"
|
|
60
|
+
if not hw.gpu:
|
|
61
|
+
ctx = self.best_context(hw.budget_gib)
|
|
62
|
+
return ctx, "CPU only (slow)" if ctx else "too large"
|
|
63
|
+
if not self.moe:
|
|
64
|
+
# Dense models slow down a lot when split, so prefer a context that stays in VRAM.
|
|
65
|
+
fast_ctx = self.best_context(hw.vram_gib)
|
|
66
|
+
if fast_ctx and fast_ctx >= MIN_USEFUL_CONTEXT:
|
|
67
|
+
return fast_ctx, "fast"
|
|
68
|
+
ctx = self.best_context(hw.budget_gib)
|
|
69
|
+
if ctx is None:
|
|
70
|
+
return None, "too large"
|
|
71
|
+
if self.memory_gib(ctx) <= hw.vram_gib:
|
|
72
|
+
return ctx, "fast"
|
|
73
|
+
return ctx, "good (experts in RAM)" if self.moe else "slow (split across GPU and CPU)"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@cache
|
|
77
|
+
def load() -> tuple[ModelSpec, ...]:
|
|
78
|
+
data = tomllib.loads(resources.files("lcode").joinpath("models.toml").read_text())
|
|
79
|
+
return tuple(ModelSpec(**entry) for entry in data["model"])
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def find(name: str) -> ModelSpec | None:
|
|
83
|
+
"""Look a model up by catalog key, Ollama tag or local variant name."""
|
|
84
|
+
for spec in load():
|
|
85
|
+
if name in (spec.key, spec.tag, spec.local_name, f"{spec.local_name}:latest") or (
|
|
86
|
+
":" not in spec.tag and name == f"{spec.tag}:latest"
|
|
87
|
+
):
|
|
88
|
+
return spec
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def recommend(hw: Hardware) -> tuple[ModelSpec, int] | None:
|
|
93
|
+
"""Best model for this machine, in order of preference:
|
|
94
|
+
|
|
95
|
+
1. the first tested model (catalog order) that fits with at least 32K of context at usable speed;
|
|
96
|
+
2. the first model in catalog order that fits with at least 64K, then 32K, at usable speed;
|
|
97
|
+
3. the smallest model that fits at all.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
def usable(spec: ModelSpec, min_ctx: int) -> int | None:
|
|
101
|
+
ctx, speed = spec.fit(hw)
|
|
102
|
+
return ctx if ctx and ctx >= min_ctx and "slow" not in speed else None
|
|
103
|
+
|
|
104
|
+
for spec in load():
|
|
105
|
+
if spec.tested and (ctx := usable(spec, MIN_USEFUL_CONTEXT)):
|
|
106
|
+
return spec, ctx
|
|
107
|
+
for min_ctx in (65536, MIN_USEFUL_CONTEXT):
|
|
108
|
+
for spec in load():
|
|
109
|
+
if ctx := usable(spec, min_ctx):
|
|
110
|
+
return spec, ctx
|
|
111
|
+
for spec in sorted(load(), key=lambda s: s.size_gb):
|
|
112
|
+
if ctx := spec.best_context(hw.budget_gib):
|
|
113
|
+
return spec, ctx
|
|
114
|
+
return None
|