lcode-cli 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lcode/__init__.py +3 -0
- lcode/__main__.py +3 -0
- lcode/agent.py +376 -0
- lcode/catalog.py +114 -0
- lcode/cli.py +437 -0
- lcode/config.py +135 -0
- lcode/hardware.py +84 -0
- lcode/models.toml +128 -0
- lcode/ollama.py +163 -0
- lcode/permissions.py +90 -0
- lcode/render.py +43 -0
- lcode/repl.py +281 -0
- lcode/tools.py +533 -0
- lcode_cli-0.1.1.dist-info/METADATA +181 -0
- lcode_cli-0.1.1.dist-info/RECORD +18 -0
- lcode_cli-0.1.1.dist-info/WHEEL +4 -0
- lcode_cli-0.1.1.dist-info/entry_points.txt +2 -0
- lcode_cli-0.1.1.dist-info/licenses/LICENSE +22 -0
lcode/cli.py
ADDED
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
"""Command-line entry point: `lcode` (chat) plus the setup, models, doctor and config subcommands."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import platform
|
|
7
|
+
import shutil
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import requests
|
|
12
|
+
from rich.console import Console
|
|
13
|
+
from rich.progress import BarColumn, DownloadColumn, Progress, TextColumn, TimeRemainingColumn, TransferSpeedColumn
|
|
14
|
+
from rich.prompt import Confirm
|
|
15
|
+
from rich.table import Table
|
|
16
|
+
|
|
17
|
+
from lcode import __version__, catalog, config
|
|
18
|
+
from lcode.agent import Agent, Settings
|
|
19
|
+
from lcode.catalog import ModelSpec
|
|
20
|
+
from lcode.config import ConfigError, format_tokens, parse_context
|
|
21
|
+
from lcode.hardware import Hardware, detect
|
|
22
|
+
from lcode.ollama import MIN_VERSION, Ollama, OllamaError, redact, version_tuple
|
|
23
|
+
|
|
24
|
+
console = Console(highlight=False)
|
|
25
|
+
|
|
26
|
+
OLLAMA_INSTALL = {
|
|
27
|
+
"linux": "curl -fsSL https://ollama.com/install.sh | sh",
|
|
28
|
+
"macos": "brew install ollama && brew services start ollama (or download the app from https://ollama.com)",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class NotInstalled(Exception):
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def fail(message: str, code: int = 1):
|
|
37
|
+
console.print(f"[red]error:[/] {message}")
|
|
38
|
+
sys.exit(code)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def check_ollama(ollama: Ollama, hw: Hardware | None = None) -> str:
|
|
42
|
+
try:
|
|
43
|
+
version = ollama.version()
|
|
44
|
+
except OllamaError as e:
|
|
45
|
+
hint = OLLAMA_INSTALL.get((hw or detect()).os, OLLAMA_INSTALL["linux"])
|
|
46
|
+
fail(f"{e}\n\nIs Ollama installed and running? Install it with:\n {hint}")
|
|
47
|
+
if version_tuple(version) < MIN_VERSION:
|
|
48
|
+
fail(f"Ollama {version} is too old; lcode needs {'.'.join(map(str, MIN_VERSION))} or newer. Update Ollama.")
|
|
49
|
+
return version
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def resolve_model(ollama: Ollama, name: str) -> tuple[str, ModelSpec | None]:
|
|
53
|
+
"""Map a catalog key or Ollama tag to an installed Ollama model name."""
|
|
54
|
+
spec = catalog.find(name)
|
|
55
|
+
installed = ollama.installed_names()
|
|
56
|
+
if spec:
|
|
57
|
+
for candidate in (spec.local_name, spec.tag):
|
|
58
|
+
if candidate in installed:
|
|
59
|
+
return candidate, spec
|
|
60
|
+
raise NotInstalled(f"{spec.name} is not installed yet. Run: lcode setup {spec.key}")
|
|
61
|
+
if name in installed or f"{name}:latest" in installed:
|
|
62
|
+
return name, None
|
|
63
|
+
raise NotInstalled(
|
|
64
|
+
f"Model '{name}' is not installed in Ollama. Pick one with `lcode models` and install it with "
|
|
65
|
+
f"`lcode setup <model>`, or pull any Ollama model with `ollama pull {name}`."
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def choose_context(
|
|
70
|
+
ollama: Ollama, model: str, spec: ModelSpec | None, requested: int | None, hw: Hardware
|
|
71
|
+
) -> tuple[int, str]:
|
|
72
|
+
"""Pick the context window: requested > largest that fits (catalog models) > 32K. Capped at the model max."""
|
|
73
|
+
try:
|
|
74
|
+
limit = ollama.max_context(model) or (spec.max_context if spec else None)
|
|
75
|
+
except OllamaError:
|
|
76
|
+
limit = spec.max_context if spec else None
|
|
77
|
+
note = ""
|
|
78
|
+
if requested:
|
|
79
|
+
ctx = requested
|
|
80
|
+
elif spec:
|
|
81
|
+
ctx = spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT
|
|
82
|
+
else:
|
|
83
|
+
ctx = 32768
|
|
84
|
+
if limit and ctx > limit:
|
|
85
|
+
ctx, note = limit, f"capped at the model maximum of {format_tokens(limit)}"
|
|
86
|
+
if spec and requested and spec.memory_gib(ctx) > hw.budget_gib:
|
|
87
|
+
note = (
|
|
88
|
+
f"needs ~{spec.memory_gib(ctx):.0f} GB but only ~{hw.budget_gib:.0f} GB is available; "
|
|
89
|
+
"expect out-of-memory errors or heavy slowdowns"
|
|
90
|
+
)
|
|
91
|
+
return ctx, note
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ----------------------------------------------------------------------------- lcode models
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def print_models(ollama: Ollama | None, hw: Hardware, current: str | None = None) -> None:
|
|
98
|
+
try:
|
|
99
|
+
installed = ollama.installed_names() if ollama else set()
|
|
100
|
+
except OllamaError:
|
|
101
|
+
installed = set()
|
|
102
|
+
rec = catalog.recommend(hw)
|
|
103
|
+
wide = console.width >= 130 # the descriptive columns don't fit an 80-column terminal
|
|
104
|
+
columns = ["Key", "Model", "Type", "Size", "Max ctx", "Fits here", "Speed", "Status"]
|
|
105
|
+
if not wide:
|
|
106
|
+
columns = [c for c in columns if c not in ("Model", "Type")]
|
|
107
|
+
table = Table(title=f"Models for this machine: {hw.describe()}", title_justify="left", header_style="bold")
|
|
108
|
+
for col in columns:
|
|
109
|
+
table.add_column(
|
|
110
|
+
col, no_wrap=col in ("Key", "Size", "Max ctx", "Fits here"), min_width=11 if col == "Status" else None
|
|
111
|
+
)
|
|
112
|
+
for spec in catalog.load():
|
|
113
|
+
ctx, speed = spec.fit(hw)
|
|
114
|
+
marks = []
|
|
115
|
+
if spec.local_name in installed or spec.tag in installed:
|
|
116
|
+
marks.append("[green]installed[/]")
|
|
117
|
+
if rec and rec[0].key == spec.key:
|
|
118
|
+
marks.append("[cyan]recommended[/]")
|
|
119
|
+
if current and catalog.find(current) is spec:
|
|
120
|
+
marks.append("[bold]current[/]")
|
|
121
|
+
marks.append("tested" if spec.tested else "[dim]untested[/]")
|
|
122
|
+
row = {
|
|
123
|
+
"Key": spec.key,
|
|
124
|
+
"Model": spec.name,
|
|
125
|
+
"Type": spec.params,
|
|
126
|
+
"Size": f"{spec.size_gb:.1f} GB",
|
|
127
|
+
"Max ctx": format_tokens(spec.max_context),
|
|
128
|
+
"Fits here": format_tokens(ctx) if ctx else "[red]no[/]",
|
|
129
|
+
"Speed": speed if wide else speed.split(" (")[0],
|
|
130
|
+
"Status": ", ".join(marks) if wide else "\n".join(marks),
|
|
131
|
+
}
|
|
132
|
+
table.add_row(*(row[c] for c in columns))
|
|
133
|
+
console.print(table)
|
|
134
|
+
console.print(
|
|
135
|
+
"[dim]Install one with [/]lcode setup <key>[dim]. 'Fits here' is the largest context that fits in memory; "
|
|
136
|
+
"choose another with --context. Any other Ollama model with tool calling also works: lcode --model <tag>[/]"
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def cmd_models(args) -> None:
|
|
141
|
+
cfg = config.load()
|
|
142
|
+
ollama = Ollama(cfg["ollama_host"])
|
|
143
|
+
print_models(ollama, detect(), cfg["model"])
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# ----------------------------------------------------------------------------- lcode setup
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def pull_with_progress(ollama: Ollama, tag: str) -> None:
|
|
150
|
+
columns = (
|
|
151
|
+
TextColumn(" {task.description}"),
|
|
152
|
+
BarColumn(),
|
|
153
|
+
DownloadColumn(),
|
|
154
|
+
TransferSpeedColumn(),
|
|
155
|
+
TimeRemainingColumn(),
|
|
156
|
+
)
|
|
157
|
+
with Progress(*columns, console=console) as progress:
|
|
158
|
+
tasks: dict[str, int] = {}
|
|
159
|
+
|
|
160
|
+
def on_event(event: dict) -> None:
|
|
161
|
+
digest, total = event.get("digest"), event.get("total")
|
|
162
|
+
if digest and total:
|
|
163
|
+
if digest not in tasks:
|
|
164
|
+
tasks[digest] = progress.add_task(f"layer {digest.split(':')[-1][:12]}", total=total)
|
|
165
|
+
progress.update(tasks[digest], completed=event.get("completed", 0))
|
|
166
|
+
elif event.get("status") and not digest:
|
|
167
|
+
progress.console.print(f" [dim]{event['status']}[/]")
|
|
168
|
+
|
|
169
|
+
ollama.pull(tag, on_event)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def cmd_setup(args) -> None:
|
|
173
|
+
cfg = config.load()
|
|
174
|
+
ollama = Ollama(cfg["ollama_host"])
|
|
175
|
+
hw = detect()
|
|
176
|
+
console.print(f"[bold]Machine:[/] {hw.describe()}")
|
|
177
|
+
console.print(f"[bold]Ollama:[/] {redact(ollama.host)} (version {check_ollama(ollama, hw)})")
|
|
178
|
+
requested = None
|
|
179
|
+
if args.context:
|
|
180
|
+
try:
|
|
181
|
+
requested = parse_context(args.context)
|
|
182
|
+
except ConfigError as e:
|
|
183
|
+
fail(str(e))
|
|
184
|
+
|
|
185
|
+
if args.model:
|
|
186
|
+
spec = catalog.find(args.model)
|
|
187
|
+
tag = spec.tag if spec else args.model
|
|
188
|
+
else:
|
|
189
|
+
rec = catalog.recommend(hw)
|
|
190
|
+
if not rec:
|
|
191
|
+
fail("no catalog model fits in this machine's memory. See `lcode models`.")
|
|
192
|
+
spec, _ = rec
|
|
193
|
+
tag = spec.tag
|
|
194
|
+
console.print(f"[bold]Recommended:[/] {spec.name} ({spec.params}) — {spec.notes}")
|
|
195
|
+
if spec:
|
|
196
|
+
ctx = requested or spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT
|
|
197
|
+
ctx = min(ctx, spec.max_context)
|
|
198
|
+
console.print(
|
|
199
|
+
f"[bold]Context:[/] {format_tokens(ctx)} tokens (~{spec.memory_gib(ctx):.0f} GB of the "
|
|
200
|
+
f"~{hw.budget_gib:.0f} GB available; speed: {spec.fit(hw)[1]})"
|
|
201
|
+
)
|
|
202
|
+
if spec.memory_gib(ctx) > hw.budget_gib:
|
|
203
|
+
console.print("[yellow]warning:[/] this probably doesn't fit in memory; pick a smaller --context or model.")
|
|
204
|
+
if not spec.tested:
|
|
205
|
+
console.print(
|
|
206
|
+
"[yellow]note:[/] this model hasn't been verified with lcode yet — please report how it goes."
|
|
207
|
+
)
|
|
208
|
+
else:
|
|
209
|
+
ctx = requested
|
|
210
|
+
console.print(f"[yellow]note:[/] {tag} is not in lcode's catalog, so its memory needs can't be estimated.")
|
|
211
|
+
|
|
212
|
+
if not args.yes and not Confirm.ask(f"Download {tag} and make it lcode's default model?", default=True):
|
|
213
|
+
console.print("Nothing changed. See all options with `lcode models`.")
|
|
214
|
+
return
|
|
215
|
+
|
|
216
|
+
try:
|
|
217
|
+
pull_with_progress(ollama, tag)
|
|
218
|
+
name = tag
|
|
219
|
+
if spec and ollama.create_text_only(spec.local_name, tag):
|
|
220
|
+
name = spec.local_name
|
|
221
|
+
console.print(f" Created [bold]{name}[/] (text-only variant: frees ~1 GB of GPU memory, no extra disk)")
|
|
222
|
+
except OllamaError as e:
|
|
223
|
+
fail(str(e))
|
|
224
|
+
|
|
225
|
+
updates: dict = {"model": spec.key if spec else tag}
|
|
226
|
+
if ctx:
|
|
227
|
+
updates["context"] = ctx
|
|
228
|
+
config.save(updates, remove=() if ctx else ("context",))
|
|
229
|
+
console.print(f"\n[green]✓ Ready.[/] Saved to {config.CONFIG_PATH}\n cd into a project and run: [bold]lcode[/]")
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
# ----------------------------------------------------------------------------- lcode doctor
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def cmd_doctor(args) -> None:
|
|
236
|
+
ok = True
|
|
237
|
+
|
|
238
|
+
def line(label: str, value: str, good: bool | None = True) -> None:
|
|
239
|
+
mark = {True: "[green]✓[/]", False: "[red]✗[/]", None: "[yellow]![/]"}[good]
|
|
240
|
+
console.print(f"{mark} [bold]{label:<10}[/] {value}")
|
|
241
|
+
|
|
242
|
+
try:
|
|
243
|
+
cfg = config.load()
|
|
244
|
+
except ConfigError as e:
|
|
245
|
+
fail(str(e))
|
|
246
|
+
hw = detect()
|
|
247
|
+
console.print(
|
|
248
|
+
f"lcode {__version__} · Python {platform.python_version()} · {platform.system()} {platform.release()} "
|
|
249
|
+
f"({platform.machine()})\n"
|
|
250
|
+
)
|
|
251
|
+
line("Hardware", hw.describe(), True if (hw.gpu or hw.unified) else None)
|
|
252
|
+
line(
|
|
253
|
+
"Config",
|
|
254
|
+
f"{config.CONFIG_PATH}" + ("" if config.CONFIG_PATH.exists() else " (not created yet)"),
|
|
255
|
+
True if config.CONFIG_PATH.exists() else None,
|
|
256
|
+
)
|
|
257
|
+
ollama = Ollama(cfg["ollama_host"])
|
|
258
|
+
try:
|
|
259
|
+
version = ollama.version()
|
|
260
|
+
fresh = version_tuple(version) >= MIN_VERSION
|
|
261
|
+
line("Ollama", f"{redact(ollama.host)} · version {version}" + ("" if fresh else " (too old)"), fresh)
|
|
262
|
+
ok &= fresh
|
|
263
|
+
except OllamaError as e:
|
|
264
|
+
line("Ollama", f"{e}", False)
|
|
265
|
+
line("", f"install: {OLLAMA_INSTALL.get(hw.os, OLLAMA_INSTALL['linux'])}", None)
|
|
266
|
+
sys.exit(1)
|
|
267
|
+
try:
|
|
268
|
+
model, spec = resolve_model(ollama, cfg["model"])
|
|
269
|
+
line("Model", f"{cfg['model']} → {model}", True)
|
|
270
|
+
ctx, note = choose_context(ollama, model, spec, cfg["context"], hw)
|
|
271
|
+
detail = f"{format_tokens(ctx)} tokens"
|
|
272
|
+
if spec:
|
|
273
|
+
detail += f" · ~{spec.memory_gib(ctx):.0f} GB needed, ~{hw.budget_gib:.0f} GB available"
|
|
274
|
+
line("Context", detail + (f" ({note})" if note else ""), None if note else True)
|
|
275
|
+
loaded = [m for m in ollama.running() if m.get("name") == model or m.get("model") == model]
|
|
276
|
+
if loaded:
|
|
277
|
+
m = loaded[0]
|
|
278
|
+
line(
|
|
279
|
+
"Loaded",
|
|
280
|
+
f"{m.get('size_vram', 0) / 1e9:.1f} GB on GPU of {m.get('size', 0) / 1e9:.1f} GB, "
|
|
281
|
+
f"context {m.get('context_length', '?')}",
|
|
282
|
+
True,
|
|
283
|
+
)
|
|
284
|
+
except NotInstalled as e:
|
|
285
|
+
line("Model", str(e), False)
|
|
286
|
+
ok = False
|
|
287
|
+
line(
|
|
288
|
+
"ripgrep",
|
|
289
|
+
"found" if shutil.which("rg") else "not found — install it for faster search",
|
|
290
|
+
bool(shutil.which("rg")) or None,
|
|
291
|
+
)
|
|
292
|
+
line("git", "found" if shutil.which("git") else "not found", bool(shutil.which("git")) or None)
|
|
293
|
+
sys.exit(0 if ok else 1)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# ----------------------------------------------------------------------------- lcode config
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def cmd_config(args) -> None:
|
|
300
|
+
try:
|
|
301
|
+
if args.action == "set":
|
|
302
|
+
config.save({args.key: args.value})
|
|
303
|
+
console.print(f"Set {args.key} = {config.read_file()[args.key]}")
|
|
304
|
+
elif args.action == "unset":
|
|
305
|
+
config.coerce(args.key, None)
|
|
306
|
+
config.save({}, remove=(args.key,))
|
|
307
|
+
console.print(f"Unset {args.key} (back to the default)")
|
|
308
|
+
elif args.action == "path":
|
|
309
|
+
console.print(str(config.CONFIG_PATH))
|
|
310
|
+
else:
|
|
311
|
+
cfg, from_file = config.load(), config.read_file()
|
|
312
|
+
table = Table(title=str(config.CONFIG_PATH), title_justify="left", header_style="bold")
|
|
313
|
+
for col in ("Setting", "Value", "Source", "Description"):
|
|
314
|
+
table.add_column(col)
|
|
315
|
+
for key, (default, _, help_text) in config.SETTINGS.items():
|
|
316
|
+
source = "file" if key in from_file else "default"
|
|
317
|
+
if cfg[key] != from_file.get(key, default):
|
|
318
|
+
source = "environment"
|
|
319
|
+
value = format_tokens(cfg[key]) if key == "context" and cfg[key] else str(cfg[key])
|
|
320
|
+
table.add_row(key, value, source, help_text)
|
|
321
|
+
console.print(table)
|
|
322
|
+
console.print("[dim]Change with: lcode config set <setting> <value>[/]")
|
|
323
|
+
except ConfigError as e:
|
|
324
|
+
fail(str(e))
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
# ----------------------------------------------------------------------------- lcode (chat)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def cmd_chat(args) -> None:
|
|
331
|
+
from lcode.repl import repl
|
|
332
|
+
|
|
333
|
+
try:
|
|
334
|
+
cfg = config.load()
|
|
335
|
+
requested_ctx = parse_context(args.context) if args.context else cfg["context"]
|
|
336
|
+
same_model = (catalog.find(args.model) or args.model) == (catalog.find(cfg["model"]) or cfg["model"])
|
|
337
|
+
if not args.context and args.model and not same_model:
|
|
338
|
+
requested_ctx = None # the saved context was sized for the saved model; fit this one instead
|
|
339
|
+
except ConfigError as e:
|
|
340
|
+
fail(str(e))
|
|
341
|
+
cwd = Path(args.repo).expanduser().resolve()
|
|
342
|
+
if not cwd.is_dir():
|
|
343
|
+
fail(f"not a directory: {cwd}")
|
|
344
|
+
ollama = Ollama(cfg["ollama_host"])
|
|
345
|
+
hw = detect()
|
|
346
|
+
check_ollama(ollama, hw)
|
|
347
|
+
try:
|
|
348
|
+
model, spec = resolve_model(ollama, args.model or cfg["model"])
|
|
349
|
+
except NotInstalled as e:
|
|
350
|
+
if not args.model and cfg["model"] == config.DEFAULT_MODEL and not config.CONFIG_PATH.exists():
|
|
351
|
+
fail("lcode isn't set up yet. Run: lcode setup")
|
|
352
|
+
fail(str(e))
|
|
353
|
+
context, note = choose_context(ollama, model, spec, requested_ctx, hw)
|
|
354
|
+
if note:
|
|
355
|
+
console.print(f"[yellow]Context {format_tokens(context)}: {note}[/]")
|
|
356
|
+
mode = "yolo" if args.yolo else ("auto-edit" if args.auto_edit else cfg["permission_mode"])
|
|
357
|
+
num_batch = cfg["num_batch"]
|
|
358
|
+
if num_batch is None and spec and spec.num_batch:
|
|
359
|
+
if model == spec.local_name:
|
|
360
|
+
num_batch = spec.num_batch
|
|
361
|
+
else: # the tuned batch size assumes the text-only variant; the vision projector needs that VRAM
|
|
362
|
+
console.print(f"[dim]Tip: run `lcode setup {spec.key}` once to create the faster text-only variant.[/]")
|
|
363
|
+
settings = Settings(
|
|
364
|
+
model=model,
|
|
365
|
+
context=context,
|
|
366
|
+
num_batch=num_batch,
|
|
367
|
+
keep_alive=cfg["keep_alive"],
|
|
368
|
+
think=cfg["think"] and not args.no_think,
|
|
369
|
+
show_thinking=args.show_thinking,
|
|
370
|
+
permission_mode=mode,
|
|
371
|
+
)
|
|
372
|
+
agent = Agent(ollama, settings, cwd, console=console)
|
|
373
|
+
repl(agent, prompt=args.prompt, resume=args.cont, hardware=hw)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
377
|
+
parser = argparse.ArgumentParser(
|
|
378
|
+
prog="lcode",
|
|
379
|
+
description="A local-first terminal coding agent powered by open-weight models via Ollama.",
|
|
380
|
+
epilog="Subcommands: lcode setup | models | doctor | config (lcode <subcommand> --help). "
|
|
381
|
+
"Docs: https://nasser1941.github.io/lcode/",
|
|
382
|
+
)
|
|
383
|
+
parser.add_argument("-p", "--prompt", help="run one request non-interactively and exit")
|
|
384
|
+
parser.add_argument("-m", "--model", help="catalog key (see `lcode models`) or any installed Ollama model")
|
|
385
|
+
parser.add_argument("--context", "--ctx", dest="context", help="context window, e.g. 65536, 128k or 1m")
|
|
386
|
+
parser.add_argument("-r", "--repo", default=".", help="working directory (default: current directory)")
|
|
387
|
+
parser.add_argument("-c", "--continue", dest="cont", action="store_true", help="resume the last session here")
|
|
388
|
+
parser.add_argument("--auto-edit", action="store_true", help="apply file edits without asking")
|
|
389
|
+
parser.add_argument("--yolo", action="store_true", help="never ask for permission (edits and commands)")
|
|
390
|
+
parser.add_argument("--no-think", action="store_true", help="disable model reasoning (faster, less accurate)")
|
|
391
|
+
parser.add_argument("--show-thinking", action="store_true", help="print the model's reasoning as it streams")
|
|
392
|
+
parser.add_argument("-V", "--version", action="version", version=f"lcode {__version__}")
|
|
393
|
+
return parser
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def build_subparsers() -> dict[str, argparse.ArgumentParser]:
|
|
397
|
+
subs = {}
|
|
398
|
+
p = argparse.ArgumentParser(prog="lcode setup", description="Download a model and make it lcode's default.")
|
|
399
|
+
p.add_argument("model", nargs="?", help="catalog key or Ollama tag (default: the best fit for this machine)")
|
|
400
|
+
p.add_argument("--context", "--ctx", dest="context", help="context window, e.g. 128k (default: largest that fits)")
|
|
401
|
+
p.add_argument("-y", "--yes", action="store_true", help="don't ask for confirmation")
|
|
402
|
+
subs["setup"] = p
|
|
403
|
+
subs["models"] = argparse.ArgumentParser(prog="lcode models", description="List models and how they fit here.")
|
|
404
|
+
subs["doctor"] = argparse.ArgumentParser(prog="lcode doctor", description="Check the installation.")
|
|
405
|
+
p = argparse.ArgumentParser(prog="lcode config", description="Show or change settings.")
|
|
406
|
+
p.add_argument("action", nargs="?", choices=["show", "set", "unset", "path"], default="show")
|
|
407
|
+
p.add_argument("key", nargs="?", choices=list(config.SETTINGS))
|
|
408
|
+
p.add_argument("value", nargs="?")
|
|
409
|
+
subs["config"] = p
|
|
410
|
+
return subs
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
COMMANDS = {"setup": cmd_setup, "models": cmd_models, "doctor": cmd_doctor, "config": cmd_config}
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def main(argv: list[str] | None = None) -> None:
|
|
417
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
418
|
+
try:
|
|
419
|
+
if argv and argv[0] in COMMANDS:
|
|
420
|
+
parser = build_subparsers()[argv[0]]
|
|
421
|
+
args = parser.parse_args(argv[1:])
|
|
422
|
+
if argv[0] == "config" and args.action in ("set", "unset") and not args.key:
|
|
423
|
+
parser.error(f"config {args.action} needs a setting name")
|
|
424
|
+
if argv[0] == "config" and args.action == "set" and args.value is None:
|
|
425
|
+
parser.error("config set needs a value")
|
|
426
|
+
COMMANDS[argv[0]](args)
|
|
427
|
+
else:
|
|
428
|
+
cmd_chat(build_parser().parse_args(argv))
|
|
429
|
+
except KeyboardInterrupt:
|
|
430
|
+
console.print("\n[dim]Cancelled.[/]")
|
|
431
|
+
sys.exit(130)
|
|
432
|
+
except requests.ConnectionError as e:
|
|
433
|
+
fail(f"lost connection to Ollama: {e}")
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
if __name__ == "__main__":
|
|
437
|
+
main()
|
lcode/config.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""User configuration (~/.config/lcode/config.toml) and on-disk state locations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
if sys.version_info >= (3, 11):
|
|
12
|
+
import tomllib
|
|
13
|
+
else: # pragma: no cover
|
|
14
|
+
import tomli as tomllib
|
|
15
|
+
|
|
16
|
+
CONFIG_DIR = Path(os.environ.get("XDG_CONFIG_HOME") or Path.home() / ".config") / "lcode"
|
|
17
|
+
CONFIG_PATH = CONFIG_DIR / "config.toml"
|
|
18
|
+
if os.environ.get("LCODE_HOME"): # sessions and prompt history
|
|
19
|
+
STATE_DIR = Path(os.environ["LCODE_HOME"])
|
|
20
|
+
else:
|
|
21
|
+
STATE_DIR = Path(os.environ.get("XDG_STATE_HOME") or Path.home() / ".local/state") / "lcode"
|
|
22
|
+
|
|
23
|
+
DEFAULT_MODEL = "qwen3.6-35b"
|
|
24
|
+
PERMISSION_MODES = ("ask", "auto-edit", "yolo")
|
|
25
|
+
|
|
26
|
+
# key -> (default, type, help)
|
|
27
|
+
SETTINGS: dict[str, tuple[object, type, str]] = {
|
|
28
|
+
"model": (DEFAULT_MODEL, str, "catalog key (see `lcode models`) or any Ollama model tag"),
|
|
29
|
+
"context": (None, int, "context window in tokens, e.g. 131072 or 128k (default: largest that fits)"),
|
|
30
|
+
"num_batch": (None, int, "prompt batch size; larger reads prompts faster but needs more VRAM"),
|
|
31
|
+
"keep_alive": ("30m", str, "how long Ollama keeps the model loaded after the last request"),
|
|
32
|
+
"ollama_host": ("http://localhost:11434", str, "Ollama server URL"),
|
|
33
|
+
"permission_mode": ("ask", str, "ask | auto-edit | yolo"),
|
|
34
|
+
"think": (True, bool, "let the model reason before answering (slower, better)"),
|
|
35
|
+
}
|
|
36
|
+
ENV_OVERRIDES = {
|
|
37
|
+
"LCODE_MODEL": "model",
|
|
38
|
+
"LCODE_CONTEXT": "context",
|
|
39
|
+
"LCODE_NUM_BATCH": "num_batch",
|
|
40
|
+
"LCODE_KEEP_ALIVE": "keep_alive",
|
|
41
|
+
"OLLAMA_HOST": "ollama_host",
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ConfigError(ValueError):
|
|
46
|
+
pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def parse_context(value: str | int) -> int:
|
|
50
|
+
"""Parse '131072', '128k', '128K' or '1m' into a token count (k = 1024)."""
|
|
51
|
+
if isinstance(value, int):
|
|
52
|
+
n = value
|
|
53
|
+
else:
|
|
54
|
+
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*([kKmM]?)\s*", str(value))
|
|
55
|
+
if not m:
|
|
56
|
+
raise ConfigError(f"invalid context size {value!r}; use e.g. 131072, 128k or 1m")
|
|
57
|
+
n = int(float(m.group(1)) * {"": 1, "k": 1024, "m": 1024 * 1024}[m.group(2).lower()])
|
|
58
|
+
if n < 2048:
|
|
59
|
+
raise ConfigError(f"context size {n} is too small (minimum 2048)")
|
|
60
|
+
return n
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def format_tokens(n: int) -> str:
|
|
64
|
+
if n >= 1024 * 1024 and n % (1024 * 1024) == 0:
|
|
65
|
+
return f"{n // (1024 * 1024)}M"
|
|
66
|
+
if n >= 1024 and n % 1024 == 0:
|
|
67
|
+
return f"{n // 1024}K"
|
|
68
|
+
return f"{n / 1000:.1f}K" if n >= 1000 else str(n)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def normalize_host(host: str) -> str:
|
|
72
|
+
host = host.strip()
|
|
73
|
+
if "://" not in host:
|
|
74
|
+
host = "http://" + host
|
|
75
|
+
return host.replace("0.0.0.0", "localhost").rstrip("/")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def coerce(key: str, value: object) -> object:
|
|
79
|
+
if key not in SETTINGS:
|
|
80
|
+
raise ConfigError(f"unknown setting {key!r}; valid: {', '.join(SETTINGS)}")
|
|
81
|
+
if value is None:
|
|
82
|
+
return None
|
|
83
|
+
_, typ, _ = SETTINGS[key]
|
|
84
|
+
if key == "context":
|
|
85
|
+
return parse_context(value) # type: ignore[arg-type]
|
|
86
|
+
if key == "ollama_host":
|
|
87
|
+
return normalize_host(str(value))
|
|
88
|
+
if key == "permission_mode" and value not in PERMISSION_MODES:
|
|
89
|
+
raise ConfigError(f"permission_mode must be one of {', '.join(PERMISSION_MODES)}")
|
|
90
|
+
if typ is bool and isinstance(value, str):
|
|
91
|
+
if value.lower() not in ("true", "false", "1", "0", "yes", "no", "on", "off"):
|
|
92
|
+
raise ConfigError(f"{key} must be true or false")
|
|
93
|
+
return value.lower() in ("true", "1", "yes", "on")
|
|
94
|
+
if typ is int:
|
|
95
|
+
try:
|
|
96
|
+
return int(value) # type: ignore[call-overload]
|
|
97
|
+
except (TypeError, ValueError) as e:
|
|
98
|
+
raise ConfigError(f"{key} must be an integer") from e
|
|
99
|
+
return typ(value)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def read_file(path: Path | None = None) -> dict:
|
|
103
|
+
path = path or CONFIG_PATH
|
|
104
|
+
if not path.is_file():
|
|
105
|
+
return {}
|
|
106
|
+
try:
|
|
107
|
+
data = tomllib.loads(path.read_text())
|
|
108
|
+
except tomllib.TOMLDecodeError as e:
|
|
109
|
+
raise ConfigError(f"{path} is not valid TOML: {e}") from e
|
|
110
|
+
return {k: coerce(k, v) for k, v in data.items() if k in SETTINGS}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def load(path: Path | None = None) -> dict:
|
|
114
|
+
"""Defaults < config file < environment variables."""
|
|
115
|
+
cfg = {k: default for k, (default, _, _) in SETTINGS.items()}
|
|
116
|
+
cfg.update(read_file(path))
|
|
117
|
+
for env, key in ENV_OVERRIDES.items():
|
|
118
|
+
if os.environ.get(env):
|
|
119
|
+
cfg[key] = coerce(key, os.environ[env])
|
|
120
|
+
return cfg
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def save(updates: dict, path: Path | None = None, remove: tuple[str, ...] = ()) -> None:
|
|
124
|
+
path = path or CONFIG_PATH
|
|
125
|
+
data = read_file(path)
|
|
126
|
+
data.update({k: coerce(k, v) for k, v in updates.items()})
|
|
127
|
+
for k in remove:
|
|
128
|
+
data.pop(k, None)
|
|
129
|
+
lines = ["# lcode configuration — see `lcode config` or https://nasser1941.github.io/lcode/configuration/"]
|
|
130
|
+
for k in SETTINGS:
|
|
131
|
+
if data.get(k) is not None:
|
|
132
|
+
v = data[k]
|
|
133
|
+
lines.append(f"{k} = {str(v).lower() if isinstance(v, bool) else json.dumps(v)}")
|
|
134
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
135
|
+
path.write_text("\n".join(lines) + "\n")
|
lcode/hardware.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Detect the machine's GPU and memory to decide which models and context sizes fit."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import platform
|
|
6
|
+
import subprocess
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
GIB = 1024**3
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Hardware:
|
|
14
|
+
os: str # "linux", "macos" or other platform.system() value
|
|
15
|
+
cpu: str
|
|
16
|
+
ram_gib: float
|
|
17
|
+
gpu: str | None = None
|
|
18
|
+
vram_gib: float = 0.0 # dedicated GPU memory (0 for Apple Silicon: memory is unified)
|
|
19
|
+
unified: bool = False # Apple Silicon
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def budget_gib(self) -> float:
|
|
23
|
+
"""Memory available for model weights + KV cache."""
|
|
24
|
+
if self.unified:
|
|
25
|
+
# macOS lets the GPU wire ~2/3 of RAM on smaller Macs and ~3/4 above 36 GB.
|
|
26
|
+
return self.ram_gib * (0.75 if self.ram_gib > 36 else 0.67)
|
|
27
|
+
# Dedicated GPU plus system RAM for offloaded layers, keeping ~8 GiB for the OS and apps.
|
|
28
|
+
return self.vram_gib + max(0.0, self.ram_gib - 8)
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def fast_gib(self) -> float:
|
|
32
|
+
"""Memory where a model runs at full accelerator speed."""
|
|
33
|
+
return self.budget_gib if self.unified else self.vram_gib
|
|
34
|
+
|
|
35
|
+
def describe(self) -> str:
|
|
36
|
+
if self.unified:
|
|
37
|
+
return f"{self.cpu}, {self.ram_gib:.0f} GB unified memory (~{self.budget_gib:.0f} GB usable by the GPU)"
|
|
38
|
+
gpu = f"{self.gpu} ({self.vram_gib:.0f} GB VRAM)" if self.gpu else "no supported GPU (CPU only)"
|
|
39
|
+
return f"{gpu}, {self.ram_gib:.0f} GB RAM"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _run(cmd: list[str]) -> str:
|
|
43
|
+
try:
|
|
44
|
+
return subprocess.run(cmd, capture_output=True, text=True, timeout=10, check=True).stdout.strip()
|
|
45
|
+
except (OSError, subprocess.SubprocessError):
|
|
46
|
+
return ""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _linux_ram_gib() -> float:
|
|
50
|
+
try:
|
|
51
|
+
with open("/proc/meminfo") as f:
|
|
52
|
+
for line in f:
|
|
53
|
+
if line.startswith("MemTotal:"):
|
|
54
|
+
return int(line.split()[1]) * 1024 / GIB
|
|
55
|
+
except OSError:
|
|
56
|
+
pass
|
|
57
|
+
return 0.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def detect() -> Hardware:
|
|
61
|
+
system = platform.system()
|
|
62
|
+
if system == "Darwin":
|
|
63
|
+
ram = int(_run(["sysctl", "-n", "hw.memsize"]) or 0) / GIB
|
|
64
|
+
cpu = _run(["sysctl", "-n", "machdep.cpu.brand_string"]) or platform.machine()
|
|
65
|
+
apple = platform.machine() == "arm64"
|
|
66
|
+
return Hardware(os="macos", cpu=cpu, ram_gib=ram, gpu=f"{cpu} GPU" if apple else None, unified=apple)
|
|
67
|
+
|
|
68
|
+
cpu = platform.processor() or platform.machine()
|
|
69
|
+
try:
|
|
70
|
+
with open("/proc/cpuinfo") as f:
|
|
71
|
+
cpu = next((ln.split(":", 1)[1].strip() for ln in f if ln.startswith("model name")), cpu)
|
|
72
|
+
except OSError:
|
|
73
|
+
pass
|
|
74
|
+
gpus = [
|
|
75
|
+
ln.rsplit(",", 1)
|
|
76
|
+
for ln in _run(["nvidia-smi", "--query-gpu=name,memory.total", "--format=csv,noheader,nounits"]).splitlines()
|
|
77
|
+
if "," in ln
|
|
78
|
+
]
|
|
79
|
+
names = [n.strip() for n, _ in gpus]
|
|
80
|
+
vram = sum(float(m) for _, m in gpus) / 1024 if gpus else 0.0
|
|
81
|
+
gpu = None
|
|
82
|
+
if names:
|
|
83
|
+
gpu = names[0] if len(names) == 1 else f"{len(names)}x {names[0]}"
|
|
84
|
+
return Hardware(os=system.lower(), cpu=cpu, ram_gib=_linux_ram_gib(), gpu=gpu, vram_gib=vram)
|