lcode-cli 0.1.1__py3-none-any.whl → 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lcode/__init__.py +1 -1
- lcode/agent.py +110 -23
- lcode/cli.py +19 -5
- lcode/limits.py +41 -0
- lcode/models.toml +21 -23
- lcode/repl.py +126 -8
- lcode/sessions.py +122 -0
- lcode/tools.py +20 -0
- {lcode_cli-0.1.1.dist-info → lcode_cli-0.1.2.dist-info}/METADATA +17 -12
- lcode_cli-0.1.2.dist-info/RECORD +20 -0
- lcode_cli-0.1.1.dist-info/RECORD +0 -18
- {lcode_cli-0.1.1.dist-info → lcode_cli-0.1.2.dist-info}/WHEEL +0 -0
- {lcode_cli-0.1.1.dist-info → lcode_cli-0.1.2.dist-info}/entry_points.txt +0 -0
- {lcode_cli-0.1.1.dist-info → lcode_cli-0.1.2.dist-info}/licenses/LICENSE +0 -0
lcode/__init__.py
CHANGED
lcode/agent.py
CHANGED
|
@@ -18,14 +18,17 @@ from rich.markdown import Markdown
|
|
|
18
18
|
from rich.panel import Panel
|
|
19
19
|
from rich.text import Text
|
|
20
20
|
|
|
21
|
-
from lcode import catalog
|
|
22
|
-
from lcode.config import
|
|
21
|
+
from lcode import catalog, limits, sessions
|
|
22
|
+
from lcode.config import format_tokens
|
|
23
23
|
from lcode.ollama import Ollama, OllamaError
|
|
24
24
|
from lcode.permissions import Permissions
|
|
25
25
|
from lcode.render import MarkdownStreamer
|
|
26
26
|
from lcode.tools import SCHEMAS, Toolbox, is_binary, parse_text_tool_calls, tree, truncate
|
|
27
27
|
|
|
28
|
-
|
|
28
|
+
GPU_MEMORY_ERRORS = ("out of memory", "illegal memory access", "cudamalloc failed")
|
|
29
|
+
SAFE_NUM_BATCH = 512 # Ollama's default prompt batch
|
|
30
|
+
MAX_STEPS_PER_TURN = 150
|
|
31
|
+
MAX_MALFORMED_CALL_RETRIES = 2 # Ollama rejects tool calls whose arguments aren't valid JSON # safety cap on tool-call iterations for one request
|
|
29
32
|
AUTO_COMPACT_RATIO = 0.85 # summarize the history when the context is this full
|
|
30
33
|
PROJECT_FILES = ("AGENTS.md", "LCODE.md", "CLAUDE.md")
|
|
31
34
|
|
|
@@ -100,6 +103,8 @@ class Agent:
|
|
|
100
103
|
self.perms = Permissions(self.console, settings.permission_mode)
|
|
101
104
|
self.tools = Toolbox(self)
|
|
102
105
|
self.session_id = self.new_session_id()
|
|
106
|
+
self.session_name = ""
|
|
107
|
+
self.session_title = ""
|
|
103
108
|
self.ctx_used = 0
|
|
104
109
|
self.last_speed = 0.0
|
|
105
110
|
self.messages: list[dict] = []
|
|
@@ -134,25 +139,63 @@ class Agent:
|
|
|
134
139
|
self.ctx_used = len(self.messages[0]["content"]) // 3
|
|
135
140
|
|
|
136
141
|
def session_file(self) -> Path:
|
|
137
|
-
return
|
|
142
|
+
return sessions.sessions_dir() / f"{self.session_id}.json"
|
|
143
|
+
|
|
144
|
+
def has_conversation(self) -> bool:
|
|
145
|
+
return any(m.get("role") == "user" for m in self.messages)
|
|
138
146
|
|
|
139
147
|
def save(self) -> None:
|
|
148
|
+
if not (self.has_conversation() or self.session_name):
|
|
149
|
+
return # nothing worth resuming
|
|
150
|
+
self.session_title = self.session_title or sessions.title_from(self.messages)
|
|
140
151
|
f = self.session_file()
|
|
141
152
|
f.parent.mkdir(parents=True, exist_ok=True)
|
|
142
|
-
|
|
153
|
+
data = {
|
|
154
|
+
"cwd": str(self.cwd),
|
|
155
|
+
"model": self.settings.model,
|
|
156
|
+
"name": self.session_name,
|
|
157
|
+
"title": self.session_title,
|
|
158
|
+
"messages": self.messages,
|
|
159
|
+
}
|
|
160
|
+
f.write_text(json.dumps(data))
|
|
161
|
+
|
|
162
|
+
def new_session(self) -> None:
|
|
163
|
+
self.reset()
|
|
164
|
+
self.session_id = self.new_session_id()
|
|
165
|
+
self.session_name = ""
|
|
166
|
+
self.session_title = ""
|
|
167
|
+
|
|
168
|
+
def rename(self, name: str) -> None:
|
|
169
|
+
self.session_name = " ".join(name.split())
|
|
170
|
+
self.save()
|
|
171
|
+
|
|
172
|
+
def load(self, info: sessions.SessionInfo) -> str:
|
|
173
|
+
"""Resume a saved session. Returns a note about the working directory, if it changed."""
|
|
174
|
+
try:
|
|
175
|
+
data = json.loads(info.path.read_text())
|
|
176
|
+
except (OSError, ValueError) as e:
|
|
177
|
+
raise OSError(f"can't read session {info.id}: {e}") from e
|
|
178
|
+
note = ""
|
|
179
|
+
saved_cwd = Path(info.cwd) if info.cwd else self.cwd
|
|
180
|
+
if saved_cwd != self.cwd:
|
|
181
|
+
if saved_cwd.is_dir():
|
|
182
|
+
self.cwd = saved_cwd.resolve()
|
|
183
|
+
note = f"Working directory is now {self.cwd}"
|
|
184
|
+
else:
|
|
185
|
+
note = f"The session's directory {saved_cwd} no longer exists; staying in {self.cwd}"
|
|
186
|
+
self.messages = [{"role": "system", "content": self.system_prompt()}, *data.get("messages", [])[1:]]
|
|
187
|
+
self.session_id = info.id
|
|
188
|
+
self.session_name = data.get("name", "")
|
|
189
|
+
self.session_title = data.get("title") or sessions.title_from(self.messages)
|
|
190
|
+
self.tools.read_mtimes.clear() # files may have changed since; the model must read them again
|
|
191
|
+
self.ctx_used = sum(len(json.dumps(m)) for m in self.messages) // 3
|
|
192
|
+
return note
|
|
143
193
|
|
|
144
194
|
def load_latest(self) -> bool:
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
continue
|
|
150
|
-
if data.get("cwd") == str(self.cwd):
|
|
151
|
-
self.messages = [{"role": "system", "content": self.system_prompt()}, *data["messages"][1:]]
|
|
152
|
-
self.session_id = f.stem
|
|
153
|
-
self.ctx_used = sum(len(json.dumps(m)) for m in self.messages) // 3
|
|
154
|
-
return True
|
|
155
|
-
return False
|
|
195
|
+
latest = sessions.list_sessions(self.cwd, limit=1)
|
|
196
|
+
if latest:
|
|
197
|
+
self.load(latest[0])
|
|
198
|
+
return bool(latest)
|
|
156
199
|
|
|
157
200
|
# -- model calls
|
|
158
201
|
def options(self) -> dict:
|
|
@@ -171,18 +214,45 @@ class Agent:
|
|
|
171
214
|
}
|
|
172
215
|
if tools:
|
|
173
216
|
payload["tools"] = tools
|
|
217
|
+
started = False
|
|
174
218
|
try:
|
|
175
|
-
|
|
219
|
+
for chunk in self.ollama.chat_stream(payload):
|
|
220
|
+
started = True
|
|
221
|
+
yield chunk
|
|
176
222
|
except OllamaError as e:
|
|
177
|
-
|
|
223
|
+
error = str(e).lower()
|
|
224
|
+
if not started and think and "think" in error and "support" in error:
|
|
178
225
|
self.settings.think = False
|
|
179
226
|
self.console.print(f"[dim]{self.settings.model} does not support reasoning; continuing without.[/]")
|
|
180
|
-
yield from self.
|
|
227
|
+
yield from self.chat(messages, tools, False)
|
|
181
228
|
return
|
|
182
|
-
if
|
|
229
|
+
if any(marker in error for marker in GPU_MEMORY_ERRORS):
|
|
230
|
+
batch = self.settings.num_batch
|
|
231
|
+
if not started and batch and batch > SAFE_NUM_BATCH:
|
|
232
|
+
# Larger batches read prompts faster but need extra VRAM that isn't always free.
|
|
233
|
+
self.settings.num_batch = SAFE_NUM_BATCH
|
|
234
|
+
self.console.print(
|
|
235
|
+
f"[yellow]The GPU ran out of memory with a prompt batch of {batch}; retrying with "
|
|
236
|
+
f"{SAFE_NUM_BATCH}.[/] [dim]To skip this retry: lcode config set num_batch {SAFE_NUM_BATCH}[/]"
|
|
237
|
+
)
|
|
238
|
+
yield from self.chat(messages, tools, think)
|
|
239
|
+
return
|
|
240
|
+
context = self.settings.context
|
|
241
|
+
if not started and context > catalog.MIN_USEFUL_CONTEXT:
|
|
242
|
+
# The context cache has to fit in memory; halve it until the model loads.
|
|
243
|
+
smaller = max(catalog.MIN_USEFUL_CONTEXT, context // 2)
|
|
244
|
+
self.settings.context = smaller
|
|
245
|
+
limits.record(self.settings.model, smaller)
|
|
246
|
+
self.console.print(
|
|
247
|
+
f"[yellow]{self.settings.model} didn't fit in GPU memory with a {format_tokens(context)} "
|
|
248
|
+
f"context; retrying with {format_tokens(smaller)}.[/] [dim]lcode will start there next "
|
|
249
|
+
"time on this machine.[/]"
|
|
250
|
+
)
|
|
251
|
+
yield from self.chat(messages, tools, think)
|
|
252
|
+
return
|
|
183
253
|
raise OllamaError(
|
|
184
|
-
f"{e}\nThe model ran out of GPU memory. Try a smaller context (/ctx 128k)
|
|
185
|
-
"
|
|
254
|
+
f"{e}\nThe model ran out of GPU memory. Try a smaller context (/ctx 128k), close other programs "
|
|
255
|
+
"using the GPU, or pick a smaller model (/models)."
|
|
186
256
|
) from e
|
|
187
257
|
raise
|
|
188
258
|
|
|
@@ -267,9 +337,26 @@ class Agent:
|
|
|
267
337
|
|
|
268
338
|
def run_turn(self, user_text: str) -> None:
|
|
269
339
|
self.messages.append({"role": "user", "content": self.expand_mentions(user_text)})
|
|
340
|
+
malformed = 0
|
|
270
341
|
for _ in range(MAX_STEPS_PER_TURN):
|
|
271
342
|
self.maybe_compact()
|
|
272
|
-
|
|
343
|
+
try:
|
|
344
|
+
calls = self.assistant_step().get("tool_calls") or []
|
|
345
|
+
except OllamaError as e:
|
|
346
|
+
if "error parsing tool call" not in str(e) or malformed >= MAX_MALFORMED_CALL_RETRIES:
|
|
347
|
+
raise
|
|
348
|
+
malformed += 1
|
|
349
|
+
reason = str(e).rsplit("err=", 1)[-1].strip() if "err=" in str(e) else "invalid JSON"
|
|
350
|
+
self.console.print("[yellow]The model wrote a malformed tool call; asking it to try again.[/]")
|
|
351
|
+
self.messages.append(
|
|
352
|
+
{
|
|
353
|
+
"role": "user",
|
|
354
|
+
"content": f"[lcode] Your last tool call could not be parsed ({reason}): its arguments "
|
|
355
|
+
"must be one complete, valid JSON object. Make the call again. If it writes a file, "
|
|
356
|
+
"make sure the whole content is included and properly escaped.",
|
|
357
|
+
}
|
|
358
|
+
)
|
|
359
|
+
continue
|
|
273
360
|
if not calls:
|
|
274
361
|
break
|
|
275
362
|
for i, call in enumerate(calls):
|
lcode/cli.py
CHANGED
|
@@ -14,7 +14,7 @@ from rich.progress import BarColumn, DownloadColumn, Progress, TextColumn, TimeR
|
|
|
14
14
|
from rich.prompt import Confirm
|
|
15
15
|
from rich.table import Table
|
|
16
16
|
|
|
17
|
-
from lcode import __version__, catalog, config
|
|
17
|
+
from lcode import __version__, catalog, config, limits
|
|
18
18
|
from lcode.agent import Agent, Settings
|
|
19
19
|
from lcode.catalog import ModelSpec
|
|
20
20
|
from lcode.config import ConfigError, format_tokens, parse_context
|
|
@@ -78,9 +78,9 @@ def choose_context(
|
|
|
78
78
|
if requested:
|
|
79
79
|
ctx = requested
|
|
80
80
|
elif spec:
|
|
81
|
-
ctx = spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT
|
|
81
|
+
ctx = limits.cap(model, spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT)
|
|
82
82
|
else:
|
|
83
|
-
ctx = 32768
|
|
83
|
+
ctx = limits.cap(model, 32768)
|
|
84
84
|
if limit and ctx > limit:
|
|
85
85
|
ctx, note = limit, f"capped at the model maximum of {format_tokens(limit)}"
|
|
86
86
|
if spec and requested and spec.memory_gib(ctx) > hw.budget_gib:
|
|
@@ -272,6 +272,13 @@ def cmd_doctor(args) -> None:
|
|
|
272
272
|
if spec:
|
|
273
273
|
detail += f" · ~{spec.memory_gib(ctx):.0f} GB needed, ~{hw.budget_gib:.0f} GB available"
|
|
274
274
|
line("Context", detail + (f" ({note})" if note else ""), None if note else True)
|
|
275
|
+
if limits.get(model):
|
|
276
|
+
line(
|
|
277
|
+
"Limit",
|
|
278
|
+
f"{format_tokens(limits.get(model))} for {model}: larger contexts ran out of GPU memory "
|
|
279
|
+
f"here (delete {limits.path()} to try again)",
|
|
280
|
+
None,
|
|
281
|
+
)
|
|
275
282
|
loaded = [m for m in ollama.running() if m.get("name") == model or m.get("model") == model]
|
|
276
283
|
if loaded:
|
|
277
284
|
m = loaded[0]
|
|
@@ -370,7 +377,7 @@ def cmd_chat(args) -> None:
|
|
|
370
377
|
permission_mode=mode,
|
|
371
378
|
)
|
|
372
379
|
agent = Agent(ollama, settings, cwd, console=console)
|
|
373
|
-
repl(agent, prompt=args.prompt,
|
|
380
|
+
repl(agent, prompt=args.prompt, hardware=hw, cont=args.cont, resume=args.resume)
|
|
374
381
|
|
|
375
382
|
|
|
376
383
|
def build_parser() -> argparse.ArgumentParser:
|
|
@@ -384,7 +391,14 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
384
391
|
parser.add_argument("-m", "--model", help="catalog key (see `lcode models`) or any installed Ollama model")
|
|
385
392
|
parser.add_argument("--context", "--ctx", dest="context", help="context window, e.g. 65536, 128k or 1m")
|
|
386
393
|
parser.add_argument("-r", "--repo", default=".", help="working directory (default: current directory)")
|
|
387
|
-
parser.add_argument("-c", "--continue", dest="cont", action="store_true", help="
|
|
394
|
+
parser.add_argument("-c", "--continue", dest="cont", action="store_true", help="continue the last session here")
|
|
395
|
+
parser.add_argument(
|
|
396
|
+
"--resume",
|
|
397
|
+
nargs="?",
|
|
398
|
+
const="",
|
|
399
|
+
metavar="SESSION",
|
|
400
|
+
help="resume a saved session: pick from a list, or give its number, name or id",
|
|
401
|
+
)
|
|
388
402
|
parser.add_argument("--auto-edit", action="store_true", help="apply file edits without asking")
|
|
389
403
|
parser.add_argument("--yolo", action="store_true", help="never ask for permission (edits and commands)")
|
|
390
404
|
parser.add_argument("--no-think", action="store_true", help="disable model reasoning (faster, less accurate)")
|
lcode/limits.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Context sizes learned to be too large on this machine, from out-of-memory errors.
|
|
2
|
+
|
|
3
|
+
When a model fails to load because its context doesn't fit in GPU memory, lcode retries with half the
|
|
4
|
+
context and records the size that was tried, so later sessions start there instead of failing first.
|
|
5
|
+
Delete the file (see `path()`) to let lcode try larger contexts again, e.g. after a GPU upgrade.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from lcode import config
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def path() -> Path:
|
|
17
|
+
return config.STATE_DIR / "limits.json"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _load() -> dict[str, int]:
|
|
21
|
+
try:
|
|
22
|
+
data = json.loads(path().read_text())
|
|
23
|
+
except (OSError, ValueError):
|
|
24
|
+
return {}
|
|
25
|
+
return {k: int(v) for k, v in data.items() if isinstance(v, int)} if isinstance(data, dict) else {}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def get(model: str) -> int | None:
|
|
29
|
+
return _load().get(model)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def record(model: str, context: int) -> None:
|
|
33
|
+
data = _load()
|
|
34
|
+
data[model] = min(context, data.get(model, context))
|
|
35
|
+
path().parent.mkdir(parents=True, exist_ok=True)
|
|
36
|
+
path().write_text(json.dumps(data, indent=2, sort_keys=True) + "\n")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def cap(model: str, context: int) -> int:
|
|
40
|
+
limit = get(model)
|
|
41
|
+
return min(context, limit) if limit else context
|
lcode/models.toml
CHANGED
|
@@ -29,7 +29,6 @@ moe = true
|
|
|
29
29
|
size_gb = 22.6
|
|
30
30
|
max_context = 262144
|
|
31
31
|
kv_kib_per_token = 22
|
|
32
|
-
num_batch = 1024
|
|
33
32
|
swe_bench_verified = 73.4
|
|
34
33
|
tested = true
|
|
35
34
|
notes = "Default. Strongest tool calling among local MoE coders; fast even with experts in RAM."
|
|
@@ -44,8 +43,8 @@ moe = false
|
|
|
44
43
|
size_gb = 17.7
|
|
45
44
|
max_context = 262144
|
|
46
45
|
kv_kib_per_token = 68
|
|
47
|
-
tested =
|
|
48
|
-
notes = "Newest Qwen. Dense:
|
|
46
|
+
tested = true
|
|
47
|
+
notes = "Newest Qwen. Dense: clean, accurate tool use; ~8 tok/s when split on a 12 GB GPU, fast when it fits in GPU/unified memory."
|
|
49
48
|
|
|
50
49
|
[[model]]
|
|
51
50
|
key = "qwen3.6-27b"
|
|
@@ -57,8 +56,8 @@ moe = false
|
|
|
57
56
|
size_gb = 17.8
|
|
58
57
|
max_context = 262144
|
|
59
58
|
kv_kib_per_token = 68
|
|
60
|
-
tested =
|
|
61
|
-
notes = "Dense coding model. Best on 24 GB+ GPUs or
|
|
59
|
+
tested = true
|
|
60
|
+
notes = "Dense coding model: clean, accurate tool use; ~7 tok/s when split on a 12 GB GPU. Best on 24 GB+ GPUs or 48 GB+ Macs."
|
|
62
61
|
|
|
63
62
|
[[model]]
|
|
64
63
|
key = "laguna-xs-2.1"
|
|
@@ -70,10 +69,9 @@ moe = true
|
|
|
70
69
|
size_gb = 20.3
|
|
71
70
|
max_context = 262144
|
|
72
71
|
kv_kib_per_token = 40
|
|
73
|
-
kv_estimated = true
|
|
74
72
|
swe_bench_verified = 70.9
|
|
75
|
-
tested =
|
|
76
|
-
notes = "Agentic-coding MoE built for local machines;
|
|
73
|
+
tested = true
|
|
74
|
+
notes = "Agentic-coding MoE built for local machines; 15-38 tok/s at 256K on a 12 GB GPU. Full KV cache in 10 of 40 layers."
|
|
77
75
|
|
|
78
76
|
[[model]]
|
|
79
77
|
key = "nemotron-3.5-lightning"
|
|
@@ -85,21 +83,8 @@ moe = true
|
|
|
85
83
|
size_gb = 25.4
|
|
86
84
|
max_context = 1048576
|
|
87
85
|
kv_kib_per_token = 7
|
|
88
|
-
tested =
|
|
89
|
-
notes = "Hybrid Mamba-Transformer: tiny KV cache, up to 1M tokens
|
|
90
|
-
|
|
91
|
-
[[model]]
|
|
92
|
-
key = "gpt-oss-20b"
|
|
93
|
-
tag = "gpt-oss:20b"
|
|
94
|
-
name = "gpt-oss 20B"
|
|
95
|
-
publisher = "OpenAI"
|
|
96
|
-
params = "21B MoE · 3.6B active"
|
|
97
|
-
moe = true
|
|
98
|
-
size_gb = 13.8
|
|
99
|
-
max_context = 131072
|
|
100
|
-
kv_kib_per_token = 24
|
|
101
|
-
tested = false
|
|
102
|
-
notes = "Compact MoE with 128K context; fits 16 GB GPUs and 24 GB Macs."
|
|
86
|
+
tested = true
|
|
87
|
+
notes = "Hybrid Mamba-Transformer: tiny KV cache, up to 1M tokens (512K on a 12 GB GPU); 44 tok/s on a 12 GB GPU."
|
|
103
88
|
|
|
104
89
|
[[model]]
|
|
105
90
|
key = "qwen3.5-9b"
|
|
@@ -114,6 +99,19 @@ kv_kib_per_token = 32
|
|
|
114
99
|
tested = true
|
|
115
100
|
notes = "Small and capable: 64 tok/s fully on a 12 GB GPU at 128K context. For 8-12 GB GPUs and 16-24 GB Macs."
|
|
116
101
|
|
|
102
|
+
[[model]]
|
|
103
|
+
key = "gpt-oss-20b"
|
|
104
|
+
tag = "gpt-oss:20b"
|
|
105
|
+
name = "gpt-oss 20B"
|
|
106
|
+
publisher = "OpenAI"
|
|
107
|
+
params = "21B MoE · 3.6B active"
|
|
108
|
+
moe = true
|
|
109
|
+
size_gb = 13.8
|
|
110
|
+
max_context = 131072
|
|
111
|
+
kv_kib_per_token = 24
|
|
112
|
+
tested = true
|
|
113
|
+
notes = "Compact MoE with 128K context: 43 tok/s on a 12 GB GPU. Sometimes invents tool arguments but corrects itself."
|
|
114
|
+
|
|
117
115
|
[[model]]
|
|
118
116
|
key = "qwen3.5-4b"
|
|
119
117
|
tag = "qwen3.5:4b"
|
lcode/repl.py
CHANGED
|
@@ -14,11 +14,13 @@ from prompt_toolkit.formatted_text import HTML
|
|
|
14
14
|
from prompt_toolkit.history import FileHistory
|
|
15
15
|
from prompt_toolkit.key_binding import KeyBindings
|
|
16
16
|
from rich.console import Group
|
|
17
|
+
from rich.markup import escape
|
|
17
18
|
from rich.panel import Panel
|
|
18
19
|
from rich.rule import Rule
|
|
20
|
+
from rich.table import Table
|
|
19
21
|
from rich.text import Text
|
|
20
22
|
|
|
21
|
-
from lcode import __version__, catalog
|
|
23
|
+
from lcode import __version__, catalog, limits, sessions
|
|
22
24
|
from lcode.agent import INIT_PROMPT, Agent
|
|
23
25
|
from lcode.catalog import MIN_USEFUL_CONTEXT
|
|
24
26
|
from lcode.config import PERMISSION_MODES, STATE_DIR, ConfigError, format_tokens, parse_context
|
|
@@ -30,6 +32,8 @@ COMMANDS = {
|
|
|
30
32
|
"/help": "Show this help",
|
|
31
33
|
"/init": "Analyze the repo and write an AGENTS.md guide (loaded at every start)",
|
|
32
34
|
"/clear": "Start a fresh conversation",
|
|
35
|
+
"/rename": "Name this session so you can find it later, e.g. /rename auth refactor",
|
|
36
|
+
"/resume": "Resume a saved session: pick from a list, or /resume <number|name> (/resume all: every folder)",
|
|
33
37
|
"/compact": "Summarize the conversation to free context (optional: what to focus on)",
|
|
34
38
|
"/context": "Show context-window usage",
|
|
35
39
|
"/ctx": "Show or change the context window, e.g. /ctx 128k",
|
|
@@ -105,6 +109,7 @@ def build_session(agent: Agent) -> PromptSession:
|
|
|
105
109
|
f" <b>{html.escape(s.model)}</b> · ctx {format_tokens(agent.ctx_used)}/{format_tokens(s.context)} "
|
|
106
110
|
f"({pct:.0f}%) · mode <{color}>{agent.perms.mode}</{color}> (shift+tab) · "
|
|
107
111
|
f"think {'on' if s.think else 'off'} · {html.escape(agent.cwd.name)}/"
|
|
112
|
+
+ (f" · <b>{html.escape(agent.session_name)}</b>" if agent.session_name else "")
|
|
108
113
|
)
|
|
109
114
|
|
|
110
115
|
STATE_DIR.mkdir(parents=True, exist_ok=True)
|
|
@@ -167,9 +172,19 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
|
|
|
167
172
|
)
|
|
168
173
|
c.print(Panel(f"{rows}\n\n [dim]{keys}[/]", title="lcode commands", border_style="cyan"))
|
|
169
174
|
elif cmd == "/clear":
|
|
170
|
-
agent.
|
|
171
|
-
|
|
172
|
-
|
|
175
|
+
agent.new_session()
|
|
176
|
+
c.print("[green]Started a new conversation.[/] The previous one is saved; /resume brings it back.")
|
|
177
|
+
elif cmd == "/rename":
|
|
178
|
+
if not arg:
|
|
179
|
+
current = f"'{escape(agent.session_name)}'" if agent.session_name else "not named yet"
|
|
180
|
+
c.print(f"This session is {current}. Name it with /rename <name>.")
|
|
181
|
+
else:
|
|
182
|
+
agent.rename(arg)
|
|
183
|
+
c.print(f"[green]Session named '{escape(agent.session_name)}'.[/] Find it later with /resume.")
|
|
184
|
+
elif cmd == "/resume":
|
|
185
|
+
info = choose_session(agent, arg)
|
|
186
|
+
if info:
|
|
187
|
+
resume_session(agent, info)
|
|
173
188
|
elif cmd == "/compact":
|
|
174
189
|
agent.compact(arg)
|
|
175
190
|
elif cmd == "/context":
|
|
@@ -208,7 +223,7 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
|
|
|
208
223
|
s.model = name
|
|
209
224
|
s.num_batch = spec.num_batch if spec and name == spec.local_name else None
|
|
210
225
|
if spec: # size the context for the new model: largest window that fits this machine
|
|
211
|
-
s.context = spec.fit(hardware)[0] or MIN_USEFUL_CONTEXT
|
|
226
|
+
s.context = limits.cap(name, spec.fit(hardware)[0] or MIN_USEFUL_CONTEXT)
|
|
212
227
|
agent.messages[0]["content"] = agent.system_prompt()
|
|
213
228
|
c.print(f"[green]Switched to {name}[/] (context {format_tokens(s.context)}).")
|
|
214
229
|
elif cmd == "/think":
|
|
@@ -241,9 +256,110 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
|
|
|
241
256
|
return True
|
|
242
257
|
|
|
243
258
|
|
|
244
|
-
def
|
|
245
|
-
if
|
|
246
|
-
|
|
259
|
+
def print_sessions(agent: Agent, found: list[sessions.SessionInfo], all_dirs: bool) -> None:
|
|
260
|
+
where = "all folders" if all_dirs else str(agent.cwd)
|
|
261
|
+
table = Table(title=f"Saved sessions · {where}", title_justify="left", header_style="bold")
|
|
262
|
+
table.add_column("#", justify="right", style="cyan")
|
|
263
|
+
table.add_column("Session")
|
|
264
|
+
table.add_column("Last used", no_wrap=True)
|
|
265
|
+
table.add_column("Requests", justify="right")
|
|
266
|
+
if all_dirs:
|
|
267
|
+
table.add_column("Folder", overflow="fold")
|
|
268
|
+
for i, info in enumerate(found, 1):
|
|
269
|
+
if info.name:
|
|
270
|
+
label = f"[bold]{escape(info.name)}[/]\n[dim]{escape(info.title or '(no requests yet)')}[/]"
|
|
271
|
+
else:
|
|
272
|
+
label = escape(info.label)
|
|
273
|
+
if info.id == agent.session_id:
|
|
274
|
+
label += " [green](current)[/]"
|
|
275
|
+
row = [str(i), label, sessions.age(info.updated), str(info.turns)]
|
|
276
|
+
if all_dirs:
|
|
277
|
+
row.append(escape(info.cwd))
|
|
278
|
+
table.add_row(*row)
|
|
279
|
+
agent.console.print(table)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def choose_session(agent: Agent, query: str = "") -> sessions.SessionInfo | None:
|
|
283
|
+
"""Find a session by number/name/id, or list them and ask. `all` lists every folder."""
|
|
284
|
+
c = agent.console
|
|
285
|
+
words = query.split()
|
|
286
|
+
all_dirs = bool(words) and words[0].lower() in ("all", "--all", "-a")
|
|
287
|
+
query = " ".join(words[1:] if all_dirs else words)
|
|
288
|
+
found = sessions.list_sessions(None if all_dirs else agent.cwd)
|
|
289
|
+
if not found and not all_dirs:
|
|
290
|
+
found, all_dirs = sessions.list_sessions(None), True
|
|
291
|
+
if found and not query:
|
|
292
|
+
c.print("[dim]No saved sessions in this folder; showing all folders.[/]")
|
|
293
|
+
if not found:
|
|
294
|
+
c.print("No saved sessions yet. Sessions are saved after every request.")
|
|
295
|
+
return None
|
|
296
|
+
if query:
|
|
297
|
+
match = sessions.find(query, found)
|
|
298
|
+
if match is None and not all_dirs:
|
|
299
|
+
match = sessions.find(query, sessions.list_sessions(None))
|
|
300
|
+
if match:
|
|
301
|
+
return match
|
|
302
|
+
c.print(f"[yellow]No session matches '{escape(query)}'.[/]")
|
|
303
|
+
print_sessions(agent, found, all_dirs)
|
|
304
|
+
try:
|
|
305
|
+
answer = input(" Resume which session? (number or name, Enter to cancel): ").strip()
|
|
306
|
+
except EOFError:
|
|
307
|
+
return None
|
|
308
|
+
if not answer:
|
|
309
|
+
return None
|
|
310
|
+
match = sessions.find(answer, found)
|
|
311
|
+
if match is None:
|
|
312
|
+
c.print(f"[yellow]No session matches '{escape(answer)}'.[/]")
|
|
313
|
+
return match
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def resume_session(agent: Agent, info: sessions.SessionInfo) -> None:
|
|
317
|
+
c = agent.console
|
|
318
|
+
if info.id == agent.session_id:
|
|
319
|
+
c.print("That's the current session.")
|
|
320
|
+
return
|
|
321
|
+
agent.save()
|
|
322
|
+
try:
|
|
323
|
+
note = agent.load(info)
|
|
324
|
+
except OSError as e:
|
|
325
|
+
c.print(f"[red]{e}[/]")
|
|
326
|
+
return
|
|
327
|
+
c.print(
|
|
328
|
+
f"[green]Resumed[/] [bold]{escape(info.label)}[/] "
|
|
329
|
+
f"[dim]({info.turns} request{'' if info.turns == 1 else 's'}, last used {sessions.age(info.updated)})[/]"
|
|
330
|
+
)
|
|
331
|
+
if note:
|
|
332
|
+
c.print(f"[yellow]{escape(note)}[/]")
|
|
333
|
+
print_recap(agent)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def print_recap(agent: Agent) -> None:
|
|
337
|
+
"""Show the last request and the start of the last answer, so it's clear where things left off."""
|
|
338
|
+
last_user = next((m for m in reversed(agent.messages) if m.get("role") == "user"), None)
|
|
339
|
+
last_answer = next(
|
|
340
|
+
(m for m in reversed(agent.messages) if m.get("role") == "assistant" and m.get("content", "").strip()), None
|
|
341
|
+
)
|
|
342
|
+
if not last_user:
|
|
343
|
+
return
|
|
344
|
+
lines = [f"[bold]You:[/] {escape(sessions.title_from([last_user]))}"]
|
|
345
|
+
if last_answer:
|
|
346
|
+
answer = " ".join(last_answer["content"].split())
|
|
347
|
+
answer = answer if len(answer) <= 300 else answer[:299].rstrip() + "…"
|
|
348
|
+
lines.append(f"[bold]lcode:[/] {escape(answer)}")
|
|
349
|
+
agent.console.print(Panel("\n".join(lines), title="Where you left off", title_align="left", border_style="dim"))
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def repl(agent: Agent, prompt: str | None, hardware: Hardware, cont: bool = False, resume: str | None = None) -> None:
|
|
353
|
+
if cont:
|
|
354
|
+
if agent.load_latest():
|
|
355
|
+
agent.console.print(f"[green]Continuing[/] [bold]{escape(agent.session_name or agent.session_title)}[/]")
|
|
356
|
+
print_recap(agent)
|
|
357
|
+
else:
|
|
358
|
+
agent.console.print("[dim]No saved session in this folder yet; starting a new one.[/]")
|
|
359
|
+
elif resume is not None:
|
|
360
|
+
info = choose_session(agent, resume)
|
|
361
|
+
if info:
|
|
362
|
+
resume_session(agent, info)
|
|
247
363
|
if prompt:
|
|
248
364
|
run_safely(agent, prompt)
|
|
249
365
|
return
|
|
@@ -260,6 +376,8 @@ def repl(agent: Agent, prompt: str | None, resume: bool, hardware: Hardware) ->
|
|
|
260
376
|
break
|
|
261
377
|
if not line:
|
|
262
378
|
continue
|
|
379
|
+
if line.lower() in sessions.EXIT_WORDS: # people type these expecting to quit, not to ask the model
|
|
380
|
+
break
|
|
263
381
|
if line.startswith("/") and not line.startswith("//"):
|
|
264
382
|
if not handle_command(agent, line, hardware):
|
|
265
383
|
break
|
lcode/sessions.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Saved conversations: listing, naming and finding sessions to resume."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime as dt
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from lcode import config
|
|
12
|
+
|
|
13
|
+
TITLE_LENGTH = 70
|
|
14
|
+
SUMMARY_PREFIX = "[Summary of our conversation so far]" # first message of a compacted conversation
|
|
15
|
+
EXIT_WORDS = {"exit", "quit", ":q", ":quit"}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class SessionInfo:
|
|
20
|
+
id: str
|
|
21
|
+
path: Path
|
|
22
|
+
cwd: str
|
|
23
|
+
name: str # set with /rename; empty if never named
|
|
24
|
+
title: str # the first request, shortened
|
|
25
|
+
model: str
|
|
26
|
+
updated: float # modification time of the session file
|
|
27
|
+
turns: int # number of requests from the user
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def label(self) -> str:
|
|
31
|
+
return self.name or self.title or "(no requests yet)"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def sessions_dir() -> Path:
|
|
35
|
+
return config.STATE_DIR / "sessions"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def title_from(messages: list[dict]) -> str:
|
|
39
|
+
"""A one-line title from the first request, without attached files or compaction summaries."""
|
|
40
|
+
compacted = False
|
|
41
|
+
for message in messages:
|
|
42
|
+
if message.get("role") != "user":
|
|
43
|
+
continue
|
|
44
|
+
content = message.get("content", "")
|
|
45
|
+
if content.startswith(SUMMARY_PREFIX):
|
|
46
|
+
compacted = True
|
|
47
|
+
continue
|
|
48
|
+
text = re.split(r"\n\n<(?:file|directory) path=", content, maxsplit=1)[0]
|
|
49
|
+
text = " ".join(text.split())
|
|
50
|
+
if text:
|
|
51
|
+
return text if len(text) <= TITLE_LENGTH else text[: TITLE_LENGTH - 1].rstrip() + "…"
|
|
52
|
+
return "Compacted conversation" if compacted else ""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def read_info(path: Path) -> SessionInfo | None:
|
|
56
|
+
try:
|
|
57
|
+
data = json.loads(path.read_text())
|
|
58
|
+
updated = path.stat().st_mtime
|
|
59
|
+
except (OSError, ValueError):
|
|
60
|
+
return None
|
|
61
|
+
messages = data.get("messages") or []
|
|
62
|
+
return SessionInfo(
|
|
63
|
+
id=path.stem,
|
|
64
|
+
path=path,
|
|
65
|
+
cwd=data.get("cwd", ""),
|
|
66
|
+
name=data.get("name", ""),
|
|
67
|
+
title=data.get("title") or title_from(messages),
|
|
68
|
+
model=data.get("model", ""),
|
|
69
|
+
updated=updated,
|
|
70
|
+
turns=sum(1 for m in messages if m.get("role") == "user"),
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def list_sessions(cwd: Path | str | None = None, limit: int = 50) -> list[SessionInfo]:
|
|
75
|
+
"""Saved sessions, most recently used first; only those for `cwd` if given."""
|
|
76
|
+
directory = sessions_dir()
|
|
77
|
+
if not directory.is_dir():
|
|
78
|
+
return []
|
|
79
|
+
files = sorted(directory.glob("*.json"), key=lambda p: p.stat().st_mtime, reverse=True)
|
|
80
|
+
found = []
|
|
81
|
+
for f in files:
|
|
82
|
+
info = read_info(f)
|
|
83
|
+
if info and (info.turns or info.name) and (cwd is None or info.cwd == str(cwd)):
|
|
84
|
+
found.append(info)
|
|
85
|
+
if len(found) >= limit:
|
|
86
|
+
break
|
|
87
|
+
return found
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def find(query: str, sessions: list[SessionInfo]) -> SessionInfo | None:
|
|
91
|
+
"""Match a list number (1-based), a name, a session id (or unique prefix) or a unique title fragment."""
|
|
92
|
+
q = query.strip()
|
|
93
|
+
if not q:
|
|
94
|
+
return None
|
|
95
|
+
if q.isdigit() and 1 <= int(q) <= len(sessions):
|
|
96
|
+
return sessions[int(q) - 1]
|
|
97
|
+
lowered = q.lower()
|
|
98
|
+
for s in sessions:
|
|
99
|
+
if s.name and s.name.lower() == lowered:
|
|
100
|
+
return s
|
|
101
|
+
for candidates in (
|
|
102
|
+
[s for s in sessions if s.id == q or s.id.startswith(q)],
|
|
103
|
+
[s for s in sessions if lowered in s.label.lower()],
|
|
104
|
+
):
|
|
105
|
+
if len(candidates) == 1:
|
|
106
|
+
return candidates[0]
|
|
107
|
+
return None
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def age(timestamp: float, now: dt.datetime | None = None) -> str:
|
|
111
|
+
now = now or dt.datetime.now()
|
|
112
|
+
then = dt.datetime.fromtimestamp(timestamp)
|
|
113
|
+
seconds = (now - then).total_seconds()
|
|
114
|
+
if seconds < 60:
|
|
115
|
+
return "just now"
|
|
116
|
+
if seconds < 3600:
|
|
117
|
+
return f"{int(seconds // 60)} min ago"
|
|
118
|
+
if then.date() == now.date():
|
|
119
|
+
return f"{int(seconds // 3600)} h ago"
|
|
120
|
+
if then.date() == (now - dt.timedelta(days=1)).date():
|
|
121
|
+
return "yesterday"
|
|
122
|
+
return then.strftime("%b %d") if then.year == now.year else then.strftime("%Y-%m-%d")
|
lcode/tools.py
CHANGED
|
@@ -4,6 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import difflib
|
|
6
6
|
import fnmatch
|
|
7
|
+
import inspect
|
|
7
8
|
import json
|
|
8
9
|
import os
|
|
9
10
|
import queue
|
|
@@ -265,6 +266,9 @@ class Toolbox:
|
|
|
265
266
|
fn = getattr(self, f"t_{name}", None)
|
|
266
267
|
if fn is None:
|
|
267
268
|
return f"Error: unknown tool '{name}'. Available: {', '.join(sorted(TOOL_NAMES))}"
|
|
269
|
+
problem = self._argument_problem(fn, args)
|
|
270
|
+
if problem:
|
|
271
|
+
return f"Error: bad arguments for {name}: {problem}"
|
|
268
272
|
try:
|
|
269
273
|
return fn(**args)
|
|
270
274
|
except ToolError as e:
|
|
@@ -274,6 +278,22 @@ class Toolbox:
|
|
|
274
278
|
except Exception as e: # surface every failure to the model instead of crashing the session
|
|
275
279
|
return f"Error: {type(e).__name__}: {e}"
|
|
276
280
|
|
|
281
|
+
@staticmethod
|
|
282
|
+
def _argument_problem(fn, args: dict) -> str:
|
|
283
|
+
"""Explain unknown or missing arguments in terms the model can act on."""
|
|
284
|
+
params = inspect.signature(fn).parameters
|
|
285
|
+
unknown = [a for a in args if a not in params]
|
|
286
|
+
missing = [p for p, spec in params.items() if spec.default is inspect.Parameter.empty and p not in args]
|
|
287
|
+
if not unknown and not missing:
|
|
288
|
+
return ""
|
|
289
|
+
parts = []
|
|
290
|
+
if unknown:
|
|
291
|
+
parts.append(f"unknown argument(s) {', '.join(map(repr, unknown))}")
|
|
292
|
+
if missing:
|
|
293
|
+
parts.append(f"missing required argument(s) {', '.join(map(repr, missing))}")
|
|
294
|
+
valid = ", ".join(p if params[p].default is inspect.Parameter.empty else f"{p} (optional)" for p in params)
|
|
295
|
+
return f"{'; '.join(parts)}. Valid arguments: {valid}."
|
|
296
|
+
|
|
277
297
|
# -- read-only
|
|
278
298
|
def t_read_file(self, path: str, offset: int = 1, limit: int = 2000) -> str:
|
|
279
299
|
p = self.resolve(path)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: lcode-cli
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: A local-first terminal coding agent powered by open-weight models on your own GPU or Mac, via Ollama.
|
|
5
5
|
Project-URL: Homepage, https://nasser1941.github.io/lcode/
|
|
6
6
|
Project-URL: Documentation, https://nasser1941.github.io/lcode/
|
|
@@ -45,6 +45,7 @@ Description-Content-Type: text/markdown
|
|
|
45
45
|
<p align="center">
|
|
46
46
|
<a href="https://github.com/nasser1941/lcode/actions/workflows/ci.yml"><img src="https://github.com/nasser1941/lcode/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
47
47
|
<a href="https://nasser1941.github.io/lcode/"><img src="https://github.com/nasser1941/lcode/actions/workflows/docs.yml/badge.svg" alt="Docs"></a>
|
|
48
|
+
<a href="https://pypi.org/project/lcode-cli/"><img src="https://img.shields.io/pypi/v/lcode-cli.svg" alt="PyPI"></a>
|
|
48
49
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
|
|
49
50
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
|
|
50
51
|
<img src="https://img.shields.io/badge/platform-Linux%20%7C%20macOS%20(Apple%20Silicon)-lightgrey.svg" alt="Linux and macOS">
|
|
@@ -75,7 +76,8 @@ through [Ollama](https://ollama.com), so your code never leaves your machine.
|
|
|
75
76
|
largest context that fits: `lcode setup` does it in one step.
|
|
76
77
|
- **Safe by default.** Every edit is shown as a diff and every command that isn't read-only needs
|
|
77
78
|
your approval. `auto-edit` and `yolo` modes when you want speed.
|
|
78
|
-
- **Bring your own model.** Eight curated open-weight models, or any Ollama model
|
|
79
|
+
- **Bring your own model.** Eight curated open-weight models, all tested end to end, or any Ollama model
|
|
80
|
+
with tool calling.
|
|
79
81
|
- **Private.** No telemetry, no accounts, no API keys.
|
|
80
82
|
|
|
81
83
|
## Install
|
|
@@ -88,14 +90,16 @@ curl -fsSL https://nasser1941.github.io/lcode/install.sh | bash
|
|
|
88
90
|
|
|
89
91
|
The installer sets up lcode with its own Python via [uv](https://docs.astral.sh/uv/), checks for
|
|
90
92
|
[Ollama](https://ollama.com) (0.30+) and runs `lcode setup`, which picks a model for your hardware
|
|
91
|
-
and downloads it. Prefer manual steps?
|
|
92
|
-
|
|
93
|
+
and downloads it. Prefer manual steps? lcode is on [PyPI](https://pypi.org/project/lcode-cli/) as
|
|
94
|
+
`lcode-cli`:
|
|
93
95
|
|
|
94
96
|
```bash
|
|
95
|
-
uv tool install
|
|
97
|
+
uv tool install lcode-cli # or: pipx install lcode-cli
|
|
96
98
|
lcode setup
|
|
97
99
|
```
|
|
98
100
|
|
|
101
|
+
See the [installation guide](https://nasser1941.github.io/lcode/installation/) for details.
|
|
102
|
+
|
|
99
103
|
## Quickstart
|
|
100
104
|
|
|
101
105
|
```bash
|
|
@@ -113,9 +117,10 @@ lcode
|
|
|
113
117
|
| | |
|
|
114
118
|
|---|---|
|
|
115
119
|
| `lcode -p "…"` | one request, no interaction (scripts, hooks) |
|
|
116
|
-
| `lcode -c` | continue the last session
|
|
120
|
+
| `lcode -c` / `lcode --resume` | continue the last session here / pick a saved session from a list |
|
|
117
121
|
| `lcode --model qwen3.5-9b --context 128k` | pick a model and context window for this session |
|
|
118
122
|
| `lcode models` / `lcode doctor` | what fits this machine / check the installation |
|
|
123
|
+
| `/rename`, `/resume` | name the current session, resume a saved one |
|
|
119
124
|
| `/model`, `/ctx 128k`, `/compact`, `/help` | switch model, resize context, summarize, list commands |
|
|
120
125
|
|
|
121
126
|
## Models
|
|
@@ -123,12 +128,12 @@ lcode
|
|
|
123
128
|
| Key | Model | Download | Max context | |
|
|
124
129
|
|---|---|---|---|---|
|
|
125
130
|
| `qwen3.6-35b` | Qwen3.6 35B-A3B Coding (MoE, 3B active) | 22.6 GB | 256K | **default**, tested |
|
|
126
|
-
| `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | |
|
|
127
|
-
| `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | |
|
|
128
|
-
| `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | |
|
|
129
|
-
| `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | |
|
|
130
|
-
| `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | |
|
|
131
|
+
| `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | tested |
|
|
132
|
+
| `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | tested |
|
|
133
|
+
| `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | tested |
|
|
134
|
+
| `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | tested |
|
|
131
135
|
| `qwen3.5-9b` | Qwen3.5 9B (dense) | 6.6 GB | 256K | tested |
|
|
136
|
+
| `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | tested |
|
|
132
137
|
| `qwen3.5-4b` | Qwen3.5 4B (dense) | 3.4 GB | 256K | tested |
|
|
133
138
|
|
|
134
139
|
What `lcode setup` picks for common machines:
|
|
@@ -140,7 +145,7 @@ What `lcode setup` picks for common machines:
|
|
|
140
145
|
| Mac with M4 Pro / M4 Max, 36 GB | qwen3.6-35b | 64K |
|
|
141
146
|
| Mac with M4 Pro, 48 GB · M4 Max, 64 GB+ | qwen3.6-35b | 256K |
|
|
142
147
|
| NVIDIA 8–24 GB + 32 GB RAM | qwen3.6-35b | 256K |
|
|
143
|
-
| NVIDIA 8 GB + 16 GB RAM |
|
|
148
|
+
| NVIDIA 8 GB + 16 GB RAM | gpt-oss-20b | 64K |
|
|
144
149
|
|
|
145
150
|
On an RTX 4080 Laptop GPU (12 GB) the default model generates 50–60 tokens/s at 256K context, and
|
|
146
151
|
qwen3.5-9b 64 tokens/s at 128K. See
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
lcode/__init__.py,sha256=DCFKMt5wGr6X-Z5XXGSODRrKPH4BQtYY3wBJ4PTfh08,117
|
|
2
|
+
lcode/__main__.py,sha256=s5N5pryX2FAu93GGZ3FxUrIXHWGzxz66lKUoaV_h8Ws,35
|
|
3
|
+
lcode/agent.py,sha256=FLWZK97MH15r7feTiYFZfbhWU_EQ9PEvwGBXbGb3AfA,21807
|
|
4
|
+
lcode/catalog.py,sha256=pZ1azX9OH2r4JE2c6E5OfwywpgS39-W2oXzoCHJhmtM,4141
|
|
5
|
+
lcode/cli.py,sha256=kRpZNze1Gsa33Z-l7eLJYXG5vFzozqOrA-KCRduJ6tU,19337
|
|
6
|
+
lcode/config.py,sha256=NvsiWFyGyDpEKHwWDrW643yYLRG4Omp8Qnp_68K2a_Y,5038
|
|
7
|
+
lcode/hardware.py,sha256=2VbknZCxTZJwJUD8YWdQ7ryEQo_LcjtoOBOHOU9Y0fw,3085
|
|
8
|
+
lcode/limits.py,sha256=oJzVQoMyTZ0_Fkpu5t3qxrf9v7Hv9S8U8BRQR5M1aKA,1237
|
|
9
|
+
lcode/models.toml,sha256=lt2G64zzo_Ryxv2ISRSNZoVD_JqCqYqsM01n8WoXGs8,4098
|
|
10
|
+
lcode/ollama.py,sha256=fELwqw7xqLH9TM7XvdI1pxqYZFeIRJBv-VSz9REdvwA,6516
|
|
11
|
+
lcode/permissions.py,sha256=6nPOlr0ARoI_ZXfugIq4r8TUZj2wE8SSvpjp9zdC578,3890
|
|
12
|
+
lcode/render.py,sha256=gMvf4W3qkLUJ4Vot3az_hP_zRDAItJC9pP-68dHFkfA,1375
|
|
13
|
+
lcode/repl.py,sha256=3mSNT9ku5rtqkX_h_mctiy4LuktS46ESzkcSlz2cjb4,16021
|
|
14
|
+
lcode/sessions.py,sha256=wOR-J0EHgJxLSPboKPHnvBcp5VTUYTj3tZWgpsa6-K4,4013
|
|
15
|
+
lcode/tools.py,sha256=TaYb1ncRO_e-XBwhB5GdAhgB6kpYlJmjetnyYFDCZzs,23198
|
|
16
|
+
lcode_cli-0.1.2.dist-info/METADATA,sha256=5BhEy2IFBX3boJjBLa639QZGH4b22CM3VhR6Pn-mnR0,8536
|
|
17
|
+
lcode_cli-0.1.2.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
18
|
+
lcode_cli-0.1.2.dist-info/entry_points.txt,sha256=pZ_uDQ98UzVAHUQeXn5HeyQWKGSH3r3T5aPHuSB89eI,41
|
|
19
|
+
lcode_cli-0.1.2.dist-info/licenses/LICENSE,sha256=JWjmoL7nNIC1XUa-dZm2hwLUgjUQMFo6k8APb0idNzs,1074
|
|
20
|
+
lcode_cli-0.1.2.dist-info/RECORD,,
|
lcode_cli-0.1.1.dist-info/RECORD
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
lcode/__init__.py,sha256=czDagA2Y7vJ-F5wm2C4CrWmBci7DnShS1yCOE3nEjbQ,117
|
|
2
|
-
lcode/__main__.py,sha256=s5N5pryX2FAu93GGZ3FxUrIXHWGzxz66lKUoaV_h8Ws,35
|
|
3
|
-
lcode/agent.py,sha256=Tc9HYFzASMpVGYWxX_U7RWqNTZK2czaPjGBjlHs-P1I,17427
|
|
4
|
-
lcode/catalog.py,sha256=pZ1azX9OH2r4JE2c6E5OfwywpgS39-W2oXzoCHJhmtM,4141
|
|
5
|
-
lcode/cli.py,sha256=7Pfq_mnZ-tIhFfe_YpVohRwt236I1an2-4TN0sWqvfg,18789
|
|
6
|
-
lcode/config.py,sha256=NvsiWFyGyDpEKHwWDrW643yYLRG4Omp8Qnp_68K2a_Y,5038
|
|
7
|
-
lcode/hardware.py,sha256=2VbknZCxTZJwJUD8YWdQ7ryEQo_LcjtoOBOHOU9Y0fw,3085
|
|
8
|
-
lcode/models.toml,sha256=23KFX9z6WPQPXZEhD1SV9OxQowHY9TEp0IDwxIPbYb4,3949
|
|
9
|
-
lcode/ollama.py,sha256=fELwqw7xqLH9TM7XvdI1pxqYZFeIRJBv-VSz9REdvwA,6516
|
|
10
|
-
lcode/permissions.py,sha256=6nPOlr0ARoI_ZXfugIq4r8TUZj2wE8SSvpjp9zdC578,3890
|
|
11
|
-
lcode/render.py,sha256=gMvf4W3qkLUJ4Vot3az_hP_zRDAItJC9pP-68dHFkfA,1375
|
|
12
|
-
lcode/repl.py,sha256=BL8bQcJ6tw4uadF2bELsFwshPe5SOKZOZQob5ybUpzM,10816
|
|
13
|
-
lcode/tools.py,sha256=99XvUhNs1YjjqXORmiu3NI9jC0FX11eJkI1sENWtq5s,22206
|
|
14
|
-
lcode_cli-0.1.1.dist-info/METADATA,sha256=3X8oSBoTEydGadmN0JHrpI9-EMGzLC9Df4n52Gq94Dc,8158
|
|
15
|
-
lcode_cli-0.1.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
16
|
-
lcode_cli-0.1.1.dist-info/entry_points.txt,sha256=pZ_uDQ98UzVAHUQeXn5HeyQWKGSH3r3T5aPHuSB89eI,41
|
|
17
|
-
lcode_cli-0.1.1.dist-info/licenses/LICENSE,sha256=JWjmoL7nNIC1XUa-dZm2hwLUgjUQMFo6k8APb0idNzs,1074
|
|
18
|
-
lcode_cli-0.1.1.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|