lcode-cli 0.1.1__py3-none-any.whl → 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lcode/__init__.py CHANGED
@@ -1,3 +1,3 @@
1
1
  """lcode — a local-first terminal coding agent powered by open-weight models via Ollama."""
2
2
 
3
- __version__ = "0.1.1"
3
+ __version__ = "0.1.2"
lcode/agent.py CHANGED
@@ -18,14 +18,17 @@ from rich.markdown import Markdown
18
18
  from rich.panel import Panel
19
19
  from rich.text import Text
20
20
 
21
- from lcode import catalog
22
- from lcode.config import STATE_DIR, format_tokens
21
+ from lcode import catalog, limits, sessions
22
+ from lcode.config import format_tokens
23
23
  from lcode.ollama import Ollama, OllamaError
24
24
  from lcode.permissions import Permissions
25
25
  from lcode.render import MarkdownStreamer
26
26
  from lcode.tools import SCHEMAS, Toolbox, is_binary, parse_text_tool_calls, tree, truncate
27
27
 
28
- MAX_STEPS_PER_TURN = 150 # safety cap on tool-call iterations for one request
28
+ GPU_MEMORY_ERRORS = ("out of memory", "illegal memory access", "cudamalloc failed")
29
+ SAFE_NUM_BATCH = 512 # Ollama's default prompt batch
30
+ MAX_STEPS_PER_TURN = 150
31
+ MAX_MALFORMED_CALL_RETRIES = 2 # Ollama rejects tool calls whose arguments aren't valid JSON # safety cap on tool-call iterations for one request
29
32
  AUTO_COMPACT_RATIO = 0.85 # summarize the history when the context is this full
30
33
  PROJECT_FILES = ("AGENTS.md", "LCODE.md", "CLAUDE.md")
31
34
 
@@ -100,6 +103,8 @@ class Agent:
100
103
  self.perms = Permissions(self.console, settings.permission_mode)
101
104
  self.tools = Toolbox(self)
102
105
  self.session_id = self.new_session_id()
106
+ self.session_name = ""
107
+ self.session_title = ""
103
108
  self.ctx_used = 0
104
109
  self.last_speed = 0.0
105
110
  self.messages: list[dict] = []
@@ -134,25 +139,63 @@ class Agent:
134
139
  self.ctx_used = len(self.messages[0]["content"]) // 3
135
140
 
136
141
  def session_file(self) -> Path:
137
- return STATE_DIR / "sessions" / f"{self.session_id}.json"
142
+ return sessions.sessions_dir() / f"{self.session_id}.json"
143
+
144
+ def has_conversation(self) -> bool:
145
+ return any(m.get("role") == "user" for m in self.messages)
138
146
 
139
147
  def save(self) -> None:
148
+ if not (self.has_conversation() or self.session_name):
149
+ return # nothing worth resuming
150
+ self.session_title = self.session_title or sessions.title_from(self.messages)
140
151
  f = self.session_file()
141
152
  f.parent.mkdir(parents=True, exist_ok=True)
142
- f.write_text(json.dumps({"cwd": str(self.cwd), "model": self.settings.model, "messages": self.messages}))
153
+ data = {
154
+ "cwd": str(self.cwd),
155
+ "model": self.settings.model,
156
+ "name": self.session_name,
157
+ "title": self.session_title,
158
+ "messages": self.messages,
159
+ }
160
+ f.write_text(json.dumps(data))
161
+
162
+ def new_session(self) -> None:
163
+ self.reset()
164
+ self.session_id = self.new_session_id()
165
+ self.session_name = ""
166
+ self.session_title = ""
167
+
168
+ def rename(self, name: str) -> None:
169
+ self.session_name = " ".join(name.split())
170
+ self.save()
171
+
172
+ def load(self, info: sessions.SessionInfo) -> str:
173
+ """Resume a saved session. Returns a note about the working directory, if it changed."""
174
+ try:
175
+ data = json.loads(info.path.read_text())
176
+ except (OSError, ValueError) as e:
177
+ raise OSError(f"can't read session {info.id}: {e}") from e
178
+ note = ""
179
+ saved_cwd = Path(info.cwd) if info.cwd else self.cwd
180
+ if saved_cwd != self.cwd:
181
+ if saved_cwd.is_dir():
182
+ self.cwd = saved_cwd.resolve()
183
+ note = f"Working directory is now {self.cwd}"
184
+ else:
185
+ note = f"The session's directory {saved_cwd} no longer exists; staying in {self.cwd}"
186
+ self.messages = [{"role": "system", "content": self.system_prompt()}, *data.get("messages", [])[1:]]
187
+ self.session_id = info.id
188
+ self.session_name = data.get("name", "")
189
+ self.session_title = data.get("title") or sessions.title_from(self.messages)
190
+ self.tools.read_mtimes.clear() # files may have changed since; the model must read them again
191
+ self.ctx_used = sum(len(json.dumps(m)) for m in self.messages) // 3
192
+ return note
143
193
 
144
194
  def load_latest(self) -> bool:
145
- for f in sorted((STATE_DIR / "sessions").glob("*.json"), reverse=True):
146
- try:
147
- data = json.loads(f.read_text())
148
- except (OSError, json.JSONDecodeError):
149
- continue
150
- if data.get("cwd") == str(self.cwd):
151
- self.messages = [{"role": "system", "content": self.system_prompt()}, *data["messages"][1:]]
152
- self.session_id = f.stem
153
- self.ctx_used = sum(len(json.dumps(m)) for m in self.messages) // 3
154
- return True
155
- return False
195
+ latest = sessions.list_sessions(self.cwd, limit=1)
196
+ if latest:
197
+ self.load(latest[0])
198
+ return bool(latest)
156
199
 
157
200
  # -- model calls
158
201
  def options(self) -> dict:
@@ -171,18 +214,45 @@ class Agent:
171
214
  }
172
215
  if tools:
173
216
  payload["tools"] = tools
217
+ started = False
174
218
  try:
175
- yield from self.ollama.chat_stream(payload)
219
+ for chunk in self.ollama.chat_stream(payload):
220
+ started = True
221
+ yield chunk
176
222
  except OllamaError as e:
177
- if think and "think" in str(e) and "support" in str(e):
223
+ error = str(e).lower()
224
+ if not started and think and "think" in error and "support" in error:
178
225
  self.settings.think = False
179
226
  self.console.print(f"[dim]{self.settings.model} does not support reasoning; continuing without.[/]")
180
- yield from self.ollama.chat_stream({**payload, "think": False})
227
+ yield from self.chat(messages, tools, False)
181
228
  return
182
- if "out of memory" in str(e).lower():
229
+ if any(marker in error for marker in GPU_MEMORY_ERRORS):
230
+ batch = self.settings.num_batch
231
+ if not started and batch and batch > SAFE_NUM_BATCH:
232
+ # Larger batches read prompts faster but need extra VRAM that isn't always free.
233
+ self.settings.num_batch = SAFE_NUM_BATCH
234
+ self.console.print(
235
+ f"[yellow]The GPU ran out of memory with a prompt batch of {batch}; retrying with "
236
+ f"{SAFE_NUM_BATCH}.[/] [dim]To skip this retry: lcode config set num_batch {SAFE_NUM_BATCH}[/]"
237
+ )
238
+ yield from self.chat(messages, tools, think)
239
+ return
240
+ context = self.settings.context
241
+ if not started and context > catalog.MIN_USEFUL_CONTEXT:
242
+ # The context cache has to fit in memory; halve it until the model loads.
243
+ smaller = max(catalog.MIN_USEFUL_CONTEXT, context // 2)
244
+ self.settings.context = smaller
245
+ limits.record(self.settings.model, smaller)
246
+ self.console.print(
247
+ f"[yellow]{self.settings.model} didn't fit in GPU memory with a {format_tokens(context)} "
248
+ f"context; retrying with {format_tokens(smaller)}.[/] [dim]lcode will start there next "
249
+ "time on this machine.[/]"
250
+ )
251
+ yield from self.chat(messages, tools, think)
252
+ return
183
253
  raise OllamaError(
184
- f"{e}\nThe model ran out of GPU memory. Try a smaller context (/ctx 128k) or a smaller "
185
- "prompt batch (`lcode config set num_batch 512`)."
254
+ f"{e}\nThe model ran out of GPU memory. Try a smaller context (/ctx 128k), close other programs "
255
+ "using the GPU, or pick a smaller model (/models)."
186
256
  ) from e
187
257
  raise
188
258
 
@@ -267,9 +337,26 @@ class Agent:
267
337
 
268
338
  def run_turn(self, user_text: str) -> None:
269
339
  self.messages.append({"role": "user", "content": self.expand_mentions(user_text)})
340
+ malformed = 0
270
341
  for _ in range(MAX_STEPS_PER_TURN):
271
342
  self.maybe_compact()
272
- calls = self.assistant_step().get("tool_calls") or []
343
+ try:
344
+ calls = self.assistant_step().get("tool_calls") or []
345
+ except OllamaError as e:
346
+ if "error parsing tool call" not in str(e) or malformed >= MAX_MALFORMED_CALL_RETRIES:
347
+ raise
348
+ malformed += 1
349
+ reason = str(e).rsplit("err=", 1)[-1].strip() if "err=" in str(e) else "invalid JSON"
350
+ self.console.print("[yellow]The model wrote a malformed tool call; asking it to try again.[/]")
351
+ self.messages.append(
352
+ {
353
+ "role": "user",
354
+ "content": f"[lcode] Your last tool call could not be parsed ({reason}): its arguments "
355
+ "must be one complete, valid JSON object. Make the call again. If it writes a file, "
356
+ "make sure the whole content is included and properly escaped.",
357
+ }
358
+ )
359
+ continue
273
360
  if not calls:
274
361
  break
275
362
  for i, call in enumerate(calls):
lcode/cli.py CHANGED
@@ -14,7 +14,7 @@ from rich.progress import BarColumn, DownloadColumn, Progress, TextColumn, TimeR
14
14
  from rich.prompt import Confirm
15
15
  from rich.table import Table
16
16
 
17
- from lcode import __version__, catalog, config
17
+ from lcode import __version__, catalog, config, limits
18
18
  from lcode.agent import Agent, Settings
19
19
  from lcode.catalog import ModelSpec
20
20
  from lcode.config import ConfigError, format_tokens, parse_context
@@ -78,9 +78,9 @@ def choose_context(
78
78
  if requested:
79
79
  ctx = requested
80
80
  elif spec:
81
- ctx = spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT
81
+ ctx = limits.cap(model, spec.fit(hw)[0] or catalog.MIN_USEFUL_CONTEXT)
82
82
  else:
83
- ctx = 32768
83
+ ctx = limits.cap(model, 32768)
84
84
  if limit and ctx > limit:
85
85
  ctx, note = limit, f"capped at the model maximum of {format_tokens(limit)}"
86
86
  if spec and requested and spec.memory_gib(ctx) > hw.budget_gib:
@@ -272,6 +272,13 @@ def cmd_doctor(args) -> None:
272
272
  if spec:
273
273
  detail += f" · ~{spec.memory_gib(ctx):.0f} GB needed, ~{hw.budget_gib:.0f} GB available"
274
274
  line("Context", detail + (f" ({note})" if note else ""), None if note else True)
275
+ if limits.get(model):
276
+ line(
277
+ "Limit",
278
+ f"{format_tokens(limits.get(model))} for {model}: larger contexts ran out of GPU memory "
279
+ f"here (delete {limits.path()} to try again)",
280
+ None,
281
+ )
275
282
  loaded = [m for m in ollama.running() if m.get("name") == model or m.get("model") == model]
276
283
  if loaded:
277
284
  m = loaded[0]
@@ -370,7 +377,7 @@ def cmd_chat(args) -> None:
370
377
  permission_mode=mode,
371
378
  )
372
379
  agent = Agent(ollama, settings, cwd, console=console)
373
- repl(agent, prompt=args.prompt, resume=args.cont, hardware=hw)
380
+ repl(agent, prompt=args.prompt, hardware=hw, cont=args.cont, resume=args.resume)
374
381
 
375
382
 
376
383
  def build_parser() -> argparse.ArgumentParser:
@@ -384,7 +391,14 @@ def build_parser() -> argparse.ArgumentParser:
384
391
  parser.add_argument("-m", "--model", help="catalog key (see `lcode models`) or any installed Ollama model")
385
392
  parser.add_argument("--context", "--ctx", dest="context", help="context window, e.g. 65536, 128k or 1m")
386
393
  parser.add_argument("-r", "--repo", default=".", help="working directory (default: current directory)")
387
- parser.add_argument("-c", "--continue", dest="cont", action="store_true", help="resume the last session here")
394
+ parser.add_argument("-c", "--continue", dest="cont", action="store_true", help="continue the last session here")
395
+ parser.add_argument(
396
+ "--resume",
397
+ nargs="?",
398
+ const="",
399
+ metavar="SESSION",
400
+ help="resume a saved session: pick from a list, or give its number, name or id",
401
+ )
388
402
  parser.add_argument("--auto-edit", action="store_true", help="apply file edits without asking")
389
403
  parser.add_argument("--yolo", action="store_true", help="never ask for permission (edits and commands)")
390
404
  parser.add_argument("--no-think", action="store_true", help="disable model reasoning (faster, less accurate)")
lcode/limits.py ADDED
@@ -0,0 +1,41 @@
1
+ """Context sizes learned to be too large on this machine, from out-of-memory errors.
2
+
3
+ When a model fails to load because its context doesn't fit in GPU memory, lcode retries with half the
4
+ context and records the size that was tried, so later sessions start there instead of failing first.
5
+ Delete the file (see `path()`) to let lcode try larger contexts again, e.g. after a GPU upgrade.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from pathlib import Path
12
+
13
+ from lcode import config
14
+
15
+
16
+ def path() -> Path:
17
+ return config.STATE_DIR / "limits.json"
18
+
19
+
20
+ def _load() -> dict[str, int]:
21
+ try:
22
+ data = json.loads(path().read_text())
23
+ except (OSError, ValueError):
24
+ return {}
25
+ return {k: int(v) for k, v in data.items() if isinstance(v, int)} if isinstance(data, dict) else {}
26
+
27
+
28
+ def get(model: str) -> int | None:
29
+ return _load().get(model)
30
+
31
+
32
+ def record(model: str, context: int) -> None:
33
+ data = _load()
34
+ data[model] = min(context, data.get(model, context))
35
+ path().parent.mkdir(parents=True, exist_ok=True)
36
+ path().write_text(json.dumps(data, indent=2, sort_keys=True) + "\n")
37
+
38
+
39
+ def cap(model: str, context: int) -> int:
40
+ limit = get(model)
41
+ return min(context, limit) if limit else context
lcode/models.toml CHANGED
@@ -29,7 +29,6 @@ moe = true
29
29
  size_gb = 22.6
30
30
  max_context = 262144
31
31
  kv_kib_per_token = 22
32
- num_batch = 1024
33
32
  swe_bench_verified = 73.4
34
33
  tested = true
35
34
  notes = "Default. Strongest tool calling among local MoE coders; fast even with experts in RAM."
@@ -44,8 +43,8 @@ moe = false
44
43
  size_gb = 17.7
45
44
  max_context = 262144
46
45
  kv_kib_per_token = 68
47
- tested = false
48
- notes = "Newest Qwen. Dense: excellent when it fits entirely in GPU/unified memory, slow when split."
46
+ tested = true
47
+ notes = "Newest Qwen. Dense: clean, accurate tool use; ~8 tok/s when split on a 12 GB GPU, fast when it fits in GPU/unified memory."
49
48
 
50
49
  [[model]]
51
50
  key = "qwen3.6-27b"
@@ -57,8 +56,8 @@ moe = false
57
56
  size_gb = 17.8
58
57
  max_context = 262144
59
58
  kv_kib_per_token = 68
60
- tested = false
61
- notes = "Dense coding model. Best on 24 GB+ GPUs or Macs with 48 GB+ unified memory."
59
+ tested = true
60
+ notes = "Dense coding model: clean, accurate tool use; ~7 tok/s when split on a 12 GB GPU. Best on 24 GB+ GPUs or 48 GB+ Macs."
62
61
 
63
62
  [[model]]
64
63
  key = "laguna-xs-2.1"
@@ -70,10 +69,9 @@ moe = true
70
69
  size_gb = 20.3
71
70
  max_context = 262144
72
71
  kv_kib_per_token = 40
73
- kv_estimated = true
74
72
  swe_bench_verified = 70.9
75
- tested = false
76
- notes = "Agentic-coding MoE built for local machines; slightly smaller than Qwen3.6 35B."
73
+ tested = true
74
+ notes = "Agentic-coding MoE built for local machines; 15-38 tok/s at 256K on a 12 GB GPU. Full KV cache in 10 of 40 layers."
77
75
 
78
76
  [[model]]
79
77
  key = "nemotron-3.5-lightning"
@@ -85,21 +83,8 @@ moe = true
85
83
  size_gb = 25.4
86
84
  max_context = 1048576
87
85
  kv_kib_per_token = 7
88
- tested = false
89
- notes = "Hybrid Mamba-Transformer: tiny KV cache, up to 1M tokens of context."
90
-
91
- [[model]]
92
- key = "gpt-oss-20b"
93
- tag = "gpt-oss:20b"
94
- name = "gpt-oss 20B"
95
- publisher = "OpenAI"
96
- params = "21B MoE · 3.6B active"
97
- moe = true
98
- size_gb = 13.8
99
- max_context = 131072
100
- kv_kib_per_token = 24
101
- tested = false
102
- notes = "Compact MoE with 128K context; fits 16 GB GPUs and 24 GB Macs."
86
+ tested = true
87
+ notes = "Hybrid Mamba-Transformer: tiny KV cache, up to 1M tokens (512K on a 12 GB GPU); 44 tok/s on a 12 GB GPU."
103
88
 
104
89
  [[model]]
105
90
  key = "qwen3.5-9b"
@@ -114,6 +99,19 @@ kv_kib_per_token = 32
114
99
  tested = true
115
100
  notes = "Small and capable: 64 tok/s fully on a 12 GB GPU at 128K context. For 8-12 GB GPUs and 16-24 GB Macs."
116
101
 
102
+ [[model]]
103
+ key = "gpt-oss-20b"
104
+ tag = "gpt-oss:20b"
105
+ name = "gpt-oss 20B"
106
+ publisher = "OpenAI"
107
+ params = "21B MoE · 3.6B active"
108
+ moe = true
109
+ size_gb = 13.8
110
+ max_context = 131072
111
+ kv_kib_per_token = 24
112
+ tested = true
113
+ notes = "Compact MoE with 128K context: 43 tok/s on a 12 GB GPU. Sometimes invents tool arguments but corrects itself."
114
+
117
115
  [[model]]
118
116
  key = "qwen3.5-4b"
119
117
  tag = "qwen3.5:4b"
lcode/repl.py CHANGED
@@ -14,11 +14,13 @@ from prompt_toolkit.formatted_text import HTML
14
14
  from prompt_toolkit.history import FileHistory
15
15
  from prompt_toolkit.key_binding import KeyBindings
16
16
  from rich.console import Group
17
+ from rich.markup import escape
17
18
  from rich.panel import Panel
18
19
  from rich.rule import Rule
20
+ from rich.table import Table
19
21
  from rich.text import Text
20
22
 
21
- from lcode import __version__, catalog
23
+ from lcode import __version__, catalog, limits, sessions
22
24
  from lcode.agent import INIT_PROMPT, Agent
23
25
  from lcode.catalog import MIN_USEFUL_CONTEXT
24
26
  from lcode.config import PERMISSION_MODES, STATE_DIR, ConfigError, format_tokens, parse_context
@@ -30,6 +32,8 @@ COMMANDS = {
30
32
  "/help": "Show this help",
31
33
  "/init": "Analyze the repo and write an AGENTS.md guide (loaded at every start)",
32
34
  "/clear": "Start a fresh conversation",
35
+ "/rename": "Name this session so you can find it later, e.g. /rename auth refactor",
36
+ "/resume": "Resume a saved session: pick from a list, or /resume <number|name> (/resume all: every folder)",
33
37
  "/compact": "Summarize the conversation to free context (optional: what to focus on)",
34
38
  "/context": "Show context-window usage",
35
39
  "/ctx": "Show or change the context window, e.g. /ctx 128k",
@@ -105,6 +109,7 @@ def build_session(agent: Agent) -> PromptSession:
105
109
  f" <b>{html.escape(s.model)}</b> · ctx {format_tokens(agent.ctx_used)}/{format_tokens(s.context)} "
106
110
  f"({pct:.0f}%) · mode <{color}>{agent.perms.mode}</{color}> (shift+tab) · "
107
111
  f"think {'on' if s.think else 'off'} · {html.escape(agent.cwd.name)}/"
112
+ + (f" · <b>{html.escape(agent.session_name)}</b>" if agent.session_name else "")
108
113
  )
109
114
 
110
115
  STATE_DIR.mkdir(parents=True, exist_ok=True)
@@ -167,9 +172,19 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
167
172
  )
168
173
  c.print(Panel(f"{rows}\n\n [dim]{keys}[/]", title="lcode commands", border_style="cyan"))
169
174
  elif cmd == "/clear":
170
- agent.reset()
171
- agent.session_id = agent.new_session_id()
172
- c.print("[green]Conversation cleared.[/]")
175
+ agent.new_session()
176
+ c.print("[green]Started a new conversation.[/] The previous one is saved; /resume brings it back.")
177
+ elif cmd == "/rename":
178
+ if not arg:
179
+ current = f"'{escape(agent.session_name)}'" if agent.session_name else "not named yet"
180
+ c.print(f"This session is {current}. Name it with /rename <name>.")
181
+ else:
182
+ agent.rename(arg)
183
+ c.print(f"[green]Session named '{escape(agent.session_name)}'.[/] Find it later with /resume.")
184
+ elif cmd == "/resume":
185
+ info = choose_session(agent, arg)
186
+ if info:
187
+ resume_session(agent, info)
173
188
  elif cmd == "/compact":
174
189
  agent.compact(arg)
175
190
  elif cmd == "/context":
@@ -208,7 +223,7 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
208
223
  s.model = name
209
224
  s.num_batch = spec.num_batch if spec and name == spec.local_name else None
210
225
  if spec: # size the context for the new model: largest window that fits this machine
211
- s.context = spec.fit(hardware)[0] or MIN_USEFUL_CONTEXT
226
+ s.context = limits.cap(name, spec.fit(hardware)[0] or MIN_USEFUL_CONTEXT)
212
227
  agent.messages[0]["content"] = agent.system_prompt()
213
228
  c.print(f"[green]Switched to {name}[/] (context {format_tokens(s.context)}).")
214
229
  elif cmd == "/think":
@@ -241,9 +256,110 @@ def handle_command(agent: Agent, line: str, hardware: Hardware) -> bool:
241
256
  return True
242
257
 
243
258
 
244
- def repl(agent: Agent, prompt: str | None, resume: bool, hardware: Hardware) -> None:
245
- if resume and agent.load_latest():
246
- agent.console.print(f"[green]Resumed session {agent.session_id} ({len(agent.messages)} messages).[/]")
259
+ def print_sessions(agent: Agent, found: list[sessions.SessionInfo], all_dirs: bool) -> None:
260
+ where = "all folders" if all_dirs else str(agent.cwd)
261
+ table = Table(title=f"Saved sessions · {where}", title_justify="left", header_style="bold")
262
+ table.add_column("#", justify="right", style="cyan")
263
+ table.add_column("Session")
264
+ table.add_column("Last used", no_wrap=True)
265
+ table.add_column("Requests", justify="right")
266
+ if all_dirs:
267
+ table.add_column("Folder", overflow="fold")
268
+ for i, info in enumerate(found, 1):
269
+ if info.name:
270
+ label = f"[bold]{escape(info.name)}[/]\n[dim]{escape(info.title or '(no requests yet)')}[/]"
271
+ else:
272
+ label = escape(info.label)
273
+ if info.id == agent.session_id:
274
+ label += " [green](current)[/]"
275
+ row = [str(i), label, sessions.age(info.updated), str(info.turns)]
276
+ if all_dirs:
277
+ row.append(escape(info.cwd))
278
+ table.add_row(*row)
279
+ agent.console.print(table)
280
+
281
+
282
+ def choose_session(agent: Agent, query: str = "") -> sessions.SessionInfo | None:
283
+ """Find a session by number/name/id, or list them and ask. `all` lists every folder."""
284
+ c = agent.console
285
+ words = query.split()
286
+ all_dirs = bool(words) and words[0].lower() in ("all", "--all", "-a")
287
+ query = " ".join(words[1:] if all_dirs else words)
288
+ found = sessions.list_sessions(None if all_dirs else agent.cwd)
289
+ if not found and not all_dirs:
290
+ found, all_dirs = sessions.list_sessions(None), True
291
+ if found and not query:
292
+ c.print("[dim]No saved sessions in this folder; showing all folders.[/]")
293
+ if not found:
294
+ c.print("No saved sessions yet. Sessions are saved after every request.")
295
+ return None
296
+ if query:
297
+ match = sessions.find(query, found)
298
+ if match is None and not all_dirs:
299
+ match = sessions.find(query, sessions.list_sessions(None))
300
+ if match:
301
+ return match
302
+ c.print(f"[yellow]No session matches '{escape(query)}'.[/]")
303
+ print_sessions(agent, found, all_dirs)
304
+ try:
305
+ answer = input(" Resume which session? (number or name, Enter to cancel): ").strip()
306
+ except EOFError:
307
+ return None
308
+ if not answer:
309
+ return None
310
+ match = sessions.find(answer, found)
311
+ if match is None:
312
+ c.print(f"[yellow]No session matches '{escape(answer)}'.[/]")
313
+ return match
314
+
315
+
316
+ def resume_session(agent: Agent, info: sessions.SessionInfo) -> None:
317
+ c = agent.console
318
+ if info.id == agent.session_id:
319
+ c.print("That's the current session.")
320
+ return
321
+ agent.save()
322
+ try:
323
+ note = agent.load(info)
324
+ except OSError as e:
325
+ c.print(f"[red]{e}[/]")
326
+ return
327
+ c.print(
328
+ f"[green]Resumed[/] [bold]{escape(info.label)}[/] "
329
+ f"[dim]({info.turns} request{'' if info.turns == 1 else 's'}, last used {sessions.age(info.updated)})[/]"
330
+ )
331
+ if note:
332
+ c.print(f"[yellow]{escape(note)}[/]")
333
+ print_recap(agent)
334
+
335
+
336
+ def print_recap(agent: Agent) -> None:
337
+ """Show the last request and the start of the last answer, so it's clear where things left off."""
338
+ last_user = next((m for m in reversed(agent.messages) if m.get("role") == "user"), None)
339
+ last_answer = next(
340
+ (m for m in reversed(agent.messages) if m.get("role") == "assistant" and m.get("content", "").strip()), None
341
+ )
342
+ if not last_user:
343
+ return
344
+ lines = [f"[bold]You:[/] {escape(sessions.title_from([last_user]))}"]
345
+ if last_answer:
346
+ answer = " ".join(last_answer["content"].split())
347
+ answer = answer if len(answer) <= 300 else answer[:299].rstrip() + "…"
348
+ lines.append(f"[bold]lcode:[/] {escape(answer)}")
349
+ agent.console.print(Panel("\n".join(lines), title="Where you left off", title_align="left", border_style="dim"))
350
+
351
+
352
+ def repl(agent: Agent, prompt: str | None, hardware: Hardware, cont: bool = False, resume: str | None = None) -> None:
353
+ if cont:
354
+ if agent.load_latest():
355
+ agent.console.print(f"[green]Continuing[/] [bold]{escape(agent.session_name or agent.session_title)}[/]")
356
+ print_recap(agent)
357
+ else:
358
+ agent.console.print("[dim]No saved session in this folder yet; starting a new one.[/]")
359
+ elif resume is not None:
360
+ info = choose_session(agent, resume)
361
+ if info:
362
+ resume_session(agent, info)
247
363
  if prompt:
248
364
  run_safely(agent, prompt)
249
365
  return
@@ -260,6 +376,8 @@ def repl(agent: Agent, prompt: str | None, resume: bool, hardware: Hardware) ->
260
376
  break
261
377
  if not line:
262
378
  continue
379
+ if line.lower() in sessions.EXIT_WORDS: # people type these expecting to quit, not to ask the model
380
+ break
263
381
  if line.startswith("/") and not line.startswith("//"):
264
382
  if not handle_command(agent, line, hardware):
265
383
  break
lcode/sessions.py ADDED
@@ -0,0 +1,122 @@
1
+ """Saved conversations: listing, naming and finding sessions to resume."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime as dt
6
+ import json
7
+ import re
8
+ from dataclasses import dataclass
9
+ from pathlib import Path
10
+
11
+ from lcode import config
12
+
13
+ TITLE_LENGTH = 70
14
+ SUMMARY_PREFIX = "[Summary of our conversation so far]" # first message of a compacted conversation
15
+ EXIT_WORDS = {"exit", "quit", ":q", ":quit"}
16
+
17
+
18
+ @dataclass(frozen=True)
19
+ class SessionInfo:
20
+ id: str
21
+ path: Path
22
+ cwd: str
23
+ name: str # set with /rename; empty if never named
24
+ title: str # the first request, shortened
25
+ model: str
26
+ updated: float # modification time of the session file
27
+ turns: int # number of requests from the user
28
+
29
+ @property
30
+ def label(self) -> str:
31
+ return self.name or self.title or "(no requests yet)"
32
+
33
+
34
+ def sessions_dir() -> Path:
35
+ return config.STATE_DIR / "sessions"
36
+
37
+
38
+ def title_from(messages: list[dict]) -> str:
39
+ """A one-line title from the first request, without attached files or compaction summaries."""
40
+ compacted = False
41
+ for message in messages:
42
+ if message.get("role") != "user":
43
+ continue
44
+ content = message.get("content", "")
45
+ if content.startswith(SUMMARY_PREFIX):
46
+ compacted = True
47
+ continue
48
+ text = re.split(r"\n\n<(?:file|directory) path=", content, maxsplit=1)[0]
49
+ text = " ".join(text.split())
50
+ if text:
51
+ return text if len(text) <= TITLE_LENGTH else text[: TITLE_LENGTH - 1].rstrip() + "…"
52
+ return "Compacted conversation" if compacted else ""
53
+
54
+
55
+ def read_info(path: Path) -> SessionInfo | None:
56
+ try:
57
+ data = json.loads(path.read_text())
58
+ updated = path.stat().st_mtime
59
+ except (OSError, ValueError):
60
+ return None
61
+ messages = data.get("messages") or []
62
+ return SessionInfo(
63
+ id=path.stem,
64
+ path=path,
65
+ cwd=data.get("cwd", ""),
66
+ name=data.get("name", ""),
67
+ title=data.get("title") or title_from(messages),
68
+ model=data.get("model", ""),
69
+ updated=updated,
70
+ turns=sum(1 for m in messages if m.get("role") == "user"),
71
+ )
72
+
73
+
74
+ def list_sessions(cwd: Path | str | None = None, limit: int = 50) -> list[SessionInfo]:
75
+ """Saved sessions, most recently used first; only those for `cwd` if given."""
76
+ directory = sessions_dir()
77
+ if not directory.is_dir():
78
+ return []
79
+ files = sorted(directory.glob("*.json"), key=lambda p: p.stat().st_mtime, reverse=True)
80
+ found = []
81
+ for f in files:
82
+ info = read_info(f)
83
+ if info and (info.turns or info.name) and (cwd is None or info.cwd == str(cwd)):
84
+ found.append(info)
85
+ if len(found) >= limit:
86
+ break
87
+ return found
88
+
89
+
90
+ def find(query: str, sessions: list[SessionInfo]) -> SessionInfo | None:
91
+ """Match a list number (1-based), a name, a session id (or unique prefix) or a unique title fragment."""
92
+ q = query.strip()
93
+ if not q:
94
+ return None
95
+ if q.isdigit() and 1 <= int(q) <= len(sessions):
96
+ return sessions[int(q) - 1]
97
+ lowered = q.lower()
98
+ for s in sessions:
99
+ if s.name and s.name.lower() == lowered:
100
+ return s
101
+ for candidates in (
102
+ [s for s in sessions if s.id == q or s.id.startswith(q)],
103
+ [s for s in sessions if lowered in s.label.lower()],
104
+ ):
105
+ if len(candidates) == 1:
106
+ return candidates[0]
107
+ return None
108
+
109
+
110
+ def age(timestamp: float, now: dt.datetime | None = None) -> str:
111
+ now = now or dt.datetime.now()
112
+ then = dt.datetime.fromtimestamp(timestamp)
113
+ seconds = (now - then).total_seconds()
114
+ if seconds < 60:
115
+ return "just now"
116
+ if seconds < 3600:
117
+ return f"{int(seconds // 60)} min ago"
118
+ if then.date() == now.date():
119
+ return f"{int(seconds // 3600)} h ago"
120
+ if then.date() == (now - dt.timedelta(days=1)).date():
121
+ return "yesterday"
122
+ return then.strftime("%b %d") if then.year == now.year else then.strftime("%Y-%m-%d")
lcode/tools.py CHANGED
@@ -4,6 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import difflib
6
6
  import fnmatch
7
+ import inspect
7
8
  import json
8
9
  import os
9
10
  import queue
@@ -265,6 +266,9 @@ class Toolbox:
265
266
  fn = getattr(self, f"t_{name}", None)
266
267
  if fn is None:
267
268
  return f"Error: unknown tool '{name}'. Available: {', '.join(sorted(TOOL_NAMES))}"
269
+ problem = self._argument_problem(fn, args)
270
+ if problem:
271
+ return f"Error: bad arguments for {name}: {problem}"
268
272
  try:
269
273
  return fn(**args)
270
274
  except ToolError as e:
@@ -274,6 +278,22 @@ class Toolbox:
274
278
  except Exception as e: # surface every failure to the model instead of crashing the session
275
279
  return f"Error: {type(e).__name__}: {e}"
276
280
 
281
+ @staticmethod
282
+ def _argument_problem(fn, args: dict) -> str:
283
+ """Explain unknown or missing arguments in terms the model can act on."""
284
+ params = inspect.signature(fn).parameters
285
+ unknown = [a for a in args if a not in params]
286
+ missing = [p for p, spec in params.items() if spec.default is inspect.Parameter.empty and p not in args]
287
+ if not unknown and not missing:
288
+ return ""
289
+ parts = []
290
+ if unknown:
291
+ parts.append(f"unknown argument(s) {', '.join(map(repr, unknown))}")
292
+ if missing:
293
+ parts.append(f"missing required argument(s) {', '.join(map(repr, missing))}")
294
+ valid = ", ".join(p if params[p].default is inspect.Parameter.empty else f"{p} (optional)" for p in params)
295
+ return f"{'; '.join(parts)}. Valid arguments: {valid}."
296
+
277
297
  # -- read-only
278
298
  def t_read_file(self, path: str, offset: int = 1, limit: int = 2000) -> str:
279
299
  p = self.resolve(path)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: lcode-cli
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: A local-first terminal coding agent powered by open-weight models on your own GPU or Mac, via Ollama.
5
5
  Project-URL: Homepage, https://nasser1941.github.io/lcode/
6
6
  Project-URL: Documentation, https://nasser1941.github.io/lcode/
@@ -45,6 +45,7 @@ Description-Content-Type: text/markdown
45
45
  <p align="center">
46
46
  <a href="https://github.com/nasser1941/lcode/actions/workflows/ci.yml"><img src="https://github.com/nasser1941/lcode/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
47
47
  <a href="https://nasser1941.github.io/lcode/"><img src="https://github.com/nasser1941/lcode/actions/workflows/docs.yml/badge.svg" alt="Docs"></a>
48
+ <a href="https://pypi.org/project/lcode-cli/"><img src="https://img.shields.io/pypi/v/lcode-cli.svg" alt="PyPI"></a>
48
49
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
49
50
  <img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
50
51
  <img src="https://img.shields.io/badge/platform-Linux%20%7C%20macOS%20(Apple%20Silicon)-lightgrey.svg" alt="Linux and macOS">
@@ -75,7 +76,8 @@ through [Ollama](https://ollama.com), so your code never leaves your machine.
75
76
  largest context that fits: `lcode setup` does it in one step.
76
77
  - **Safe by default.** Every edit is shown as a diff and every command that isn't read-only needs
77
78
  your approval. `auto-edit` and `yolo` modes when you want speed.
78
- - **Bring your own model.** Eight curated open-weight models, or any Ollama model with tool calling.
79
+ - **Bring your own model.** Eight curated open-weight models, all tested end to end, or any Ollama model
80
+ with tool calling.
79
81
  - **Private.** No telemetry, no accounts, no API keys.
80
82
 
81
83
  ## Install
@@ -88,14 +90,16 @@ curl -fsSL https://nasser1941.github.io/lcode/install.sh | bash
88
90
 
89
91
  The installer sets up lcode with its own Python via [uv](https://docs.astral.sh/uv/), checks for
90
92
  [Ollama](https://ollama.com) (0.30+) and runs `lcode setup`, which picks a model for your hardware
91
- and downloads it. Prefer manual steps? See the
92
- [installation guide](https://nasser1941.github.io/lcode/installation/), or:
93
+ and downloads it. Prefer manual steps? lcode is on [PyPI](https://pypi.org/project/lcode-cli/) as
94
+ `lcode-cli`:
93
95
 
94
96
  ```bash
95
- uv tool install git+https://github.com/nasser1941/lcode
97
+ uv tool install lcode-cli # or: pipx install lcode-cli
96
98
  lcode setup
97
99
  ```
98
100
 
101
+ See the [installation guide](https://nasser1941.github.io/lcode/installation/) for details.
102
+
99
103
  ## Quickstart
100
104
 
101
105
  ```bash
@@ -113,9 +117,10 @@ lcode
113
117
  | | |
114
118
  |---|---|
115
119
  | `lcode -p "…"` | one request, no interaction (scripts, hooks) |
116
- | `lcode -c` | continue the last session in this directory |
120
+ | `lcode -c` / `lcode --resume` | continue the last session here / pick a saved session from a list |
117
121
  | `lcode --model qwen3.5-9b --context 128k` | pick a model and context window for this session |
118
122
  | `lcode models` / `lcode doctor` | what fits this machine / check the installation |
123
+ | `/rename`, `/resume` | name the current session, resume a saved one |
119
124
  | `/model`, `/ctx 128k`, `/compact`, `/help` | switch model, resize context, summarize, list commands |
120
125
 
121
126
  ## Models
@@ -123,12 +128,12 @@ lcode
123
128
  | Key | Model | Download | Max context | |
124
129
  |---|---|---|---|---|
125
130
  | `qwen3.6-35b` | Qwen3.6 35B-A3B Coding (MoE, 3B active) | 22.6 GB | 256K | **default**, tested |
126
- | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | |
127
- | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | |
128
- | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | |
129
- | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | |
130
- | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | |
131
+ | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | tested |
132
+ | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | tested |
133
+ | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | tested |
134
+ | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | tested |
131
135
  | `qwen3.5-9b` | Qwen3.5 9B (dense) | 6.6 GB | 256K | tested |
136
+ | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | tested |
132
137
  | `qwen3.5-4b` | Qwen3.5 4B (dense) | 3.4 GB | 256K | tested |
133
138
 
134
139
  What `lcode setup` picks for common machines:
@@ -140,7 +145,7 @@ What `lcode setup` picks for common machines:
140
145
  | Mac with M4 Pro / M4 Max, 36 GB | qwen3.6-35b | 64K |
141
146
  | Mac with M4 Pro, 48 GB · M4 Max, 64 GB+ | qwen3.6-35b | 256K |
142
147
  | NVIDIA 8–24 GB + 32 GB RAM | qwen3.6-35b | 256K |
143
- | NVIDIA 8 GB + 16 GB RAM | qwen3.5-4b | 64K |
148
+ | NVIDIA 8 GB + 16 GB RAM | gpt-oss-20b | 64K |
144
149
 
145
150
  On an RTX 4080 Laptop GPU (12 GB) the default model generates 50–60 tokens/s at 256K context, and
146
151
  qwen3.5-9b 64 tokens/s at 128K. See
@@ -0,0 +1,20 @@
1
+ lcode/__init__.py,sha256=DCFKMt5wGr6X-Z5XXGSODRrKPH4BQtYY3wBJ4PTfh08,117
2
+ lcode/__main__.py,sha256=s5N5pryX2FAu93GGZ3FxUrIXHWGzxz66lKUoaV_h8Ws,35
3
+ lcode/agent.py,sha256=FLWZK97MH15r7feTiYFZfbhWU_EQ9PEvwGBXbGb3AfA,21807
4
+ lcode/catalog.py,sha256=pZ1azX9OH2r4JE2c6E5OfwywpgS39-W2oXzoCHJhmtM,4141
5
+ lcode/cli.py,sha256=kRpZNze1Gsa33Z-l7eLJYXG5vFzozqOrA-KCRduJ6tU,19337
6
+ lcode/config.py,sha256=NvsiWFyGyDpEKHwWDrW643yYLRG4Omp8Qnp_68K2a_Y,5038
7
+ lcode/hardware.py,sha256=2VbknZCxTZJwJUD8YWdQ7ryEQo_LcjtoOBOHOU9Y0fw,3085
8
+ lcode/limits.py,sha256=oJzVQoMyTZ0_Fkpu5t3qxrf9v7Hv9S8U8BRQR5M1aKA,1237
9
+ lcode/models.toml,sha256=lt2G64zzo_Ryxv2ISRSNZoVD_JqCqYqsM01n8WoXGs8,4098
10
+ lcode/ollama.py,sha256=fELwqw7xqLH9TM7XvdI1pxqYZFeIRJBv-VSz9REdvwA,6516
11
+ lcode/permissions.py,sha256=6nPOlr0ARoI_ZXfugIq4r8TUZj2wE8SSvpjp9zdC578,3890
12
+ lcode/render.py,sha256=gMvf4W3qkLUJ4Vot3az_hP_zRDAItJC9pP-68dHFkfA,1375
13
+ lcode/repl.py,sha256=3mSNT9ku5rtqkX_h_mctiy4LuktS46ESzkcSlz2cjb4,16021
14
+ lcode/sessions.py,sha256=wOR-J0EHgJxLSPboKPHnvBcp5VTUYTj3tZWgpsa6-K4,4013
15
+ lcode/tools.py,sha256=TaYb1ncRO_e-XBwhB5GdAhgB6kpYlJmjetnyYFDCZzs,23198
16
+ lcode_cli-0.1.2.dist-info/METADATA,sha256=5BhEy2IFBX3boJjBLa639QZGH4b22CM3VhR6Pn-mnR0,8536
17
+ lcode_cli-0.1.2.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
18
+ lcode_cli-0.1.2.dist-info/entry_points.txt,sha256=pZ_uDQ98UzVAHUQeXn5HeyQWKGSH3r3T5aPHuSB89eI,41
19
+ lcode_cli-0.1.2.dist-info/licenses/LICENSE,sha256=JWjmoL7nNIC1XUa-dZm2hwLUgjUQMFo6k8APb0idNzs,1074
20
+ lcode_cli-0.1.2.dist-info/RECORD,,
@@ -1,18 +0,0 @@
1
- lcode/__init__.py,sha256=czDagA2Y7vJ-F5wm2C4CrWmBci7DnShS1yCOE3nEjbQ,117
2
- lcode/__main__.py,sha256=s5N5pryX2FAu93GGZ3FxUrIXHWGzxz66lKUoaV_h8Ws,35
3
- lcode/agent.py,sha256=Tc9HYFzASMpVGYWxX_U7RWqNTZK2czaPjGBjlHs-P1I,17427
4
- lcode/catalog.py,sha256=pZ1azX9OH2r4JE2c6E5OfwywpgS39-W2oXzoCHJhmtM,4141
5
- lcode/cli.py,sha256=7Pfq_mnZ-tIhFfe_YpVohRwt236I1an2-4TN0sWqvfg,18789
6
- lcode/config.py,sha256=NvsiWFyGyDpEKHwWDrW643yYLRG4Omp8Qnp_68K2a_Y,5038
7
- lcode/hardware.py,sha256=2VbknZCxTZJwJUD8YWdQ7ryEQo_LcjtoOBOHOU9Y0fw,3085
8
- lcode/models.toml,sha256=23KFX9z6WPQPXZEhD1SV9OxQowHY9TEp0IDwxIPbYb4,3949
9
- lcode/ollama.py,sha256=fELwqw7xqLH9TM7XvdI1pxqYZFeIRJBv-VSz9REdvwA,6516
10
- lcode/permissions.py,sha256=6nPOlr0ARoI_ZXfugIq4r8TUZj2wE8SSvpjp9zdC578,3890
11
- lcode/render.py,sha256=gMvf4W3qkLUJ4Vot3az_hP_zRDAItJC9pP-68dHFkfA,1375
12
- lcode/repl.py,sha256=BL8bQcJ6tw4uadF2bELsFwshPe5SOKZOZQob5ybUpzM,10816
13
- lcode/tools.py,sha256=99XvUhNs1YjjqXORmiu3NI9jC0FX11eJkI1sENWtq5s,22206
14
- lcode_cli-0.1.1.dist-info/METADATA,sha256=3X8oSBoTEydGadmN0JHrpI9-EMGzLC9Df4n52Gq94Dc,8158
15
- lcode_cli-0.1.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
16
- lcode_cli-0.1.1.dist-info/entry_points.txt,sha256=pZ_uDQ98UzVAHUQeXn5HeyQWKGSH3r3T5aPHuSB89eI,41
17
- lcode_cli-0.1.1.dist-info/licenses/LICENSE,sha256=JWjmoL7nNIC1XUa-dZm2hwLUgjUQMFo6k8APb0idNzs,1074
18
- lcode_cli-0.1.1.dist-info/RECORD,,