lcode-cli 0.1.1__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/CHANGELOG.md +41 -1
  2. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/PKG-INFO +17 -12
  3. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/README.md +16 -11
  4. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/configuration.md +8 -5
  5. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/how-it-works.md +4 -1
  6. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/installation.md +14 -6
  7. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/models.md +49 -16
  8. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/quickstart.md +3 -2
  9. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/troubleshooting.md +11 -4
  10. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/usage.md +34 -4
  11. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/install.sh +4 -3
  12. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/__init__.py +1 -1
  13. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/agent.py +110 -23
  14. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/cli.py +19 -5
  15. lcode_cli-0.1.2/src/lcode/limits.py +41 -0
  16. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/models.toml +21 -23
  17. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/repl.py +126 -8
  18. lcode_cli-0.1.2/src/lcode/sessions.py +122 -0
  19. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/tools.py +20 -0
  20. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/conftest.py +1 -1
  21. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_agent.py +81 -0
  22. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_catalog.py +1 -1
  23. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_cli.py +13 -0
  24. lcode_cli-0.1.2/tests/test_sessions.py +174 -0
  25. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_tools.py +6 -1
  26. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.editorconfig +0 -0
  27. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/CODEOWNERS +0 -0
  28. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  29. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  30. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  31. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/ISSUE_TEMPLATE/model_request.yml +0 -0
  32. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  33. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/dependabot.yml +0 -0
  34. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/workflows/ci.yml +0 -0
  35. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/workflows/docs.yml +0 -0
  36. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.github/workflows/release.yml +0 -0
  37. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.gitignore +0 -0
  38. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/.pre-commit-config.yaml +0 -0
  39. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/AGENTS.md +0 -0
  40. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/CODE_OF_CONDUCT.md +0 -0
  41. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/CONTRIBUTING.md +0 -0
  42. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/LICENSE +0 -0
  43. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/SECURITY.md +0 -0
  44. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/assets/demo.svg +0 -0
  45. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/assets/extra.css +0 -0
  46. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/assets/logo.svg +0 -0
  47. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/assets/social-preview.png +0 -0
  48. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/changelog.md +0 -0
  49. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/contributing.md +0 -0
  50. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/docs/index.md +0 -0
  51. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/mkdocs.yml +0 -0
  52. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/overrides/main.html +0 -0
  53. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/pyproject.toml +0 -0
  54. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/scripts/record_demo.py +0 -0
  55. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/scripts/social-preview.html +0 -0
  56. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/__main__.py +0 -0
  57. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/catalog.py +0 -0
  58. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/config.py +0 -0
  59. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/hardware.py +0 -0
  60. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/ollama.py +0 -0
  61. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/permissions.py +0 -0
  62. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/src/lcode/render.py +0 -0
  63. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_config.py +0 -0
  64. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/tests/test_permissions.py +0 -0
  65. {lcode_cli-0.1.1 → lcode_cli-0.1.2}/uv.lock +0 -0
@@ -6,6 +6,45 @@ All notable changes to lcode are documented here. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.2] - 2026-09-29
10
+
11
+ ### Added
12
+
13
+ - All eight catalog models are now tested end to end (`qwen3.8-27b`, `qwen3.6-27b`, `laguna-xs-2.1`,
14
+ `nemotron-3.5-lightning` and `gpt-oss-20b` joined the three tested before), with a results table
15
+ in the models guide. `laguna-xs-2.1`'s cache size is now measured rather than estimated.
16
+ - When a model fails to load because its context doesn't fit in GPU memory, lcode retries with half the
17
+ context and remembers the size that worked for that model (`~/.local/state/lcode/limits.json`, shown
18
+ by `lcode doctor`), so later sessions don't fail first. Found with `nemotron-3.5-lightning`, whose 1M
19
+ context doesn't fit on a 12 GB GPU.
20
+ - Named sessions and a session picker: `/rename <name>` names the current session, `/resume` lists
21
+ saved sessions and resumes the one you pick (by number, name, id or title), `/resume all` shows
22
+ every folder, and `lcode --resume [SESSION]` does the same from the shell. Resuming shows a short
23
+ recap of where you left off.
24
+
25
+ ### Changed
26
+
27
+ - `lcode setup` recommends `gpt-oss-20b` at 64K (instead of `qwen3.5-4b`) for 8 GB GPUs with 16 GB of
28
+ RAM, and prefers `qwen3.5-9b` over `gpt-oss-20b` where both fit.
29
+ - lcode is on PyPI as `lcode-cli`; the installer now installs the latest release from PyPI instead of
30
+ the `main` branch.
31
+
32
+ ### Fixed
33
+
34
+ - `qwen3.6-35b` could crash Ollama with "CUDA error: an illegal memory access" on 12 GB GPUs: its
35
+ prompt batch of 1024 left too little VRAM for the model's speculative-decoding context. The default
36
+ is now Ollama's 512 (1024 remains available with `lcode config set num_batch 1024`), and when the GPU
37
+ runs out of memory before answering, lcode retries automatically with a batch of 512.
38
+ - A tool call with invalid JSON arguments no longer aborts the request: lcode tells the model what
39
+ was wrong and lets it try again. Unknown or missing tool arguments are reported with the list of
40
+ valid arguments.
41
+ - `lcode -c` continues the most recently *used* session in the folder, not the most recently created
42
+ one. Empty sessions are no longer saved.
43
+ - `/rename` saves the session right away, so a session named before its first request shows up in
44
+ `/resume`.
45
+ - Typing `exit` or `quit` (without a slash) quits instead of being sent to the model as a request.
46
+ - Compacted sessions keep their original title instead of showing the summary header.
47
+
9
48
  ## [0.1.1] - 2026-09-28
10
49
 
11
50
  ### Added
@@ -50,6 +89,7 @@ First public release.
50
89
  - `AGENTS.md` project instructions and `/init` to generate them.
51
90
  - One-line installer for Linux and macOS.
52
91
 
53
- [Unreleased]: https://github.com/nasser1941/lcode/compare/v0.1.1...HEAD
92
+ [Unreleased]: https://github.com/nasser1941/lcode/compare/v0.1.2...HEAD
93
+ [0.1.2]: https://github.com/nasser1941/lcode/compare/v0.1.1...v0.1.2
54
94
  [0.1.1]: https://github.com/nasser1941/lcode/compare/v0.1.0...v0.1.1
55
95
  [0.1.0]: https://github.com/nasser1941/lcode/releases/tag/v0.1.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: lcode-cli
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: A local-first terminal coding agent powered by open-weight models on your own GPU or Mac, via Ollama.
5
5
  Project-URL: Homepage, https://nasser1941.github.io/lcode/
6
6
  Project-URL: Documentation, https://nasser1941.github.io/lcode/
@@ -45,6 +45,7 @@ Description-Content-Type: text/markdown
45
45
  <p align="center">
46
46
  <a href="https://github.com/nasser1941/lcode/actions/workflows/ci.yml"><img src="https://github.com/nasser1941/lcode/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
47
47
  <a href="https://nasser1941.github.io/lcode/"><img src="https://github.com/nasser1941/lcode/actions/workflows/docs.yml/badge.svg" alt="Docs"></a>
48
+ <a href="https://pypi.org/project/lcode-cli/"><img src="https://img.shields.io/pypi/v/lcode-cli.svg" alt="PyPI"></a>
48
49
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
49
50
  <img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
50
51
  <img src="https://img.shields.io/badge/platform-Linux%20%7C%20macOS%20(Apple%20Silicon)-lightgrey.svg" alt="Linux and macOS">
@@ -75,7 +76,8 @@ through [Ollama](https://ollama.com), so your code never leaves your machine.
75
76
  largest context that fits: `lcode setup` does it in one step.
76
77
  - **Safe by default.** Every edit is shown as a diff and every command that isn't read-only needs
77
78
  your approval. `auto-edit` and `yolo` modes when you want speed.
78
- - **Bring your own model.** Eight curated open-weight models, or any Ollama model with tool calling.
79
+ - **Bring your own model.** Eight curated open-weight models, all tested end to end, or any Ollama model
80
+ with tool calling.
79
81
  - **Private.** No telemetry, no accounts, no API keys.
80
82
 
81
83
  ## Install
@@ -88,14 +90,16 @@ curl -fsSL https://nasser1941.github.io/lcode/install.sh | bash
88
90
 
89
91
  The installer sets up lcode with its own Python via [uv](https://docs.astral.sh/uv/), checks for
90
92
  [Ollama](https://ollama.com) (0.30+) and runs `lcode setup`, which picks a model for your hardware
91
- and downloads it. Prefer manual steps? See the
92
- [installation guide](https://nasser1941.github.io/lcode/installation/), or:
93
+ and downloads it. Prefer manual steps? lcode is on [PyPI](https://pypi.org/project/lcode-cli/) as
94
+ `lcode-cli`:
93
95
 
94
96
  ```bash
95
- uv tool install git+https://github.com/nasser1941/lcode
97
+ uv tool install lcode-cli # or: pipx install lcode-cli
96
98
  lcode setup
97
99
  ```
98
100
 
101
+ See the [installation guide](https://nasser1941.github.io/lcode/installation/) for details.
102
+
99
103
  ## Quickstart
100
104
 
101
105
  ```bash
@@ -113,9 +117,10 @@ lcode
113
117
  | | |
114
118
  |---|---|
115
119
  | `lcode -p "…"` | one request, no interaction (scripts, hooks) |
116
- | `lcode -c` | continue the last session in this directory |
120
+ | `lcode -c` / `lcode --resume` | continue the last session here / pick a saved session from a list |
117
121
  | `lcode --model qwen3.5-9b --context 128k` | pick a model and context window for this session |
118
122
  | `lcode models` / `lcode doctor` | what fits this machine / check the installation |
123
+ | `/rename`, `/resume` | name the current session, resume a saved one |
119
124
  | `/model`, `/ctx 128k`, `/compact`, `/help` | switch model, resize context, summarize, list commands |
120
125
 
121
126
  ## Models
@@ -123,12 +128,12 @@ lcode
123
128
  | Key | Model | Download | Max context | |
124
129
  |---|---|---|---|---|
125
130
  | `qwen3.6-35b` | Qwen3.6 35B-A3B Coding (MoE, 3B active) | 22.6 GB | 256K | **default**, tested |
126
- | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | |
127
- | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | |
128
- | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | |
129
- | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | |
130
- | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | |
131
+ | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | tested |
132
+ | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | tested |
133
+ | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | tested |
134
+ | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | tested |
131
135
  | `qwen3.5-9b` | Qwen3.5 9B (dense) | 6.6 GB | 256K | tested |
136
+ | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | tested |
132
137
  | `qwen3.5-4b` | Qwen3.5 4B (dense) | 3.4 GB | 256K | tested |
133
138
 
134
139
  What `lcode setup` picks for common machines:
@@ -140,7 +145,7 @@ What `lcode setup` picks for common machines:
140
145
  | Mac with M4 Pro / M4 Max, 36 GB | qwen3.6-35b | 64K |
141
146
  | Mac with M4 Pro, 48 GB · M4 Max, 64 GB+ | qwen3.6-35b | 256K |
142
147
  | NVIDIA 8–24 GB + 32 GB RAM | qwen3.6-35b | 256K |
143
- | NVIDIA 8 GB + 16 GB RAM | qwen3.5-4b | 64K |
148
+ | NVIDIA 8 GB + 16 GB RAM | gpt-oss-20b | 64K |
144
149
 
145
150
  On an RTX 4080 Laptop GPU (12 GB) the default model generates 50–60 tokens/s at 256K context, and
146
151
  qwen3.5-9b 64 tokens/s at 128K. See
@@ -12,6 +12,7 @@
12
12
  <p align="center">
13
13
  <a href="https://github.com/nasser1941/lcode/actions/workflows/ci.yml"><img src="https://github.com/nasser1941/lcode/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
14
14
  <a href="https://nasser1941.github.io/lcode/"><img src="https://github.com/nasser1941/lcode/actions/workflows/docs.yml/badge.svg" alt="Docs"></a>
15
+ <a href="https://pypi.org/project/lcode-cli/"><img src="https://img.shields.io/pypi/v/lcode-cli.svg" alt="PyPI"></a>
15
16
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
16
17
  <img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
17
18
  <img src="https://img.shields.io/badge/platform-Linux%20%7C%20macOS%20(Apple%20Silicon)-lightgrey.svg" alt="Linux and macOS">
@@ -42,7 +43,8 @@ through [Ollama](https://ollama.com), so your code never leaves your machine.
42
43
  largest context that fits: `lcode setup` does it in one step.
43
44
  - **Safe by default.** Every edit is shown as a diff and every command that isn't read-only needs
44
45
  your approval. `auto-edit` and `yolo` modes when you want speed.
45
- - **Bring your own model.** Eight curated open-weight models, or any Ollama model with tool calling.
46
+ - **Bring your own model.** Eight curated open-weight models, all tested end to end, or any Ollama model
47
+ with tool calling.
46
48
  - **Private.** No telemetry, no accounts, no API keys.
47
49
 
48
50
  ## Install
@@ -55,14 +57,16 @@ curl -fsSL https://nasser1941.github.io/lcode/install.sh | bash
55
57
 
56
58
  The installer sets up lcode with its own Python via [uv](https://docs.astral.sh/uv/), checks for
57
59
  [Ollama](https://ollama.com) (0.30+) and runs `lcode setup`, which picks a model for your hardware
58
- and downloads it. Prefer manual steps? See the
59
- [installation guide](https://nasser1941.github.io/lcode/installation/), or:
60
+ and downloads it. Prefer manual steps? lcode is on [PyPI](https://pypi.org/project/lcode-cli/) as
61
+ `lcode-cli`:
60
62
 
61
63
  ```bash
62
- uv tool install git+https://github.com/nasser1941/lcode
64
+ uv tool install lcode-cli # or: pipx install lcode-cli
63
65
  lcode setup
64
66
  ```
65
67
 
68
+ See the [installation guide](https://nasser1941.github.io/lcode/installation/) for details.
69
+
66
70
  ## Quickstart
67
71
 
68
72
  ```bash
@@ -80,9 +84,10 @@ lcode
80
84
  | | |
81
85
  |---|---|
82
86
  | `lcode -p "…"` | one request, no interaction (scripts, hooks) |
83
- | `lcode -c` | continue the last session in this directory |
87
+ | `lcode -c` / `lcode --resume` | continue the last session here / pick a saved session from a list |
84
88
  | `lcode --model qwen3.5-9b --context 128k` | pick a model and context window for this session |
85
89
  | `lcode models` / `lcode doctor` | what fits this machine / check the installation |
90
+ | `/rename`, `/resume` | name the current session, resume a saved one |
86
91
  | `/model`, `/ctx 128k`, `/compact`, `/help` | switch model, resize context, summarize, list commands |
87
92
 
88
93
  ## Models
@@ -90,12 +95,12 @@ lcode
90
95
  | Key | Model | Download | Max context | |
91
96
  |---|---|---|---|---|
92
97
  | `qwen3.6-35b` | Qwen3.6 35B-A3B Coding (MoE, 3B active) | 22.6 GB | 256K | **default**, tested |
93
- | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | |
94
- | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | |
95
- | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | |
96
- | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | |
97
- | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | |
98
+ | `qwen3.8-27b` | Qwen3.8 27B (dense) | 17.7 GB | 256K | tested |
99
+ | `qwen3.6-27b` | Qwen3.6 27B Coding (dense) | 17.8 GB | 256K | tested |
100
+ | `laguna-xs-2.1` | Poolside Laguna XS 2.1 (MoE, 3B active) | 20.3 GB | 256K | tested |
101
+ | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning (hybrid MoE) | 25.4 GB | 1M | tested |
98
102
  | `qwen3.5-9b` | Qwen3.5 9B (dense) | 6.6 GB | 256K | tested |
103
+ | `gpt-oss-20b` | OpenAI gpt-oss 20B (MoE) | 13.8 GB | 128K | tested |
99
104
  | `qwen3.5-4b` | Qwen3.5 4B (dense) | 3.4 GB | 256K | tested |
100
105
 
101
106
  What `lcode setup` picks for common machines:
@@ -107,7 +112,7 @@ What `lcode setup` picks for common machines:
107
112
  | Mac with M4 Pro / M4 Max, 36 GB | qwen3.6-35b | 64K |
108
113
  | Mac with M4 Pro, 48 GB · M4 Max, 64 GB+ | qwen3.6-35b | 256K |
109
114
  | NVIDIA 8–24 GB + 32 GB RAM | qwen3.6-35b | 256K |
110
- | NVIDIA 8 GB + 16 GB RAM | qwen3.5-4b | 64K |
115
+ | NVIDIA 8 GB + 16 GB RAM | gpt-oss-20b | 64K |
111
116
 
112
117
  On an RTX 4080 Laptop GPU (12 GB) the default model generates 50–60 tokens/s at 256K context, and
113
118
  qwen3.5-9b 64 tokens/s at 128K. See
@@ -17,7 +17,7 @@ lcode config path # print the file location
17
17
  |---|---|---|
18
18
  | `model` | `qwen3.6-35b` | Catalog key (see `lcode models`) or any installed Ollama model tag |
19
19
  | `context` | largest that fits | Context window in tokens; accepts `128k`, `1m` |
20
- | `num_batch` | per model | Prompt batch size. Larger reads prompts faster but needs more GPU memory |
20
+ | `num_batch` | 512 | Prompt batch size. Larger reads prompts faster but needs more GPU memory |
21
21
  | `keep_alive` | `30m` | How long Ollama keeps the model in memory after the last request |
22
22
  | `ollama_host` | `http://localhost:11434` | Ollama server URL |
23
23
  | `permission_mode` | `ask` | `ask`, `auto-edit` or `yolo` |
@@ -53,14 +53,17 @@ Environment variables override the file, which is useful for one-off runs and CI
53
53
  | `~/.config/lcode/config.toml` | Settings |
54
54
  | `~/.local/state/lcode/sessions/` | Saved conversations (for `lcode -c`) |
55
55
  | `~/.local/state/lcode/history` | Prompt history (++up++ in the prompt) |
56
+ | `~/.local/state/lcode/limits.json` | Context sizes that ran out of GPU memory on this machine (safe to delete) |
56
57
 
57
58
  Sessions contain everything the model read, including file contents. Delete the folder to clear them.
58
59
 
59
60
  ## Tuning for speed and memory
60
61
 
61
- - **Out-of-memory errors:** lower the context (`/ctx 128k`) or the prompt batch
62
- (`lcode config set num_batch 512`).
63
- - **Slow first answers on big files:** a larger `num_batch` (1024–2048) reads prompts faster if your
64
- GPU has room. lcode uses 1024 for `qwen3.6-35b`, the most that fits a 12 GB GPU next to a 256K cache.
62
+ - **Out-of-memory errors:** lower the context (`/ctx 128k`), close other programs using the GPU, or
63
+ pick a smaller model (`/models`).
64
+ - **Slow first answers on big files:** a larger prompt batch reads prompts faster if your GPU has
65
+ room: `lcode config set num_batch 1024` makes `qwen3.6-35b` read prompts ~1.8x faster (~500 instead
66
+ of ~280 tokens/s on a 12 GB GPU). On a 12 GB GPU at 256K context it only fits when little else uses
67
+ VRAM; if the GPU runs out of memory before answering, lcode retries automatically with 512.
65
68
  - **Faster answers, less accuracy:** `--no-think` or `/think off`.
66
69
  - **Keep the model warm:** `keep_alive = "2h"` avoids reload delays between sessions but holds the memory.
@@ -52,7 +52,10 @@ Two lessons from tuning the default model on a 12 GB GPU shaped the defaults:
52
52
  out-of-memory crashes at large batch sizes. lcode creates text-only variants (`lcode-<key>`) that
53
53
  reuse the downloaded weights.
54
54
  - **Prompt batch size dominates prompt-reading speed** when experts sit in system RAM: 512 → 280
55
- tokens/s, 1024 → 500 tokens/s. 2048 (~700 tokens/s) doesn't fit next to a 256K cache on 12 GB.
55
+ tokens/s, 1024 → 500 tokens/s. But the model's speculative-decoding context grows with the batch
56
+ (0.8 GB at 512, 1.1 GB at 1024), and at 1024 it only fits next to a 256K cache on 12 GB when little
57
+ else uses VRAM; otherwise CUDA fails with "illegal memory access". lcode therefore uses Ollama's
58
+ default of 512 and, if a larger configured batch runs out of memory, retries with 512.
56
59
 
57
60
  ## Security model
58
61
 
@@ -52,21 +52,29 @@ curl -fsSL https://nasser1941.github.io/lcode/install.sh | bash
52
52
  ```
53
53
 
54
54
  The [installer](https://github.com/nasser1941/lcode/blob/main/install.sh) installs
55
- [uv](https://docs.astral.sh/uv/) if needed, installs lcode with its own Python into `~/.local/bin`,
56
- checks that Ollama is running, and starts `lcode setup`.
55
+ [uv](https://docs.astral.sh/uv/) if needed, installs the latest release of lcode from
56
+ [PyPI](https://pypi.org/project/lcode-cli/) with its own Python into `~/.local/bin`, checks that
57
+ Ollama is running, and starts `lcode setup`.
57
58
 
58
59
  ??? note "Prefer to do it by hand?"
59
60
 
60
- With [uv](https://docs.astral.sh/uv/getting-started/installation/):
61
+ lcode is published on PyPI as `lcode-cli` (the command is `lcode`). With
62
+ [uv](https://docs.astral.sh/uv/getting-started/installation/):
61
63
 
62
64
  ```bash
63
- uv tool install git+https://github.com/nasser1941/lcode
65
+ uv tool install lcode-cli
64
66
  ```
65
67
 
66
68
  Or with [pipx](https://pipx.pypa.io):
67
69
 
68
70
  ```bash
69
- pipx install git+https://github.com/nasser1941/lcode
71
+ pipx install lcode-cli
72
+ ```
73
+
74
+ To try the latest unreleased changes from `main`:
75
+
76
+ ```bash
77
+ uv tool install git+https://github.com/nasser1941/lcode
70
78
  ```
71
79
 
72
80
  lcode needs Python 3.10 or newer. The macOS system Python is too old, which is why uv (it
@@ -138,7 +146,7 @@ steps. NVIDIA GPUs work inside WSL2 with the regular Windows driver.
138
146
  ## Upgrade and uninstall
139
147
 
140
148
  ```bash
141
- uv tool upgrade lcode-cli # upgrade
149
+ uv tool upgrade lcode-cli # upgrade (pipx: pipx upgrade lcode-cli)
142
150
  uv tool uninstall lcode-cli # remove lcode
143
151
  rm -rf ~/.config/lcode ~/.local/state/lcode # remove settings and saved sessions
144
152
  ollama rm lcode-qwen3.6-35b qwen3.6:35b-a3b-coding # remove downloaded models
@@ -8,17 +8,17 @@ size, and it works with any other Ollama model that supports tool calling.
8
8
  | Key | Model | Type | Download | Max context | Status |
9
9
  |---|---|---|---|---|---|
10
10
  | `qwen3.6-35b` | Qwen3.6 35B-A3B Coding | MoE, 3B active | 22.6 GB | 256K | **default**, tested |
11
- | `qwen3.8-27b` | Qwen3.8 27B | dense | 17.7 GB | 256K | untested |
12
- | `qwen3.6-27b` | Qwen3.6 27B Coding | dense | 17.8 GB | 256K | untested |
13
- | `laguna-xs-2.1` | Poolside Laguna XS 2.1 | MoE, 3B active | 20.3 GB | 256K | untested |
14
- | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning | hybrid MoE, 3B active | 25.4 GB | 1M | untested |
15
- | `gpt-oss-20b` | OpenAI gpt-oss 20B | MoE, 3.6B active | 13.8 GB | 128K | untested |
11
+ | `qwen3.8-27b` | Qwen3.8 27B | dense | 17.7 GB | 256K | tested |
12
+ | `qwen3.6-27b` | Qwen3.6 27B Coding | dense | 17.8 GB | 256K | tested |
13
+ | `laguna-xs-2.1` | Poolside Laguna XS 2.1 | MoE, 3B active | 20.3 GB | 256K | tested |
14
+ | `nemotron-3.5-lightning` | NVIDIA Nemotron 3.5 Lightning | hybrid MoE, 3B active | 25.4 GB | 1M | tested |
16
15
  | `qwen3.5-9b` | Qwen3.5 9B | dense | 6.6 GB | 256K | tested |
16
+ | `gpt-oss-20b` | OpenAI gpt-oss 20B | MoE, 3.6B active | 13.8 GB | 128K | tested |
17
17
  | `qwen3.5-4b` | Qwen3.5 4B | dense | 3.4 GB | 256K | tested |
18
18
 
19
- *Tested* means the maintainers verified multi-step tool use (exploring a repo, editing files,
20
- running commands) end to end. Untested models are expected to work; please
21
- [report how they do](https://github.com/nasser1941/lcode/issues/new?template=model_request.yml).
19
+ *Tested* means the model passed lcode's two acceptance tasks end to end (see
20
+ [test results](#test-results)). Results on other hardware are very welcome: please
21
+ [report how a model does](https://github.com/nasser1941/lcode/issues/new?template=model_request.yml).
22
22
 
23
23
  Run `lcode models` to see the same list with what fits on **your** machine:
24
24
 
@@ -112,10 +112,10 @@ makes long windows affordable:
112
112
  | Model | KV cache per token | 32K | 128K | 256K |
113
113
  |---|---|---|---|---|
114
114
  | qwen3.6-35b | 22 KiB | 23 GB | 25 GB | 28 GB (measured: 26 GB) |
115
- | qwen3.8-27b / qwen3.6-27b | 68 KiB | 20 GB | 26 GB | 35 GB |
116
- | laguna-xs-2.1 | ~40 KiB (estimated) | 21 GB | 25 GB | 30 GB |
117
- | nemotron-3.5-lightning | 7 KiB | 25 GB | 26 GB | 26 GB (1M: 32 GB) |
118
- | gpt-oss-20b | 24 KiB | 15 GB | 17 GB | — |
115
+ | qwen3.8-27b / qwen3.6-27b | 68 KiB | 20 GB | 26 GB (measured: 24 GB) | 35 GB |
116
+ | laguna-xs-2.1 | 40 KiB | 21 GB | 25 GB | 30 GB (measured: 23 GB) |
117
+ | nemotron-3.5-lightning | 7 KiB | 25 GB | 26 GB | 26 GB (measured: 26 GB; 1M: 32 GB) |
118
+ | gpt-oss-20b | 24 KiB | 15 GB | 17 GB (measured: 14 GB) | — |
119
119
  | qwen3.5-9b | 32 KiB | 8 GB | 11 GB (measured: 9.8 GB) | 15 GB (measured: 16 GB) |
120
120
  | qwen3.5-4b | 32 KiB | 5 GB | 8 GB (measured: 8.0 GB) | 12 GB |
121
121
 
@@ -140,11 +140,44 @@ results on your machine are welcome in the
140
140
  | Mac with M4 Pro / M4 Max, 36 GB | ~24 GB | qwen3.6-35b | 64K | fast |
141
141
  | Mac with M4 Pro, 48 GB | ~36 GB | qwen3.6-35b | 256K | fast |
142
142
  | Mac with M4 Max, 64–128 GB | 48–96 GB | qwen3.6-35b | 256K | fast |
143
- | NVIDIA 8 GB + 16 GB RAM | ~16 GB | qwen3.5-4b | 64K | fast |
143
+ | NVIDIA 8 GB + 16 GB RAM | ~16 GB | gpt-oss-20b | 64K | good (experts in RAM) |
144
144
  | NVIDIA 8–16 GB + 32 GB RAM | 32–40 GB | qwen3.6-35b | 256K | good (experts in RAM) |
145
145
  | NVIDIA 24 GB + 64 GB RAM | ~80 GB | qwen3.6-35b | 256K | good (experts in RAM) |
146
146
  | CPU only, 32 GB RAM | ~24 GB | qwen3.5-4b | 256K | slow |
147
147
 
148
+ ### Test results
149
+
150
+ Every catalog model runs the same two tasks through lcode with its default settings for the machine
151
+ (the context lcode picks, `yolo` mode, reasoning on):
152
+
153
+ 1. **Bug fix:** tests fail in a small project; the model must find the bug, fix it and re-run the tests.
154
+ 2. **Repo question + script:** in the `requests` source, say where the `Authorization` header is
155
+ stripped on redirects with `file:line` citations, then write and run an `ast`-based script.
156
+
157
+ On an RTX 4080 Laptop GPU (12 GB), i9-13980HX, 32 GB RAM:
158
+
159
+ | Model | Context | Memory | On GPU | Bug fix | Repo question + script | Speed |
160
+ |---|---|---|---|---|---|---|
161
+ | qwen3.6-35b | 256K | 23.0 GB | 25% | ✓ 30 s | ✓ 45 s | 50–55 tok/s |
162
+ | qwen3.8-27b | 128K | 24.4 GB | 29% | ✓ 86 s | ✓ 284 s | 7–8 tok/s |
163
+ | qwen3.6-27b | 128K | 24.3 GB | 29% | ✓ 78 s | ✓ 267 s | 7 tok/s |
164
+ | laguna-xs-2.1 | 256K | 22.6 GB | 15% | ✓ 67 s | ✓ 195 s | 15–38 tok/s |
165
+ | nemotron-3.5-lightning | 512K² | 27.2 GB | 25% | ✓ 62 s | ✓ 127 s | 44 tok/s |
166
+ | qwen3.5-9b | 128K | 9.8 GB | 100% | ✓ 21 s | ✓ 37 s (2 of 3 runs)¹ | 61–64 tok/s |
167
+ | gpt-oss-20b | 128K | 14.3 GB | 54% | ✓ 16 s | ✓ 36 s | 43 tok/s |
168
+ | qwen3.5-4b | 128K | 8.0 GB | 100% | ✓ 21 s | ✓ 24 s | 93–97 tok/s |
169
+
170
+ ¹ In one run the 9B saved the script in the wrong folder. `gpt-oss-20b` and `nemotron-3.5-lightning`
171
+ sometimes call tools with arguments that don't exist; they correct themselves from lcode's error
172
+ messages.
173
+ ² At its full 1M context Nemotron's cache (7 GB) has to sit in VRAM next to the model and doesn't fit
174
+ on a 12 GB GPU; 512K (3.5 GB) loads fine. lcode handles this by itself: when a model doesn't fit, it
175
+ retries with half the context and remembers the size that worked (see
176
+ [Troubleshooting](troubleshooting.md#out-of-memory-or-cuda-error-an-illegal-memory-access)).
177
+
178
+ The dense 27B models are accurate but slow here because only ~30% of them fits in 12 GB of VRAM;
179
+ on a 24 GB GPU or a Mac with enough unified memory they run fully accelerated.
180
+
148
181
  ### Measured performance
149
182
 
150
183
  On an RTX 4080 Laptop GPU (12 GB) with an i9-13980HX and 32 GB RAM, `qwen3.6-35b` at 256K context:
@@ -152,7 +185,7 @@ On an RTX 4080 Laptop GPU (12 GB) with an i9-13980HX and 32 GB RAM, `qwen3.6-35b
152
185
  | | |
153
186
  |---|---|
154
187
  | Generation | 50–60 tokens/s on code (speculative decoding with the model's multi-token prediction) |
155
- | Prompt reading | ~500 tokens/s: a 60K-token chunk of code takes about 2 minutes the first time |
188
+ | Prompt reading | ~280 tokens/s (~500 with `num_batch 1024`): a 60K-token chunk of code takes 2–4 minutes the first time |
156
189
  | Follow-up turns | start in 1–2 s: Ollama reuses the cached prompt |
157
190
  | A real task | "explain X with file:line citations, then write and run a script" in about 2 minutes |
158
191
 
@@ -180,5 +213,5 @@ Apple Silicon numbers are not measured yet; please share yours.
180
213
 
181
214
  Some models (the Qwen3.6 family, for example) ship with a vision encoder that a coding agent doesn't
182
215
  use. `lcode setup` creates a text-only variant named `lcode-<key>` that reuses the downloaded
183
- weights, so it takes no extra disk space, and frees about 1 GB of GPU memory. On a 12 GB GPU that
184
- memory is what allows a larger prompt batch, which reads prompts ~1.8x faster.
216
+ weights, so it takes no extra disk space, and frees about 1 GB of GPU memory for the context cache
217
+ (and for a larger prompt batch, where it fits; see [tuning](configuration.md#tuning-for-speed-and-memory)).
@@ -12,7 +12,7 @@ lcode
12
12
 
13
13
  ```text
14
14
  ╭──────────────────────────────────────────────╮
15
- │ lcode v0.1.0 — local coding agent │
15
+ │ lcode v0.1.2 — local coding agent │
16
16
  │ │
17
17
  │ model lcode-qwen3.6-35b │
18
18
  │ context 256K tokens │
@@ -86,7 +86,8 @@ Read-only commands such as `ls`, `cat`, `grep` and `git status` run without aski
86
86
 
87
87
  - ++ctrl+c++ interrupts the model or a running command; you keep the conversation.
88
88
  - `/compact` summarizes a long conversation to free context (this also happens automatically at 85%).
89
- - `lcode -c` resumes the last session in the current directory.
89
+ - `lcode -c` continues the last session in this folder; `/rename` names a session and `/resume`
90
+ picks one from a list.
90
91
  - `/help` lists everything else.
91
92
 
92
93
  Next: [everything lcode can do](usage.md), or [choosing models and context windows](models.md).
@@ -34,10 +34,17 @@ export PATH="$HOME/.local/bin:$PATH"
34
34
 
35
35
  ### Out of memory, or `CUDA error: an illegal memory access`
36
36
 
37
- The model plus its context don't fit.
37
+ The model plus its context don't fit. lcode handles the common cases itself: if a model fails to load
38
+ because of GPU memory, it first retries with a prompt batch of 512 (if you raised it), then with half
39
+ the context, until it loads. The context that worked is remembered in `~/.local/state/lcode/limits.json`,
40
+ so the next session starts there; `lcode doctor` shows it. Delete that file to let lcode try larger
41
+ contexts again (for example after a GPU upgrade, or if another program was using the GPU at the time).
42
+
43
+ If it still fails:
38
44
 
39
45
  1. Lower the context: `/ctx 128k` in a session or `lcode config set context 128k`.
40
- 2. Lower the prompt batch: `lcode config set num_batch 512`.
46
+ 2. If you raised the prompt batch, lower it again: `lcode config unset num_batch`. (When the GPU runs
47
+ out of memory before answering, lcode already retries once with a batch of 512.)
41
48
  3. Make sure no other model is loaded (`ollama ps`, then `ollama stop <name>`) and that other GPU
42
49
  apps are closed.
43
50
  4. Use a smaller model: `lcode models`.
@@ -49,8 +56,8 @@ of the text-only variant, run `lcode setup <key>` once to create the variant.
49
56
 
50
57
  - The first request loads the model (10–45 s). lcode starts loading in the background as soon as it
51
58
  starts, and `keep_alive` keeps it loaded between requests.
52
- - Reading many large files takes a while the first time (~500 tokens/s on a 12 GB GPU); follow-up
53
- turns reuse the cache.
59
+ - Reading many large files takes a while the first time (~280 tokens/s on a 12 GB GPU, ~500 with
60
+ `lcode config set num_batch 1024` if it fits); follow-up turns reuse the cache.
54
61
  - `ollama ps` shows how much of the model is on the GPU. Dense models are much slower when split;
55
62
  prefer the MoE models in `lcode models` on smaller GPUs.
56
63
  - `--no-think` skips reasoning for simple requests.
@@ -16,7 +16,8 @@ lcode config [set|unset] # show or change settings
16
16
  | `-m, --model NAME` | Catalog key (see `lcode models`) or any installed Ollama model |
17
17
  | `--context SIZE` | Context window, e.g. `65536`, `128k`, `1m` |
18
18
  | `-r, --repo DIR` | Work in another directory |
19
- | `-c, --continue` | Resume the last session in this directory |
19
+ | `-c, --continue` | Continue the most recently used session in this directory |
20
+ | `--resume [SESSION]` | Resume a saved session: pick from a list, or give its number, name or id |
20
21
  | `--auto-edit` | Apply file edits without asking (commands still ask) |
21
22
  | `--yolo` | Never ask for permission |
22
23
  | `--no-think` | Turn off the model's reasoning: faster, less accurate |
@@ -44,7 +45,9 @@ reasoning is on.
44
45
  |---|---|
45
46
  | `/help` | List commands and keys |
46
47
  | `/init` | Analyze the repository and write `AGENTS.md` (read at every start) |
47
- | `/clear` | Start a new conversation |
48
+ | `/clear` | Start a new conversation (the current one stays saved) |
49
+ | `/rename NAME` | Name the current session so it's easy to find later |
50
+ | `/resume [SESSION]` | Resume a saved session: pick from a list, or give its number or name; `/resume all` lists every folder |
48
51
  | `/compact [focus]` | Summarize the conversation to free context |
49
52
  | `/context` | Show how full the context window is |
50
53
  | `/ctx [size]` | Show or change the context window |
@@ -100,8 +103,35 @@ conventions. `/init` writes a first version for you.
100
103
 
101
104
  ## Sessions
102
105
 
103
- Every session is saved to `~/.local/state/lcode/sessions/`. `lcode -c` resumes the most recent
104
- session for the current directory, with the full conversation.
106
+ Every conversation is saved after each request to `~/.local/state/lcode/sessions/`, so you can
107
+ leave and pick up where you stopped.
108
+
109
+ ```text
110
+ ❯ /rename auth refactor # name the current session
111
+ ❯ /resume # list this folder's sessions and pick one
112
+ ```
113
+
114
+ ```text
115
+ Saved sessions · /home/you/code/my-project
116
+ # Session Last used Requests
117
+ 1 auth refactor 2 h ago 14
118
+ Explain how login tokens are validated
119
+ 2 The tests in tests/test_parser.py fail… yesterday 6
120
+ Resume which session? (number or name, Enter to cancel): 1
121
+ ```
122
+
123
+ Unnamed sessions are listed by their first request. After resuming, lcode shows your last request
124
+ and the start of its last answer so you know where you left off.
125
+
126
+ | From the shell | From a session | Resumes |
127
+ |---|---|---|
128
+ | `lcode -c` | | the most recently used session in this folder |
129
+ | `lcode --resume` | `/resume` | one you pick from a list |
130
+ | `lcode --resume "auth refactor"` | `/resume auth refactor` | a session by name, list number, id or a unique part of its title |
131
+ | | `/resume all` | a session from any folder (lcode switches to that folder) |
132
+
133
+ `/clear` starts a new conversation and keeps the old one saved. Files may have changed since a
134
+ session was saved, so after resuming the model has to read a file again before editing it.
105
135
 
106
136
  ## Scripting
107
137
 
@@ -5,14 +5,15 @@
5
5
  #
6
6
  # What it does, step by step:
7
7
  # 1. installs uv (https://docs.astral.sh/uv/) if missing; uv provides an isolated Python 3.12
8
- # 2. installs lcode with `uv tool install` into ~/.local/bin
8
+ # 2. installs lcode from PyPI (package lcode-cli) with `uv tool install` into ~/.local/bin
9
9
  # 3. checks that Ollama is installed and running, and tells you how to install it if not
10
10
  # 4. runs `lcode setup`, which picks the best model for your hardware and downloads it
11
11
  #
12
- # Environment overrides: LCODE_SOURCE (pip-style source to install from), LCODE_SKIP_SETUP=1.
12
+ # Environment overrides: LCODE_SOURCE (what to install, e.g. git+https://github.com/nasser1941/lcode for
13
+ # the latest main), LCODE_SKIP_SETUP=1.
13
14
  set -euo pipefail
14
15
 
15
- SOURCE="${LCODE_SOURCE:-git+https://github.com/nasser1941/lcode@main}"
16
+ SOURCE="${LCODE_SOURCE:-lcode-cli}"
16
17
 
17
18
  bold() { printf '\033[1m%s\033[0m\n' "$*"; }
18
19
  info() { printf '\033[1;36m==>\033[0m %s\n' "$*"; }
@@ -1,3 +1,3 @@
1
1
  """lcode — a local-first terminal coding agent powered by open-weight models via Ollama."""
2
2
 
3
- __version__ = "0.1.1"
3
+ __version__ = "0.1.2"