lcode-cli 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lcode/models.toml ADDED
@@ -0,0 +1,128 @@
1
+ # The model catalog: open-weight models lcode knows how to size and set up.
2
+ #
3
+ # Want another model? Open a "Model request" issue or a pull request that adds an entry here
4
+ # (see CONTRIBUTING.md). Any other Ollama model with tool calling also works with
5
+ # `lcode --model <ollama-tag>`; lcode just can't estimate its memory needs.
6
+ #
7
+ # Fields
8
+ # key short name used on the command line (`lcode --model <key>`)
9
+ # tag Ollama model tag that `lcode setup` pulls
10
+ # size_gb download size (GB, 10^9 bytes)
11
+ # max_context largest context window the model supports (tokens)
12
+ # kv_kib_per_token KV-cache memory per token of context at f16 (KiB), computed from the GGUF
13
+ # metadata: attention layers x KV heads x (key_length + value_length) x 2 bytes
14
+ # kv_estimated true when the attention layout had to be assumed rather than read
15
+ # moe Mixture-of-Experts: runs well even when part of it sits in system RAM
16
+ # num_batch prompt batch size lcode uses by default (omit for Ollama's default of 512)
17
+ # tested verified end to end with lcode's tools by the maintainers
18
+ #
19
+ # The order matters: `lcode setup` recommends the first *tested* model that fits the machine, and
20
+ # otherwise the first model in this order that fits with a useful context, so keep stronger models first.
21
+
22
+ [[model]]
23
+ key = "qwen3.6-35b"
24
+ tag = "qwen3.6:35b-a3b-coding"
25
+ name = "Qwen3.6 35B-A3B Coding"
26
+ publisher = "Qwen (Alibaba)"
27
+ params = "35B MoE · 3B active"
28
+ moe = true
29
+ size_gb = 22.6
30
+ max_context = 262144
31
+ kv_kib_per_token = 22
32
+ num_batch = 1024
33
+ swe_bench_verified = 73.4
34
+ tested = true
35
+ notes = "Default. Strongest tool calling among local MoE coders; fast even with experts in RAM."
36
+
37
+ [[model]]
38
+ key = "qwen3.8-27b"
39
+ tag = "qwen3.8:27b"
40
+ name = "Qwen3.8 27B"
41
+ publisher = "Qwen (Alibaba)"
42
+ params = "27B dense"
43
+ moe = false
44
+ size_gb = 17.7
45
+ max_context = 262144
46
+ kv_kib_per_token = 68
47
+ tested = false
48
+ notes = "Newest Qwen. Dense: excellent when it fits entirely in GPU/unified memory, slow when split."
49
+
50
+ [[model]]
51
+ key = "qwen3.6-27b"
52
+ tag = "qwen3.6:27b-coding"
53
+ name = "Qwen3.6 27B Coding"
54
+ publisher = "Qwen (Alibaba)"
55
+ params = "27B dense"
56
+ moe = false
57
+ size_gb = 17.8
58
+ max_context = 262144
59
+ kv_kib_per_token = 68
60
+ tested = false
61
+ notes = "Dense coding model. Best on 24 GB+ GPUs or Macs with 48 GB+ unified memory."
62
+
63
+ [[model]]
64
+ key = "laguna-xs-2.1"
65
+ tag = "laguna-xs-2.1"
66
+ name = "Laguna XS 2.1"
67
+ publisher = "Poolside"
68
+ params = "33B MoE · 3B active"
69
+ moe = true
70
+ size_gb = 20.3
71
+ max_context = 262144
72
+ kv_kib_per_token = 40
73
+ kv_estimated = true
74
+ swe_bench_verified = 70.9
75
+ tested = false
76
+ notes = "Agentic-coding MoE built for local machines; slightly smaller than Qwen3.6 35B."
77
+
78
+ [[model]]
79
+ key = "nemotron-3.5-lightning"
80
+ tag = "nemotron-3.5-lightning:30b"
81
+ name = "Nemotron 3.5 Lightning"
82
+ publisher = "NVIDIA"
83
+ params = "30B MoE · 3B active"
84
+ moe = true
85
+ size_gb = 25.4
86
+ max_context = 1048576
87
+ kv_kib_per_token = 7
88
+ tested = false
89
+ notes = "Hybrid Mamba-Transformer: tiny KV cache, up to 1M tokens of context."
90
+
91
+ [[model]]
92
+ key = "gpt-oss-20b"
93
+ tag = "gpt-oss:20b"
94
+ name = "gpt-oss 20B"
95
+ publisher = "OpenAI"
96
+ params = "21B MoE · 3.6B active"
97
+ moe = true
98
+ size_gb = 13.8
99
+ max_context = 131072
100
+ kv_kib_per_token = 24
101
+ tested = false
102
+ notes = "Compact MoE with 128K context; fits 16 GB GPUs and 24 GB Macs."
103
+
104
+ [[model]]
105
+ key = "qwen3.5-9b"
106
+ tag = "qwen3.5:9b"
107
+ name = "Qwen3.5 9B"
108
+ publisher = "Qwen (Alibaba)"
109
+ params = "9B dense"
110
+ moe = false
111
+ size_gb = 6.6
112
+ max_context = 262144
113
+ kv_kib_per_token = 32
114
+ tested = true
115
+ notes = "Small and capable: 64 tok/s fully on a 12 GB GPU at 128K context. For 8-12 GB GPUs and 16-24 GB Macs."
116
+
117
+ [[model]]
118
+ key = "qwen3.5-4b"
119
+ tag = "qwen3.5:4b"
120
+ name = "Qwen3.5 4B"
121
+ publisher = "Qwen (Alibaba)"
122
+ params = "4B dense"
123
+ moe = false
124
+ size_gb = 3.4
125
+ max_context = 262144
126
+ kv_kib_per_token = 32
127
+ tested = true
128
+ notes = "Smallest option: ~95 tok/s on a 12 GB GPU. Makes more mistakes but fixes them when commands fail."
lcode/ollama.py ADDED
@@ -0,0 +1,163 @@
1
+ """A small client for the Ollama HTTP API (works with native, Docker and remote servers)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from collections.abc import Callable, Iterator
8
+ from urllib.parse import urlsplit, urlunsplit
9
+
10
+ import requests
11
+
12
+ MIN_VERSION = (0, 30, 0) # first release that runs the Qwen3.5/3.6 hybrid-attention architecture
13
+
14
+
15
+ class OllamaError(RuntimeError):
16
+ pass
17
+
18
+
19
+ def redact(url: str) -> str:
20
+ """Hide credentials embedded in a URL (http://user:pass@host) before printing it."""
21
+ parts = urlsplit(url)
22
+ if parts.username or parts.password:
23
+ host = parts.hostname or ""
24
+ if parts.port:
25
+ host += f":{parts.port}"
26
+ return urlunsplit((parts.scheme, f"***@{host}", parts.path, parts.query, parts.fragment))
27
+ return url
28
+
29
+
30
+ def version_tuple(version: str) -> tuple[int, ...]:
31
+ return tuple(int(x) for x in re.findall(r"\d+", version)[:3])
32
+
33
+
34
+ class Ollama:
35
+ def __init__(self, host: str):
36
+ self.host = host.rstrip("/")
37
+
38
+ def _url(self, path: str) -> str:
39
+ return f"{self.host}{path}"
40
+
41
+ def _post(self, path: str, payload: dict, timeout: float | None = 30) -> dict:
42
+ try:
43
+ r = requests.post(self._url(path), json=payload, timeout=timeout)
44
+ except requests.RequestException as e:
45
+ raise OllamaError(f"cannot reach Ollama at {redact(self.host)}: {e}") from e
46
+ if r.status_code != 200:
47
+ raise OllamaError(f"Ollama {path} failed ({r.status_code}): {r.text[:300]}")
48
+ return r.json()
49
+
50
+ # -- info
51
+ def version(self) -> str:
52
+ try:
53
+ return requests.get(self._url("/api/version"), timeout=5).json()["version"]
54
+ except (requests.RequestException, ValueError, KeyError) as e:
55
+ raise OllamaError(f"cannot reach Ollama at {redact(self.host)}: {e}") from e
56
+
57
+ def list_models(self) -> list[dict]:
58
+ try:
59
+ return requests.get(self._url("/api/tags"), timeout=10).json().get("models", [])
60
+ except (requests.RequestException, ValueError) as e:
61
+ raise OllamaError(f"cannot reach Ollama at {redact(self.host)}: {e}") from e
62
+
63
+ def installed_names(self) -> set[str]:
64
+ names = set()
65
+ for m in self.list_models():
66
+ names.add(m["name"])
67
+ if m["name"].endswith(":latest"):
68
+ names.add(m["name"][: -len(":latest")])
69
+ return names
70
+
71
+ def show(self, model: str) -> dict:
72
+ return self._post("/api/show", {"model": model})
73
+
74
+ def max_context(self, model: str) -> int | None:
75
+ info = self.show(model).get("model_info", {})
76
+ return next((int(v) for k, v in info.items() if k.endswith(".context_length")), None)
77
+
78
+ def running(self) -> list[dict]:
79
+ try:
80
+ return requests.get(self._url("/api/ps"), timeout=5).json().get("models", [])
81
+ except (requests.RequestException, ValueError):
82
+ return []
83
+
84
+ # -- management
85
+ def pull(self, model: str, on_progress: Callable[[dict], None]) -> None:
86
+ try:
87
+ with requests.post(
88
+ self._url("/api/pull"), json={"model": model, "stream": True}, stream=True, timeout=(10, None)
89
+ ) as r:
90
+ if r.status_code != 200:
91
+ raise OllamaError(f"pull failed ({r.status_code}): {r.text[:300]}")
92
+ for line in r.iter_lines():
93
+ if line:
94
+ event = json.loads(line)
95
+ if "error" in event:
96
+ raise OllamaError(f"pull failed: {event['error']}")
97
+ on_progress(event)
98
+ except requests.RequestException as e:
99
+ raise OllamaError(f"pull failed: {e}") from e
100
+
101
+ def create_text_only(self, name: str, base: str) -> bool:
102
+ """Create `name` from `base` without its vision projector. Returns False if `base` has none.
103
+
104
+ Only the language-model weights are referenced (no copy, no extra disk). Dropping the
105
+ projector saves ~1 GB of VRAM, which Ollama under-counts on small GPUs.
106
+ """
107
+ show = self.show(base)
108
+ modelfile = show.get("modelfile", "")
109
+ blobs = re.findall(r"^FROM .*?sha256[-:]([0-9a-f]{64})", modelfile, re.M)
110
+ if len(blobs) < 2 and "projector_info" not in show:
111
+ return False
112
+ if not blobs:
113
+ raise OllamaError(f"could not find the weights of {base}")
114
+ params: dict[str, list[str]] = {}
115
+ for line in show.get("parameters", "").splitlines():
116
+ key, _, value = line.strip().partition(" ")
117
+ if key:
118
+ params.setdefault(key, []).append(value.strip().strip('"'))
119
+ payload: dict = {
120
+ "model": name,
121
+ "files": {"model.gguf": f"sha256:{blobs[0]}"},
122
+ "template": show.get("template", ""),
123
+ "parameters": {k: (v if k == "stop" else _number(v[-1])) for k, v in params.items()},
124
+ "stream": False,
125
+ }
126
+ for key in ("renderer", "parser"):
127
+ m = re.search(rf"^{key.upper()} (\S+)", modelfile, re.M)
128
+ if m:
129
+ payload[key] = m.group(1)
130
+ if self._post("/api/create", payload, timeout=600).get("status") != "success":
131
+ raise OllamaError(f"creating {name} failed")
132
+ return True
133
+
134
+ # -- inference
135
+ def chat_stream(self, payload: dict) -> Iterator[dict]:
136
+ try:
137
+ r = requests.post(self._url("/api/chat"), json={**payload, "stream": True}, stream=True, timeout=(10, None))
138
+ except requests.RequestException as e:
139
+ raise OllamaError(f"cannot reach Ollama at {redact(self.host)}: {e}") from e
140
+ with r:
141
+ if r.status_code != 200:
142
+ raise OllamaError(f"Ollama error {r.status_code}: {r.text[:500]}")
143
+ for line in r.iter_lines():
144
+ if line:
145
+ chunk = json.loads(line)
146
+ if "error" in chunk:
147
+ raise OllamaError(f"Ollama error: {chunk['error']}")
148
+ yield chunk
149
+
150
+ def load(self, model: str, options: dict, keep_alive: str) -> None:
151
+ """Load a model into memory without generating anything."""
152
+ self._post(
153
+ "/api/chat", {"model": model, "messages": [], "keep_alive": keep_alive, "options": options}, timeout=900
154
+ )
155
+
156
+
157
+ def _number(value: str) -> int | float | str:
158
+ for cast in (int, float):
159
+ try:
160
+ return cast(value)
161
+ except ValueError:
162
+ pass
163
+ return value
lcode/permissions.py ADDED
@@ -0,0 +1,90 @@
1
+ """Permission prompts for file edits and shell commands."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import shlex
7
+
8
+ from rich.console import Console, RenderableType
9
+ from rich.panel import Panel
10
+
11
+ from lcode.config import PERMISSION_MODES
12
+
13
+ # Commands that only read state run without asking (when used without redirection or chaining).
14
+ SAFE_COMMANDS = {
15
+ "ls", "cat", "head", "tail", "wc", "pwd", "grep", "rg", "find", "tree", "file", "stat", "du", "df",
16
+ "echo", "which", "whoami", "uname", "date", "sort", "uniq", "cut", "nl", "diff", "basename", "dirname",
17
+ "realpath", "readlink", "nvidia-smi", "sw_vers",
18
+ } # fmt: skip
19
+ SAFE_GIT = {"status", "log", "diff", "show", "branch", "ls-files", "rev-parse", "remote", "blame", "shortlog"}
20
+ # For these, "always allow" is remembered per subcommand (e.g. `git commit`, not all of `git`).
21
+ MULTI_WORD = {"git", "npm", "pnpm", "yarn", "pip", "uv", "docker", "cargo", "go", "kubectl", "python", "python3", "brew"} # fmt: skip # noqa: E501
22
+
23
+
24
+ def bash_key(command: str) -> str:
25
+ """The key "always allow" remembers: the program (plus subcommand for tools like git), ignoring `cd dir &&`."""
26
+ command = re.sub(r"^\s*(cd\s+\S+\s*&&\s*)+", "", command)
27
+ try:
28
+ words = shlex.split(command)
29
+ except ValueError:
30
+ words = command.split()
31
+ if not words:
32
+ return "bash:"
33
+ if words[0] in MULTI_WORD and len(words) > 1:
34
+ if words[1] in ("-m", "-c") and len(words) > 2: # python -m <module>
35
+ return f"bash:{' '.join(words[:3])}"
36
+ return f"bash:{words[0]} {words[1]}"
37
+ return f"bash:{words[0]}"
38
+
39
+
40
+ def is_read_only(command: str) -> bool:
41
+ if any(tok in command for tok in (";", "&", ">", "`", "$(", "<(")) or "\n" in command:
42
+ return False
43
+ if re.search(r"-exec|-delete|-ok\b|-fprint", command):
44
+ return False
45
+ for segment in command.split("|"):
46
+ try:
47
+ words = shlex.split(segment)
48
+ except ValueError:
49
+ return False
50
+ if not words:
51
+ return False
52
+ if words[0] == "git":
53
+ if len(words) < 2 or words[1] not in SAFE_GIT:
54
+ return False
55
+ elif words[0] not in SAFE_COMMANDS:
56
+ return False
57
+ return True
58
+
59
+
60
+ class Permissions:
61
+ def __init__(self, console: Console, mode: str = "ask"):
62
+ self.console = console
63
+ self.mode = mode
64
+ self.always: set[str] = set()
65
+
66
+ def cycle(self) -> None:
67
+ self.mode = PERMISSION_MODES[(PERMISSION_MODES.index(self.mode) + 1) % len(PERMISSION_MODES)]
68
+
69
+ def request(self, key: str, kind: str, title: str, body: RenderableType) -> tuple[bool, str]:
70
+ """Ask the user. kind is 'edit' or 'bash'. Returns (allowed, message for the model)."""
71
+ if self.mode == "yolo" or key in self.always or (kind == "edit" and self.mode == "auto-edit"):
72
+ return True, ""
73
+ self.console.print(Panel(body, title=title, title_align="left", border_style="yellow"))
74
+ scope = key.split(":", 1)[1] if kind == "bash" else "file edits"
75
+ try:
76
+ answer = input(f" Allow? [y]es / [a]lways for '{scope}' this session / [n]o (+ optional reason): ")
77
+ except EOFError:
78
+ return False, "The user denied this action (no input available)."
79
+ answer = answer.strip()
80
+ if answer.lower() in ("y", "yes", ""):
81
+ return True, ""
82
+ if answer.lower() in ("a", "always"):
83
+ self.always.add(key if kind == "bash" else "edit")
84
+ return True, ""
85
+ # "n", "no", "n <reason>" or any other text, which is taken as the reason.
86
+ reason = re.sub(r"^(no|n)\b[\s,:-]*", "", answer, flags=re.I).strip()
87
+ message = "The user denied this action."
88
+ if reason:
89
+ return False, f"{message} User says: {reason}"
90
+ return False, f"{message} Ask the user how to proceed or choose a different approach."
lcode/render.py ADDED
@@ -0,0 +1,43 @@
1
+ """Terminal rendering helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from rich.console import Console
6
+ from rich.markdown import Markdown
7
+
8
+
9
+ class MarkdownStreamer:
10
+ """Render streamed markdown block by block, at blank lines outside code fences.
11
+
12
+ Rendering whole blocks (instead of re-rendering a live region) keeps scrollback clean and
13
+ works in every terminal.
14
+ """
15
+
16
+ def __init__(self, console: Console):
17
+ self.console = console
18
+ self.buf = ""
19
+
20
+ def feed(self, text: str) -> None:
21
+ self.buf += text
22
+ cut, offset, in_fence = 0, 0, False
23
+ for line in self.buf.split("\n")[:-1]: # the last line may still be incomplete
24
+ stripped = line.strip()
25
+ if stripped.startswith(("```", "~~~")):
26
+ in_fence = not in_fence
27
+ if not in_fence:
28
+ cut = offset + len(line) + 1
29
+ elif not in_fence and stripped == "":
30
+ cut = offset + len(line) + 1
31
+ offset += len(line) + 1
32
+ if cut:
33
+ self._render(self.buf[:cut])
34
+ self.buf = self.buf[cut:]
35
+
36
+ def flush(self) -> None:
37
+ self._render(self.buf)
38
+ self.buf = ""
39
+
40
+ def _render(self, text: str) -> None:
41
+ if text.strip():
42
+ self.console.print(Markdown(text.strip("\n"), code_theme="monokai"))
43
+ self.console.print()