notebook-llm-cli 0.1.1__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {notebook_llm_cli-0.1.1/notebook_llm_cli.egg-info → notebook_llm_cli-0.2.0}/PKG-INFO +19 -6
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/README.md +18 -5
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/__init__.py +4 -2
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/cli.py +192 -37
- notebook_llm_cli-0.2.0/notebook_llm/disk.py +73 -0
- notebook_llm_cli-0.2.0/notebook_llm/integrations.py +167 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/library.py +86 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/manager.py +35 -4
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/server.py +6 -1
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0/notebook_llm_cli.egg-info}/PKG-INFO +19 -6
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm_cli.egg-info/SOURCES.txt +2 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/pyproject.toml +1 -1
- notebook_llm_cli-0.2.0/tests/test_core.py +175 -0
- notebook_llm_cli-0.1.1/tests/test_core.py +0 -77
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/LICENSE +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/__main__.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/client.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/config.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/env.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/estimate.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/estimate_data.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/gpu.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/installer.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/tunnel.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm/ui.py +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm_cli.egg-info/dependency_links.txt +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm_cli.egg-info/entry_points.txt +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm_cli.egg-info/requires.txt +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/notebook_llm_cli.egg-info/top_level.txt +0 -0
- {notebook_llm_cli-0.1.1 → notebook_llm_cli-0.2.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: notebook-llm-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Interactive CLI to run Ollama LLMs on Kaggle/Colab GPUs: GPU detection, fit and tokens/sec estimates, library search, public tunnel.
|
|
5
5
|
Author-email: Shashan Lumbhani <lumbhanishashan1510@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -45,17 +45,29 @@ Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm
|
|
|
45
45
|
1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
|
|
46
46
|
2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
|
|
47
47
|
3. Recommended models for my GPU
|
|
48
|
-
4.
|
|
49
|
-
5.
|
|
50
|
-
6.
|
|
51
|
-
7.
|
|
52
|
-
8.
|
|
48
|
+
4. **Quant picker** - compare every quantization (q2 / q4_K_M / q5 / q8 / fp16 ...) of one model size: exact size, fit, ~tok/s, quality, and a suggested pick. Also reachable from any list by typing `q3` (= quants of row 3)
|
|
49
|
+
5. Installed models - load / unload / benchmark (real tok/s) / delete
|
|
50
|
+
6. **Connect your tools** - ready-to-paste settings for Continue, Cline, Aider, Open WebUI, Claude Code, Cursor, Antigravity IDE and Python/curl, filled in with your tunnel link, model and context length. Save them all to a file with `s`
|
|
51
|
+
7. Tunnel - open, show link, close, new link
|
|
52
|
+
8. Server - start / stop / restart / logs / GPU report
|
|
53
|
+
9. Settings - context length, KV cache type, flash attention, keep-alive, models folder (saved)
|
|
54
|
+
10. Benchmark the active model
|
|
55
|
+
|
|
56
|
+
### Disk-space check
|
|
57
|
+
Before every download the tool compares the model size (+5% +1 GB) with free space in the models folder. If it won't fit you can delete models on the spot, pick a smaller quant, or continue anyway. The models folder is chosen automatically (the disk with the most free space, e.g. `/kaggle/temp`) and shown in the header; change it in Settings.
|
|
58
|
+
|
|
59
|
+
### Tool notes
|
|
60
|
+
- Cursor sends requests from its own servers, so it needs the public tunnel link (that's what this tool gives you).
|
|
61
|
+
- Claude Code uses Ollama's Anthropic-compatible API (`ANTHROPIC_BASE_URL`); it needs a tool-capable model and a large context.
|
|
62
|
+
- Antigravity IDE has no custom-model setting, so use the Continue/Cline extension inside it.
|
|
53
63
|
|
|
54
64
|
## Non-interactive
|
|
55
65
|
```
|
|
56
66
|
notebook-llm gpus # hardware + recommended table
|
|
57
67
|
notebook-llm search coder
|
|
58
68
|
notebook-llm check llama3.1:70b --ctx 8192
|
|
69
|
+
notebook-llm quants qwen2.5-coder:7b # compare quantizations
|
|
70
|
+
notebook-llm connect cursor # or: continue cline aider openwebui claude-code antigravity python (none = all)
|
|
59
71
|
notebook-llm quickstart --model auto
|
|
60
72
|
notebook-llm status | stop
|
|
61
73
|
```
|
|
@@ -66,6 +78,7 @@ from notebook_llm import NotebookLLM
|
|
|
66
78
|
llm = NotebookLLM()
|
|
67
79
|
url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
|
|
68
80
|
llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
|
|
81
|
+
print(llm.connect("cursor")) # settings text for a tool; llm.connect() = all tools
|
|
69
82
|
```
|
|
70
83
|
|
|
71
84
|
## How the estimates work
|
|
@@ -23,17 +23,29 @@ Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm
|
|
|
23
23
|
1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
|
|
24
24
|
2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
|
|
25
25
|
3. Recommended models for my GPU
|
|
26
|
-
4.
|
|
27
|
-
5.
|
|
28
|
-
6.
|
|
29
|
-
7.
|
|
30
|
-
8.
|
|
26
|
+
4. **Quant picker** - compare every quantization (q2 / q4_K_M / q5 / q8 / fp16 ...) of one model size: exact size, fit, ~tok/s, quality, and a suggested pick. Also reachable from any list by typing `q3` (= quants of row 3)
|
|
27
|
+
5. Installed models - load / unload / benchmark (real tok/s) / delete
|
|
28
|
+
6. **Connect your tools** - ready-to-paste settings for Continue, Cline, Aider, Open WebUI, Claude Code, Cursor, Antigravity IDE and Python/curl, filled in with your tunnel link, model and context length. Save them all to a file with `s`
|
|
29
|
+
7. Tunnel - open, show link, close, new link
|
|
30
|
+
8. Server - start / stop / restart / logs / GPU report
|
|
31
|
+
9. Settings - context length, KV cache type, flash attention, keep-alive, models folder (saved)
|
|
32
|
+
10. Benchmark the active model
|
|
33
|
+
|
|
34
|
+
### Disk-space check
|
|
35
|
+
Before every download the tool compares the model size (+5% +1 GB) with free space in the models folder. If it won't fit you can delete models on the spot, pick a smaller quant, or continue anyway. The models folder is chosen automatically (the disk with the most free space, e.g. `/kaggle/temp`) and shown in the header; change it in Settings.
|
|
36
|
+
|
|
37
|
+
### Tool notes
|
|
38
|
+
- Cursor sends requests from its own servers, so it needs the public tunnel link (that's what this tool gives you).
|
|
39
|
+
- Claude Code uses Ollama's Anthropic-compatible API (`ANTHROPIC_BASE_URL`); it needs a tool-capable model and a large context.
|
|
40
|
+
- Antigravity IDE has no custom-model setting, so use the Continue/Cline extension inside it.
|
|
31
41
|
|
|
32
42
|
## Non-interactive
|
|
33
43
|
```
|
|
34
44
|
notebook-llm gpus # hardware + recommended table
|
|
35
45
|
notebook-llm search coder
|
|
36
46
|
notebook-llm check llama3.1:70b --ctx 8192
|
|
47
|
+
notebook-llm quants qwen2.5-coder:7b # compare quantizations
|
|
48
|
+
notebook-llm connect cursor # or: continue cline aider openwebui claude-code antigravity python (none = all)
|
|
37
49
|
notebook-llm quickstart --model auto
|
|
38
50
|
notebook-llm status | stop
|
|
39
51
|
```
|
|
@@ -44,6 +56,7 @@ from notebook_llm import NotebookLLM
|
|
|
44
56
|
llm = NotebookLLM()
|
|
45
57
|
url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
|
|
46
58
|
llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
|
|
59
|
+
print(llm.connect("cursor")) # settings text for a tool; llm.connect() = all tools
|
|
47
60
|
```
|
|
48
61
|
|
|
49
62
|
## How the estimates work
|
|
@@ -5,8 +5,10 @@ from .library import LibModel, Candidate, search, make_candidate
|
|
|
5
5
|
from .manager import NotebookLLM, Settings
|
|
6
6
|
from .cli import run, main
|
|
7
7
|
from .env import is_kaggle
|
|
8
|
+
from .disk import DiskSpaceError
|
|
9
|
+
from .integrations import render as render_config, TOOLS
|
|
8
10
|
|
|
9
11
|
__all__ = ["NotebookLLM", "Settings", "run", "main", "GPU", "GPUInfo", "Hardware",
|
|
10
12
|
"detect_gpus", "detect_hardware", "Estimate", "estimate", "LibModel",
|
|
11
|
-
"Candidate", "search", "make_candidate", "is_kaggle"]
|
|
12
|
-
__version__ = "0.
|
|
13
|
+
"Candidate", "search", "make_candidate", "is_kaggle", "DiskSpaceError", "render_config", "TOOLS"]
|
|
14
|
+
__version__ = "0.2.0"
|
|
@@ -1,11 +1,13 @@
|
|
|
1
1
|
"""Interactive CLI. Run `notebook-llm` (terminal) or `notebook_llm.run()` (notebook cell)."""
|
|
2
2
|
import argparse
|
|
3
|
-
import
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
4
5
|
|
|
5
|
-
from . import library, ui
|
|
6
|
+
from . import disk, library, ui
|
|
6
7
|
from .env import is_kaggle
|
|
8
|
+
from .integrations import TOOLS
|
|
7
9
|
from .manager import NotebookLLM
|
|
8
|
-
from .ui import
|
|
10
|
+
from .ui import ask, c, confirm, pick, table, title
|
|
9
11
|
|
|
10
12
|
|
|
11
13
|
# ---------- rendering helpers ---------------------------------------
|
|
@@ -18,7 +20,6 @@ def speed_cell(est) -> str:
|
|
|
18
20
|
|
|
19
21
|
|
|
20
22
|
def show_candidates(cands, llm):
|
|
21
|
-
ctx = cands[0].est.context if cands else "-"
|
|
22
23
|
ui.info(f"Hardware: {llm.hw.summary()}")
|
|
23
24
|
ui.info("NEEDS = weights + KV cache (auto context) + overhead. tok/s is an estimate (+-30%).")
|
|
24
25
|
rows = []
|
|
@@ -49,6 +50,7 @@ def header(llm: NotebookLLM):
|
|
|
49
50
|
print(f" VRAM in use: {llm.hw.gpu.used_vram_gb} / {llm.hw.gpu.total_vram_gb} GB")
|
|
50
51
|
elif is_kaggle():
|
|
51
52
|
ui.warn("No GPU. In Kaggle: Settings > Accelerator > GPU T4 x2, then restart.")
|
|
53
|
+
print(f" Disk : {disk.free_gb(llm.models_dir):.0f} GB free ({llm.models_dir})")
|
|
52
54
|
up = llm.server.is_running()
|
|
53
55
|
print(f" Server : {c('running', 'green') if up else c('stopped', 'red')}")
|
|
54
56
|
if up:
|
|
@@ -59,7 +61,7 @@ def header(llm: NotebookLLM):
|
|
|
59
61
|
except Exception:
|
|
60
62
|
pass
|
|
61
63
|
print(f" Model : {llm.settings.model or '(none selected)'}")
|
|
62
|
-
print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else
|
|
64
|
+
print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else 'closed'}")
|
|
63
65
|
|
|
64
66
|
|
|
65
67
|
# ---------- flows ----------------------------------------------------
|
|
@@ -75,6 +77,24 @@ def ensure_server(llm, model=None) -> bool:
|
|
|
75
77
|
return False
|
|
76
78
|
|
|
77
79
|
|
|
80
|
+
def ensure_disk(llm, cd) -> bool:
|
|
81
|
+
"""Check free space before a download. Returns True if OK to proceed."""
|
|
82
|
+
chk = llm.space_for(cd.size_gb)
|
|
83
|
+
if chk.ok:
|
|
84
|
+
ui.info(f"Disk check: {chk.free_gb:.0f} GB free, need ~{chk.needed_gb:.0f} GB. OK.")
|
|
85
|
+
return True
|
|
86
|
+
ui.err(f"Not enough disk space in {chk.path}: {chk.free_gb:.1f} GB free, need ~{chk.needed_gb:.1f} GB.")
|
|
87
|
+
print(" Ways to fix it: delete models you don't need, pick a smaller size or quantization "
|
|
88
|
+
"(type qN in the list), or change the models folder in Settings.")
|
|
89
|
+
if llm.installed() and confirm("Open 'Installed models' to free space now?"):
|
|
90
|
+
flow_installed(llm)
|
|
91
|
+
chk = llm.space_for(cd.size_gb)
|
|
92
|
+
if chk.ok:
|
|
93
|
+
ui.ok(f"Now {chk.free_gb:.0f} GB free. Continuing.")
|
|
94
|
+
return True
|
|
95
|
+
return confirm("Try the download anyway?", False)
|
|
96
|
+
|
|
97
|
+
|
|
78
98
|
def after_load(llm):
|
|
79
99
|
if not llm.url and confirm("Open a public tunnel now?"):
|
|
80
100
|
flow_open_tunnel(llm)
|
|
@@ -94,8 +114,10 @@ def model_actions(llm, cd):
|
|
|
94
114
|
return
|
|
95
115
|
try:
|
|
96
116
|
if not llm.client.has_model(cd.ref):
|
|
117
|
+
if not ensure_disk(llm, cd):
|
|
118
|
+
return
|
|
97
119
|
print(f" Pulling {cd.ref} ({cd.size_gb:.1f} GB)...")
|
|
98
|
-
llm.pull(cd.ref)
|
|
120
|
+
llm.pull(cd.ref, cd.size_gb, force=True) # already checked above
|
|
99
121
|
if ch == "1":
|
|
100
122
|
print(" Loading into VRAM...")
|
|
101
123
|
llm.load(cd.ref)
|
|
@@ -103,6 +125,8 @@ def model_actions(llm, cd):
|
|
|
103
125
|
after_load(llm)
|
|
104
126
|
except Exception as e:
|
|
105
127
|
ui.err(str(e))
|
|
128
|
+
if "space" in str(e).lower():
|
|
129
|
+
ui.warn("The disk filled up during the download. Delete a model or pick a smaller one.")
|
|
106
130
|
|
|
107
131
|
|
|
108
132
|
def _result_rows(res, offset=0):
|
|
@@ -165,30 +189,99 @@ def flow_search(llm):
|
|
|
165
189
|
return
|
|
166
190
|
|
|
167
191
|
|
|
192
|
+
def _select_from_candidates(llm, cands, extra_hint=""):
|
|
193
|
+
"""Number = select, qN = quant picker for row N. Returns after handling."""
|
|
194
|
+
v = ask(f"Choose a number, qN to compare quantizations of row N{extra_hint}. Enter = back")
|
|
195
|
+
if not v or v in ("b", "q"):
|
|
196
|
+
return "back"
|
|
197
|
+
m = re.fullmatch(r"q\s*(\d+)", v.lower())
|
|
198
|
+
if m and 1 <= int(m.group(1)) <= len(cands):
|
|
199
|
+
flow_quants(llm, cands[int(m.group(1)) - 1].ref)
|
|
200
|
+
return "again"
|
|
201
|
+
if v.isdigit() and 1 <= int(v) <= len(cands):
|
|
202
|
+
model_actions(llm, cands[int(v) - 1])
|
|
203
|
+
return "back"
|
|
204
|
+
return v # custom tag
|
|
205
|
+
|
|
206
|
+
|
|
168
207
|
def flow_pick_tag(llm, m):
|
|
169
208
|
while True:
|
|
170
209
|
print(f"\n Checking sizes for {m.name} (registry lookup)...")
|
|
171
210
|
cands = library.candidates_for(m, llm.hw, llm.settings.context, llm.settings.kv_cache)
|
|
172
211
|
title(m.name)
|
|
173
212
|
show_candidates(cands, llm)
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
if not v or v in ("b", "q"):
|
|
213
|
+
r = _select_from_candidates(llm, cands, " or type a custom tag")
|
|
214
|
+
if r == "back":
|
|
177
215
|
return
|
|
178
|
-
if
|
|
179
|
-
|
|
180
|
-
model_actions(llm, library.make_candidate(f"{m.name}:{
|
|
216
|
+
if r == "again":
|
|
217
|
+
continue
|
|
218
|
+
model_actions(llm, library.make_candidate(f"{m.name}:{r}", llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
181
219
|
return
|
|
182
220
|
|
|
183
221
|
|
|
184
222
|
def flow_recommended(llm):
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
223
|
+
while True:
|
|
224
|
+
cands = [x for x in llm.recommended() if x.est.verdict != "no"] or llm.recommended()
|
|
225
|
+
title("Recommended for your hardware")
|
|
226
|
+
show_candidates(cands, llm)
|
|
227
|
+
r = _select_from_candidates(llm, cands)
|
|
228
|
+
if r == "again":
|
|
229
|
+
continue
|
|
230
|
+
if r not in ("back",):
|
|
231
|
+
continue
|
|
232
|
+
return
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def flow_quants(llm, ref=None):
|
|
236
|
+
"""Compare every quantization (q4/q5/q8/fp16...) of one model size: fit, speed, quality."""
|
|
237
|
+
if not ref:
|
|
238
|
+
ref = ask("Model and size, e.g. qwen2.5-coder:7b (or just a name like qwen2.5-coder)")
|
|
239
|
+
if not ref or ref in ("q", "b"):
|
|
240
|
+
return
|
|
241
|
+
name, _, tag = ref.partition(":")
|
|
242
|
+
if not library.size_of(tag):
|
|
243
|
+
print(f" Looking up sizes for {name}...")
|
|
244
|
+
try:
|
|
245
|
+
lm = next((x for x in library.search(name) if x.name == name), None)
|
|
246
|
+
except library.LibraryError as e:
|
|
247
|
+
ui.err(str(e))
|
|
248
|
+
return
|
|
249
|
+
sizes = [s for s in (lm.sizes if lm else []) if "cloud" not in s]
|
|
250
|
+
if not sizes:
|
|
251
|
+
ui.warn(f"No sized tags found for '{name}'. Use the form name:size, e.g. qwen2.5-coder:7b")
|
|
252
|
+
return
|
|
253
|
+
for i, s in enumerate(sizes, 1):
|
|
254
|
+
print(f" {i}) {name}:{s}")
|
|
255
|
+
i = pick("Which size", len(sizes))
|
|
256
|
+
if i is None:
|
|
257
|
+
return
|
|
258
|
+
tag = sizes[i]
|
|
259
|
+
size_tag = library.size_of(tag)
|
|
260
|
+
print(f" Fetching quantizations of {name}:{size_tag} (registry lookups)...")
|
|
261
|
+
try:
|
|
262
|
+
rows = library.quant_candidates(name, size_tag, llm.hw, llm.settings.context, llm.settings.kv_cache)
|
|
263
|
+
except library.LibraryError as e:
|
|
264
|
+
ui.err(str(e))
|
|
265
|
+
return
|
|
266
|
+
best = library.suggest_quant(rows)
|
|
267
|
+
title(f"Quantizations of {name}:{size_tag}")
|
|
268
|
+
ui.info(f"Hardware: {llm.hw.summary()}")
|
|
269
|
+
out = []
|
|
270
|
+
for i, r in enumerate(rows, 1):
|
|
271
|
+
cd = r.cand
|
|
272
|
+
mark = c("<- suggested", "green") if best is r else ""
|
|
273
|
+
out.append((i, r.tag, r.variant, r.quant or "-", library.quality_of(r.quant),
|
|
274
|
+
f"{cd.size_gb:.1f} GB" + ("" if cd.exact else "~"), f"{cd.est.needed_gb:.1f} GB",
|
|
275
|
+
fit_cell(cd.est), speed_cell(cd.est), mark))
|
|
276
|
+
table(["#", "TAG", "VARIANT", "QUANT", "QUALITY", "SIZE", "NEEDS", "FIT", "TOK/S", ""], out, right_cols=(0, 5, 6, 8))
|
|
277
|
+
if best:
|
|
278
|
+
ui.info(f"Suggested = best quality that fits your GPU and stays at 15+ tok/s: {name}:{best.tag}")
|
|
279
|
+
else:
|
|
280
|
+
ui.warn("Nothing fits comfortably. Try a smaller size (low-bit quants lose quality).")
|
|
281
|
+
ui.info("Lower quant = smaller and faster but less accurate. q4_K_M is the usual sweet spot; q8 is near-lossless.")
|
|
282
|
+
i = pick("Select which", len(rows))
|
|
190
283
|
if i is not None:
|
|
191
|
-
model_actions(llm,
|
|
284
|
+
model_actions(llm, rows[i].cand)
|
|
192
285
|
|
|
193
286
|
|
|
194
287
|
def flow_installed(llm):
|
|
@@ -208,6 +301,7 @@ def flow_installed(llm):
|
|
|
208
301
|
rows.append((i, m["name"], f"{cd.size_gb:.1f} GB", fit_cell(cd.est), speed_cell(cd.est), mark, active))
|
|
209
302
|
title("Installed models")
|
|
210
303
|
table(["#", "MODEL", "SIZE", "FIT", "TOK/S", "STATE", ""], rows, right_cols=(0, 2, 4))
|
|
304
|
+
ui.info(f"Disk free: {disk.free_gb(llm.models_dir):.0f} GB in {llm.models_dir}")
|
|
211
305
|
i = pick("Manage which", len(inst))
|
|
212
306
|
if i is None:
|
|
213
307
|
return
|
|
@@ -226,7 +320,7 @@ def flow_installed(llm):
|
|
|
226
320
|
flow_benchmark(llm, name)
|
|
227
321
|
elif ch == "4" and confirm(f"Delete {name}?", False):
|
|
228
322
|
llm.delete(name)
|
|
229
|
-
ui.ok("Deleted.")
|
|
323
|
+
ui.ok(f"Deleted. Disk free now: {disk.free_gb(llm.models_dir):.0f} GB")
|
|
230
324
|
except Exception as e:
|
|
231
325
|
ui.err(str(e))
|
|
232
326
|
|
|
@@ -253,28 +347,64 @@ def flow_benchmark(llm, name=None):
|
|
|
253
347
|
ui.warn("Model is partly on CPU. Try a smaller model/quant or lower context in Settings.")
|
|
254
348
|
|
|
255
349
|
|
|
256
|
-
def
|
|
350
|
+
def open_tunnel_quiet(llm):
|
|
257
351
|
if not ensure_server(llm):
|
|
258
|
-
return
|
|
352
|
+
return None
|
|
259
353
|
try:
|
|
260
|
-
|
|
354
|
+
return llm.open_tunnel()
|
|
261
355
|
except Exception as e:
|
|
262
356
|
ui.err(str(e))
|
|
357
|
+
return None
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def flow_open_tunnel(llm):
|
|
361
|
+
url = open_tunnel_quiet(llm)
|
|
362
|
+
if not url:
|
|
263
363
|
return
|
|
264
364
|
print("\n " + c(url, "bold", "green"))
|
|
265
365
|
ui.warn("Anyone with this link can use your GPU. It changes every time you reopen the tunnel.")
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
366
|
+
ui.info("Next: menu 'Connect your tools' prints settings for Cline, Aider, Cursor, Claude Code and more.")
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def flow_connect(llm):
|
|
370
|
+
"""Print ready-to-paste settings for coding tools."""
|
|
371
|
+
if not llm.settings.model:
|
|
372
|
+
ui.warn("Load a model first (Quick start / Recommended / Search).")
|
|
373
|
+
return
|
|
374
|
+
if not llm.url:
|
|
375
|
+
if not confirm("These tools need the public tunnel link. Open it now?"):
|
|
376
|
+
return
|
|
377
|
+
if not open_tunnel_quiet(llm):
|
|
378
|
+
return
|
|
379
|
+
while True:
|
|
380
|
+
title(f"Connect your tools ({llm.settings.model})")
|
|
381
|
+
print(f" Link: {c(llm.url, 'bold', 'green')}\n")
|
|
382
|
+
for i, (_, label, _) in enumerate(TOOLS, 1):
|
|
383
|
+
print(f" {c(i, 'cyan')}) {label}")
|
|
384
|
+
print(f" {c('a', 'cyan')}) Show all {c('s', 'cyan')}) Save all to a file")
|
|
385
|
+
v = ask("Choose (Enter = back)").lower()
|
|
386
|
+
if not v or v in ("b", "q"):
|
|
387
|
+
return
|
|
388
|
+
if v == "a":
|
|
389
|
+
print("\n" + llm.connect())
|
|
390
|
+
elif v == "s":
|
|
391
|
+
base = "/kaggle/working" if os.path.isdir("/kaggle/working") else os.getcwd()
|
|
392
|
+
path = os.path.join(base, "notebook_llm_connect.txt")
|
|
393
|
+
with open(path, "w") as f:
|
|
394
|
+
f.write(llm.connect())
|
|
395
|
+
ui.ok(f"Saved to {path}")
|
|
396
|
+
elif v.isdigit() and 1 <= int(v) <= len(TOOLS):
|
|
397
|
+
print("\n" + llm.connect(TOOLS[int(v) - 1][0]))
|
|
398
|
+
print()
|
|
269
399
|
|
|
270
400
|
|
|
271
401
|
def flow_tunnel(llm):
|
|
272
402
|
if not llm.url:
|
|
273
403
|
return flow_open_tunnel(llm)
|
|
274
|
-
print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1)
|
|
404
|
+
print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1) Connect your tools (Cline, Aider, Cursor...)\n 2) Close tunnel\n 3) New tunnel (new link)\n Enter) back")
|
|
275
405
|
ch = ask("Choose")
|
|
276
406
|
if ch == "1":
|
|
277
|
-
|
|
407
|
+
flow_connect(llm)
|
|
278
408
|
elif ch == "2":
|
|
279
409
|
llm.close_tunnel()
|
|
280
410
|
ui.ok("Tunnel closed.")
|
|
@@ -305,7 +435,8 @@ def flow_server(llm):
|
|
|
305
435
|
def flow_settings(llm):
|
|
306
436
|
s = llm.settings
|
|
307
437
|
print(f"\n 1) Context length : {s.context or 'auto'}\n 2) KV cache type : {s.kv_cache} (f16 | q8_0 | q4_0)\n"
|
|
308
|
-
f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n
|
|
438
|
+
f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n"
|
|
439
|
+
f" 5) Models folder : {llm.models_dir} ({disk.free_gb(llm.models_dir):.0f} GB free)\n Enter) back")
|
|
309
440
|
ch = ask("Change which")
|
|
310
441
|
if ch == "1":
|
|
311
442
|
v = ask("Context tokens (number, or 'auto')", "auto")
|
|
@@ -317,13 +448,20 @@ def flow_settings(llm):
|
|
|
317
448
|
s.flash_attention = confirm("Enable flash attention?", s.flash_attention)
|
|
318
449
|
elif ch == "4":
|
|
319
450
|
s.keep_alive = ask("Keep-alive", s.keep_alive)
|
|
451
|
+
elif ch == "5":
|
|
452
|
+
ui.warn("Models already downloaded stay in the old folder and will not be visible from the new one.")
|
|
453
|
+
for i, d in enumerate(disk.candidate_dirs(), 1):
|
|
454
|
+
print(f" {i}) {d} ({disk.free_gb(d):.0f} GB free)")
|
|
455
|
+
v = ask("Folder path or number", llm.models_dir)
|
|
456
|
+
dirs = disk.candidate_dirs()
|
|
457
|
+
s.models_dir = dirs[int(v) - 1] if v.isdigit() and 1 <= int(v) <= len(dirs) else v
|
|
320
458
|
else:
|
|
321
459
|
return
|
|
322
460
|
s.save()
|
|
323
461
|
ui.ok("Saved.")
|
|
324
462
|
if llm.server.is_running() and confirm("Restart server to apply now?"):
|
|
325
463
|
llm.restart_server()
|
|
326
|
-
if s.model:
|
|
464
|
+
if s.model and llm.client.has_model(s.model):
|
|
327
465
|
llm.load(s.model)
|
|
328
466
|
|
|
329
467
|
|
|
@@ -338,18 +476,21 @@ def flow_quickstart(llm):
|
|
|
338
476
|
ui.err(str(e))
|
|
339
477
|
return
|
|
340
478
|
ui.ok(f"Ready: {llm.settings.model}")
|
|
341
|
-
print("\n " + c(url, "bold", "green"))
|
|
342
|
-
|
|
479
|
+
print("\n " + c(url or "(no tunnel)", "bold", "green"))
|
|
480
|
+
if url and confirm("Show setup for your coding tools now?"):
|
|
481
|
+
flow_connect(llm)
|
|
343
482
|
|
|
344
483
|
|
|
345
484
|
MENU = [
|
|
346
485
|
("Quick start (auto: install, model, tunnel)", flow_quickstart),
|
|
347
486
|
("Search Ollama library", flow_search),
|
|
348
487
|
("Recommended models for my GPU", flow_recommended),
|
|
488
|
+
("Quant picker (compare q4 / q5 / q8 / fp16)", flow_quants),
|
|
349
489
|
("Installed models (load / benchmark / delete)", flow_installed),
|
|
490
|
+
("Connect your tools (Cline, Aider, Cursor, Claude Code...)", flow_connect),
|
|
350
491
|
("Tunnel (open / get link / close)", flow_tunnel),
|
|
351
492
|
("Server (start / stop / logs)", flow_server),
|
|
352
|
-
("Settings (context, KV cache, keep-alive)", flow_settings),
|
|
493
|
+
("Settings (context, KV cache, keep-alive, folder)", flow_settings),
|
|
353
494
|
("Benchmark active model", flow_benchmark),
|
|
354
495
|
]
|
|
355
496
|
|
|
@@ -385,6 +526,10 @@ def main(argv=None) -> None:
|
|
|
385
526
|
k = sub.add_parser("check", help="will MODEL fit? estimated tok/s")
|
|
386
527
|
k.add_argument("model")
|
|
387
528
|
k.add_argument("--ctx", type=int)
|
|
529
|
+
qn = sub.add_parser("quants", help="compare quantizations, e.g. qwen2.5-coder:7b")
|
|
530
|
+
qn.add_argument("model")
|
|
531
|
+
cn = sub.add_parser("connect", help="print settings for a tool (continue, cline, aider, openwebui, claude-code, cursor, antigravity, python)")
|
|
532
|
+
cn.add_argument("tool", nargs="?")
|
|
388
533
|
q = sub.add_parser("quickstart", help="install, start, pull best model, open tunnel")
|
|
389
534
|
q.add_argument("--model", default="auto")
|
|
390
535
|
q.add_argument("--no-tunnel", action="store_true")
|
|
@@ -392,7 +537,6 @@ def main(argv=None) -> None:
|
|
|
392
537
|
sub.add_parser("stop", help="stop tunnel and server")
|
|
393
538
|
a = p.parse_args(argv)
|
|
394
539
|
if a.no_color:
|
|
395
|
-
import os
|
|
396
540
|
os.environ["NO_COLOR"] = "1"
|
|
397
541
|
|
|
398
542
|
if a.cmd is None:
|
|
@@ -403,14 +547,25 @@ def main(argv=None) -> None:
|
|
|
403
547
|
show_candidates(llm.recommended(), llm)
|
|
404
548
|
elif a.cmd == "search":
|
|
405
549
|
for m in library.search(a.query)[:20]:
|
|
406
|
-
print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}")
|
|
550
|
+
print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}{' [cloud]' if m.cloud else ''}")
|
|
407
551
|
elif a.cmd == "check":
|
|
408
|
-
|
|
409
|
-
|
|
552
|
+
show_estimate_detail(library.make_candidate(a.model, llm.hw, a.ctx, llm.settings.kv_cache))
|
|
553
|
+
elif a.cmd == "quants":
|
|
554
|
+
name, _, tag = a.model.partition(":")
|
|
555
|
+
rows = library.quant_candidates(name, library.size_of(tag) or tag, llm.hw, llm.settings.context, llm.settings.kv_cache)
|
|
556
|
+
best = library.suggest_quant(rows)
|
|
557
|
+
out = [(r.tag, library.quality_of(r.quant), f"{r.cand.size_gb:.1f} GB", f"{r.cand.est.needed_gb:.1f} GB",
|
|
558
|
+
fit_cell(r.cand.est), speed_cell(r.cand.est), "<- suggested" if r is best else "") for r in rows]
|
|
559
|
+
table(["TAG", "QUALITY", "SIZE", "NEEDS", "FIT", "TOK/S", ""], out, right_cols=(2, 3, 5))
|
|
560
|
+
elif a.cmd == "connect":
|
|
561
|
+
try:
|
|
562
|
+
print(llm.connect(a.tool))
|
|
563
|
+
except KeyError as e:
|
|
564
|
+
raise SystemExit(str(e).strip("\"'"))
|
|
410
565
|
elif a.cmd == "quickstart":
|
|
411
566
|
url = llm.quickstart(a.model, tunnel=not a.no_tunnel)
|
|
412
567
|
print(url or "")
|
|
413
|
-
print(llm.
|
|
568
|
+
print(llm.connect())
|
|
414
569
|
elif a.cmd == "status":
|
|
415
570
|
print(f"server={llm.server.is_running()} model={llm.settings.model} tunnel={llm.url}")
|
|
416
571
|
elif a.cmd == "stop":
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Disk-space checks for model downloads."""
|
|
2
|
+
import os
|
|
3
|
+
import shutil
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class DiskSpaceError(RuntimeError):
|
|
8
|
+
pass
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class SpaceCheck:
|
|
13
|
+
ok: bool
|
|
14
|
+
free_gb: float
|
|
15
|
+
needed_gb: float
|
|
16
|
+
path: str
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def default_models_dir() -> str:
|
|
20
|
+
return os.environ.get("OLLAMA_MODELS") or os.path.expanduser("~/.ollama/models")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _existing_parent(path: str) -> str:
|
|
24
|
+
p = os.path.abspath(path)
|
|
25
|
+
while not os.path.exists(p):
|
|
26
|
+
parent = os.path.dirname(p)
|
|
27
|
+
if parent == p:
|
|
28
|
+
break
|
|
29
|
+
p = parent
|
|
30
|
+
return p
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def free_gb(path: str) -> float:
|
|
34
|
+
return shutil.disk_usage(_existing_parent(path)).free / 1024 ** 3
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def dir_size_gb(path: str) -> float:
|
|
38
|
+
total = 0
|
|
39
|
+
for root, _, files in os.walk(path):
|
|
40
|
+
for f in files:
|
|
41
|
+
try:
|
|
42
|
+
total += os.path.getsize(os.path.join(root, f))
|
|
43
|
+
except OSError:
|
|
44
|
+
pass
|
|
45
|
+
return total / 1024 ** 3
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def has_models(path: str) -> bool:
|
|
49
|
+
blobs = os.path.join(path, "blobs")
|
|
50
|
+
return os.path.isdir(blobs) and any(True for _ in os.scandir(blobs))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def candidate_dirs():
|
|
54
|
+
out = [default_models_dir()]
|
|
55
|
+
for base in ("/kaggle/temp", "/tmp"):
|
|
56
|
+
if os.path.isdir(base):
|
|
57
|
+
out.append(os.path.join(base, "ollama_models"))
|
|
58
|
+
return out
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def pick_models_dir() -> str:
|
|
62
|
+
"""Keep the default if it already holds models; otherwise use the disk with the most free space."""
|
|
63
|
+
default = default_models_dir()
|
|
64
|
+
if has_models(default):
|
|
65
|
+
return default
|
|
66
|
+
best = max(candidate_dirs(), key=free_gb)
|
|
67
|
+
return best if free_gb(best) > free_gb(default) + 2 else default
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def check_space(size_gb: float, path: str) -> SpaceCheck:
|
|
71
|
+
needed = size_gb * 1.05 + 1.0 # download + 5% + 1 GB headroom
|
|
72
|
+
free = free_gb(path)
|
|
73
|
+
return SpaceCheck(free >= needed, free, needed, path)
|