notebook-llm-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- notebook_llm/__init__.py +12 -0
- notebook_llm/__main__.py +3 -0
- notebook_llm/cli.py +422 -0
- notebook_llm/client.py +83 -0
- notebook_llm/config.py +32 -0
- notebook_llm/env.py +9 -0
- notebook_llm/estimate.py +128 -0
- notebook_llm/estimate_data.py +38 -0
- notebook_llm/gpu.py +127 -0
- notebook_llm/installer.py +67 -0
- notebook_llm/library.py +212 -0
- notebook_llm/manager.py +169 -0
- notebook_llm/server.py +101 -0
- notebook_llm/tunnel.py +63 -0
- notebook_llm/ui.py +71 -0
- notebook_llm_cli-0.1.0.dist-info/METADATA +79 -0
- notebook_llm_cli-0.1.0.dist-info/RECORD +21 -0
- notebook_llm_cli-0.1.0.dist-info/WHEEL +5 -0
- notebook_llm_cli-0.1.0.dist-info/entry_points.txt +3 -0
- notebook_llm_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
- notebook_llm_cli-0.1.0.dist-info/top_level.txt +1 -0
notebook_llm/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""notebook_llm: manage Ollama LLMs on notebook GPUs (Kaggle/Colab) from one interactive CLI."""
|
|
2
|
+
from .gpu import GPU, GPUInfo, Hardware, detect_gpus, detect_hardware
|
|
3
|
+
from .estimate import Estimate, estimate
|
|
4
|
+
from .library import LibModel, Candidate, search, make_candidate
|
|
5
|
+
from .manager import NotebookLLM, Settings
|
|
6
|
+
from .cli import run, main
|
|
7
|
+
from .env import is_kaggle
|
|
8
|
+
|
|
9
|
+
__all__ = ["NotebookLLM", "Settings", "run", "main", "GPU", "GPUInfo", "Hardware",
|
|
10
|
+
"detect_gpus", "detect_hardware", "Estimate", "estimate", "LibModel",
|
|
11
|
+
"Candidate", "search", "make_candidate", "is_kaggle"]
|
|
12
|
+
__version__ = "0.1.0"
|
notebook_llm/__main__.py
ADDED
notebook_llm/cli.py
ADDED
|
@@ -0,0 +1,422 @@
|
|
|
1
|
+
"""Interactive CLI. Run `notebook-llm` (terminal) or `notebook_llm.run()` (notebook cell)."""
|
|
2
|
+
import argparse
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
from . import library, ui
|
|
6
|
+
from .env import is_kaggle
|
|
7
|
+
from .manager import NotebookLLM
|
|
8
|
+
from .ui import c, ask, confirm, pick, table, title
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# ---------- rendering helpers ---------------------------------------
|
|
12
|
+
def fit_cell(est) -> str:
|
|
13
|
+
return c(est.label, ui.VERDICT_STYLE[est.verdict])
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def speed_cell(est) -> str:
|
|
17
|
+
return f"~{est.tok_s:.0f}" if est.tok_s else "-"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def show_candidates(cands, llm):
|
|
21
|
+
ctx = cands[0].est.context if cands else "-"
|
|
22
|
+
ui.info(f"Hardware: {llm.hw.summary()}")
|
|
23
|
+
ui.info("NEEDS = weights + KV cache (auto context) + overhead. tok/s is an estimate (+-30%).")
|
|
24
|
+
rows = []
|
|
25
|
+
for i, cd in enumerate(cands, 1):
|
|
26
|
+
size = f"{cd.size_gb:.1f} GB" + ("" if cd.exact else "~")
|
|
27
|
+
moe = " MoE" if cd.active_b and cd.params_b and cd.active_b < cd.params_b * 0.8 else ""
|
|
28
|
+
rows.append((i, cd.ref + c(moe, "gray"), size, f"{cd.est.needed_gb:.1f} GB", cd.est.context,
|
|
29
|
+
fit_cell(cd.est), speed_cell(cd.est)))
|
|
30
|
+
table(["#", "MODEL", "SIZE", "NEEDS", "CTX", "FIT", "TOK/S"], rows, right_cols=(0, 2, 3, 4, 6))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def show_estimate_detail(cd):
|
|
34
|
+
e = cd.est
|
|
35
|
+
print(f"\n {c(cd.ref, 'bold')} ({'exact' if cd.exact else 'estimated'} size {cd.size_gb:.1f} GB)")
|
|
36
|
+
print(f" Weights {e.weights_gb:.1f} GB + KV cache {e.kv_gb:.1f} GB (ctx {e.context}) + overhead {e.overhead_gb:.1f} GB"
|
|
37
|
+
f" = {c(f'{e.needed_gb:.1f} GB', 'bold')}")
|
|
38
|
+
print(f" Verdict: {fit_cell(e)} Speed: {c(speed_cell(e) + ' tok/s', 'bold')} "
|
|
39
|
+
f"On GPU: {e.gpu_fraction * 100:.0f}% of weights")
|
|
40
|
+
if e.note:
|
|
41
|
+
ui.warn(e.note)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def header(llm: NotebookLLM):
|
|
45
|
+
llm.refresh_hardware()
|
|
46
|
+
title("notebook_llm")
|
|
47
|
+
print(f" {llm.hw.summary()}")
|
|
48
|
+
if llm.hw.gpu.available:
|
|
49
|
+
print(f" VRAM in use: {llm.hw.gpu.used_vram_gb} / {llm.hw.gpu.total_vram_gb} GB")
|
|
50
|
+
elif is_kaggle():
|
|
51
|
+
ui.warn("No GPU. In Kaggle: Settings > Accelerator > GPU T4 x2, then restart.")
|
|
52
|
+
up = llm.server.is_running()
|
|
53
|
+
print(f" Server : {c('running', 'green') if up else c('stopped', 'red')}")
|
|
54
|
+
if up:
|
|
55
|
+
try:
|
|
56
|
+
for m in llm.loaded():
|
|
57
|
+
pct = 100 * m.get("size_vram", 0) / m["size"] if m.get("size") else 0
|
|
58
|
+
print(f" Loaded : {c(m['name'], 'bold')} ({m.get('size', 0) / 1024 ** 3:.1f} GB, {pct:.0f}% on GPU)")
|
|
59
|
+
except Exception:
|
|
60
|
+
pass
|
|
61
|
+
print(f" Model : {llm.settings.model or '(none selected)'}")
|
|
62
|
+
print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else c('closed', 'gray')}")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ---------- flows ----------------------------------------------------
|
|
66
|
+
def ensure_server(llm, model=None) -> bool:
|
|
67
|
+
if llm.server.is_running():
|
|
68
|
+
return True
|
|
69
|
+
print(" Installing/starting Ollama...")
|
|
70
|
+
try:
|
|
71
|
+
llm.start_server(model)
|
|
72
|
+
return True
|
|
73
|
+
except Exception as e:
|
|
74
|
+
ui.err(str(e))
|
|
75
|
+
return False
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def after_load(llm):
|
|
79
|
+
if not llm.url and confirm("Open a public tunnel now?"):
|
|
80
|
+
flow_open_tunnel(llm)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def model_actions(llm, cd):
|
|
84
|
+
show_estimate_detail(cd)
|
|
85
|
+
if cd.est.verdict == "no":
|
|
86
|
+
ui.warn("This model will not run on your hardware.")
|
|
87
|
+
if not confirm("Pull anyway?", False):
|
|
88
|
+
return
|
|
89
|
+
print("\n 1) Pull + load now (make it the active model)\n 2) Pull only\n Enter) back")
|
|
90
|
+
ch = ask("Choose")
|
|
91
|
+
if ch not in ("1", "2"):
|
|
92
|
+
return
|
|
93
|
+
if not ensure_server(llm, cd.ref):
|
|
94
|
+
return
|
|
95
|
+
try:
|
|
96
|
+
if not llm.client.has_model(cd.ref):
|
|
97
|
+
print(f" Pulling {cd.ref} ({cd.size_gb:.1f} GB)...")
|
|
98
|
+
llm.pull(cd.ref)
|
|
99
|
+
if ch == "1":
|
|
100
|
+
print(" Loading into VRAM...")
|
|
101
|
+
llm.load(cd.ref)
|
|
102
|
+
ui.ok(f"{cd.ref} is loaded and will stay in memory.")
|
|
103
|
+
after_load(llm)
|
|
104
|
+
except Exception as e:
|
|
105
|
+
ui.err(str(e))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _result_rows(res, offset=0):
|
|
109
|
+
rows = []
|
|
110
|
+
for i, m in enumerate(res, 1 + offset):
|
|
111
|
+
sizes = ", ".join(m.sizes[:6]) + (" ..." if len(m.sizes) > 6 else "")
|
|
112
|
+
if m.cloud:
|
|
113
|
+
sizes = (sizes + " " if sizes else "") + c("cloud", "yellow")
|
|
114
|
+
desc = m.description.split(". ")[0]
|
|
115
|
+
rows.append((i, c(m.name, "bold"), m.pulls, sizes or "-", ",".join(m.capabilities[:3]),
|
|
116
|
+
(desc[:46] + "...") if len(desc) > 46 else desc))
|
|
117
|
+
return rows
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def flow_search(llm):
|
|
121
|
+
print(" Search ollama.com/search. Examples: coder | /vision | qwen /tools | /newest")
|
|
122
|
+
print(" Filters: /tools /vision /thinking /embedding /cloud /newest /popular. Or paste a model like qwen3:8b")
|
|
123
|
+
q = ask("Search (blank = all models)")
|
|
124
|
+
if q in ("q", "b"):
|
|
125
|
+
return
|
|
126
|
+
if q.startswith("http") and "/library/" in q:
|
|
127
|
+
q = q.split("/library/", 1)[1].strip("/")
|
|
128
|
+
if ":" in q and " " not in q: # direct model reference
|
|
129
|
+
return model_actions(llm, library.make_candidate(q, llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
130
|
+
words = [w for w in q.split() if not w.startswith("/")]
|
|
131
|
+
flags = {w[1:].lower() for w in q.split() if w.startswith("/")}
|
|
132
|
+
cap = next((f for f in ("tools", "vision", "thinking", "embedding", "cloud") if f in flags), None)
|
|
133
|
+
sort = "newest" if "newest" in flags else "popular" if "popular" in flags else None
|
|
134
|
+
query = " ".join(words)
|
|
135
|
+
|
|
136
|
+
shown, page, hidden = [], 1, 0
|
|
137
|
+
while True:
|
|
138
|
+
print(f" Fetching page {page} from ollama.com...")
|
|
139
|
+
try:
|
|
140
|
+
res = library.search(query, page, sort, cap)
|
|
141
|
+
except library.LibraryError as e:
|
|
142
|
+
ui.err(str(e))
|
|
143
|
+
if not shown:
|
|
144
|
+
return
|
|
145
|
+
res = []
|
|
146
|
+
names = {m.name for m in shown}
|
|
147
|
+
res = [m for m in res if m.name not in names]
|
|
148
|
+
if cap != "cloud":
|
|
149
|
+
hidden += sum(1 for m in res if not m.local)
|
|
150
|
+
res = [m for m in res if m.local]
|
|
151
|
+
if not res and not shown:
|
|
152
|
+
ui.warn("No local models found (cloud-only models are hidden; add /cloud to show them).")
|
|
153
|
+
return
|
|
154
|
+
title(f"ollama.com: '{query or 'all models'}'{' /' + cap if cap else ''}{' /' + sort if sort else ''}")
|
|
155
|
+
table(["#", "NAME", "PULLS", "SIZES", "CAPS", "DESCRIPTION"], _result_rows(res, len(shown)), right_cols=(0,))
|
|
156
|
+
shown += res
|
|
157
|
+
if hidden:
|
|
158
|
+
ui.info(f"({hidden} cloud-only models hidden: they run on Ollama's servers, not your GPU)")
|
|
159
|
+
v = ask(f"Open which model (1-{len(shown)}), m = more, Enter = back")
|
|
160
|
+
if v.lower() == "m":
|
|
161
|
+
page += 1
|
|
162
|
+
continue
|
|
163
|
+
if v.isdigit() and 1 <= int(v) <= len(shown):
|
|
164
|
+
return flow_pick_tag(llm, shown[int(v) - 1])
|
|
165
|
+
return
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def flow_pick_tag(llm, m):
|
|
169
|
+
while True:
|
|
170
|
+
print(f"\n Checking sizes for {m.name} (registry lookup)...")
|
|
171
|
+
cands = library.candidates_for(m, llm.hw, llm.settings.context, llm.settings.kv_cache)
|
|
172
|
+
title(m.name)
|
|
173
|
+
show_candidates(cands, llm)
|
|
174
|
+
print(" Type a number, or a custom tag (e.g. 14b-instruct-q8_0). Enter = back.")
|
|
175
|
+
v = ask("Choose")
|
|
176
|
+
if not v or v in ("b", "q"):
|
|
177
|
+
return
|
|
178
|
+
if v.isdigit() and 1 <= int(v) <= len(cands):
|
|
179
|
+
return model_actions(llm, cands[int(v) - 1])
|
|
180
|
+
model_actions(llm, library.make_candidate(f"{m.name}:{v}", llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
181
|
+
return
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def flow_recommended(llm):
|
|
185
|
+
cands = llm.recommended()
|
|
186
|
+
cands = [x for x in cands if x.est.verdict != "no"] or cands
|
|
187
|
+
title("Recommended for your hardware")
|
|
188
|
+
show_candidates(cands, llm)
|
|
189
|
+
i = pick("Select model", len(cands))
|
|
190
|
+
if i is not None:
|
|
191
|
+
model_actions(llm, library.make_candidate(cands[i].ref, llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def flow_installed(llm):
|
|
195
|
+
if not ensure_server(llm):
|
|
196
|
+
return
|
|
197
|
+
inst = llm.installed()
|
|
198
|
+
if not inst:
|
|
199
|
+
ui.warn("No models installed yet. Use 'Search' or 'Recommended'.")
|
|
200
|
+
return
|
|
201
|
+
loaded = {m["name"] for m in llm.loaded()}
|
|
202
|
+
rows = []
|
|
203
|
+
for i, m in enumerate(inst, 1):
|
|
204
|
+
cd = llm.candidate(m["name"])
|
|
205
|
+
cd.size_gb = m.get("size", 0) / 1024 ** 3 or cd.size_gb
|
|
206
|
+
mark = c("loaded", "green") if m["name"] in loaded else ""
|
|
207
|
+
active = c("*", "cyan") if m["name"] == llm.settings.model else ""
|
|
208
|
+
rows.append((i, m["name"], f"{cd.size_gb:.1f} GB", fit_cell(cd.est), speed_cell(cd.est), mark, active))
|
|
209
|
+
title("Installed models")
|
|
210
|
+
table(["#", "MODEL", "SIZE", "FIT", "TOK/S", "STATE", ""], rows, right_cols=(0, 2, 4))
|
|
211
|
+
i = pick("Manage which", len(inst))
|
|
212
|
+
if i is None:
|
|
213
|
+
return
|
|
214
|
+
name = inst[i]["name"]
|
|
215
|
+
print(f"\n {c(name, 'bold')}\n 1) Load / make active\n 2) Unload from VRAM\n 3) Benchmark (real tok/s)\n 4) Delete\n Enter) back")
|
|
216
|
+
ch = ask("Choose")
|
|
217
|
+
try:
|
|
218
|
+
if ch == "1":
|
|
219
|
+
llm.load(name)
|
|
220
|
+
ui.ok("Loaded.")
|
|
221
|
+
after_load(llm)
|
|
222
|
+
elif ch == "2":
|
|
223
|
+
llm.unload(name)
|
|
224
|
+
ui.ok("Unloaded.")
|
|
225
|
+
elif ch == "3":
|
|
226
|
+
flow_benchmark(llm, name)
|
|
227
|
+
elif ch == "4" and confirm(f"Delete {name}?", False):
|
|
228
|
+
llm.delete(name)
|
|
229
|
+
ui.ok("Deleted.")
|
|
230
|
+
except Exception as e:
|
|
231
|
+
ui.err(str(e))
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def flow_benchmark(llm, name=None):
|
|
235
|
+
name = name or llm.settings.model
|
|
236
|
+
if not name or not ensure_server(llm):
|
|
237
|
+
ui.warn("Select/load a model first.")
|
|
238
|
+
return
|
|
239
|
+
print(f" Benchmarking {name} (first run includes load time)...")
|
|
240
|
+
try:
|
|
241
|
+
r = llm.benchmark(name)
|
|
242
|
+
except Exception as e:
|
|
243
|
+
ui.err(str(e))
|
|
244
|
+
return
|
|
245
|
+
est = llm.candidate(name).est
|
|
246
|
+
gen = c(f"{r['tok_s']:.1f} tok/s", "bold", "green")
|
|
247
|
+
guess = f" (estimate was ~{est.tok_s:.0f})" if est.tok_s else ""
|
|
248
|
+
print(f" Generation : {gen}{guess}")
|
|
249
|
+
print(f" Prompt eval: {r['prompt_tok_s']:.0f} tok/s Load time: {r['load_s']:.1f}s")
|
|
250
|
+
if "gpu_pct" in r:
|
|
251
|
+
print(f" Placement : {r['gpu_pct']:.0f}% on GPU ({r['loaded_gb']:.1f} GB in memory)")
|
|
252
|
+
if r["gpu_pct"] < 99:
|
|
253
|
+
ui.warn("Model is partly on CPU. Try a smaller model/quant or lower context in Settings.")
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def flow_open_tunnel(llm):
|
|
257
|
+
if not ensure_server(llm):
|
|
258
|
+
return
|
|
259
|
+
try:
|
|
260
|
+
url = llm.open_tunnel()
|
|
261
|
+
except Exception as e:
|
|
262
|
+
ui.err(str(e))
|
|
263
|
+
return
|
|
264
|
+
print("\n " + c(url, "bold", "green"))
|
|
265
|
+
ui.warn("Anyone with this link can use your GPU. It changes every time you reopen the tunnel.")
|
|
266
|
+
if llm.settings.model:
|
|
267
|
+
print("\n Paste into Continue config.yaml:\n")
|
|
268
|
+
print(llm.continue_config())
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def flow_tunnel(llm):
|
|
272
|
+
if not llm.url:
|
|
273
|
+
return flow_open_tunnel(llm)
|
|
274
|
+
print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1) Show Continue config\n 2) Close tunnel\n 3) New tunnel (new link)\n Enter) back")
|
|
275
|
+
ch = ask("Choose")
|
|
276
|
+
if ch == "1":
|
|
277
|
+
print("\n" + llm.continue_config())
|
|
278
|
+
elif ch == "2":
|
|
279
|
+
llm.close_tunnel()
|
|
280
|
+
ui.ok("Tunnel closed.")
|
|
281
|
+
elif ch == "3":
|
|
282
|
+
llm.close_tunnel()
|
|
283
|
+
flow_open_tunnel(llm)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def flow_server(llm):
|
|
287
|
+
up = llm.server.is_running()
|
|
288
|
+
print(f"\n Server is {c('running', 'green') if up else c('stopped', 'red')}\n 1) {'Restart' if up else 'Start'}\n 2) Stop\n 3) Show log tail + GPU report\n Enter) back")
|
|
289
|
+
ch = ask("Choose")
|
|
290
|
+
try:
|
|
291
|
+
if ch == "1":
|
|
292
|
+
llm.restart_server() if up else llm.start_server()
|
|
293
|
+
ui.ok("Server running.")
|
|
294
|
+
elif ch == "2":
|
|
295
|
+
llm.stop_server()
|
|
296
|
+
ui.ok("Stopped.")
|
|
297
|
+
elif ch == "3":
|
|
298
|
+
print(llm.server.gpu_report() or "(no GPU lines in log)")
|
|
299
|
+
print(c("--- log ---", "gray"))
|
|
300
|
+
print(llm.server.log_tail(15))
|
|
301
|
+
except Exception as e:
|
|
302
|
+
ui.err(str(e))
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def flow_settings(llm):
|
|
306
|
+
s = llm.settings
|
|
307
|
+
print(f"\n 1) Context length : {s.context or 'auto'}\n 2) KV cache type : {s.kv_cache} (f16 | q8_0 | q4_0)\n"
|
|
308
|
+
f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n Enter) back")
|
|
309
|
+
ch = ask("Change which")
|
|
310
|
+
if ch == "1":
|
|
311
|
+
v = ask("Context tokens (number, or 'auto')", "auto")
|
|
312
|
+
s.context = None if v == "auto" else int(v) if v.isdigit() else s.context
|
|
313
|
+
elif ch == "2":
|
|
314
|
+
v = ask("KV cache", s.kv_cache)
|
|
315
|
+
s.kv_cache = v if v in ("f16", "q8_0", "q4_0") else s.kv_cache
|
|
316
|
+
elif ch == "3":
|
|
317
|
+
s.flash_attention = confirm("Enable flash attention?", s.flash_attention)
|
|
318
|
+
elif ch == "4":
|
|
319
|
+
s.keep_alive = ask("Keep-alive", s.keep_alive)
|
|
320
|
+
else:
|
|
321
|
+
return
|
|
322
|
+
s.save()
|
|
323
|
+
ui.ok("Saved.")
|
|
324
|
+
if llm.server.is_running() and confirm("Restart server to apply now?"):
|
|
325
|
+
llm.restart_server()
|
|
326
|
+
if s.model:
|
|
327
|
+
llm.load(s.model)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def flow_quickstart(llm):
|
|
331
|
+
c0 = llm.best_fit()
|
|
332
|
+
show_estimate_detail(c0)
|
|
333
|
+
print("\n Quick start = install, start server, pull the best-fitting model, load it, open tunnel.")
|
|
334
|
+
m = ask("Model to use ('auto' = shown above)", "auto")
|
|
335
|
+
try:
|
|
336
|
+
url = llm.quickstart(m)
|
|
337
|
+
except Exception as e:
|
|
338
|
+
ui.err(str(e))
|
|
339
|
+
return
|
|
340
|
+
ui.ok(f"Ready: {llm.settings.model}")
|
|
341
|
+
print("\n " + c(url, "bold", "green"))
|
|
342
|
+
print("\n" + llm.continue_config())
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
MENU = [
|
|
346
|
+
("Quick start (auto: install, model, tunnel)", flow_quickstart),
|
|
347
|
+
("Search Ollama library", flow_search),
|
|
348
|
+
("Recommended models for my GPU", flow_recommended),
|
|
349
|
+
("Installed models (load / benchmark / delete)", flow_installed),
|
|
350
|
+
("Tunnel (open / get link / close)", flow_tunnel),
|
|
351
|
+
("Server (start / stop / logs)", flow_server),
|
|
352
|
+
("Settings (context, KV cache, keep-alive)", flow_settings),
|
|
353
|
+
("Benchmark active model", flow_benchmark),
|
|
354
|
+
]
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def run(argv=None) -> None:
|
|
358
|
+
"""Interactive menu. Call this from a notebook cell: `import notebook_llm; notebook_llm.run()`."""
|
|
359
|
+
llm = NotebookLLM()
|
|
360
|
+
while True:
|
|
361
|
+
header(llm)
|
|
362
|
+
print()
|
|
363
|
+
for i, (label, _) in enumerate(MENU, 1):
|
|
364
|
+
print(f" {c(i, 'cyan')}) {label}")
|
|
365
|
+
print(f" {c('0', 'cyan')}) Quit (server and tunnel keep running)")
|
|
366
|
+
try:
|
|
367
|
+
ch = ask("Choose")
|
|
368
|
+
if ch in ("0", "q", "quit", "exit"):
|
|
369
|
+
return
|
|
370
|
+
if ch.isdigit() and 1 <= int(ch) <= len(MENU):
|
|
371
|
+
MENU[int(ch) - 1][1](llm)
|
|
372
|
+
except KeyboardInterrupt:
|
|
373
|
+
print()
|
|
374
|
+
return
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
# ---------- non-interactive commands (for `!notebook-llm ...`) ----------
|
|
378
|
+
def main(argv=None) -> None:
|
|
379
|
+
p = argparse.ArgumentParser(prog="notebook-llm", description="Run with no arguments for the interactive menu.")
|
|
380
|
+
p.add_argument("--no-color", action="store_true")
|
|
381
|
+
sub = p.add_subparsers(dest="cmd")
|
|
382
|
+
sub.add_parser("gpus", help="show hardware + recommended models")
|
|
383
|
+
s = sub.add_parser("search", help="search the Ollama library")
|
|
384
|
+
s.add_argument("query", nargs="?", default="")
|
|
385
|
+
k = sub.add_parser("check", help="will MODEL fit? estimated tok/s")
|
|
386
|
+
k.add_argument("model")
|
|
387
|
+
k.add_argument("--ctx", type=int)
|
|
388
|
+
q = sub.add_parser("quickstart", help="install, start, pull best model, open tunnel")
|
|
389
|
+
q.add_argument("--model", default="auto")
|
|
390
|
+
q.add_argument("--no-tunnel", action="store_true")
|
|
391
|
+
sub.add_parser("status")
|
|
392
|
+
sub.add_parser("stop", help="stop tunnel and server")
|
|
393
|
+
a = p.parse_args(argv)
|
|
394
|
+
if a.no_color:
|
|
395
|
+
import os
|
|
396
|
+
os.environ["NO_COLOR"] = "1"
|
|
397
|
+
|
|
398
|
+
if a.cmd is None:
|
|
399
|
+
return run()
|
|
400
|
+
llm = NotebookLLM()
|
|
401
|
+
if a.cmd == "gpus":
|
|
402
|
+
print(llm.hw.summary())
|
|
403
|
+
show_candidates(llm.recommended(), llm)
|
|
404
|
+
elif a.cmd == "search":
|
|
405
|
+
for m in library.search(a.query)[:20]:
|
|
406
|
+
print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}")
|
|
407
|
+
elif a.cmd == "check":
|
|
408
|
+
cd = library.make_candidate(a.model, llm.hw, a.ctx, llm.settings.kv_cache)
|
|
409
|
+
show_estimate_detail(cd)
|
|
410
|
+
elif a.cmd == "quickstart":
|
|
411
|
+
url = llm.quickstart(a.model, tunnel=not a.no_tunnel)
|
|
412
|
+
print(url or "")
|
|
413
|
+
print(llm.continue_config())
|
|
414
|
+
elif a.cmd == "status":
|
|
415
|
+
print(f"server={llm.server.is_running()} model={llm.settings.model} tunnel={llm.url}")
|
|
416
|
+
elif a.cmd == "stop":
|
|
417
|
+
llm.close_tunnel()
|
|
418
|
+
llm.stop_server()
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
if __name__ == "__main__":
|
|
422
|
+
main()
|
notebook_llm/client.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Thin Ollama HTTP client."""
|
|
2
|
+
import json
|
|
3
|
+
import sys
|
|
4
|
+
from typing import Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class OllamaClient:
|
|
10
|
+
def __init__(self, base_url: str = "http://127.0.0.1:11434"):
|
|
11
|
+
self.base = base_url.rstrip("/")
|
|
12
|
+
|
|
13
|
+
def list_models(self) -> List[Dict]:
|
|
14
|
+
r = requests.get(f"{self.base}/api/tags", timeout=10)
|
|
15
|
+
r.raise_for_status()
|
|
16
|
+
return r.json().get("models", [])
|
|
17
|
+
|
|
18
|
+
def has_model(self, name: str) -> bool:
|
|
19
|
+
names = {m["name"] for m in self.list_models()}
|
|
20
|
+
return name in names or f"{name}:latest" in names
|
|
21
|
+
|
|
22
|
+
def running(self) -> List[Dict]:
|
|
23
|
+
r = requests.get(f"{self.base}/api/ps", timeout=10)
|
|
24
|
+
r.raise_for_status()
|
|
25
|
+
return r.json().get("models", [])
|
|
26
|
+
|
|
27
|
+
def pull(self, name: str, show_progress: bool = True) -> None:
|
|
28
|
+
with requests.post(f"{self.base}/api/pull", json={"model": name, "stream": True},
|
|
29
|
+
stream=True, timeout=None) as r:
|
|
30
|
+
r.raise_for_status()
|
|
31
|
+
last = ""
|
|
32
|
+
for line in r.iter_lines():
|
|
33
|
+
if not line:
|
|
34
|
+
continue
|
|
35
|
+
d = json.loads(line)
|
|
36
|
+
if "error" in d:
|
|
37
|
+
raise RuntimeError(d["error"])
|
|
38
|
+
if not show_progress:
|
|
39
|
+
continue
|
|
40
|
+
total, done = d.get("total"), d.get("completed")
|
|
41
|
+
msg = (f"{d.get('status', '')} {done * 100 // total}% "
|
|
42
|
+
f"({done / 1e9:.1f}/{total / 1e9:.1f} GB)") if total and done is not None else d.get("status", "")
|
|
43
|
+
if msg != last:
|
|
44
|
+
sys.stdout.write(f"\r {msg:<70}")
|
|
45
|
+
sys.stdout.flush()
|
|
46
|
+
last = msg
|
|
47
|
+
if show_progress:
|
|
48
|
+
print()
|
|
49
|
+
|
|
50
|
+
def delete(self, name: str) -> None:
|
|
51
|
+
requests.delete(f"{self.base}/api/delete", json={"model": name}, timeout=30).raise_for_status()
|
|
52
|
+
|
|
53
|
+
def load(self, name: str, keep_alive=-1) -> None:
|
|
54
|
+
requests.post(f"{self.base}/api/generate",
|
|
55
|
+
json={"model": name, "prompt": "", "stream": False, "keep_alive": keep_alive},
|
|
56
|
+
timeout=900).raise_for_status()
|
|
57
|
+
|
|
58
|
+
def unload(self, name: str) -> None:
|
|
59
|
+
self.load(name, keep_alive=0)
|
|
60
|
+
|
|
61
|
+
def chat(self, name: str, prompt: str, system: Optional[str] = None, keep_alive=-1) -> str:
|
|
62
|
+
msgs = ([{"role": "system", "content": system}] if system else []) + [{"role": "user", "content": prompt}]
|
|
63
|
+
r = requests.post(f"{self.base}/api/chat",
|
|
64
|
+
json={"model": name, "messages": msgs, "stream": False, "keep_alive": keep_alive}, timeout=900)
|
|
65
|
+
r.raise_for_status()
|
|
66
|
+
return r.json()["message"]["content"]
|
|
67
|
+
|
|
68
|
+
def benchmark(self, name: str, num_predict: int = 128) -> Dict[str, float]:
|
|
69
|
+
"""Measure real tokens/sec using Ollama's own timing counters."""
|
|
70
|
+
r = requests.post(f"{self.base}/api/generate", json={
|
|
71
|
+
"model": name, "stream": False, "keep_alive": -1,
|
|
72
|
+
"prompt": "Write a Python function that checks whether a number is prime, with a short explanation.",
|
|
73
|
+
"options": {"num_predict": num_predict, "temperature": 0}}, timeout=900)
|
|
74
|
+
r.raise_for_status()
|
|
75
|
+
d = r.json()
|
|
76
|
+
gen = d.get("eval_count", 0) / max(d.get("eval_duration", 1), 1) * 1e9
|
|
77
|
+
pre = d.get("prompt_eval_count", 0) / max(d.get("prompt_eval_duration", 1), 1) * 1e9
|
|
78
|
+
out = {"tok_s": gen, "prompt_tok_s": pre, "load_s": d.get("load_duration", 0) / 1e9, "tokens": d.get("eval_count", 0)}
|
|
79
|
+
for m in self.running():
|
|
80
|
+
if m.get("name", "").startswith(name.split(":")[0]) and m.get("size"):
|
|
81
|
+
out["gpu_pct"] = 100 * m.get("size_vram", 0) / m["size"]
|
|
82
|
+
out["loaded_gb"] = m["size"] / 1024 ** 3
|
|
83
|
+
return out
|
notebook_llm/config.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Continue (IDE extension) config for the tunnel URL."""
|
|
2
|
+
from typing import Optional
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def continue_config(api_base: str, model: str, autocomplete_model: Optional[str] = None) -> str:
|
|
6
|
+
cfg = f"""name: Notebook LLM
|
|
7
|
+
version: 1.0.0
|
|
8
|
+
schema: v1
|
|
9
|
+
models:
|
|
10
|
+
- name: {model}
|
|
11
|
+
provider: ollama
|
|
12
|
+
model: {model}
|
|
13
|
+
apiBase: {api_base}
|
|
14
|
+
roles:
|
|
15
|
+
- chat
|
|
16
|
+
- edit
|
|
17
|
+
- apply
|
|
18
|
+
"""
|
|
19
|
+
if autocomplete_model:
|
|
20
|
+
cfg += f""" - name: {autocomplete_model} (autocomplete)
|
|
21
|
+
provider: ollama
|
|
22
|
+
model: {autocomplete_model}
|
|
23
|
+
apiBase: {api_base}
|
|
24
|
+
roles:
|
|
25
|
+
- autocomplete
|
|
26
|
+
"""
|
|
27
|
+
cfg += f""" - name: Autodetect
|
|
28
|
+
provider: ollama
|
|
29
|
+
model: AUTODETECT
|
|
30
|
+
apiBase: {api_base}
|
|
31
|
+
"""
|
|
32
|
+
return cfg
|