notebook-llm-cli 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. notebook_llm_cli-0.1.0/LICENSE +21 -0
  2. notebook_llm_cli-0.1.0/PKG-INFO +79 -0
  3. notebook_llm_cli-0.1.0/README.md +57 -0
  4. notebook_llm_cli-0.1.0/notebook_llm/__init__.py +12 -0
  5. notebook_llm_cli-0.1.0/notebook_llm/__main__.py +3 -0
  6. notebook_llm_cli-0.1.0/notebook_llm/cli.py +422 -0
  7. notebook_llm_cli-0.1.0/notebook_llm/client.py +83 -0
  8. notebook_llm_cli-0.1.0/notebook_llm/config.py +32 -0
  9. notebook_llm_cli-0.1.0/notebook_llm/env.py +9 -0
  10. notebook_llm_cli-0.1.0/notebook_llm/estimate.py +128 -0
  11. notebook_llm_cli-0.1.0/notebook_llm/estimate_data.py +38 -0
  12. notebook_llm_cli-0.1.0/notebook_llm/gpu.py +127 -0
  13. notebook_llm_cli-0.1.0/notebook_llm/installer.py +67 -0
  14. notebook_llm_cli-0.1.0/notebook_llm/library.py +212 -0
  15. notebook_llm_cli-0.1.0/notebook_llm/manager.py +169 -0
  16. notebook_llm_cli-0.1.0/notebook_llm/server.py +101 -0
  17. notebook_llm_cli-0.1.0/notebook_llm/tunnel.py +63 -0
  18. notebook_llm_cli-0.1.0/notebook_llm/ui.py +71 -0
  19. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/PKG-INFO +79 -0
  20. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/SOURCES.txt +25 -0
  21. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/dependency_links.txt +1 -0
  22. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/entry_points.txt +3 -0
  23. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/requires.txt +1 -0
  24. notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/top_level.txt +1 -0
  25. notebook_llm_cli-0.1.0/pyproject.toml +34 -0
  26. notebook_llm_cli-0.1.0/setup.cfg +4 -0
  27. notebook_llm_cli-0.1.0/tests/test_core.py +77 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Shashan Lumbhani
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.4
2
+ Name: notebook-llm-cli
3
+ Version: 0.1.0
4
+ Summary: Interactive CLI to run Ollama LLMs on Kaggle/Colab GPUs: GPU detection, fit and tokens/sec estimates, library search, public tunnel.
5
+ Author-email: Shashan Lumbhani <lumbhanishashan1510@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/soni-shashan/notebook-llm
8
+ Project-URL: Issues, https://github.com/soni-shashan/notebook-llm/issues
9
+ Keywords: ollama,llm,kaggle,colab,gpu,cloudflare-tunnel,cli
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: POSIX :: Linux
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Requires-Python: >=3.8
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ Requires-Dist: requests>=2.25
21
+ Dynamic: license-file
22
+
23
+ # notebook_llm
24
+
25
+ One interactive CLI to run LLMs on a Kaggle/Colab GPU with Ollama: detect GPUs, check whether a model fits, estimate tokens/sec, search the Ollama library, pull, load, benchmark, and open a public tunnel link.
26
+
27
+ ## Install (Kaggle notebook: Internet ON, Accelerator = GPU)
28
+
29
+ ```python
30
+ !pip install -q notebook-llm
31
+ ```
32
+
33
+ From a local folder (copy it to a writable place first, /kaggle/input is read-only):
34
+
35
+ ```python
36
+ !pip install -q ./notebook_llm # or: pip install notebook_llm
37
+ import notebook_llm
38
+ notebook_llm.run() # opens the interactive menu (input boxes appear in the cell)
39
+ ```
40
+
41
+ In a real terminal just run `notebook-llm`.
42
+ Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm.run()` there.
43
+
44
+ ## Menu
45
+ 1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
46
+ 2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
47
+ 3. Recommended models for my GPU
48
+ 4. Installed models - load / unload / benchmark (real tok/s) / delete
49
+ 5. Tunnel - open, show link, Continue config, close, new link
50
+ 6. Server - start / stop / restart / logs / GPU report
51
+ 7. Settings - context length, KV cache type, flash attention, keep-alive (saved)
52
+ 8. Benchmark the active model
53
+
54
+ ## Non-interactive
55
+ ```
56
+ notebook-llm gpus # hardware + recommended table
57
+ notebook-llm search coder
58
+ notebook-llm check llama3.1:70b --ctx 8192
59
+ notebook-llm quickstart --model auto
60
+ notebook-llm status | stop
61
+ ```
62
+
63
+ ## Python API
64
+ ```python
65
+ from notebook_llm import NotebookLLM
66
+ llm = NotebookLLM()
67
+ url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
68
+ llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
69
+ ```
70
+
71
+ ## How the estimates work
72
+ - **Needs** = weights (exact size from registry.ollama.ai, else params x bytes/param) + KV cache (scales with context and KV type) + ~0.7 GB per GPU.
73
+ - **FIT**: FITS (<=92% of total VRAM), TIGHT (<=100%), SLOW (spills to RAM), TOO BIG.
74
+ - **tok/s** = memory bandwidth / bytes read per token, using a built-in GPU bandwidth table (T4, P100, V100, A100, L4, RTX...). MoE models only read their active experts (e.g. qwen3-coder:30b ~3.3B active) so they are much faster than dense models of the same size. Multi-GPU is layer-split, so it adds capacity, not speed.
75
+ - Estimates are +-30%. Use Benchmark for the real number.
76
+
77
+ ## Notes
78
+ - The tunnel link has no authentication; anyone with it can use your GPU.
79
+ - NVIDIA GPUs only. Library search scrapes ollama.com; if it is unreachable a built-in list is used.
@@ -0,0 +1,57 @@
1
+ # notebook_llm
2
+
3
+ One interactive CLI to run LLMs on a Kaggle/Colab GPU with Ollama: detect GPUs, check whether a model fits, estimate tokens/sec, search the Ollama library, pull, load, benchmark, and open a public tunnel link.
4
+
5
+ ## Install (Kaggle notebook: Internet ON, Accelerator = GPU)
6
+
7
+ ```python
8
+ !pip install -q notebook-llm
9
+ ```
10
+
11
+ From a local folder (copy it to a writable place first, /kaggle/input is read-only):
12
+
13
+ ```python
14
+ !pip install -q ./notebook_llm # or: pip install notebook_llm
15
+ import notebook_llm
16
+ notebook_llm.run() # opens the interactive menu (input boxes appear in the cell)
17
+ ```
18
+
19
+ In a real terminal just run `notebook-llm`.
20
+ Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm.run()` there.
21
+
22
+ ## Menu
23
+ 1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
24
+ 2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
25
+ 3. Recommended models for my GPU
26
+ 4. Installed models - load / unload / benchmark (real tok/s) / delete
27
+ 5. Tunnel - open, show link, Continue config, close, new link
28
+ 6. Server - start / stop / restart / logs / GPU report
29
+ 7. Settings - context length, KV cache type, flash attention, keep-alive (saved)
30
+ 8. Benchmark the active model
31
+
32
+ ## Non-interactive
33
+ ```
34
+ notebook-llm gpus # hardware + recommended table
35
+ notebook-llm search coder
36
+ notebook-llm check llama3.1:70b --ctx 8192
37
+ notebook-llm quickstart --model auto
38
+ notebook-llm status | stop
39
+ ```
40
+
41
+ ## Python API
42
+ ```python
43
+ from notebook_llm import NotebookLLM
44
+ llm = NotebookLLM()
45
+ url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
46
+ llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
47
+ ```
48
+
49
+ ## How the estimates work
50
+ - **Needs** = weights (exact size from registry.ollama.ai, else params x bytes/param) + KV cache (scales with context and KV type) + ~0.7 GB per GPU.
51
+ - **FIT**: FITS (<=92% of total VRAM), TIGHT (<=100%), SLOW (spills to RAM), TOO BIG.
52
+ - **tok/s** = memory bandwidth / bytes read per token, using a built-in GPU bandwidth table (T4, P100, V100, A100, L4, RTX...). MoE models only read their active experts (e.g. qwen3-coder:30b ~3.3B active) so they are much faster than dense models of the same size. Multi-GPU is layer-split, so it adds capacity, not speed.
53
+ - Estimates are +-30%. Use Benchmark for the real number.
54
+
55
+ ## Notes
56
+ - The tunnel link has no authentication; anyone with it can use your GPU.
57
+ - NVIDIA GPUs only. Library search scrapes ollama.com; if it is unreachable a built-in list is used.
@@ -0,0 +1,12 @@
1
+ """notebook_llm: manage Ollama LLMs on notebook GPUs (Kaggle/Colab) from one interactive CLI."""
2
+ from .gpu import GPU, GPUInfo, Hardware, detect_gpus, detect_hardware
3
+ from .estimate import Estimate, estimate
4
+ from .library import LibModel, Candidate, search, make_candidate
5
+ from .manager import NotebookLLM, Settings
6
+ from .cli import run, main
7
+ from .env import is_kaggle
8
+
9
+ __all__ = ["NotebookLLM", "Settings", "run", "main", "GPU", "GPUInfo", "Hardware",
10
+ "detect_gpus", "detect_hardware", "Estimate", "estimate", "LibModel",
11
+ "Candidate", "search", "make_candidate", "is_kaggle"]
12
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
@@ -0,0 +1,422 @@
1
+ """Interactive CLI. Run `notebook-llm` (terminal) or `notebook_llm.run()` (notebook cell)."""
2
+ import argparse
3
+ import sys
4
+
5
+ from . import library, ui
6
+ from .env import is_kaggle
7
+ from .manager import NotebookLLM
8
+ from .ui import c, ask, confirm, pick, table, title
9
+
10
+
11
+ # ---------- rendering helpers ---------------------------------------
12
+ def fit_cell(est) -> str:
13
+ return c(est.label, ui.VERDICT_STYLE[est.verdict])
14
+
15
+
16
+ def speed_cell(est) -> str:
17
+ return f"~{est.tok_s:.0f}" if est.tok_s else "-"
18
+
19
+
20
+ def show_candidates(cands, llm):
21
+ ctx = cands[0].est.context if cands else "-"
22
+ ui.info(f"Hardware: {llm.hw.summary()}")
23
+ ui.info("NEEDS = weights + KV cache (auto context) + overhead. tok/s is an estimate (+-30%).")
24
+ rows = []
25
+ for i, cd in enumerate(cands, 1):
26
+ size = f"{cd.size_gb:.1f} GB" + ("" if cd.exact else "~")
27
+ moe = " MoE" if cd.active_b and cd.params_b and cd.active_b < cd.params_b * 0.8 else ""
28
+ rows.append((i, cd.ref + c(moe, "gray"), size, f"{cd.est.needed_gb:.1f} GB", cd.est.context,
29
+ fit_cell(cd.est), speed_cell(cd.est)))
30
+ table(["#", "MODEL", "SIZE", "NEEDS", "CTX", "FIT", "TOK/S"], rows, right_cols=(0, 2, 3, 4, 6))
31
+
32
+
33
+ def show_estimate_detail(cd):
34
+ e = cd.est
35
+ print(f"\n {c(cd.ref, 'bold')} ({'exact' if cd.exact else 'estimated'} size {cd.size_gb:.1f} GB)")
36
+ print(f" Weights {e.weights_gb:.1f} GB + KV cache {e.kv_gb:.1f} GB (ctx {e.context}) + overhead {e.overhead_gb:.1f} GB"
37
+ f" = {c(f'{e.needed_gb:.1f} GB', 'bold')}")
38
+ print(f" Verdict: {fit_cell(e)} Speed: {c(speed_cell(e) + ' tok/s', 'bold')} "
39
+ f"On GPU: {e.gpu_fraction * 100:.0f}% of weights")
40
+ if e.note:
41
+ ui.warn(e.note)
42
+
43
+
44
+ def header(llm: NotebookLLM):
45
+ llm.refresh_hardware()
46
+ title("notebook_llm")
47
+ print(f" {llm.hw.summary()}")
48
+ if llm.hw.gpu.available:
49
+ print(f" VRAM in use: {llm.hw.gpu.used_vram_gb} / {llm.hw.gpu.total_vram_gb} GB")
50
+ elif is_kaggle():
51
+ ui.warn("No GPU. In Kaggle: Settings > Accelerator > GPU T4 x2, then restart.")
52
+ up = llm.server.is_running()
53
+ print(f" Server : {c('running', 'green') if up else c('stopped', 'red')}")
54
+ if up:
55
+ try:
56
+ for m in llm.loaded():
57
+ pct = 100 * m.get("size_vram", 0) / m["size"] if m.get("size") else 0
58
+ print(f" Loaded : {c(m['name'], 'bold')} ({m.get('size', 0) / 1024 ** 3:.1f} GB, {pct:.0f}% on GPU)")
59
+ except Exception:
60
+ pass
61
+ print(f" Model : {llm.settings.model or '(none selected)'}")
62
+ print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else c('closed', 'gray')}")
63
+
64
+
65
+ # ---------- flows ----------------------------------------------------
66
+ def ensure_server(llm, model=None) -> bool:
67
+ if llm.server.is_running():
68
+ return True
69
+ print(" Installing/starting Ollama...")
70
+ try:
71
+ llm.start_server(model)
72
+ return True
73
+ except Exception as e:
74
+ ui.err(str(e))
75
+ return False
76
+
77
+
78
+ def after_load(llm):
79
+ if not llm.url and confirm("Open a public tunnel now?"):
80
+ flow_open_tunnel(llm)
81
+
82
+
83
+ def model_actions(llm, cd):
84
+ show_estimate_detail(cd)
85
+ if cd.est.verdict == "no":
86
+ ui.warn("This model will not run on your hardware.")
87
+ if not confirm("Pull anyway?", False):
88
+ return
89
+ print("\n 1) Pull + load now (make it the active model)\n 2) Pull only\n Enter) back")
90
+ ch = ask("Choose")
91
+ if ch not in ("1", "2"):
92
+ return
93
+ if not ensure_server(llm, cd.ref):
94
+ return
95
+ try:
96
+ if not llm.client.has_model(cd.ref):
97
+ print(f" Pulling {cd.ref} ({cd.size_gb:.1f} GB)...")
98
+ llm.pull(cd.ref)
99
+ if ch == "1":
100
+ print(" Loading into VRAM...")
101
+ llm.load(cd.ref)
102
+ ui.ok(f"{cd.ref} is loaded and will stay in memory.")
103
+ after_load(llm)
104
+ except Exception as e:
105
+ ui.err(str(e))
106
+
107
+
108
+ def _result_rows(res, offset=0):
109
+ rows = []
110
+ for i, m in enumerate(res, 1 + offset):
111
+ sizes = ", ".join(m.sizes[:6]) + (" ..." if len(m.sizes) > 6 else "")
112
+ if m.cloud:
113
+ sizes = (sizes + " " if sizes else "") + c("cloud", "yellow")
114
+ desc = m.description.split(". ")[0]
115
+ rows.append((i, c(m.name, "bold"), m.pulls, sizes or "-", ",".join(m.capabilities[:3]),
116
+ (desc[:46] + "...") if len(desc) > 46 else desc))
117
+ return rows
118
+
119
+
120
+ def flow_search(llm):
121
+ print(" Search ollama.com/search. Examples: coder | /vision | qwen /tools | /newest")
122
+ print(" Filters: /tools /vision /thinking /embedding /cloud /newest /popular. Or paste a model like qwen3:8b")
123
+ q = ask("Search (blank = all models)")
124
+ if q in ("q", "b"):
125
+ return
126
+ if q.startswith("http") and "/library/" in q:
127
+ q = q.split("/library/", 1)[1].strip("/")
128
+ if ":" in q and " " not in q: # direct model reference
129
+ return model_actions(llm, library.make_candidate(q, llm.hw, llm.settings.context, llm.settings.kv_cache))
130
+ words = [w for w in q.split() if not w.startswith("/")]
131
+ flags = {w[1:].lower() for w in q.split() if w.startswith("/")}
132
+ cap = next((f for f in ("tools", "vision", "thinking", "embedding", "cloud") if f in flags), None)
133
+ sort = "newest" if "newest" in flags else "popular" if "popular" in flags else None
134
+ query = " ".join(words)
135
+
136
+ shown, page, hidden = [], 1, 0
137
+ while True:
138
+ print(f" Fetching page {page} from ollama.com...")
139
+ try:
140
+ res = library.search(query, page, sort, cap)
141
+ except library.LibraryError as e:
142
+ ui.err(str(e))
143
+ if not shown:
144
+ return
145
+ res = []
146
+ names = {m.name for m in shown}
147
+ res = [m for m in res if m.name not in names]
148
+ if cap != "cloud":
149
+ hidden += sum(1 for m in res if not m.local)
150
+ res = [m for m in res if m.local]
151
+ if not res and not shown:
152
+ ui.warn("No local models found (cloud-only models are hidden; add /cloud to show them).")
153
+ return
154
+ title(f"ollama.com: '{query or 'all models'}'{' /' + cap if cap else ''}{' /' + sort if sort else ''}")
155
+ table(["#", "NAME", "PULLS", "SIZES", "CAPS", "DESCRIPTION"], _result_rows(res, len(shown)), right_cols=(0,))
156
+ shown += res
157
+ if hidden:
158
+ ui.info(f"({hidden} cloud-only models hidden: they run on Ollama's servers, not your GPU)")
159
+ v = ask(f"Open which model (1-{len(shown)}), m = more, Enter = back")
160
+ if v.lower() == "m":
161
+ page += 1
162
+ continue
163
+ if v.isdigit() and 1 <= int(v) <= len(shown):
164
+ return flow_pick_tag(llm, shown[int(v) - 1])
165
+ return
166
+
167
+
168
+ def flow_pick_tag(llm, m):
169
+ while True:
170
+ print(f"\n Checking sizes for {m.name} (registry lookup)...")
171
+ cands = library.candidates_for(m, llm.hw, llm.settings.context, llm.settings.kv_cache)
172
+ title(m.name)
173
+ show_candidates(cands, llm)
174
+ print(" Type a number, or a custom tag (e.g. 14b-instruct-q8_0). Enter = back.")
175
+ v = ask("Choose")
176
+ if not v or v in ("b", "q"):
177
+ return
178
+ if v.isdigit() and 1 <= int(v) <= len(cands):
179
+ return model_actions(llm, cands[int(v) - 1])
180
+ model_actions(llm, library.make_candidate(f"{m.name}:{v}", llm.hw, llm.settings.context, llm.settings.kv_cache))
181
+ return
182
+
183
+
184
+ def flow_recommended(llm):
185
+ cands = llm.recommended()
186
+ cands = [x for x in cands if x.est.verdict != "no"] or cands
187
+ title("Recommended for your hardware")
188
+ show_candidates(cands, llm)
189
+ i = pick("Select model", len(cands))
190
+ if i is not None:
191
+ model_actions(llm, library.make_candidate(cands[i].ref, llm.hw, llm.settings.context, llm.settings.kv_cache))
192
+
193
+
194
+ def flow_installed(llm):
195
+ if not ensure_server(llm):
196
+ return
197
+ inst = llm.installed()
198
+ if not inst:
199
+ ui.warn("No models installed yet. Use 'Search' or 'Recommended'.")
200
+ return
201
+ loaded = {m["name"] for m in llm.loaded()}
202
+ rows = []
203
+ for i, m in enumerate(inst, 1):
204
+ cd = llm.candidate(m["name"])
205
+ cd.size_gb = m.get("size", 0) / 1024 ** 3 or cd.size_gb
206
+ mark = c("loaded", "green") if m["name"] in loaded else ""
207
+ active = c("*", "cyan") if m["name"] == llm.settings.model else ""
208
+ rows.append((i, m["name"], f"{cd.size_gb:.1f} GB", fit_cell(cd.est), speed_cell(cd.est), mark, active))
209
+ title("Installed models")
210
+ table(["#", "MODEL", "SIZE", "FIT", "TOK/S", "STATE", ""], rows, right_cols=(0, 2, 4))
211
+ i = pick("Manage which", len(inst))
212
+ if i is None:
213
+ return
214
+ name = inst[i]["name"]
215
+ print(f"\n {c(name, 'bold')}\n 1) Load / make active\n 2) Unload from VRAM\n 3) Benchmark (real tok/s)\n 4) Delete\n Enter) back")
216
+ ch = ask("Choose")
217
+ try:
218
+ if ch == "1":
219
+ llm.load(name)
220
+ ui.ok("Loaded.")
221
+ after_load(llm)
222
+ elif ch == "2":
223
+ llm.unload(name)
224
+ ui.ok("Unloaded.")
225
+ elif ch == "3":
226
+ flow_benchmark(llm, name)
227
+ elif ch == "4" and confirm(f"Delete {name}?", False):
228
+ llm.delete(name)
229
+ ui.ok("Deleted.")
230
+ except Exception as e:
231
+ ui.err(str(e))
232
+
233
+
234
+ def flow_benchmark(llm, name=None):
235
+ name = name or llm.settings.model
236
+ if not name or not ensure_server(llm):
237
+ ui.warn("Select/load a model first.")
238
+ return
239
+ print(f" Benchmarking {name} (first run includes load time)...")
240
+ try:
241
+ r = llm.benchmark(name)
242
+ except Exception as e:
243
+ ui.err(str(e))
244
+ return
245
+ est = llm.candidate(name).est
246
+ gen = c(f"{r['tok_s']:.1f} tok/s", "bold", "green")
247
+ guess = f" (estimate was ~{est.tok_s:.0f})" if est.tok_s else ""
248
+ print(f" Generation : {gen}{guess}")
249
+ print(f" Prompt eval: {r['prompt_tok_s']:.0f} tok/s Load time: {r['load_s']:.1f}s")
250
+ if "gpu_pct" in r:
251
+ print(f" Placement : {r['gpu_pct']:.0f}% on GPU ({r['loaded_gb']:.1f} GB in memory)")
252
+ if r["gpu_pct"] < 99:
253
+ ui.warn("Model is partly on CPU. Try a smaller model/quant or lower context in Settings.")
254
+
255
+
256
+ def flow_open_tunnel(llm):
257
+ if not ensure_server(llm):
258
+ return
259
+ try:
260
+ url = llm.open_tunnel()
261
+ except Exception as e:
262
+ ui.err(str(e))
263
+ return
264
+ print("\n " + c(url, "bold", "green"))
265
+ ui.warn("Anyone with this link can use your GPU. It changes every time you reopen the tunnel.")
266
+ if llm.settings.model:
267
+ print("\n Paste into Continue config.yaml:\n")
268
+ print(llm.continue_config())
269
+
270
+
271
+ def flow_tunnel(llm):
272
+ if not llm.url:
273
+ return flow_open_tunnel(llm)
274
+ print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1) Show Continue config\n 2) Close tunnel\n 3) New tunnel (new link)\n Enter) back")
275
+ ch = ask("Choose")
276
+ if ch == "1":
277
+ print("\n" + llm.continue_config())
278
+ elif ch == "2":
279
+ llm.close_tunnel()
280
+ ui.ok("Tunnel closed.")
281
+ elif ch == "3":
282
+ llm.close_tunnel()
283
+ flow_open_tunnel(llm)
284
+
285
+
286
+ def flow_server(llm):
287
+ up = llm.server.is_running()
288
+ print(f"\n Server is {c('running', 'green') if up else c('stopped', 'red')}\n 1) {'Restart' if up else 'Start'}\n 2) Stop\n 3) Show log tail + GPU report\n Enter) back")
289
+ ch = ask("Choose")
290
+ try:
291
+ if ch == "1":
292
+ llm.restart_server() if up else llm.start_server()
293
+ ui.ok("Server running.")
294
+ elif ch == "2":
295
+ llm.stop_server()
296
+ ui.ok("Stopped.")
297
+ elif ch == "3":
298
+ print(llm.server.gpu_report() or "(no GPU lines in log)")
299
+ print(c("--- log ---", "gray"))
300
+ print(llm.server.log_tail(15))
301
+ except Exception as e:
302
+ ui.err(str(e))
303
+
304
+
305
+ def flow_settings(llm):
306
+ s = llm.settings
307
+ print(f"\n 1) Context length : {s.context or 'auto'}\n 2) KV cache type : {s.kv_cache} (f16 | q8_0 | q4_0)\n"
308
+ f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n Enter) back")
309
+ ch = ask("Change which")
310
+ if ch == "1":
311
+ v = ask("Context tokens (number, or 'auto')", "auto")
312
+ s.context = None if v == "auto" else int(v) if v.isdigit() else s.context
313
+ elif ch == "2":
314
+ v = ask("KV cache", s.kv_cache)
315
+ s.kv_cache = v if v in ("f16", "q8_0", "q4_0") else s.kv_cache
316
+ elif ch == "3":
317
+ s.flash_attention = confirm("Enable flash attention?", s.flash_attention)
318
+ elif ch == "4":
319
+ s.keep_alive = ask("Keep-alive", s.keep_alive)
320
+ else:
321
+ return
322
+ s.save()
323
+ ui.ok("Saved.")
324
+ if llm.server.is_running() and confirm("Restart server to apply now?"):
325
+ llm.restart_server()
326
+ if s.model:
327
+ llm.load(s.model)
328
+
329
+
330
+ def flow_quickstart(llm):
331
+ c0 = llm.best_fit()
332
+ show_estimate_detail(c0)
333
+ print("\n Quick start = install, start server, pull the best-fitting model, load it, open tunnel.")
334
+ m = ask("Model to use ('auto' = shown above)", "auto")
335
+ try:
336
+ url = llm.quickstart(m)
337
+ except Exception as e:
338
+ ui.err(str(e))
339
+ return
340
+ ui.ok(f"Ready: {llm.settings.model}")
341
+ print("\n " + c(url, "bold", "green"))
342
+ print("\n" + llm.continue_config())
343
+
344
+
345
+ MENU = [
346
+ ("Quick start (auto: install, model, tunnel)", flow_quickstart),
347
+ ("Search Ollama library", flow_search),
348
+ ("Recommended models for my GPU", flow_recommended),
349
+ ("Installed models (load / benchmark / delete)", flow_installed),
350
+ ("Tunnel (open / get link / close)", flow_tunnel),
351
+ ("Server (start / stop / logs)", flow_server),
352
+ ("Settings (context, KV cache, keep-alive)", flow_settings),
353
+ ("Benchmark active model", flow_benchmark),
354
+ ]
355
+
356
+
357
+ def run(argv=None) -> None:
358
+ """Interactive menu. Call this from a notebook cell: `import notebook_llm; notebook_llm.run()`."""
359
+ llm = NotebookLLM()
360
+ while True:
361
+ header(llm)
362
+ print()
363
+ for i, (label, _) in enumerate(MENU, 1):
364
+ print(f" {c(i, 'cyan')}) {label}")
365
+ print(f" {c('0', 'cyan')}) Quit (server and tunnel keep running)")
366
+ try:
367
+ ch = ask("Choose")
368
+ if ch in ("0", "q", "quit", "exit"):
369
+ return
370
+ if ch.isdigit() and 1 <= int(ch) <= len(MENU):
371
+ MENU[int(ch) - 1][1](llm)
372
+ except KeyboardInterrupt:
373
+ print()
374
+ return
375
+
376
+
377
+ # ---------- non-interactive commands (for `!notebook-llm ...`) ----------
378
+ def main(argv=None) -> None:
379
+ p = argparse.ArgumentParser(prog="notebook-llm", description="Run with no arguments for the interactive menu.")
380
+ p.add_argument("--no-color", action="store_true")
381
+ sub = p.add_subparsers(dest="cmd")
382
+ sub.add_parser("gpus", help="show hardware + recommended models")
383
+ s = sub.add_parser("search", help="search the Ollama library")
384
+ s.add_argument("query", nargs="?", default="")
385
+ k = sub.add_parser("check", help="will MODEL fit? estimated tok/s")
386
+ k.add_argument("model")
387
+ k.add_argument("--ctx", type=int)
388
+ q = sub.add_parser("quickstart", help="install, start, pull best model, open tunnel")
389
+ q.add_argument("--model", default="auto")
390
+ q.add_argument("--no-tunnel", action="store_true")
391
+ sub.add_parser("status")
392
+ sub.add_parser("stop", help="stop tunnel and server")
393
+ a = p.parse_args(argv)
394
+ if a.no_color:
395
+ import os
396
+ os.environ["NO_COLOR"] = "1"
397
+
398
+ if a.cmd is None:
399
+ return run()
400
+ llm = NotebookLLM()
401
+ if a.cmd == "gpus":
402
+ print(llm.hw.summary())
403
+ show_candidates(llm.recommended(), llm)
404
+ elif a.cmd == "search":
405
+ for m in library.search(a.query)[:20]:
406
+ print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}")
407
+ elif a.cmd == "check":
408
+ cd = library.make_candidate(a.model, llm.hw, a.ctx, llm.settings.kv_cache)
409
+ show_estimate_detail(cd)
410
+ elif a.cmd == "quickstart":
411
+ url = llm.quickstart(a.model, tunnel=not a.no_tunnel)
412
+ print(url or "")
413
+ print(llm.continue_config())
414
+ elif a.cmd == "status":
415
+ print(f"server={llm.server.is_running()} model={llm.settings.model} tunnel={llm.url}")
416
+ elif a.cmd == "stop":
417
+ llm.close_tunnel()
418
+ llm.stop_server()
419
+
420
+
421
+ if __name__ == "__main__":
422
+ main()