notebook-llm-cli 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- notebook_llm_cli-0.1.0/LICENSE +21 -0
- notebook_llm_cli-0.1.0/PKG-INFO +79 -0
- notebook_llm_cli-0.1.0/README.md +57 -0
- notebook_llm_cli-0.1.0/notebook_llm/__init__.py +12 -0
- notebook_llm_cli-0.1.0/notebook_llm/__main__.py +3 -0
- notebook_llm_cli-0.1.0/notebook_llm/cli.py +422 -0
- notebook_llm_cli-0.1.0/notebook_llm/client.py +83 -0
- notebook_llm_cli-0.1.0/notebook_llm/config.py +32 -0
- notebook_llm_cli-0.1.0/notebook_llm/env.py +9 -0
- notebook_llm_cli-0.1.0/notebook_llm/estimate.py +128 -0
- notebook_llm_cli-0.1.0/notebook_llm/estimate_data.py +38 -0
- notebook_llm_cli-0.1.0/notebook_llm/gpu.py +127 -0
- notebook_llm_cli-0.1.0/notebook_llm/installer.py +67 -0
- notebook_llm_cli-0.1.0/notebook_llm/library.py +212 -0
- notebook_llm_cli-0.1.0/notebook_llm/manager.py +169 -0
- notebook_llm_cli-0.1.0/notebook_llm/server.py +101 -0
- notebook_llm_cli-0.1.0/notebook_llm/tunnel.py +63 -0
- notebook_llm_cli-0.1.0/notebook_llm/ui.py +71 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/PKG-INFO +79 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/SOURCES.txt +25 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/dependency_links.txt +1 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/entry_points.txt +3 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/requires.txt +1 -0
- notebook_llm_cli-0.1.0/notebook_llm_cli.egg-info/top_level.txt +1 -0
- notebook_llm_cli-0.1.0/pyproject.toml +34 -0
- notebook_llm_cli-0.1.0/setup.cfg +4 -0
- notebook_llm_cli-0.1.0/tests/test_core.py +77 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Shashan Lumbhani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: notebook-llm-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Interactive CLI to run Ollama LLMs on Kaggle/Colab GPUs: GPU detection, fit and tokens/sec estimates, library search, public tunnel.
|
|
5
|
+
Author-email: Shashan Lumbhani <lumbhanishashan1510@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/soni-shashan/notebook-llm
|
|
8
|
+
Project-URL: Issues, https://github.com/soni-shashan/notebook-llm/issues
|
|
9
|
+
Keywords: ollama,llm,kaggle,colab,gpu,cloudflare-tunnel,cli
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Requires-Python: >=3.8
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: requests>=2.25
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
# notebook_llm
|
|
24
|
+
|
|
25
|
+
One interactive CLI to run LLMs on a Kaggle/Colab GPU with Ollama: detect GPUs, check whether a model fits, estimate tokens/sec, search the Ollama library, pull, load, benchmark, and open a public tunnel link.
|
|
26
|
+
|
|
27
|
+
## Install (Kaggle notebook: Internet ON, Accelerator = GPU)
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
!pip install -q notebook-llm
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
From a local folder (copy it to a writable place first, /kaggle/input is read-only):
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
!pip install -q ./notebook_llm # or: pip install notebook_llm
|
|
37
|
+
import notebook_llm
|
|
38
|
+
notebook_llm.run() # opens the interactive menu (input boxes appear in the cell)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
In a real terminal just run `notebook-llm`.
|
|
42
|
+
Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm.run()` there.
|
|
43
|
+
|
|
44
|
+
## Menu
|
|
45
|
+
1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
|
|
46
|
+
2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
|
|
47
|
+
3. Recommended models for my GPU
|
|
48
|
+
4. Installed models - load / unload / benchmark (real tok/s) / delete
|
|
49
|
+
5. Tunnel - open, show link, Continue config, close, new link
|
|
50
|
+
6. Server - start / stop / restart / logs / GPU report
|
|
51
|
+
7. Settings - context length, KV cache type, flash attention, keep-alive (saved)
|
|
52
|
+
8. Benchmark the active model
|
|
53
|
+
|
|
54
|
+
## Non-interactive
|
|
55
|
+
```
|
|
56
|
+
notebook-llm gpus # hardware + recommended table
|
|
57
|
+
notebook-llm search coder
|
|
58
|
+
notebook-llm check llama3.1:70b --ctx 8192
|
|
59
|
+
notebook-llm quickstart --model auto
|
|
60
|
+
notebook-llm status | stop
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Python API
|
|
64
|
+
```python
|
|
65
|
+
from notebook_llm import NotebookLLM
|
|
66
|
+
llm = NotebookLLM()
|
|
67
|
+
url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
|
|
68
|
+
llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## How the estimates work
|
|
72
|
+
- **Needs** = weights (exact size from registry.ollama.ai, else params x bytes/param) + KV cache (scales with context and KV type) + ~0.7 GB per GPU.
|
|
73
|
+
- **FIT**: FITS (<=92% of total VRAM), TIGHT (<=100%), SLOW (spills to RAM), TOO BIG.
|
|
74
|
+
- **tok/s** = memory bandwidth / bytes read per token, using a built-in GPU bandwidth table (T4, P100, V100, A100, L4, RTX...). MoE models only read their active experts (e.g. qwen3-coder:30b ~3.3B active) so they are much faster than dense models of the same size. Multi-GPU is layer-split, so it adds capacity, not speed.
|
|
75
|
+
- Estimates are +-30%. Use Benchmark for the real number.
|
|
76
|
+
|
|
77
|
+
## Notes
|
|
78
|
+
- The tunnel link has no authentication; anyone with it can use your GPU.
|
|
79
|
+
- NVIDIA GPUs only. Library search scrapes ollama.com; if it is unreachable a built-in list is used.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# notebook_llm
|
|
2
|
+
|
|
3
|
+
One interactive CLI to run LLMs on a Kaggle/Colab GPU with Ollama: detect GPUs, check whether a model fits, estimate tokens/sec, search the Ollama library, pull, load, benchmark, and open a public tunnel link.
|
|
4
|
+
|
|
5
|
+
## Install (Kaggle notebook: Internet ON, Accelerator = GPU)
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
!pip install -q notebook-llm
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
From a local folder (copy it to a writable place first, /kaggle/input is read-only):
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
!pip install -q ./notebook_llm # or: pip install notebook_llm
|
|
15
|
+
import notebook_llm
|
|
16
|
+
notebook_llm.run() # opens the interactive menu (input boxes appear in the cell)
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
In a real terminal just run `notebook-llm`.
|
|
20
|
+
Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm.run()` there.
|
|
21
|
+
|
|
22
|
+
## Menu
|
|
23
|
+
1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
|
|
24
|
+
2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
|
|
25
|
+
3. Recommended models for my GPU
|
|
26
|
+
4. Installed models - load / unload / benchmark (real tok/s) / delete
|
|
27
|
+
5. Tunnel - open, show link, Continue config, close, new link
|
|
28
|
+
6. Server - start / stop / restart / logs / GPU report
|
|
29
|
+
7. Settings - context length, KV cache type, flash attention, keep-alive (saved)
|
|
30
|
+
8. Benchmark the active model
|
|
31
|
+
|
|
32
|
+
## Non-interactive
|
|
33
|
+
```
|
|
34
|
+
notebook-llm gpus # hardware + recommended table
|
|
35
|
+
notebook-llm search coder
|
|
36
|
+
notebook-llm check llama3.1:70b --ctx 8192
|
|
37
|
+
notebook-llm quickstart --model auto
|
|
38
|
+
notebook-llm status | stop
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Python API
|
|
42
|
+
```python
|
|
43
|
+
from notebook_llm import NotebookLLM
|
|
44
|
+
llm = NotebookLLM()
|
|
45
|
+
url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
|
|
46
|
+
llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## How the estimates work
|
|
50
|
+
- **Needs** = weights (exact size from registry.ollama.ai, else params x bytes/param) + KV cache (scales with context and KV type) + ~0.7 GB per GPU.
|
|
51
|
+
- **FIT**: FITS (<=92% of total VRAM), TIGHT (<=100%), SLOW (spills to RAM), TOO BIG.
|
|
52
|
+
- **tok/s** = memory bandwidth / bytes read per token, using a built-in GPU bandwidth table (T4, P100, V100, A100, L4, RTX...). MoE models only read their active experts (e.g. qwen3-coder:30b ~3.3B active) so they are much faster than dense models of the same size. Multi-GPU is layer-split, so it adds capacity, not speed.
|
|
53
|
+
- Estimates are +-30%. Use Benchmark for the real number.
|
|
54
|
+
|
|
55
|
+
## Notes
|
|
56
|
+
- The tunnel link has no authentication; anyone with it can use your GPU.
|
|
57
|
+
- NVIDIA GPUs only. Library search scrapes ollama.com; if it is unreachable a built-in list is used.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""notebook_llm: manage Ollama LLMs on notebook GPUs (Kaggle/Colab) from one interactive CLI."""
|
|
2
|
+
from .gpu import GPU, GPUInfo, Hardware, detect_gpus, detect_hardware
|
|
3
|
+
from .estimate import Estimate, estimate
|
|
4
|
+
from .library import LibModel, Candidate, search, make_candidate
|
|
5
|
+
from .manager import NotebookLLM, Settings
|
|
6
|
+
from .cli import run, main
|
|
7
|
+
from .env import is_kaggle
|
|
8
|
+
|
|
9
|
+
__all__ = ["NotebookLLM", "Settings", "run", "main", "GPU", "GPUInfo", "Hardware",
|
|
10
|
+
"detect_gpus", "detect_hardware", "Estimate", "estimate", "LibModel",
|
|
11
|
+
"Candidate", "search", "make_candidate", "is_kaggle"]
|
|
12
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,422 @@
|
|
|
1
|
+
"""Interactive CLI. Run `notebook-llm` (terminal) or `notebook_llm.run()` (notebook cell)."""
|
|
2
|
+
import argparse
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
from . import library, ui
|
|
6
|
+
from .env import is_kaggle
|
|
7
|
+
from .manager import NotebookLLM
|
|
8
|
+
from .ui import c, ask, confirm, pick, table, title
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# ---------- rendering helpers ---------------------------------------
|
|
12
|
+
def fit_cell(est) -> str:
|
|
13
|
+
return c(est.label, ui.VERDICT_STYLE[est.verdict])
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def speed_cell(est) -> str:
|
|
17
|
+
return f"~{est.tok_s:.0f}" if est.tok_s else "-"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def show_candidates(cands, llm):
|
|
21
|
+
ctx = cands[0].est.context if cands else "-"
|
|
22
|
+
ui.info(f"Hardware: {llm.hw.summary()}")
|
|
23
|
+
ui.info("NEEDS = weights + KV cache (auto context) + overhead. tok/s is an estimate (+-30%).")
|
|
24
|
+
rows = []
|
|
25
|
+
for i, cd in enumerate(cands, 1):
|
|
26
|
+
size = f"{cd.size_gb:.1f} GB" + ("" if cd.exact else "~")
|
|
27
|
+
moe = " MoE" if cd.active_b and cd.params_b and cd.active_b < cd.params_b * 0.8 else ""
|
|
28
|
+
rows.append((i, cd.ref + c(moe, "gray"), size, f"{cd.est.needed_gb:.1f} GB", cd.est.context,
|
|
29
|
+
fit_cell(cd.est), speed_cell(cd.est)))
|
|
30
|
+
table(["#", "MODEL", "SIZE", "NEEDS", "CTX", "FIT", "TOK/S"], rows, right_cols=(0, 2, 3, 4, 6))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def show_estimate_detail(cd):
|
|
34
|
+
e = cd.est
|
|
35
|
+
print(f"\n {c(cd.ref, 'bold')} ({'exact' if cd.exact else 'estimated'} size {cd.size_gb:.1f} GB)")
|
|
36
|
+
print(f" Weights {e.weights_gb:.1f} GB + KV cache {e.kv_gb:.1f} GB (ctx {e.context}) + overhead {e.overhead_gb:.1f} GB"
|
|
37
|
+
f" = {c(f'{e.needed_gb:.1f} GB', 'bold')}")
|
|
38
|
+
print(f" Verdict: {fit_cell(e)} Speed: {c(speed_cell(e) + ' tok/s', 'bold')} "
|
|
39
|
+
f"On GPU: {e.gpu_fraction * 100:.0f}% of weights")
|
|
40
|
+
if e.note:
|
|
41
|
+
ui.warn(e.note)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def header(llm: NotebookLLM):
|
|
45
|
+
llm.refresh_hardware()
|
|
46
|
+
title("notebook_llm")
|
|
47
|
+
print(f" {llm.hw.summary()}")
|
|
48
|
+
if llm.hw.gpu.available:
|
|
49
|
+
print(f" VRAM in use: {llm.hw.gpu.used_vram_gb} / {llm.hw.gpu.total_vram_gb} GB")
|
|
50
|
+
elif is_kaggle():
|
|
51
|
+
ui.warn("No GPU. In Kaggle: Settings > Accelerator > GPU T4 x2, then restart.")
|
|
52
|
+
up = llm.server.is_running()
|
|
53
|
+
print(f" Server : {c('running', 'green') if up else c('stopped', 'red')}")
|
|
54
|
+
if up:
|
|
55
|
+
try:
|
|
56
|
+
for m in llm.loaded():
|
|
57
|
+
pct = 100 * m.get("size_vram", 0) / m["size"] if m.get("size") else 0
|
|
58
|
+
print(f" Loaded : {c(m['name'], 'bold')} ({m.get('size', 0) / 1024 ** 3:.1f} GB, {pct:.0f}% on GPU)")
|
|
59
|
+
except Exception:
|
|
60
|
+
pass
|
|
61
|
+
print(f" Model : {llm.settings.model or '(none selected)'}")
|
|
62
|
+
print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else c('closed', 'gray')}")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ---------- flows ----------------------------------------------------
|
|
66
|
+
def ensure_server(llm, model=None) -> bool:
|
|
67
|
+
if llm.server.is_running():
|
|
68
|
+
return True
|
|
69
|
+
print(" Installing/starting Ollama...")
|
|
70
|
+
try:
|
|
71
|
+
llm.start_server(model)
|
|
72
|
+
return True
|
|
73
|
+
except Exception as e:
|
|
74
|
+
ui.err(str(e))
|
|
75
|
+
return False
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def after_load(llm):
|
|
79
|
+
if not llm.url and confirm("Open a public tunnel now?"):
|
|
80
|
+
flow_open_tunnel(llm)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def model_actions(llm, cd):
|
|
84
|
+
show_estimate_detail(cd)
|
|
85
|
+
if cd.est.verdict == "no":
|
|
86
|
+
ui.warn("This model will not run on your hardware.")
|
|
87
|
+
if not confirm("Pull anyway?", False):
|
|
88
|
+
return
|
|
89
|
+
print("\n 1) Pull + load now (make it the active model)\n 2) Pull only\n Enter) back")
|
|
90
|
+
ch = ask("Choose")
|
|
91
|
+
if ch not in ("1", "2"):
|
|
92
|
+
return
|
|
93
|
+
if not ensure_server(llm, cd.ref):
|
|
94
|
+
return
|
|
95
|
+
try:
|
|
96
|
+
if not llm.client.has_model(cd.ref):
|
|
97
|
+
print(f" Pulling {cd.ref} ({cd.size_gb:.1f} GB)...")
|
|
98
|
+
llm.pull(cd.ref)
|
|
99
|
+
if ch == "1":
|
|
100
|
+
print(" Loading into VRAM...")
|
|
101
|
+
llm.load(cd.ref)
|
|
102
|
+
ui.ok(f"{cd.ref} is loaded and will stay in memory.")
|
|
103
|
+
after_load(llm)
|
|
104
|
+
except Exception as e:
|
|
105
|
+
ui.err(str(e))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _result_rows(res, offset=0):
|
|
109
|
+
rows = []
|
|
110
|
+
for i, m in enumerate(res, 1 + offset):
|
|
111
|
+
sizes = ", ".join(m.sizes[:6]) + (" ..." if len(m.sizes) > 6 else "")
|
|
112
|
+
if m.cloud:
|
|
113
|
+
sizes = (sizes + " " if sizes else "") + c("cloud", "yellow")
|
|
114
|
+
desc = m.description.split(". ")[0]
|
|
115
|
+
rows.append((i, c(m.name, "bold"), m.pulls, sizes or "-", ",".join(m.capabilities[:3]),
|
|
116
|
+
(desc[:46] + "...") if len(desc) > 46 else desc))
|
|
117
|
+
return rows
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def flow_search(llm):
|
|
121
|
+
print(" Search ollama.com/search. Examples: coder | /vision | qwen /tools | /newest")
|
|
122
|
+
print(" Filters: /tools /vision /thinking /embedding /cloud /newest /popular. Or paste a model like qwen3:8b")
|
|
123
|
+
q = ask("Search (blank = all models)")
|
|
124
|
+
if q in ("q", "b"):
|
|
125
|
+
return
|
|
126
|
+
if q.startswith("http") and "/library/" in q:
|
|
127
|
+
q = q.split("/library/", 1)[1].strip("/")
|
|
128
|
+
if ":" in q and " " not in q: # direct model reference
|
|
129
|
+
return model_actions(llm, library.make_candidate(q, llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
130
|
+
words = [w for w in q.split() if not w.startswith("/")]
|
|
131
|
+
flags = {w[1:].lower() for w in q.split() if w.startswith("/")}
|
|
132
|
+
cap = next((f for f in ("tools", "vision", "thinking", "embedding", "cloud") if f in flags), None)
|
|
133
|
+
sort = "newest" if "newest" in flags else "popular" if "popular" in flags else None
|
|
134
|
+
query = " ".join(words)
|
|
135
|
+
|
|
136
|
+
shown, page, hidden = [], 1, 0
|
|
137
|
+
while True:
|
|
138
|
+
print(f" Fetching page {page} from ollama.com...")
|
|
139
|
+
try:
|
|
140
|
+
res = library.search(query, page, sort, cap)
|
|
141
|
+
except library.LibraryError as e:
|
|
142
|
+
ui.err(str(e))
|
|
143
|
+
if not shown:
|
|
144
|
+
return
|
|
145
|
+
res = []
|
|
146
|
+
names = {m.name for m in shown}
|
|
147
|
+
res = [m for m in res if m.name not in names]
|
|
148
|
+
if cap != "cloud":
|
|
149
|
+
hidden += sum(1 for m in res if not m.local)
|
|
150
|
+
res = [m for m in res if m.local]
|
|
151
|
+
if not res and not shown:
|
|
152
|
+
ui.warn("No local models found (cloud-only models are hidden; add /cloud to show them).")
|
|
153
|
+
return
|
|
154
|
+
title(f"ollama.com: '{query or 'all models'}'{' /' + cap if cap else ''}{' /' + sort if sort else ''}")
|
|
155
|
+
table(["#", "NAME", "PULLS", "SIZES", "CAPS", "DESCRIPTION"], _result_rows(res, len(shown)), right_cols=(0,))
|
|
156
|
+
shown += res
|
|
157
|
+
if hidden:
|
|
158
|
+
ui.info(f"({hidden} cloud-only models hidden: they run on Ollama's servers, not your GPU)")
|
|
159
|
+
v = ask(f"Open which model (1-{len(shown)}), m = more, Enter = back")
|
|
160
|
+
if v.lower() == "m":
|
|
161
|
+
page += 1
|
|
162
|
+
continue
|
|
163
|
+
if v.isdigit() and 1 <= int(v) <= len(shown):
|
|
164
|
+
return flow_pick_tag(llm, shown[int(v) - 1])
|
|
165
|
+
return
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def flow_pick_tag(llm, m):
|
|
169
|
+
while True:
|
|
170
|
+
print(f"\n Checking sizes for {m.name} (registry lookup)...")
|
|
171
|
+
cands = library.candidates_for(m, llm.hw, llm.settings.context, llm.settings.kv_cache)
|
|
172
|
+
title(m.name)
|
|
173
|
+
show_candidates(cands, llm)
|
|
174
|
+
print(" Type a number, or a custom tag (e.g. 14b-instruct-q8_0). Enter = back.")
|
|
175
|
+
v = ask("Choose")
|
|
176
|
+
if not v or v in ("b", "q"):
|
|
177
|
+
return
|
|
178
|
+
if v.isdigit() and 1 <= int(v) <= len(cands):
|
|
179
|
+
return model_actions(llm, cands[int(v) - 1])
|
|
180
|
+
model_actions(llm, library.make_candidate(f"{m.name}:{v}", llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
181
|
+
return
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def flow_recommended(llm):
|
|
185
|
+
cands = llm.recommended()
|
|
186
|
+
cands = [x for x in cands if x.est.verdict != "no"] or cands
|
|
187
|
+
title("Recommended for your hardware")
|
|
188
|
+
show_candidates(cands, llm)
|
|
189
|
+
i = pick("Select model", len(cands))
|
|
190
|
+
if i is not None:
|
|
191
|
+
model_actions(llm, library.make_candidate(cands[i].ref, llm.hw, llm.settings.context, llm.settings.kv_cache))
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def flow_installed(llm):
|
|
195
|
+
if not ensure_server(llm):
|
|
196
|
+
return
|
|
197
|
+
inst = llm.installed()
|
|
198
|
+
if not inst:
|
|
199
|
+
ui.warn("No models installed yet. Use 'Search' or 'Recommended'.")
|
|
200
|
+
return
|
|
201
|
+
loaded = {m["name"] for m in llm.loaded()}
|
|
202
|
+
rows = []
|
|
203
|
+
for i, m in enumerate(inst, 1):
|
|
204
|
+
cd = llm.candidate(m["name"])
|
|
205
|
+
cd.size_gb = m.get("size", 0) / 1024 ** 3 or cd.size_gb
|
|
206
|
+
mark = c("loaded", "green") if m["name"] in loaded else ""
|
|
207
|
+
active = c("*", "cyan") if m["name"] == llm.settings.model else ""
|
|
208
|
+
rows.append((i, m["name"], f"{cd.size_gb:.1f} GB", fit_cell(cd.est), speed_cell(cd.est), mark, active))
|
|
209
|
+
title("Installed models")
|
|
210
|
+
table(["#", "MODEL", "SIZE", "FIT", "TOK/S", "STATE", ""], rows, right_cols=(0, 2, 4))
|
|
211
|
+
i = pick("Manage which", len(inst))
|
|
212
|
+
if i is None:
|
|
213
|
+
return
|
|
214
|
+
name = inst[i]["name"]
|
|
215
|
+
print(f"\n {c(name, 'bold')}\n 1) Load / make active\n 2) Unload from VRAM\n 3) Benchmark (real tok/s)\n 4) Delete\n Enter) back")
|
|
216
|
+
ch = ask("Choose")
|
|
217
|
+
try:
|
|
218
|
+
if ch == "1":
|
|
219
|
+
llm.load(name)
|
|
220
|
+
ui.ok("Loaded.")
|
|
221
|
+
after_load(llm)
|
|
222
|
+
elif ch == "2":
|
|
223
|
+
llm.unload(name)
|
|
224
|
+
ui.ok("Unloaded.")
|
|
225
|
+
elif ch == "3":
|
|
226
|
+
flow_benchmark(llm, name)
|
|
227
|
+
elif ch == "4" and confirm(f"Delete {name}?", False):
|
|
228
|
+
llm.delete(name)
|
|
229
|
+
ui.ok("Deleted.")
|
|
230
|
+
except Exception as e:
|
|
231
|
+
ui.err(str(e))
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def flow_benchmark(llm, name=None):
|
|
235
|
+
name = name or llm.settings.model
|
|
236
|
+
if not name or not ensure_server(llm):
|
|
237
|
+
ui.warn("Select/load a model first.")
|
|
238
|
+
return
|
|
239
|
+
print(f" Benchmarking {name} (first run includes load time)...")
|
|
240
|
+
try:
|
|
241
|
+
r = llm.benchmark(name)
|
|
242
|
+
except Exception as e:
|
|
243
|
+
ui.err(str(e))
|
|
244
|
+
return
|
|
245
|
+
est = llm.candidate(name).est
|
|
246
|
+
gen = c(f"{r['tok_s']:.1f} tok/s", "bold", "green")
|
|
247
|
+
guess = f" (estimate was ~{est.tok_s:.0f})" if est.tok_s else ""
|
|
248
|
+
print(f" Generation : {gen}{guess}")
|
|
249
|
+
print(f" Prompt eval: {r['prompt_tok_s']:.0f} tok/s Load time: {r['load_s']:.1f}s")
|
|
250
|
+
if "gpu_pct" in r:
|
|
251
|
+
print(f" Placement : {r['gpu_pct']:.0f}% on GPU ({r['loaded_gb']:.1f} GB in memory)")
|
|
252
|
+
if r["gpu_pct"] < 99:
|
|
253
|
+
ui.warn("Model is partly on CPU. Try a smaller model/quant or lower context in Settings.")
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def flow_open_tunnel(llm):
|
|
257
|
+
if not ensure_server(llm):
|
|
258
|
+
return
|
|
259
|
+
try:
|
|
260
|
+
url = llm.open_tunnel()
|
|
261
|
+
except Exception as e:
|
|
262
|
+
ui.err(str(e))
|
|
263
|
+
return
|
|
264
|
+
print("\n " + c(url, "bold", "green"))
|
|
265
|
+
ui.warn("Anyone with this link can use your GPU. It changes every time you reopen the tunnel.")
|
|
266
|
+
if llm.settings.model:
|
|
267
|
+
print("\n Paste into Continue config.yaml:\n")
|
|
268
|
+
print(llm.continue_config())
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def flow_tunnel(llm):
|
|
272
|
+
if not llm.url:
|
|
273
|
+
return flow_open_tunnel(llm)
|
|
274
|
+
print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1) Show Continue config\n 2) Close tunnel\n 3) New tunnel (new link)\n Enter) back")
|
|
275
|
+
ch = ask("Choose")
|
|
276
|
+
if ch == "1":
|
|
277
|
+
print("\n" + llm.continue_config())
|
|
278
|
+
elif ch == "2":
|
|
279
|
+
llm.close_tunnel()
|
|
280
|
+
ui.ok("Tunnel closed.")
|
|
281
|
+
elif ch == "3":
|
|
282
|
+
llm.close_tunnel()
|
|
283
|
+
flow_open_tunnel(llm)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def flow_server(llm):
|
|
287
|
+
up = llm.server.is_running()
|
|
288
|
+
print(f"\n Server is {c('running', 'green') if up else c('stopped', 'red')}\n 1) {'Restart' if up else 'Start'}\n 2) Stop\n 3) Show log tail + GPU report\n Enter) back")
|
|
289
|
+
ch = ask("Choose")
|
|
290
|
+
try:
|
|
291
|
+
if ch == "1":
|
|
292
|
+
llm.restart_server() if up else llm.start_server()
|
|
293
|
+
ui.ok("Server running.")
|
|
294
|
+
elif ch == "2":
|
|
295
|
+
llm.stop_server()
|
|
296
|
+
ui.ok("Stopped.")
|
|
297
|
+
elif ch == "3":
|
|
298
|
+
print(llm.server.gpu_report() or "(no GPU lines in log)")
|
|
299
|
+
print(c("--- log ---", "gray"))
|
|
300
|
+
print(llm.server.log_tail(15))
|
|
301
|
+
except Exception as e:
|
|
302
|
+
ui.err(str(e))
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def flow_settings(llm):
|
|
306
|
+
s = llm.settings
|
|
307
|
+
print(f"\n 1) Context length : {s.context or 'auto'}\n 2) KV cache type : {s.kv_cache} (f16 | q8_0 | q4_0)\n"
|
|
308
|
+
f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n Enter) back")
|
|
309
|
+
ch = ask("Change which")
|
|
310
|
+
if ch == "1":
|
|
311
|
+
v = ask("Context tokens (number, or 'auto')", "auto")
|
|
312
|
+
s.context = None if v == "auto" else int(v) if v.isdigit() else s.context
|
|
313
|
+
elif ch == "2":
|
|
314
|
+
v = ask("KV cache", s.kv_cache)
|
|
315
|
+
s.kv_cache = v if v in ("f16", "q8_0", "q4_0") else s.kv_cache
|
|
316
|
+
elif ch == "3":
|
|
317
|
+
s.flash_attention = confirm("Enable flash attention?", s.flash_attention)
|
|
318
|
+
elif ch == "4":
|
|
319
|
+
s.keep_alive = ask("Keep-alive", s.keep_alive)
|
|
320
|
+
else:
|
|
321
|
+
return
|
|
322
|
+
s.save()
|
|
323
|
+
ui.ok("Saved.")
|
|
324
|
+
if llm.server.is_running() and confirm("Restart server to apply now?"):
|
|
325
|
+
llm.restart_server()
|
|
326
|
+
if s.model:
|
|
327
|
+
llm.load(s.model)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def flow_quickstart(llm):
|
|
331
|
+
c0 = llm.best_fit()
|
|
332
|
+
show_estimate_detail(c0)
|
|
333
|
+
print("\n Quick start = install, start server, pull the best-fitting model, load it, open tunnel.")
|
|
334
|
+
m = ask("Model to use ('auto' = shown above)", "auto")
|
|
335
|
+
try:
|
|
336
|
+
url = llm.quickstart(m)
|
|
337
|
+
except Exception as e:
|
|
338
|
+
ui.err(str(e))
|
|
339
|
+
return
|
|
340
|
+
ui.ok(f"Ready: {llm.settings.model}")
|
|
341
|
+
print("\n " + c(url, "bold", "green"))
|
|
342
|
+
print("\n" + llm.continue_config())
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
MENU = [
|
|
346
|
+
("Quick start (auto: install, model, tunnel)", flow_quickstart),
|
|
347
|
+
("Search Ollama library", flow_search),
|
|
348
|
+
("Recommended models for my GPU", flow_recommended),
|
|
349
|
+
("Installed models (load / benchmark / delete)", flow_installed),
|
|
350
|
+
("Tunnel (open / get link / close)", flow_tunnel),
|
|
351
|
+
("Server (start / stop / logs)", flow_server),
|
|
352
|
+
("Settings (context, KV cache, keep-alive)", flow_settings),
|
|
353
|
+
("Benchmark active model", flow_benchmark),
|
|
354
|
+
]
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def run(argv=None) -> None:
|
|
358
|
+
"""Interactive menu. Call this from a notebook cell: `import notebook_llm; notebook_llm.run()`."""
|
|
359
|
+
llm = NotebookLLM()
|
|
360
|
+
while True:
|
|
361
|
+
header(llm)
|
|
362
|
+
print()
|
|
363
|
+
for i, (label, _) in enumerate(MENU, 1):
|
|
364
|
+
print(f" {c(i, 'cyan')}) {label}")
|
|
365
|
+
print(f" {c('0', 'cyan')}) Quit (server and tunnel keep running)")
|
|
366
|
+
try:
|
|
367
|
+
ch = ask("Choose")
|
|
368
|
+
if ch in ("0", "q", "quit", "exit"):
|
|
369
|
+
return
|
|
370
|
+
if ch.isdigit() and 1 <= int(ch) <= len(MENU):
|
|
371
|
+
MENU[int(ch) - 1][1](llm)
|
|
372
|
+
except KeyboardInterrupt:
|
|
373
|
+
print()
|
|
374
|
+
return
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
# ---------- non-interactive commands (for `!notebook-llm ...`) ----------
|
|
378
|
+
def main(argv=None) -> None:
|
|
379
|
+
p = argparse.ArgumentParser(prog="notebook-llm", description="Run with no arguments for the interactive menu.")
|
|
380
|
+
p.add_argument("--no-color", action="store_true")
|
|
381
|
+
sub = p.add_subparsers(dest="cmd")
|
|
382
|
+
sub.add_parser("gpus", help="show hardware + recommended models")
|
|
383
|
+
s = sub.add_parser("search", help="search the Ollama library")
|
|
384
|
+
s.add_argument("query", nargs="?", default="")
|
|
385
|
+
k = sub.add_parser("check", help="will MODEL fit? estimated tok/s")
|
|
386
|
+
k.add_argument("model")
|
|
387
|
+
k.add_argument("--ctx", type=int)
|
|
388
|
+
q = sub.add_parser("quickstart", help="install, start, pull best model, open tunnel")
|
|
389
|
+
q.add_argument("--model", default="auto")
|
|
390
|
+
q.add_argument("--no-tunnel", action="store_true")
|
|
391
|
+
sub.add_parser("status")
|
|
392
|
+
sub.add_parser("stop", help="stop tunnel and server")
|
|
393
|
+
a = p.parse_args(argv)
|
|
394
|
+
if a.no_color:
|
|
395
|
+
import os
|
|
396
|
+
os.environ["NO_COLOR"] = "1"
|
|
397
|
+
|
|
398
|
+
if a.cmd is None:
|
|
399
|
+
return run()
|
|
400
|
+
llm = NotebookLLM()
|
|
401
|
+
if a.cmd == "gpus":
|
|
402
|
+
print(llm.hw.summary())
|
|
403
|
+
show_candidates(llm.recommended(), llm)
|
|
404
|
+
elif a.cmd == "search":
|
|
405
|
+
for m in library.search(a.query)[:20]:
|
|
406
|
+
print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}")
|
|
407
|
+
elif a.cmd == "check":
|
|
408
|
+
cd = library.make_candidate(a.model, llm.hw, a.ctx, llm.settings.kv_cache)
|
|
409
|
+
show_estimate_detail(cd)
|
|
410
|
+
elif a.cmd == "quickstart":
|
|
411
|
+
url = llm.quickstart(a.model, tunnel=not a.no_tunnel)
|
|
412
|
+
print(url or "")
|
|
413
|
+
print(llm.continue_config())
|
|
414
|
+
elif a.cmd == "status":
|
|
415
|
+
print(f"server={llm.server.is_running()} model={llm.settings.model} tunnel={llm.url}")
|
|
416
|
+
elif a.cmd == "stop":
|
|
417
|
+
llm.close_tunnel()
|
|
418
|
+
llm.stop_server()
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
if __name__ == "__main__":
|
|
422
|
+
main()
|