notebook-llm-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- notebook_llm/__init__.py +12 -0
- notebook_llm/__main__.py +3 -0
- notebook_llm/cli.py +422 -0
- notebook_llm/client.py +83 -0
- notebook_llm/config.py +32 -0
- notebook_llm/env.py +9 -0
- notebook_llm/estimate.py +128 -0
- notebook_llm/estimate_data.py +38 -0
- notebook_llm/gpu.py +127 -0
- notebook_llm/installer.py +67 -0
- notebook_llm/library.py +212 -0
- notebook_llm/manager.py +169 -0
- notebook_llm/server.py +101 -0
- notebook_llm/tunnel.py +63 -0
- notebook_llm/ui.py +71 -0
- notebook_llm_cli-0.1.0.dist-info/METADATA +79 -0
- notebook_llm_cli-0.1.0.dist-info/RECORD +21 -0
- notebook_llm_cli-0.1.0.dist-info/WHEEL +5 -0
- notebook_llm_cli-0.1.0.dist-info/entry_points.txt +3 -0
- notebook_llm_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
- notebook_llm_cli-0.1.0.dist-info/top_level.txt +1 -0
notebook_llm/manager.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""NotebookLLM: the object the CLI (and notebook users) drive."""
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
from dataclasses import asdict, dataclass
|
|
5
|
+
from typing import Dict, List, Optional
|
|
6
|
+
|
|
7
|
+
from . import installer
|
|
8
|
+
from .client import OllamaClient
|
|
9
|
+
from .config import continue_config
|
|
10
|
+
from .env import is_kaggle, log
|
|
11
|
+
from .gpu import detect_hardware, detect_gpus
|
|
12
|
+
from .library import Candidate, make_candidate
|
|
13
|
+
from .server import OllamaServer
|
|
14
|
+
from .tunnel import CloudflareTunnel
|
|
15
|
+
|
|
16
|
+
CONFIG_PATH = os.path.expanduser("~/.notebook_llm/config.json")
|
|
17
|
+
|
|
18
|
+
# Curated list for "recommended for my GPU". Coding models first (they are auto-picked first).
|
|
19
|
+
CURATED = [
|
|
20
|
+
"qwen3-coder:30b", "qwen2.5-coder:32b", "qwen2.5-coder:14b", "qwen2.5-coder:7b",
|
|
21
|
+
"qwen2.5-coder:3b", "qwen2.5-coder:1.5b",
|
|
22
|
+
"gpt-oss:20b", "qwen3:30b", "qwen3:14b", "qwen3:8b", "qwen3:4b",
|
|
23
|
+
"gemma3:27b", "gemma3:12b", "gemma3:4b", "deepseek-r1:32b", "deepseek-r1:14b", "deepseek-r1:8b",
|
|
24
|
+
"phi4:14b", "llama3.1:8b", "llama3.2:3b", "mistral:7b",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Settings:
|
|
30
|
+
model: Optional[str] = None
|
|
31
|
+
context: Optional[int] = None # None = auto from VRAM
|
|
32
|
+
keep_alive: str = "-1"
|
|
33
|
+
flash_attention: bool = True
|
|
34
|
+
kv_cache: str = "q8_0" # f16 | q8_0 | q4_0 (needs flash attention)
|
|
35
|
+
port: int = 11434
|
|
36
|
+
|
|
37
|
+
@classmethod
|
|
38
|
+
def load(cls) -> "Settings":
|
|
39
|
+
try:
|
|
40
|
+
with open(CONFIG_PATH) as f:
|
|
41
|
+
return cls(**{k: v for k, v in json.load(f).items() if k in cls.__dataclass_fields__})
|
|
42
|
+
except (OSError, ValueError):
|
|
43
|
+
return cls()
|
|
44
|
+
|
|
45
|
+
def save(self) -> None:
|
|
46
|
+
os.makedirs(os.path.dirname(CONFIG_PATH), exist_ok=True)
|
|
47
|
+
with open(CONFIG_PATH, "w") as f:
|
|
48
|
+
json.dump(asdict(self), f, indent=2)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class NotebookLLM:
|
|
52
|
+
def __init__(self, settings: Optional[Settings] = None):
|
|
53
|
+
self.settings = settings or Settings.load()
|
|
54
|
+
self.hw = detect_hardware()
|
|
55
|
+
self.client = OllamaClient(f"http://127.0.0.1:{self.settings.port}")
|
|
56
|
+
self.tunnel: Optional[CloudflareTunnel] = CloudflareTunnel.discover(
|
|
57
|
+
installer.cloudflared_path() or "cloudflared", f"http://127.0.0.1:{self.settings.port}")
|
|
58
|
+
self.server = self._make_server()
|
|
59
|
+
|
|
60
|
+
# ---- hardware / estimates -------------------------------------
|
|
61
|
+
def refresh_hardware(self) -> None:
|
|
62
|
+
self.hw.gpu = detect_gpus()
|
|
63
|
+
|
|
64
|
+
def candidate(self, ref: str, online: bool = False) -> Candidate:
|
|
65
|
+
return make_candidate(ref, self.hw, self.settings.context, self.settings.kv_cache, online)
|
|
66
|
+
|
|
67
|
+
def recommended(self) -> List[Candidate]:
|
|
68
|
+
return [self.candidate(r) for r in CURATED]
|
|
69
|
+
|
|
70
|
+
def best_fit(self) -> Candidate:
|
|
71
|
+
"""First curated coder model that fits comfortably; else the smallest one."""
|
|
72
|
+
cands = self.recommended()
|
|
73
|
+
for c in cands[:6]:
|
|
74
|
+
if c.est.verdict == "fits":
|
|
75
|
+
return c
|
|
76
|
+
return cands[5]
|
|
77
|
+
|
|
78
|
+
# ---- server ----------------------------------------------------
|
|
79
|
+
def _make_server(self, model: Optional[str] = None) -> OllamaServer:
|
|
80
|
+
s = self.settings
|
|
81
|
+
ctx = s.context
|
|
82
|
+
ref = model or s.model
|
|
83
|
+
if ctx is None and ref:
|
|
84
|
+
ctx = self.candidate(ref).est.context
|
|
85
|
+
return OllamaServer(port=s.port, gpu_indices=self.hw.gpu.indices, keep_alive=s.keep_alive,
|
|
86
|
+
context_length=ctx, flash_attention=s.flash_attention, kv_cache_type=s.kv_cache)
|
|
87
|
+
|
|
88
|
+
def ensure_installed(self) -> None:
|
|
89
|
+
installer.install_ollama()
|
|
90
|
+
|
|
91
|
+
def start_server(self, model: Optional[str] = None) -> None:
|
|
92
|
+
self.ensure_installed()
|
|
93
|
+
self.server = self._make_server(model)
|
|
94
|
+
self.server.start()
|
|
95
|
+
|
|
96
|
+
def restart_server(self, model: Optional[str] = None) -> None:
|
|
97
|
+
self.server.stop()
|
|
98
|
+
self.start_server(model)
|
|
99
|
+
|
|
100
|
+
def stop_server(self) -> None:
|
|
101
|
+
self.server.stop()
|
|
102
|
+
|
|
103
|
+
# ---- models ----------------------------------------------------
|
|
104
|
+
def installed(self) -> List[Dict]:
|
|
105
|
+
return self.client.list_models() if self.server.is_running() else []
|
|
106
|
+
|
|
107
|
+
def loaded(self) -> List[Dict]:
|
|
108
|
+
return self.client.running() if self.server.is_running() else []
|
|
109
|
+
|
|
110
|
+
def pull(self, ref: str) -> None:
|
|
111
|
+
self.client.pull(ref)
|
|
112
|
+
|
|
113
|
+
def load(self, ref: str) -> None:
|
|
114
|
+
"""Make `ref` the active model. Restarts the server if its context should change."""
|
|
115
|
+
want = self.candidate(ref).est.context
|
|
116
|
+
if self.settings.context is None and self.server.context_length not in (None, want) and self.server.is_running():
|
|
117
|
+
log(f"Restarting server with context {want} for this model...")
|
|
118
|
+
self.restart_server(ref)
|
|
119
|
+
self.client.load(ref, keep_alive=-1 if self.settings.keep_alive == "-1" else self.settings.keep_alive)
|
|
120
|
+
self.settings.model = ref
|
|
121
|
+
self.settings.save()
|
|
122
|
+
|
|
123
|
+
def unload(self, ref: str) -> None:
|
|
124
|
+
self.client.unload(ref)
|
|
125
|
+
|
|
126
|
+
def delete(self, ref: str) -> None:
|
|
127
|
+
self.client.delete(ref)
|
|
128
|
+
if self.settings.model == ref:
|
|
129
|
+
self.settings.model = None
|
|
130
|
+
self.settings.save()
|
|
131
|
+
|
|
132
|
+
def benchmark(self, ref: Optional[str] = None) -> Dict[str, float]:
|
|
133
|
+
return self.client.benchmark(ref or self.settings.model)
|
|
134
|
+
|
|
135
|
+
def chat(self, prompt: str, system: Optional[str] = None) -> str:
|
|
136
|
+
return self.client.chat(self.settings.model, prompt, system)
|
|
137
|
+
|
|
138
|
+
# ---- tunnel ----------------------------------------------------
|
|
139
|
+
@property
|
|
140
|
+
def url(self) -> Optional[str]:
|
|
141
|
+
return self.tunnel.url if self.tunnel else None
|
|
142
|
+
|
|
143
|
+
def open_tunnel(self) -> str:
|
|
144
|
+
if self.tunnel and self.tunnel.url:
|
|
145
|
+
return self.tunnel.url
|
|
146
|
+
self.tunnel = CloudflareTunnel(installer.install_cloudflared(), self.server.url)
|
|
147
|
+
return self.tunnel.start()
|
|
148
|
+
|
|
149
|
+
def close_tunnel(self) -> None:
|
|
150
|
+
if self.tunnel:
|
|
151
|
+
self.tunnel.stop()
|
|
152
|
+
self.tunnel = None
|
|
153
|
+
|
|
154
|
+
def continue_config(self, autocomplete: Optional[str] = None) -> str:
|
|
155
|
+
return continue_config(self.url or self.server.url, self.settings.model or "MODEL", autocomplete)
|
|
156
|
+
|
|
157
|
+
# ---- one call --------------------------------------------------
|
|
158
|
+
def quickstart(self, model: str = "auto", tunnel: bool = True) -> Optional[str]:
|
|
159
|
+
"""Install -> start -> pull -> load -> tunnel. Returns the public URL."""
|
|
160
|
+
self.refresh_hardware()
|
|
161
|
+
log(self.hw.summary())
|
|
162
|
+
c = self.best_fit() if model == "auto" else self.candidate(model)
|
|
163
|
+
log(f"Model: {c.ref} ({c.est.label}, ~{c.est.tok_s:.0f} tok/s est.)" if c.est.tok_s else f"Model: {c.ref}")
|
|
164
|
+
self.settings.model = c.ref
|
|
165
|
+
self.start_server(c.ref)
|
|
166
|
+
if not self.client.has_model(c.ref):
|
|
167
|
+
self.pull(c.ref)
|
|
168
|
+
self.load(c.ref)
|
|
169
|
+
return self.open_tunnel() if tunnel else None
|
notebook_llm/server.py
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Manage the background `ollama serve` process."""
|
|
2
|
+
import os
|
|
3
|
+
import signal
|
|
4
|
+
import subprocess
|
|
5
|
+
import time
|
|
6
|
+
from typing import Dict, List, Optional
|
|
7
|
+
|
|
8
|
+
import requests
|
|
9
|
+
|
|
10
|
+
LOG_PATH = "/tmp/ollama.log"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class OllamaServer:
|
|
14
|
+
def __init__(self, port: int = 11434, gpu_indices: Optional[List[int]] = None,
|
|
15
|
+
keep_alive: str = "-1", context_length: Optional[int] = None,
|
|
16
|
+
flash_attention: bool = True, kv_cache_type: Optional[str] = "q8_0",
|
|
17
|
+
extra_env: Optional[Dict[str, str]] = None, host: str = "127.0.0.1"):
|
|
18
|
+
self.host, self.port = host, port
|
|
19
|
+
self.gpu_indices = gpu_indices or []
|
|
20
|
+
self.keep_alive = keep_alive
|
|
21
|
+
self.context_length = context_length
|
|
22
|
+
self.flash_attention = flash_attention
|
|
23
|
+
self.kv_cache_type = kv_cache_type
|
|
24
|
+
self.extra_env = extra_env or {}
|
|
25
|
+
self.process: Optional[subprocess.Popen] = None
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def url(self) -> str:
|
|
29
|
+
return f"http://{self.host}:{self.port}"
|
|
30
|
+
|
|
31
|
+
def is_running(self) -> bool:
|
|
32
|
+
try:
|
|
33
|
+
return requests.get(f"{self.url}/api/tags", timeout=2).ok
|
|
34
|
+
except requests.RequestException:
|
|
35
|
+
return False
|
|
36
|
+
|
|
37
|
+
def _env(self) -> Dict[str, str]:
|
|
38
|
+
env = os.environ.copy()
|
|
39
|
+
env["OLLAMA_HOST"] = f"{self.host}:{self.port}"
|
|
40
|
+
env["OLLAMA_KEEP_ALIVE"] = self.keep_alive
|
|
41
|
+
if self.gpu_indices:
|
|
42
|
+
env["CUDA_VISIBLE_DEVICES"] = ",".join(map(str, self.gpu_indices))
|
|
43
|
+
if len(self.gpu_indices) > 1:
|
|
44
|
+
env["OLLAMA_SCHED_SPREAD"] = "1"
|
|
45
|
+
if self.context_length:
|
|
46
|
+
env["OLLAMA_CONTEXT_LENGTH"] = str(self.context_length)
|
|
47
|
+
if self.flash_attention:
|
|
48
|
+
env["OLLAMA_FLASH_ATTENTION"] = "1"
|
|
49
|
+
if self.kv_cache_type and self.kv_cache_type != "f16":
|
|
50
|
+
env["OLLAMA_KV_CACHE_TYPE"] = self.kv_cache_type
|
|
51
|
+
env.update(self.extra_env)
|
|
52
|
+
return env
|
|
53
|
+
|
|
54
|
+
def start(self, timeout: int = 90) -> None:
|
|
55
|
+
if self.is_running():
|
|
56
|
+
return
|
|
57
|
+
with open(LOG_PATH, "w") as lf:
|
|
58
|
+
self.process = subprocess.Popen(
|
|
59
|
+
["ollama", "serve"], env=self._env(), stdout=lf, stderr=subprocess.STDOUT,
|
|
60
|
+
stdin=subprocess.DEVNULL, start_new_session=True)
|
|
61
|
+
deadline = time.time() + timeout
|
|
62
|
+
while time.time() < deadline:
|
|
63
|
+
if self.is_running():
|
|
64
|
+
return
|
|
65
|
+
if self.process.poll() is not None:
|
|
66
|
+
raise RuntimeError(f"ollama exited early:\n{self.log_tail()}")
|
|
67
|
+
time.sleep(1)
|
|
68
|
+
raise TimeoutError(f"Ollama not ready after {timeout}s:\n{self.log_tail()}")
|
|
69
|
+
|
|
70
|
+
def stop(self) -> None:
|
|
71
|
+
if self.process and self.process.poll() is None:
|
|
72
|
+
try:
|
|
73
|
+
os.killpg(os.getpgid(self.process.pid), signal.SIGTERM)
|
|
74
|
+
except ProcessLookupError:
|
|
75
|
+
pass
|
|
76
|
+
else: # started by an earlier CLI session
|
|
77
|
+
subprocess.run(["pkill", "-f", "ollama serve"], check=False)
|
|
78
|
+
self.process = None
|
|
79
|
+
for _ in range(15):
|
|
80
|
+
if not self.is_running():
|
|
81
|
+
break
|
|
82
|
+
time.sleep(1)
|
|
83
|
+
|
|
84
|
+
def restart(self) -> None:
|
|
85
|
+
self.stop()
|
|
86
|
+
self.start()
|
|
87
|
+
|
|
88
|
+
def log_tail(self, n: int = 30) -> str:
|
|
89
|
+
try:
|
|
90
|
+
with open(LOG_PATH) as f:
|
|
91
|
+
return "".join(f.readlines()[-n:])
|
|
92
|
+
except FileNotFoundError:
|
|
93
|
+
return ""
|
|
94
|
+
|
|
95
|
+
def gpu_report(self) -> str:
|
|
96
|
+
try:
|
|
97
|
+
with open(LOG_PATH) as f:
|
|
98
|
+
lines = [l.strip() for l in f if any(k in l.lower() for k in ("inference compute", "library=", "cuda"))]
|
|
99
|
+
except FileNotFoundError:
|
|
100
|
+
return ""
|
|
101
|
+
return "\n".join(lines[-6:])
|
notebook_llm/tunnel.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Cloudflare quick tunnel (no account needed)."""
|
|
2
|
+
import os
|
|
3
|
+
import re
|
|
4
|
+
import signal
|
|
5
|
+
import subprocess
|
|
6
|
+
import time
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
LOG_PATH = "/tmp/cloudflared.log"
|
|
10
|
+
_URL_RE = re.compile(r"https://[a-z0-9-]+\.trycloudflare\.com")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class CloudflareTunnel:
|
|
14
|
+
def __init__(self, binary: str, target: str = "http://127.0.0.1:11434"):
|
|
15
|
+
self.binary, self.target = binary, target
|
|
16
|
+
self.process: Optional[subprocess.Popen] = None
|
|
17
|
+
self.url: Optional[str] = None
|
|
18
|
+
|
|
19
|
+
@classmethod
|
|
20
|
+
def discover(cls, binary: str = "cloudflared", target: str = "http://127.0.0.1:11434"):
|
|
21
|
+
"""Re-attach to a tunnel started by an earlier session."""
|
|
22
|
+
r = subprocess.run(["pgrep", "-f", "cloudflared.*tunnel"], capture_output=True, text=True)
|
|
23
|
+
if r.returncode != 0:
|
|
24
|
+
return None
|
|
25
|
+
try:
|
|
26
|
+
m = _URL_RE.search(open(LOG_PATH).read())
|
|
27
|
+
except FileNotFoundError:
|
|
28
|
+
return None
|
|
29
|
+
if not m:
|
|
30
|
+
return None
|
|
31
|
+
t = cls(binary, target)
|
|
32
|
+
t.url = m.group(0)
|
|
33
|
+
return t
|
|
34
|
+
|
|
35
|
+
def start(self, timeout: int = 45) -> str:
|
|
36
|
+
host = self.target.split("://", 1)[-1]
|
|
37
|
+
with open(LOG_PATH, "w") as lf:
|
|
38
|
+
self.process = subprocess.Popen(
|
|
39
|
+
[self.binary, "tunnel", "--no-autoupdate", "--url", self.target, "--http-host-header", host],
|
|
40
|
+
stdout=lf, stderr=subprocess.STDOUT, stdin=subprocess.DEVNULL, start_new_session=True)
|
|
41
|
+
deadline = time.time() + timeout
|
|
42
|
+
while time.time() < deadline:
|
|
43
|
+
time.sleep(1)
|
|
44
|
+
try:
|
|
45
|
+
m = _URL_RE.search(open(LOG_PATH).read())
|
|
46
|
+
except FileNotFoundError:
|
|
47
|
+
m = None
|
|
48
|
+
if m:
|
|
49
|
+
self.url = m.group(0)
|
|
50
|
+
return self.url
|
|
51
|
+
if self.process.poll() is not None:
|
|
52
|
+
break
|
|
53
|
+
raise RuntimeError("Could not get a tunnel URL. Internet enabled? Log:\n" + open(LOG_PATH).read()[-1500:])
|
|
54
|
+
|
|
55
|
+
def stop(self) -> None:
|
|
56
|
+
if self.process and self.process.poll() is None:
|
|
57
|
+
try:
|
|
58
|
+
os.killpg(os.getpgid(self.process.pid), signal.SIGTERM)
|
|
59
|
+
except ProcessLookupError:
|
|
60
|
+
pass
|
|
61
|
+
else:
|
|
62
|
+
subprocess.run(["pkill", "-f", "cloudflared.*tunnel"], check=False)
|
|
63
|
+
self.process, self.url = None, None
|
notebook_llm/ui.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Tiny terminal UI helpers (no dependencies). Works in Kaggle/Jupyter cells too."""
|
|
2
|
+
import os
|
|
3
|
+
import re
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
_ANSI = re.compile(r"\x1b\[[0-9;]*m")
|
|
7
|
+
CODES = {"red": "31", "green": "32", "yellow": "33", "blue": "34", "magenta": "35",
|
|
8
|
+
"cyan": "36", "gray": "37", "bold": "1", "dim": "2"}
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def color_on() -> bool:
|
|
12
|
+
if os.environ.get("NO_COLOR"):
|
|
13
|
+
return False
|
|
14
|
+
return sys.stdout.isatty() or "ipykernel" in sys.modules
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def c(text, *styles) -> str:
|
|
18
|
+
if not color_on() or not styles:
|
|
19
|
+
return str(text)
|
|
20
|
+
return "".join(f"\x1b[{CODES[s]}m" for s in styles) + str(text) + "\x1b[0m"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def vlen(s: str) -> int:
|
|
24
|
+
return len(_ANSI.sub("", s))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def pad(s: str, w: int, right: bool = False) -> str:
|
|
28
|
+
gap = " " * max(0, w - vlen(s))
|
|
29
|
+
return gap + s if right else s + gap
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def table(headers, rows, right_cols=()):
|
|
33
|
+
widths = [max(vlen(str(x)) for x in col) for col in zip(headers, *rows)] if rows else [len(h) for h in headers]
|
|
34
|
+
line = " ".join(pad(c(h, "bold"), widths[i], i in right_cols) for i, h in enumerate(headers))
|
|
35
|
+
print(" " + line)
|
|
36
|
+
for r in rows:
|
|
37
|
+
print(" " + " ".join(pad(str(x), widths[i], i in right_cols) for i, x in enumerate(r)))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def title(text: str) -> None:
|
|
41
|
+
print("\n" + c(f"== {text} ==", "bold", "cyan"))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def info(msg): print(c(msg, "gray"))
|
|
45
|
+
def ok(msg): print(c("OK ", "green") + msg)
|
|
46
|
+
def warn(msg): print(c("! ", "yellow") + msg)
|
|
47
|
+
def err(msg): print(c("ERR ", "red") + msg)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def ask(prompt: str, default: str = "") -> str:
|
|
51
|
+
try:
|
|
52
|
+
v = input(f"> {prompt}{f' [{default}]' if default else ''}: ").strip()
|
|
53
|
+
except EOFError:
|
|
54
|
+
return "q"
|
|
55
|
+
return v or default
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def confirm(prompt: str, default: bool = True) -> bool:
|
|
59
|
+
v = ask(f"{prompt} ({'Y/n' if default else 'y/N'})").lower()
|
|
60
|
+
return default if not v else v.startswith("y")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def pick(prompt: str, n: int):
|
|
64
|
+
"""Returns 0-based index, or None for blank/back."""
|
|
65
|
+
v = ask(f"{prompt} (1-{n}, Enter = back)")
|
|
66
|
+
if v.isdigit() and 1 <= int(v) <= n:
|
|
67
|
+
return int(v) - 1
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
VERDICT_STYLE = {"fits": "green", "tight": "yellow", "offload": "yellow", "no": "red", "cpu": "yellow"}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: notebook-llm-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Interactive CLI to run Ollama LLMs on Kaggle/Colab GPUs: GPU detection, fit and tokens/sec estimates, library search, public tunnel.
|
|
5
|
+
Author-email: Shashan Lumbhani <lumbhanishashan1510@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/soni-shashan/notebook-llm
|
|
8
|
+
Project-URL: Issues, https://github.com/soni-shashan/notebook-llm/issues
|
|
9
|
+
Keywords: ollama,llm,kaggle,colab,gpu,cloudflare-tunnel,cli
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Requires-Python: >=3.8
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: requests>=2.25
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
# notebook_llm
|
|
24
|
+
|
|
25
|
+
One interactive CLI to run LLMs on a Kaggle/Colab GPU with Ollama: detect GPUs, check whether a model fits, estimate tokens/sec, search the Ollama library, pull, load, benchmark, and open a public tunnel link.
|
|
26
|
+
|
|
27
|
+
## Install (Kaggle notebook: Internet ON, Accelerator = GPU)
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
!pip install -q notebook-llm
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
From a local folder (copy it to a writable place first, /kaggle/input is read-only):
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
!pip install -q ./notebook_llm # or: pip install notebook_llm
|
|
37
|
+
import notebook_llm
|
|
38
|
+
notebook_llm.run() # opens the interactive menu (input boxes appear in the cell)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
In a real terminal just run `notebook-llm`.
|
|
42
|
+
Note: `!notebook-llm` cannot take keyboard input in Kaggle, so use `notebook_llm.run()` there.
|
|
43
|
+
|
|
44
|
+
## Menu
|
|
45
|
+
1. Quick start - installs everything, picks the best model that fits, loads it, opens the tunnel
|
|
46
|
+
2. Search Ollama library - live search of ollama.com/search (paged with `m`, filters like `/tools /vision /thinking /embedding /newest`, or paste `model:tag` / a library URL). Cloud-only models are hidden because they don't run on your GPU (`/cloud` shows them). Pick a model to see FIT verdict and ~tok/s for every size, then pull/load
|
|
47
|
+
3. Recommended models for my GPU
|
|
48
|
+
4. Installed models - load / unload / benchmark (real tok/s) / delete
|
|
49
|
+
5. Tunnel - open, show link, Continue config, close, new link
|
|
50
|
+
6. Server - start / stop / restart / logs / GPU report
|
|
51
|
+
7. Settings - context length, KV cache type, flash attention, keep-alive (saved)
|
|
52
|
+
8. Benchmark the active model
|
|
53
|
+
|
|
54
|
+
## Non-interactive
|
|
55
|
+
```
|
|
56
|
+
notebook-llm gpus # hardware + recommended table
|
|
57
|
+
notebook-llm search coder
|
|
58
|
+
notebook-llm check llama3.1:70b --ctx 8192
|
|
59
|
+
notebook-llm quickstart --model auto
|
|
60
|
+
notebook-llm status | stop
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Python API
|
|
64
|
+
```python
|
|
65
|
+
from notebook_llm import NotebookLLM
|
|
66
|
+
llm = NotebookLLM()
|
|
67
|
+
url = llm.quickstart() # install -> serve -> pull -> load -> tunnel
|
|
68
|
+
llm.chat("hello"); llm.benchmark(); llm.close_tunnel(); llm.stop_server()
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## How the estimates work
|
|
72
|
+
- **Needs** = weights (exact size from registry.ollama.ai, else params x bytes/param) + KV cache (scales with context and KV type) + ~0.7 GB per GPU.
|
|
73
|
+
- **FIT**: FITS (<=92% of total VRAM), TIGHT (<=100%), SLOW (spills to RAM), TOO BIG.
|
|
74
|
+
- **tok/s** = memory bandwidth / bytes read per token, using a built-in GPU bandwidth table (T4, P100, V100, A100, L4, RTX...). MoE models only read their active experts (e.g. qwen3-coder:30b ~3.3B active) so they are much faster than dense models of the same size. Multi-GPU is layer-split, so it adds capacity, not speed.
|
|
75
|
+
- Estimates are +-30%. Use Benchmark for the real number.
|
|
76
|
+
|
|
77
|
+
## Notes
|
|
78
|
+
- The tunnel link has no authentication; anyone with it can use your GPU.
|
|
79
|
+
- NVIDIA GPUs only. Library search scrapes ollama.com; if it is unreachable a built-in list is used.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
notebook_llm/__init__.py,sha256=KbK3VgAduO-E0kbCD703YO3ioCaqLl6vJEijKi69hXM,622
|
|
2
|
+
notebook_llm/__main__.py,sha256=bYt9eEaoRQWdejEHFD8REx9jxVEdZptECFsV7F49Ink,30
|
|
3
|
+
notebook_llm/cli.py,sha256=C8r7uIY6aaEseoKFZ6hxHH4b2MojcsRaoY-QsGf0_mE,16297
|
|
4
|
+
notebook_llm/client.py,sha256=IbFhh5LfSkdGWyq37tjdm1g4j_b-hURsQBIKXW31crk,3811
|
|
5
|
+
notebook_llm/config.py,sha256=kahnWiZEJYG6kkFO_c5ZBI3MPnRV6wxs0cxH34vhc0I,716
|
|
6
|
+
notebook_llm/env.py,sha256=uwPTr9yx08V1ii9y9gvdbMzH9fuHzr7JrdHBAyePu5U,199
|
|
7
|
+
notebook_llm/estimate.py,sha256=Pobt0BpjaWfCYMk5nXjPFf6TLmsqOIDlqZa64PiqyuI,4731
|
|
8
|
+
notebook_llm/estimate_data.py,sha256=Cj-k7xOPFXwGqrcW37oEzbB1EzHWmpXmn8IuwhpnoPE,1689
|
|
9
|
+
notebook_llm/gpu.py,sha256=iL3b6TJdkhssN3MAHMwT1fwMMtR4b5oOqYOFzhzRDf4,3477
|
|
10
|
+
notebook_llm/installer.py,sha256=W-rYf6algJXiL_1gQAZ0rivMIpnXzAa6JRQecZcg2HY,2233
|
|
11
|
+
notebook_llm/library.py,sha256=bIe8SIvRwYyyf6uONkU2kraOYV18W6aqvCbspE3dNg0,8700
|
|
12
|
+
notebook_llm/manager.py,sha256=QAukry6f2L9Rm4mpVBOQjteIjbW0T9OFKeAReQ-_W9I,6777
|
|
13
|
+
notebook_llm/server.py,sha256=SMcLD4glFTnVpewi6tEVwoeVnTg3v9E7BHFmbyNKwFE,3715
|
|
14
|
+
notebook_llm/tunnel.py,sha256=EcNc48Fuyzq1SBDXdpGjZIJ5tywJAeMNwYQ061mHqAU,2345
|
|
15
|
+
notebook_llm/ui.py,sha256=v9T_MJDG6kqqTS4Roqv-1aiOKWLVqgmD6FnDddb-Nsg,2170
|
|
16
|
+
notebook_llm_cli-0.1.0.dist-info/licenses/LICENSE,sha256=X2BYxFa5xVFrpM5_P2bdUjsQWaJQ3cwsAxI8QbTzQjc,1073
|
|
17
|
+
notebook_llm_cli-0.1.0.dist-info/METADATA,sha256=UWrYctFLqc9HXen0MIKO3vZ7MSZt9xxzSfCYJuoVaO8,3788
|
|
18
|
+
notebook_llm_cli-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
19
|
+
notebook_llm_cli-0.1.0.dist-info/entry_points.txt,sha256=YQ3lVVfGIlLaO8Tu0oKpmLYJAieGZjz0KxYNO2Jiys0,92
|
|
20
|
+
notebook_llm_cli-0.1.0.dist-info/top_level.txt,sha256=XGp57YPO1z_LAIcu660qjYFHMaw_LEsfaQM9xmdNPbs,13
|
|
21
|
+
notebook_llm_cli-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Shashan Lumbhani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
notebook_llm
|