notebook-llm-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ """notebook_llm: manage Ollama LLMs on notebook GPUs (Kaggle/Colab) from one interactive CLI."""
2
+ from .gpu import GPU, GPUInfo, Hardware, detect_gpus, detect_hardware
3
+ from .estimate import Estimate, estimate
4
+ from .library import LibModel, Candidate, search, make_candidate
5
+ from .manager import NotebookLLM, Settings
6
+ from .cli import run, main
7
+ from .env import is_kaggle
8
+
9
+ __all__ = ["NotebookLLM", "Settings", "run", "main", "GPU", "GPUInfo", "Hardware",
10
+ "detect_gpus", "detect_hardware", "Estimate", "estimate", "LibModel",
11
+ "Candidate", "search", "make_candidate", "is_kaggle"]
12
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
notebook_llm/cli.py ADDED
@@ -0,0 +1,422 @@
1
+ """Interactive CLI. Run `notebook-llm` (terminal) or `notebook_llm.run()` (notebook cell)."""
2
+ import argparse
3
+ import sys
4
+
5
+ from . import library, ui
6
+ from .env import is_kaggle
7
+ from .manager import NotebookLLM
8
+ from .ui import c, ask, confirm, pick, table, title
9
+
10
+
11
+ # ---------- rendering helpers ---------------------------------------
12
+ def fit_cell(est) -> str:
13
+ return c(est.label, ui.VERDICT_STYLE[est.verdict])
14
+
15
+
16
+ def speed_cell(est) -> str:
17
+ return f"~{est.tok_s:.0f}" if est.tok_s else "-"
18
+
19
+
20
+ def show_candidates(cands, llm):
21
+ ctx = cands[0].est.context if cands else "-"
22
+ ui.info(f"Hardware: {llm.hw.summary()}")
23
+ ui.info("NEEDS = weights + KV cache (auto context) + overhead. tok/s is an estimate (+-30%).")
24
+ rows = []
25
+ for i, cd in enumerate(cands, 1):
26
+ size = f"{cd.size_gb:.1f} GB" + ("" if cd.exact else "~")
27
+ moe = " MoE" if cd.active_b and cd.params_b and cd.active_b < cd.params_b * 0.8 else ""
28
+ rows.append((i, cd.ref + c(moe, "gray"), size, f"{cd.est.needed_gb:.1f} GB", cd.est.context,
29
+ fit_cell(cd.est), speed_cell(cd.est)))
30
+ table(["#", "MODEL", "SIZE", "NEEDS", "CTX", "FIT", "TOK/S"], rows, right_cols=(0, 2, 3, 4, 6))
31
+
32
+
33
+ def show_estimate_detail(cd):
34
+ e = cd.est
35
+ print(f"\n {c(cd.ref, 'bold')} ({'exact' if cd.exact else 'estimated'} size {cd.size_gb:.1f} GB)")
36
+ print(f" Weights {e.weights_gb:.1f} GB + KV cache {e.kv_gb:.1f} GB (ctx {e.context}) + overhead {e.overhead_gb:.1f} GB"
37
+ f" = {c(f'{e.needed_gb:.1f} GB', 'bold')}")
38
+ print(f" Verdict: {fit_cell(e)} Speed: {c(speed_cell(e) + ' tok/s', 'bold')} "
39
+ f"On GPU: {e.gpu_fraction * 100:.0f}% of weights")
40
+ if e.note:
41
+ ui.warn(e.note)
42
+
43
+
44
+ def header(llm: NotebookLLM):
45
+ llm.refresh_hardware()
46
+ title("notebook_llm")
47
+ print(f" {llm.hw.summary()}")
48
+ if llm.hw.gpu.available:
49
+ print(f" VRAM in use: {llm.hw.gpu.used_vram_gb} / {llm.hw.gpu.total_vram_gb} GB")
50
+ elif is_kaggle():
51
+ ui.warn("No GPU. In Kaggle: Settings > Accelerator > GPU T4 x2, then restart.")
52
+ up = llm.server.is_running()
53
+ print(f" Server : {c('running', 'green') if up else c('stopped', 'red')}")
54
+ if up:
55
+ try:
56
+ for m in llm.loaded():
57
+ pct = 100 * m.get("size_vram", 0) / m["size"] if m.get("size") else 0
58
+ print(f" Loaded : {c(m['name'], 'bold')} ({m.get('size', 0) / 1024 ** 3:.1f} GB, {pct:.0f}% on GPU)")
59
+ except Exception:
60
+ pass
61
+ print(f" Model : {llm.settings.model or '(none selected)'}")
62
+ print(f" Tunnel : {c(llm.url, 'bold', 'green') if llm.url else c('closed', 'gray')}")
63
+
64
+
65
+ # ---------- flows ----------------------------------------------------
66
+ def ensure_server(llm, model=None) -> bool:
67
+ if llm.server.is_running():
68
+ return True
69
+ print(" Installing/starting Ollama...")
70
+ try:
71
+ llm.start_server(model)
72
+ return True
73
+ except Exception as e:
74
+ ui.err(str(e))
75
+ return False
76
+
77
+
78
+ def after_load(llm):
79
+ if not llm.url and confirm("Open a public tunnel now?"):
80
+ flow_open_tunnel(llm)
81
+
82
+
83
+ def model_actions(llm, cd):
84
+ show_estimate_detail(cd)
85
+ if cd.est.verdict == "no":
86
+ ui.warn("This model will not run on your hardware.")
87
+ if not confirm("Pull anyway?", False):
88
+ return
89
+ print("\n 1) Pull + load now (make it the active model)\n 2) Pull only\n Enter) back")
90
+ ch = ask("Choose")
91
+ if ch not in ("1", "2"):
92
+ return
93
+ if not ensure_server(llm, cd.ref):
94
+ return
95
+ try:
96
+ if not llm.client.has_model(cd.ref):
97
+ print(f" Pulling {cd.ref} ({cd.size_gb:.1f} GB)...")
98
+ llm.pull(cd.ref)
99
+ if ch == "1":
100
+ print(" Loading into VRAM...")
101
+ llm.load(cd.ref)
102
+ ui.ok(f"{cd.ref} is loaded and will stay in memory.")
103
+ after_load(llm)
104
+ except Exception as e:
105
+ ui.err(str(e))
106
+
107
+
108
+ def _result_rows(res, offset=0):
109
+ rows = []
110
+ for i, m in enumerate(res, 1 + offset):
111
+ sizes = ", ".join(m.sizes[:6]) + (" ..." if len(m.sizes) > 6 else "")
112
+ if m.cloud:
113
+ sizes = (sizes + " " if sizes else "") + c("cloud", "yellow")
114
+ desc = m.description.split(". ")[0]
115
+ rows.append((i, c(m.name, "bold"), m.pulls, sizes or "-", ",".join(m.capabilities[:3]),
116
+ (desc[:46] + "...") if len(desc) > 46 else desc))
117
+ return rows
118
+
119
+
120
+ def flow_search(llm):
121
+ print(" Search ollama.com/search. Examples: coder | /vision | qwen /tools | /newest")
122
+ print(" Filters: /tools /vision /thinking /embedding /cloud /newest /popular. Or paste a model like qwen3:8b")
123
+ q = ask("Search (blank = all models)")
124
+ if q in ("q", "b"):
125
+ return
126
+ if q.startswith("http") and "/library/" in q:
127
+ q = q.split("/library/", 1)[1].strip("/")
128
+ if ":" in q and " " not in q: # direct model reference
129
+ return model_actions(llm, library.make_candidate(q, llm.hw, llm.settings.context, llm.settings.kv_cache))
130
+ words = [w for w in q.split() if not w.startswith("/")]
131
+ flags = {w[1:].lower() for w in q.split() if w.startswith("/")}
132
+ cap = next((f for f in ("tools", "vision", "thinking", "embedding", "cloud") if f in flags), None)
133
+ sort = "newest" if "newest" in flags else "popular" if "popular" in flags else None
134
+ query = " ".join(words)
135
+
136
+ shown, page, hidden = [], 1, 0
137
+ while True:
138
+ print(f" Fetching page {page} from ollama.com...")
139
+ try:
140
+ res = library.search(query, page, sort, cap)
141
+ except library.LibraryError as e:
142
+ ui.err(str(e))
143
+ if not shown:
144
+ return
145
+ res = []
146
+ names = {m.name for m in shown}
147
+ res = [m for m in res if m.name not in names]
148
+ if cap != "cloud":
149
+ hidden += sum(1 for m in res if not m.local)
150
+ res = [m for m in res if m.local]
151
+ if not res and not shown:
152
+ ui.warn("No local models found (cloud-only models are hidden; add /cloud to show them).")
153
+ return
154
+ title(f"ollama.com: '{query or 'all models'}'{' /' + cap if cap else ''}{' /' + sort if sort else ''}")
155
+ table(["#", "NAME", "PULLS", "SIZES", "CAPS", "DESCRIPTION"], _result_rows(res, len(shown)), right_cols=(0,))
156
+ shown += res
157
+ if hidden:
158
+ ui.info(f"({hidden} cloud-only models hidden: they run on Ollama's servers, not your GPU)")
159
+ v = ask(f"Open which model (1-{len(shown)}), m = more, Enter = back")
160
+ if v.lower() == "m":
161
+ page += 1
162
+ continue
163
+ if v.isdigit() and 1 <= int(v) <= len(shown):
164
+ return flow_pick_tag(llm, shown[int(v) - 1])
165
+ return
166
+
167
+
168
+ def flow_pick_tag(llm, m):
169
+ while True:
170
+ print(f"\n Checking sizes for {m.name} (registry lookup)...")
171
+ cands = library.candidates_for(m, llm.hw, llm.settings.context, llm.settings.kv_cache)
172
+ title(m.name)
173
+ show_candidates(cands, llm)
174
+ print(" Type a number, or a custom tag (e.g. 14b-instruct-q8_0). Enter = back.")
175
+ v = ask("Choose")
176
+ if not v or v in ("b", "q"):
177
+ return
178
+ if v.isdigit() and 1 <= int(v) <= len(cands):
179
+ return model_actions(llm, cands[int(v) - 1])
180
+ model_actions(llm, library.make_candidate(f"{m.name}:{v}", llm.hw, llm.settings.context, llm.settings.kv_cache))
181
+ return
182
+
183
+
184
+ def flow_recommended(llm):
185
+ cands = llm.recommended()
186
+ cands = [x for x in cands if x.est.verdict != "no"] or cands
187
+ title("Recommended for your hardware")
188
+ show_candidates(cands, llm)
189
+ i = pick("Select model", len(cands))
190
+ if i is not None:
191
+ model_actions(llm, library.make_candidate(cands[i].ref, llm.hw, llm.settings.context, llm.settings.kv_cache))
192
+
193
+
194
+ def flow_installed(llm):
195
+ if not ensure_server(llm):
196
+ return
197
+ inst = llm.installed()
198
+ if not inst:
199
+ ui.warn("No models installed yet. Use 'Search' or 'Recommended'.")
200
+ return
201
+ loaded = {m["name"] for m in llm.loaded()}
202
+ rows = []
203
+ for i, m in enumerate(inst, 1):
204
+ cd = llm.candidate(m["name"])
205
+ cd.size_gb = m.get("size", 0) / 1024 ** 3 or cd.size_gb
206
+ mark = c("loaded", "green") if m["name"] in loaded else ""
207
+ active = c("*", "cyan") if m["name"] == llm.settings.model else ""
208
+ rows.append((i, m["name"], f"{cd.size_gb:.1f} GB", fit_cell(cd.est), speed_cell(cd.est), mark, active))
209
+ title("Installed models")
210
+ table(["#", "MODEL", "SIZE", "FIT", "TOK/S", "STATE", ""], rows, right_cols=(0, 2, 4))
211
+ i = pick("Manage which", len(inst))
212
+ if i is None:
213
+ return
214
+ name = inst[i]["name"]
215
+ print(f"\n {c(name, 'bold')}\n 1) Load / make active\n 2) Unload from VRAM\n 3) Benchmark (real tok/s)\n 4) Delete\n Enter) back")
216
+ ch = ask("Choose")
217
+ try:
218
+ if ch == "1":
219
+ llm.load(name)
220
+ ui.ok("Loaded.")
221
+ after_load(llm)
222
+ elif ch == "2":
223
+ llm.unload(name)
224
+ ui.ok("Unloaded.")
225
+ elif ch == "3":
226
+ flow_benchmark(llm, name)
227
+ elif ch == "4" and confirm(f"Delete {name}?", False):
228
+ llm.delete(name)
229
+ ui.ok("Deleted.")
230
+ except Exception as e:
231
+ ui.err(str(e))
232
+
233
+
234
+ def flow_benchmark(llm, name=None):
235
+ name = name or llm.settings.model
236
+ if not name or not ensure_server(llm):
237
+ ui.warn("Select/load a model first.")
238
+ return
239
+ print(f" Benchmarking {name} (first run includes load time)...")
240
+ try:
241
+ r = llm.benchmark(name)
242
+ except Exception as e:
243
+ ui.err(str(e))
244
+ return
245
+ est = llm.candidate(name).est
246
+ gen = c(f"{r['tok_s']:.1f} tok/s", "bold", "green")
247
+ guess = f" (estimate was ~{est.tok_s:.0f})" if est.tok_s else ""
248
+ print(f" Generation : {gen}{guess}")
249
+ print(f" Prompt eval: {r['prompt_tok_s']:.0f} tok/s Load time: {r['load_s']:.1f}s")
250
+ if "gpu_pct" in r:
251
+ print(f" Placement : {r['gpu_pct']:.0f}% on GPU ({r['loaded_gb']:.1f} GB in memory)")
252
+ if r["gpu_pct"] < 99:
253
+ ui.warn("Model is partly on CPU. Try a smaller model/quant or lower context in Settings.")
254
+
255
+
256
+ def flow_open_tunnel(llm):
257
+ if not ensure_server(llm):
258
+ return
259
+ try:
260
+ url = llm.open_tunnel()
261
+ except Exception as e:
262
+ ui.err(str(e))
263
+ return
264
+ print("\n " + c(url, "bold", "green"))
265
+ ui.warn("Anyone with this link can use your GPU. It changes every time you reopen the tunnel.")
266
+ if llm.settings.model:
267
+ print("\n Paste into Continue config.yaml:\n")
268
+ print(llm.continue_config())
269
+
270
+
271
+ def flow_tunnel(llm):
272
+ if not llm.url:
273
+ return flow_open_tunnel(llm)
274
+ print(f"\n Tunnel: {c(llm.url, 'bold', 'green')}\n 1) Show Continue config\n 2) Close tunnel\n 3) New tunnel (new link)\n Enter) back")
275
+ ch = ask("Choose")
276
+ if ch == "1":
277
+ print("\n" + llm.continue_config())
278
+ elif ch == "2":
279
+ llm.close_tunnel()
280
+ ui.ok("Tunnel closed.")
281
+ elif ch == "3":
282
+ llm.close_tunnel()
283
+ flow_open_tunnel(llm)
284
+
285
+
286
+ def flow_server(llm):
287
+ up = llm.server.is_running()
288
+ print(f"\n Server is {c('running', 'green') if up else c('stopped', 'red')}\n 1) {'Restart' if up else 'Start'}\n 2) Stop\n 3) Show log tail + GPU report\n Enter) back")
289
+ ch = ask("Choose")
290
+ try:
291
+ if ch == "1":
292
+ llm.restart_server() if up else llm.start_server()
293
+ ui.ok("Server running.")
294
+ elif ch == "2":
295
+ llm.stop_server()
296
+ ui.ok("Stopped.")
297
+ elif ch == "3":
298
+ print(llm.server.gpu_report() or "(no GPU lines in log)")
299
+ print(c("--- log ---", "gray"))
300
+ print(llm.server.log_tail(15))
301
+ except Exception as e:
302
+ ui.err(str(e))
303
+
304
+
305
+ def flow_settings(llm):
306
+ s = llm.settings
307
+ print(f"\n 1) Context length : {s.context or 'auto'}\n 2) KV cache type : {s.kv_cache} (f16 | q8_0 | q4_0)\n"
308
+ f" 3) Flash attention: {s.flash_attention}\n 4) Keep-alive : {s.keep_alive} (-1 = forever, or e.g. 30m)\n Enter) back")
309
+ ch = ask("Change which")
310
+ if ch == "1":
311
+ v = ask("Context tokens (number, or 'auto')", "auto")
312
+ s.context = None if v == "auto" else int(v) if v.isdigit() else s.context
313
+ elif ch == "2":
314
+ v = ask("KV cache", s.kv_cache)
315
+ s.kv_cache = v if v in ("f16", "q8_0", "q4_0") else s.kv_cache
316
+ elif ch == "3":
317
+ s.flash_attention = confirm("Enable flash attention?", s.flash_attention)
318
+ elif ch == "4":
319
+ s.keep_alive = ask("Keep-alive", s.keep_alive)
320
+ else:
321
+ return
322
+ s.save()
323
+ ui.ok("Saved.")
324
+ if llm.server.is_running() and confirm("Restart server to apply now?"):
325
+ llm.restart_server()
326
+ if s.model:
327
+ llm.load(s.model)
328
+
329
+
330
+ def flow_quickstart(llm):
331
+ c0 = llm.best_fit()
332
+ show_estimate_detail(c0)
333
+ print("\n Quick start = install, start server, pull the best-fitting model, load it, open tunnel.")
334
+ m = ask("Model to use ('auto' = shown above)", "auto")
335
+ try:
336
+ url = llm.quickstart(m)
337
+ except Exception as e:
338
+ ui.err(str(e))
339
+ return
340
+ ui.ok(f"Ready: {llm.settings.model}")
341
+ print("\n " + c(url, "bold", "green"))
342
+ print("\n" + llm.continue_config())
343
+
344
+
345
+ MENU = [
346
+ ("Quick start (auto: install, model, tunnel)", flow_quickstart),
347
+ ("Search Ollama library", flow_search),
348
+ ("Recommended models for my GPU", flow_recommended),
349
+ ("Installed models (load / benchmark / delete)", flow_installed),
350
+ ("Tunnel (open / get link / close)", flow_tunnel),
351
+ ("Server (start / stop / logs)", flow_server),
352
+ ("Settings (context, KV cache, keep-alive)", flow_settings),
353
+ ("Benchmark active model", flow_benchmark),
354
+ ]
355
+
356
+
357
+ def run(argv=None) -> None:
358
+ """Interactive menu. Call this from a notebook cell: `import notebook_llm; notebook_llm.run()`."""
359
+ llm = NotebookLLM()
360
+ while True:
361
+ header(llm)
362
+ print()
363
+ for i, (label, _) in enumerate(MENU, 1):
364
+ print(f" {c(i, 'cyan')}) {label}")
365
+ print(f" {c('0', 'cyan')}) Quit (server and tunnel keep running)")
366
+ try:
367
+ ch = ask("Choose")
368
+ if ch in ("0", "q", "quit", "exit"):
369
+ return
370
+ if ch.isdigit() and 1 <= int(ch) <= len(MENU):
371
+ MENU[int(ch) - 1][1](llm)
372
+ except KeyboardInterrupt:
373
+ print()
374
+ return
375
+
376
+
377
+ # ---------- non-interactive commands (for `!notebook-llm ...`) ----------
378
+ def main(argv=None) -> None:
379
+ p = argparse.ArgumentParser(prog="notebook-llm", description="Run with no arguments for the interactive menu.")
380
+ p.add_argument("--no-color", action="store_true")
381
+ sub = p.add_subparsers(dest="cmd")
382
+ sub.add_parser("gpus", help="show hardware + recommended models")
383
+ s = sub.add_parser("search", help="search the Ollama library")
384
+ s.add_argument("query", nargs="?", default="")
385
+ k = sub.add_parser("check", help="will MODEL fit? estimated tok/s")
386
+ k.add_argument("model")
387
+ k.add_argument("--ctx", type=int)
388
+ q = sub.add_parser("quickstart", help="install, start, pull best model, open tunnel")
389
+ q.add_argument("--model", default="auto")
390
+ q.add_argument("--no-tunnel", action="store_true")
391
+ sub.add_parser("status")
392
+ sub.add_parser("stop", help="stop tunnel and server")
393
+ a = p.parse_args(argv)
394
+ if a.no_color:
395
+ import os
396
+ os.environ["NO_COLOR"] = "1"
397
+
398
+ if a.cmd is None:
399
+ return run()
400
+ llm = NotebookLLM()
401
+ if a.cmd == "gpus":
402
+ print(llm.hw.summary())
403
+ show_candidates(llm.recommended(), llm)
404
+ elif a.cmd == "search":
405
+ for m in library.search(a.query)[:20]:
406
+ print(f"{m.name:28} {m.pulls:>8} {', '.join(m.sizes)}")
407
+ elif a.cmd == "check":
408
+ cd = library.make_candidate(a.model, llm.hw, a.ctx, llm.settings.kv_cache)
409
+ show_estimate_detail(cd)
410
+ elif a.cmd == "quickstart":
411
+ url = llm.quickstart(a.model, tunnel=not a.no_tunnel)
412
+ print(url or "")
413
+ print(llm.continue_config())
414
+ elif a.cmd == "status":
415
+ print(f"server={llm.server.is_running()} model={llm.settings.model} tunnel={llm.url}")
416
+ elif a.cmd == "stop":
417
+ llm.close_tunnel()
418
+ llm.stop_server()
419
+
420
+
421
+ if __name__ == "__main__":
422
+ main()
notebook_llm/client.py ADDED
@@ -0,0 +1,83 @@
1
+ """Thin Ollama HTTP client."""
2
+ import json
3
+ import sys
4
+ from typing import Dict, List, Optional
5
+
6
+ import requests
7
+
8
+
9
+ class OllamaClient:
10
+ def __init__(self, base_url: str = "http://127.0.0.1:11434"):
11
+ self.base = base_url.rstrip("/")
12
+
13
+ def list_models(self) -> List[Dict]:
14
+ r = requests.get(f"{self.base}/api/tags", timeout=10)
15
+ r.raise_for_status()
16
+ return r.json().get("models", [])
17
+
18
+ def has_model(self, name: str) -> bool:
19
+ names = {m["name"] for m in self.list_models()}
20
+ return name in names or f"{name}:latest" in names
21
+
22
+ def running(self) -> List[Dict]:
23
+ r = requests.get(f"{self.base}/api/ps", timeout=10)
24
+ r.raise_for_status()
25
+ return r.json().get("models", [])
26
+
27
+ def pull(self, name: str, show_progress: bool = True) -> None:
28
+ with requests.post(f"{self.base}/api/pull", json={"model": name, "stream": True},
29
+ stream=True, timeout=None) as r:
30
+ r.raise_for_status()
31
+ last = ""
32
+ for line in r.iter_lines():
33
+ if not line:
34
+ continue
35
+ d = json.loads(line)
36
+ if "error" in d:
37
+ raise RuntimeError(d["error"])
38
+ if not show_progress:
39
+ continue
40
+ total, done = d.get("total"), d.get("completed")
41
+ msg = (f"{d.get('status', '')} {done * 100 // total}% "
42
+ f"({done / 1e9:.1f}/{total / 1e9:.1f} GB)") if total and done is not None else d.get("status", "")
43
+ if msg != last:
44
+ sys.stdout.write(f"\r {msg:<70}")
45
+ sys.stdout.flush()
46
+ last = msg
47
+ if show_progress:
48
+ print()
49
+
50
+ def delete(self, name: str) -> None:
51
+ requests.delete(f"{self.base}/api/delete", json={"model": name}, timeout=30).raise_for_status()
52
+
53
+ def load(self, name: str, keep_alive=-1) -> None:
54
+ requests.post(f"{self.base}/api/generate",
55
+ json={"model": name, "prompt": "", "stream": False, "keep_alive": keep_alive},
56
+ timeout=900).raise_for_status()
57
+
58
+ def unload(self, name: str) -> None:
59
+ self.load(name, keep_alive=0)
60
+
61
+ def chat(self, name: str, prompt: str, system: Optional[str] = None, keep_alive=-1) -> str:
62
+ msgs = ([{"role": "system", "content": system}] if system else []) + [{"role": "user", "content": prompt}]
63
+ r = requests.post(f"{self.base}/api/chat",
64
+ json={"model": name, "messages": msgs, "stream": False, "keep_alive": keep_alive}, timeout=900)
65
+ r.raise_for_status()
66
+ return r.json()["message"]["content"]
67
+
68
+ def benchmark(self, name: str, num_predict: int = 128) -> Dict[str, float]:
69
+ """Measure real tokens/sec using Ollama's own timing counters."""
70
+ r = requests.post(f"{self.base}/api/generate", json={
71
+ "model": name, "stream": False, "keep_alive": -1,
72
+ "prompt": "Write a Python function that checks whether a number is prime, with a short explanation.",
73
+ "options": {"num_predict": num_predict, "temperature": 0}}, timeout=900)
74
+ r.raise_for_status()
75
+ d = r.json()
76
+ gen = d.get("eval_count", 0) / max(d.get("eval_duration", 1), 1) * 1e9
77
+ pre = d.get("prompt_eval_count", 0) / max(d.get("prompt_eval_duration", 1), 1) * 1e9
78
+ out = {"tok_s": gen, "prompt_tok_s": pre, "load_s": d.get("load_duration", 0) / 1e9, "tokens": d.get("eval_count", 0)}
79
+ for m in self.running():
80
+ if m.get("name", "").startswith(name.split(":")[0]) and m.get("size"):
81
+ out["gpu_pct"] = 100 * m.get("size_vram", 0) / m["size"]
82
+ out["loaded_gb"] = m["size"] / 1024 ** 3
83
+ return out
notebook_llm/config.py ADDED
@@ -0,0 +1,32 @@
1
+ """Continue (IDE extension) config for the tunnel URL."""
2
+ from typing import Optional
3
+
4
+
5
+ def continue_config(api_base: str, model: str, autocomplete_model: Optional[str] = None) -> str:
6
+ cfg = f"""name: Notebook LLM
7
+ version: 1.0.0
8
+ schema: v1
9
+ models:
10
+ - name: {model}
11
+ provider: ollama
12
+ model: {model}
13
+ apiBase: {api_base}
14
+ roles:
15
+ - chat
16
+ - edit
17
+ - apply
18
+ """
19
+ if autocomplete_model:
20
+ cfg += f""" - name: {autocomplete_model} (autocomplete)
21
+ provider: ollama
22
+ model: {autocomplete_model}
23
+ apiBase: {api_base}
24
+ roles:
25
+ - autocomplete
26
+ """
27
+ cfg += f""" - name: Autodetect
28
+ provider: ollama
29
+ model: AUTODETECT
30
+ apiBase: {api_base}
31
+ """
32
+ return cfg
notebook_llm/env.py ADDED
@@ -0,0 +1,9 @@
1
+ import os
2
+
3
+
4
+ def is_kaggle() -> bool:
5
+ return bool(os.environ.get("KAGGLE_KERNEL_RUN_TYPE")) or os.path.isdir("/kaggle")
6
+
7
+
8
+ def log(msg: str) -> None:
9
+ print(f"[notebook_llm] {msg}", flush=True)