ollama-mtop 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ollama_mtop-0.2.0/.gitignore +28 -0
- ollama_mtop-0.2.0/CHANGELOG.md +47 -0
- ollama_mtop-0.2.0/LICENSE +21 -0
- ollama_mtop-0.2.0/PKG-INFO +216 -0
- ollama_mtop-0.2.0/README.md +188 -0
- ollama_mtop-0.2.0/pyproject.toml +55 -0
- ollama_mtop-0.2.0/src/mtop/__init__.py +894 -0
- ollama_mtop-0.2.0/src/mtop/__main__.py +4 -0
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.egg-info/
|
|
6
|
+
dist/
|
|
7
|
+
build/
|
|
8
|
+
*.egg
|
|
9
|
+
|
|
10
|
+
# Virtual environments
|
|
11
|
+
.venv/
|
|
12
|
+
venv/
|
|
13
|
+
env/
|
|
14
|
+
|
|
15
|
+
# IDE
|
|
16
|
+
.idea/
|
|
17
|
+
.vscode/
|
|
18
|
+
*.swp
|
|
19
|
+
*.swo
|
|
20
|
+
*~
|
|
21
|
+
|
|
22
|
+
# OS
|
|
23
|
+
.DS_Store
|
|
24
|
+
Thumbs.db
|
|
25
|
+
|
|
26
|
+
# Build artifacts
|
|
27
|
+
*.whl
|
|
28
|
+
*.tar.gz
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.2.0 — 2026-07-05
|
|
4
|
+
|
|
5
|
+
### Architecture
|
|
6
|
+
- **Background collector thread.** All blocking I/O (`docker inspect`/`stats`/`exec`,
|
|
7
|
+
`nvidia-smi`, Ollama API calls with up to 5 s timeouts) moved out of the render
|
|
8
|
+
loop into a daemon collector publishing immutable snapshots. The curses loop
|
|
9
|
+
polls keys at a fixed 100 ms and only draws — a hung API or slow docker daemon
|
|
10
|
+
can no longer freeze the UI. Stale snapshots (> 3× interval) are flagged in the
|
|
11
|
+
header.
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
- `OLLAMA_HOST` values without a scheme (`gpu-rig:11434`, `0.0.0.0:11434`) —
|
|
15
|
+
valid for Ollama itself — crashed urllib. Now normalized to `http://`.
|
|
16
|
+
- Unified-memory platforms (GB10 Spark, Jetson/Orin) where `nvidia-smi` exists
|
|
17
|
+
but reports memory as `[N/A]`: the VRAM bar silently vanished. Memory is now
|
|
18
|
+
patched from `/proc/meminfo` while numeric util/temp from `nvidia-smi` is kept.
|
|
19
|
+
- CPU bar normalized against the container's effective CPU limit
|
|
20
|
+
(`--cpus` / quota via `HostConfig`) instead of the host core count.
|
|
21
|
+
- `GPU_REFRESH`/`STATS_REFRESH` caps were frozen at startup; runtime `+`/`-`
|
|
22
|
+
interval changes now propagate to the slow-path cadence.
|
|
23
|
+
- Detail strings (CPU %, MEM usage, VRAM MiB) were drawn at hardcoded columns
|
|
24
|
+
(x=60/62) and overlapped bars on terminals < ~80 cols; now right-aligned to
|
|
25
|
+
the frame and dropped cleanly when there is no room.
|
|
26
|
+
- Dead install paths in README: curl one-liner pointed at a nonexistent `main`
|
|
27
|
+
branch; `python -m mtop` claimed to work without install (src-layout).
|
|
28
|
+
- Redundant `nodelay(True)` (overridden by `timeout()`) removed.
|
|
29
|
+
|
|
30
|
+
### Added
|
|
31
|
+
- `--json` — one-shot snapshot as JSON on stdout, exit code 1 on unhealthy.
|
|
32
|
+
For cron, Prometheus textfile collectors, Ansible facts.
|
|
33
|
+
- `--no-docker` — API-only mode for remote Ollama instances; skips all local
|
|
34
|
+
docker calls instead of mixing remote models with local container stats.
|
|
35
|
+
- `o` key — toggle the raw `ollama ps` section (default off; it duplicated
|
|
36
|
+
`/api/ps` at the cost of a `docker exec` per refresh).
|
|
37
|
+
- GPU probe strategy memoization (host vs `docker exec` vs unified) — no more
|
|
38
|
+
re-forking failed `nvidia-smi` probes every cycle on GPU-less boxes; memo
|
|
39
|
+
resets on failure so driver restarts are picked up.
|
|
40
|
+
|
|
41
|
+
### Internal
|
|
42
|
+
- Type hints unified on `X | None` (3.10+ baseline), `typing.Optional` dropped.
|
|
43
|
+
- `[tool.ruff]` config added (line-length 100, py310).
|
|
44
|
+
|
|
45
|
+
## 0.1.0
|
|
46
|
+
|
|
47
|
+
Initial release.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Quaerendir
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ollama-mtop
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: htop for Ollama — curses-based TUI monitor for models, GPU, and Docker containers
|
|
5
|
+
Project-URL: Homepage, https://github.com/Quaerendir/mtop
|
|
6
|
+
Project-URL: Repository, https://github.com/Quaerendir/mtop
|
|
7
|
+
Project-URL: Issues, https://github.com/Quaerendir/mtop/issues
|
|
8
|
+
Author-email: Quaerendir <42608192+Quaerendir@users.noreply.github.com>
|
|
9
|
+
Maintainer-email: Quaerendir <42608192+Quaerendir@users.noreply.github.com>
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: curses,docker,gpu,inference,jetson,llm,monitor,nvidia,ollama,tui
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Environment :: Console :: Curses
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: System Administrators
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
25
|
+
Classifier: Topic :: System :: Monitoring
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# mtop
|
|
30
|
+
|
|
31
|
+
**htop for Ollama** — a curses-based TUI that monitors your models, GPU, and Docker container in real time. Zero flicker. Zero dependencies beyond Python 3.10+.
|
|
32
|
+
|
|
33
|
+

|
|
34
|
+

|
|
35
|
+

|
|
36
|
+
|
|
37
|
+
<!-- TODO: Replace with actual screenshot/gif -->
|
|
38
|
+
<!--  -->
|
|
39
|
+
|
|
40
|
+
## Why?
|
|
41
|
+
|
|
42
|
+
There are web dashboards, Prometheus exporters, and chat TUIs for Ollama. But there's no **terminal monitor** — something you SSH into a box and just run, like `htop` or `nvtop`, to see what models are loaded, how much VRAM they're eating, and whether the container is healthy.
|
|
43
|
+
|
|
44
|
+
`mtop` fills that gap. One file, one command, pure stdlib Python.
|
|
45
|
+
|
|
46
|
+
## Features
|
|
47
|
+
|
|
48
|
+
- **Zero-flicker display** — curses double-buffered rendering, no `clear` + print loops
|
|
49
|
+
- **Loaded models** — name, VRAM/RAM split, context length, processor type, TTL countdown
|
|
50
|
+
- **Container health** — status indicator (●/✗/○), uptime, CPU & memory with progress bars
|
|
51
|
+
- **GPU monitoring** — NVIDIA desktop GPUs via `nvidia-smi`, with utilization and VRAM bars
|
|
52
|
+
- **Jetson / Tegra / NVIDIA Spark** — automatic fallback to unified memory via `/proc/meminfo`
|
|
53
|
+
- **Non-blocking UI** — all I/O (docker, nvidia-smi, HTTP) runs in a background collector thread; the interface stays responsive at 100 ms even when the API hangs, and stale data is flagged
|
|
54
|
+
- **Interactive** — `q` to quit, `+`/`-` to adjust refresh interval, `o` to toggle raw `ollama ps`
|
|
55
|
+
- **Scriptable** — `--json` one-shot mode for cron, Prometheus textfile collectors, or Ansible facts (exit code 1 on unhealthy)
|
|
56
|
+
- **API-only mode** — `--no-docker` for monitoring remote Ollama instances without local docker calls
|
|
57
|
+
- **cgroup-aware CPU bar** — normalizes against the container's `--cpus`/quota limit, not the host core count
|
|
58
|
+
- **Docker-aware** — talks to both the Ollama API and `docker exec ollama ps`
|
|
59
|
+
- **Respects `$OLLAMA_HOST`** — works with remote Ollama instances out of the box
|
|
60
|
+
- **Zero external dependencies** — only Python stdlib (`curses`, `urllib`, `json`, `subprocess`)
|
|
61
|
+
|
|
62
|
+
## Quick Start
|
|
63
|
+
|
|
64
|
+
### One-liner (no install)
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
curl -fsSL https://raw.githubusercontent.com/Quaerendir/mtop/master/src/mtop/__init__.py -o mtop.py
|
|
68
|
+
chmod +x mtop.py
|
|
69
|
+
./mtop.py
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
### pip install
|
|
73
|
+
|
|
74
|
+
> Not yet published to PyPI — coming with the first tagged release. Until then, use the one-liner or install from source.
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
pip install ollama-mtop # (pending)
|
|
78
|
+
mtop
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
### From source
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
85
|
+
cd mtop
|
|
86
|
+
pip install -e .
|
|
87
|
+
mtop
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Run directly from a clone (no install)
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
94
|
+
cd mtop
|
|
95
|
+
PYTHONPATH=src python -m mtop
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## Usage
|
|
99
|
+
|
|
100
|
+
```
|
|
101
|
+
mtop [-c CONTAINER] [-i INTERVAL] [-u URL] [--no-gpu] [--no-docker] [--json] [-V] [-h]
|
|
102
|
+
|
|
103
|
+
Options:
|
|
104
|
+
-c, --container NAME Docker container name (default: ollama)
|
|
105
|
+
-i, --interval SECS Refresh interval in seconds (default: 1.0)
|
|
106
|
+
-u, --api-url URL Ollama API base URL (default: $OLLAMA_HOST or http://localhost:11434)
|
|
107
|
+
Scheme-less values (gpu-rig:11434) are accepted, like Ollama itself
|
|
108
|
+
--no-gpu Disable GPU monitoring section
|
|
109
|
+
--no-docker API-only mode: skip all docker calls (remote instances)
|
|
110
|
+
--json Print one snapshot as JSON and exit (exit 1 on unhealthy)
|
|
111
|
+
-V, --version Show version
|
|
112
|
+
-h, --help Show help
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
### Examples
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
# Monitor a custom container name
|
|
119
|
+
mtop -c my-ollama
|
|
120
|
+
|
|
121
|
+
# Slower refresh for remote/metered connections
|
|
122
|
+
mtop -i 5
|
|
123
|
+
|
|
124
|
+
# Monitor a remote Ollama instance — API only, no local docker/GPU noise
|
|
125
|
+
mtop -u 192.168.1.100:11434 --no-docker
|
|
126
|
+
|
|
127
|
+
# One-shot health/state snapshot for scripting
|
|
128
|
+
mtop --json | jq '.models[].name'
|
|
129
|
+
|
|
130
|
+
# Using OLLAMA_HOST environment variable
|
|
131
|
+
export OLLAMA_HOST=http://gpu-rig:11434
|
|
132
|
+
mtop
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
### Interactive Keys
|
|
136
|
+
|
|
137
|
+
| Key | Action |
|
|
138
|
+
|-----|--------|
|
|
139
|
+
| `q` / `ESC` | Quit |
|
|
140
|
+
| `+` | Decrease refresh interval (faster) |
|
|
141
|
+
| `-` | Increase refresh interval (slower) |
|
|
142
|
+
| `o` | Toggle raw `ollama ps` section |
|
|
143
|
+
|
|
144
|
+
## Display Layout
|
|
145
|
+
|
|
146
|
+
```
|
|
147
|
+
─── mtop v0.2.0 — Ollama Model Monitor ───
|
|
148
|
+
host: gpu-rig container: ● ollama up: 3d 14h 2026-03-11 15:42:01
|
|
149
|
+
────────────────────────────────────────────────────────────────────────────────
|
|
150
|
+
CONTAINER RESOURCES
|
|
151
|
+
CPU [████░░░░░░░░░░░░░░░░░░░░░░░░░░] 12.3%
|
|
152
|
+
MEM [██████████████░░░░░░░░░░░░░░░░] 45.2% 14.2GiB / 31.4GiB
|
|
153
|
+
GPU
|
|
154
|
+
[0] NVIDIA GeForce RTX 4090 42°C
|
|
155
|
+
UTIL [████████░░░░░░░░░░░░░░░░░] 32.0%
|
|
156
|
+
VRAM [██████████████████░░░░░░░] 72.4% 17382 / 24000 MiB
|
|
157
|
+
LOADED MODELS
|
|
158
|
+
MODEL VRAM RAM CTX PROCESSOR EXPIRES
|
|
159
|
+
──────────────────────────────────────────────────────────────────────────────────────────────
|
|
160
|
+
qwen2.5-coder:32b-instruct-q8_0 18.42 G 0.00 G 32768 GPU 4m 32s left
|
|
161
|
+
OLLAMA PS (raw)
|
|
162
|
+
NAME SIZE PROCESSOR UNTIL
|
|
163
|
+
qwen2.5-coder:32b-instruct-q8_0 19.8 GB 100% GPU 4 minutes from now
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Supported Platforms
|
|
167
|
+
|
|
168
|
+
| Platform | GPU Monitoring | Notes |
|
|
169
|
+
|----------|---------------|-------|
|
|
170
|
+
| Linux x86_64 + NVIDIA | ✅ Full | `nvidia-smi` on host or in container |
|
|
171
|
+
| NVIDIA Jetson / Orin | ✅ Unified memory | Falls back to `/proc/meminfo` |
|
|
172
|
+
| NVIDIA GB10 Spark | ✅ Unified memory | Tegra-based, same fallback |
|
|
173
|
+
| Linux without GPU | ✅ (no GPU section) | Use `--no-gpu` to hide the section |
|
|
174
|
+
| macOS | ⚠️ Partial | curses works, no `nvidia-smi`; Docker Desktop only |
|
|
175
|
+
| WSL2 | ⚠️ Partial | Works if Docker + nvidia-container-toolkit configured |
|
|
176
|
+
|
|
177
|
+
## Requirements
|
|
178
|
+
|
|
179
|
+
- **Python 3.10+** (uses `match`-era type hints like `list[str]`, `X | Y`)
|
|
180
|
+
- **Docker** (for container monitoring)
|
|
181
|
+
- **Ollama** running in a Docker container (or accessible via API)
|
|
182
|
+
- **nvidia-smi** (optional, for GPU stats)
|
|
183
|
+
|
|
184
|
+
## Roadmap
|
|
185
|
+
|
|
186
|
+
- [ ] Record terminal sessions with `asciinema` for README gif
|
|
187
|
+
- [ ] AMD ROCm GPU support (`rocm-smi`)
|
|
188
|
+
- [ ] Apple Silicon GPU stats (via `powermetrics`)
|
|
189
|
+
- [ ] Model pull progress tracking
|
|
190
|
+
- [ ] Multiple container / multi-host support
|
|
191
|
+
- [x] Configurable layout (raw `ollama ps` toggle; more sections to follow)
|
|
192
|
+
- [ ] Model actions — unload on keypress (`keep_alive: 0`), extend TTL
|
|
193
|
+
- [ ] Sparkline history for CPU/GPU utilization (braille chars, stdlib deque)
|
|
194
|
+
- [ ] systemd/bare-metal Ollama support (cgroup v2 stats, no Docker required)
|
|
195
|
+
- [ ] Log panel (tail Ollama container logs)
|
|
196
|
+
- [ ] Request rate / tokens-per-second from Ollama API
|
|
197
|
+
|
|
198
|
+
## Contributing
|
|
199
|
+
|
|
200
|
+
PRs welcome. Keep it stdlib-only — the zero-dependency constraint is a feature, not a limitation.
|
|
201
|
+
|
|
202
|
+
```bash
|
|
203
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
204
|
+
cd mtop
|
|
205
|
+
pip install -e .
|
|
206
|
+
# hack on src/mtop/__init__.py
|
|
207
|
+
mtop
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
## License
|
|
211
|
+
|
|
212
|
+
MIT — see [LICENSE](LICENSE).
|
|
213
|
+
|
|
214
|
+
## Acknowledgements
|
|
215
|
+
|
|
216
|
+
Built as a collaboration between a human homelab geek and Claude (Anthropic) during a late-night infrastructure session. The original bash prototype migrated to Python/curses because fighting `tput` and `jq` in a loop was getting old.
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# mtop
|
|
2
|
+
|
|
3
|
+
**htop for Ollama** — a curses-based TUI that monitors your models, GPU, and Docker container in real time. Zero flicker. Zero dependencies beyond Python 3.10+.
|
|
4
|
+
|
|
5
|
+

|
|
6
|
+

|
|
7
|
+

|
|
8
|
+
|
|
9
|
+
<!-- TODO: Replace with actual screenshot/gif -->
|
|
10
|
+
<!--  -->
|
|
11
|
+
|
|
12
|
+
## Why?
|
|
13
|
+
|
|
14
|
+
There are web dashboards, Prometheus exporters, and chat TUIs for Ollama. But there's no **terminal monitor** — something you SSH into a box and just run, like `htop` or `nvtop`, to see what models are loaded, how much VRAM they're eating, and whether the container is healthy.
|
|
15
|
+
|
|
16
|
+
`mtop` fills that gap. One file, one command, pure stdlib Python.
|
|
17
|
+
|
|
18
|
+
## Features
|
|
19
|
+
|
|
20
|
+
- **Zero-flicker display** — curses double-buffered rendering, no `clear` + print loops
|
|
21
|
+
- **Loaded models** — name, VRAM/RAM split, context length, processor type, TTL countdown
|
|
22
|
+
- **Container health** — status indicator (●/✗/○), uptime, CPU & memory with progress bars
|
|
23
|
+
- **GPU monitoring** — NVIDIA desktop GPUs via `nvidia-smi`, with utilization and VRAM bars
|
|
24
|
+
- **Jetson / Tegra / NVIDIA Spark** — automatic fallback to unified memory via `/proc/meminfo`
|
|
25
|
+
- **Non-blocking UI** — all I/O (docker, nvidia-smi, HTTP) runs in a background collector thread; the interface stays responsive at 100 ms even when the API hangs, and stale data is flagged
|
|
26
|
+
- **Interactive** — `q` to quit, `+`/`-` to adjust refresh interval, `o` to toggle raw `ollama ps`
|
|
27
|
+
- **Scriptable** — `--json` one-shot mode for cron, Prometheus textfile collectors, or Ansible facts (exit code 1 on unhealthy)
|
|
28
|
+
- **API-only mode** — `--no-docker` for monitoring remote Ollama instances without local docker calls
|
|
29
|
+
- **cgroup-aware CPU bar** — normalizes against the container's `--cpus`/quota limit, not the host core count
|
|
30
|
+
- **Docker-aware** — talks to both the Ollama API and `docker exec ollama ps`
|
|
31
|
+
- **Respects `$OLLAMA_HOST`** — works with remote Ollama instances out of the box
|
|
32
|
+
- **Zero external dependencies** — only Python stdlib (`curses`, `urllib`, `json`, `subprocess`)
|
|
33
|
+
|
|
34
|
+
## Quick Start
|
|
35
|
+
|
|
36
|
+
### One-liner (no install)
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
curl -fsSL https://raw.githubusercontent.com/Quaerendir/mtop/master/src/mtop/__init__.py -o mtop.py
|
|
40
|
+
chmod +x mtop.py
|
|
41
|
+
./mtop.py
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
### pip install
|
|
45
|
+
|
|
46
|
+
> Not yet published to PyPI — coming with the first tagged release. Until then, use the one-liner or install from source.
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install ollama-mtop # (pending)
|
|
50
|
+
mtop
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
### From source
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
57
|
+
cd mtop
|
|
58
|
+
pip install -e .
|
|
59
|
+
mtop
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
### Run directly from a clone (no install)
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
66
|
+
cd mtop
|
|
67
|
+
PYTHONPATH=src python -m mtop
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## Usage
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
mtop [-c CONTAINER] [-i INTERVAL] [-u URL] [--no-gpu] [--no-docker] [--json] [-V] [-h]
|
|
74
|
+
|
|
75
|
+
Options:
|
|
76
|
+
-c, --container NAME Docker container name (default: ollama)
|
|
77
|
+
-i, --interval SECS Refresh interval in seconds (default: 1.0)
|
|
78
|
+
-u, --api-url URL Ollama API base URL (default: $OLLAMA_HOST or http://localhost:11434)
|
|
79
|
+
Scheme-less values (gpu-rig:11434) are accepted, like Ollama itself
|
|
80
|
+
--no-gpu Disable GPU monitoring section
|
|
81
|
+
--no-docker API-only mode: skip all docker calls (remote instances)
|
|
82
|
+
--json Print one snapshot as JSON and exit (exit 1 on unhealthy)
|
|
83
|
+
-V, --version Show version
|
|
84
|
+
-h, --help Show help
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### Examples
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
# Monitor a custom container name
|
|
91
|
+
mtop -c my-ollama
|
|
92
|
+
|
|
93
|
+
# Slower refresh for remote/metered connections
|
|
94
|
+
mtop -i 5
|
|
95
|
+
|
|
96
|
+
# Monitor a remote Ollama instance — API only, no local docker/GPU noise
|
|
97
|
+
mtop -u 192.168.1.100:11434 --no-docker
|
|
98
|
+
|
|
99
|
+
# One-shot health/state snapshot for scripting
|
|
100
|
+
mtop --json | jq '.models[].name'
|
|
101
|
+
|
|
102
|
+
# Using OLLAMA_HOST environment variable
|
|
103
|
+
export OLLAMA_HOST=http://gpu-rig:11434
|
|
104
|
+
mtop
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Interactive Keys
|
|
108
|
+
|
|
109
|
+
| Key | Action |
|
|
110
|
+
|-----|--------|
|
|
111
|
+
| `q` / `ESC` | Quit |
|
|
112
|
+
| `+` | Decrease refresh interval (faster) |
|
|
113
|
+
| `-` | Increase refresh interval (slower) |
|
|
114
|
+
| `o` | Toggle raw `ollama ps` section |
|
|
115
|
+
|
|
116
|
+
## Display Layout
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
─── mtop v0.2.0 — Ollama Model Monitor ───
|
|
120
|
+
host: gpu-rig container: ● ollama up: 3d 14h 2026-03-11 15:42:01
|
|
121
|
+
────────────────────────────────────────────────────────────────────────────────
|
|
122
|
+
CONTAINER RESOURCES
|
|
123
|
+
CPU [████░░░░░░░░░░░░░░░░░░░░░░░░░░] 12.3%
|
|
124
|
+
MEM [██████████████░░░░░░░░░░░░░░░░] 45.2% 14.2GiB / 31.4GiB
|
|
125
|
+
GPU
|
|
126
|
+
[0] NVIDIA GeForce RTX 4090 42°C
|
|
127
|
+
UTIL [████████░░░░░░░░░░░░░░░░░] 32.0%
|
|
128
|
+
VRAM [██████████████████░░░░░░░] 72.4% 17382 / 24000 MiB
|
|
129
|
+
LOADED MODELS
|
|
130
|
+
MODEL VRAM RAM CTX PROCESSOR EXPIRES
|
|
131
|
+
──────────────────────────────────────────────────────────────────────────────────────────────
|
|
132
|
+
qwen2.5-coder:32b-instruct-q8_0 18.42 G 0.00 G 32768 GPU 4m 32s left
|
|
133
|
+
OLLAMA PS (raw)
|
|
134
|
+
NAME SIZE PROCESSOR UNTIL
|
|
135
|
+
qwen2.5-coder:32b-instruct-q8_0 19.8 GB 100% GPU 4 minutes from now
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Supported Platforms
|
|
139
|
+
|
|
140
|
+
| Platform | GPU Monitoring | Notes |
|
|
141
|
+
|----------|---------------|-------|
|
|
142
|
+
| Linux x86_64 + NVIDIA | ✅ Full | `nvidia-smi` on host or in container |
|
|
143
|
+
| NVIDIA Jetson / Orin | ✅ Unified memory | Falls back to `/proc/meminfo` |
|
|
144
|
+
| NVIDIA GB10 Spark | ✅ Unified memory | Tegra-based, same fallback |
|
|
145
|
+
| Linux without GPU | ✅ (no GPU section) | Use `--no-gpu` to hide the section |
|
|
146
|
+
| macOS | ⚠️ Partial | curses works, no `nvidia-smi`; Docker Desktop only |
|
|
147
|
+
| WSL2 | ⚠️ Partial | Works if Docker + nvidia-container-toolkit configured |
|
|
148
|
+
|
|
149
|
+
## Requirements
|
|
150
|
+
|
|
151
|
+
- **Python 3.10+** (uses `match`-era type hints like `list[str]`, `X | Y`)
|
|
152
|
+
- **Docker** (for container monitoring)
|
|
153
|
+
- **Ollama** running in a Docker container (or accessible via API)
|
|
154
|
+
- **nvidia-smi** (optional, for GPU stats)
|
|
155
|
+
|
|
156
|
+
## Roadmap
|
|
157
|
+
|
|
158
|
+
- [ ] Record terminal sessions with `asciinema` for README gif
|
|
159
|
+
- [ ] AMD ROCm GPU support (`rocm-smi`)
|
|
160
|
+
- [ ] Apple Silicon GPU stats (via `powermetrics`)
|
|
161
|
+
- [ ] Model pull progress tracking
|
|
162
|
+
- [ ] Multiple container / multi-host support
|
|
163
|
+
- [x] Configurable layout (raw `ollama ps` toggle; more sections to follow)
|
|
164
|
+
- [ ] Model actions — unload on keypress (`keep_alive: 0`), extend TTL
|
|
165
|
+
- [ ] Sparkline history for CPU/GPU utilization (braille chars, stdlib deque)
|
|
166
|
+
- [ ] systemd/bare-metal Ollama support (cgroup v2 stats, no Docker required)
|
|
167
|
+
- [ ] Log panel (tail Ollama container logs)
|
|
168
|
+
- [ ] Request rate / tokens-per-second from Ollama API
|
|
169
|
+
|
|
170
|
+
## Contributing
|
|
171
|
+
|
|
172
|
+
PRs welcome. Keep it stdlib-only — the zero-dependency constraint is a feature, not a limitation.
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
git clone https://github.com/Quaerendir/mtop.git
|
|
176
|
+
cd mtop
|
|
177
|
+
pip install -e .
|
|
178
|
+
# hack on src/mtop/__init__.py
|
|
179
|
+
mtop
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
## License
|
|
183
|
+
|
|
184
|
+
MIT — see [LICENSE](LICENSE).
|
|
185
|
+
|
|
186
|
+
## Acknowledgements
|
|
187
|
+
|
|
188
|
+
Built as a collaboration between a human homelab geek and Claude (Anthropic) during a late-night infrastructure session. The original bash prototype migrated to Python/curses because fighting `tput` and `jq` in a loop was getting old.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ollama-mtop"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "htop for Ollama — curses-based TUI monitor for models, GPU, and Docker containers"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text = "MIT"}
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Quaerendir", email = "42608192+Quaerendir@users.noreply.github.com" },
|
|
14
|
+
]
|
|
15
|
+
maintainers = [
|
|
16
|
+
{ name = "Quaerendir", email = "42608192+Quaerendir@users.noreply.github.com" },
|
|
17
|
+
]
|
|
18
|
+
keywords = [
|
|
19
|
+
"ollama", "monitor", "tui", "curses", "docker",
|
|
20
|
+
"gpu", "nvidia", "jetson", "llm", "inference",
|
|
21
|
+
]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 4 - Beta",
|
|
24
|
+
"Environment :: Console :: Curses",
|
|
25
|
+
"Intended Audience :: Developers",
|
|
26
|
+
"Intended Audience :: System Administrators",
|
|
27
|
+
"License :: OSI Approved :: MIT License",
|
|
28
|
+
"Operating System :: POSIX :: Linux",
|
|
29
|
+
"Programming Language :: Python :: 3",
|
|
30
|
+
"Programming Language :: Python :: 3.10",
|
|
31
|
+
"Programming Language :: Python :: 3.11",
|
|
32
|
+
"Programming Language :: Python :: 3.12",
|
|
33
|
+
"Programming Language :: Python :: 3.13",
|
|
34
|
+
"Topic :: System :: Monitoring",
|
|
35
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Homepage = "https://github.com/Quaerendir/mtop"
|
|
40
|
+
Repository = "https://github.com/Quaerendir/mtop"
|
|
41
|
+
Issues = "https://github.com/Quaerendir/mtop/issues"
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
mtop = "mtop:main"
|
|
45
|
+
|
|
46
|
+
[tool.hatch.build.targets.wheel]
|
|
47
|
+
packages = ["src/mtop"]
|
|
48
|
+
|
|
49
|
+
[tool.ruff]
|
|
50
|
+
line-length = 100
|
|
51
|
+
target-version = "py310"
|
|
52
|
+
|
|
53
|
+
[tool.ruff.lint]
|
|
54
|
+
select = ["E", "F", "W", "B", "SIM", "UP", "C4"]
|
|
55
|
+
ignore = ["E741", "SIM105"] # SIM105: contextlib.suppress has per-call overhead in the render hot path
|