nvmon 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nvmon-0.1.0.dist-info/METADATA +69 -0
- nvmon-0.1.0.dist-info/RECORD +6 -0
- nvmon-0.1.0.dist-info/WHEEL +4 -0
- nvmon-0.1.0.dist-info/entry_points.txt +2 -0
- nvmon-0.1.0.dist-info/licenses/LICENSE +21 -0
- nvmon.py +751 -0
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: nvmon
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A fancy NVIDIA GPU monitor for the terminal
|
|
5
|
+
Project-URL: Source, https://github.com/Han-DongHeun/nvmon
|
|
6
|
+
Project-URL: Issues, https://github.com/Han-DongHeun/nvmon/issues
|
|
7
|
+
Author: Han DongHeun
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: gpu,monitor,nvidia,nvml,terminal,tui
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: System Administrators
|
|
15
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
16
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Topic :: System :: Monitoring
|
|
19
|
+
Requires-Python: >=3.6
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# nvmon
|
|
23
|
+
|
|
24
|
+
A fancy NVIDIA GPU monitor for the terminal. Single Python file, no dependencies.
|
|
25
|
+
|
|
26
|
+

|
|
27
|
+
|
|
28
|
+
On a wide terminal, eight GPUs go into two columns:
|
|
29
|
+
|
|
30
|
+

|
|
31
|
+
|
|
32
|
+
## Install
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
uv tool install nvmon # or: pipx install nvmon
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
From GitHub: `uv tool install git+https://github.com/Han-DongHeun/nvmon`, update with
|
|
39
|
+
`uv tool upgrade nvmon`. On a server without internet, copy `nvmon.py` over and run it with any
|
|
40
|
+
Python 3.6+.
|
|
41
|
+
|
|
42
|
+
## Usage
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
nvmon # Esc, q or Ctrl+C to quit
|
|
46
|
+
nvmon -i 1 # refresh every second (default: 0.5)
|
|
47
|
+
nvmon -g 0,2,4-7 # only these GPUs
|
|
48
|
+
ssh -t HOST nvmon # over ssh: -t gives it a terminal
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## What it shows
|
|
52
|
+
|
|
53
|
+
- **GPU**: share of time any kernel ran. **cores / tensor** (H100 and newer): share of SMs busy
|
|
54
|
+
and Tensor Core activity. A GPU can read 100 % while its cores sit at 30 %.
|
|
55
|
+
- **MEM**, **PCIe** traffic in each direction against the link's top speed, power, temperature,
|
|
56
|
+
fan and clock.
|
|
57
|
+
- Processes on each GPU, grouped by account (your own unlabelled), script names for Python.
|
|
58
|
+
- Warnings only when something is off: `SLOWED: power cap | too hot | hw brake` when the clock is held back,
|
|
59
|
+
`PCIe DEGRADED: x16 -> x8` when the card runs on fewer lanes than it supports.
|
|
60
|
+
|
|
61
|
+
## Notes
|
|
62
|
+
|
|
63
|
+
- Needs an NVIDIA driver and a truecolor terminal. In tmux, add
|
|
64
|
+
`set -ag terminal-overrides ",*:RGB"` to `~/.tmux.conf`.
|
|
65
|
+
- On Windows, run it in Windows Terminal; Git Bash's default window is not a terminal to Python.
|
|
66
|
+
|
|
67
|
+
## License
|
|
68
|
+
|
|
69
|
+
MIT
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
nvmon.py,sha256=dwbcgyRkEcflcFwPBGGDGgKk30QZLHFdKqkMRaWhd9E,33637
|
|
2
|
+
nvmon-0.1.0.dist-info/METADATA,sha256=fzuI5e_t1JS5od7awT4Ex6ow5fgl7xR5NBmzwnESa_k,2497
|
|
3
|
+
nvmon-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
4
|
+
nvmon-0.1.0.dist-info/entry_points.txt,sha256=vDqTDCl58ogjFNrY3CieSTOvoqZMdqfsmbsLDKkKazo,37
|
|
5
|
+
nvmon-0.1.0.dist-info/licenses/LICENSE,sha256=3iKjr_ugOfDusXLMyFJWvySMJuLwoEoUeB_fT1TvA3c,1069
|
|
6
|
+
nvmon-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Han DongHeun
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
nvmon.py
ADDED
|
@@ -0,0 +1,751 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""nvmon: a compact, btop-style NVIDIA GPU monitor with zero dependencies.
|
|
3
|
+
|
|
4
|
+
One box per GPU: utilization history on the left (0 % at the bottom, 100 % at
|
|
5
|
+
the top); utilization, memory and PCIe traffic on the right; name, power,
|
|
6
|
+
warnings, temperature, fan and clock on the top edge; processes on the bottom
|
|
7
|
+
edge. It talks to the NVML library that ships with the NVIDIA driver through
|
|
8
|
+
ctypes and draws with plain ANSI escape codes, so a single file runs on any
|
|
9
|
+
Python >= 3.6.
|
|
10
|
+
|
|
11
|
+
nvmon [-i SECONDS] [-g 0,2,4-7] q / Esc / Ctrl+C to quit
|
|
12
|
+
"""
|
|
13
|
+
import argparse
|
|
14
|
+
import ctypes
|
|
15
|
+
import itertools
|
|
16
|
+
import math
|
|
17
|
+
import os
|
|
18
|
+
import signal
|
|
19
|
+
import socket
|
|
20
|
+
import sys
|
|
21
|
+
import time
|
|
22
|
+
from collections import deque
|
|
23
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
24
|
+
from ctypes import byref, c_int, c_uint, c_ulonglong, c_void_p
|
|
25
|
+
from typing import NamedTuple, Optional
|
|
26
|
+
|
|
27
|
+
__version__ = "0.1.0"
|
|
28
|
+
|
|
29
|
+
INFO_W = 32 # width of the stats column
|
|
30
|
+
CHROME_W = 7 # "│ " + " │ " + " │" around graph and stats
|
|
31
|
+
MIN_GRAPH_W = 30 # a narrower graph is not worth a second column
|
|
32
|
+
MAX_INNER_H = 6 # tallest panel: the full stats column (graph: 6 x 8 levels)
|
|
33
|
+
HISTORY = 1024 # samples kept per GPU; wider than any terminal
|
|
34
|
+
BLOCKS = " ▁▂▃▄▅▆▇█" # a character cell filled in 1/8 steps
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _fg(n):
|
|
38
|
+
return "\x1b[38;5;{}m".format(n)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _rgb(rgb):
|
|
42
|
+
return "\x1b[38;2;{};{};{}m".format(*rgb)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _gradient(stops):
|
|
46
|
+
"""Truecolor codes for 0..100, interpolated through (position, (r, g, b)) `stops`."""
|
|
47
|
+
colours = []
|
|
48
|
+
for i in range(101):
|
|
49
|
+
x = min(max(i, stops[0][0]), stops[-1][0])
|
|
50
|
+
(x0, c0), (x1, c1) = next(pair for pair in zip(stops, stops[1:]) if x <= pair[1][0])
|
|
51
|
+
colours.append(_rgb([round(a + (b - a) * (x - x0) / (x1 - x0)) for a, b in zip(c0, c1)]))
|
|
52
|
+
return colours
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
GREEN, YELLOW, ORANGE, RED = (95, 175, 95), (215, 215, 95), (215, 135, 95), (215, 95, 95)
|
|
56
|
+
# Utilization and shares: green -> yellow -> orange -> red over 0-100 %.
|
|
57
|
+
HEAT = _gradient([(0, GREEN), (20, (175, 215, 95)), (40, YELLOW), (60, (215, 175, 95)), (80, ORANGE), (100, RED)])
|
|
58
|
+
# Temperature in °C: idle GPUs sit at 30-45, busy ones at 60-80, most throttle from about 85-90.
|
|
59
|
+
TEMP = _gradient([(30, (95, 135, 215)), (45, (95, 175, 175)), (60, GREEN), (72, YELLOW), (80, ORANGE), (88, RED)])
|
|
60
|
+
# Warnings on the top edge: yellow = worth a look, orange = slowed, red = act.
|
|
61
|
+
WARN, SLOW, ALERT = _rgb(YELLOW), _rgb(ORANGE), _rgb(RED)
|
|
62
|
+
# Everything else sticks to the 256-colour palette.
|
|
63
|
+
DIM, FAINT, PROC = _fg(240), _fg(237), _fg(110)
|
|
64
|
+
BOLD, RESET = "\x1b[1m", "\x1b[0m"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def heat(t):
|
|
68
|
+
"""Colour for a position t in [0, 1] of the 0-100 % scale."""
|
|
69
|
+
return HEAT[round(min(max(t, 0), 1) * 100)]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ── NVML (libnvidia-ml / nvml.dll, part of the NVIDIA driver) ───────────────
|
|
73
|
+
|
|
74
|
+
NVML_SUCCESS = 0
|
|
75
|
+
NVML_TEMPERATURE_GPU = 0
|
|
76
|
+
NVML_CLOCK_GRAPHICS = 0
|
|
77
|
+
NVML_PCIE_UTIL_TX_BYTES, NVML_PCIE_UTIL_RX_BYTES = 0, 1
|
|
78
|
+
NVML_DEVICE_NAME_BUFFER_SIZE = 96
|
|
79
|
+
NVML_SYSTEM_DRIVER_VERSION_BUFFER_SIZE = 80
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class Utilization(ctypes.Structure):
|
|
83
|
+
_fields_ = [("gpu", c_uint), ("memory", c_uint)]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class Memory(ctypes.Structure):
|
|
87
|
+
_fields_ = [("total", c_ulonglong), ("free", c_ulonglong), ("used", c_ulonglong)]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class MemoryV2(ctypes.Structure):
|
|
91
|
+
_fields_ = [("version", c_uint), ("total", c_ulonglong), ("reserved", c_ulonglong),
|
|
92
|
+
("free", c_ulonglong), ("used", c_ulonglong)]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class TemperatureV1(ctypes.Structure):
|
|
96
|
+
_fields_ = [("version", c_uint), ("sensorType", c_int), ("temperature", c_int)]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class ProcessInfo(ctypes.Structure): # nvmlProcessInfo_v2_t
|
|
100
|
+
_fields_ = [("pid", c_uint), ("usedGpuMemory", c_ulonglong),
|
|
101
|
+
("gpuInstanceId", c_uint), ("computeInstanceId", c_uint)]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
NVML_VALUE_NOT_AVAILABLE = 2**64 - 1 # usedGpuMemory under Windows WDDM
|
|
105
|
+
MAX_PROCESSES = 64
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class GpmSupport(ctypes.Structure): # nvmlGpmSupport_t
|
|
109
|
+
_fields_ = [("version", c_uint), ("isSupportedDevice", c_uint)]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class GpmMetric(ctypes.Structure): # nvmlGpmMetric_t
|
|
113
|
+
_fields_ = [("metricId", c_uint), ("nvmlReturn", c_int), ("value", ctypes.c_double),
|
|
114
|
+
("shortName", ctypes.c_char_p), ("longName", ctypes.c_char_p), ("unit", ctypes.c_char_p)]
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# The GPM metrics we read: SM busy %, Tensor Core activity %, PCIe MiB/s from and to the GPU.
|
|
118
|
+
GPM_METRICS = (2, 5, 20, 21)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class GpmMetricsGet(ctypes.Structure): # nvmlGpmMetricsGet_t, sized for GPM_METRICS
|
|
122
|
+
_fields_ = [("version", c_uint), ("numMetrics", c_uint), ("sample1", c_void_p), ("sample2", c_void_p),
|
|
123
|
+
("metrics", GpmMetric * len(GPM_METRICS))]
|
|
124
|
+
|
|
125
|
+
# nvmlClocksEventReason bits that mean "held back", most serious first. The others (idle,
|
|
126
|
+
# application clocks, sync boost, display) are normal operation and stay hidden.
|
|
127
|
+
SLOWDOWNS = [(0x20 | 0x40, "SLOWED: too hot", ALERT), # software / hardware thermal slowdown
|
|
128
|
+
(0x08 | 0x80, "SLOWED: hw brake", ALERT), # hardware slowdown / power brake
|
|
129
|
+
(0x04, "SLOWED: power cap", SLOW)] # software power cap
|
|
130
|
+
|
|
131
|
+
# PCIe payload bandwidth per lane and direction, GB/s, by link generation.
|
|
132
|
+
PCIE_LANE_GBS = {1: 0.25, 2: 0.5, 3: 0.985, 4: 1.969, 5: 3.938, 6: 7.563}
|
|
133
|
+
|
|
134
|
+
# NVML_STRUCT_VERSION(): struct size with the version number in the top byte.
|
|
135
|
+
MEMORY_V2 = ctypes.sizeof(MemoryV2) | 2 << 24
|
|
136
|
+
TEMPERATURE_V1 = ctypes.sizeof(TemperatureV1) | 1 << 24
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class NvmlError(Exception):
|
|
140
|
+
pass
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def load_nvml():
|
|
144
|
+
"""Open the driver's NVML library and initialise it."""
|
|
145
|
+
if os.name == "nt":
|
|
146
|
+
paths = [os.path.join(os.environ.get("WINDIR", r"C:\Windows"), "System32", "nvml.dll"),
|
|
147
|
+
os.path.join(os.environ.get("ProgramFiles", r"C:\Program Files"),
|
|
148
|
+
"NVIDIA Corporation", "NVSMI", "nvml.dll")]
|
|
149
|
+
else:
|
|
150
|
+
paths = ["libnvidia-ml.so.1"]
|
|
151
|
+
for path in paths:
|
|
152
|
+
try:
|
|
153
|
+
nv = ctypes.CDLL(path) # CDLL releases the GIL during calls, so GPUs poll in parallel
|
|
154
|
+
except OSError:
|
|
155
|
+
continue
|
|
156
|
+
nv.nvmlErrorString.restype = ctypes.c_char_p
|
|
157
|
+
check(nv, nv.nvmlInit_v2())
|
|
158
|
+
return nv
|
|
159
|
+
raise NvmlError("NVML library not found - is the NVIDIA driver installed?")
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def check(nv, rc):
|
|
163
|
+
if rc != NVML_SUCCESS:
|
|
164
|
+
raise NvmlError(nv.nvmlErrorString(rc).decode())
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class Sample(NamedTuple):
|
|
168
|
+
util: Optional[int] # %
|
|
169
|
+
temp: Optional[int] # °C
|
|
170
|
+
fan: Optional[int] # %
|
|
171
|
+
power: Optional[float] # W
|
|
172
|
+
power_limit: Optional[float] # W
|
|
173
|
+
clock: Optional[int] # graphics clock, MHz
|
|
174
|
+
mem_used: Optional[float] # GiB
|
|
175
|
+
mem_total: Optional[float] # GiB
|
|
176
|
+
tx: Optional[float] # PCIe GPU -> CPU, bytes/s
|
|
177
|
+
rx: Optional[float] # PCIe CPU -> GPU, bytes/s
|
|
178
|
+
pcie_width: Optional[int] # current link width (lanes)
|
|
179
|
+
cores: Optional[float] # % of SMs busy (GPM, Hopper and newer)
|
|
180
|
+
tensor: Optional[float] # % Tensor Core activity (GPM)
|
|
181
|
+
slowdown: Optional[tuple] # (label, colour) when the clock is held back
|
|
182
|
+
processes: list # [(name, GiB or None, owner or None)]
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class Gpu:
|
|
186
|
+
def __init__(self, nv, index, handle):
|
|
187
|
+
self.nv, self.index, self.handle = nv, index, handle
|
|
188
|
+
name = ctypes.create_string_buffer(NVML_DEVICE_NAME_BUFFER_SIZE)
|
|
189
|
+
nv.nvmlDeviceGetName(handle, name, NVML_DEVICE_NAME_BUFFER_SIZE)
|
|
190
|
+
name = name.value.decode("utf-8", "replace")
|
|
191
|
+
self.name = name[len("NVIDIA "):] if name.startswith("NVIDIA ") else name
|
|
192
|
+
self.history = deque(maxlen=HISTORY)
|
|
193
|
+
self.now = None
|
|
194
|
+
self.pcie_max_gen = self._uint("nvmlDeviceGetMaxPcieLinkGeneration")
|
|
195
|
+
self.pcie_max_width = self._uint("nvmlDeviceGetMaxPcieLinkWidth")
|
|
196
|
+
self._gpm = self._gpm_samples()
|
|
197
|
+
self.has_activity = self._gpm is not None
|
|
198
|
+
self._names = {} # pid -> (name, owner): fixed for a process's life, so read once
|
|
199
|
+
|
|
200
|
+
def _ok(self, fn, *args):
|
|
201
|
+
return getattr(self.nv, fn)(self.handle, *args) == NVML_SUCCESS
|
|
202
|
+
|
|
203
|
+
def _uint(self, fn, *args):
|
|
204
|
+
"""Scalar query; None when this GPU does not support it (e.g. fan on an H100)."""
|
|
205
|
+
value = c_uint()
|
|
206
|
+
return value.value if self._ok(fn, *args, byref(value)) else None
|
|
207
|
+
|
|
208
|
+
def _memory(self):
|
|
209
|
+
"""(used, total) bytes. v2 (R510+) leaves out driver-reserved memory, like nvidia-smi."""
|
|
210
|
+
if hasattr(self.nv, "nvmlDeviceGetMemoryInfo_v2"):
|
|
211
|
+
mem = MemoryV2(version=MEMORY_V2)
|
|
212
|
+
if self._ok("nvmlDeviceGetMemoryInfo_v2", byref(mem)):
|
|
213
|
+
return mem.used, mem.total
|
|
214
|
+
mem = Memory()
|
|
215
|
+
return (mem.used, mem.total) if self._ok("nvmlDeviceGetMemoryInfo", byref(mem)) else (None, None)
|
|
216
|
+
|
|
217
|
+
def _temperature(self):
|
|
218
|
+
"""nvmlDeviceGetTemperature is deprecated; its replacement only exists on R565+ drivers."""
|
|
219
|
+
if hasattr(self.nv, "nvmlDeviceGetTemperatureV"):
|
|
220
|
+
temp = TemperatureV1(version=TEMPERATURE_V1, sensorType=NVML_TEMPERATURE_GPU)
|
|
221
|
+
if self._ok("nvmlDeviceGetTemperatureV", byref(temp)):
|
|
222
|
+
return temp.temperature
|
|
223
|
+
return self._uint("nvmlDeviceGetTemperature", NVML_TEMPERATURE_GPU)
|
|
224
|
+
|
|
225
|
+
def _gpm_samples(self):
|
|
226
|
+
"""Two GPM sample buffers, or None where GPM is unavailable (it needs Hopper or newer)."""
|
|
227
|
+
support = GpmSupport(version=1)
|
|
228
|
+
if not (hasattr(self.nv, "nvmlGpmQueryDeviceSupport")
|
|
229
|
+
and self._ok("nvmlGpmQueryDeviceSupport", byref(support)) and support.isSupportedDevice):
|
|
230
|
+
return None
|
|
231
|
+
samples = [c_void_p(), c_void_p()]
|
|
232
|
+
if any(self.nv.nvmlGpmSampleAlloc(byref(sample)) != NVML_SUCCESS for sample in samples):
|
|
233
|
+
return None
|
|
234
|
+
# Take the first sample now, on the main thread: NVML crashes when a device's
|
|
235
|
+
# first GPM sample is taken from several threads at once.
|
|
236
|
+
if self.nv.nvmlGpmSampleGet(self.handle, samples[0]) != NVML_SUCCESS:
|
|
237
|
+
return None
|
|
238
|
+
return samples # [previous, current], kept for the whole run
|
|
239
|
+
|
|
240
|
+
def _activity(self):
|
|
241
|
+
"""GPM_METRICS values between this poll and the previous one (Nones without GPM)."""
|
|
242
|
+
none = (None,) * len(GPM_METRICS)
|
|
243
|
+
if self._gpm is None or self.nv.nvmlGpmSampleGet(self.handle, self._gpm[1]) != NVML_SUCCESS:
|
|
244
|
+
return none
|
|
245
|
+
previous, current = self._gpm
|
|
246
|
+
self._gpm.reverse() # this sample is the baseline for the next poll
|
|
247
|
+
query = GpmMetricsGet(version=1, numMetrics=len(GPM_METRICS), sample1=previous, sample2=current)
|
|
248
|
+
for metric, metric_id in zip(query.metrics, GPM_METRICS):
|
|
249
|
+
metric.metricId = metric_id
|
|
250
|
+
if self.nv.nvmlGpmMetricsGet(byref(query)) != NVML_SUCCESS:
|
|
251
|
+
return none
|
|
252
|
+
return tuple(None if m.nvmlReturn != NVML_SUCCESS else m.value for m in query.metrics)
|
|
253
|
+
|
|
254
|
+
def _pcie(self, counter):
|
|
255
|
+
"""PCIe bytes/s; this query samples a counter for 20 ms, so it is only a fallback for GPM."""
|
|
256
|
+
kb = self._uint("nvmlDeviceGetPcieThroughput", counter)
|
|
257
|
+
return None if kb is None else kb * 1024
|
|
258
|
+
|
|
259
|
+
def _slowdown(self):
|
|
260
|
+
"""(label, colour) for why the clock is held back, or None."""
|
|
261
|
+
fn = ("nvmlDeviceGetCurrentClocksEventReasons" if hasattr(self.nv, "nvmlDeviceGetCurrentClocksEventReasons")
|
|
262
|
+
else "nvmlDeviceGetCurrentClocksThrottleReasons") # the pre-R535 name
|
|
263
|
+
reasons = c_ulonglong()
|
|
264
|
+
if not self._ok(fn, byref(reasons)):
|
|
265
|
+
return None
|
|
266
|
+
return next(((label, colour) for bits, label, colour in SLOWDOWNS if reasons.value & bits), None)
|
|
267
|
+
|
|
268
|
+
def _processes(self):
|
|
269
|
+
infos, count = (ProcessInfo * MAX_PROCESSES)(), c_uint(MAX_PROCESSES)
|
|
270
|
+
if not (hasattr(self.nv, "nvmlDeviceGetComputeRunningProcesses_v3") and
|
|
271
|
+
self._ok("nvmlDeviceGetComputeRunningProcesses_v3", byref(count), infos)):
|
|
272
|
+
return []
|
|
273
|
+
running = infos[:count.value]
|
|
274
|
+
self._names = {p.pid: self._names.get(p.pid) or (self._process_name(p.pid), process_owner(p.pid))
|
|
275
|
+
for p in running}
|
|
276
|
+
procs = []
|
|
277
|
+
for p in running:
|
|
278
|
+
name, owner = self._names[p.pid]
|
|
279
|
+
procs.append((name, None if p.usedGpuMemory == NVML_VALUE_NOT_AVAILABLE else p.usedGpuMemory / 2**30, owner))
|
|
280
|
+
return procs
|
|
281
|
+
|
|
282
|
+
def _process_name(self, pid):
|
|
283
|
+
"""Short name; for a Python interpreter, the script or module it runs (train.py, torch.distributed.run)."""
|
|
284
|
+
try:
|
|
285
|
+
with open("/proc/{}/cmdline".format(pid), "rb") as f:
|
|
286
|
+
argv = [a for a in f.read().decode("utf-8", "replace").split("\0") if a]
|
|
287
|
+
except OSError: # not Linux, or the process already exited
|
|
288
|
+
argv = []
|
|
289
|
+
if not argv:
|
|
290
|
+
name = ctypes.create_string_buffer(256)
|
|
291
|
+
if self.nv.nvmlSystemGetProcessName(pid, name, 256) != NVML_SUCCESS:
|
|
292
|
+
return str(pid)
|
|
293
|
+
argv = [name.value.decode("utf-8", "replace")]
|
|
294
|
+
exe = os.path.basename(argv[0])
|
|
295
|
+
if exe.startswith("python"):
|
|
296
|
+
args = iter(argv[1:])
|
|
297
|
+
for arg in args:
|
|
298
|
+
if arg == "-m":
|
|
299
|
+
return next(args, exe)
|
|
300
|
+
if not arg.startswith("-"):
|
|
301
|
+
return os.path.basename(arg)
|
|
302
|
+
return exe
|
|
303
|
+
|
|
304
|
+
def poll(self):
|
|
305
|
+
util = Utilization()
|
|
306
|
+
util = util.gpu if self._ok("nvmlDeviceGetUtilizationRates", byref(util)) else None
|
|
307
|
+
used, total = self._memory()
|
|
308
|
+
power = self._uint("nvmlDeviceGetPowerUsage") # mW
|
|
309
|
+
limit = self._uint("nvmlDeviceGetEnforcedPowerLimit") # mW
|
|
310
|
+
cores, tensor, gpm_tx, gpm_rx = self._activity()
|
|
311
|
+
self.now = Sample(
|
|
312
|
+
util=util,
|
|
313
|
+
temp=self._temperature(),
|
|
314
|
+
fan=self._uint("nvmlDeviceGetFanSpeed"),
|
|
315
|
+
power=None if power is None else power / 1000,
|
|
316
|
+
power_limit=None if limit is None else limit / 1000,
|
|
317
|
+
clock=self._uint("nvmlDeviceGetClockInfo", NVML_CLOCK_GRAPHICS),
|
|
318
|
+
mem_used=None if used is None else used / 2**30,
|
|
319
|
+
mem_total=None if total is None else total / 2**30,
|
|
320
|
+
tx=self._pcie(NVML_PCIE_UTIL_TX_BYTES) if gpm_tx is None else gpm_tx * 2**20,
|
|
321
|
+
rx=self._pcie(NVML_PCIE_UTIL_RX_BYTES) if gpm_rx is None else gpm_rx * 2**20,
|
|
322
|
+
pcie_width=self._uint("nvmlDeviceGetCurrPcieLinkWidth"),
|
|
323
|
+
cores=cores,
|
|
324
|
+
tensor=tensor,
|
|
325
|
+
slowdown=self._slowdown(),
|
|
326
|
+
processes=self._processes(),
|
|
327
|
+
)
|
|
328
|
+
self.history.append(util or 0)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def versions(nv):
|
|
332
|
+
"""'driver 580.173.02 CUDA 13.0'; fixed while the program runs, so read once."""
|
|
333
|
+
driver, cuda, parts = ctypes.create_string_buffer(NVML_SYSTEM_DRIVER_VERSION_BUFFER_SIZE), c_int(), []
|
|
334
|
+
if nv.nvmlSystemGetDriverVersion(driver, NVML_SYSTEM_DRIVER_VERSION_BUFFER_SIZE) == NVML_SUCCESS:
|
|
335
|
+
parts.append("driver " + driver.value.decode())
|
|
336
|
+
if nv.nvmlSystemGetCudaDriverVersion_v2(byref(cuda)) == NVML_SUCCESS: # e.g. 13000 -> 13.0
|
|
337
|
+
parts.append("CUDA {}.{}".format(cuda.value // 1000, cuda.value % 1000 // 10))
|
|
338
|
+
return " ".join(parts)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def process_owner(pid):
|
|
342
|
+
"""Account running `pid`; None for our own processes or where there is no /proc (Windows)."""
|
|
343
|
+
try:
|
|
344
|
+
import pwd
|
|
345
|
+
with open("/proc/{}/status".format(pid)) as f:
|
|
346
|
+
uid = next(int(line.split()[1]) for line in f if line.startswith("Uid:"))
|
|
347
|
+
except (ImportError, OSError, StopIteration):
|
|
348
|
+
return None
|
|
349
|
+
if uid == os.getuid():
|
|
350
|
+
return None
|
|
351
|
+
try:
|
|
352
|
+
return pwd.getpwuid(uid).pw_name
|
|
353
|
+
except KeyError: # no passwd entry, e.g. inside a container
|
|
354
|
+
return str(uid)
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def open_gpus(nv, wanted=None):
|
|
358
|
+
"""GPUs whose index is in `wanted` (all when None)."""
|
|
359
|
+
count = c_uint()
|
|
360
|
+
check(nv, nv.nvmlDeviceGetCount_v2(byref(count)))
|
|
361
|
+
gpus = []
|
|
362
|
+
for index in range(count.value):
|
|
363
|
+
if wanted is not None and index not in wanted:
|
|
364
|
+
continue
|
|
365
|
+
handle = c_void_p()
|
|
366
|
+
# The count includes GPUs we may not open (cgroups, /dev/nvidiaN permissions): skip those.
|
|
367
|
+
if nv.nvmlDeviceGetHandleByIndex_v2(index, byref(handle)) == NVML_SUCCESS:
|
|
368
|
+
gpus.append(Gpu(nv, index, handle))
|
|
369
|
+
return gpus
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
# ── rendering ────────────────────────────────────────────────────────────────
|
|
373
|
+
# A line is a list of (text, style) segments; every character is one cell wide.
|
|
374
|
+
|
|
375
|
+
# A horizontal rule across the stats column, joined to the borders.
|
|
376
|
+
RULE = [(" ├" + "─" * (INFO_W + 2) + "┤", DIM)]
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def width_of(line):
|
|
380
|
+
return sum(len(text) for text, _ in line)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def clip(line, width):
|
|
384
|
+
"""`line` cut to at most `width` cells."""
|
|
385
|
+
out = []
|
|
386
|
+
for text, style in line:
|
|
387
|
+
if width <= 0:
|
|
388
|
+
break
|
|
389
|
+
out.append((text[:width], style))
|
|
390
|
+
width -= len(text)
|
|
391
|
+
return out
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def pad(line, width):
|
|
395
|
+
"""`line` padded with spaces to `width` cells."""
|
|
396
|
+
return line + [(" " * (width - width_of(line)), "")]
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def graph(history, width, height):
|
|
400
|
+
"""Stacked block chart, newest sample on the right, top row first.
|
|
401
|
+
|
|
402
|
+
Each cell takes the colour of the level its top reaches: full cells shade row by row,
|
|
403
|
+
and the ragged top edge shows the exact colour of each value.
|
|
404
|
+
"""
|
|
405
|
+
vals = list(history)[-width:] if width > 0 else []
|
|
406
|
+
steps = height * 8
|
|
407
|
+
# Any non-zero value gets at least one sub-level so light load stays visible.
|
|
408
|
+
levels = [0] * (width - len(vals)) + [max(round(v * steps / 100), 1 if v else 0) for v in vals]
|
|
409
|
+
rows = []
|
|
410
|
+
for r in reversed(range(height)):
|
|
411
|
+
fills = [min(8, max(0, lv - r * 8)) for lv in levels]
|
|
412
|
+
# Blank and full cells share the row colour so they join into long runs.
|
|
413
|
+
cells = [(BLOCKS[f], heat((r * 8 + (f or 8)) / steps)) for f in fills]
|
|
414
|
+
rows.append([("".join(text for text, _ in run), style)
|
|
415
|
+
for style, run in itertools.groupby(cells, key=lambda cell: cell[1])])
|
|
416
|
+
return rows
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def row(left, right=()):
|
|
420
|
+
"""One stats line: `left`, then the `right` segments flush right; exactly INFO_W cells."""
|
|
421
|
+
room = INFO_W - width_of(right)
|
|
422
|
+
return pad(clip(left, room - 1 if right else room), room) + list(right)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def share(part, whole):
|
|
426
|
+
"""A 4-cell percentage, coloured by its value."""
|
|
427
|
+
if part is None or not whole:
|
|
428
|
+
return [(" -", "")]
|
|
429
|
+
return [("{:.0f}%".format(100 * part / whole).rjust(4), heat(part / whole))]
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def num(value, fmt):
|
|
433
|
+
return "-" if value is None else fmt.format(value)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def info(gpu, height):
|
|
437
|
+
s = gpu.now
|
|
438
|
+
total = num(s.mem_total, "{:.1f}")
|
|
439
|
+
busy = [("GPU ", "")] + share(s.util, 100)
|
|
440
|
+
if gpu.has_activity: # "cores" = share of SMs at work, "tensor" = Tensor Core activity
|
|
441
|
+
busy += [(" cores " + share(s.cores, 100)[0][0] + " tensor " + share(s.tensor, 100)[0][0], FAINT)]
|
|
442
|
+
# A link's top speed: its generation's per-lane rate times the lanes it runs on now.
|
|
443
|
+
lanes = s.pcie_width or gpu.pcie_max_width
|
|
444
|
+
top = PCIE_LANE_GBS.get(gpu.pcie_max_gen, 0) * (lanes or 0)
|
|
445
|
+
cap = (" / {:.0f}".format(top), WARN if degraded(gpu) else "") if top else ("", "")
|
|
446
|
+
|
|
447
|
+
def link(label, rate): # " CPU -> GPU ... 0.16 / 63 GB/s", flush right like MEM's used / total
|
|
448
|
+
return row([(label, "")], [("-" if rate is None else "{:.2f}".format(rate / 1e9), ""), cap, (" GB/s", "")])
|
|
449
|
+
|
|
450
|
+
gpu_row = row(busy)
|
|
451
|
+
mem_row = row([("MEM ", "")] + share(s.mem_used, s.mem_total),
|
|
452
|
+
[("{} / {} GiB".format(num(s.mem_used, "{:.1f}"), total), "")])
|
|
453
|
+
title, to_gpu, to_cpu = row([("PCIe transfer", "")]), link(" CPU -> GPU", s.rx), link(" GPU -> CPU", s.tx)
|
|
454
|
+
# The rule (None) after MEM and the PCIe title only appear when there is room.
|
|
455
|
+
if height >= 6:
|
|
456
|
+
return [gpu_row, mem_row, None, title, to_gpu, to_cpu] + [row([])] * (height - 6)
|
|
457
|
+
if height == 5:
|
|
458
|
+
return [gpu_row, mem_row, title, to_gpu, to_cpu]
|
|
459
|
+
return [gpu_row, mem_row, to_gpu, to_cpu][:height]
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def degraded(gpu):
|
|
463
|
+
"""True when the PCIe link runs on fewer lanes than card and slot allow, e.g. a loose card."""
|
|
464
|
+
s = gpu.now
|
|
465
|
+
return bool(s.pcie_width and gpu.pcie_max_width and s.pcie_width < gpu.pcie_max_width)
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def panel(gpu, width, height):
|
|
469
|
+
graph_w = max(0, width - CHROME_W - INFO_W)
|
|
470
|
+
s = gpu.now
|
|
471
|
+
split = 2 + graph_w + 1 # column of the graph | stats divider
|
|
472
|
+
# Both sides as (segments, rank); when space runs out the lowest rank goes first:
|
|
473
|
+
# fan, name, power, clock, temperature, PCIe warning, slowdown warning. The GPU number stays.
|
|
474
|
+
# Fixed widths keep things from shifting.
|
|
475
|
+
limit = num(s.power_limit, "{:.0f}") # power padded to the limit's width
|
|
476
|
+
label = [([("GPU {}".format(gpu.index), BOLD)], None), ([(" " + gpu.name, "")], 2),
|
|
477
|
+
([(" │ ", DIM), ("{} / {} W".format(num(s.power, "{:.0f}").rjust(len(limit)), limit), "")], 3)]
|
|
478
|
+
parts = [] if s.slowdown is None else [([s.slowdown], 9)]
|
|
479
|
+
if degraded(gpu):
|
|
480
|
+
parts += [([("PCIe DEGRADED: x{} -> x{}".format(gpu.pcie_max_width, s.pcie_width), WARN)], 8)]
|
|
481
|
+
parts += [] if s.temp is None else [([("{:>3}°C".format(s.temp), TEMP[min(max(s.temp, 0), 100)])], 5)]
|
|
482
|
+
parts += [] if s.fan is None else [([("FAN {:>3}%".format(s.fan), "")], 1)]
|
|
483
|
+
parts += [] if s.clock is None else [([("{:>4} MHz".format(s.clock), s.slowdown[1] if s.slowdown else "")], 4)]
|
|
484
|
+
top = edge(width, label, parts)
|
|
485
|
+
bottom = bottom_edge(width, split, process_label(s.processes, split - 4))
|
|
486
|
+
body = [[("│ ", DIM)] + g + (RULE if i is None
|
|
487
|
+
else [(" │ ", DIM)] + i + [(" │", DIM)])
|
|
488
|
+
for g, i in zip(graph(gpu.history, graph_w, height), info(gpu, height))]
|
|
489
|
+
return [top] + body + [bottom]
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def edge(width, label, parts):
|
|
493
|
+
"""Top border, "╭─ label ───── parts ─╮", fitted to `width`.
|
|
494
|
+
|
|
495
|
+
`label` and `parts` are (segments, rank) groups. While they do not fit, the group with the
|
|
496
|
+
lowest rank goes, whichever side it is on, keeping the others in order; a group shows whole
|
|
497
|
+
or not at all. The first label group (rank None: the GPU number) always stays, cut if need be.
|
|
498
|
+
"""
|
|
499
|
+
label, parts = list(label), list(parts)
|
|
500
|
+
while True:
|
|
501
|
+
note = [seg for i, (group, _) in enumerate(parts) for seg in ([(" ", "")] if i else []) + group]
|
|
502
|
+
tail = ([(" ", "")] + note + [(" ", "")] if note else []) + [("─╮", DIM)]
|
|
503
|
+
kept = [seg for group, _ in label for seg in group]
|
|
504
|
+
if width_of(kept) + width_of(tail) + 5 <= width: # 5 = "╭─", a space either side of the label, one "─"
|
|
505
|
+
break
|
|
506
|
+
droppable = [(rank, side, group) for side in (label, parts) for group, rank in side if rank is not None]
|
|
507
|
+
if not droppable:
|
|
508
|
+
break
|
|
509
|
+
_, side, group = min(droppable, key=lambda item: item[0])
|
|
510
|
+
side.remove(next(item for item in side if item[0] is group))
|
|
511
|
+
kept = clip(kept, width - 5 - width_of(tail))
|
|
512
|
+
head = [("╭─", DIM)] + ([(" ", "")] + kept + [(" ", "")] if kept else [])
|
|
513
|
+
return head + [("─" * max(0, width - width_of(head) - width_of(tail)), DIM)] + tail
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
def process_label(processes, room):
|
|
517
|
+
"""Processes grouped by owner, owners and processes by memory, biggest first:
|
|
518
|
+
"train.py 15.1G eval.py 0.5G kim: bench.py 3.2G" (our own processes carry no owner).
|
|
519
|
+
As many whole entries as fit in `room` cells, then "+N" for the rest."""
|
|
520
|
+
def mem(process):
|
|
521
|
+
return process[1] or 0
|
|
522
|
+
groups = {}
|
|
523
|
+
for process in sorted(processes, key=mem, reverse=True):
|
|
524
|
+
groups.setdefault(process[2], []).append(process)
|
|
525
|
+
ordered = sorted(groups.values(), key=lambda group: sum(map(mem, group)), reverse=True)
|
|
526
|
+
entries = []
|
|
527
|
+
for group in ordered:
|
|
528
|
+
for j, (name, used, owner) in enumerate(group):
|
|
529
|
+
gap = "" if not entries else " " if j else " "
|
|
530
|
+
entries.append([(gap, "")] + ([(owner + ": ", DIM)] if owner and not j else []) + [(name, PROC)]
|
|
531
|
+
+ ([] if used is None else [(" {:.1f}G".format(used), DIM)]))
|
|
532
|
+
label = []
|
|
533
|
+
for i, entry in enumerate(entries):
|
|
534
|
+
rest = len(entries) - i - 1
|
|
535
|
+
if width_of(label + entry) + (len(" +{}".format(rest)) if rest else 0) > room:
|
|
536
|
+
# The first entry is always shown, cut if need be; later ones collapse into the count.
|
|
537
|
+
return clip(entry, room) if not label else label + [(" +{}".format(rest + 1), DIM)]
|
|
538
|
+
label += entry
|
|
539
|
+
return label
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
def bottom_edge(width, split, label):
|
|
543
|
+
"""Bottom border with `label` flush right against column `split`, the graph | stats divider."""
|
|
544
|
+
label = clip(label, split - 4) # keep "╰─" and a space on each side
|
|
545
|
+
middle = [(" ", "")] + label + [(" ", "")] if label else []
|
|
546
|
+
return ([("╰" + "─" * max(0, split - 1 - width_of(middle)), DIM)] + middle
|
|
547
|
+
+ [("─" * max(0, width - split - 1) + "╯", DIM)])
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def render(gpus, width, height):
|
|
551
|
+
"""The whole screen: exactly `height` lines of exactly `width` cells."""
|
|
552
|
+
# Two columns only when one column cannot show every GPU at full height.
|
|
553
|
+
two_fit = width // 2 >= CHROME_W + INFO_W + MIN_GRAPH_W
|
|
554
|
+
cols = 2 if two_fit and len(gpus) * (MAX_INNER_H + 2) > height else 1
|
|
555
|
+
rows = math.ceil(len(gpus) / cols)
|
|
556
|
+
inner = max(1, min(MAX_INNER_H, height // rows - 2))
|
|
557
|
+
lines = []
|
|
558
|
+
for r in range(rows):
|
|
559
|
+
panels = [panel(g, width // cols, inner) for g in gpus[r * cols:(r + 1) * cols]]
|
|
560
|
+
lines += [[seg for part in parts for seg in part] for parts in zip(*panels)]
|
|
561
|
+
lines = [pad(clip(line, width), width) for line in lines[:height]]
|
|
562
|
+
return lines + [[(" " * width, "")]] * (height - len(lines))
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def spread(width, left, right):
|
|
566
|
+
"""`left`, then `right` flush right: one line of exactly `width` cells."""
|
|
567
|
+
room = width - width_of(right)
|
|
568
|
+
return clip(pad(clip(left, room - 1), room) + right, width)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def header(width, interval, driver):
|
|
572
|
+
"""Top line: name, version, host, local time | driver, refresh interval."""
|
|
573
|
+
now = time.time()
|
|
574
|
+
stamp = time.strftime("%Y-%m-%d %a %H:%M:%S", time.localtime(now)) + ".{:02d}".format(int(now % 1 * 100))
|
|
575
|
+
left = [("nvmon", BOLD), (" " + __version__, DIM), (" " + socket.gethostname(), ""), (" " + stamp, "")]
|
|
576
|
+
right = [("refresh ", DIM), ("{:g}s".format(interval), "")]
|
|
577
|
+
if driver and width_of(left) + 2 + len(driver) + 3 + width_of(right) <= width: # dropped when narrow
|
|
578
|
+
right = [(driver + " ", DIM)] + right
|
|
579
|
+
return spread(width, left, right)
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
def footer(width):
|
|
583
|
+
"""Bottom line: how to quit."""
|
|
584
|
+
return spread(width, [], [("Esc / q quit", DIM)])
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def paint(lines):
|
|
588
|
+
"""Join lines into one string, sending a style code only where the style changes."""
|
|
589
|
+
out, current = [], ""
|
|
590
|
+
for i, line in enumerate(lines):
|
|
591
|
+
if i:
|
|
592
|
+
out.append("\r\n")
|
|
593
|
+
for text, style in line:
|
|
594
|
+
if style != current:
|
|
595
|
+
# One colour replaces another directly; anything else (bold, plain) needs a reset first.
|
|
596
|
+
colours = style.startswith("\x1b[38") and current.startswith("\x1b[38")
|
|
597
|
+
out.append(style if colours else RESET + style)
|
|
598
|
+
current = style
|
|
599
|
+
out.append(text)
|
|
600
|
+
return "".join(out) + RESET
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
# ── terminal ─────────────────────────────────────────────────────────────────
|
|
604
|
+
|
|
605
|
+
class Screen:
|
|
606
|
+
"""Alternate screen, hidden cursor, no auto-wrap, unbuffered keys; all restored on exit."""
|
|
607
|
+
|
|
608
|
+
def __enter__(self):
|
|
609
|
+
self._keys = sys.stdin.isatty()
|
|
610
|
+
self._restore = [enable_vt(), raw_keys() if self._keys else (lambda: None)]
|
|
611
|
+
self._write("\x1b[?1049h\x1b[?25l\x1b[?7l")
|
|
612
|
+
return self
|
|
613
|
+
|
|
614
|
+
def __exit__(self, *exc):
|
|
615
|
+
self._write(RESET + "\x1b[?7h\x1b[?25h\x1b[?1049l")
|
|
616
|
+
for undo in self._restore:
|
|
617
|
+
undo()
|
|
618
|
+
|
|
619
|
+
def wait(self, seconds):
|
|
620
|
+
"""Sleep for `seconds`; returns True early if q or Esc is pressed."""
|
|
621
|
+
end = time.monotonic() + seconds
|
|
622
|
+
if not self._keys:
|
|
623
|
+
time.sleep(seconds)
|
|
624
|
+
return False
|
|
625
|
+
if os.name == "nt":
|
|
626
|
+
import msvcrt
|
|
627
|
+
while True:
|
|
628
|
+
while msvcrt.kbhit():
|
|
629
|
+
key = msvcrt.getwch()
|
|
630
|
+
if key in ("\x00", "\xe0"): # arrows and function keys come as two characters
|
|
631
|
+
msvcrt.getwch()
|
|
632
|
+
elif key in ("q", "Q", "\x1b"):
|
|
633
|
+
return True
|
|
634
|
+
left = end - time.monotonic()
|
|
635
|
+
if left <= 0:
|
|
636
|
+
return False
|
|
637
|
+
time.sleep(min(left, 0.05))
|
|
638
|
+
import select
|
|
639
|
+
fd = sys.stdin.fileno()
|
|
640
|
+
while True:
|
|
641
|
+
left = end - time.monotonic()
|
|
642
|
+
if left <= 0 or not select.select([fd], [], [], left)[0]:
|
|
643
|
+
return False
|
|
644
|
+
keys = os.read(fd, 64)
|
|
645
|
+
# A lone ESC is the Esc key; arrows and the like arrive as ESC [ ... sequences.
|
|
646
|
+
if b"q" in keys or b"Q" in keys or keys == b"\x1b":
|
|
647
|
+
return True
|
|
648
|
+
if not keys: # stdin closed: nothing more to read, just sleep
|
|
649
|
+
time.sleep(max(0, end - time.monotonic()))
|
|
650
|
+
return False
|
|
651
|
+
|
|
652
|
+
def draw(self, lines):
|
|
653
|
+
# 2026 = synchronized output: terminals that know it swap the frame in at once.
|
|
654
|
+
self._write("\x1b[?2026h\x1b[H" + paint(lines) + "\x1b[?2026l")
|
|
655
|
+
|
|
656
|
+
@staticmethod
|
|
657
|
+
def _write(text):
|
|
658
|
+
# Raw UTF-8 bytes: works even where Python picked an ASCII stdout (LANG=C on 3.6).
|
|
659
|
+
sys.stdout.buffer.write(text.encode("utf-8"))
|
|
660
|
+
sys.stdout.buffer.flush()
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
def enable_vt():
|
|
664
|
+
"""Make the Windows console interpret ANSI codes; returns a function that undoes it."""
|
|
665
|
+
if os.name != "nt":
|
|
666
|
+
return lambda: None
|
|
667
|
+
kernel32 = ctypes.windll.kernel32
|
|
668
|
+
kernel32.GetStdHandle.restype = c_void_p
|
|
669
|
+
handle = c_void_p(kernel32.GetStdHandle(-11)) # STD_OUTPUT_HANDLE
|
|
670
|
+
mode = ctypes.c_ulong()
|
|
671
|
+
if not kernel32.GetConsoleMode(handle, byref(mode)):
|
|
672
|
+
return lambda: None
|
|
673
|
+
# ENABLE_VIRTUAL_TERMINAL_PROCESSING | DISABLE_NEWLINE_AUTO_RETURN (VT-style end of line)
|
|
674
|
+
kernel32.SetConsoleMode(handle, mode.value | 0x0004 | 0x0008)
|
|
675
|
+
return lambda: kernel32.SetConsoleMode(handle, mode.value)
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def raw_keys():
|
|
679
|
+
"""Deliver key presses at once and without echo; returns a function that undoes it."""
|
|
680
|
+
if os.name == "nt":
|
|
681
|
+
return lambda: None # msvcrt already reads single keys without echo
|
|
682
|
+
import termios
|
|
683
|
+
import tty
|
|
684
|
+
fd = sys.stdin.fileno()
|
|
685
|
+
saved = termios.tcgetattr(fd)
|
|
686
|
+
tty.setcbreak(fd) # Ctrl+C keeps working: cbreak leaves signal keys alone
|
|
687
|
+
return lambda: termios.tcsetattr(fd, termios.TCSADRAIN, saved)
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def gpu_list(spec):
|
|
691
|
+
"""argparse type: '0,2,4-7' -> {0, 2, 4, 5, 6, 7}."""
|
|
692
|
+
picked = set()
|
|
693
|
+
try:
|
|
694
|
+
for part in spec.split(","):
|
|
695
|
+
first, _, last = part.strip().partition("-")
|
|
696
|
+
picked.update(range(int(first), int(last or first) + 1))
|
|
697
|
+
except ValueError:
|
|
698
|
+
picked = set()
|
|
699
|
+
if not picked:
|
|
700
|
+
raise argparse.ArgumentTypeError("expected GPU numbers such as 0,2,4-7, got {!r}".format(spec))
|
|
701
|
+
return picked
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def main():
|
|
705
|
+
parser = argparse.ArgumentParser(prog="nvmon", description="Compact btop-style NVIDIA GPU monitor.")
|
|
706
|
+
parser.add_argument("-i", "--interval", type=float, default=0.5, metavar="SEC",
|
|
707
|
+
help="seconds between updates (default: 0.5)")
|
|
708
|
+
parser.add_argument("-g", "--gpus", type=gpu_list, metavar="LIST",
|
|
709
|
+
help="only these GPUs, e.g. 0,2,4-7 (default: all)")
|
|
710
|
+
parser.add_argument("-V", "--version", action="version", version="nvmon " + __version__)
|
|
711
|
+
args = parser.parse_args()
|
|
712
|
+
if not 0 < args.interval < math.inf: # also rejects NaN
|
|
713
|
+
parser.error("--interval must be a positive number of seconds")
|
|
714
|
+
if not sys.stdout.isatty():
|
|
715
|
+
sys.exit("nvmon: output is not a terminal (over ssh, use: ssh -t HOST nvmon)")
|
|
716
|
+
|
|
717
|
+
try:
|
|
718
|
+
nv = load_nvml()
|
|
719
|
+
except NvmlError as e:
|
|
720
|
+
sys.exit("nvmon: {}".format(e))
|
|
721
|
+
try:
|
|
722
|
+
gpus = open_gpus(nv, args.gpus)
|
|
723
|
+
missing = sorted((args.gpus or set()) - {g.index for g in gpus})
|
|
724
|
+
if missing:
|
|
725
|
+
sys.exit("nvmon: GPU {} not found or not accessible".format(", ".join(map(str, missing))))
|
|
726
|
+
if not gpus:
|
|
727
|
+
sys.exit("nvmon: no accessible NVIDIA GPU")
|
|
728
|
+
driver = versions(nv)
|
|
729
|
+
signal.signal(signal.SIGTERM, lambda signum, frame: sys.exit(128 + signum))
|
|
730
|
+
with ThreadPoolExecutor(len(gpus)) as pool, Screen() as screen:
|
|
731
|
+
deadline = time.monotonic()
|
|
732
|
+
while True:
|
|
733
|
+
list(pool.map(Gpu.poll, gpus))
|
|
734
|
+
width, height = os.get_terminal_size()
|
|
735
|
+
screen.draw([header(width, args.interval, driver)] + render(gpus, width, height - 2)
|
|
736
|
+
+ [footer(width)])
|
|
737
|
+
# Fixed-rate ticks; a late tick restarts the schedule instead of bursting.
|
|
738
|
+
now = time.monotonic()
|
|
739
|
+
deadline = max(deadline + args.interval, now)
|
|
740
|
+
if screen.wait(deadline - now):
|
|
741
|
+
break
|
|
742
|
+
except KeyboardInterrupt:
|
|
743
|
+
pass
|
|
744
|
+
except NvmlError as e:
|
|
745
|
+
sys.exit("nvmon: {}".format(e))
|
|
746
|
+
finally:
|
|
747
|
+
nv.nvmlShutdown()
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
if __name__ == "__main__":
|
|
751
|
+
main()
|