nodeview-slurm 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nodeview/snapshot.py ADDED
@@ -0,0 +1,72 @@
1
+ """Static snapshot of the current state, for --once.
2
+
3
+ Handy when you don't want an interactive interface: inside a script, in a
4
+ pipe, in a .bashrc, or on a terminal too small for the TUI.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from datetime import datetime
9
+
10
+ from rich.console import Console
11
+ from rich.panel import Panel
12
+
13
+ from . import analysis as an
14
+ from . import config
15
+ from . import i18n
16
+ from .i18n import t
17
+ from . import palette
18
+ from . import render
19
+ from .source import fetch
20
+
21
+
22
+ def _panel(content, title: str) -> Panel:
23
+ return Panel(content, title=f"[bold cyan]{title}[/]", title_align="left",
24
+ border_style="cyan", padding=(0, 1))
25
+
26
+
27
+ def _saved_palette():
28
+ """The static snapshot also follows the theme chosen in the TUI: if you
29
+ picked a light theme, printing with the dark palette would be inconsistent."""
30
+ name = config.load().get('theme')
31
+ if isinstance(name, str):
32
+ try:
33
+ from textual.theme import BUILTIN_THEMES
34
+ theme = BUILTIN_THEMES.get(name)
35
+ if theme is not None:
36
+ return palette.from_theme(theme)
37
+ except Exception:
38
+ pass
39
+ return palette.default()
40
+
41
+
42
+ def print_once(width=None, cluster=False) -> int:
43
+ # the static snapshot also follows the language chosen in the TUI
44
+ lang = config.load().get('lang')
45
+ if isinstance(lang, str):
46
+ i18n.use(lang)
47
+ palette.use(_saved_palette())
48
+ console = Console(width=width)
49
+ try:
50
+ snap = fetch()
51
+ except Exception as exc:
52
+ console.print(f"[bold red]{t('once.error')}[/] {exc}")
53
+ return 1
54
+
55
+ when = datetime.fromtimestamp(snap.taken_at).strftime('%H:%M:%S')
56
+ console.print("[bold]nodeview[/] [dim]"
57
+ + t('once.header', me=snap.me, nodes=len(snap.nodes), when=when,
58
+ secs=f"{snap.duration:.2f}") + "[/]")
59
+ if cluster:
60
+ occupants = palette.Occupants(an.occupant_counts(snap), me=snap.me)
61
+ console.print(_panel(render.totals_strip(snap), t('card.totals')))
62
+ console.print(_panel(render.node_table(snap, occupants), t('card.nodes')))
63
+ console.print(_panel(render.breakdown(snap, occupants), t('card.usage')))
64
+ console.print(render.legend(occupants))
65
+ console.print(_panel(render.specs_table(snap), t('card.specs')))
66
+ else:
67
+ console.print(_panel(render.my_jobs(snap), t('card.my_jobs')))
68
+ console.print(_panel(render.where_do_i_fit(snap), t('card.fit')))
69
+ console.print(_panel(render.cluster(snap), t('card.cluster')))
70
+ from .profile import read_profile
71
+ console.print(_panel(render.my_account(read_profile()), t('card.account')))
72
+ return 0
nodeview/source.py ADDED
@@ -0,0 +1,306 @@
1
+ """Reading Slurm state.
2
+
3
+ Two sources, each for what it does best:
4
+
5
+ - `scontrol -d -o show node` (in nodes.py) for nodes: gives everything the
6
+ cluster exposes, including the actual GPU model, the indices of busy GPUs,
7
+ allocated RAM versus truly free RAM, and CPU load;
8
+ - **nodeocc**'s readers for jobs, reservations and QOS limits, which already
9
+ know how to interpret this cluster's quirks.
10
+
11
+ The result is our own dataclasses: the rest of the app never touches
12
+ nodeocc's internal classes.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import math
17
+ import os
18
+ import pwd
19
+ import time
20
+ from dataclasses import dataclass, field
21
+ from typing import Optional
22
+
23
+ from .i18n import t
24
+ from .nodes import NodeSpec, read_gpu_owners, read_nodes, short_name
25
+
26
+
27
+ class SourceUnavailable(RuntimeError):
28
+ """nodeocc is not installed or cannot be imported."""
29
+
30
+
31
+ class _ReaderShim:
32
+ """Silent stand-in for nodeocc's Singleton.
33
+
34
+ The readers only use the instance for err()/log()/timeme(), but the real
35
+ constructor also runs master election, opens UDP sockets and writes
36
+ portfiles in /tmp. None of that is needed here, and joining in would
37
+ disturb an already running nodeocc: we install this in its place.
38
+ """
39
+
40
+ def __getattr__(self, name):
41
+ return lambda *args, **kwargs: None
42
+
43
+
44
+ def _ensure_readers():
45
+ try:
46
+ from view.curses_multiwindow import Singleton
47
+ except ImportError as exc: # pragma: no cover
48
+ from .i18n import t
49
+ raise SourceUnavailable(t('src.missing')) from exc
50
+
51
+ if Singleton.getInstance(create=False) is None:
52
+ # access via the mangled name: the only way to register the instance
53
+ # without running __init__ and its side effects
54
+ Singleton._Singleton__instance = _ReaderShim()
55
+
56
+ import readers.slurmreader as slurmreader
57
+ return slurmreader
58
+
59
+
60
+ def _clean(value, default=None):
61
+ """pandas nan and empty strings -> None."""
62
+ if value is None:
63
+ return default
64
+ if isinstance(value, float) and math.isnan(value):
65
+ return default
66
+ if isinstance(value, str) and not value.strip():
67
+ return default
68
+ return value
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class Occupancy:
73
+ """The slice of a node taken by a job."""
74
+ jobid: str
75
+ job_name: str
76
+ user: str
77
+ group: Optional[str]
78
+ account: Optional[str]
79
+ partition: Optional[str]
80
+ qos: Optional[str]
81
+ gpus: int
82
+ cpus: int
83
+ mem_mb: float
84
+ mine: bool
85
+
86
+
87
+ @dataclass
88
+ class Node(NodeSpec):
89
+ """A node: Slurm's specs plus who is running on it."""
90
+ reserved: str = 'no' # 'no' | 'yes' | 'pending'
91
+ occupants: tuple = ()
92
+ gpu_owners: dict = field(default_factory=dict) # GPU index -> user
93
+
94
+ @property
95
+ def status(self) -> str:
96
+ """'ok' | 'drain' | 'down', the classification used across the app."""
97
+ state = self.state.upper()
98
+ if any(b in state for b in ('DOWN', 'FAIL', 'INVAL', 'UNKNOWN', 'NOT_RESPONDING')):
99
+ return 'down'
100
+ if 'DRAIN' in state or 'MAINT' in state:
101
+ return 'drain'
102
+ return 'ok'
103
+
104
+ @property
105
+ def healthy(self) -> bool:
106
+ return self.status == 'ok'
107
+
108
+ @property
109
+ def schedulable(self) -> bool:
110
+ return self.healthy and self.reserved == 'no' and self.gpus_free > 0
111
+
112
+ # NB: NodeSpec.mem_free_mb is the RAM the OS reports as free; to decide
113
+ # where a job fits, what matters is mem_unalloc_mb, i.e. what Slurm hasn't
114
+ # assigned yet. A job may have reserved memory without touching it yet,
115
+ # and the two figures diverge a lot.
116
+
117
+ @property
118
+ def users(self) -> tuple:
119
+ seen = []
120
+ for occ in self.occupants:
121
+ if occ.user not in seen:
122
+ seen.append(occ.user)
123
+ return tuple(seen)
124
+
125
+
126
+ @dataclass(frozen=True)
127
+ class Job:
128
+ jobid: str
129
+ name: str
130
+ user: str
131
+ group: Optional[str]
132
+ state: str
133
+ runtime: str
134
+ reason: Optional[str]
135
+ partition: str
136
+ account: Optional[str]
137
+ qos: Optional[str]
138
+ priority: Optional[int]
139
+ gpus: int
140
+ cpus: int
141
+ mem_mb: float
142
+ nodes: tuple
143
+ mine: bool
144
+
145
+ @property
146
+ def running(self) -> bool:
147
+ return self.state in ('R', 'CG')
148
+
149
+ @property
150
+ def pending(self) -> bool:
151
+ return not self.running
152
+
153
+
154
+ @dataclass(frozen=True)
155
+ class Maintenance:
156
+ nodes: tuple
157
+ start: str
158
+ end: str
159
+
160
+
161
+ @dataclass
162
+ class Snapshot:
163
+ nodes: list
164
+ jobs: list
165
+ gpu_limit_user: Optional[float]
166
+ gpu_limit_group: Optional[float]
167
+ maintenances: list
168
+ me: str
169
+ taken_at: float
170
+ duration: float = 0.0
171
+
172
+ @property
173
+ def my_jobs(self) -> list:
174
+ return [j for j in self.jobs if j.mine]
175
+
176
+ @property
177
+ def running_jobs(self) -> list:
178
+ return [j for j in self.jobs if j.running]
179
+
180
+ @property
181
+ def pending_jobs(self) -> list:
182
+ return [j for j in self.jobs if j.pending]
183
+
184
+ @property
185
+ def healthy_nodes(self) -> list:
186
+ return [n for n in self.nodes if n.healthy]
187
+
188
+ @property
189
+ def gpus_total(self) -> int:
190
+ return sum(n.gpus_total for n in self.healthy_nodes)
191
+
192
+ @property
193
+ def gpus_used(self) -> int:
194
+ return sum(n.gpus_used for n in self.nodes)
195
+
196
+ @property
197
+ def gpus_free(self) -> int:
198
+ return sum(n.gpus_free for n in self.nodes if n.schedulable)
199
+
200
+ @property
201
+ def cpus_total(self) -> int:
202
+ return sum(n.cpus_total for n in self.healthy_nodes)
203
+
204
+ @property
205
+ def cpus_used(self) -> int:
206
+ return sum(n.cpus_alloc for n in self.nodes)
207
+
208
+ @property
209
+ def mem_total_mb(self) -> float:
210
+ return sum(n.mem_total_mb for n in self.healthy_nodes)
211
+
212
+ @property
213
+ def mem_used_mb(self) -> float:
214
+ return sum(n.mem_alloc_mb for n in self.nodes)
215
+
216
+ @property
217
+ def unhealthy_nodes(self) -> list:
218
+ return [n for n in self.nodes if not n.healthy]
219
+
220
+
221
+ def current_user() -> str:
222
+ return os.environ.get('USER') or pwd.getpwuid(os.getuid()).pw_name
223
+
224
+
225
+ def _reserved_map(inf) -> dict:
226
+ return {n.name: (n.reserved or 'no') for n in getattr(inf, 'nodes', []) or []}
227
+
228
+
229
+ def fetch() -> Snapshot:
230
+ """A full read of the cluster. It takes a few tenths of a second and is
231
+ meant to run in a worker thread."""
232
+ slurmreader = _ensure_readers()
233
+ me = current_user()
234
+ started = time.perf_counter()
235
+
236
+ specs = read_nodes()
237
+ owners = read_gpu_owners()
238
+ inf = slurmreader.read_infrastructure()
239
+ raw_jobs, _ = slurmreader.read_jobs()
240
+
241
+ reserved = _reserved_map(inf)
242
+
243
+ jobs = []
244
+ per_node = {}
245
+ for j in raw_jobs:
246
+ job = Job(
247
+ jobid=str(j.jobid),
248
+ name=str(_clean(j.name, t('fallback.job_name'))),
249
+ user=str(j.user),
250
+ group=_clean(j.user_group),
251
+ state=str(j.state),
252
+ runtime=str(_clean(j.runtime, '-')),
253
+ reason=_clean(j.reason),
254
+ partition=str(_clean(j.partition, '-')),
255
+ account=_clean(j.account),
256
+ qos=None,
257
+ priority=int(j.priority) if _clean(j.priority) is not None else None,
258
+ gpus=sum(int(jl.n_gpus or 0) for jl in j.joblets),
259
+ cpus=sum(int(jl.cpus or 0) for jl in j.joblets),
260
+ mem_mb=sum(float(jl.mem or 0) for jl in j.joblets),
261
+ nodes=tuple(jl.node for jl in j.joblets if jl.node),
262
+ mine=(j.user == me),
263
+ )
264
+ jobs.append(job)
265
+
266
+ if job.running:
267
+ for jl in j.joblets:
268
+ if not jl.node:
269
+ continue
270
+ per_node.setdefault(jl.node, []).append(Occupancy(
271
+ jobid=job.jobid,
272
+ job_name=job.name,
273
+ user=job.user,
274
+ group=job.group,
275
+ account=job.account,
276
+ partition=job.partition,
277
+ qos=job.qos,
278
+ gpus=int(jl.n_gpus or 0),
279
+ cpus=int(jl.cpus or 0),
280
+ mem_mb=float(jl.mem or 0),
281
+ mine=job.mine,
282
+ ))
283
+
284
+ nodes = []
285
+ for spec in specs:
286
+ occ = tuple(sorted(per_node.get(spec.name, []),
287
+ key=lambda o: (-o.gpus, o.user, o.jobid)))
288
+ nodes.append(Node(**vars(spec), reserved=reserved.get(spec.name, 'no'),
289
+ occupants=occ, gpu_owners=owners.get(spec.name, {})))
290
+
291
+ maintenances = [
292
+ Maintenance(tuple(short_name(n) for n in (m.nodes or ())),
293
+ str(m.start_time), str(m.end_time))
294
+ for m in (getattr(inf, 'maintenances', None) or [])
295
+ ]
296
+
297
+ return Snapshot(
298
+ nodes=nodes,
299
+ jobs=jobs,
300
+ gpu_limit_user=_clean(inf.gpu_limit_pu),
301
+ gpu_limit_group=_clean(inf.gpu_limit_grp),
302
+ maintenances=maintenances,
303
+ me=me,
304
+ taken_at=time.time(),
305
+ duration=time.perf_counter() - started,
306
+ )
@@ -0,0 +1,41 @@
1
+ Metadata-Version: 2.5
2
+ Name: nodeview-slurm
3
+ Version: 0.1.0
4
+ Summary: GPU status of a Slurm cluster, from the user's point of view.
5
+ Project-URL: Homepage, https://codeberg.org/danielrossi/NodeView
6
+ Project-URL: Issues, https://codeberg.org/danielrossi/NodeView/issues
7
+ Author-email: Daniel Rossi <daniel.rossi@unimore.it>
8
+ Keywords: gpu,hpc,slurm,textual,tui
9
+ Classifier: Environment :: Console
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Operating System :: POSIX :: Linux
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: System :: Monitoring
14
+ Requires-Python: >=3.9
15
+ Requires-Dist: nodeocc<1.1,>=1.0.26
16
+ Requires-Dist: rich>=13
17
+ Requires-Dist: textual>=8.0
18
+ Description-Content-Type: text/markdown
19
+
20
+ # NodeView
21
+
22
+ GPU status of a Slurm cluster, from the user's point of view: your GPUs, the
23
+ cluster's nodes and the queue, in a terminal interface.
24
+
25
+ ## Installation
26
+
27
+ ```
28
+ python3 -m pip install nodeview-slurm
29
+ ```
30
+
31
+ ## Usage
32
+
33
+ ```
34
+ nodeview # interactive interface
35
+ nodeview --once # print a snapshot and exit
36
+ nodeview --once --cluster
37
+ nodeview --interval 5 # refresh the state every 5 seconds
38
+ ```
39
+
40
+ Requires Python 3.9+ and the Slurm commands (`scontrol`, `squeue`, `sinfo`,
41
+ `sacct`, ...) on your `PATH`.
@@ -0,0 +1,18 @@
1
+ nodeview/__init__.py,sha256=FAAjqB48n7gij9f-r4wOtF3sgZUJK7E-ezpr72mvg2U,101
2
+ nodeview/__main__.py,sha256=kXRknv5vG_c8Fqw8MFWJKNpVkib70WxXBbnQ2Tb-OVU,1219
3
+ nodeview/analysis.py,sha256=rBo-Mg0S9XjeIuPAl8bqABTs4XA4lWdeoJ5qiyCRHL8,9997
4
+ nodeview/app.py,sha256=q7QYboR0DmBwJCrPC3AGXVhX0jriMktOTi7wnRoC3og,15750
5
+ nodeview/app.tcss,sha256=UdlfEBOcFFIyFQ2M-1UBErtSX10le3xYZjxFLztq940,601
6
+ nodeview/config.py,sha256=ybg8NB2RDOFI3ufK1wEis80rJhXT1SqnqPAchnH2s08,1869
7
+ nodeview/i18n.py,sha256=Wb40MSJdhMoDXX6g_XpV4JDqul4ba47ulSrADIc8CT8,18887
8
+ nodeview/manual.py,sha256=Ovm93ateDtaYSQxFvxrPKsRDfFSkfAYMWx2-BbFf2Aw,6901
9
+ nodeview/nodes.py,sha256=IuRUkuBeqcaEuhFVMhVneMcqUhhiZLzkXp0blVytdDA,10668
10
+ nodeview/palette.py,sha256=_GVZxJbVnYpuOdFUtGQ1jyITKNmorlUBP1bmyp_J0t0,11400
11
+ nodeview/profile.py,sha256=ni8wxL8Ak1DmyBJOd4CnULkWFSzAQkfvGtvFU2_lols,9453
12
+ nodeview/render.py,sha256=VSAodSvdJYYLPe1mBBVqtOBlHHU6LS1yO39nODFRq5c,27358
13
+ nodeview/snapshot.py,sha256=TFlgHojQwlfFSP_hjTyCvUsS_D-w3_5tFDZBJkzvGik,2672
14
+ nodeview/source.py,sha256=9TEfaZ8zYgIfpTulABKkVKiYSN5K_fwUM3n7yhD5u2k,8786
15
+ nodeview_slurm-0.1.0.dist-info/METADATA,sha256=FSPhUZmHZn-8LMNkmrdpam3aZH8jD3MrtlJqmCffmL0,1251
16
+ nodeview_slurm-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
17
+ nodeview_slurm-0.1.0.dist-info/entry_points.txt,sha256=T-Is8Gv-vdgiGkbE36usLI3DxS0lD2dlGqS8gXhg9uI,52
18
+ nodeview_slurm-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ nodeview = nodeview.__main__:main