nodeview-slurm 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ build/
4
+ dist/
5
+ *.egg-info/
6
+ .venv/
@@ -0,0 +1,41 @@
1
+ Metadata-Version: 2.5
2
+ Name: nodeview-slurm
3
+ Version: 0.1.0
4
+ Summary: GPU status of a Slurm cluster, from the user's point of view.
5
+ Project-URL: Homepage, https://codeberg.org/danielrossi/NodeView
6
+ Project-URL: Issues, https://codeberg.org/danielrossi/NodeView/issues
7
+ Author-email: Daniel Rossi <daniel.rossi@unimore.it>
8
+ Keywords: gpu,hpc,slurm,textual,tui
9
+ Classifier: Environment :: Console
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Operating System :: POSIX :: Linux
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: System :: Monitoring
14
+ Requires-Python: >=3.9
15
+ Requires-Dist: nodeocc<1.1,>=1.0.26
16
+ Requires-Dist: rich>=13
17
+ Requires-Dist: textual>=8.0
18
+ Description-Content-Type: text/markdown
19
+
20
+ # NodeView
21
+
22
+ GPU status of a Slurm cluster, from the user's point of view: your GPUs, the
23
+ cluster's nodes and the queue, in a terminal interface.
24
+
25
+ ## Installation
26
+
27
+ ```
28
+ python3 -m pip install nodeview-slurm
29
+ ```
30
+
31
+ ## Usage
32
+
33
+ ```
34
+ nodeview # interactive interface
35
+ nodeview --once # print a snapshot and exit
36
+ nodeview --once --cluster
37
+ nodeview --interval 5 # refresh the state every 5 seconds
38
+ ```
39
+
40
+ Requires Python 3.9+ and the Slurm commands (`scontrol`, `squeue`, `sinfo`,
41
+ `sacct`, ...) on your `PATH`.
@@ -0,0 +1,22 @@
1
+ # NodeView
2
+
3
+ GPU status of a Slurm cluster, from the user's point of view: your GPUs, the
4
+ cluster's nodes and the queue, in a terminal interface.
5
+
6
+ ## Installation
7
+
8
+ ```
9
+ python3 -m pip install nodeview-slurm
10
+ ```
11
+
12
+ ## Usage
13
+
14
+ ```
15
+ nodeview # interactive interface
16
+ nodeview --once # print a snapshot and exit
17
+ nodeview --once --cluster
18
+ nodeview --interval 5 # refresh the state every 5 seconds
19
+ ```
20
+
21
+ Requires Python 3.9+ and the Slurm commands (`scontrol`, `squeue`, `sinfo`,
22
+ `sacct`, ...) on your `PATH`.
@@ -0,0 +1,37 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "nodeview-slurm"
7
+ dynamic = ["version"]
8
+ description = "GPU status of a Slurm cluster, from the user's point of view."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ authors = [{ name = "Daniel Rossi", email = "daniel.rossi@unimore.it" }]
12
+ keywords = ["slurm", "gpu", "hpc", "tui", "textual"]
13
+ classifiers = [
14
+ "Environment :: Console",
15
+ "Intended Audience :: Science/Research",
16
+ "Operating System :: POSIX :: Linux",
17
+ "Programming Language :: Python :: 3",
18
+ "Topic :: System :: Monitoring",
19
+ ]
20
+ dependencies = [
21
+ "textual>=8.0",
22
+ "rich>=13",
23
+ "nodeocc>=1.0.26,<1.1",
24
+ ]
25
+
26
+ [project.urls]
27
+ Homepage = "https://codeberg.org/danielrossi/NodeView"
28
+ Issues = "https://codeberg.org/danielrossi/NodeView/issues"
29
+
30
+ [project.scripts]
31
+ nodeview = "nodeview.__main__:main"
32
+
33
+ [tool.hatch.version]
34
+ path = "src/nodeview/__init__.py"
35
+
36
+ [tool.hatch.build.targets.wheel]
37
+ packages = ["src/nodeview"]
@@ -0,0 +1,3 @@
1
+ """NodeView: GPU status of a Slurm cluster, from the user's point of view."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,32 @@
1
+ """Entry point: TUI by default, static snapshot with --once."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import sys
6
+
7
+
8
+ def main(argv=None) -> int:
9
+ parser = argparse.ArgumentParser(
10
+ prog='nodeview',
11
+ description="GPU status of a Slurm cluster, from the user's point of view.")
12
+ parser.add_argument('--once', action='store_true',
13
+ help="print a snapshot and exit, without the interactive interface")
14
+ parser.add_argument('--interval', type=float, default=10.0, metavar='SEC',
15
+ help="how often to refresh the state (default: 10s)")
16
+ parser.add_argument('--cluster', action='store_true',
17
+ help="with --once, print the cluster page instead of yours")
18
+ parser.add_argument('--width', type=int, default=None, metavar='N',
19
+ help="force the output width for --once (useful in pipes and scripts)")
20
+ args = parser.parse_args(argv)
21
+
22
+ if args.once:
23
+ from .snapshot import print_once
24
+ return print_once(width=args.width, cluster=args.cluster)
25
+
26
+ from .app import NodeView
27
+ NodeView(interval=args.interval).run()
28
+ return 0
29
+
30
+
31
+ if __name__ == '__main__':
32
+ sys.exit(main())
@@ -0,0 +1,298 @@
1
+ """Everything that can be derived from a Snapshot.
2
+
3
+ Deliberately kept apart from the UI: the same functions serve both the TUI
4
+ and the static --once snapshot.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from dataclasses import dataclass
9
+ from datetime import datetime
10
+ from typing import Optional
11
+
12
+ from .i18n import t
13
+ from .source import Node, Snapshot
14
+
15
+ # Slurm's codes are cryptic, and they're the first thing you look at when a
16
+ # job won't start. The texts live in the language catalog; only codes here.
17
+ REASON_CODES = (
18
+ 'QOSGrpGRES', 'QOSGrpGRESMinutes', 'QOSGrpMemLimit', 'QOSGrpCpuLimit',
19
+ 'QOSMaxGRESPerUser', 'QOSMaxCpuPerUserLimit', 'QOSMaxMemPerUser',
20
+ 'QOSMaxJobsPerUserLimit', 'Dependency', 'Resources', 'Priority',
21
+ 'BeginTime', 'ReqNodeNotAvail', 'JobArrayTaskLimit',
22
+ )
23
+
24
+
25
+ def _reason_texts(code: str) -> tuple:
26
+ return (t(f'reason.{code}.label'), t(f'reason.{code}.detail'))
27
+
28
+
29
+ def explain_reason(reason: Optional[str]) -> tuple:
30
+ """(short label, explanation) for a Slurm pending reason."""
31
+ if not reason or reason.strip() in ('None', 'none'):
32
+ return _reason_texts('unknown')
33
+ key = reason.split(',')[0].strip()
34
+ if key in REASON_CODES:
35
+ return _reason_texts(key)
36
+ for code in REASON_CODES:
37
+ if key.startswith(code):
38
+ return _reason_texts(code)
39
+ return (reason, "")
40
+
41
+
42
+ @dataclass
43
+ class Fit:
44
+ """Answers: if I ask for N GPUs, where do I fit right now?"""
45
+ gpus: int
46
+ nodes: list
47
+ over_user_quota: bool
48
+
49
+ @property
50
+ def possible(self) -> bool:
51
+ return bool(self.nodes) and not self.over_user_quota
52
+
53
+
54
+ @dataclass
55
+ class Quota:
56
+ used: int
57
+ limit: Optional[float]
58
+
59
+ @property
60
+ def headroom(self) -> Optional[int]:
61
+ if self.limit is None:
62
+ return None
63
+ return max(0, int(self.limit) - self.used)
64
+
65
+ @property
66
+ def ratio(self) -> float:
67
+ if not self.limit:
68
+ return 0.0
69
+ return min(1.0, self.used / self.limit)
70
+
71
+
72
+ def my_gpus(snap: Snapshot) -> int:
73
+ return sum(j.gpus for j in snap.my_jobs if j.running)
74
+
75
+
76
+ def my_quota(snap: Snapshot) -> Quota:
77
+ return Quota(used=my_gpus(snap), limit=snap.gpu_limit_user)
78
+
79
+
80
+ def gpus_by_group(snap: Snapshot) -> list:
81
+ """GPUs in use per unix group, from the hungriest to the least."""
82
+ totals = {}
83
+ for job in snap.running_jobs:
84
+ key = job.group or t('fallback.group')
85
+ totals[key] = totals.get(key, 0) + job.gpus
86
+ return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)
87
+
88
+
89
+ def gpus_by_account(snap: Snapshot) -> list:
90
+ """GPUs in use per Slurm account: that's where courses (cvcs) stand apart
91
+ from projects and theses."""
92
+ totals = {}
93
+ for job in snap.running_jobs:
94
+ key = job.account or t('fallback.account')
95
+ totals[key] = totals.get(key, 0) + job.gpus
96
+ return sorted(((k, v) for k, v in totals.items() if v),
97
+ key=lambda kv: kv[1], reverse=True)
98
+
99
+
100
+ def by_gpu_model(snap: Snapshot) -> list:
101
+ """(model, used, total): tells which GPU type is still free."""
102
+ totals = {}
103
+ for node in snap.nodes:
104
+ if not node.healthy:
105
+ continue
106
+ used, total = totals.get(node.gpu_model, (0, 0))
107
+ totals[node.gpu_model] = (used + node.gpus_used, total + node.gpus_total)
108
+ return sorted(((m, u, t) for m, (u, t) in totals.items()),
109
+ key=lambda r: (-(r[2] - r[1]), r[0]))
110
+
111
+
112
+ def by_partition(snap: Snapshot) -> list:
113
+ """(partition, nodes, total GPUs)."""
114
+ totals = {}
115
+ for node in snap.nodes:
116
+ for part in node.partitions:
117
+ n, g = totals.get(part, (0, 0))
118
+ totals[part] = (n + 1, g + node.gpus_total)
119
+ return sorted(((p, n, g) for p, (n, g) in totals.items()),
120
+ key=lambda r: -r[2])
121
+
122
+
123
+ def occupant_counts(snap: Snapshot) -> dict:
124
+ """Busy GPUs per user: this is the input to the identity palette."""
125
+ counts = {}
126
+ for node in snap.nodes:
127
+ for occ in node.occupants:
128
+ counts[occ.user] = counts.get(occ.user, 0) + max(occ.gpus, 0)
129
+ return counts
130
+
131
+
132
+ def candidate_nodes(snap: Snapshot, gpus: int, min_mem_gb: float = 0.0) -> list:
133
+ """Nodes that would accept a `gpus`-GPU job right away.
134
+
135
+ It also takes at least one CPU and some free RAM per GPU: a node with free
136
+ GPUs but no CPUs left isn't really usable.
137
+ """
138
+ out = []
139
+ for node in snap.nodes:
140
+ if not node.schedulable or node.gpus_free < gpus:
141
+ continue
142
+ if node.cpus_free < gpus:
143
+ continue
144
+ if node.mem_unalloc_mb < max(min_mem_gb * 1024, 1024 * gpus):
145
+ continue
146
+ out.append(node)
147
+ return sorted(out, key=lambda n: (-n.gpus_free, -n.mem_free_mb, n.name))
148
+
149
+
150
+ def fits(snap: Snapshot, sizes=(1, 2, 4, 8), min_mem_gb: float = 0.0) -> list:
151
+ quota = my_quota(snap)
152
+ room = quota.headroom
153
+ out = []
154
+ for size in sizes:
155
+ over = room is not None and size > room
156
+ out.append(Fit(
157
+ gpus=size,
158
+ nodes=[] if over else candidate_nodes(snap, size, min_mem_gb),
159
+ over_user_quota=over,
160
+ ))
161
+ return out
162
+
163
+
164
+ def queue_reasons(snap: Snapshot) -> list:
165
+ """(raw reason, label, explanation, count) sorted by frequency."""
166
+ counts = {}
167
+ for job in snap.pending_jobs:
168
+ key = (job.reason or 'None').split(',')[0].strip()
169
+ counts[key] = counts.get(key, 0) + 1
170
+ rows = []
171
+ for raw, n in sorted(counts.items(), key=lambda kv: kv[1], reverse=True):
172
+ label, detail = explain_reason(raw)
173
+ rows.append((raw, label, detail, n))
174
+ return rows
175
+
176
+
177
+ def headline(snap: Snapshot) -> Optional[str]:
178
+ """One sentence that explains the queue, when there's something to say.
179
+
180
+ The interesting case: a long queue but free GPUs. It means the bottleneck
181
+ is a quota, not the hardware, and nobody notices by looking at usage
182
+ alone.
183
+ """
184
+ pending = len(snap.pending_jobs)
185
+ if not pending:
186
+ return None
187
+ reasons = queue_reasons(snap)
188
+ if not reasons:
189
+ return None
190
+ raw, label, _, n = reasons[0]
191
+ free = snap.gpus_free
192
+ if raw.startswith('QOS') and free > 0 and n >= max(2, pending // 3):
193
+ return t('head.quota_not_gpus', n=n, total=pending, label=label, free=free)
194
+ if raw == 'Resources' and free == 0:
195
+ return t('head.all_busy', n=n)
196
+ return t('head.generic', n=n, total=pending, label=label)
197
+
198
+
199
+ def warnings(snap: Snapshot) -> list:
200
+ """Warnings as (level, text). Level: 'info' | 'warn' | 'bad'."""
201
+ out = []
202
+ for status, level, key in (('down', 'bad', 'warn.down'), ('drain', 'warn', 'warn.drain')):
203
+ hit = [n for n in snap.nodes if n.status == status]
204
+ if hit:
205
+ noun = t('warn.node_one' if len(hit) == 1 else 'warn.node_many')
206
+ out.append((level, t(key, n=len(hit), noun=noun,
207
+ nodes=", ".join(n.name for n in hit))))
208
+
209
+ reserved = [n for n in snap.nodes if n.reserved != 'no' and n.healthy]
210
+ if reserved:
211
+ noun = t('warn.reserved_one' if len(reserved) == 1 else 'warn.reserved_many')
212
+ out.append(('info', t('warn.reserved', n=len(reserved), noun=noun,
213
+ nodes=", ".join(n.name for n in reserved))))
214
+
215
+ for m in snap.maintenances:
216
+ out.append((maintenance_level(m), describe_maintenance(m, len(snap.nodes))))
217
+ return out
218
+
219
+
220
+ def _parse_when(value: str) -> Optional[datetime]:
221
+ for fmt in ('%Y-%m-%d %H:%M:%S', '%Y-%m-%dT%H:%M:%S'):
222
+ try:
223
+ return datetime.strptime(str(value), fmt)
224
+ except (ValueError, TypeError):
225
+ continue
226
+ return None
227
+
228
+
229
+ def days_to_maintenance(m) -> Optional[float]:
230
+ start = _parse_when(m.start)
231
+ if start is None:
232
+ return None
233
+ return (start - datetime.now()).total_seconds() / 86400
234
+
235
+
236
+ def maintenance_level(m) -> str:
237
+ days = days_to_maintenance(m)
238
+ if days is None:
239
+ return 'info'
240
+ if days <= 0:
241
+ return 'bad'
242
+ return 'warn' if days <= 7 else 'info'
243
+
244
+
245
+ def describe_maintenance(m, total_nodes: int) -> str:
246
+ """Compact: the full node list takes three lines and says nothing."""
247
+ start, end = _parse_when(m.start), _parse_when(m.end)
248
+ n = len(m.nodes)
249
+ if n >= max(1, total_nodes - 1):
250
+ where = t('warn.whole_cluster')
251
+ elif n <= 4:
252
+ where = ", ".join(m.nodes)
253
+ else:
254
+ where = t('warn.n_nodes_sample', n=n, sample=', '.join(m.nodes[:3]))
255
+
256
+ if start is None:
257
+ return t('warn.maint_planned', where=where)
258
+
259
+ days = days_to_maintenance(m)
260
+ when = start.strftime('%d/%m %H:%M')
261
+ if days is not None and days <= 0:
262
+ quando = t('warn.maint_now')
263
+ elif days is not None and days < 1:
264
+ quando = t('warn.maint_hours', n=int(days * 24))
265
+ elif days is not None:
266
+ d = int(days)
267
+ quando = t('warn.maint_days', n=d,
268
+ noun=t('warn.day_one' if d == 1 else 'warn.day_many'))
269
+ else:
270
+ quando = when
271
+
272
+ lasting = ""
273
+ if end is not None and start is not None:
274
+ g = max(1, round((end - start).total_seconds() / 86400))
275
+ lasting = t('warn.lasting', n=g,
276
+ noun=t('warn.day_one' if g == 1 else 'warn.day_many'))
277
+ return t('warn.maint', when=quando, date=when, lasting=lasting, where=where)
278
+
279
+
280
+ def uptime(node) -> Optional[str]:
281
+ """How long the node has been up, in readable form."""
282
+ start = _parse_when((node.boot_time or '').replace('T', ' '))
283
+ if start is None:
284
+ return None
285
+ days = (datetime.now() - start).days
286
+ if days >= 1:
287
+ return f"{days} " + t('warn.day_one' if days == 1 else 'warn.day_many')
288
+ hours = int((datetime.now() - start).total_seconds() // 3600)
289
+ return t('time.hours', n=hours)
290
+
291
+
292
+ def node_sort_key(node: Node):
293
+ """Nodes you can fit on first, then the busiest, broken ones last."""
294
+ if not node.healthy:
295
+ return (2, 0, node.name)
296
+ if node.gpus_free > 0 and node.reserved == 'no':
297
+ return (0, -node.gpus_free, node.name)
298
+ return (1, -node.gpus_free, node.name)