nodeview-slurm 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nodeview/__init__.py +3 -0
- nodeview/__main__.py +32 -0
- nodeview/analysis.py +298 -0
- nodeview/app.py +423 -0
- nodeview/app.tcss +35 -0
- nodeview/config.py +61 -0
- nodeview/i18n.py +455 -0
- nodeview/manual.py +176 -0
- nodeview/nodes.py +318 -0
- nodeview/palette.py +325 -0
- nodeview/profile.py +284 -0
- nodeview/render.py +688 -0
- nodeview/snapshot.py +72 -0
- nodeview/source.py +306 -0
- nodeview_slurm-0.1.0.dist-info/METADATA +41 -0
- nodeview_slurm-0.1.0.dist-info/RECORD +18 -0
- nodeview_slurm-0.1.0.dist-info/WHEEL +4 -0
- nodeview_slurm-0.1.0.dist-info/entry_points.txt +2 -0
nodeview/__init__.py
ADDED
nodeview/__main__.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Entry point: TUI by default, static snapshot with --once."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import sys
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def main(argv=None) -> int:
|
|
9
|
+
parser = argparse.ArgumentParser(
|
|
10
|
+
prog='nodeview',
|
|
11
|
+
description="GPU status of a Slurm cluster, from the user's point of view.")
|
|
12
|
+
parser.add_argument('--once', action='store_true',
|
|
13
|
+
help="print a snapshot and exit, without the interactive interface")
|
|
14
|
+
parser.add_argument('--interval', type=float, default=10.0, metavar='SEC',
|
|
15
|
+
help="how often to refresh the state (default: 10s)")
|
|
16
|
+
parser.add_argument('--cluster', action='store_true',
|
|
17
|
+
help="with --once, print the cluster page instead of yours")
|
|
18
|
+
parser.add_argument('--width', type=int, default=None, metavar='N',
|
|
19
|
+
help="force the output width for --once (useful in pipes and scripts)")
|
|
20
|
+
args = parser.parse_args(argv)
|
|
21
|
+
|
|
22
|
+
if args.once:
|
|
23
|
+
from .snapshot import print_once
|
|
24
|
+
return print_once(width=args.width, cluster=args.cluster)
|
|
25
|
+
|
|
26
|
+
from .app import NodeView
|
|
27
|
+
NodeView(interval=args.interval).run()
|
|
28
|
+
return 0
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
if __name__ == '__main__':
|
|
32
|
+
sys.exit(main())
|
nodeview/analysis.py
ADDED
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""Everything that can be derived from a Snapshot.
|
|
2
|
+
|
|
3
|
+
Deliberately kept apart from the UI: the same functions serve both the TUI
|
|
4
|
+
and the static --once snapshot.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from datetime import datetime
|
|
10
|
+
from typing import Optional
|
|
11
|
+
|
|
12
|
+
from .i18n import t
|
|
13
|
+
from .source import Node, Snapshot
|
|
14
|
+
|
|
15
|
+
# Slurm's codes are cryptic, and they're the first thing you look at when a
|
|
16
|
+
# job won't start. The texts live in the language catalog; only codes here.
|
|
17
|
+
REASON_CODES = (
|
|
18
|
+
'QOSGrpGRES', 'QOSGrpGRESMinutes', 'QOSGrpMemLimit', 'QOSGrpCpuLimit',
|
|
19
|
+
'QOSMaxGRESPerUser', 'QOSMaxCpuPerUserLimit', 'QOSMaxMemPerUser',
|
|
20
|
+
'QOSMaxJobsPerUserLimit', 'Dependency', 'Resources', 'Priority',
|
|
21
|
+
'BeginTime', 'ReqNodeNotAvail', 'JobArrayTaskLimit',
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _reason_texts(code: str) -> tuple:
|
|
26
|
+
return (t(f'reason.{code}.label'), t(f'reason.{code}.detail'))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def explain_reason(reason: Optional[str]) -> tuple:
|
|
30
|
+
"""(short label, explanation) for a Slurm pending reason."""
|
|
31
|
+
if not reason or reason.strip() in ('None', 'none'):
|
|
32
|
+
return _reason_texts('unknown')
|
|
33
|
+
key = reason.split(',')[0].strip()
|
|
34
|
+
if key in REASON_CODES:
|
|
35
|
+
return _reason_texts(key)
|
|
36
|
+
for code in REASON_CODES:
|
|
37
|
+
if key.startswith(code):
|
|
38
|
+
return _reason_texts(code)
|
|
39
|
+
return (reason, "")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Fit:
|
|
44
|
+
"""Answers: if I ask for N GPUs, where do I fit right now?"""
|
|
45
|
+
gpus: int
|
|
46
|
+
nodes: list
|
|
47
|
+
over_user_quota: bool
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def possible(self) -> bool:
|
|
51
|
+
return bool(self.nodes) and not self.over_user_quota
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class Quota:
|
|
56
|
+
used: int
|
|
57
|
+
limit: Optional[float]
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def headroom(self) -> Optional[int]:
|
|
61
|
+
if self.limit is None:
|
|
62
|
+
return None
|
|
63
|
+
return max(0, int(self.limit) - self.used)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def ratio(self) -> float:
|
|
67
|
+
if not self.limit:
|
|
68
|
+
return 0.0
|
|
69
|
+
return min(1.0, self.used / self.limit)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def my_gpus(snap: Snapshot) -> int:
|
|
73
|
+
return sum(j.gpus for j in snap.my_jobs if j.running)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def my_quota(snap: Snapshot) -> Quota:
|
|
77
|
+
return Quota(used=my_gpus(snap), limit=snap.gpu_limit_user)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def gpus_by_group(snap: Snapshot) -> list:
|
|
81
|
+
"""GPUs in use per unix group, from the hungriest to the least."""
|
|
82
|
+
totals = {}
|
|
83
|
+
for job in snap.running_jobs:
|
|
84
|
+
key = job.group or t('fallback.group')
|
|
85
|
+
totals[key] = totals.get(key, 0) + job.gpus
|
|
86
|
+
return sorted(totals.items(), key=lambda kv: kv[1], reverse=True)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def gpus_by_account(snap: Snapshot) -> list:
|
|
90
|
+
"""GPUs in use per Slurm account: that's where courses (cvcs) stand apart
|
|
91
|
+
from projects and theses."""
|
|
92
|
+
totals = {}
|
|
93
|
+
for job in snap.running_jobs:
|
|
94
|
+
key = job.account or t('fallback.account')
|
|
95
|
+
totals[key] = totals.get(key, 0) + job.gpus
|
|
96
|
+
return sorted(((k, v) for k, v in totals.items() if v),
|
|
97
|
+
key=lambda kv: kv[1], reverse=True)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def by_gpu_model(snap: Snapshot) -> list:
|
|
101
|
+
"""(model, used, total): tells which GPU type is still free."""
|
|
102
|
+
totals = {}
|
|
103
|
+
for node in snap.nodes:
|
|
104
|
+
if not node.healthy:
|
|
105
|
+
continue
|
|
106
|
+
used, total = totals.get(node.gpu_model, (0, 0))
|
|
107
|
+
totals[node.gpu_model] = (used + node.gpus_used, total + node.gpus_total)
|
|
108
|
+
return sorted(((m, u, t) for m, (u, t) in totals.items()),
|
|
109
|
+
key=lambda r: (-(r[2] - r[1]), r[0]))
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def by_partition(snap: Snapshot) -> list:
|
|
113
|
+
"""(partition, nodes, total GPUs)."""
|
|
114
|
+
totals = {}
|
|
115
|
+
for node in snap.nodes:
|
|
116
|
+
for part in node.partitions:
|
|
117
|
+
n, g = totals.get(part, (0, 0))
|
|
118
|
+
totals[part] = (n + 1, g + node.gpus_total)
|
|
119
|
+
return sorted(((p, n, g) for p, (n, g) in totals.items()),
|
|
120
|
+
key=lambda r: -r[2])
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def occupant_counts(snap: Snapshot) -> dict:
|
|
124
|
+
"""Busy GPUs per user: this is the input to the identity palette."""
|
|
125
|
+
counts = {}
|
|
126
|
+
for node in snap.nodes:
|
|
127
|
+
for occ in node.occupants:
|
|
128
|
+
counts[occ.user] = counts.get(occ.user, 0) + max(occ.gpus, 0)
|
|
129
|
+
return counts
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def candidate_nodes(snap: Snapshot, gpus: int, min_mem_gb: float = 0.0) -> list:
|
|
133
|
+
"""Nodes that would accept a `gpus`-GPU job right away.
|
|
134
|
+
|
|
135
|
+
It also takes at least one CPU and some free RAM per GPU: a node with free
|
|
136
|
+
GPUs but no CPUs left isn't really usable.
|
|
137
|
+
"""
|
|
138
|
+
out = []
|
|
139
|
+
for node in snap.nodes:
|
|
140
|
+
if not node.schedulable or node.gpus_free < gpus:
|
|
141
|
+
continue
|
|
142
|
+
if node.cpus_free < gpus:
|
|
143
|
+
continue
|
|
144
|
+
if node.mem_unalloc_mb < max(min_mem_gb * 1024, 1024 * gpus):
|
|
145
|
+
continue
|
|
146
|
+
out.append(node)
|
|
147
|
+
return sorted(out, key=lambda n: (-n.gpus_free, -n.mem_free_mb, n.name))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def fits(snap: Snapshot, sizes=(1, 2, 4, 8), min_mem_gb: float = 0.0) -> list:
|
|
151
|
+
quota = my_quota(snap)
|
|
152
|
+
room = quota.headroom
|
|
153
|
+
out = []
|
|
154
|
+
for size in sizes:
|
|
155
|
+
over = room is not None and size > room
|
|
156
|
+
out.append(Fit(
|
|
157
|
+
gpus=size,
|
|
158
|
+
nodes=[] if over else candidate_nodes(snap, size, min_mem_gb),
|
|
159
|
+
over_user_quota=over,
|
|
160
|
+
))
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def queue_reasons(snap: Snapshot) -> list:
|
|
165
|
+
"""(raw reason, label, explanation, count) sorted by frequency."""
|
|
166
|
+
counts = {}
|
|
167
|
+
for job in snap.pending_jobs:
|
|
168
|
+
key = (job.reason or 'None').split(',')[0].strip()
|
|
169
|
+
counts[key] = counts.get(key, 0) + 1
|
|
170
|
+
rows = []
|
|
171
|
+
for raw, n in sorted(counts.items(), key=lambda kv: kv[1], reverse=True):
|
|
172
|
+
label, detail = explain_reason(raw)
|
|
173
|
+
rows.append((raw, label, detail, n))
|
|
174
|
+
return rows
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def headline(snap: Snapshot) -> Optional[str]:
|
|
178
|
+
"""One sentence that explains the queue, when there's something to say.
|
|
179
|
+
|
|
180
|
+
The interesting case: a long queue but free GPUs. It means the bottleneck
|
|
181
|
+
is a quota, not the hardware, and nobody notices by looking at usage
|
|
182
|
+
alone.
|
|
183
|
+
"""
|
|
184
|
+
pending = len(snap.pending_jobs)
|
|
185
|
+
if not pending:
|
|
186
|
+
return None
|
|
187
|
+
reasons = queue_reasons(snap)
|
|
188
|
+
if not reasons:
|
|
189
|
+
return None
|
|
190
|
+
raw, label, _, n = reasons[0]
|
|
191
|
+
free = snap.gpus_free
|
|
192
|
+
if raw.startswith('QOS') and free > 0 and n >= max(2, pending // 3):
|
|
193
|
+
return t('head.quota_not_gpus', n=n, total=pending, label=label, free=free)
|
|
194
|
+
if raw == 'Resources' and free == 0:
|
|
195
|
+
return t('head.all_busy', n=n)
|
|
196
|
+
return t('head.generic', n=n, total=pending, label=label)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def warnings(snap: Snapshot) -> list:
|
|
200
|
+
"""Warnings as (level, text). Level: 'info' | 'warn' | 'bad'."""
|
|
201
|
+
out = []
|
|
202
|
+
for status, level, key in (('down', 'bad', 'warn.down'), ('drain', 'warn', 'warn.drain')):
|
|
203
|
+
hit = [n for n in snap.nodes if n.status == status]
|
|
204
|
+
if hit:
|
|
205
|
+
noun = t('warn.node_one' if len(hit) == 1 else 'warn.node_many')
|
|
206
|
+
out.append((level, t(key, n=len(hit), noun=noun,
|
|
207
|
+
nodes=", ".join(n.name for n in hit))))
|
|
208
|
+
|
|
209
|
+
reserved = [n for n in snap.nodes if n.reserved != 'no' and n.healthy]
|
|
210
|
+
if reserved:
|
|
211
|
+
noun = t('warn.reserved_one' if len(reserved) == 1 else 'warn.reserved_many')
|
|
212
|
+
out.append(('info', t('warn.reserved', n=len(reserved), noun=noun,
|
|
213
|
+
nodes=", ".join(n.name for n in reserved))))
|
|
214
|
+
|
|
215
|
+
for m in snap.maintenances:
|
|
216
|
+
out.append((maintenance_level(m), describe_maintenance(m, len(snap.nodes))))
|
|
217
|
+
return out
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _parse_when(value: str) -> Optional[datetime]:
|
|
221
|
+
for fmt in ('%Y-%m-%d %H:%M:%S', '%Y-%m-%dT%H:%M:%S'):
|
|
222
|
+
try:
|
|
223
|
+
return datetime.strptime(str(value), fmt)
|
|
224
|
+
except (ValueError, TypeError):
|
|
225
|
+
continue
|
|
226
|
+
return None
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def days_to_maintenance(m) -> Optional[float]:
|
|
230
|
+
start = _parse_when(m.start)
|
|
231
|
+
if start is None:
|
|
232
|
+
return None
|
|
233
|
+
return (start - datetime.now()).total_seconds() / 86400
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def maintenance_level(m) -> str:
|
|
237
|
+
days = days_to_maintenance(m)
|
|
238
|
+
if days is None:
|
|
239
|
+
return 'info'
|
|
240
|
+
if days <= 0:
|
|
241
|
+
return 'bad'
|
|
242
|
+
return 'warn' if days <= 7 else 'info'
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def describe_maintenance(m, total_nodes: int) -> str:
|
|
246
|
+
"""Compact: the full node list takes three lines and says nothing."""
|
|
247
|
+
start, end = _parse_when(m.start), _parse_when(m.end)
|
|
248
|
+
n = len(m.nodes)
|
|
249
|
+
if n >= max(1, total_nodes - 1):
|
|
250
|
+
where = t('warn.whole_cluster')
|
|
251
|
+
elif n <= 4:
|
|
252
|
+
where = ", ".join(m.nodes)
|
|
253
|
+
else:
|
|
254
|
+
where = t('warn.n_nodes_sample', n=n, sample=', '.join(m.nodes[:3]))
|
|
255
|
+
|
|
256
|
+
if start is None:
|
|
257
|
+
return t('warn.maint_planned', where=where)
|
|
258
|
+
|
|
259
|
+
days = days_to_maintenance(m)
|
|
260
|
+
when = start.strftime('%d/%m %H:%M')
|
|
261
|
+
if days is not None and days <= 0:
|
|
262
|
+
quando = t('warn.maint_now')
|
|
263
|
+
elif days is not None and days < 1:
|
|
264
|
+
quando = t('warn.maint_hours', n=int(days * 24))
|
|
265
|
+
elif days is not None:
|
|
266
|
+
d = int(days)
|
|
267
|
+
quando = t('warn.maint_days', n=d,
|
|
268
|
+
noun=t('warn.day_one' if d == 1 else 'warn.day_many'))
|
|
269
|
+
else:
|
|
270
|
+
quando = when
|
|
271
|
+
|
|
272
|
+
lasting = ""
|
|
273
|
+
if end is not None and start is not None:
|
|
274
|
+
g = max(1, round((end - start).total_seconds() / 86400))
|
|
275
|
+
lasting = t('warn.lasting', n=g,
|
|
276
|
+
noun=t('warn.day_one' if g == 1 else 'warn.day_many'))
|
|
277
|
+
return t('warn.maint', when=quando, date=when, lasting=lasting, where=where)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def uptime(node) -> Optional[str]:
|
|
281
|
+
"""How long the node has been up, in readable form."""
|
|
282
|
+
start = _parse_when((node.boot_time or '').replace('T', ' '))
|
|
283
|
+
if start is None:
|
|
284
|
+
return None
|
|
285
|
+
days = (datetime.now() - start).days
|
|
286
|
+
if days >= 1:
|
|
287
|
+
return f"{days} " + t('warn.day_one' if days == 1 else 'warn.day_many')
|
|
288
|
+
hours = int((datetime.now() - start).total_seconds() // 3600)
|
|
289
|
+
return t('time.hours', n=hours)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def node_sort_key(node: Node):
|
|
293
|
+
"""Nodes you can fit on first, then the busiest, broken ones last."""
|
|
294
|
+
if not node.healthy:
|
|
295
|
+
return (2, 0, node.name)
|
|
296
|
+
if node.gpus_free > 0 and node.reserved == 'no':
|
|
297
|
+
return (0, -node.gpus_free, node.name)
|
|
298
|
+
return (1, -node.gpus_free, node.name)
|