nodeview-slurm 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nodeview/__init__.py +3 -0
- nodeview/__main__.py +32 -0
- nodeview/analysis.py +298 -0
- nodeview/app.py +423 -0
- nodeview/app.tcss +35 -0
- nodeview/config.py +61 -0
- nodeview/i18n.py +455 -0
- nodeview/manual.py +176 -0
- nodeview/nodes.py +318 -0
- nodeview/palette.py +325 -0
- nodeview/profile.py +284 -0
- nodeview/render.py +688 -0
- nodeview/snapshot.py +72 -0
- nodeview/source.py +306 -0
- nodeview_slurm-0.1.0.dist-info/METADATA +41 -0
- nodeview_slurm-0.1.0.dist-info/RECORD +18 -0
- nodeview_slurm-0.1.0.dist-info/WHEEL +4 -0
- nodeview_slurm-0.1.0.dist-info/entry_points.txt +2 -0
nodeview/profile.py
ADDED
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
"""Everything the cluster knows about you.
|
|
2
|
+
|
|
3
|
+
This is information you rarely look at (project budgets, disk quotas, account
|
|
4
|
+
expiration) but that matters a lot when it does: an exhausted budget or a full
|
|
5
|
+
quota blocks your jobs without telling you why.
|
|
6
|
+
|
|
7
|
+
It costs ~1.7 s in total, almost all spent in the three local commands
|
|
8
|
+
`susage`, `squota` and `sexpiration`, which read a separately updated cache.
|
|
9
|
+
Too much for the job refresh loop: it's read at a much slower pace.
|
|
10
|
+
|
|
11
|
+
Each source is independent and may be missing: on a cluster without the local
|
|
12
|
+
commands the other sections still show up.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
import subprocess
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from datetime import datetime, timedelta
|
|
20
|
+
from typing import Optional
|
|
21
|
+
|
|
22
|
+
_ANSI = re.compile(r'\x1b\[[0-9;]*m')
|
|
23
|
+
|
|
24
|
+
HISTORY_DAYS = 30
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _run(args, timeout=25) -> Optional[str]:
|
|
28
|
+
try:
|
|
29
|
+
out = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
|
|
30
|
+
except (OSError, subprocess.SubprocessError):
|
|
31
|
+
return None
|
|
32
|
+
if out.returncode != 0 and not out.stdout.strip():
|
|
33
|
+
return None
|
|
34
|
+
return out.stdout
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _parse_tabulate(text: str):
|
|
38
|
+
"""Parses the fixed-width tables printed by the local commands.
|
|
39
|
+
|
|
40
|
+
The row of dashes under the header gives the exact column boundaries:
|
|
41
|
+
more reliable than splitting on spaces, since the headers contain some
|
|
42
|
+
("Total usage (%)").
|
|
43
|
+
"""
|
|
44
|
+
lines = _ANSI.sub('', text or '').splitlines()
|
|
45
|
+
sep = None
|
|
46
|
+
for i, line in enumerate(lines):
|
|
47
|
+
stripped = line.strip()
|
|
48
|
+
if stripped and set(stripped) <= set('- ') and '-' in stripped:
|
|
49
|
+
sep = i
|
|
50
|
+
break
|
|
51
|
+
if sep is None or sep == 0:
|
|
52
|
+
return [], []
|
|
53
|
+
|
|
54
|
+
spans = [(m.start(), m.end()) for m in re.finditer(r'-+', lines[sep])]
|
|
55
|
+
if not spans:
|
|
56
|
+
return [], []
|
|
57
|
+
spans[-1] = (spans[-1][0], 10_000) # the last column may overflow
|
|
58
|
+
|
|
59
|
+
def cells(line):
|
|
60
|
+
return [line[a:b].strip() for a, b in spans]
|
|
61
|
+
|
|
62
|
+
headers = cells(lines[sep - 1])
|
|
63
|
+
rows = []
|
|
64
|
+
for line in lines[sep + 1:]:
|
|
65
|
+
if not line.strip():
|
|
66
|
+
break
|
|
67
|
+
row = cells(line)
|
|
68
|
+
if any(row):
|
|
69
|
+
rows.append(row)
|
|
70
|
+
return headers, rows
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _num(value) -> Optional[float]:
|
|
74
|
+
try:
|
|
75
|
+
return float(str(value).replace(',', '').strip())
|
|
76
|
+
except (TypeError, ValueError):
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass(frozen=True)
|
|
81
|
+
class Budget:
|
|
82
|
+
"""A project's compute budget, in standard hours."""
|
|
83
|
+
project: str
|
|
84
|
+
used: Optional[float]
|
|
85
|
+
total: Optional[float]
|
|
86
|
+
pct: Optional[float]
|
|
87
|
+
year_used: Optional[float]
|
|
88
|
+
year_total: Optional[float]
|
|
89
|
+
year_pct: Optional[float]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class DiskQuota:
|
|
94
|
+
filesystem: str
|
|
95
|
+
owner: str
|
|
96
|
+
used_gb: Optional[float]
|
|
97
|
+
quota_gb: Optional[float]
|
|
98
|
+
pct: Optional[float]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass(frozen=True)
|
|
102
|
+
class Expiration:
|
|
103
|
+
date: str
|
|
104
|
+
days_left: Optional[int]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True)
|
|
108
|
+
class Share:
|
|
109
|
+
"""Fairshare: how much you've used relative to your allotted share."""
|
|
110
|
+
account: str
|
|
111
|
+
norm_shares: Optional[float]
|
|
112
|
+
effective_usage: Optional[float]
|
|
113
|
+
factor: Optional[float]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass(frozen=True)
|
|
117
|
+
class History:
|
|
118
|
+
days: int
|
|
119
|
+
completed: int = 0
|
|
120
|
+
failed: int = 0
|
|
121
|
+
cancelled: int = 0
|
|
122
|
+
timeout: int = 0
|
|
123
|
+
other: int = 0
|
|
124
|
+
|
|
125
|
+
@property
|
|
126
|
+
def finished(self) -> int:
|
|
127
|
+
return self.completed + self.failed + self.cancelled + self.timeout + self.other
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def failure_rate(self) -> Optional[float]:
|
|
131
|
+
return (self.failed + self.timeout) / self.finished if self.finished else None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass
|
|
135
|
+
class Profile:
|
|
136
|
+
budgets: list = field(default_factory=list)
|
|
137
|
+
quotas: list = field(default_factory=list)
|
|
138
|
+
expiration: Optional[Expiration] = None
|
|
139
|
+
shares: list = field(default_factory=list)
|
|
140
|
+
accounts: dict = field(default_factory=dict) # account -> list of QOS
|
|
141
|
+
history: Optional[History] = None
|
|
142
|
+
taken_at: float = 0.0
|
|
143
|
+
duration: float = 0.0
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def empty(self) -> bool:
|
|
147
|
+
return not (self.budgets or self.quotas or self.expiration
|
|
148
|
+
or self.shares or self.accounts or self.history)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _read_budgets() -> list:
|
|
152
|
+
headers, rows = _parse_tabulate(_run(['susage']) or '')
|
|
153
|
+
if not rows:
|
|
154
|
+
return []
|
|
155
|
+
|
|
156
|
+
def find(*needles):
|
|
157
|
+
for i, h in enumerate(headers):
|
|
158
|
+
low = h.lower()
|
|
159
|
+
if all(n in low for n in needles):
|
|
160
|
+
return i
|
|
161
|
+
return None
|
|
162
|
+
|
|
163
|
+
i_proj = find('project') or 0
|
|
164
|
+
i_tot = find('total', 'budget')
|
|
165
|
+
i_use = find('total', 'usage')
|
|
166
|
+
i_pct = find('total', 'usage', '%')
|
|
167
|
+
# "Total usage" and "Total usage (%)" look alike: exclude the percentage
|
|
168
|
+
if i_use is not None and i_use == i_pct:
|
|
169
|
+
i_use = next((i for i, h in enumerate(headers)
|
|
170
|
+
if 'usage' in h.lower() and '%' not in h and 'total' in h.lower()), None)
|
|
171
|
+
i_ytot = find('budget', '20')
|
|
172
|
+
i_yuse = next((i for i, h in enumerate(headers)
|
|
173
|
+
if 'usage' in h.lower() and '%' not in h and re.search(r'20\d\d', h)), None)
|
|
174
|
+
i_ypct = next((i for i, h in enumerate(headers)
|
|
175
|
+
if 'usage' in h.lower() and '%' in h and re.search(r'20\d\d', h)), None)
|
|
176
|
+
|
|
177
|
+
def cell(row, idx):
|
|
178
|
+
return _num(row[idx]) if idx is not None and idx < len(row) else None
|
|
179
|
+
|
|
180
|
+
return [Budget(project=row[i_proj], used=cell(row, i_use), total=cell(row, i_tot),
|
|
181
|
+
pct=cell(row, i_pct), year_used=cell(row, i_yuse),
|
|
182
|
+
year_total=cell(row, i_ytot), year_pct=cell(row, i_ypct))
|
|
183
|
+
for row in rows if row[i_proj]]
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _read_quotas() -> list:
|
|
187
|
+
headers, rows = _parse_tabulate(_run(['squota']) or '')
|
|
188
|
+
if not rows:
|
|
189
|
+
return []
|
|
190
|
+
i_gb = [i for i, h in enumerate(headers) if 'gb' in h.lower() and '%' not in h]
|
|
191
|
+
i_pct = next((i for i, h in enumerate(headers) if '%' in h and 'gb' in h.lower()), None)
|
|
192
|
+
i_use = i_gb[0] if i_gb else None
|
|
193
|
+
i_quota = i_gb[1] if len(i_gb) > 1 else None
|
|
194
|
+
|
|
195
|
+
out = []
|
|
196
|
+
for row in rows:
|
|
197
|
+
if len(row) < 2 or not row[0]:
|
|
198
|
+
continue
|
|
199
|
+
out.append(DiskQuota(
|
|
200
|
+
filesystem=row[0], owner=row[1],
|
|
201
|
+
used_gb=_num(row[i_use]) if i_use is not None and i_use < len(row) else None,
|
|
202
|
+
quota_gb=_num(row[i_quota]) if i_quota is not None and i_quota < len(row) else None,
|
|
203
|
+
pct=_num(row[i_pct]) if i_pct is not None and i_pct < len(row) else None,
|
|
204
|
+
))
|
|
205
|
+
return out
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _read_expiration() -> Optional[Expiration]:
|
|
209
|
+
headers, rows = _parse_tabulate(_run(['sexpiration']) or '')
|
|
210
|
+
if not rows:
|
|
211
|
+
return None
|
|
212
|
+
row = rows[0]
|
|
213
|
+
date = row[1] if len(row) > 1 else ''
|
|
214
|
+
days = _num(row[2]) if len(row) > 2 else None
|
|
215
|
+
return Expiration(date=date, days_left=int(days) if days is not None else None)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _read_shares() -> list:
|
|
219
|
+
text = _run(['sshare', '-U', '-P', '-n',
|
|
220
|
+
'-o', 'Account,NormShares,EffectvUsage,FairShare'])
|
|
221
|
+
out = []
|
|
222
|
+
for line in (text or '').splitlines():
|
|
223
|
+
parts = [p.strip() for p in _ANSI.sub('', line).split('|')]
|
|
224
|
+
if len(parts) < 4 or not parts[0]:
|
|
225
|
+
continue
|
|
226
|
+
out.append(Share(account=parts[0], norm_shares=_num(parts[1]),
|
|
227
|
+
effective_usage=_num(parts[2]), factor=_num(parts[3])))
|
|
228
|
+
return out
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _read_accounts() -> dict:
|
|
232
|
+
import os
|
|
233
|
+
user = os.environ.get('USER') or ''
|
|
234
|
+
text = _run(['sacctmgr', '-nP', 'show', 'assoc', f'user={user}',
|
|
235
|
+
'format=Account,QOS'])
|
|
236
|
+
accounts = {}
|
|
237
|
+
for line in (text or '').splitlines():
|
|
238
|
+
parts = line.split('|')
|
|
239
|
+
if len(parts) < 2 or not parts[0].strip():
|
|
240
|
+
continue
|
|
241
|
+
accounts[parts[0].strip()] = [q for q in parts[1].split(',') if q.strip()]
|
|
242
|
+
return accounts
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _read_history(days: int = HISTORY_DAYS) -> Optional[History]:
|
|
246
|
+
since = (datetime.now() - timedelta(days=days)).strftime('%Y-%m-%d')
|
|
247
|
+
text = _run(['sacct', '-n', '-X', '-P', '-S', since, '-o', 'State'])
|
|
248
|
+
if text is None:
|
|
249
|
+
return None
|
|
250
|
+
counts = {'completed': 0, 'failed': 0, 'cancelled': 0, 'timeout': 0, 'other': 0}
|
|
251
|
+
for line in text.splitlines():
|
|
252
|
+
state = line.strip().upper()
|
|
253
|
+
if not state:
|
|
254
|
+
continue
|
|
255
|
+
if state.startswith('COMPLETED'):
|
|
256
|
+
counts['completed'] += 1
|
|
257
|
+
elif state.startswith('FAILED') or state.startswith('NODE_FAIL') or state.startswith('OUT_OF'):
|
|
258
|
+
counts['failed'] += 1
|
|
259
|
+
elif state.startswith('CANCELLED'):
|
|
260
|
+
counts['cancelled'] += 1
|
|
261
|
+
elif state.startswith('TIMEOUT'):
|
|
262
|
+
counts['timeout'] += 1
|
|
263
|
+
elif state.startswith('RUNNING') or state.startswith('PENDING'):
|
|
264
|
+
continue # not finished yet, don't count
|
|
265
|
+
else:
|
|
266
|
+
counts['other'] += 1
|
|
267
|
+
return History(days=days, **counts)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def read_profile() -> Profile:
|
|
271
|
+
import time
|
|
272
|
+
started = time.perf_counter()
|
|
273
|
+
profile = Profile()
|
|
274
|
+
# each source on its own: if a command is missing, the others remain
|
|
275
|
+
for attr, reader in (('budgets', _read_budgets), ('quotas', _read_quotas),
|
|
276
|
+
('expiration', _read_expiration), ('shares', _read_shares),
|
|
277
|
+
('accounts', _read_accounts), ('history', _read_history)):
|
|
278
|
+
try:
|
|
279
|
+
setattr(profile, attr, reader())
|
|
280
|
+
except Exception:
|
|
281
|
+
pass
|
|
282
|
+
profile.taken_at = time.time()
|
|
283
|
+
profile.duration = time.perf_counter() - started
|
|
284
|
+
return profile
|