nodeview-slurm 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: nodeview-slurm
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: GPU status of a Slurm cluster, from the user's point of view.
5
5
  Project-URL: Homepage, https://codeberg.org/danielrossi/NodeView
6
6
  Project-URL: Issues, https://codeberg.org/danielrossi/NodeView/issues
@@ -1,3 +1,3 @@
1
1
  """NodeView: GPU status of a Slurm cluster, from the user's point of view."""
2
2
 
3
- __version__ = "0.1.0"
3
+ __version__ = "0.2.0"
@@ -29,6 +29,10 @@ from . import palette
29
29
  from . import render
30
30
  from .profile import Profile, read_profile
31
31
  from .source import Snapshot, fetch
32
+ from .usage import GpuUsage, read_gpu_usage
33
+
34
+ # GPU hours move slowly and the call costs ~0.2 s: every few minutes is plenty
35
+ USAGE_EVERY = 300.0
32
36
 
33
37
 
34
38
  def _retranslate(cls, bindings) -> None:
@@ -137,6 +141,8 @@ class MainScreen(DataScreen):
137
141
  yield Static(id="fit")
138
142
  with Card(t('card.cluster'), classes="card"):
139
143
  yield Static(id="cluster")
144
+ with Card(t('card.gpu_hours'), classes="card"):
145
+ yield Static(id="usage")
140
146
  # at the bottom on purpose: this is info you rarely look at
141
147
  with Card(t('card.account'), classes="card"):
142
148
  yield Static(id="account")
@@ -146,8 +152,16 @@ class MainScreen(DataScreen):
146
152
  self.query_one("#mine", Static).update(render.my_jobs(snap))
147
153
  self.query_one("#fit", Static).update(render.where_do_i_fit(snap))
148
154
  self.query_one("#cluster", Static).update(render.cluster(snap))
155
+ self.apply_usage()
149
156
  self.apply_profile(getattr(self.app, "profile", None))
150
157
 
158
+ def apply_usage(self) -> None:
159
+ try:
160
+ widget = self.query_one("#usage", Static)
161
+ except Exception:
162
+ return
163
+ widget.update(render.gpu_usage(self.app.usage, loaded=self.app.usage_loaded))
164
+
151
165
  def apply_profile(self, prof) -> None:
152
166
  try:
153
167
  widget = self.query_one("#account", Static)
@@ -268,6 +282,8 @@ class NodeView(App):
268
282
  self._error: str | None = None
269
283
  self._restored = False
270
284
  self.profile: Profile | None = None
285
+ self.usage: GpuUsage | None = None
286
+ self.usage_loaded = False
271
287
 
272
288
  def on_mount(self) -> None:
273
289
  self._restore_prefs()
@@ -278,6 +294,8 @@ class NodeView(App):
278
294
  # budget, quotas and expiration cost ~1.7s and don't change minute to
279
295
  # minute: they're read only once at startup, outside the loop
280
296
  self._load_profile()
297
+ self._load_usage()
298
+ self.set_interval(USAGE_EVERY, self._load_usage)
281
299
 
282
300
  # --- data -------------------------------------------------------------
283
301
 
@@ -303,6 +321,23 @@ class NodeView(App):
303
321
  return
304
322
  self.call_from_thread(self._on_profile, prof)
305
323
 
324
+ @work(exclusive=True, thread=True, group="usage")
325
+ def _load_usage(self) -> None:
326
+ try:
327
+ usage = read_gpu_usage()
328
+ except Exception:
329
+ usage = None
330
+ self.call_from_thread(self._on_usage, usage)
331
+
332
+ def _on_usage(self, usage: GpuUsage | None) -> None:
333
+ # a failed read keeps the last good one: better slightly stale than empty
334
+ if usage is not None or not self.usage_loaded:
335
+ self.usage = usage
336
+ self.usage_loaded = True
337
+ screen = self.screen
338
+ if isinstance(screen, MainScreen):
339
+ screen.apply_usage()
340
+
306
341
  def _on_profile(self, prof: Profile) -> None:
307
342
  self.profile = prof
308
343
  screen = self.screen
@@ -0,0 +1,101 @@
1
+ """Board power of GPU models, for the energy estimate.
2
+
3
+ Slurm doesn't expose it: `CurrentWatts` stays 0 without an energy plugin, and
4
+ gres.conf has no field for it. So the model name Slurm gives (a node feature
5
+ like 'gpu_L40S_45G', a GRES type like 'a100', 'nvidia_h100_80gb_hbm3') is
6
+ normalized and looked up here.
7
+
8
+ Values are the manufacturers' board power (TDP / TBP) in watts, per GPU as
9
+ Slurm counts them: dual-GPU boards (K80, M60) are halved, and so on. Models
10
+ sold in several form factors take the datacenter one (A100 is SXM, 400 W);
11
+ the PCIe variants have their own keys when the name says so.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from typing import Optional
17
+
18
+ TDP_W = {
19
+ # --- NVIDIA datacenter ------------------------------------------------
20
+ 'k10': 112, 'k20': 225, 'k20x': 235, 'k40': 235, 'k80': 150,
21
+ 'm4': 50, 'm6': 100, 'm10': 56, 'm40': 250, 'm60': 150,
22
+ 'p4': 75, 'p6': 90, 'p40': 250, 'p100': 250, 'p100sxm2': 300,
23
+ # 'v100sxm2' needs its own key, or the V100S would claim it as a prefix
24
+ 'v100': 300, 'v100pcie': 250, 'v100s': 250, 'v100sxm2': 300, 'v100sxm3': 350,
25
+ 't4': 70, 't4g': 70,
26
+ 'a2': 60, 'a10': 150, 'a10g': 150, 'a16': 62, 'a30': 165, 'a40': 300,
27
+ 'a100': 400, 'a100pcie': 300, 'a800': 400, 'a800pcie': 300,
28
+ 'h100': 700, 'h100pcie': 350, 'h100nvl': 400, 'h800': 700, 'h800pcie': 350,
29
+ 'h200': 700, 'h200nvl': 600, 'h20': 400, 'gh200': 900,
30
+ 'l4': 72, 'l20': 275, 'l40': 300, 'l40s': 350,
31
+ 'b100': 700, 'b200': 1000, 'gb200': 1200, 'b300': 1100, 'gb300': 1400,
32
+
33
+ # --- NVIDIA workstation -----------------------------------------------
34
+ 'k2000': 51, 'k4000': 80, 'k5000': 122, 'k6000': 225,
35
+ 'm2000': 75, 'm4000': 120, 'm5000': 150, 'm6000': 250,
36
+ 'p400': 30, 'p600': 40, 'p620': 40, 'p1000': 47, 'p2000': 75, 'p2200': 75,
37
+ 'p4000': 105, 'p5000': 180, 'p6000': 250, 'gp100': 235, 'gv100': 250,
38
+ # Quadro RTX (Turing): the name is just 'RTX 6000' once 'rtx' is dropped
39
+ '4000': 160, '5000': 230, '6000': 295, '8000': 295,
40
+ 'a400': 50, 'a1000': 50, 'a2000': 70, 'a4000': 140, 'a4500': 200,
41
+ 'a5000': 230, 'a5500': 230, 'a6000': 300,
42
+ '2000ada': 70, '4000ada': 130, '4000sff': 70, '4500ada': 210,
43
+ '5000ada': 250, '5880ada': 285, '6000ada': 300,
44
+ 'pro4000': 140, 'pro4500': 200, 'pro5000': 300, 'pro6000': 600,
45
+ 'pro6000bmaxq': 300, 'pro6000blackwellmaxq': 300,
46
+ 'pro6000bse': 600, 'pro6000blackwellserver': 600,
47
+
48
+ # --- NVIDIA GeForce / TITAN -------------------------------------------
49
+ '750ti': 60, '960': 120, '970': 145, '980': 165, '980ti': 250,
50
+ 'titanx': 250, 'titanxp': 250, 'titanv': 250, 'titanrtx': 280,
51
+ '1050': 75, '1050ti': 75, '1060': 120, '1070': 150, '1070ti': 180,
52
+ '1080': 180, '1080ti': 250,
53
+ '1650': 75, '1660': 120, '1660super': 125, '1660ti': 120,
54
+ '2060': 160, '2060super': 175, '2070': 175, '2070super': 215,
55
+ '2080': 215, '2080super': 250, '2080ti': 250,
56
+ '3050': 130, '3060': 170, '3060ti': 200, '3070': 220, '3070ti': 290,
57
+ '3080': 320, '3080ti': 350, '3090': 350, '3090ti': 450,
58
+ '4060': 115, '4060ti': 160, '4070': 200, '4070super': 220, '4070ti': 285,
59
+ '4070tisuper': 285, '4080': 320, '4080super': 320, '4090': 450, '4090d': 425,
60
+ '5060': 145, '5060ti': 180, '5070': 250, '5070ti': 300, '5080': 360, '5090': 575,
61
+
62
+ # --- AMD --------------------------------------------------------------
63
+ 'mi25': 300, 'mi50': 300, 'mi60': 300, 'mi100': 300, 'mi210': 300,
64
+ 'mi250': 500, 'mi250x': 560, 'mi300a': 550, 'mi300x': 750, 'mi308x': 650,
65
+ 'mi325x': 1000, 'mi350x': 1000, 'mi355x': 1400,
66
+ 'provii': 250, 'prov520': 225, 'prov620': 300,
67
+ 'prow5700': 205, 'prow6600': 100, 'prow6800': 250,
68
+ 'prow7600': 130, 'prow7700': 190, 'prow7800': 260, 'prow7900': 295,
69
+ 'rx6800': 250, 'rx6800xt': 300, 'rx6900xt': 300, 'rx6950xt': 335,
70
+ 'rx7800xt': 263, 'rx7900gre': 260, 'rx7900xt': 315, 'rx7900xtx': 355,
71
+ 'rx9070': 220, 'rx9070xt': 304,
72
+
73
+ # --- Intel ------------------------------------------------------------
74
+ 'max1100': 300, 'max1550': 600, 'flex140': 75, 'flex170': 150,
75
+ 'a770': 225, 'a750': 225, 'b580': 190, 'gaudi2': 600, 'gaudi3': 900,
76
+ }
77
+
78
+ # vendor and family words that don't tell models apart
79
+ _NOISE = {'nvidia', 'tesla', 'geforce', 'quadro', 'amd', 'radeon', 'instinct',
80
+ 'intel', 'arc', 'data', 'center', 'datacenter', 'gpu', 'generation'}
81
+ _MEMORY = re.compile(r'\d+(gb?|gib)|hbm\d?e?|g?ddr\d+x?')
82
+
83
+
84
+ def model_key(model: str) -> str:
85
+ """'RTX A5000 24G' -> 'a5000'; 'nvidia_h100_80gb_hbm3' -> 'h100'."""
86
+ words = re.sub(r'[_\-/.]', ' ', (model or '').lower()).split()
87
+ words = [w for w in words if w not in _NOISE and not _MEMORY.fullmatch(w)]
88
+ return re.sub(r'^(rtx|gtx)', '', ''.join(words))
89
+
90
+
91
+ def tdp_watts(model: str) -> Optional[int]:
92
+ """Board power of a GPU model as Slurm names it, None if unknown.
93
+
94
+ Longest matching prefix wins: '2080ti' before '2080', 'l40s' before 'l40',
95
+ and 'a100sxm4' still finds 'a100'.
96
+ """
97
+ key = model_key(model)
98
+ if not key:
99
+ return None
100
+ matches = [k for k in TDP_W if key.startswith(k)]
101
+ return TDP_W[max(matches, key=len)] if matches else None
@@ -30,6 +30,7 @@ IT = {
30
30
  'card.fit': "DOVE ENTRO ADESSO",
31
31
  'card.cluster': "CLUSTER",
32
32
  'card.account': "IL MIO ACCOUNT",
33
+ 'card.gpu_hours': "LE MIE ORE GPU",
33
34
  'card.totals': "IL CLUSTER IN TRE NUMERI",
34
35
  'card.nodes': "NODI",
35
36
  'card.nodes_free': "NODI · solo quelli con GPU libere",
@@ -145,6 +146,23 @@ IT = {
145
146
  'acct.accounts': " ACCOUNT ",
146
147
  'acct.qos': " QOS ",
147
148
 
149
+ # --- GPU hours -------------------------------------------------------
150
+ 'usage.none': "nessun dato: sacct non risponde",
151
+ 'usage.last_days': "ULTIMI {n} GIORNI",
152
+ 'usage.per_month': "ORE GPU AL MESE",
153
+ 'usage.weekdays': "L M M G V S D",
154
+ 'usage.months': "gen feb mar apr mag giu lug ago set ott nov dic",
155
+ 'usage.day': "{day} {month}",
156
+ 'usage.avg_day': "media {h} h/giorno",
157
+ 'usage.peak': "picco {h} h il {day}",
158
+ 'usage.col_hours': "ore GPU",
159
+ 'usage.col_energy': "energia stimata",
160
+ 'usage.row_month': "questo mese",
161
+ 'usage.row_year': "anno {year}",
162
+ 'usage.row_window': "ultimi {n} mesi",
163
+ 'usage.row_avg': "media al mese",
164
+ 'usage.unpriced': "escluse {h} h su GPU senza TDP noto",
165
+
148
166
  # --- pending reasons -------------------------------------------------
149
167
  'reason.unknown.label': "in valutazione",
150
168
  'reason.unknown.detail': "lo scheduler la sta ancora guardando",
@@ -232,6 +250,7 @@ EN = {
232
250
  'card.fit': "WHERE I FIT RIGHT NOW",
233
251
  'card.cluster': "CLUSTER",
234
252
  'card.account': "MY ACCOUNT",
253
+ 'card.gpu_hours': "MY GPU HOURS",
235
254
  'card.totals': "THE CLUSTER IN THREE NUMBERS",
236
255
  'card.nodes': "NODES",
237
256
  'card.nodes_free': "NODES · only those with free GPUs",
@@ -340,6 +359,22 @@ EN = {
340
359
  'acct.accounts': " ACCOUNTS ",
341
360
  'acct.qos': " QOS ",
342
361
 
362
+ 'usage.none': "no data: sacct is not responding",
363
+ 'usage.last_days': "LAST {n} DAYS",
364
+ 'usage.per_month': "GPU HOURS PER MONTH",
365
+ 'usage.weekdays': "M T W T F S S",
366
+ 'usage.months': "Jan Feb Mar Apr May Jun Jul Aug Sep Oct Nov Dec",
367
+ 'usage.day': "{month} {day}",
368
+ 'usage.avg_day': "average {h} h/day",
369
+ 'usage.peak': "peak {h} h on {day}",
370
+ 'usage.col_hours': "GPU hours",
371
+ 'usage.col_energy': "estimated energy",
372
+ 'usage.row_month': "this month",
373
+ 'usage.row_year': "year {year}",
374
+ 'usage.row_window': "last {n} months",
375
+ 'usage.row_avg': "monthly average",
376
+ 'usage.unpriced': "{h} h on GPUs with unknown TDP left out",
377
+
343
378
  'reason.unknown.label': "under evaluation",
344
379
  'reason.unknown.detail': "the scheduler is still looking at it",
345
380
  'reason.QOSGrpGRES.label': "group GPU quota exhausted",
@@ -87,12 +87,20 @@ def _gpu_indices(gres_used: str) -> list:
87
87
  return sorted(set(out))
88
88
 
89
89
 
90
- def _model(features: str) -> str:
91
- """'gpu_RTX_A5000_24G' -> 'RTX A5000 24G'; several features -> first gpu_*"""
90
+ def _model(features: str, gres: str = '') -> str:
91
+ """'gpu_RTX_A5000_24G' -> 'RTX A5000 24G'; several features -> first gpu_*
92
+
93
+ Clusters that don't name the GPU in the features often do it in the GRES
94
+ type instead: 'gpu:a100:4(S:0-1)' -> 'a100'.
95
+ """
92
96
  for feat in (features or '').split(','):
93
97
  feat = feat.strip()
94
98
  if feat.startswith('gpu_'):
95
99
  return feat[4:].replace('_', ' ')
100
+ for entry in (gres or '').split(','):
101
+ parts = entry.split('(')[0].split(':')
102
+ if len(parts) == 3 and parts[0] == 'gpu' and parts[1] not in ('', '(null)', 'null'):
103
+ return parts[1]
96
104
  first = (features or '').split(',')[0].strip()
97
105
  return t('na') if first in ('', '(null)', 'null') else first
98
106
 
@@ -296,7 +304,7 @@ def read_nodes() -> list:
296
304
  state=d.get('State', '?'),
297
305
  reason=None if reason in ('', 'none', 'N/A') else reason,
298
306
  partitions=tuple(p for p in d.get('Partitions', '').split(',') if p),
299
- gpu_model=_model(d.get('AvailableFeatures', '')),
307
+ gpu_model=_model(d.get('AvailableFeatures', ''), d.get('Gres', '')),
300
308
  gpus_total=_gpu_count(d.get('Gres', '')),
301
309
  gpus_used=int(alloc.get('gpu', _gpu_count(d.get('GresUsed', '')))),
302
310
  gpu_indices_used=tuple(_gpu_indices(d.get('GresUsed', ''))),
@@ -9,12 +9,17 @@ the right job, and never let a color travel alone.
9
9
  """
10
10
  from __future__ import annotations
11
11
 
12
+ import math
13
+ from datetime import timedelta
14
+
12
15
  from rich import box
16
+ from rich.columns import Columns
13
17
  from rich.console import Group
14
18
  from rich.table import Table
15
19
  from rich.text import Text
16
20
 
17
21
  from . import analysis as an
22
+ from . import i18n
18
23
  from . import manual as manual_data
19
24
  from .i18n import t
20
25
  from . import palette
@@ -22,6 +27,8 @@ from .source import Snapshot
22
27
 
23
28
  BUSY, FREE = "●", "○" # single GPUs: the shape tells them apart even without color
24
29
  BLOCK, EMPTY = "█", "▁" # continuous meters
30
+ CELL = "▆▆" # one day in the heatmap: the missing top leaves a gap between weeks
31
+ EIGHTHS = " ▁▂▃▄▅▆▇█" # bar tops, in eighths of a row
25
32
 
26
33
  # internal key -> (icon, color role). The displayed text comes from the
27
34
  # catalog: the key must not change when the language changes.
@@ -424,6 +431,186 @@ def my_account(prof):
424
431
  return Group(*blocks)
425
432
 
426
433
 
434
+ def hours(value: float) -> str:
435
+ if 0 < value < 1:
436
+ return "<1"
437
+ sep = '.' if i18n.active() == 'it' else ','
438
+ return f"{value:,.0f}".replace(',', sep)
439
+
440
+
441
+ def _day_label(d) -> str:
442
+ months = t('usage.months').split()
443
+ return t('usage.day', day=d.day, month=months[d.month - 1])
444
+
445
+
446
+ def _nice_ceiling(value: float) -> float:
447
+ """The round number just above `value`, for the top of the axis."""
448
+ if value <= 0:
449
+ return 1.0
450
+ power = 10 ** math.floor(math.log10(value))
451
+ for step in (1, 2, 2.5, 5, 10):
452
+ if step * power >= value:
453
+ return step * power
454
+ return 10 * power
455
+
456
+
457
+ def _day_heatmap(u) -> Text:
458
+ """Calendar of the last days: a row per week, darker means more hours.
459
+
460
+ The scale is relative to the busiest day, which the legend names. Zero is
461
+ a neutral grey rather than the lightest blue: "nothing" must not look like
462
+ "a little". Each week's total sits on the right, so the numbers are there
463
+ even for those who can't tell the shades apart.
464
+ """
465
+ pal = _p()
466
+ by_day = dict(u.days)
467
+ first, last = u.days[0][0], u.days[-1][0]
468
+ peak_day, peak = max(u.days, key=lambda dh: dh[1])
469
+ monday = first - timedelta(days=first.weekday())
470
+ weeks = [monday + timedelta(weeks=i) for i in range((last - monday).days // 7 + 1)]
471
+ width = max(len(_day_label(w)) for w in weeks)
472
+
473
+ def cell(h):
474
+ return pal.rule if h <= 0 else pal.fill(h / peak)
475
+
476
+ out = Text(no_wrap=True, overflow="crop")
477
+ out.append(t('usage.last_days', n=len(u.days)), style=f"bold {pal.ink_muted}")
478
+ out.append(f" {hours(u.total_days)} h\n", style=f"bold {pal.ink}")
479
+ out.append(" " * (width + 2), style=pal.ink_muted)
480
+ out.append(" ".join(f"{w:<2}" for w in t('usage.weekdays').split()) + "\n",
481
+ style=pal.ink_muted)
482
+ for week in weeks:
483
+ out.append(f"{_day_label(week):>{width}} ", style=pal.ink_muted)
484
+ total = 0.0
485
+ for i in range(7):
486
+ d = week + timedelta(days=i)
487
+ if first <= d <= last:
488
+ total += by_day[d]
489
+ out.append(CELL, style=cell(by_day[d]))
490
+ else:
491
+ out.append(" ")
492
+ out.append(" ")
493
+ out.append(f"{hours(total):>6} h\n", style=pal.ink_soft)
494
+
495
+ out.append(" " * (width + 2) + "0 ", style=pal.ink_muted)
496
+ for color in (pal.rule,) + tuple(pal.sequential):
497
+ out.append(CELL, style=color)
498
+ out.append(" ")
499
+ out.append(f"{hours(peak)} h\n", style=pal.ink_muted)
500
+ out.append(t('usage.avg_day', h=hours(u.total_days / len(u.days))), style=pal.ink_soft)
501
+ if peak > 0:
502
+ out.append(" · ", style=pal.ink_muted)
503
+ out.append(t('usage.peak', h=hours(peak), day=_day_label(peak_day)),
504
+ style=pal.ink_soft)
505
+ return out
506
+
507
+
508
+ def _energy_cells(values) -> tuple:
509
+ """Unit and 3-character cells for the energy row under the bars.
510
+
511
+ Kilowatt-hours while they fit in three digits, megawatt-hours with one
512
+ decimal beyond: a column is four characters, and a fourth digit would glue
513
+ one month to the next.
514
+ """
515
+ sep = ',' if i18n.active() == 'it' else '.'
516
+ if max(values, default=0) < 999.5:
517
+ return 'kWh', ["·" if v <= 0 else hours(v) for v in values]
518
+ return 'MWh', ["·" if v <= 0 else f"{v / 1000:.1f}".replace('.', sep) for v in values]
519
+
520
+
521
+ def _month_bars(u, height: int = 8) -> Text:
522
+ """Vertical bars, one per month, on a single axis starting at zero.
523
+
524
+ One series, so one hue: the most visible step of the same ramp as the
525
+ heatmap. The axis carries its unit and the years sit on the baseline,
526
+ under the month they start with, so the chart reads without a legend.
527
+ Right under the month names, the estimated energy of each month; then a
528
+ small table with the totals, hours and energy side by side.
529
+ """
530
+ pal = _p()
531
+ values = [h for _, h in u.months]
532
+ top = _nice_ceiling(max(max(values), 10)) # an axis up to '<1' says nothing
533
+ ticks = {height: f"{hours(top)} h", height // 2: f"{hours(top / 2)} h"}
534
+ unit, cells = _energy_cells(u.month_kwh)
535
+ gutter = max(len(unit), *(len(v) for v in ticks.values()))
536
+ color = pal.sequential[-1] if pal.sequential else pal.ink
537
+
538
+ out = Text(no_wrap=True, overflow="crop")
539
+ out.append(t('usage.per_month') + "\n", style=f"bold {pal.ink_muted}")
540
+ for row in range(height, 0, -1):
541
+ tick = ticks.get(row)
542
+ out.append(f"{tick:>{gutter}} ┤" if tick else " " * gutter + " │", style=pal.ink_muted)
543
+ for v in values:
544
+ level = v / top * height - (row - 1)
545
+ if level >= 1:
546
+ bar = BLOCK * 3
547
+ elif level > 0:
548
+ bar = EIGHTHS[max(1, round(level * 8))] * 3
549
+ else:
550
+ bar = " "
551
+ out.append(" " + bar, style=color)
552
+ out.append("\n")
553
+
554
+ # baseline with the years written into it: '0 └─2025──────────2026───'
555
+ base = ["─"] * (4 * len(values))
556
+ for i, (m, _) in enumerate(u.months):
557
+ # under January, and under the first month unless January comes right after it
558
+ if m.month == 1 or (i == 0 and m.month < 12):
559
+ base[4 * i + 1:4 * i + 5] = f"{m.year}"
560
+ out.append(f"{'0':>{gutter}} └", style=pal.ink_muted)
561
+ for ch in base:
562
+ out.append(ch, style=pal.ink_muted if ch == "─" else pal.ink_soft)
563
+ out.append("\n")
564
+
565
+ names = t('usage.months').split()
566
+ out.append(" " * (gutter + 2) + "".join(f" {names[m.month - 1]:<3}" for m, _ in u.months)
567
+ + "\n", style=pal.ink_muted)
568
+ out.append(f"{unit:>{gutter}} ", style=pal.ink_muted)
569
+ out.append("".join(f"{c:>4}" for c in cells) + "\n", style=pal.ink_soft)
570
+
571
+ out.append("\n")
572
+ out.append(_totals_table(u))
573
+ return out
574
+
575
+
576
+ def _totals_table(u) -> Text:
577
+ """Hours and estimated energy for this month, this year and the window."""
578
+ pal = _p()
579
+ n = len(u.months)
580
+ rows = [
581
+ (t('usage.row_month'), u.months[-1][1], u.month_kwh[-1]),
582
+ (t('usage.row_year', year=u.taken_at.year), u.year_hours, u.year_kwh),
583
+ (t('usage.row_window', n=n), u.total_months, sum(u.month_kwh)),
584
+ (t('usage.row_avg'), u.total_months / n, sum(u.month_kwh) / n),
585
+ ]
586
+ head_h, head_e = t('usage.col_hours'), t('usage.col_energy')
587
+ label_w = max(len(r[0]) for r in rows)
588
+ hours_w = max(len(head_h), *(len(hours(r[1])) + 2 for r in rows))
589
+ energy_w = max(len(head_e), *(len(hours(r[2])) + 4 for r in rows))
590
+
591
+ out = Text(no_wrap=True, overflow="crop")
592
+ out.append(" " * label_w + f" {head_h:>{hours_w}} {head_e:>{energy_w}}\n",
593
+ style=pal.ink_muted)
594
+ for i, (label, h, kwh) in enumerate(rows):
595
+ strong = i < 2
596
+ out.append(f"{label:<{label_w}} ", style=pal.ink_soft)
597
+ out.append(f"{hours(h) + ' h':>{hours_w}} ", style=f"bold {pal.ink}" if strong else pal.ink_soft)
598
+ out.append(f"{hours(kwh) + ' kWh':>{energy_w}}", style=f"bold {pal.ink}" if strong else pal.ink_soft)
599
+ if i < len(rows) - 1:
600
+ out.append("\n")
601
+ if u.unpriced_hours >= 1:
602
+ out.append("\n" + t('usage.unpriced', h=hours(u.unpriced_hours)), style=pal.ink_muted)
603
+ return out
604
+
605
+
606
+ def gpu_usage(u, loaded: bool = True):
607
+ """GPU hours: the last days as a calendar, the last months as bars."""
608
+ pal = _p()
609
+ if u is None:
610
+ return Text(t('app.loading') if not loaded else t('usage.none'), style=pal.ink_muted)
611
+ return Columns([_day_heatmap(u), _month_bars(u)], padding=(0, 6))
612
+
613
+
427
614
  # --------------------------------------------------------------------------
428
615
  # "cluster" screen
429
616
  # --------------------------------------------------------------------------
@@ -67,6 +67,8 @@ def print_once(width=None, cluster=False) -> int:
67
67
  console.print(_panel(render.my_jobs(snap), t('card.my_jobs')))
68
68
  console.print(_panel(render.where_do_i_fit(snap), t('card.fit')))
69
69
  console.print(_panel(render.cluster(snap), t('card.cluster')))
70
+ from .usage import read_gpu_usage
71
+ console.print(_panel(render.gpu_usage(read_gpu_usage()), t('card.gpu_hours')))
70
72
  from .profile import read_profile
71
73
  console.print(_panel(render.my_account(read_profile()), t('card.account')))
72
74
  return 0
@@ -0,0 +1,204 @@
1
+ """GPU hours you've used, per day and per month, and the energy they took.
2
+
3
+ A single `sacct` call covers the whole window (a year of jobs and steps takes
4
+ ~0.6 s) and each job's GPUs × time is split across the days and months it
5
+ overlaps. That's what `sreport` does too, but sreport rolls up once an hour and
6
+ misses the jobs still running: here today is already counted.
7
+
8
+ `-D` matters: without it a requeued job shows only its last run, and the hours
9
+ of the earlier ones vanish.
10
+
11
+ Energy is an estimate, not a measurement: without an energy plugin Slurm
12
+ records `ConsumedEnergy=0`. It combines what Slurm does know, the GPU model of
13
+ the node and each job's average GPU utilization, with the model's board power.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import re
18
+ import subprocess
19
+ from dataclasses import dataclass
20
+ from datetime import date, datetime, timedelta
21
+ from typing import Optional
22
+
23
+ from .gpu_power import tdp_watts
24
+ from .nodes import _expand_nodelist, read_nodes, short_name
25
+
26
+ DAYS = 30
27
+ MONTHS = 12
28
+
29
+ _GPUS = re.compile(r'(?:^|,)gres/gpu=(\d+)')
30
+ _UTIL = re.compile(r'(?:^|,)gres/gpuutil=(\d+)')
31
+ _TYPED = re.compile(r'(?:^|,)gres/gpu:([^=,]+)=\d+') # 'gres/gpu:a100=2'
32
+
33
+ # An allocated GPU draws power even at 0% utilization: roughly this share of
34
+ # its board power, the rest grows with utilization.
35
+ IDLE_SHARE = 0.15
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class GpuUsage:
40
+ days: tuple # ((date, hours), ...) oldest first, ending today
41
+ months: tuple # ((first day of the month, hours), ...) oldest first
42
+ month_kwh: tuple # estimated energy, aligned with `months`
43
+ month_priced: tuple # the GPU hours behind `month_kwh`: those with a known model
44
+ taken_at: datetime
45
+
46
+ @property
47
+ def total_days(self) -> float:
48
+ return sum(h for _, h in self.days)
49
+
50
+ @property
51
+ def total_months(self) -> float:
52
+ return sum(h for _, h in self.months)
53
+
54
+ def _this_year(self, values) -> float:
55
+ return sum(v for (m, _), v in zip(self.months, values) if m.year == self.taken_at.year)
56
+
57
+ @property
58
+ def year_kwh(self) -> float:
59
+ """January to December of the current year: the window always covers it."""
60
+ return self._this_year(self.month_kwh)
61
+
62
+ @property
63
+ def year_hours(self) -> float:
64
+ return self._this_year(h for _, h in self.months)
65
+
66
+ @property
67
+ def unpriced_hours(self) -> float:
68
+ """GPU hours on models without a known board power, left out of the energy."""
69
+ return self._this_year(h for _, h in self.months) - self._this_year(self.month_priced)
70
+
71
+
72
+ def _month_start(d: date, back: int = 0) -> date:
73
+ index = d.year * 12 + (d.month - 1) - back
74
+ return date(index // 12, index % 12 + 1, 1)
75
+
76
+
77
+ def _ts(value: str) -> Optional[datetime]:
78
+ try:
79
+ return datetime.strptime(value.strip(), '%Y-%m-%dT%H:%M:%S')
80
+ except ValueError:
81
+ return None # 'None' / 'Unknown': not started yet
82
+
83
+
84
+ def _gpu_count(tres: str) -> int:
85
+ """GPUs in an AllocTRES: the plain 'gres/gpu=N', or the typed ones summed
86
+ when the plain one is missing."""
87
+ match = _GPUS.search(tres)
88
+ if match:
89
+ return int(match.group(1))
90
+ return sum(int(m.group(1)) for m in re.finditer(r'(?:^|,)gres/gpu:[^=,]+=(\d+)', tres))
91
+
92
+
93
+ def _runs(text: str, now: datetime, node_tdp: dict) -> list:
94
+ """(start, end, gpus, watts) for every run that held at least one GPU.
95
+
96
+ Allocation lines give time, GPUs and node; step lines ('123.batch', '123.0')
97
+ carry the measured utilization, and the busiest step stands for the job.
98
+ The GPU model comes from the node, or failing that from a typed GRES in
99
+ the job ('gres/gpu:a100=2'); `watts` is None when neither is known. Jobs without a
100
+ utilization reading (too short, still running) get the average of the
101
+ others, weighted by GPU hours.
102
+ """
103
+ allocs, util = [], {}
104
+ for line in text.splitlines():
105
+ parts = line.split('|')
106
+ if len(parts) < 6:
107
+ continue
108
+ jobid, start, end, tres, nodelist, usage = parts[:6]
109
+ if '.' in jobid:
110
+ match = _UTIL.search(usage)
111
+ if match:
112
+ base = jobid.split('.')[0]
113
+ util[base] = max(util.get(base, 0), int(match.group(1)))
114
+ continue
115
+ begin = _ts(start)
116
+ if begin is None:
117
+ continue
118
+ gpus = _gpu_count(tres)
119
+ finish = _ts(end) or now # still running
120
+ if gpus and finish > begin:
121
+ nodes = _expand_nodelist(nodelist)
122
+ tdp = node_tdp.get(short_name(nodes[0])) if nodes else None
123
+ typed = _TYPED.search(tres)
124
+ if tdp is None and typed:
125
+ tdp = tdp_watts(typed.group(1))
126
+ allocs.append((jobid, begin, finish, gpus, tdp))
127
+
128
+ def share(jobid, gpus):
129
+ value = util[jobid]
130
+ return min(100, value / gpus if value > 100 else value) / 100
131
+
132
+ known = [(a, (a[2] - a[1]).total_seconds() * a[3]) for a in allocs if a[0] in util]
133
+ weight = sum(w for _, w in known)
134
+ fallback = sum(share(a[0], a[3]) * w for a, w in known) / weight if weight else 0.5
135
+
136
+ out = []
137
+ for jobid, begin, finish, gpus, tdp in allocs:
138
+ u = share(jobid, gpus) if jobid in util else fallback
139
+ watts = gpus * tdp * (IDLE_SHARE + (1 - IDLE_SHARE) * u) if tdp else None
140
+ out.append((begin, finish, gpus, watts))
141
+ return out
142
+
143
+
144
+ def _spread(intervals: list, edges: list) -> list:
145
+ """Sum of weight × hours falling in each [edges[i], edges[i+1]) bucket.
146
+
147
+ `intervals` holds (start, end, weight): GPUs for GPU hours, kW for kWh.
148
+ """
149
+ totals = [0.0] * (len(edges) - 1)
150
+ lo, hi = edges[0], edges[-1]
151
+ for start, end, weight in intervals:
152
+ start, end = max(start, lo), min(end, hi)
153
+ if end <= start:
154
+ continue
155
+ for i in range(len(totals)):
156
+ a, b = max(start, edges[i]), min(end, edges[i + 1])
157
+ if b > a:
158
+ totals[i] += weight * (b - a).total_seconds() / 3600
159
+ return totals
160
+
161
+
162
+ def summarize(text: str, now: datetime, node_tdp: Optional[dict] = None) -> GpuUsage:
163
+ today = now.date()
164
+ day_starts = [today - timedelta(days=DAYS - 1 - i) for i in range(DAYS)]
165
+ month_starts = [_month_start(today, MONTHS - 1 - i) for i in range(MONTHS)]
166
+ runs = _runs(text, now, node_tdp or {})
167
+
168
+ def edges(starts, after):
169
+ return [datetime.combine(d, datetime.min.time()) for d in starts + [after]]
170
+
171
+ by_day = edges(day_starts, today + timedelta(days=1))
172
+ by_month = edges(month_starts, _month_start(today, -1))
173
+ gpus = [(s, e, g) for s, e, g, _ in runs]
174
+ priced = [(s, e, g) for s, e, g, w in runs if w is not None]
175
+ kilowatts = [(s, e, w / 1000) for s, e, _, w in runs if w is not None]
176
+ return GpuUsage(days=tuple(zip(day_starts, _spread(gpus, by_day))),
177
+ months=tuple(zip(month_starts, _spread(gpus, by_month))),
178
+ month_kwh=tuple(_spread(kilowatts, by_month)),
179
+ month_priced=tuple(_spread(priced, by_month)),
180
+ taken_at=now)
181
+
182
+
183
+ def _node_tdp() -> dict:
184
+ """Node name -> board power of its GPUs; empty if the nodes can't be read."""
185
+ try:
186
+ return {n.name: tdp_watts(n.gpu_model) for n in read_nodes()}
187
+ except Exception:
188
+ return {}
189
+
190
+
191
+ def read_gpu_usage() -> Optional[GpuUsage]:
192
+ now = datetime.now()
193
+ since = min(_month_start(now.date(), MONTHS - 1),
194
+ now.date() - timedelta(days=DAYS - 1))
195
+ try:
196
+ out = subprocess.run(
197
+ ['sacct', '-n', '-D', '-P', '-S', since.isoformat(), '-E', 'now',
198
+ '-o', 'JobID,Start,End,AllocTRES,NodeList,TRESUsageInAve'],
199
+ capture_output=True, text=True, timeout=25)
200
+ except (OSError, subprocess.SubprocessError):
201
+ return None
202
+ if out.returncode != 0:
203
+ return None
204
+ return summarize(out.stdout, now, _node_tdp())
File without changes