tuneplane-node 0.3.15__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,497 @@
1
+ """The node daemon: a narrow job API on a GPU node.
2
+
3
+ What it is
4
+ ──────────────────────────────────────────────────────────────────────────────
5
+ The node-side component of the node backend (kubelet is the analogue under K8s): the
6
+ console calls it over HTTP to launch, observe and stop job containers. **The daemon is
7
+ control-plane infrastructure and runs no training** -- a job is still a per-job
8
+ container, with exactly the local backend's semantics (one throwaway container per
9
+ job, destroyed when it finishes).
10
+
11
+ Reuse rather than rewrite
12
+ ──────────────────────────────────────────────────────────────────────────────
13
+ Every core mechanism inside the node is the local backend's implementation:
14
+ container runtime node/runtime.py (the docker / podman CLI wrapper)
15
+ GPU allocation node/allocator.py (allocation lock, labels as the truth,
16
+ health probing)
17
+ image prefetch the same explicit pull on its own long timeout as LocalExecutor
18
+
19
+ So restarting a node loses no state at all: the truth about occupancy is in the
20
+ container labels and one `docker ps` rebuilds it -- the property the local backend's
21
+ design bought in the first place.
22
+
23
+ Security model
24
+ ──────────────────────────────────────────────────────────────────────────────
25
+ Only the job verbs are exposed (launch/observe/stop/logs/reap), never arbitrary
26
+ container operations; a bearer token is mandatory (without one the daemon refuses to
27
+ start rather than running open). Against the alternative -- the console holding each
28
+ node's docker socket, which is equivalent to root -- this narrows the privilege
29
+ surface.
30
+
31
+ Asynchronous semantics of an image pull
32
+ ──────────────────────────────────────────────────────────────────────────────
33
+ A first pull of tens of GB takes tens of minutes and cannot hang off a single HTTP
34
+ request. When the image is not on the machine, launch returns phase=pulling at once
35
+ and a background task finishes the pull, then allocates GPUs and starts the container;
36
+ observe on the console side reports PENDING with the pull explained meanwhile. A
37
+ failed pull becomes an explicit failed state carrying the reason, rather than
38
+ disappearing quietly. If the daemon restarts mid-pull the pending record is gone and
39
+ no container exists, so reconciliation converges on STOPPED -- which the docs state,
40
+ and resubmitting is the answer.
41
+ """
42
+ from __future__ import annotations
43
+
44
+ import asyncio
45
+ import logging
46
+ import secrets
47
+ import shutil
48
+ import time
49
+ from dataclasses import dataclass, field
50
+ from weakref import WeakValueDictionary
51
+
52
+ from fastapi import Depends, FastAPI, HTTPException, Request
53
+
54
+ from tuneplane_node.allocator import GpuAllocator, NoCapacity
55
+ from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
56
+ from tuneplane_node.runtime import age_seconds as _age_seconds
57
+ from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
58
+
59
+ log = logging.getLogger(__name__)
60
+
61
+
62
+ #: How long a failed pending record is kept (seconds). It has to outlive a round of
63
+ #: reconciliation -- the console must see the failure reason before it can record it in
64
+ #: the ledger -- and after that it is nothing but memory held for no reason.
65
+ _FAILED_PENDING_TTL_S = 1800.0
66
+
67
+
68
+ @dataclass
69
+ class PendingLaunch:
70
+ """The explicit state of one asynchronous launch, while its image is pulling."""
71
+
72
+ run_id: str
73
+ image: str
74
+ #: ★ How many cards this job has been promised. It has to be recorded: the pull
75
+ #: takes tens of minutes, no container exists during it, and the truth about
76
+ #: occupancy is the container labels -- so occupancy() cannot see the job at all.
77
+ #: Without this field health reports the promised cards as free, the control
78
+ #: plane's _place puts later jobs on the same node, and by the time the image is
79
+ #: pulled the cards are long gone and the job ends in "please resubmit". Queueing
80
+ #: for tens of minutes and then running nothing is the worst kind of failure.
81
+ requested_gpus: int = 0
82
+ phase: str = "pulling" # pulling | starting | failed
83
+ error: str = ""
84
+ started_at: float = field(default_factory=time.time)
85
+
86
+ @property
87
+ def reserves_gpus(self) -> bool:
88
+ """Still waiting for cards; a failed launch holds none."""
89
+ return self.phase != "failed"
90
+
91
+
92
+ class NodeState:
93
+ """The daemon's in-process state. runtime and allocator are injectable, for test doubles."""
94
+
95
+ def __init__(self, settings, *, runtime=None, allocator=None):
96
+ self.settings = settings
97
+ self._runtime = runtime
98
+ self._allocator = allocator
99
+ self.pending: dict[str, PendingLaunch] = {}
100
+ #: Strong references to the background pull tasks. asyncio holds only a weak
101
+ #: one, so a create_task whose result is dropped can be garbage-collected
102
+ #: mid-pull -- and the job then sits in pulling forever, with no exception and
103
+ #: nothing in the log.
104
+ self._tasks: set = set()
105
+ #: Background launches, by container ref, so a stop can find the one it
106
+ #: is stopping. Without this a stop during a pull popped the pending
107
+ #: record and returned ok, while the task went on to allocate GPUs and
108
+ #: start the container minutes later -- the console had already released
109
+ #: the quota, so nothing ever reclaimed those cards.
110
+ self._launches: dict[str, "asyncio.Task"] = {}
111
+ #: Refs whose launch was cancelled. Checked at both await boundaries of
112
+ #: `_pull_then_start`, because cancelling a task only helps if it is
113
+ #: suspended: a cancel that lands while the container is being created
114
+ #: has to be answered by tearing that container down instead.
115
+ self.cancelled: set[str] = set()
116
+ self._control_locks: WeakValueDictionary[str, asyncio.Lock] = WeakValueDictionary()
117
+
118
+ def control_lock(self, ref: str) -> asyncio.Lock:
119
+ """Serialize create/start with stop/remove for one container.
120
+
121
+ Holders and waiters keep the lock alive; completed container identities
122
+ do not accumulate in a daemon that may run for months.
123
+ """
124
+ lock = self._control_locks.get(ref)
125
+ if lock is None:
126
+ lock = asyncio.Lock()
127
+ self._control_locks[ref] = lock
128
+ return lock
129
+
130
+ def spawn(self, coro, *, ref: str = "") -> None:
131
+ """Start a background task and hold a reference to it until it finishes."""
132
+ task = asyncio.get_running_loop().create_task(coro)
133
+ self._tasks.add(task)
134
+ task.add_done_callback(self._tasks.discard)
135
+ if ref:
136
+ self._launches[ref] = task
137
+ task.add_done_callback(lambda _t, r=ref: self._launches.pop(r, None))
138
+
139
+ def begin_launch(self, ref: str) -> None:
140
+ """A fresh launch for `ref` clears any tombstone an earlier one left."""
141
+ self.cancelled.discard(ref)
142
+
143
+ def cancel_launch(self, ref: str) -> None:
144
+ """Stop a launch in flight, whatever stage it has reached."""
145
+ self.cancelled.add(ref)
146
+ task = self._launches.pop(ref, None)
147
+ if task is not None and not task.done():
148
+ task.cancel()
149
+ self.pending.pop(ref, None)
150
+
151
+ def is_cancelled(self, ref: str) -> bool:
152
+ return ref in self.cancelled
153
+
154
+ def reserved_gpus(self) -> int:
155
+ """Cards promised to pending jobs whose container has not started yet."""
156
+ self.expire_failed_pending()
157
+ return sum(p.requested_gpus for p in self.pending.values() if p.reserves_gpus)
158
+
159
+ def expire_failed_pending(self, now: float | None = None) -> int:
160
+ """Drop failed records that have expired.
161
+
162
+ On the normal path the console reconciles to a terminal state and then calls
163
+ remove/stop, which takes the record with it. But a FAILED training job stays in
164
+ the ledger so it can be retried automatically, remove never comes, and the
165
+ records pile up in the daemon's memory. Hence a TTL, rather than relying on the
166
+ caller to collect them.
167
+ """
168
+ now = now if now is not None else time.time()
169
+ stale = [
170
+ name for name, p in self.pending.items()
171
+ if p.phase == "failed" and (now - p.started_at) > _FAILED_PENDING_TTL_S
172
+ ]
173
+ for name in stale:
174
+ self.pending.pop(name, None)
175
+ return len(stale)
176
+
177
+ @property
178
+ def runtime(self):
179
+ if self._runtime is None:
180
+ self._runtime = detect_runtime(
181
+ getattr(self.settings, "local_runtime", "auto"),
182
+ state_dir=self.settings.storage.state_root / "local",
183
+ timeout=float(getattr(self.settings, "local_cli_timeout", 60.0)),
184
+ )
185
+ return self._runtime
186
+
187
+ @property
188
+ def allocator(self) -> GpuAllocator:
189
+ if self._allocator is None:
190
+ self._allocator = GpuAllocator(self.runtime, self.settings)
191
+ return self._allocator
192
+
193
+
194
+ def _node_series(settings) -> str:
195
+ """What this node was configured as, reported as-is.
196
+
197
+ Not mapped to a hardware series here. The series registry belongs to the
198
+ console -- it is editable in its admin page -- so a node that mapped would be
199
+ a second answer to a question one place already owns, and the two would drift
200
+ the first time an administrator added a card. The node reports facts (the
201
+ configured profile, and the card name the driver gives); the console
202
+ interprets them.
203
+ """
204
+ return str(getattr(settings, "cluster_profile", "") or "").strip().lower()
205
+
206
+
207
+ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
208
+ token = str(getattr(settings, "node_token", "") or "").strip()
209
+ if not token:
210
+ # Fail closed: with no token, the node's ability to launch containers is open to
211
+ # anything on the private network.
212
+ raise ValueError("the node daemon requires TUNEPLANE_NODE_TOKEN to be set explicitly")
213
+
214
+ state = NodeState(settings, runtime=runtime, allocator=allocator)
215
+ app = FastAPI(title="tuneplane-node", docs_url=None, redoc_url=None, openapi_url=None)
216
+ app.state.agent = state
217
+
218
+ expected_auth = f"Bearer {token}"
219
+
220
+ async def _auth(request: Request) -> None:
221
+ header = request.headers.get("authorization", "")
222
+ # Constant-time comparison: `!=` returns at the first differing byte and the
223
+ # token has a fixed length, which makes probing it byte by byte practical from
224
+ # inside the private network.
225
+ if not secrets.compare_digest(header, expected_auth):
226
+ raise HTTPException(401, "invalid node token")
227
+ wire = request.headers.get("x-tuneplane-node-protocol", "")
228
+ # The protocol header is required. It used to read `if wire and ...`,
229
+ # which let a request without one straight through -- so "both sides
230
+ # check on every request" actually meant "checked when sent", and a
231
+ # console that sent none passed silently and was handled by whatever
232
+ # each side happened to assume. That is the exact case fail-closed
233
+ # exists to prevent.
234
+ if not wire:
235
+ raise HTTPException(
236
+ 400,
237
+ f"missing X-TunePlane-Node-Protocol header (this node speaks wire v{PROTOCOL}); "
238
+ "upgrade the console, or check that no proxy strips the header",
239
+ )
240
+ if wire != str(PROTOCOL):
241
+ raise HTTPException(409, f"wire protocol mismatch: console={wire}, node={PROTOCOL}")
242
+
243
+ auth = Depends(_auth)
244
+
245
+ # ── Launch ──────────────────────────────────────────────────────────────
246
+
247
+ async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
248
+ async def _run(gpus: list[int]) -> str:
249
+ # The node writes the pick back itself: GPU passthrough follows the node's
250
+ # own setting (a simulated node turns it off) while the allocation label is
251
+ # written either way -- the same semantics as the local backend's
252
+ # simulation mode.
253
+ passthrough = bool(getattr(settings, "local_gpu_passthrough", True))
254
+ spec.gpus = gpus if passthrough else []
255
+ spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
256
+ return await state.runtime.run(spec)
257
+
258
+ gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
259
+ return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
260
+
261
+ async def _pull_then_start(name: str, spec, run_id: str, requested_gpus: int) -> None:
262
+ # `.get`, not `[...]`: a stop landing before this task's first slice has
263
+ # already popped the record, and a KeyError here would be an unhandled
264
+ # exception in a background task rather than the no-op it should be.
265
+ pending = state.pending.get(name)
266
+ if pending is None or state.is_cancelled(name):
267
+ return
268
+ try:
269
+ timeout = float(getattr(settings, "local_image_pull_timeout_s", 3600.0))
270
+ await state.runtime.pull(spec.image, timeout=timeout)
271
+ async with state.control_lock(name):
272
+ # The pull is documented as taking tens of minutes. Anything can have
273
+ # happened in that window, and the thing that usually did is a stop.
274
+ if state.is_cancelled(name) or state.pending.get(name) is not pending:
275
+ return
276
+ pending.phase = "starting"
277
+ await _start_container(spec, run_id, requested_gpus)
278
+ if state.is_cancelled(name):
279
+ # The cancel landed *during* create+start. The container exists
280
+ # and holds real cards, and the caller was told the stop
281
+ # succeeded, so tearing it down here is the only thing that can.
282
+ await state.runtime.stop(name, timeout=5)
283
+ await state.runtime.remove(name)
284
+ return
285
+ state.pending.pop(name, None)
286
+ except asyncio.CancelledError:
287
+ # A cancel while suspended. The container was never created, so
288
+ # there is nothing to tear down -- but the record must not be left
289
+ # behind reserving cards in `reserved_gpus()`.
290
+ state.pending.pop(name, None)
291
+ raise
292
+ except NoCapacity as exc:
293
+ # A concurrent job took the cards during the pull: fail explicitly and say
294
+ # what to do next rather than queueing on the node -- queueing belongs to
295
+ # the console's scheduler, and a second queue must not grow here.
296
+ pending.phase = "failed"
297
+ pending.error = f"the GPUs were taken during the image pull ({exc}); submit again"
298
+ except Exception as exc: # noqa: BLE001
299
+ pending.phase = "failed"
300
+ pending.error = f"image pull or launch failed: {exc}"
301
+ log.exception("asynchronous launch of %s failed", name)
302
+
303
+ @app.post("/node/v1/launch", dependencies=[auth])
304
+ async def launch(body: dict):
305
+ run_id = str(body.get("run_id") or "").strip()
306
+ requested_gpus = int(body.get("requested_gpus") or 0)
307
+ if not run_id or "container" not in body:
308
+ raise HTTPException(422, "launch requires run_id and container")
309
+ spec = spec_from_dict(body["container"])
310
+
311
+ async with state.control_lock(spec.name):
312
+ # Repeated launches observe the container created by their predecessor.
313
+ existing = await state.runtime.inspect(spec.name)
314
+ if existing.exists and existing.status in ("running", "created", "paused"):
315
+ return {"phase": "running", "idempotent": True, "gpus": existing.gpus}
316
+ if existing.exists:
317
+ await state.runtime.remove(spec.name)
318
+ if (pending := state.pending.get(spec.name)) and pending.phase != "failed":
319
+ return {"phase": pending.phase, "idempotent": True, "image": pending.image}
320
+
321
+ state.begin_launch(spec.name)
322
+ if not await state.runtime.image_exists(spec.image):
323
+ # Reserve capacity while the image is pulled in the background.
324
+ state.pending[spec.name] = PendingLaunch(
325
+ run_id=run_id, image=spec.image, requested_gpus=requested_gpus,
326
+ )
327
+ state.spawn(
328
+ _pull_then_start(spec.name, spec, run_id, requested_gpus), ref=spec.name,
329
+ )
330
+ return {"phase": "pulling", "image": spec.image,
331
+ "requested_gpus": requested_gpus}
332
+ try:
333
+ return await _start_container(spec, run_id, requested_gpus)
334
+ except NoCapacity as exc:
335
+ raise HTTPException(409, str(exc)) from exc
336
+
337
+ # ── Observe ─────────────────────────────────────────────────────────────
338
+
339
+ @app.post("/node/v1/observe", dependencies=[auth])
340
+ async def observe(body: dict):
341
+ refs = [str(x) for x in (body.get("refs") or [])]
342
+ states = {st.name: st for st in await state.runtime.ps()}
343
+ out: dict[str, dict] = {}
344
+ for ref in refs:
345
+ if st := states.get(ref):
346
+ out[ref] = state_to_dict(st)
347
+ continue
348
+ if pending := state.pending.get(ref):
349
+ if pending.phase == "failed":
350
+ out[ref] = {
351
+ **state_to_dict(ContainerState(name=ref, exists=True, status="dead")),
352
+ "pending_error": pending.error,
353
+ }
354
+ else:
355
+ # Pulling or starting: presented to the ledger as created (-> PENDING),
356
+ # never as gone.
357
+ out[ref] = {
358
+ **state_to_dict(ContainerState(name=ref, exists=True, status="created")),
359
+ "pending_phase": pending.phase,
360
+ "pending_image": pending.image,
361
+ # How long it has been waiting. The control plane uses it to say
362
+ # "still pulling, 12 minutes in" instead of leaving somebody
363
+ # staring at a motionless PENDING, guessing whether it is stuck.
364
+ "pending_elapsed_s": round(time.time() - pending.started_at, 1),
365
+ "pending_gpus": pending.requested_gpus,
366
+ }
367
+ continue
368
+ out[ref] = state_to_dict(ContainerState(name=ref, exists=False))
369
+ return {"states": out}
370
+
371
+ # ── Control and logs ────────────────────────────────────────────────────
372
+
373
+ @app.post("/node/v1/stop", dependencies=[auth])
374
+ async def stop(body: dict):
375
+ ref = str(body.get("ref") or "")
376
+ # Cancelling the pending record is not enough: the background pull task
377
+ # has to be stopped too. Popping the record alone left it running, so it
378
+ # finished the pull, allocated the GPUs and started the container --
379
+ # after the console had already released the quota for a run it was told
380
+ # had stopped, which is how those cards ended up held by nothing.
381
+ async with state.control_lock(ref):
382
+ state.cancel_launch(ref)
383
+ ok = await state.runtime.stop(
384
+ ref, timeout=int(body.get("timeout") or getattr(settings, "local_stop_timeout", 30))
385
+ )
386
+ return {"ok": ok}
387
+
388
+ @app.post("/node/v1/remove", dependencies=[auth])
389
+ async def remove(body: dict):
390
+ ref = str(body.get("ref") or "")
391
+ async with state.control_lock(ref):
392
+ state.cancel_launch(ref)
393
+ return {"ok": await state.runtime.remove(ref)}
394
+
395
+ @app.get("/node/v1/logs", dependencies=[auth])
396
+ async def logs(ref: str, tail: int = 2000):
397
+ return {"logs": await state.runtime.logs(ref, tail_lines=tail)}
398
+
399
+ @app.post("/node/v1/reap", dependencies=[auth])
400
+ async def reap(body: dict):
401
+ ttl = int(body.get("max_age_s") or 0) or int(
402
+ getattr(settings, "local_container_ttl_s", 600)
403
+ )
404
+ now = time.time()
405
+ removed = 0
406
+ for st in await state.runtime.ps():
407
+ if st.status in ("running", "paused"):
408
+ continue
409
+ if st.status == "created":
410
+ # Wreckage from a successful create whose start failed. It never
411
+ # reaches `exited` on its own, yet it carries LABEL_GPUS and so
412
+ # `occupancy()` counts it as holding its cards -- leaving it is a
413
+ # permanent GPU leak. `LocalExecutor.reap` has handled this since
414
+ # it was found there; the node backend skipped every `created`
415
+ # container unconditionally.
416
+ #
417
+ # Only one that never started and is past its TTL: a healthy
418
+ # launch is `created` for milliseconds, because
419
+ # `allocate_and_run` holds the lock across create and start.
420
+ if not st.never_started:
421
+ continue
422
+ if ttl > 0 and _age_seconds(st.created_at, now) < ttl:
423
+ continue
424
+ elif ttl > 0 and _age_seconds(st.finished_at, now) < ttl:
425
+ continue
426
+ if await state.runtime.remove(st.name):
427
+ removed += 1
428
+ return {"removed": removed}
429
+
430
+ # ── Health and capacity ─────────────────────────────────────────────────
431
+
432
+ @app.get("/node/v1/health", dependencies=[auth])
433
+ async def health():
434
+ base = await state.runtime.health()
435
+ if not base.get("ok"):
436
+ return {"ok": False, "runtime": base, "protocol": PROTOCOL}
437
+ try:
438
+ occ = await state.allocator.occupancy()
439
+ except Exception as exc: # noqa: BLE001
440
+ return {"ok": False, "runtime": base, "protocol": PROTOCOL,
441
+ "detail": f"GPU probe failed: {exc}"}
442
+ # The node's view of the shared storage root. The console reports its
443
+ # own; both should be the same filesystem, and a node that disagrees is
444
+ # exactly the misconfiguration worth surfacing here.
445
+ disk = None
446
+ root = settings.storage.root
447
+ if settings.storage_configured and root.exists():
448
+ try:
449
+ du = shutil.disk_usage(root)
450
+ disk = {"path": str(root), "used_pct": round(du.used / du.total * 100.0, 1),
451
+ "free_gib": round(du.free / 1024**3, 1)}
452
+ except OSError:
453
+ disk = None
454
+ # ★ free = physically free minus the cards promised to jobs still pulling. The
455
+ # control plane's _place reads nothing else, so without the subtraction it
456
+ # promises the same cards twice and the job waits tens of minutes to find out
457
+ # it cannot have them.
458
+ reserved = state.reserved_gpus()
459
+ free_gpus = occ.free[reserved:] if reserved else occ.free
460
+ return {
461
+ "ok": True,
462
+ "protocol": PROTOCOL,
463
+ "runtime": base,
464
+ # This node's series. Where nodes hold different cards, it is what lets the
465
+ # control plane place a job on the right ones and count its capacity in the
466
+ # right bucket -- without it every node is taken to hold whatever the
467
+ # console's own cluster_profile says, an H100 node's cards are reported as
468
+ # H200, and every control that works per series stops working.
469
+ "series": _node_series(settings),
470
+ "gpus_total": occ.total,
471
+ "gpus_free": max(0, len(occ.free) - reserved),
472
+ "gpus_reserved": reserved,
473
+ "free_gpus": free_gpus,
474
+ "external": sorted(occ.external),
475
+ # When detection is broken `external` is always empty, so the control plane
476
+ # has to be able to tell "nobody holds a card on this node" from "this node
477
+ # cannot see whether anybody does" -- otherwise an inflated free count is
478
+ # undiscoverable.
479
+ "external_probe_ok": occ.external_probe_ok,
480
+ "unhealthy": {str(k): v for k, v in occ.unhealthy.items()},
481
+ "by_job": {str(k): v for k, v in occ.by_job.items()},
482
+ "gpu_detail": occ.explain(),
483
+ "disk": disk,
484
+ "pending_pulls": {
485
+ name: {"image": p.image, "phase": p.phase, "error": p.error}
486
+ for name, p in state.pending.items()
487
+ },
488
+ }
489
+
490
+ return app
491
+
492
+
493
+ def make_app() -> FastAPI:
494
+ """The uvicorn factory entry point, used by `tuneplane-node serve`."""
495
+ from tuneplane_node.settings import NodeSettings
496
+
497
+ return create_node_app(NodeSettings())
@@ -0,0 +1,163 @@
1
+ """What this machine has, collected in one place.
2
+
3
+ **This is the tier-3 seam.** Everything the platform does with an inventory --
4
+ refusing a shared-mount fleet, matching a series, deciding a node can hold a job
5
+ -- works from the dataclass below, and only `collect()` touches hardware. A
6
+ runner with no GPU therefore exercises registration, gating, drain and the whole
7
+ console surface; one function stays unproven, and its size is known (CLAUDE.md,
8
+ *Where a check runs*).
9
+
10
+ It answers only what the console cannot see for itself and cannot be told
11
+ truthfully by configuration: the driver, the container runtime, the cards, and
12
+ whether the shared storage root is actually here.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import asyncio
17
+ import os
18
+ import shutil
19
+ from dataclasses import asdict, dataclass
20
+ from typing import Any
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class NodeInventory:
25
+ #: What the platform calls these cards, e.g. "h100". Empty when the driver
26
+ #: reports a card the series registry has never heard of.
27
+ accelerator_series: str = ""
28
+ #: What the driver calls them, e.g. "NVIDIA H100 80GB HBM3". Kept alongside
29
+ #: the series because an unmapped card is exactly the case where an operator
30
+ #: needs to see the real name to add it to the registry.
31
+ accelerator_name: str = ""
32
+ accelerator_count: int = 0
33
+ driver: str = ""
34
+ runtime: str = ""
35
+ #: Whether the Storage Root is present at the same absolute path the console
36
+ #: uses. A shared-mount Fleet refuses a node that answers False, because the
37
+ #: alternative is accepting it and failing at the first launch with an
38
+ #: unrelated-looking I/O error.
39
+ storage_root_visible: bool = False
40
+ disk_free_bytes: int = 0
41
+ cpu_count: int = 0
42
+ memory_bytes: int = 0
43
+
44
+ def to_dict(self) -> dict[str, Any]:
45
+ return asdict(self)
46
+
47
+
48
+ async def probe_gpus() -> tuple[str, str, int]:
49
+ """(card name, driver version, card count), straight from the driver.
50
+
51
+ The one place in the platform that shells out to `nvidia-smi`, and the
52
+ reason this module is the tier-3 seam.
53
+
54
+ **Asking the driver rather than reading configuration is the point.** The
55
+ node-side series used to come from this machine's own `TUNEPLANE_CLUSTER_PROFILE`,
56
+ which is a value an operator sets and can forget: a mixed-GPU fleet whose
57
+ nodes were never configured reports no series at all, and every placement
58
+ rule that filters by series silently stops filtering. The machine knows what
59
+ cards it has; nothing else reliably does.
60
+
61
+ A machine with no `nvidia-smi` answers ("", "", 0), which the console reads
62
+ as unknown. That is a fact, and guessing from the container image would not
63
+ be.
64
+ """
65
+ try:
66
+ proc = await asyncio.create_subprocess_exec(
67
+ "nvidia-smi", "--query-gpu=name,driver_version", "--format=csv,noheader",
68
+ stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.DEVNULL,
69
+ )
70
+ out, _ = await asyncio.wait_for(proc.communicate(), timeout=10)
71
+ except (FileNotFoundError, OSError, asyncio.TimeoutError):
72
+ return "", "", 0
73
+ if proc.returncode != 0:
74
+ return "", "", 0
75
+ # One line per card. They are homogeneous on any machine the platform can
76
+ # place a pool on, so the first line names them all; the line count is the
77
+ # card count, which is also what the allocator sees.
78
+ lines = [ln.strip() for ln in (out or b"").decode(errors="replace").splitlines() if ln.strip()]
79
+ if not lines:
80
+ return "", "", 0
81
+ name, _, driver = lines[0].partition(",")
82
+ return name.strip(), driver.strip(), len(lines)
83
+
84
+
85
+ def _memory_bytes() -> int:
86
+ try:
87
+ return int(os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES"))
88
+ except (ValueError, OSError, AttributeError):
89
+ # Not every platform exposes these, and a missing figure is not worth
90
+ # failing a registration over.
91
+ return 0
92
+
93
+
94
+ def _series_of(gpu_name: str, settings) -> str:
95
+ """What this node was configured as, reported as-is.
96
+
97
+ **Not mapped to a hardware series here.** The registry that knows which card
98
+ names belong to which series is the console's, editable in its admin page; a
99
+ node that mapped would be a second answer to a question one place already
100
+ owns, and the two would drift the first time an administrator added a card.
101
+
102
+ So the node reports two facts -- the profile an operator declared for this
103
+ machine, and the name the driver gives the cards -- and the console decides
104
+ what they mean. Both may be empty, and empty is honest: placement treats an
105
+ unknown series as "do not exclude this node" rather than pretending to know.
106
+ """
107
+ from tuneplane_node.daemon import _node_series
108
+
109
+ return _node_series(settings)
110
+
111
+
112
+ async def collect(settings, *, runtime=None, allocator=None) -> NodeInventory:
113
+ """Ask this machine what it is. Never raises: a partial answer beats none.
114
+
115
+ A field the machine cannot answer stays at its zero value, which the console
116
+ reads as "unknown" rather than "none" -- the two matter for capacity but not
117
+ for whether the node may join, and inventing a number here would be worse
118
+ than an empty one.
119
+ """
120
+ from tuneplane_node.allocator import GpuAllocator
121
+ from tuneplane_node.runtime import detect_runtime
122
+
123
+ runtime = runtime or detect_runtime(settings)
124
+ allocator = allocator or GpuAllocator(settings, runtime=runtime)
125
+
126
+ runtime_name = ""
127
+ driver = ""
128
+ try:
129
+ health = await runtime.health()
130
+ runtime_name = f"{getattr(runtime, 'name', '')}://{health.get('version') or ''}".strip("/:")
131
+ except Exception: # noqa: BLE001
132
+ runtime_name = str(getattr(runtime, "name", "") or "")
133
+
134
+ gpu_name, driver, probed = await probe_gpus()
135
+ count = probed
136
+ try:
137
+ # The allocator's figure wins when it has one: it is what actually
138
+ # decides placement on this node, and a disagreement between it and the
139
+ # driver is the allocator's to explain.
140
+ count = int((await allocator.occupancy()).total) or probed
141
+ except Exception: # noqa: BLE001
142
+ count = probed
143
+
144
+ root = settings.storage.root
145
+ visible = bool(getattr(settings, "storage_configured", False)) and root.exists()
146
+ free = 0
147
+ if visible:
148
+ try:
149
+ free = int(shutil.disk_usage(root).free)
150
+ except OSError:
151
+ visible = False
152
+
153
+ return NodeInventory(
154
+ accelerator_series=_series_of(gpu_name, settings),
155
+ accelerator_name=gpu_name,
156
+ accelerator_count=count,
157
+ driver=driver,
158
+ runtime=runtime_name,
159
+ storage_root_visible=visible,
160
+ disk_free_bytes=free,
161
+ cpu_count=os.cpu_count() or 0,
162
+ memory_bytes=_memory_bytes(),
163
+ )