tuneplane-node 0.3.15__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tuneplane_node/__init__.py +1 -0
- tuneplane_node/allocator.py +596 -0
- tuneplane_node/cli.py +189 -0
- tuneplane_node/daemon.py +497 -0
- tuneplane_node/inventory.py +163 -0
- tuneplane_node/join.py +199 -0
- tuneplane_node/runtime.py +611 -0
- tuneplane_node/settings.py +77 -0
- tuneplane_node/wire.py +85 -0
- tuneplane_node-0.3.15.dist-info/METADATA +12 -0
- tuneplane_node-0.3.15.dist-info/RECORD +13 -0
- tuneplane_node-0.3.15.dist-info/WHEEL +4 -0
- tuneplane_node-0.3.15.dist-info/entry_points.txt +2 -0
tuneplane_node/daemon.py
ADDED
|
@@ -0,0 +1,497 @@
|
|
|
1
|
+
"""The node daemon: a narrow job API on a GPU node.
|
|
2
|
+
|
|
3
|
+
What it is
|
|
4
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
5
|
+
The node-side component of the node backend (kubelet is the analogue under K8s): the
|
|
6
|
+
console calls it over HTTP to launch, observe and stop job containers. **The daemon is
|
|
7
|
+
control-plane infrastructure and runs no training** -- a job is still a per-job
|
|
8
|
+
container, with exactly the local backend's semantics (one throwaway container per
|
|
9
|
+
job, destroyed when it finishes).
|
|
10
|
+
|
|
11
|
+
Reuse rather than rewrite
|
|
12
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
13
|
+
Every core mechanism inside the node is the local backend's implementation:
|
|
14
|
+
container runtime node/runtime.py (the docker / podman CLI wrapper)
|
|
15
|
+
GPU allocation node/allocator.py (allocation lock, labels as the truth,
|
|
16
|
+
health probing)
|
|
17
|
+
image prefetch the same explicit pull on its own long timeout as LocalExecutor
|
|
18
|
+
|
|
19
|
+
So restarting a node loses no state at all: the truth about occupancy is in the
|
|
20
|
+
container labels and one `docker ps` rebuilds it -- the property the local backend's
|
|
21
|
+
design bought in the first place.
|
|
22
|
+
|
|
23
|
+
Security model
|
|
24
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
25
|
+
Only the job verbs are exposed (launch/observe/stop/logs/reap), never arbitrary
|
|
26
|
+
container operations; a bearer token is mandatory (without one the daemon refuses to
|
|
27
|
+
start rather than running open). Against the alternative -- the console holding each
|
|
28
|
+
node's docker socket, which is equivalent to root -- this narrows the privilege
|
|
29
|
+
surface.
|
|
30
|
+
|
|
31
|
+
Asynchronous semantics of an image pull
|
|
32
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
33
|
+
A first pull of tens of GB takes tens of minutes and cannot hang off a single HTTP
|
|
34
|
+
request. When the image is not on the machine, launch returns phase=pulling at once
|
|
35
|
+
and a background task finishes the pull, then allocates GPUs and starts the container;
|
|
36
|
+
observe on the console side reports PENDING with the pull explained meanwhile. A
|
|
37
|
+
failed pull becomes an explicit failed state carrying the reason, rather than
|
|
38
|
+
disappearing quietly. If the daemon restarts mid-pull the pending record is gone and
|
|
39
|
+
no container exists, so reconciliation converges on STOPPED -- which the docs state,
|
|
40
|
+
and resubmitting is the answer.
|
|
41
|
+
"""
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import asyncio
|
|
45
|
+
import logging
|
|
46
|
+
import secrets
|
|
47
|
+
import shutil
|
|
48
|
+
import time
|
|
49
|
+
from dataclasses import dataclass, field
|
|
50
|
+
from weakref import WeakValueDictionary
|
|
51
|
+
|
|
52
|
+
from fastapi import Depends, FastAPI, HTTPException, Request
|
|
53
|
+
|
|
54
|
+
from tuneplane_node.allocator import GpuAllocator, NoCapacity
|
|
55
|
+
from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
|
|
56
|
+
from tuneplane_node.runtime import age_seconds as _age_seconds
|
|
57
|
+
from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
|
|
58
|
+
|
|
59
|
+
log = logging.getLogger(__name__)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
#: How long a failed pending record is kept (seconds). It has to outlive a round of
|
|
63
|
+
#: reconciliation -- the console must see the failure reason before it can record it in
|
|
64
|
+
#: the ledger -- and after that it is nothing but memory held for no reason.
|
|
65
|
+
_FAILED_PENDING_TTL_S = 1800.0
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class PendingLaunch:
|
|
70
|
+
"""The explicit state of one asynchronous launch, while its image is pulling."""
|
|
71
|
+
|
|
72
|
+
run_id: str
|
|
73
|
+
image: str
|
|
74
|
+
#: ★ How many cards this job has been promised. It has to be recorded: the pull
|
|
75
|
+
#: takes tens of minutes, no container exists during it, and the truth about
|
|
76
|
+
#: occupancy is the container labels -- so occupancy() cannot see the job at all.
|
|
77
|
+
#: Without this field health reports the promised cards as free, the control
|
|
78
|
+
#: plane's _place puts later jobs on the same node, and by the time the image is
|
|
79
|
+
#: pulled the cards are long gone and the job ends in "please resubmit". Queueing
|
|
80
|
+
#: for tens of minutes and then running nothing is the worst kind of failure.
|
|
81
|
+
requested_gpus: int = 0
|
|
82
|
+
phase: str = "pulling" # pulling | starting | failed
|
|
83
|
+
error: str = ""
|
|
84
|
+
started_at: float = field(default_factory=time.time)
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def reserves_gpus(self) -> bool:
|
|
88
|
+
"""Still waiting for cards; a failed launch holds none."""
|
|
89
|
+
return self.phase != "failed"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class NodeState:
|
|
93
|
+
"""The daemon's in-process state. runtime and allocator are injectable, for test doubles."""
|
|
94
|
+
|
|
95
|
+
def __init__(self, settings, *, runtime=None, allocator=None):
|
|
96
|
+
self.settings = settings
|
|
97
|
+
self._runtime = runtime
|
|
98
|
+
self._allocator = allocator
|
|
99
|
+
self.pending: dict[str, PendingLaunch] = {}
|
|
100
|
+
#: Strong references to the background pull tasks. asyncio holds only a weak
|
|
101
|
+
#: one, so a create_task whose result is dropped can be garbage-collected
|
|
102
|
+
#: mid-pull -- and the job then sits in pulling forever, with no exception and
|
|
103
|
+
#: nothing in the log.
|
|
104
|
+
self._tasks: set = set()
|
|
105
|
+
#: Background launches, by container ref, so a stop can find the one it
|
|
106
|
+
#: is stopping. Without this a stop during a pull popped the pending
|
|
107
|
+
#: record and returned ok, while the task went on to allocate GPUs and
|
|
108
|
+
#: start the container minutes later -- the console had already released
|
|
109
|
+
#: the quota, so nothing ever reclaimed those cards.
|
|
110
|
+
self._launches: dict[str, "asyncio.Task"] = {}
|
|
111
|
+
#: Refs whose launch was cancelled. Checked at both await boundaries of
|
|
112
|
+
#: `_pull_then_start`, because cancelling a task only helps if it is
|
|
113
|
+
#: suspended: a cancel that lands while the container is being created
|
|
114
|
+
#: has to be answered by tearing that container down instead.
|
|
115
|
+
self.cancelled: set[str] = set()
|
|
116
|
+
self._control_locks: WeakValueDictionary[str, asyncio.Lock] = WeakValueDictionary()
|
|
117
|
+
|
|
118
|
+
def control_lock(self, ref: str) -> asyncio.Lock:
|
|
119
|
+
"""Serialize create/start with stop/remove for one container.
|
|
120
|
+
|
|
121
|
+
Holders and waiters keep the lock alive; completed container identities
|
|
122
|
+
do not accumulate in a daemon that may run for months.
|
|
123
|
+
"""
|
|
124
|
+
lock = self._control_locks.get(ref)
|
|
125
|
+
if lock is None:
|
|
126
|
+
lock = asyncio.Lock()
|
|
127
|
+
self._control_locks[ref] = lock
|
|
128
|
+
return lock
|
|
129
|
+
|
|
130
|
+
def spawn(self, coro, *, ref: str = "") -> None:
|
|
131
|
+
"""Start a background task and hold a reference to it until it finishes."""
|
|
132
|
+
task = asyncio.get_running_loop().create_task(coro)
|
|
133
|
+
self._tasks.add(task)
|
|
134
|
+
task.add_done_callback(self._tasks.discard)
|
|
135
|
+
if ref:
|
|
136
|
+
self._launches[ref] = task
|
|
137
|
+
task.add_done_callback(lambda _t, r=ref: self._launches.pop(r, None))
|
|
138
|
+
|
|
139
|
+
def begin_launch(self, ref: str) -> None:
|
|
140
|
+
"""A fresh launch for `ref` clears any tombstone an earlier one left."""
|
|
141
|
+
self.cancelled.discard(ref)
|
|
142
|
+
|
|
143
|
+
def cancel_launch(self, ref: str) -> None:
|
|
144
|
+
"""Stop a launch in flight, whatever stage it has reached."""
|
|
145
|
+
self.cancelled.add(ref)
|
|
146
|
+
task = self._launches.pop(ref, None)
|
|
147
|
+
if task is not None and not task.done():
|
|
148
|
+
task.cancel()
|
|
149
|
+
self.pending.pop(ref, None)
|
|
150
|
+
|
|
151
|
+
def is_cancelled(self, ref: str) -> bool:
|
|
152
|
+
return ref in self.cancelled
|
|
153
|
+
|
|
154
|
+
def reserved_gpus(self) -> int:
|
|
155
|
+
"""Cards promised to pending jobs whose container has not started yet."""
|
|
156
|
+
self.expire_failed_pending()
|
|
157
|
+
return sum(p.requested_gpus for p in self.pending.values() if p.reserves_gpus)
|
|
158
|
+
|
|
159
|
+
def expire_failed_pending(self, now: float | None = None) -> int:
|
|
160
|
+
"""Drop failed records that have expired.
|
|
161
|
+
|
|
162
|
+
On the normal path the console reconciles to a terminal state and then calls
|
|
163
|
+
remove/stop, which takes the record with it. But a FAILED training job stays in
|
|
164
|
+
the ledger so it can be retried automatically, remove never comes, and the
|
|
165
|
+
records pile up in the daemon's memory. Hence a TTL, rather than relying on the
|
|
166
|
+
caller to collect them.
|
|
167
|
+
"""
|
|
168
|
+
now = now if now is not None else time.time()
|
|
169
|
+
stale = [
|
|
170
|
+
name for name, p in self.pending.items()
|
|
171
|
+
if p.phase == "failed" and (now - p.started_at) > _FAILED_PENDING_TTL_S
|
|
172
|
+
]
|
|
173
|
+
for name in stale:
|
|
174
|
+
self.pending.pop(name, None)
|
|
175
|
+
return len(stale)
|
|
176
|
+
|
|
177
|
+
@property
|
|
178
|
+
def runtime(self):
|
|
179
|
+
if self._runtime is None:
|
|
180
|
+
self._runtime = detect_runtime(
|
|
181
|
+
getattr(self.settings, "local_runtime", "auto"),
|
|
182
|
+
state_dir=self.settings.storage.state_root / "local",
|
|
183
|
+
timeout=float(getattr(self.settings, "local_cli_timeout", 60.0)),
|
|
184
|
+
)
|
|
185
|
+
return self._runtime
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def allocator(self) -> GpuAllocator:
|
|
189
|
+
if self._allocator is None:
|
|
190
|
+
self._allocator = GpuAllocator(self.runtime, self.settings)
|
|
191
|
+
return self._allocator
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _node_series(settings) -> str:
|
|
195
|
+
"""What this node was configured as, reported as-is.
|
|
196
|
+
|
|
197
|
+
Not mapped to a hardware series here. The series registry belongs to the
|
|
198
|
+
console -- it is editable in its admin page -- so a node that mapped would be
|
|
199
|
+
a second answer to a question one place already owns, and the two would drift
|
|
200
|
+
the first time an administrator added a card. The node reports facts (the
|
|
201
|
+
configured profile, and the card name the driver gives); the console
|
|
202
|
+
interprets them.
|
|
203
|
+
"""
|
|
204
|
+
return str(getattr(settings, "cluster_profile", "") or "").strip().lower()
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
208
|
+
token = str(getattr(settings, "node_token", "") or "").strip()
|
|
209
|
+
if not token:
|
|
210
|
+
# Fail closed: with no token, the node's ability to launch containers is open to
|
|
211
|
+
# anything on the private network.
|
|
212
|
+
raise ValueError("the node daemon requires TUNEPLANE_NODE_TOKEN to be set explicitly")
|
|
213
|
+
|
|
214
|
+
state = NodeState(settings, runtime=runtime, allocator=allocator)
|
|
215
|
+
app = FastAPI(title="tuneplane-node", docs_url=None, redoc_url=None, openapi_url=None)
|
|
216
|
+
app.state.agent = state
|
|
217
|
+
|
|
218
|
+
expected_auth = f"Bearer {token}"
|
|
219
|
+
|
|
220
|
+
async def _auth(request: Request) -> None:
|
|
221
|
+
header = request.headers.get("authorization", "")
|
|
222
|
+
# Constant-time comparison: `!=` returns at the first differing byte and the
|
|
223
|
+
# token has a fixed length, which makes probing it byte by byte practical from
|
|
224
|
+
# inside the private network.
|
|
225
|
+
if not secrets.compare_digest(header, expected_auth):
|
|
226
|
+
raise HTTPException(401, "invalid node token")
|
|
227
|
+
wire = request.headers.get("x-tuneplane-node-protocol", "")
|
|
228
|
+
# The protocol header is required. It used to read `if wire and ...`,
|
|
229
|
+
# which let a request without one straight through -- so "both sides
|
|
230
|
+
# check on every request" actually meant "checked when sent", and a
|
|
231
|
+
# console that sent none passed silently and was handled by whatever
|
|
232
|
+
# each side happened to assume. That is the exact case fail-closed
|
|
233
|
+
# exists to prevent.
|
|
234
|
+
if not wire:
|
|
235
|
+
raise HTTPException(
|
|
236
|
+
400,
|
|
237
|
+
f"missing X-TunePlane-Node-Protocol header (this node speaks wire v{PROTOCOL}); "
|
|
238
|
+
"upgrade the console, or check that no proxy strips the header",
|
|
239
|
+
)
|
|
240
|
+
if wire != str(PROTOCOL):
|
|
241
|
+
raise HTTPException(409, f"wire protocol mismatch: console={wire}, node={PROTOCOL}")
|
|
242
|
+
|
|
243
|
+
auth = Depends(_auth)
|
|
244
|
+
|
|
245
|
+
# ── Launch ──────────────────────────────────────────────────────────────
|
|
246
|
+
|
|
247
|
+
async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
|
|
248
|
+
async def _run(gpus: list[int]) -> str:
|
|
249
|
+
# The node writes the pick back itself: GPU passthrough follows the node's
|
|
250
|
+
# own setting (a simulated node turns it off) while the allocation label is
|
|
251
|
+
# written either way -- the same semantics as the local backend's
|
|
252
|
+
# simulation mode.
|
|
253
|
+
passthrough = bool(getattr(settings, "local_gpu_passthrough", True))
|
|
254
|
+
spec.gpus = gpus if passthrough else []
|
|
255
|
+
spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
|
|
256
|
+
return await state.runtime.run(spec)
|
|
257
|
+
|
|
258
|
+
gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
|
|
259
|
+
return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
|
|
260
|
+
|
|
261
|
+
async def _pull_then_start(name: str, spec, run_id: str, requested_gpus: int) -> None:
|
|
262
|
+
# `.get`, not `[...]`: a stop landing before this task's first slice has
|
|
263
|
+
# already popped the record, and a KeyError here would be an unhandled
|
|
264
|
+
# exception in a background task rather than the no-op it should be.
|
|
265
|
+
pending = state.pending.get(name)
|
|
266
|
+
if pending is None or state.is_cancelled(name):
|
|
267
|
+
return
|
|
268
|
+
try:
|
|
269
|
+
timeout = float(getattr(settings, "local_image_pull_timeout_s", 3600.0))
|
|
270
|
+
await state.runtime.pull(spec.image, timeout=timeout)
|
|
271
|
+
async with state.control_lock(name):
|
|
272
|
+
# The pull is documented as taking tens of minutes. Anything can have
|
|
273
|
+
# happened in that window, and the thing that usually did is a stop.
|
|
274
|
+
if state.is_cancelled(name) or state.pending.get(name) is not pending:
|
|
275
|
+
return
|
|
276
|
+
pending.phase = "starting"
|
|
277
|
+
await _start_container(spec, run_id, requested_gpus)
|
|
278
|
+
if state.is_cancelled(name):
|
|
279
|
+
# The cancel landed *during* create+start. The container exists
|
|
280
|
+
# and holds real cards, and the caller was told the stop
|
|
281
|
+
# succeeded, so tearing it down here is the only thing that can.
|
|
282
|
+
await state.runtime.stop(name, timeout=5)
|
|
283
|
+
await state.runtime.remove(name)
|
|
284
|
+
return
|
|
285
|
+
state.pending.pop(name, None)
|
|
286
|
+
except asyncio.CancelledError:
|
|
287
|
+
# A cancel while suspended. The container was never created, so
|
|
288
|
+
# there is nothing to tear down -- but the record must not be left
|
|
289
|
+
# behind reserving cards in `reserved_gpus()`.
|
|
290
|
+
state.pending.pop(name, None)
|
|
291
|
+
raise
|
|
292
|
+
except NoCapacity as exc:
|
|
293
|
+
# A concurrent job took the cards during the pull: fail explicitly and say
|
|
294
|
+
# what to do next rather than queueing on the node -- queueing belongs to
|
|
295
|
+
# the console's scheduler, and a second queue must not grow here.
|
|
296
|
+
pending.phase = "failed"
|
|
297
|
+
pending.error = f"the GPUs were taken during the image pull ({exc}); submit again"
|
|
298
|
+
except Exception as exc: # noqa: BLE001
|
|
299
|
+
pending.phase = "failed"
|
|
300
|
+
pending.error = f"image pull or launch failed: {exc}"
|
|
301
|
+
log.exception("asynchronous launch of %s failed", name)
|
|
302
|
+
|
|
303
|
+
@app.post("/node/v1/launch", dependencies=[auth])
|
|
304
|
+
async def launch(body: dict):
|
|
305
|
+
run_id = str(body.get("run_id") or "").strip()
|
|
306
|
+
requested_gpus = int(body.get("requested_gpus") or 0)
|
|
307
|
+
if not run_id or "container" not in body:
|
|
308
|
+
raise HTTPException(422, "launch requires run_id and container")
|
|
309
|
+
spec = spec_from_dict(body["container"])
|
|
310
|
+
|
|
311
|
+
async with state.control_lock(spec.name):
|
|
312
|
+
# Repeated launches observe the container created by their predecessor.
|
|
313
|
+
existing = await state.runtime.inspect(spec.name)
|
|
314
|
+
if existing.exists and existing.status in ("running", "created", "paused"):
|
|
315
|
+
return {"phase": "running", "idempotent": True, "gpus": existing.gpus}
|
|
316
|
+
if existing.exists:
|
|
317
|
+
await state.runtime.remove(spec.name)
|
|
318
|
+
if (pending := state.pending.get(spec.name)) and pending.phase != "failed":
|
|
319
|
+
return {"phase": pending.phase, "idempotent": True, "image": pending.image}
|
|
320
|
+
|
|
321
|
+
state.begin_launch(spec.name)
|
|
322
|
+
if not await state.runtime.image_exists(spec.image):
|
|
323
|
+
# Reserve capacity while the image is pulled in the background.
|
|
324
|
+
state.pending[spec.name] = PendingLaunch(
|
|
325
|
+
run_id=run_id, image=spec.image, requested_gpus=requested_gpus,
|
|
326
|
+
)
|
|
327
|
+
state.spawn(
|
|
328
|
+
_pull_then_start(spec.name, spec, run_id, requested_gpus), ref=spec.name,
|
|
329
|
+
)
|
|
330
|
+
return {"phase": "pulling", "image": spec.image,
|
|
331
|
+
"requested_gpus": requested_gpus}
|
|
332
|
+
try:
|
|
333
|
+
return await _start_container(spec, run_id, requested_gpus)
|
|
334
|
+
except NoCapacity as exc:
|
|
335
|
+
raise HTTPException(409, str(exc)) from exc
|
|
336
|
+
|
|
337
|
+
# ── Observe ─────────────────────────────────────────────────────────────
|
|
338
|
+
|
|
339
|
+
@app.post("/node/v1/observe", dependencies=[auth])
|
|
340
|
+
async def observe(body: dict):
|
|
341
|
+
refs = [str(x) for x in (body.get("refs") or [])]
|
|
342
|
+
states = {st.name: st for st in await state.runtime.ps()}
|
|
343
|
+
out: dict[str, dict] = {}
|
|
344
|
+
for ref in refs:
|
|
345
|
+
if st := states.get(ref):
|
|
346
|
+
out[ref] = state_to_dict(st)
|
|
347
|
+
continue
|
|
348
|
+
if pending := state.pending.get(ref):
|
|
349
|
+
if pending.phase == "failed":
|
|
350
|
+
out[ref] = {
|
|
351
|
+
**state_to_dict(ContainerState(name=ref, exists=True, status="dead")),
|
|
352
|
+
"pending_error": pending.error,
|
|
353
|
+
}
|
|
354
|
+
else:
|
|
355
|
+
# Pulling or starting: presented to the ledger as created (-> PENDING),
|
|
356
|
+
# never as gone.
|
|
357
|
+
out[ref] = {
|
|
358
|
+
**state_to_dict(ContainerState(name=ref, exists=True, status="created")),
|
|
359
|
+
"pending_phase": pending.phase,
|
|
360
|
+
"pending_image": pending.image,
|
|
361
|
+
# How long it has been waiting. The control plane uses it to say
|
|
362
|
+
# "still pulling, 12 minutes in" instead of leaving somebody
|
|
363
|
+
# staring at a motionless PENDING, guessing whether it is stuck.
|
|
364
|
+
"pending_elapsed_s": round(time.time() - pending.started_at, 1),
|
|
365
|
+
"pending_gpus": pending.requested_gpus,
|
|
366
|
+
}
|
|
367
|
+
continue
|
|
368
|
+
out[ref] = state_to_dict(ContainerState(name=ref, exists=False))
|
|
369
|
+
return {"states": out}
|
|
370
|
+
|
|
371
|
+
# ── Control and logs ────────────────────────────────────────────────────
|
|
372
|
+
|
|
373
|
+
@app.post("/node/v1/stop", dependencies=[auth])
|
|
374
|
+
async def stop(body: dict):
|
|
375
|
+
ref = str(body.get("ref") or "")
|
|
376
|
+
# Cancelling the pending record is not enough: the background pull task
|
|
377
|
+
# has to be stopped too. Popping the record alone left it running, so it
|
|
378
|
+
# finished the pull, allocated the GPUs and started the container --
|
|
379
|
+
# after the console had already released the quota for a run it was told
|
|
380
|
+
# had stopped, which is how those cards ended up held by nothing.
|
|
381
|
+
async with state.control_lock(ref):
|
|
382
|
+
state.cancel_launch(ref)
|
|
383
|
+
ok = await state.runtime.stop(
|
|
384
|
+
ref, timeout=int(body.get("timeout") or getattr(settings, "local_stop_timeout", 30))
|
|
385
|
+
)
|
|
386
|
+
return {"ok": ok}
|
|
387
|
+
|
|
388
|
+
@app.post("/node/v1/remove", dependencies=[auth])
|
|
389
|
+
async def remove(body: dict):
|
|
390
|
+
ref = str(body.get("ref") or "")
|
|
391
|
+
async with state.control_lock(ref):
|
|
392
|
+
state.cancel_launch(ref)
|
|
393
|
+
return {"ok": await state.runtime.remove(ref)}
|
|
394
|
+
|
|
395
|
+
@app.get("/node/v1/logs", dependencies=[auth])
|
|
396
|
+
async def logs(ref: str, tail: int = 2000):
|
|
397
|
+
return {"logs": await state.runtime.logs(ref, tail_lines=tail)}
|
|
398
|
+
|
|
399
|
+
@app.post("/node/v1/reap", dependencies=[auth])
|
|
400
|
+
async def reap(body: dict):
|
|
401
|
+
ttl = int(body.get("max_age_s") or 0) or int(
|
|
402
|
+
getattr(settings, "local_container_ttl_s", 600)
|
|
403
|
+
)
|
|
404
|
+
now = time.time()
|
|
405
|
+
removed = 0
|
|
406
|
+
for st in await state.runtime.ps():
|
|
407
|
+
if st.status in ("running", "paused"):
|
|
408
|
+
continue
|
|
409
|
+
if st.status == "created":
|
|
410
|
+
# Wreckage from a successful create whose start failed. It never
|
|
411
|
+
# reaches `exited` on its own, yet it carries LABEL_GPUS and so
|
|
412
|
+
# `occupancy()` counts it as holding its cards -- leaving it is a
|
|
413
|
+
# permanent GPU leak. `LocalExecutor.reap` has handled this since
|
|
414
|
+
# it was found there; the node backend skipped every `created`
|
|
415
|
+
# container unconditionally.
|
|
416
|
+
#
|
|
417
|
+
# Only one that never started and is past its TTL: a healthy
|
|
418
|
+
# launch is `created` for milliseconds, because
|
|
419
|
+
# `allocate_and_run` holds the lock across create and start.
|
|
420
|
+
if not st.never_started:
|
|
421
|
+
continue
|
|
422
|
+
if ttl > 0 and _age_seconds(st.created_at, now) < ttl:
|
|
423
|
+
continue
|
|
424
|
+
elif ttl > 0 and _age_seconds(st.finished_at, now) < ttl:
|
|
425
|
+
continue
|
|
426
|
+
if await state.runtime.remove(st.name):
|
|
427
|
+
removed += 1
|
|
428
|
+
return {"removed": removed}
|
|
429
|
+
|
|
430
|
+
# ── Health and capacity ─────────────────────────────────────────────────
|
|
431
|
+
|
|
432
|
+
@app.get("/node/v1/health", dependencies=[auth])
|
|
433
|
+
async def health():
|
|
434
|
+
base = await state.runtime.health()
|
|
435
|
+
if not base.get("ok"):
|
|
436
|
+
return {"ok": False, "runtime": base, "protocol": PROTOCOL}
|
|
437
|
+
try:
|
|
438
|
+
occ = await state.allocator.occupancy()
|
|
439
|
+
except Exception as exc: # noqa: BLE001
|
|
440
|
+
return {"ok": False, "runtime": base, "protocol": PROTOCOL,
|
|
441
|
+
"detail": f"GPU probe failed: {exc}"}
|
|
442
|
+
# The node's view of the shared storage root. The console reports its
|
|
443
|
+
# own; both should be the same filesystem, and a node that disagrees is
|
|
444
|
+
# exactly the misconfiguration worth surfacing here.
|
|
445
|
+
disk = None
|
|
446
|
+
root = settings.storage.root
|
|
447
|
+
if settings.storage_configured and root.exists():
|
|
448
|
+
try:
|
|
449
|
+
du = shutil.disk_usage(root)
|
|
450
|
+
disk = {"path": str(root), "used_pct": round(du.used / du.total * 100.0, 1),
|
|
451
|
+
"free_gib": round(du.free / 1024**3, 1)}
|
|
452
|
+
except OSError:
|
|
453
|
+
disk = None
|
|
454
|
+
# ★ free = physically free minus the cards promised to jobs still pulling. The
|
|
455
|
+
# control plane's _place reads nothing else, so without the subtraction it
|
|
456
|
+
# promises the same cards twice and the job waits tens of minutes to find out
|
|
457
|
+
# it cannot have them.
|
|
458
|
+
reserved = state.reserved_gpus()
|
|
459
|
+
free_gpus = occ.free[reserved:] if reserved else occ.free
|
|
460
|
+
return {
|
|
461
|
+
"ok": True,
|
|
462
|
+
"protocol": PROTOCOL,
|
|
463
|
+
"runtime": base,
|
|
464
|
+
# This node's series. Where nodes hold different cards, it is what lets the
|
|
465
|
+
# control plane place a job on the right ones and count its capacity in the
|
|
466
|
+
# right bucket -- without it every node is taken to hold whatever the
|
|
467
|
+
# console's own cluster_profile says, an H100 node's cards are reported as
|
|
468
|
+
# H200, and every control that works per series stops working.
|
|
469
|
+
"series": _node_series(settings),
|
|
470
|
+
"gpus_total": occ.total,
|
|
471
|
+
"gpus_free": max(0, len(occ.free) - reserved),
|
|
472
|
+
"gpus_reserved": reserved,
|
|
473
|
+
"free_gpus": free_gpus,
|
|
474
|
+
"external": sorted(occ.external),
|
|
475
|
+
# When detection is broken `external` is always empty, so the control plane
|
|
476
|
+
# has to be able to tell "nobody holds a card on this node" from "this node
|
|
477
|
+
# cannot see whether anybody does" -- otherwise an inflated free count is
|
|
478
|
+
# undiscoverable.
|
|
479
|
+
"external_probe_ok": occ.external_probe_ok,
|
|
480
|
+
"unhealthy": {str(k): v for k, v in occ.unhealthy.items()},
|
|
481
|
+
"by_job": {str(k): v for k, v in occ.by_job.items()},
|
|
482
|
+
"gpu_detail": occ.explain(),
|
|
483
|
+
"disk": disk,
|
|
484
|
+
"pending_pulls": {
|
|
485
|
+
name: {"image": p.image, "phase": p.phase, "error": p.error}
|
|
486
|
+
for name, p in state.pending.items()
|
|
487
|
+
},
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
return app
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def make_app() -> FastAPI:
|
|
494
|
+
"""The uvicorn factory entry point, used by `tuneplane-node serve`."""
|
|
495
|
+
from tuneplane_node.settings import NodeSettings
|
|
496
|
+
|
|
497
|
+
return create_node_app(NodeSettings())
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""What this machine has, collected in one place.
|
|
2
|
+
|
|
3
|
+
**This is the tier-3 seam.** Everything the platform does with an inventory --
|
|
4
|
+
refusing a shared-mount fleet, matching a series, deciding a node can hold a job
|
|
5
|
+
-- works from the dataclass below, and only `collect()` touches hardware. A
|
|
6
|
+
runner with no GPU therefore exercises registration, gating, drain and the whole
|
|
7
|
+
console surface; one function stays unproven, and its size is known (CLAUDE.md,
|
|
8
|
+
*Where a check runs*).
|
|
9
|
+
|
|
10
|
+
It answers only what the console cannot see for itself and cannot be told
|
|
11
|
+
truthfully by configuration: the driver, the container runtime, the cards, and
|
|
12
|
+
whether the shared storage root is actually here.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import asyncio
|
|
17
|
+
import os
|
|
18
|
+
import shutil
|
|
19
|
+
from dataclasses import asdict, dataclass
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class NodeInventory:
|
|
25
|
+
#: What the platform calls these cards, e.g. "h100". Empty when the driver
|
|
26
|
+
#: reports a card the series registry has never heard of.
|
|
27
|
+
accelerator_series: str = ""
|
|
28
|
+
#: What the driver calls them, e.g. "NVIDIA H100 80GB HBM3". Kept alongside
|
|
29
|
+
#: the series because an unmapped card is exactly the case where an operator
|
|
30
|
+
#: needs to see the real name to add it to the registry.
|
|
31
|
+
accelerator_name: str = ""
|
|
32
|
+
accelerator_count: int = 0
|
|
33
|
+
driver: str = ""
|
|
34
|
+
runtime: str = ""
|
|
35
|
+
#: Whether the Storage Root is present at the same absolute path the console
|
|
36
|
+
#: uses. A shared-mount Fleet refuses a node that answers False, because the
|
|
37
|
+
#: alternative is accepting it and failing at the first launch with an
|
|
38
|
+
#: unrelated-looking I/O error.
|
|
39
|
+
storage_root_visible: bool = False
|
|
40
|
+
disk_free_bytes: int = 0
|
|
41
|
+
cpu_count: int = 0
|
|
42
|
+
memory_bytes: int = 0
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> dict[str, Any]:
|
|
45
|
+
return asdict(self)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
async def probe_gpus() -> tuple[str, str, int]:
|
|
49
|
+
"""(card name, driver version, card count), straight from the driver.
|
|
50
|
+
|
|
51
|
+
The one place in the platform that shells out to `nvidia-smi`, and the
|
|
52
|
+
reason this module is the tier-3 seam.
|
|
53
|
+
|
|
54
|
+
**Asking the driver rather than reading configuration is the point.** The
|
|
55
|
+
node-side series used to come from this machine's own `TUNEPLANE_CLUSTER_PROFILE`,
|
|
56
|
+
which is a value an operator sets and can forget: a mixed-GPU fleet whose
|
|
57
|
+
nodes were never configured reports no series at all, and every placement
|
|
58
|
+
rule that filters by series silently stops filtering. The machine knows what
|
|
59
|
+
cards it has; nothing else reliably does.
|
|
60
|
+
|
|
61
|
+
A machine with no `nvidia-smi` answers ("", "", 0), which the console reads
|
|
62
|
+
as unknown. That is a fact, and guessing from the container image would not
|
|
63
|
+
be.
|
|
64
|
+
"""
|
|
65
|
+
try:
|
|
66
|
+
proc = await asyncio.create_subprocess_exec(
|
|
67
|
+
"nvidia-smi", "--query-gpu=name,driver_version", "--format=csv,noheader",
|
|
68
|
+
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.DEVNULL,
|
|
69
|
+
)
|
|
70
|
+
out, _ = await asyncio.wait_for(proc.communicate(), timeout=10)
|
|
71
|
+
except (FileNotFoundError, OSError, asyncio.TimeoutError):
|
|
72
|
+
return "", "", 0
|
|
73
|
+
if proc.returncode != 0:
|
|
74
|
+
return "", "", 0
|
|
75
|
+
# One line per card. They are homogeneous on any machine the platform can
|
|
76
|
+
# place a pool on, so the first line names them all; the line count is the
|
|
77
|
+
# card count, which is also what the allocator sees.
|
|
78
|
+
lines = [ln.strip() for ln in (out or b"").decode(errors="replace").splitlines() if ln.strip()]
|
|
79
|
+
if not lines:
|
|
80
|
+
return "", "", 0
|
|
81
|
+
name, _, driver = lines[0].partition(",")
|
|
82
|
+
return name.strip(), driver.strip(), len(lines)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _memory_bytes() -> int:
|
|
86
|
+
try:
|
|
87
|
+
return int(os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES"))
|
|
88
|
+
except (ValueError, OSError, AttributeError):
|
|
89
|
+
# Not every platform exposes these, and a missing figure is not worth
|
|
90
|
+
# failing a registration over.
|
|
91
|
+
return 0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _series_of(gpu_name: str, settings) -> str:
|
|
95
|
+
"""What this node was configured as, reported as-is.
|
|
96
|
+
|
|
97
|
+
**Not mapped to a hardware series here.** The registry that knows which card
|
|
98
|
+
names belong to which series is the console's, editable in its admin page; a
|
|
99
|
+
node that mapped would be a second answer to a question one place already
|
|
100
|
+
owns, and the two would drift the first time an administrator added a card.
|
|
101
|
+
|
|
102
|
+
So the node reports two facts -- the profile an operator declared for this
|
|
103
|
+
machine, and the name the driver gives the cards -- and the console decides
|
|
104
|
+
what they mean. Both may be empty, and empty is honest: placement treats an
|
|
105
|
+
unknown series as "do not exclude this node" rather than pretending to know.
|
|
106
|
+
"""
|
|
107
|
+
from tuneplane_node.daemon import _node_series
|
|
108
|
+
|
|
109
|
+
return _node_series(settings)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
async def collect(settings, *, runtime=None, allocator=None) -> NodeInventory:
|
|
113
|
+
"""Ask this machine what it is. Never raises: a partial answer beats none.
|
|
114
|
+
|
|
115
|
+
A field the machine cannot answer stays at its zero value, which the console
|
|
116
|
+
reads as "unknown" rather than "none" -- the two matter for capacity but not
|
|
117
|
+
for whether the node may join, and inventing a number here would be worse
|
|
118
|
+
than an empty one.
|
|
119
|
+
"""
|
|
120
|
+
from tuneplane_node.allocator import GpuAllocator
|
|
121
|
+
from tuneplane_node.runtime import detect_runtime
|
|
122
|
+
|
|
123
|
+
runtime = runtime or detect_runtime(settings)
|
|
124
|
+
allocator = allocator or GpuAllocator(settings, runtime=runtime)
|
|
125
|
+
|
|
126
|
+
runtime_name = ""
|
|
127
|
+
driver = ""
|
|
128
|
+
try:
|
|
129
|
+
health = await runtime.health()
|
|
130
|
+
runtime_name = f"{getattr(runtime, 'name', '')}://{health.get('version') or ''}".strip("/:")
|
|
131
|
+
except Exception: # noqa: BLE001
|
|
132
|
+
runtime_name = str(getattr(runtime, "name", "") or "")
|
|
133
|
+
|
|
134
|
+
gpu_name, driver, probed = await probe_gpus()
|
|
135
|
+
count = probed
|
|
136
|
+
try:
|
|
137
|
+
# The allocator's figure wins when it has one: it is what actually
|
|
138
|
+
# decides placement on this node, and a disagreement between it and the
|
|
139
|
+
# driver is the allocator's to explain.
|
|
140
|
+
count = int((await allocator.occupancy()).total) or probed
|
|
141
|
+
except Exception: # noqa: BLE001
|
|
142
|
+
count = probed
|
|
143
|
+
|
|
144
|
+
root = settings.storage.root
|
|
145
|
+
visible = bool(getattr(settings, "storage_configured", False)) and root.exists()
|
|
146
|
+
free = 0
|
|
147
|
+
if visible:
|
|
148
|
+
try:
|
|
149
|
+
free = int(shutil.disk_usage(root).free)
|
|
150
|
+
except OSError:
|
|
151
|
+
visible = False
|
|
152
|
+
|
|
153
|
+
return NodeInventory(
|
|
154
|
+
accelerator_series=_series_of(gpu_name, settings),
|
|
155
|
+
accelerator_name=gpu_name,
|
|
156
|
+
accelerator_count=count,
|
|
157
|
+
driver=driver,
|
|
158
|
+
runtime=runtime_name,
|
|
159
|
+
storage_root_visible=visible,
|
|
160
|
+
disk_free_bytes=free,
|
|
161
|
+
cpu_count=os.cpu_count() or 0,
|
|
162
|
+
memory_bytes=_memory_bytes(),
|
|
163
|
+
)
|