tuneplane-node 0.3.15__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tuneplane_node/__init__.py +1 -0
- tuneplane_node/allocator.py +596 -0
- tuneplane_node/cli.py +189 -0
- tuneplane_node/daemon.py +497 -0
- tuneplane_node/inventory.py +163 -0
- tuneplane_node/join.py +199 -0
- tuneplane_node/runtime.py +611 -0
- tuneplane_node/settings.py +77 -0
- tuneplane_node/wire.py +85 -0
- tuneplane_node-0.3.15.dist-info/METADATA +12 -0
- tuneplane_node-0.3.15.dist-info/RECORD +13 -0
- tuneplane_node-0.3.15.dist-info/WHEEL +4 -0
- tuneplane_node-0.3.15.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,611 @@
|
|
|
1
|
+
"""Thin wrapper over a container runtime: docker / podman / bare process.
|
|
2
|
+
|
|
3
|
+
Why the CLI rather than an SDK
|
|
4
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
5
|
+
docker's Python SDK is synchronous and blocking, which does not fit this service's
|
|
6
|
+
async stack (either wrap it in a thread pool or take a community fork); podman's
|
|
7
|
+
HTTP API is a different interface again, so both would mean two clients to keep.
|
|
8
|
+
|
|
9
|
+
Six actions are all that is used: run / inspect / logs / stop / rm / ps. podman's
|
|
10
|
+
CLI is deliberately docker-compatible and the flags for those six are all but
|
|
11
|
+
identical -- the CLI is the greatest common factor of the two, and
|
|
12
|
+
`asyncio.create_subprocess_exec` is async without help.
|
|
13
|
+
|
|
14
|
+
**argv lists throughout, never a shell.** A job's entrypoint comes from an
|
|
15
|
+
experiment the user uploaded and its env holds user-controlled values; one place
|
|
16
|
+
that joins either into a shell string is an injection surface.
|
|
17
|
+
|
|
18
|
+
The trade-off between the three runtimes
|
|
19
|
+
──────────────────────────────────────────────────────────────────────────────
|
|
20
|
+
docker the default. Widest ecosystem, `--gpus` natively
|
|
21
|
+
podman daemonless, can run rootless; GPUs through CDI
|
|
22
|
+
(`--device nvidia.com/gpu=N`)
|
|
23
|
+
process ★ no isolation, only for a dev machine with no container runtime
|
|
24
|
+
|
|
25
|
+
The cost of process mode has to be stated plainly: it falls back to "the process
|
|
26
|
+
shares everything with the host", which is the problem containerisation was adopted
|
|
27
|
+
to solve -- a crashed training process can leave GPU memory held, and jobs have no
|
|
28
|
+
resource boundary between them. Never the default.
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import asyncio
|
|
33
|
+
import json
|
|
34
|
+
import logging
|
|
35
|
+
import os
|
|
36
|
+
import shutil
|
|
37
|
+
import signal
|
|
38
|
+
import time
|
|
39
|
+
from dataclasses import dataclass, field
|
|
40
|
+
from pathlib import Path
|
|
41
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
42
|
+
|
|
43
|
+
log = logging.getLogger(__name__)
|
|
44
|
+
|
|
45
|
+
#: Label prefix. Container labels are **the only truth about allocation** -- the
|
|
46
|
+
#: console rebuilds "which cards are held by whom" from them after a restart, so
|
|
47
|
+
#: this cannot live in memory alone.
|
|
48
|
+
LABEL_RUN_ID = "tuneplane.run-id"
|
|
49
|
+
LABEL_GPUS = "tuneplane.gpus"
|
|
50
|
+
LABEL_USER = "tuneplane.user"
|
|
51
|
+
LABEL_KIND = "tuneplane.kind"
|
|
52
|
+
#: Which execution of that run this container is (Job.attempt on the console
|
|
53
|
+
#: side). The name is a pure function of the run id, so it is reused by every
|
|
54
|
+
#: later attempt; this label is the only thing on the container that tells them
|
|
55
|
+
#: apart, and it is what a delayed cleanup checks before removing it.
|
|
56
|
+
LABEL_ATTEMPT = "tuneplane.attempt"
|
|
57
|
+
|
|
58
|
+
#: Container name prefix. A fixed prefix rather than a random name is what makes a
|
|
59
|
+
#: repeated launch of the same run_id produce one job instead of two.
|
|
60
|
+
NAME_PREFIX = "tuneplane-"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def container_name(run_id: str) -> str:
|
|
64
|
+
"""run_id -> container name. A docker name allows [a-zA-Z0-9][a-zA-Z0-9_.-]*."""
|
|
65
|
+
safe = "".join(ch if (ch.isalnum() or ch in "_.-") else "-" for ch in (run_id or ""))
|
|
66
|
+
return f"{NAME_PREFIX}{safe.strip('-') or 'job'}"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class ContainerSpec:
|
|
71
|
+
"""Everything one container launch needs, already resolved: the runtime decides nothing."""
|
|
72
|
+
|
|
73
|
+
name: str
|
|
74
|
+
image: str
|
|
75
|
+
#: Entry argv. **Not a shell string** -- where shell semantics are wanted, write
|
|
76
|
+
#: ["bash", "-lc", "..."] explicitly, so the injection surface stays at the caller
|
|
77
|
+
#: instead of being spread through here.
|
|
78
|
+
command: list[str]
|
|
79
|
+
env: dict[str, str] = field(default_factory=dict)
|
|
80
|
+
#: Physical GPU indices. An empty list means no cards (export, eval and the like).
|
|
81
|
+
gpus: list[int] = field(default_factory=list)
|
|
82
|
+
#: (host path, path inside the container, read-only)
|
|
83
|
+
mounts: list[tuple[str, str, bool]] = field(default_factory=list)
|
|
84
|
+
labels: dict[str, str] = field(default_factory=dict)
|
|
85
|
+
workdir: str = ""
|
|
86
|
+
shm_size: str = ""
|
|
87
|
+
cpu_limit: str = ""
|
|
88
|
+
memory_limit: str = ""
|
|
89
|
+
#: Training reports its status back to the console, and sharing the host's network
|
|
90
|
+
#: is the least work for the single-machine, private-network case.
|
|
91
|
+
network: str = "host"
|
|
92
|
+
user: str = ""
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass
|
|
96
|
+
class ContainerState:
|
|
97
|
+
"""The one result shape inspect / ps return."""
|
|
98
|
+
|
|
99
|
+
name: str
|
|
100
|
+
exists: bool = False
|
|
101
|
+
#: The runtime's own status: created / running / exited / dead / paused ...
|
|
102
|
+
status: str = ""
|
|
103
|
+
exit_code: Optional[int] = None
|
|
104
|
+
labels: dict[str, str] = field(default_factory=dict)
|
|
105
|
+
created_at: str = ""
|
|
106
|
+
started_at: str = ""
|
|
107
|
+
finished_at: str = ""
|
|
108
|
+
#: Killed by the kernel OOM killer. Recorded on its own because its exit code is
|
|
109
|
+
#: 137 too, exactly like a `stop` that escalated to SIGKILL -- reading the exit code
|
|
110
|
+
#: alone reports "out of memory" as "the user stopped it", and what to do next is
|
|
111
|
+
#: the opposite in each case (lower the batch size vs. nothing at all).
|
|
112
|
+
oom_killed: bool = False
|
|
113
|
+
|
|
114
|
+
@property
|
|
115
|
+
def gpus(self) -> list[int]:
|
|
116
|
+
"""Which cards this container holds, recovered from its labels."""
|
|
117
|
+
raw = self.labels.get(LABEL_GPUS, "")
|
|
118
|
+
return [int(x) for x in raw.split(",") if x.strip().isdigit()]
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def run_id(self) -> str:
|
|
122
|
+
return self.labels.get(LABEL_RUN_ID, "")
|
|
123
|
+
|
|
124
|
+
@property
|
|
125
|
+
def attempt(self) -> int:
|
|
126
|
+
"""Which execution of that run this is. 0 when the label is absent,
|
|
127
|
+
which is what a container created before the label existed looks like."""
|
|
128
|
+
raw = self.labels.get(LABEL_ATTEMPT, "")
|
|
129
|
+
return int(raw) if raw.strip().lstrip("-").isdigit() else 0
|
|
130
|
+
|
|
131
|
+
@property
|
|
132
|
+
def never_started(self) -> bool:
|
|
133
|
+
"""Created successfully but never once started.
|
|
134
|
+
|
|
135
|
+
docker reports a zero StartedAt (0001-01-01) for a container that never ran.
|
|
136
|
+
Such a container never reaches exited on its own, yet it carries LABEL_GPUS and
|
|
137
|
+
so counts as holding cards -- reclaim has to be able to recognise it.
|
|
138
|
+
"""
|
|
139
|
+
return not self.started_at or self.started_at.startswith("0001-01-01")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class RuntimeError_(RuntimeError):
|
|
143
|
+
"""A runtime call failed. Carries stderr -- that is where docker puts the reason."""
|
|
144
|
+
|
|
145
|
+
def __init__(self, message: str, *, code: int = 0, stderr: str = ""):
|
|
146
|
+
super().__init__(message)
|
|
147
|
+
self.code = code
|
|
148
|
+
self.stderr = stderr
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@runtime_checkable
|
|
152
|
+
class ContainerRuntime(Protocol):
|
|
153
|
+
name: str
|
|
154
|
+
|
|
155
|
+
async def run(self, spec: ContainerSpec) -> str: ...
|
|
156
|
+
|
|
157
|
+
async def inspect(self, name: str) -> ContainerState: ...
|
|
158
|
+
|
|
159
|
+
async def logs(self, name: str, *, tail_lines: int = 2000) -> str: ...
|
|
160
|
+
|
|
161
|
+
async def stop(self, name: str, *, timeout: int = 30) -> bool: ...
|
|
162
|
+
|
|
163
|
+
async def remove(self, name: str) -> bool: ...
|
|
164
|
+
|
|
165
|
+
async def ps(self) -> list[ContainerState]: ...
|
|
166
|
+
|
|
167
|
+
async def health(self) -> dict: ...
|
|
168
|
+
|
|
169
|
+
async def image_exists(self, image: str) -> bool: ...
|
|
170
|
+
|
|
171
|
+
async def pull(self, image: str, *, timeout: float) -> None: ...
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ── CLI runtime (docker / podman share it) ───────────────────────────────────
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
class CliRuntime:
|
|
178
|
+
"""docker / podman. Their CLIs agree on these six actions; only the GPU flags differ."""
|
|
179
|
+
|
|
180
|
+
def __init__(self, binary: str, *, timeout: float = 60.0):
|
|
181
|
+
self.binary = binary
|
|
182
|
+
self.name = Path(binary).name
|
|
183
|
+
self.timeout = timeout
|
|
184
|
+
|
|
185
|
+
# ── The underlying call ─────────────────────────────────────────────────
|
|
186
|
+
|
|
187
|
+
async def _exec(
|
|
188
|
+
self, *args: str, check: bool = True, timeout: float | None = None
|
|
189
|
+
) -> tuple[int, str, str]:
|
|
190
|
+
limit = timeout if timeout is not None else self.timeout
|
|
191
|
+
proc = await asyncio.create_subprocess_exec(
|
|
192
|
+
self.binary, *args,
|
|
193
|
+
stdout=asyncio.subprocess.PIPE,
|
|
194
|
+
stderr=asyncio.subprocess.PIPE,
|
|
195
|
+
)
|
|
196
|
+
try:
|
|
197
|
+
out, err = await asyncio.wait_for(proc.communicate(), timeout=limit)
|
|
198
|
+
except asyncio.TimeoutError:
|
|
199
|
+
proc.kill()
|
|
200
|
+
raise RuntimeError_(f"{self.name} {args[0]} timed out after {limit}s") from None
|
|
201
|
+
code = proc.returncode or 0
|
|
202
|
+
stdout, stderr = out.decode(errors="replace"), err.decode(errors="replace")
|
|
203
|
+
if check and code != 0:
|
|
204
|
+
raise RuntimeError_(
|
|
205
|
+
f"{self.name} {args[0]} failed (exit {code}): {stderr.strip()[:400]}",
|
|
206
|
+
code=code, stderr=stderr,
|
|
207
|
+
)
|
|
208
|
+
return code, stdout, stderr
|
|
209
|
+
|
|
210
|
+
# ── GPU flags: the only real difference between docker and podman ───────
|
|
211
|
+
|
|
212
|
+
def _gpu_args(self, gpus: list[int]) -> list[str]:
|
|
213
|
+
if not gpus:
|
|
214
|
+
return []
|
|
215
|
+
ids = ",".join(str(i) for i in gpus)
|
|
216
|
+
if self.name == "podman":
|
|
217
|
+
# podman goes through CDI, one --device per card. Requires that
|
|
218
|
+
# `nvidia-ctk cdi generate` has already been run.
|
|
219
|
+
return [arg for i in gpus for arg in ("--device", f"nvidia.com/gpu={i}")]
|
|
220
|
+
# ★ The double quotes are load-bearing, not stray escaping: the docker CLI parses
|
|
221
|
+
# the value of --gpus with a csv.Reader, so a bare `device=0,1` is split into
|
|
222
|
+
# ["device=0", "1"] and every field without an `=` is read as a count -- one
|
|
223
|
+
# DeviceRequest then carries both Count and DeviceIDs and the daemon answers
|
|
224
|
+
# `cannot set both Count and DeviceIDs on device request` (nvidia-docker#1026).
|
|
225
|
+
# The documented form is --gpus '"device=0,1"' under a shell; this goes straight
|
|
226
|
+
# to argv with no shell, so the double quotes have to stay inside the string. A
|
|
227
|
+
# single card is quoted too: same parse, one fewer branch.
|
|
228
|
+
return ["--gpus", f'"device={ids}"']
|
|
229
|
+
|
|
230
|
+
# ── The six actions ─────────────────────────────────────────────────────
|
|
231
|
+
|
|
232
|
+
async def run(self, spec: ContainerSpec) -> str:
|
|
233
|
+
args = ["run", "-d", "--name", spec.name]
|
|
234
|
+
|
|
235
|
+
# ★ No --rm. The exit code and the logs are still needed once the container has
|
|
236
|
+
# exited; reclaiming it is reap()'s job, with the same semantics as KubeRay's
|
|
237
|
+
# ttlSecondsAfterFinished.
|
|
238
|
+
|
|
239
|
+
args += self._gpu_args(spec.gpus)
|
|
240
|
+
for k, v in spec.env.items():
|
|
241
|
+
args += ["-e", f"{k}={v}"]
|
|
242
|
+
for k, v in spec.labels.items():
|
|
243
|
+
args += ["--label", f"{k}={v}"]
|
|
244
|
+
for host, cont, ro in spec.mounts:
|
|
245
|
+
args += ["-v", f"{host}:{cont}:ro" if ro else f"{host}:{cont}"]
|
|
246
|
+
if spec.workdir:
|
|
247
|
+
args += ["-w", spec.workdir]
|
|
248
|
+
if spec.shm_size:
|
|
249
|
+
args += ["--shm-size", spec.shm_size]
|
|
250
|
+
if spec.cpu_limit:
|
|
251
|
+
args += ["--cpus", spec.cpu_limit]
|
|
252
|
+
if spec.memory_limit:
|
|
253
|
+
args += ["--memory", spec.memory_limit]
|
|
254
|
+
if spec.network:
|
|
255
|
+
args += ["--network", spec.network]
|
|
256
|
+
if spec.user:
|
|
257
|
+
args += ["--user", spec.user]
|
|
258
|
+
# A training process has to be able to finish cleanly on stop (to write a
|
|
259
|
+
# checkpoint), so leave the signal a path to reach it.
|
|
260
|
+
args += ["--stop-signal", "SIGTERM"]
|
|
261
|
+
# `command` is the entry argv (see ContainerSpec), so it has to *replace*
|
|
262
|
+
# the image's ENTRYPOINT rather than be appended to it as CMD.
|
|
263
|
+
#
|
|
264
|
+
# Without `--entrypoint` it is appended, and an image whose entrypoint is
|
|
265
|
+
# a CLI rather than an exec-wrapper receives our argv as its own flags. A
|
|
266
|
+
# Playground session on `vllm/vllm-openai` -- whose ENTRYPOINT is the
|
|
267
|
+
# `vllm` command -- died at once with
|
|
268
|
+
# vllm: error: unrecognized arguments: -lc python3 -m vllm...
|
|
269
|
+
# which is this `bash -lc` arriving where a subcommand was expected. The
|
|
270
|
+
# same launch works on kuberay, because a Kubernetes container's
|
|
271
|
+
# `command` overrides the entrypoint by definition -- so the two backends
|
|
272
|
+
# disagreed about the one field that says what to run, and the process
|
|
273
|
+
# runtime (which execs the argv directly) agreed with Kubernetes.
|
|
274
|
+
#
|
|
275
|
+
# What this gives up is an image's own entrypoint wrapper, such as
|
|
276
|
+
# nvidia_entrypoint.sh on the NGC bases. That wrapper prints a banner and
|
|
277
|
+
# `exec "$@"`, and the kuberay path has been skipping it since it was
|
|
278
|
+
# written, which is the evidence that nothing depends on it.
|
|
279
|
+
if spec.command:
|
|
280
|
+
args += ["--entrypoint", spec.command[0]]
|
|
281
|
+
args.append(spec.image)
|
|
282
|
+
args += spec.command[1:]
|
|
283
|
+
|
|
284
|
+
_, out, _ = await self._exec(*args)
|
|
285
|
+
return out.strip()
|
|
286
|
+
|
|
287
|
+
async def inspect(self, name: str) -> ContainerState:
|
|
288
|
+
code, out, _ = await self._exec("inspect", name, check=False)
|
|
289
|
+
if code != 0:
|
|
290
|
+
return ContainerState(name=name, exists=False)
|
|
291
|
+
try:
|
|
292
|
+
data = json.loads(out)[0]
|
|
293
|
+
except (json.JSONDecodeError, IndexError, KeyError):
|
|
294
|
+
return ContainerState(name=name, exists=False)
|
|
295
|
+
return _state_from_inspect(name, data)
|
|
296
|
+
|
|
297
|
+
async def logs(self, name: str, *, tail_lines: int = 2000) -> str:
|
|
298
|
+
# docker logs keeps the training run's stdout and stderr apart; both are wanted.
|
|
299
|
+
code, out, err = await self._exec(
|
|
300
|
+
"logs", "--timestamps", "--tail", str(tail_lines), name, check=False
|
|
301
|
+
)
|
|
302
|
+
if code != 0:
|
|
303
|
+
return f"(the container logs could not be read: {err.strip()[:200]})"
|
|
304
|
+
return out + err
|
|
305
|
+
|
|
306
|
+
async def stop(self, name: str, *, timeout: int = 30) -> bool:
|
|
307
|
+
code, _, err = await self._exec("stop", "-t", str(timeout), name, check=False)
|
|
308
|
+
if code == 0:
|
|
309
|
+
return True
|
|
310
|
+
# Already gone = the goal is met, not a failure.
|
|
311
|
+
if "no such container" in err.lower():
|
|
312
|
+
return True
|
|
313
|
+
log.warning("failed to stop container %s: %s", name, err.strip()[:200])
|
|
314
|
+
return False
|
|
315
|
+
|
|
316
|
+
async def remove(self, name: str) -> bool:
|
|
317
|
+
code, _, err = await self._exec("rm", "-f", name, check=False)
|
|
318
|
+
return code == 0 or "no such container" in err.lower()
|
|
319
|
+
|
|
320
|
+
async def ps(self) -> list[ContainerState]:
|
|
321
|
+
"""List every container this platform owns, exited ones included.
|
|
322
|
+
|
|
323
|
+
`--filter label=` filters on the key alone, so what comes back is the platform's
|
|
324
|
+
own containers and nothing else running on the host is touched -- which matters,
|
|
325
|
+
because reclaim relies on it.
|
|
326
|
+
"""
|
|
327
|
+
_, out, _ = await self._exec(
|
|
328
|
+
"ps", "-a", "--filter", f"label={LABEL_RUN_ID}", "--format", "{{.Names}}"
|
|
329
|
+
)
|
|
330
|
+
names = [n.strip() for n in out.splitlines() if n.strip()]
|
|
331
|
+
if not names:
|
|
332
|
+
return []
|
|
333
|
+
states = await asyncio.gather(
|
|
334
|
+
*(self.inspect(n) for n in names), return_exceptions=True
|
|
335
|
+
)
|
|
336
|
+
return [s for s in states if isinstance(s, ContainerState) and s.exists]
|
|
337
|
+
|
|
338
|
+
async def health(self) -> dict:
|
|
339
|
+
try:
|
|
340
|
+
_, out, _ = await self._exec("version", "--format", "{{.Server.Version}}")
|
|
341
|
+
except RuntimeError_ as e:
|
|
342
|
+
return {"backend": self.name, "ok": False, "detail": str(e)}
|
|
343
|
+
return {"backend": self.name, "ok": True, "version": out.strip()}
|
|
344
|
+
|
|
345
|
+
async def image_exists(self, image: str) -> bool:
|
|
346
|
+
code, _, _ = await self._exec("image", "inspect", image, check=False)
|
|
347
|
+
return code == 0
|
|
348
|
+
|
|
349
|
+
async def pull(self, image: str, *, timeout: float) -> None:
|
|
350
|
+
"""Pull an image explicitly, on a timeout of its own.
|
|
351
|
+
|
|
352
|
+
A training image runs to tens of GB and does not fit inside the default CLI
|
|
353
|
+
timeout. Skip this and `run` pulls implicitly and hits local_cli_timeout (60s):
|
|
354
|
+
the first job to use a new image always fails to start, reported as nothing but
|
|
355
|
+
a timeout, which is close to undiagnosable from the outside.
|
|
356
|
+
"""
|
|
357
|
+
await self._exec("pull", image, timeout=timeout)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def _state_from_inspect(name: str, data: dict) -> ContainerState:
|
|
361
|
+
state = data.get("State") or {}
|
|
362
|
+
cfg = data.get("Config") or {}
|
|
363
|
+
status = str(state.get("Status") or "").lower()
|
|
364
|
+
# ExitCode is 0 while running, which means "has not exited yet" rather than "exited
|
|
365
|
+
# successfully" -- taking it as 0 records a running job as SUCCEEDED in the ledger,
|
|
366
|
+
# releasing its quota while the cards are still held.
|
|
367
|
+
exit_code = None if status in ("running", "created", "paused") else state.get("ExitCode")
|
|
368
|
+
return ContainerState(
|
|
369
|
+
name=name,
|
|
370
|
+
exists=True,
|
|
371
|
+
status=status,
|
|
372
|
+
exit_code=exit_code if exit_code is None else int(exit_code),
|
|
373
|
+
labels={k: str(v) for k, v in (cfg.get("Labels") or {}).items()},
|
|
374
|
+
created_at=str(data.get("Created") or ""),
|
|
375
|
+
started_at=str(state.get("StartedAt") or ""),
|
|
376
|
+
finished_at=str(state.get("FinishedAt") or ""),
|
|
377
|
+
oom_killed=bool(state.get("OOMKilled")),
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
# ── Bare process runtime ─────────────────────────────────────────────────────
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
class ProcessRuntime:
|
|
385
|
+
"""The fallback when there is no container runtime. **No isolation; dev machines only.**
|
|
386
|
+
|
|
387
|
+
Three real differences from the container modes, to be known before choosing it:
|
|
388
|
+
|
|
389
|
+
1. Held GPU memory comes back. A container exiting destroys its cgroup and the
|
|
390
|
+
memory is certain to be released; a bare process that crashes can leave a
|
|
391
|
+
zombie holding a card, which is the reason containers were adopted here.
|
|
392
|
+
2. No resource boundary. One job going OOM takes the whole machine with it.
|
|
393
|
+
3. State has to be stored here. docker keeps a container's metadata; a bare
|
|
394
|
+
process is only a pid -- after a console restart the pid alone cannot give the
|
|
395
|
+
exit code, so this writes a state file.
|
|
396
|
+
"""
|
|
397
|
+
|
|
398
|
+
name = "process"
|
|
399
|
+
|
|
400
|
+
def __init__(self, state_dir: Path, *, timeout: float = 60.0):
|
|
401
|
+
self.state_dir = Path(state_dir)
|
|
402
|
+
self.timeout = timeout
|
|
403
|
+
|
|
404
|
+
def _meta(self, name: str) -> Path:
|
|
405
|
+
return self.state_dir / f"{name}.json"
|
|
406
|
+
|
|
407
|
+
def _log(self, name: str) -> Path:
|
|
408
|
+
return self.state_dir / f"{name}.log"
|
|
409
|
+
|
|
410
|
+
def _rc(self, name: str) -> Path:
|
|
411
|
+
return self.state_dir / f"{name}.rc"
|
|
412
|
+
|
|
413
|
+
async def run(self, spec: ContainerSpec) -> str:
|
|
414
|
+
self.state_dir.mkdir(parents=True, exist_ok=True)
|
|
415
|
+
env = {**os.environ, **spec.env}
|
|
416
|
+
if spec.gpus:
|
|
417
|
+
env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
|
|
418
|
+
else:
|
|
419
|
+
# Set empty rather than left unset: unset, the process sees every card and
|
|
420
|
+
# can fill them all.
|
|
421
|
+
env["CUDA_VISIBLE_DEVICES"] = ""
|
|
422
|
+
|
|
423
|
+
rc_path, log_path = self._rc(spec.name), self._log(spec.name)
|
|
424
|
+
rc_path.unlink(missing_ok=True)
|
|
425
|
+
|
|
426
|
+
# The exit code has to stay readable across a console restart, so wrap the
|
|
427
|
+
# command in a shell that writes it to a file. That shell does nothing but run
|
|
428
|
+
# the command and record its status; the user's command arrives through "$@"
|
|
429
|
+
# rather than being interpolated -- interpolation is injection.
|
|
430
|
+
wrapper = 'set -o pipefail; "$@"; echo $? > "$TUNEPLANE_RC_FILE"'
|
|
431
|
+
env["TUNEPLANE_RC_FILE"] = str(rc_path)
|
|
432
|
+
|
|
433
|
+
with open(log_path, "wb") as fh:
|
|
434
|
+
proc = await asyncio.create_subprocess_exec(
|
|
435
|
+
"bash", "-c", wrapper, "bash", *spec.command,
|
|
436
|
+
stdout=fh, stderr=asyncio.subprocess.STDOUT,
|
|
437
|
+
cwd=spec.workdir or None,
|
|
438
|
+
env=env,
|
|
439
|
+
start_new_session=True, # its own process group, so stop can kill all of it
|
|
440
|
+
)
|
|
441
|
+
self._meta(spec.name).write_text(json.dumps({
|
|
442
|
+
"name": spec.name, "pid": proc.pid, "labels": spec.labels,
|
|
443
|
+
"started_at": time.time(),
|
|
444
|
+
}), encoding="utf-8")
|
|
445
|
+
return str(proc.pid)
|
|
446
|
+
|
|
447
|
+
def _alive(self, pid: int) -> bool:
|
|
448
|
+
try:
|
|
449
|
+
os.kill(pid, 0)
|
|
450
|
+
except ProcessLookupError:
|
|
451
|
+
return False
|
|
452
|
+
except PermissionError:
|
|
453
|
+
return True # it exists, it just belongs to another user
|
|
454
|
+
return True
|
|
455
|
+
|
|
456
|
+
async def inspect(self, name: str) -> ContainerState:
|
|
457
|
+
meta_path = self._meta(name)
|
|
458
|
+
if not meta_path.is_file():
|
|
459
|
+
return ContainerState(name=name, exists=False)
|
|
460
|
+
try:
|
|
461
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
462
|
+
except json.JSONDecodeError:
|
|
463
|
+
return ContainerState(name=name, exists=False)
|
|
464
|
+
|
|
465
|
+
rc_path = self._rc(name)
|
|
466
|
+
if rc_path.is_file():
|
|
467
|
+
try:
|
|
468
|
+
code = int(rc_path.read_text(encoding="utf-8").strip() or 1)
|
|
469
|
+
except ValueError:
|
|
470
|
+
code = 1
|
|
471
|
+
return ContainerState(
|
|
472
|
+
name=name, exists=True, status="exited", exit_code=code,
|
|
473
|
+
labels=meta.get("labels") or {},
|
|
474
|
+
)
|
|
475
|
+
if self._alive(int(meta.get("pid") or 0)):
|
|
476
|
+
return ContainerState(
|
|
477
|
+
name=name, exists=True, status="running", labels=meta.get("labels") or {}
|
|
478
|
+
)
|
|
479
|
+
# The process is gone but wrote no exit code: it was SIGKILLed, and the OOM
|
|
480
|
+
# killer is the usual reason.
|
|
481
|
+
return ContainerState(
|
|
482
|
+
name=name, exists=True, status="exited", exit_code=137,
|
|
483
|
+
labels=meta.get("labels") or {},
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
async def logs(self, name: str, *, tail_lines: int = 2000) -> str:
|
|
487
|
+
path = self._log(name)
|
|
488
|
+
if not path.is_file():
|
|
489
|
+
return ""
|
|
490
|
+
lines = path.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
491
|
+
return "\n".join(lines[-tail_lines:])
|
|
492
|
+
|
|
493
|
+
async def stop(self, name: str, *, timeout: int = 30) -> bool:
|
|
494
|
+
meta_path = self._meta(name)
|
|
495
|
+
if not meta_path.is_file():
|
|
496
|
+
return True
|
|
497
|
+
try:
|
|
498
|
+
pid = int(json.loads(meta_path.read_text(encoding="utf-8")).get("pid") or 0)
|
|
499
|
+
except (json.JSONDecodeError, ValueError):
|
|
500
|
+
return True
|
|
501
|
+
if not pid:
|
|
502
|
+
return True
|
|
503
|
+
# Kill the whole process group: training spawns a crowd of children (Ray
|
|
504
|
+
# workers, dataloaders), and killing only the parent leaves orphans behind
|
|
505
|
+
# still holding GPU memory.
|
|
506
|
+
for sig, wait in ((signal.SIGTERM, timeout), (signal.SIGKILL, 0)):
|
|
507
|
+
try:
|
|
508
|
+
os.killpg(os.getpgid(pid), sig)
|
|
509
|
+
except (ProcessLookupError, PermissionError):
|
|
510
|
+
return True
|
|
511
|
+
if not wait:
|
|
512
|
+
break
|
|
513
|
+
for _ in range(wait * 2):
|
|
514
|
+
await asyncio.sleep(0.5)
|
|
515
|
+
if not self._alive(pid):
|
|
516
|
+
return True
|
|
517
|
+
return True
|
|
518
|
+
|
|
519
|
+
async def remove(self, name: str) -> bool:
|
|
520
|
+
for p in (self._meta(name), self._log(name), self._rc(name)):
|
|
521
|
+
p.unlink(missing_ok=True)
|
|
522
|
+
return True
|
|
523
|
+
|
|
524
|
+
async def ps(self) -> list[ContainerState]:
|
|
525
|
+
if not self.state_dir.is_dir():
|
|
526
|
+
return []
|
|
527
|
+
out = []
|
|
528
|
+
for meta in sorted(self.state_dir.glob("*.json")):
|
|
529
|
+
state = await self.inspect(meta.stem)
|
|
530
|
+
if state.exists:
|
|
531
|
+
out.append(state)
|
|
532
|
+
return out
|
|
533
|
+
|
|
534
|
+
async def health(self) -> dict:
|
|
535
|
+
return {
|
|
536
|
+
"backend": self.name, "ok": True,
|
|
537
|
+
"detail": "bare-process mode: no isolation, and GPU memory may be left behind -- for a development machine only",
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
async def image_exists(self, image: str) -> bool:
|
|
541
|
+
del image
|
|
542
|
+
return True # a bare process has no image, so it is always ready
|
|
543
|
+
|
|
544
|
+
async def pull(self, image: str, *, timeout: float) -> None:
|
|
545
|
+
del image, timeout
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
# ── Detection ────────────────────────────────────────────────────────────────
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def detect_runtime(preference: str, *, state_dir: Path, timeout: float = 60.0) -> ContainerRuntime:
|
|
552
|
+
"""Build the runtime the preference names.
|
|
553
|
+
|
|
554
|
+
`auto` tries docker then podman, and **raises when neither is there rather than
|
|
555
|
+
falling back to process**.
|
|
556
|
+
|
|
557
|
+
A silent downgrade has two consequences, and the second one is fatal:
|
|
558
|
+
1. It leaves people believing they run containerised jobs, until held GPU memory
|
|
559
|
+
one day shows there was no isolation.
|
|
560
|
+
2. ★ process mode keeps its state in a state_dir of its own. Jobs started under
|
|
561
|
+
docker are invisible when looked up in process mode -- reconciliation then
|
|
562
|
+
judges **every running job finished**, releasing quota while the cards stay
|
|
563
|
+
held. One misconfiguration is enough to scramble the whole ledger.
|
|
564
|
+
|
|
565
|
+
To run bare processes, set TUNEPLANE_LOCAL_RUNTIME=process explicitly, so that the
|
|
566
|
+
choice is on the record in the configuration.
|
|
567
|
+
"""
|
|
568
|
+
pref = (preference or "auto").strip().lower()
|
|
569
|
+
|
|
570
|
+
if pref == "process":
|
|
571
|
+
return ProcessRuntime(state_dir, timeout=timeout)
|
|
572
|
+
|
|
573
|
+
if pref in ("docker", "podman"):
|
|
574
|
+
path = shutil.which(pref)
|
|
575
|
+
if not path:
|
|
576
|
+
raise RuntimeError_(
|
|
577
|
+
f"TUNEPLANE_LOCAL_RUNTIME={pref} but {pref} is not on PATH. "
|
|
578
|
+
f"Use auto to fall back automatically, or process (no isolation)."
|
|
579
|
+
)
|
|
580
|
+
return CliRuntime(path, timeout=timeout)
|
|
581
|
+
|
|
582
|
+
if pref != "auto":
|
|
583
|
+
raise RuntimeError_(
|
|
584
|
+
f"unknown runtime {preference!r} (one of: auto, docker, podman, process)"
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
for candidate in ("docker", "podman"):
|
|
588
|
+
if path := shutil.which(candidate):
|
|
589
|
+
return CliRuntime(path, timeout=timeout)
|
|
590
|
+
raise RuntimeError_(
|
|
591
|
+
"Neither docker nor podman is on PATH. The local backend needs a container runtime; "
|
|
592
|
+
"to really run bare processes (no isolation, GPU memory may be left held), set "
|
|
593
|
+
"TUNEPLANE_LOCAL_RUNTIME=process explicitly."
|
|
594
|
+
)
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def age_seconds(finished_at: str, now: float) -> float:
|
|
598
|
+
"""How long ago a container exited. Unparseable means "long ago".
|
|
599
|
+
|
|
600
|
+
Reclaiming an exited container early costs nothing; letting wreckage pile up
|
|
601
|
+
because a timestamp format did not parse costs disk on every node.
|
|
602
|
+
"""
|
|
603
|
+
if not finished_at:
|
|
604
|
+
return float("inf")
|
|
605
|
+
from datetime import datetime
|
|
606
|
+
|
|
607
|
+
try:
|
|
608
|
+
ts = datetime.fromisoformat(finished_at.replace("Z", "+00:00"))
|
|
609
|
+
except ValueError:
|
|
610
|
+
return float("inf")
|
|
611
|
+
return max(0.0, now - ts.timestamp())
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""What a node needs to know about itself.
|
|
2
|
+
|
|
3
|
+
A deliberate subset of the console's `WebSettings`, sharing its `TUNEPLANE_` prefix
|
|
4
|
+
so an operator configures both the same way and a machine that is also the
|
|
5
|
+
console reads one file.
|
|
6
|
+
|
|
7
|
+
**Small on purpose.** The console has 235 settings and a node needs thirteen of
|
|
8
|
+
them; importing the big one to get those thirteen is what dragged
|
|
9
|
+
`python-jose`, `passlib` and SQLAlchemy onto every GPU box. Anything added here
|
|
10
|
+
should be something the machine itself has to answer -- where its containers run,
|
|
11
|
+
which cards it has, where the shared filesystem is. Everything about scheduling,
|
|
12
|
+
quota, auth and storage policy belongs to the console.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Optional
|
|
19
|
+
|
|
20
|
+
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
21
|
+
from tuneplane.layout import StorageLayout
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class NodeSettings(BaseSettings):
|
|
25
|
+
model_config = SettingsConfigDict(
|
|
26
|
+
env_prefix="TUNEPLANE_",
|
|
27
|
+
env_file=".env",
|
|
28
|
+
env_file_encoding="utf-8",
|
|
29
|
+
env_ignore_empty=True,
|
|
30
|
+
extra="ignore",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
# ── who this node answers to ────────────────────────────────────────────
|
|
34
|
+
#: Shared with the console. Missing refuses startup rather than exposing the
|
|
35
|
+
#: node's container API unauthenticated.
|
|
36
|
+
node_token: str = ""
|
|
37
|
+
#: Where the identity and credential a join produced are kept. Not under the
|
|
38
|
+
#: storage root: a node that cannot see the shared mount still has to be able
|
|
39
|
+
#: to say who it is.
|
|
40
|
+
node_state_dir: str = "~/.tuneplane-node"
|
|
41
|
+
|
|
42
|
+
# ── how it runs containers ──────────────────────────────────────────────
|
|
43
|
+
local_runtime: str = ""
|
|
44
|
+
local_cli_timeout: float = 60.0
|
|
45
|
+
local_stop_timeout: int = 30
|
|
46
|
+
local_container_ttl_s: int = 86400
|
|
47
|
+
local_image_pull_timeout_s: float = 3600.0
|
|
48
|
+
local_gpu_passthrough: str = ""
|
|
49
|
+
|
|
50
|
+
# ── what cards it has ───────────────────────────────────────────────────
|
|
51
|
+
local_gpu_count: int = 0
|
|
52
|
+
local_check_gpu_health: bool = True
|
|
53
|
+
local_check_external_gpus: bool = True
|
|
54
|
+
#: The profile an operator declared for this machine. Reported to the console
|
|
55
|
+
#: as-is: mapping a profile or a card name to a hardware series is the
|
|
56
|
+
#: console's registry to own, and a node that guessed would be a second
|
|
57
|
+
#: answer to a question one place already owns.
|
|
58
|
+
cluster_profile: str = ""
|
|
59
|
+
|
|
60
|
+
# ── where the shared filesystem is ──────────────────────────────────────
|
|
61
|
+
storage_root: Optional[Path] = None
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def storage_configured(self) -> bool:
|
|
65
|
+
return self.storage_root is not None
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def storage(self) -> StorageLayout:
|
|
69
|
+
"""The same layout the console derives, from the same root.
|
|
70
|
+
|
|
71
|
+
Falls back to `./.tuneplane-data/storage` so a developer running the agent in a
|
|
72
|
+
checkout gets a working tree without configuring anything -- the same
|
|
73
|
+
rule `WebSettings.storage` follows, and it has to be the same rule or the
|
|
74
|
+
two sides disagree about where a job's files are.
|
|
75
|
+
"""
|
|
76
|
+
root = self.storage_root or Path(os.getcwd()) / ".tuneplane-data" / "storage"
|
|
77
|
+
return StorageLayout(root)
|