tuneplane-node 0.3.15__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,611 @@
1
+ """Thin wrapper over a container runtime: docker / podman / bare process.
2
+
3
+ Why the CLI rather than an SDK
4
+ ──────────────────────────────────────────────────────────────────────────────
5
+ docker's Python SDK is synchronous and blocking, which does not fit this service's
6
+ async stack (either wrap it in a thread pool or take a community fork); podman's
7
+ HTTP API is a different interface again, so both would mean two clients to keep.
8
+
9
+ Six actions are all that is used: run / inspect / logs / stop / rm / ps. podman's
10
+ CLI is deliberately docker-compatible and the flags for those six are all but
11
+ identical -- the CLI is the greatest common factor of the two, and
12
+ `asyncio.create_subprocess_exec` is async without help.
13
+
14
+ **argv lists throughout, never a shell.** A job's entrypoint comes from an
15
+ experiment the user uploaded and its env holds user-controlled values; one place
16
+ that joins either into a shell string is an injection surface.
17
+
18
+ The trade-off between the three runtimes
19
+ ──────────────────────────────────────────────────────────────────────────────
20
+ docker the default. Widest ecosystem, `--gpus` natively
21
+ podman daemonless, can run rootless; GPUs through CDI
22
+ (`--device nvidia.com/gpu=N`)
23
+ process ★ no isolation, only for a dev machine with no container runtime
24
+
25
+ The cost of process mode has to be stated plainly: it falls back to "the process
26
+ shares everything with the host", which is the problem containerisation was adopted
27
+ to solve -- a crashed training process can leave GPU memory held, and jobs have no
28
+ resource boundary between them. Never the default.
29
+ """
30
+ from __future__ import annotations
31
+
32
+ import asyncio
33
+ import json
34
+ import logging
35
+ import os
36
+ import shutil
37
+ import signal
38
+ import time
39
+ from dataclasses import dataclass, field
40
+ from pathlib import Path
41
+ from typing import Optional, Protocol, runtime_checkable
42
+
43
+ log = logging.getLogger(__name__)
44
+
45
+ #: Label prefix. Container labels are **the only truth about allocation** -- the
46
+ #: console rebuilds "which cards are held by whom" from them after a restart, so
47
+ #: this cannot live in memory alone.
48
+ LABEL_RUN_ID = "tuneplane.run-id"
49
+ LABEL_GPUS = "tuneplane.gpus"
50
+ LABEL_USER = "tuneplane.user"
51
+ LABEL_KIND = "tuneplane.kind"
52
+ #: Which execution of that run this container is (Job.attempt on the console
53
+ #: side). The name is a pure function of the run id, so it is reused by every
54
+ #: later attempt; this label is the only thing on the container that tells them
55
+ #: apart, and it is what a delayed cleanup checks before removing it.
56
+ LABEL_ATTEMPT = "tuneplane.attempt"
57
+
58
+ #: Container name prefix. A fixed prefix rather than a random name is what makes a
59
+ #: repeated launch of the same run_id produce one job instead of two.
60
+ NAME_PREFIX = "tuneplane-"
61
+
62
+
63
+ def container_name(run_id: str) -> str:
64
+ """run_id -> container name. A docker name allows [a-zA-Z0-9][a-zA-Z0-9_.-]*."""
65
+ safe = "".join(ch if (ch.isalnum() or ch in "_.-") else "-" for ch in (run_id or ""))
66
+ return f"{NAME_PREFIX}{safe.strip('-') or 'job'}"
67
+
68
+
69
+ @dataclass
70
+ class ContainerSpec:
71
+ """Everything one container launch needs, already resolved: the runtime decides nothing."""
72
+
73
+ name: str
74
+ image: str
75
+ #: Entry argv. **Not a shell string** -- where shell semantics are wanted, write
76
+ #: ["bash", "-lc", "..."] explicitly, so the injection surface stays at the caller
77
+ #: instead of being spread through here.
78
+ command: list[str]
79
+ env: dict[str, str] = field(default_factory=dict)
80
+ #: Physical GPU indices. An empty list means no cards (export, eval and the like).
81
+ gpus: list[int] = field(default_factory=list)
82
+ #: (host path, path inside the container, read-only)
83
+ mounts: list[tuple[str, str, bool]] = field(default_factory=list)
84
+ labels: dict[str, str] = field(default_factory=dict)
85
+ workdir: str = ""
86
+ shm_size: str = ""
87
+ cpu_limit: str = ""
88
+ memory_limit: str = ""
89
+ #: Training reports its status back to the console, and sharing the host's network
90
+ #: is the least work for the single-machine, private-network case.
91
+ network: str = "host"
92
+ user: str = ""
93
+
94
+
95
+ @dataclass
96
+ class ContainerState:
97
+ """The one result shape inspect / ps return."""
98
+
99
+ name: str
100
+ exists: bool = False
101
+ #: The runtime's own status: created / running / exited / dead / paused ...
102
+ status: str = ""
103
+ exit_code: Optional[int] = None
104
+ labels: dict[str, str] = field(default_factory=dict)
105
+ created_at: str = ""
106
+ started_at: str = ""
107
+ finished_at: str = ""
108
+ #: Killed by the kernel OOM killer. Recorded on its own because its exit code is
109
+ #: 137 too, exactly like a `stop` that escalated to SIGKILL -- reading the exit code
110
+ #: alone reports "out of memory" as "the user stopped it", and what to do next is
111
+ #: the opposite in each case (lower the batch size vs. nothing at all).
112
+ oom_killed: bool = False
113
+
114
+ @property
115
+ def gpus(self) -> list[int]:
116
+ """Which cards this container holds, recovered from its labels."""
117
+ raw = self.labels.get(LABEL_GPUS, "")
118
+ return [int(x) for x in raw.split(",") if x.strip().isdigit()]
119
+
120
+ @property
121
+ def run_id(self) -> str:
122
+ return self.labels.get(LABEL_RUN_ID, "")
123
+
124
+ @property
125
+ def attempt(self) -> int:
126
+ """Which execution of that run this is. 0 when the label is absent,
127
+ which is what a container created before the label existed looks like."""
128
+ raw = self.labels.get(LABEL_ATTEMPT, "")
129
+ return int(raw) if raw.strip().lstrip("-").isdigit() else 0
130
+
131
+ @property
132
+ def never_started(self) -> bool:
133
+ """Created successfully but never once started.
134
+
135
+ docker reports a zero StartedAt (0001-01-01) for a container that never ran.
136
+ Such a container never reaches exited on its own, yet it carries LABEL_GPUS and
137
+ so counts as holding cards -- reclaim has to be able to recognise it.
138
+ """
139
+ return not self.started_at or self.started_at.startswith("0001-01-01")
140
+
141
+
142
+ class RuntimeError_(RuntimeError):
143
+ """A runtime call failed. Carries stderr -- that is where docker puts the reason."""
144
+
145
+ def __init__(self, message: str, *, code: int = 0, stderr: str = ""):
146
+ super().__init__(message)
147
+ self.code = code
148
+ self.stderr = stderr
149
+
150
+
151
+ @runtime_checkable
152
+ class ContainerRuntime(Protocol):
153
+ name: str
154
+
155
+ async def run(self, spec: ContainerSpec) -> str: ...
156
+
157
+ async def inspect(self, name: str) -> ContainerState: ...
158
+
159
+ async def logs(self, name: str, *, tail_lines: int = 2000) -> str: ...
160
+
161
+ async def stop(self, name: str, *, timeout: int = 30) -> bool: ...
162
+
163
+ async def remove(self, name: str) -> bool: ...
164
+
165
+ async def ps(self) -> list[ContainerState]: ...
166
+
167
+ async def health(self) -> dict: ...
168
+
169
+ async def image_exists(self, image: str) -> bool: ...
170
+
171
+ async def pull(self, image: str, *, timeout: float) -> None: ...
172
+
173
+
174
+ # ── CLI runtime (docker / podman share it) ───────────────────────────────────
175
+
176
+
177
+ class CliRuntime:
178
+ """docker / podman. Their CLIs agree on these six actions; only the GPU flags differ."""
179
+
180
+ def __init__(self, binary: str, *, timeout: float = 60.0):
181
+ self.binary = binary
182
+ self.name = Path(binary).name
183
+ self.timeout = timeout
184
+
185
+ # ── The underlying call ─────────────────────────────────────────────────
186
+
187
+ async def _exec(
188
+ self, *args: str, check: bool = True, timeout: float | None = None
189
+ ) -> tuple[int, str, str]:
190
+ limit = timeout if timeout is not None else self.timeout
191
+ proc = await asyncio.create_subprocess_exec(
192
+ self.binary, *args,
193
+ stdout=asyncio.subprocess.PIPE,
194
+ stderr=asyncio.subprocess.PIPE,
195
+ )
196
+ try:
197
+ out, err = await asyncio.wait_for(proc.communicate(), timeout=limit)
198
+ except asyncio.TimeoutError:
199
+ proc.kill()
200
+ raise RuntimeError_(f"{self.name} {args[0]} timed out after {limit}s") from None
201
+ code = proc.returncode or 0
202
+ stdout, stderr = out.decode(errors="replace"), err.decode(errors="replace")
203
+ if check and code != 0:
204
+ raise RuntimeError_(
205
+ f"{self.name} {args[0]} failed (exit {code}): {stderr.strip()[:400]}",
206
+ code=code, stderr=stderr,
207
+ )
208
+ return code, stdout, stderr
209
+
210
+ # ── GPU flags: the only real difference between docker and podman ───────
211
+
212
+ def _gpu_args(self, gpus: list[int]) -> list[str]:
213
+ if not gpus:
214
+ return []
215
+ ids = ",".join(str(i) for i in gpus)
216
+ if self.name == "podman":
217
+ # podman goes through CDI, one --device per card. Requires that
218
+ # `nvidia-ctk cdi generate` has already been run.
219
+ return [arg for i in gpus for arg in ("--device", f"nvidia.com/gpu={i}")]
220
+ # ★ The double quotes are load-bearing, not stray escaping: the docker CLI parses
221
+ # the value of --gpus with a csv.Reader, so a bare `device=0,1` is split into
222
+ # ["device=0", "1"] and every field without an `=` is read as a count -- one
223
+ # DeviceRequest then carries both Count and DeviceIDs and the daemon answers
224
+ # `cannot set both Count and DeviceIDs on device request` (nvidia-docker#1026).
225
+ # The documented form is --gpus '"device=0,1"' under a shell; this goes straight
226
+ # to argv with no shell, so the double quotes have to stay inside the string. A
227
+ # single card is quoted too: same parse, one fewer branch.
228
+ return ["--gpus", f'"device={ids}"']
229
+
230
+ # ── The six actions ─────────────────────────────────────────────────────
231
+
232
+ async def run(self, spec: ContainerSpec) -> str:
233
+ args = ["run", "-d", "--name", spec.name]
234
+
235
+ # ★ No --rm. The exit code and the logs are still needed once the container has
236
+ # exited; reclaiming it is reap()'s job, with the same semantics as KubeRay's
237
+ # ttlSecondsAfterFinished.
238
+
239
+ args += self._gpu_args(spec.gpus)
240
+ for k, v in spec.env.items():
241
+ args += ["-e", f"{k}={v}"]
242
+ for k, v in spec.labels.items():
243
+ args += ["--label", f"{k}={v}"]
244
+ for host, cont, ro in spec.mounts:
245
+ args += ["-v", f"{host}:{cont}:ro" if ro else f"{host}:{cont}"]
246
+ if spec.workdir:
247
+ args += ["-w", spec.workdir]
248
+ if spec.shm_size:
249
+ args += ["--shm-size", spec.shm_size]
250
+ if spec.cpu_limit:
251
+ args += ["--cpus", spec.cpu_limit]
252
+ if spec.memory_limit:
253
+ args += ["--memory", spec.memory_limit]
254
+ if spec.network:
255
+ args += ["--network", spec.network]
256
+ if spec.user:
257
+ args += ["--user", spec.user]
258
+ # A training process has to be able to finish cleanly on stop (to write a
259
+ # checkpoint), so leave the signal a path to reach it.
260
+ args += ["--stop-signal", "SIGTERM"]
261
+ # `command` is the entry argv (see ContainerSpec), so it has to *replace*
262
+ # the image's ENTRYPOINT rather than be appended to it as CMD.
263
+ #
264
+ # Without `--entrypoint` it is appended, and an image whose entrypoint is
265
+ # a CLI rather than an exec-wrapper receives our argv as its own flags. A
266
+ # Playground session on `vllm/vllm-openai` -- whose ENTRYPOINT is the
267
+ # `vllm` command -- died at once with
268
+ # vllm: error: unrecognized arguments: -lc python3 -m vllm...
269
+ # which is this `bash -lc` arriving where a subcommand was expected. The
270
+ # same launch works on kuberay, because a Kubernetes container's
271
+ # `command` overrides the entrypoint by definition -- so the two backends
272
+ # disagreed about the one field that says what to run, and the process
273
+ # runtime (which execs the argv directly) agreed with Kubernetes.
274
+ #
275
+ # What this gives up is an image's own entrypoint wrapper, such as
276
+ # nvidia_entrypoint.sh on the NGC bases. That wrapper prints a banner and
277
+ # `exec "$@"`, and the kuberay path has been skipping it since it was
278
+ # written, which is the evidence that nothing depends on it.
279
+ if spec.command:
280
+ args += ["--entrypoint", spec.command[0]]
281
+ args.append(spec.image)
282
+ args += spec.command[1:]
283
+
284
+ _, out, _ = await self._exec(*args)
285
+ return out.strip()
286
+
287
+ async def inspect(self, name: str) -> ContainerState:
288
+ code, out, _ = await self._exec("inspect", name, check=False)
289
+ if code != 0:
290
+ return ContainerState(name=name, exists=False)
291
+ try:
292
+ data = json.loads(out)[0]
293
+ except (json.JSONDecodeError, IndexError, KeyError):
294
+ return ContainerState(name=name, exists=False)
295
+ return _state_from_inspect(name, data)
296
+
297
+ async def logs(self, name: str, *, tail_lines: int = 2000) -> str:
298
+ # docker logs keeps the training run's stdout and stderr apart; both are wanted.
299
+ code, out, err = await self._exec(
300
+ "logs", "--timestamps", "--tail", str(tail_lines), name, check=False
301
+ )
302
+ if code != 0:
303
+ return f"(the container logs could not be read: {err.strip()[:200]})"
304
+ return out + err
305
+
306
+ async def stop(self, name: str, *, timeout: int = 30) -> bool:
307
+ code, _, err = await self._exec("stop", "-t", str(timeout), name, check=False)
308
+ if code == 0:
309
+ return True
310
+ # Already gone = the goal is met, not a failure.
311
+ if "no such container" in err.lower():
312
+ return True
313
+ log.warning("failed to stop container %s: %s", name, err.strip()[:200])
314
+ return False
315
+
316
+ async def remove(self, name: str) -> bool:
317
+ code, _, err = await self._exec("rm", "-f", name, check=False)
318
+ return code == 0 or "no such container" in err.lower()
319
+
320
+ async def ps(self) -> list[ContainerState]:
321
+ """List every container this platform owns, exited ones included.
322
+
323
+ `--filter label=` filters on the key alone, so what comes back is the platform's
324
+ own containers and nothing else running on the host is touched -- which matters,
325
+ because reclaim relies on it.
326
+ """
327
+ _, out, _ = await self._exec(
328
+ "ps", "-a", "--filter", f"label={LABEL_RUN_ID}", "--format", "{{.Names}}"
329
+ )
330
+ names = [n.strip() for n in out.splitlines() if n.strip()]
331
+ if not names:
332
+ return []
333
+ states = await asyncio.gather(
334
+ *(self.inspect(n) for n in names), return_exceptions=True
335
+ )
336
+ return [s for s in states if isinstance(s, ContainerState) and s.exists]
337
+
338
+ async def health(self) -> dict:
339
+ try:
340
+ _, out, _ = await self._exec("version", "--format", "{{.Server.Version}}")
341
+ except RuntimeError_ as e:
342
+ return {"backend": self.name, "ok": False, "detail": str(e)}
343
+ return {"backend": self.name, "ok": True, "version": out.strip()}
344
+
345
+ async def image_exists(self, image: str) -> bool:
346
+ code, _, _ = await self._exec("image", "inspect", image, check=False)
347
+ return code == 0
348
+
349
+ async def pull(self, image: str, *, timeout: float) -> None:
350
+ """Pull an image explicitly, on a timeout of its own.
351
+
352
+ A training image runs to tens of GB and does not fit inside the default CLI
353
+ timeout. Skip this and `run` pulls implicitly and hits local_cli_timeout (60s):
354
+ the first job to use a new image always fails to start, reported as nothing but
355
+ a timeout, which is close to undiagnosable from the outside.
356
+ """
357
+ await self._exec("pull", image, timeout=timeout)
358
+
359
+
360
+ def _state_from_inspect(name: str, data: dict) -> ContainerState:
361
+ state = data.get("State") or {}
362
+ cfg = data.get("Config") or {}
363
+ status = str(state.get("Status") or "").lower()
364
+ # ExitCode is 0 while running, which means "has not exited yet" rather than "exited
365
+ # successfully" -- taking it as 0 records a running job as SUCCEEDED in the ledger,
366
+ # releasing its quota while the cards are still held.
367
+ exit_code = None if status in ("running", "created", "paused") else state.get("ExitCode")
368
+ return ContainerState(
369
+ name=name,
370
+ exists=True,
371
+ status=status,
372
+ exit_code=exit_code if exit_code is None else int(exit_code),
373
+ labels={k: str(v) for k, v in (cfg.get("Labels") or {}).items()},
374
+ created_at=str(data.get("Created") or ""),
375
+ started_at=str(state.get("StartedAt") or ""),
376
+ finished_at=str(state.get("FinishedAt") or ""),
377
+ oom_killed=bool(state.get("OOMKilled")),
378
+ )
379
+
380
+
381
+ # ── Bare process runtime ─────────────────────────────────────────────────────
382
+
383
+
384
+ class ProcessRuntime:
385
+ """The fallback when there is no container runtime. **No isolation; dev machines only.**
386
+
387
+ Three real differences from the container modes, to be known before choosing it:
388
+
389
+ 1. Held GPU memory comes back. A container exiting destroys its cgroup and the
390
+ memory is certain to be released; a bare process that crashes can leave a
391
+ zombie holding a card, which is the reason containers were adopted here.
392
+ 2. No resource boundary. One job going OOM takes the whole machine with it.
393
+ 3. State has to be stored here. docker keeps a container's metadata; a bare
394
+ process is only a pid -- after a console restart the pid alone cannot give the
395
+ exit code, so this writes a state file.
396
+ """
397
+
398
+ name = "process"
399
+
400
+ def __init__(self, state_dir: Path, *, timeout: float = 60.0):
401
+ self.state_dir = Path(state_dir)
402
+ self.timeout = timeout
403
+
404
+ def _meta(self, name: str) -> Path:
405
+ return self.state_dir / f"{name}.json"
406
+
407
+ def _log(self, name: str) -> Path:
408
+ return self.state_dir / f"{name}.log"
409
+
410
+ def _rc(self, name: str) -> Path:
411
+ return self.state_dir / f"{name}.rc"
412
+
413
+ async def run(self, spec: ContainerSpec) -> str:
414
+ self.state_dir.mkdir(parents=True, exist_ok=True)
415
+ env = {**os.environ, **spec.env}
416
+ if spec.gpus:
417
+ env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
418
+ else:
419
+ # Set empty rather than left unset: unset, the process sees every card and
420
+ # can fill them all.
421
+ env["CUDA_VISIBLE_DEVICES"] = ""
422
+
423
+ rc_path, log_path = self._rc(spec.name), self._log(spec.name)
424
+ rc_path.unlink(missing_ok=True)
425
+
426
+ # The exit code has to stay readable across a console restart, so wrap the
427
+ # command in a shell that writes it to a file. That shell does nothing but run
428
+ # the command and record its status; the user's command arrives through "$@"
429
+ # rather than being interpolated -- interpolation is injection.
430
+ wrapper = 'set -o pipefail; "$@"; echo $? > "$TUNEPLANE_RC_FILE"'
431
+ env["TUNEPLANE_RC_FILE"] = str(rc_path)
432
+
433
+ with open(log_path, "wb") as fh:
434
+ proc = await asyncio.create_subprocess_exec(
435
+ "bash", "-c", wrapper, "bash", *spec.command,
436
+ stdout=fh, stderr=asyncio.subprocess.STDOUT,
437
+ cwd=spec.workdir or None,
438
+ env=env,
439
+ start_new_session=True, # its own process group, so stop can kill all of it
440
+ )
441
+ self._meta(spec.name).write_text(json.dumps({
442
+ "name": spec.name, "pid": proc.pid, "labels": spec.labels,
443
+ "started_at": time.time(),
444
+ }), encoding="utf-8")
445
+ return str(proc.pid)
446
+
447
+ def _alive(self, pid: int) -> bool:
448
+ try:
449
+ os.kill(pid, 0)
450
+ except ProcessLookupError:
451
+ return False
452
+ except PermissionError:
453
+ return True # it exists, it just belongs to another user
454
+ return True
455
+
456
+ async def inspect(self, name: str) -> ContainerState:
457
+ meta_path = self._meta(name)
458
+ if not meta_path.is_file():
459
+ return ContainerState(name=name, exists=False)
460
+ try:
461
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
462
+ except json.JSONDecodeError:
463
+ return ContainerState(name=name, exists=False)
464
+
465
+ rc_path = self._rc(name)
466
+ if rc_path.is_file():
467
+ try:
468
+ code = int(rc_path.read_text(encoding="utf-8").strip() or 1)
469
+ except ValueError:
470
+ code = 1
471
+ return ContainerState(
472
+ name=name, exists=True, status="exited", exit_code=code,
473
+ labels=meta.get("labels") or {},
474
+ )
475
+ if self._alive(int(meta.get("pid") or 0)):
476
+ return ContainerState(
477
+ name=name, exists=True, status="running", labels=meta.get("labels") or {}
478
+ )
479
+ # The process is gone but wrote no exit code: it was SIGKILLed, and the OOM
480
+ # killer is the usual reason.
481
+ return ContainerState(
482
+ name=name, exists=True, status="exited", exit_code=137,
483
+ labels=meta.get("labels") or {},
484
+ )
485
+
486
+ async def logs(self, name: str, *, tail_lines: int = 2000) -> str:
487
+ path = self._log(name)
488
+ if not path.is_file():
489
+ return ""
490
+ lines = path.read_text(encoding="utf-8", errors="replace").splitlines()
491
+ return "\n".join(lines[-tail_lines:])
492
+
493
+ async def stop(self, name: str, *, timeout: int = 30) -> bool:
494
+ meta_path = self._meta(name)
495
+ if not meta_path.is_file():
496
+ return True
497
+ try:
498
+ pid = int(json.loads(meta_path.read_text(encoding="utf-8")).get("pid") or 0)
499
+ except (json.JSONDecodeError, ValueError):
500
+ return True
501
+ if not pid:
502
+ return True
503
+ # Kill the whole process group: training spawns a crowd of children (Ray
504
+ # workers, dataloaders), and killing only the parent leaves orphans behind
505
+ # still holding GPU memory.
506
+ for sig, wait in ((signal.SIGTERM, timeout), (signal.SIGKILL, 0)):
507
+ try:
508
+ os.killpg(os.getpgid(pid), sig)
509
+ except (ProcessLookupError, PermissionError):
510
+ return True
511
+ if not wait:
512
+ break
513
+ for _ in range(wait * 2):
514
+ await asyncio.sleep(0.5)
515
+ if not self._alive(pid):
516
+ return True
517
+ return True
518
+
519
+ async def remove(self, name: str) -> bool:
520
+ for p in (self._meta(name), self._log(name), self._rc(name)):
521
+ p.unlink(missing_ok=True)
522
+ return True
523
+
524
+ async def ps(self) -> list[ContainerState]:
525
+ if not self.state_dir.is_dir():
526
+ return []
527
+ out = []
528
+ for meta in sorted(self.state_dir.glob("*.json")):
529
+ state = await self.inspect(meta.stem)
530
+ if state.exists:
531
+ out.append(state)
532
+ return out
533
+
534
+ async def health(self) -> dict:
535
+ return {
536
+ "backend": self.name, "ok": True,
537
+ "detail": "bare-process mode: no isolation, and GPU memory may be left behind -- for a development machine only",
538
+ }
539
+
540
+ async def image_exists(self, image: str) -> bool:
541
+ del image
542
+ return True # a bare process has no image, so it is always ready
543
+
544
+ async def pull(self, image: str, *, timeout: float) -> None:
545
+ del image, timeout
546
+
547
+
548
+ # ── Detection ────────────────────────────────────────────────────────────────
549
+
550
+
551
+ def detect_runtime(preference: str, *, state_dir: Path, timeout: float = 60.0) -> ContainerRuntime:
552
+ """Build the runtime the preference names.
553
+
554
+ `auto` tries docker then podman, and **raises when neither is there rather than
555
+ falling back to process**.
556
+
557
+ A silent downgrade has two consequences, and the second one is fatal:
558
+ 1. It leaves people believing they run containerised jobs, until held GPU memory
559
+ one day shows there was no isolation.
560
+ 2. ★ process mode keeps its state in a state_dir of its own. Jobs started under
561
+ docker are invisible when looked up in process mode -- reconciliation then
562
+ judges **every running job finished**, releasing quota while the cards stay
563
+ held. One misconfiguration is enough to scramble the whole ledger.
564
+
565
+ To run bare processes, set TUNEPLANE_LOCAL_RUNTIME=process explicitly, so that the
566
+ choice is on the record in the configuration.
567
+ """
568
+ pref = (preference or "auto").strip().lower()
569
+
570
+ if pref == "process":
571
+ return ProcessRuntime(state_dir, timeout=timeout)
572
+
573
+ if pref in ("docker", "podman"):
574
+ path = shutil.which(pref)
575
+ if not path:
576
+ raise RuntimeError_(
577
+ f"TUNEPLANE_LOCAL_RUNTIME={pref} but {pref} is not on PATH. "
578
+ f"Use auto to fall back automatically, or process (no isolation)."
579
+ )
580
+ return CliRuntime(path, timeout=timeout)
581
+
582
+ if pref != "auto":
583
+ raise RuntimeError_(
584
+ f"unknown runtime {preference!r} (one of: auto, docker, podman, process)"
585
+ )
586
+
587
+ for candidate in ("docker", "podman"):
588
+ if path := shutil.which(candidate):
589
+ return CliRuntime(path, timeout=timeout)
590
+ raise RuntimeError_(
591
+ "Neither docker nor podman is on PATH. The local backend needs a container runtime; "
592
+ "to really run bare processes (no isolation, GPU memory may be left held), set "
593
+ "TUNEPLANE_LOCAL_RUNTIME=process explicitly."
594
+ )
595
+
596
+
597
+ def age_seconds(finished_at: str, now: float) -> float:
598
+ """How long ago a container exited. Unparseable means "long ago".
599
+
600
+ Reclaiming an exited container early costs nothing; letting wreckage pile up
601
+ because a timestamp format did not parse costs disk on every node.
602
+ """
603
+ if not finished_at:
604
+ return float("inf")
605
+ from datetime import datetime
606
+
607
+ try:
608
+ ts = datetime.fromisoformat(finished_at.replace("Z", "+00:00"))
609
+ except ValueError:
610
+ return float("inf")
611
+ return max(0.0, now - ts.timestamp())
@@ -0,0 +1,77 @@
1
+ """What a node needs to know about itself.
2
+
3
+ A deliberate subset of the console's `WebSettings`, sharing its `TUNEPLANE_` prefix
4
+ so an operator configures both the same way and a machine that is also the
5
+ console reads one file.
6
+
7
+ **Small on purpose.** The console has 235 settings and a node needs thirteen of
8
+ them; importing the big one to get those thirteen is what dragged
9
+ `python-jose`, `passlib` and SQLAlchemy onto every GPU box. Anything added here
10
+ should be something the machine itself has to answer -- where its containers run,
11
+ which cards it has, where the shared filesystem is. Everything about scheduling,
12
+ quota, auth and storage policy belongs to the console.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ from pathlib import Path
18
+ from typing import Optional
19
+
20
+ from pydantic_settings import BaseSettings, SettingsConfigDict
21
+ from tuneplane.layout import StorageLayout
22
+
23
+
24
+ class NodeSettings(BaseSettings):
25
+ model_config = SettingsConfigDict(
26
+ env_prefix="TUNEPLANE_",
27
+ env_file=".env",
28
+ env_file_encoding="utf-8",
29
+ env_ignore_empty=True,
30
+ extra="ignore",
31
+ )
32
+
33
+ # ── who this node answers to ────────────────────────────────────────────
34
+ #: Shared with the console. Missing refuses startup rather than exposing the
35
+ #: node's container API unauthenticated.
36
+ node_token: str = ""
37
+ #: Where the identity and credential a join produced are kept. Not under the
38
+ #: storage root: a node that cannot see the shared mount still has to be able
39
+ #: to say who it is.
40
+ node_state_dir: str = "~/.tuneplane-node"
41
+
42
+ # ── how it runs containers ──────────────────────────────────────────────
43
+ local_runtime: str = ""
44
+ local_cli_timeout: float = 60.0
45
+ local_stop_timeout: int = 30
46
+ local_container_ttl_s: int = 86400
47
+ local_image_pull_timeout_s: float = 3600.0
48
+ local_gpu_passthrough: str = ""
49
+
50
+ # ── what cards it has ───────────────────────────────────────────────────
51
+ local_gpu_count: int = 0
52
+ local_check_gpu_health: bool = True
53
+ local_check_external_gpus: bool = True
54
+ #: The profile an operator declared for this machine. Reported to the console
55
+ #: as-is: mapping a profile or a card name to a hardware series is the
56
+ #: console's registry to own, and a node that guessed would be a second
57
+ #: answer to a question one place already owns.
58
+ cluster_profile: str = ""
59
+
60
+ # ── where the shared filesystem is ──────────────────────────────────────
61
+ storage_root: Optional[Path] = None
62
+
63
+ @property
64
+ def storage_configured(self) -> bool:
65
+ return self.storage_root is not None
66
+
67
+ @property
68
+ def storage(self) -> StorageLayout:
69
+ """The same layout the console derives, from the same root.
70
+
71
+ Falls back to `./.tuneplane-data/storage` so a developer running the agent in a
72
+ checkout gets a working tree without configuring anything -- the same
73
+ rule `WebSettings.storage` follows, and it has to be the same rule or the
74
+ two sides disagree about where a job's files are.
75
+ """
76
+ root = self.storage_root or Path(os.getcwd()) / ".tuneplane-data" / "storage"
77
+ return StorageLayout(root)