tuneplane-node 0.3.15__tar.gz → 0.3.18__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/.gitignore +8 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/PKG-INFO +1 -1
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/pyproject.toml +1 -1
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/daemon.py +45 -1
- tuneplane_node-0.3.18/src/tuneplane_node/host_resources.py +206 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/runtime.py +73 -7
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/settings.py +10 -0
- tuneplane_node-0.3.18/tests/test_host_resources.py +221 -0
- tuneplane_node-0.3.18/tests/test_host_resources_docker.py +62 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/tests/test_runtime.py +88 -4
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/LICENSE +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/NOTICE +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/__init__.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/allocator.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/cli.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/inventory.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/join.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/src/tuneplane_node/wire.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.18}/tests/test_allocator.py +0 -0
|
@@ -10,6 +10,7 @@ venv/
|
|
|
10
10
|
.python-version
|
|
11
11
|
.pytest_cache/
|
|
12
12
|
.ruff_cache/
|
|
13
|
+
server/native/target/
|
|
13
14
|
|
|
14
15
|
# Frontend
|
|
15
16
|
web/node_modules/
|
|
@@ -57,3 +58,10 @@ outputs/
|
|
|
57
58
|
|
|
58
59
|
# Agent worktrees, created by the harness and never part of a commit.
|
|
59
60
|
.claude/worktrees/
|
|
61
|
+
|
|
62
|
+
# Independent public website build
|
|
63
|
+
web/dist-website/
|
|
64
|
+
:memory:.ses
|
|
65
|
+
|
|
66
|
+
# Shipped, independently packaged plugin artifacts.
|
|
67
|
+
!server/src/tuneplane_server/_plugins/**/*.whl
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
# the local backend runs the same container runtime and allocator on its own host.
|
|
11
11
|
[project]
|
|
12
12
|
name = "tuneplane-node"
|
|
13
|
-
version = "0.3.
|
|
13
|
+
version = "0.3.18"
|
|
14
14
|
description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
|
|
15
15
|
requires-python = ">=3.12"
|
|
16
16
|
# Same licence as the control plane: this is infrastructure an organization runs,
|
|
@@ -47,11 +47,13 @@ import secrets
|
|
|
47
47
|
import shutil
|
|
48
48
|
import time
|
|
49
49
|
from dataclasses import dataclass, field
|
|
50
|
+
from pathlib import Path
|
|
50
51
|
from weakref import WeakValueDictionary
|
|
51
52
|
|
|
52
53
|
from fastapi import Depends, FastAPI, HTTPException, Request
|
|
53
54
|
|
|
54
55
|
from tuneplane_node.allocator import GpuAllocator, NoCapacity
|
|
56
|
+
from tuneplane_node.host_resources import HostResources, claim_host, read_host_capacity
|
|
55
57
|
from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
|
|
56
58
|
from tuneplane_node.runtime import age_seconds as _age_seconds
|
|
57
59
|
from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
|
|
@@ -244,6 +246,19 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
244
246
|
|
|
245
247
|
# ── Launch ──────────────────────────────────────────────────────────────
|
|
246
248
|
|
|
249
|
+
def host_capacity():
|
|
250
|
+
cpus = getattr(settings, "local_allocatable_cpus", 0)
|
|
251
|
+
memory = getattr(settings, "local_allocatable_memory_gb", 0)
|
|
252
|
+
if not cpus and not memory:
|
|
253
|
+
return None
|
|
254
|
+
if state.runtime.name not in ("docker", "podman"):
|
|
255
|
+
raise ValueError("CPU admission requires Docker or Podman")
|
|
256
|
+
return HostResources(cpus, memory * 1024**3)
|
|
257
|
+
|
|
258
|
+
def host_lock():
|
|
259
|
+
# A node-local path: separate machines must never share reservation ledgers.
|
|
260
|
+
return Path(getattr(settings, "node_state_dir", "~/.tuneplane-node")).expanduser() / "host-allocation.lock"
|
|
261
|
+
|
|
247
262
|
async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
|
|
248
263
|
async def _run(gpus: list[int]) -> str:
|
|
249
264
|
# The node writes the pick back itself: GPU passthrough follows the node's
|
|
@@ -255,6 +270,25 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
255
270
|
spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
|
|
256
271
|
return await state.runtime.run(spec)
|
|
257
272
|
|
|
273
|
+
capacity = host_capacity()
|
|
274
|
+
if requested_gpus == 0 or capacity is not None:
|
|
275
|
+
request = HostResources.from_labels(spec.labels) or (capacity if requested_gpus > 0 else None)
|
|
276
|
+
if capacity is None or request is None:
|
|
277
|
+
raise NoCapacity("CPU-only execution requires configured host capacity and explicit CPU/memory requests")
|
|
278
|
+
spec.labels.update(request.labels())
|
|
279
|
+
spec.cpu_limit = str(request.cpus)
|
|
280
|
+
spec.memory_limit = str(request.memory_bytes)
|
|
281
|
+
async with claim_host(state.runtime, capacity, request, run_id=run_id, lock_path=host_lock()) as claim:
|
|
282
|
+
if claim.existing is not None:
|
|
283
|
+
return {"phase": "running", "container_id": claim.existing.name, "gpus": claim.existing.gpus, "idempotent": True}
|
|
284
|
+
if requested_gpus == 0:
|
|
285
|
+
cid = await claim.run(lambda: _run([]))
|
|
286
|
+
gpus = []
|
|
287
|
+
else:
|
|
288
|
+
gpus, cid = await state.allocator.allocate_and_run(
|
|
289
|
+
run_id, requested_gpus, lambda ids: claim.run(lambda: _run(ids))
|
|
290
|
+
)
|
|
291
|
+
return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
|
|
258
292
|
gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
|
|
259
293
|
return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
|
|
260
294
|
|
|
@@ -294,7 +328,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
294
328
|
# what to do next rather than queueing on the node -- queueing belongs to
|
|
295
329
|
# the console's scheduler, and a second queue must not grow here.
|
|
296
330
|
pending.phase = "failed"
|
|
297
|
-
pending.error = f"the
|
|
331
|
+
pending.error = f"the resources were taken during the image pull ({exc}); submit again"
|
|
298
332
|
except Exception as exc: # noqa: BLE001
|
|
299
333
|
pending.phase = "failed"
|
|
300
334
|
pending.error = f"image pull or launch failed: {exc}"
|
|
@@ -439,6 +473,15 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
439
473
|
except Exception as exc: # noqa: BLE001
|
|
440
474
|
return {"ok": False, "runtime": base, "protocol": PROTOCOL,
|
|
441
475
|
"detail": f"GPU probe failed: {exc}"}
|
|
476
|
+
host = None
|
|
477
|
+
try:
|
|
478
|
+
capacity = host_capacity()
|
|
479
|
+
if capacity is not None:
|
|
480
|
+
usage = await read_host_capacity(state.runtime, capacity, lock_path=host_lock())
|
|
481
|
+
host = {"cpus": capacity.cpus, "memory_bytes": capacity.memory_bytes,
|
|
482
|
+
"free_cpus": usage.free_cpus, "free_memory_bytes": usage.free_memory_bytes}
|
|
483
|
+
except Exception as exc:
|
|
484
|
+
log.warning("host capacity unavailable: %s", exc)
|
|
442
485
|
# The node's view of the shared storage root. The console reports its
|
|
443
486
|
# own; both should be the same filesystem, and a node that disagrees is
|
|
444
487
|
# exactly the misconfiguration worth surfacing here.
|
|
@@ -467,6 +510,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
467
510
|
# console's own cluster_profile says, an H100 node's cards are reported as
|
|
468
511
|
# H200, and every control that works per series stops working.
|
|
469
512
|
"series": _node_series(settings),
|
|
513
|
+
"host_resources": host,
|
|
470
514
|
"gpus_total": occ.total,
|
|
471
515
|
"gpus_free": max(0, len(occ.free) - reserved),
|
|
472
516
|
"gpus_reserved": reserved,
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Host admission reconstructed from container labels under a shared file lock.
|
|
2
|
+
|
|
3
|
+
The lock file must be shared by every allocator of the same local runtime and
|
|
4
|
+
must never be unlinked. Container lifetime is reservation lifetime: a restart
|
|
5
|
+
reads the same labels, and an exited container no longer holds CPU or memory.
|
|
6
|
+
Before a container becomes observable, a durable intent covers uncertain startup.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import fcntl
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
15
|
+
import tempfile
|
|
16
|
+
from contextlib import asynccontextmanager
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from tuneplane_node.allocator import NoCapacity
|
|
21
|
+
from tuneplane_node.runtime import ContainerState
|
|
22
|
+
|
|
23
|
+
LABEL_CPUS = "tuneplane.host-cpus"
|
|
24
|
+
LABEL_MEMORY = "tuneplane.host-memory-bytes"
|
|
25
|
+
log = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class HostResources:
|
|
30
|
+
cpus: int
|
|
31
|
+
memory_bytes: int
|
|
32
|
+
|
|
33
|
+
def __post_init__(self):
|
|
34
|
+
for value in (self.cpus, self.memory_bytes):
|
|
35
|
+
if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
|
|
36
|
+
raise ValueError("host CPU and memory resources must be positive integers")
|
|
37
|
+
|
|
38
|
+
def labels(self) -> dict[str, str]:
|
|
39
|
+
return {LABEL_CPUS: str(self.cpus), LABEL_MEMORY: str(self.memory_bytes)}
|
|
40
|
+
|
|
41
|
+
@classmethod
|
|
42
|
+
def from_labels(cls, labels: dict[str, str]) -> HostResources | None:
|
|
43
|
+
try:
|
|
44
|
+
return cls(int(labels[LABEL_CPUS]), int(labels[LABEL_MEMORY]))
|
|
45
|
+
except (KeyError, ValueError, TypeError):
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class HostCapacity:
|
|
51
|
+
"""One host's observed occupancy, including unresolved launch intents."""
|
|
52
|
+
|
|
53
|
+
allocatable: HostResources
|
|
54
|
+
used_cpus: int
|
|
55
|
+
used_memory_bytes: int
|
|
56
|
+
live_runs: frozenset[str]
|
|
57
|
+
pending_runs: frozenset[str]
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def free_cpus(self) -> int:
|
|
61
|
+
return max(0, self.allocatable.cpus - self.used_cpus)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def free_memory_bytes(self) -> int:
|
|
65
|
+
return max(0, self.allocatable.memory_bytes - self.used_memory_bytes)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def single_host_request(pools) -> HostResources:
|
|
69
|
+
"""The direct single-host shape, shared by scheduling and all CPU executors."""
|
|
70
|
+
if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
|
|
71
|
+
raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
|
|
72
|
+
pool = pools[0]
|
|
73
|
+
if pool.scratch_gb:
|
|
74
|
+
raise ValueError("direct single-host jobs do not yet enforce per-job scratch space")
|
|
75
|
+
return HostResources(pool.cpus, pool.memory_gb * 1024**3)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _occupancy(capacity, states, pending) -> HostCapacity:
|
|
79
|
+
visible = {state.run_id for state in states if state.exists}
|
|
80
|
+
unresolved = {key: value for key, value in pending.items() if key not in visible}
|
|
81
|
+
live = [state for state in states if state.exists and state.status != "exited"]
|
|
82
|
+
held = [HostResources.from_labels(state.labels) or capacity for state in live]
|
|
83
|
+
held.extend(unresolved.values())
|
|
84
|
+
return HostCapacity(
|
|
85
|
+
allocatable=capacity,
|
|
86
|
+
used_cpus=sum(value.cpus for value in held),
|
|
87
|
+
used_memory_bytes=sum(value.memory_bytes for value in held),
|
|
88
|
+
live_runs=frozenset(state.run_id for state in live),
|
|
89
|
+
pending_runs=frozenset(unresolved),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@asynccontextmanager
|
|
94
|
+
async def _host_lock(lock_path: Path):
|
|
95
|
+
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
96
|
+
with lock_path.open("a+b") as lock:
|
|
97
|
+
while True:
|
|
98
|
+
try:
|
|
99
|
+
fcntl.flock(lock.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
100
|
+
break
|
|
101
|
+
except BlockingIOError:
|
|
102
|
+
await asyncio.sleep(0.05)
|
|
103
|
+
try:
|
|
104
|
+
yield
|
|
105
|
+
finally:
|
|
106
|
+
fcntl.flock(lock.fileno(), fcntl.LOCK_UN)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
async def read_host_capacity(runtime, capacity: HostResources, *, lock_path: Path) -> HostCapacity:
|
|
110
|
+
"""Read under the launch lock without reconciling or rewriting the ledger."""
|
|
111
|
+
async with _host_lock(lock_path):
|
|
112
|
+
states = await runtime.ps(strict=True)
|
|
113
|
+
return _occupancy(capacity, states, _read_pending(lock_path.with_suffix(".json")))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class HostClaim:
|
|
118
|
+
existing: ContainerState | None = None
|
|
119
|
+
started: bool = False
|
|
120
|
+
|
|
121
|
+
async def run(self, start):
|
|
122
|
+
"""Mark the point after which a failed call may still create a workload."""
|
|
123
|
+
if self.existing is not None:
|
|
124
|
+
raise ValueError("an existing host allocation must not launch another workload")
|
|
125
|
+
self.started = True
|
|
126
|
+
return await start()
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _read_pending(path: Path) -> dict[str, HostResources]:
|
|
130
|
+
try:
|
|
131
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
132
|
+
return {run_id: HostResources(**value) for run_id, value in raw.items()}
|
|
133
|
+
except FileNotFoundError:
|
|
134
|
+
return {}
|
|
135
|
+
except (ValueError, TypeError, AttributeError) as exc:
|
|
136
|
+
raise NoCapacity("host reservation ledger is unreadable; restore it before admitting jobs") from exc
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _write_pending(path: Path, pending: dict[str, HostResources]) -> None:
|
|
140
|
+
payload = {key: {"cpus": value.cpus, "memory_bytes": value.memory_bytes} for key, value in pending.items()}
|
|
141
|
+
with tempfile.NamedTemporaryFile(mode="w", encoding="utf-8", dir=path.parent, delete=False) as out:
|
|
142
|
+
temporary = Path(out.name)
|
|
143
|
+
try:
|
|
144
|
+
json.dump(payload, out)
|
|
145
|
+
out.flush()
|
|
146
|
+
os.fsync(out.fileno())
|
|
147
|
+
os.replace(temporary, path)
|
|
148
|
+
directory = os.open(path.parent, os.O_RDONLY)
|
|
149
|
+
try:
|
|
150
|
+
os.fsync(directory)
|
|
151
|
+
finally:
|
|
152
|
+
os.close(directory)
|
|
153
|
+
finally:
|
|
154
|
+
temporary.unlink(missing_ok=True)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
@asynccontextmanager
|
|
158
|
+
async def claim_host(runtime, capacity: HostResources, request: HostResources, *, run_id: str, lock_path: Path):
|
|
159
|
+
"""Hold admission through container creation; yield a launch claim or an existing run.
|
|
160
|
+
|
|
161
|
+
Unknown live allocations consume the entire host allowance. Unreadable
|
|
162
|
+
inventory fails closed. The caller must put the request labels on its new
|
|
163
|
+
container and must not launch another container when an existing run is yielded.
|
|
164
|
+
"""
|
|
165
|
+
if request.cpus > capacity.cpus or request.memory_bytes > capacity.memory_bytes:
|
|
166
|
+
raise NoCapacity("the job's host-resource request exceeds this host's allocatable capacity")
|
|
167
|
+
pending_path = lock_path.with_suffix(".json")
|
|
168
|
+
async with _host_lock(lock_path):
|
|
169
|
+
states = await runtime.ps(strict=True)
|
|
170
|
+
pending = _read_pending(pending_path)
|
|
171
|
+
# A visible container takes over its durable intent, including when
|
|
172
|
+
# it already exited. An absent one is an uncertain launch, not free capacity.
|
|
173
|
+
for state in states:
|
|
174
|
+
if state.exists:
|
|
175
|
+
pending.pop(state.run_id, None)
|
|
176
|
+
_write_pending(pending_path, pending)
|
|
177
|
+
live = [state for state in states if state.exists and state.status != "exited"]
|
|
178
|
+
existing = next((state for state in live if state.run_id == run_id), None)
|
|
179
|
+
if existing is not None:
|
|
180
|
+
yield HostClaim(existing=existing)
|
|
181
|
+
return
|
|
182
|
+
if run_id in pending:
|
|
183
|
+
raise NoCapacity("this run has an unresolved host reservation; reconcile its launch before retrying")
|
|
184
|
+
occupancy = _occupancy(capacity, states, pending)
|
|
185
|
+
if request.cpus > occupancy.free_cpus or request.memory_bytes > occupancy.free_memory_bytes:
|
|
186
|
+
raise NoCapacity(
|
|
187
|
+
f"host resources unavailable: need {request.cpus} CPUs and {request.memory_bytes} memory bytes; "
|
|
188
|
+
f"free {occupancy.free_cpus} CPUs and {occupancy.free_memory_bytes} memory bytes"
|
|
189
|
+
)
|
|
190
|
+
pending[run_id] = request
|
|
191
|
+
_write_pending(pending_path, pending)
|
|
192
|
+
claim = HostClaim()
|
|
193
|
+
try:
|
|
194
|
+
yield claim
|
|
195
|
+
finally:
|
|
196
|
+
# Failed/timed-out CLI calls can still create a container later.
|
|
197
|
+
# Keep their intent across restarts until a workload is observed.
|
|
198
|
+
try:
|
|
199
|
+
if not claim.started:
|
|
200
|
+
pending.pop(run_id, None)
|
|
201
|
+
for state in await runtime.ps(strict=True):
|
|
202
|
+
if state.exists:
|
|
203
|
+
pending.pop(state.run_id, None)
|
|
204
|
+
_write_pending(pending_path, pending)
|
|
205
|
+
except Exception:
|
|
206
|
+
log.exception("host launch outcome is unknown; retaining durable reservations")
|
|
@@ -30,6 +30,7 @@ resource boundary between them. Never the default.
|
|
|
30
30
|
from __future__ import annotations
|
|
31
31
|
|
|
32
32
|
import asyncio
|
|
33
|
+
import contextlib
|
|
33
34
|
import json
|
|
34
35
|
import logging
|
|
35
36
|
import os
|
|
@@ -57,7 +58,38 @@ LABEL_ATTEMPT = "tuneplane.attempt"
|
|
|
57
58
|
|
|
58
59
|
#: Container name prefix. A fixed prefix rather than a random name is what makes a
|
|
59
60
|
#: repeated launch of the same run_id produce one job instead of two.
|
|
60
|
-
|
|
61
|
+
#:
|
|
62
|
+
#: Two letters because the name is read, not parsed: nothing finds a container
|
|
63
|
+
#: by it -- `ps` filters on `LABEL_RUN_ID` -- so all the prefix does is tell an
|
|
64
|
+
#: operator reading `docker ps` on a shared machine which containers are the
|
|
65
|
+
#: platform's. It was `tuneplane-`, ten characters in front of an id that is
|
|
66
|
+
#: itself eighteen, on a name the console and the CLI both print.
|
|
67
|
+
NAME_PREFIX = "tp-"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
#: Environment names that carry credentials: tokens, keys, vended storage sessions.
|
|
71
|
+
_SECRET_HINTS = ("TOKEN", "SECRET", "KEY", "PASSWORD", "PASSWD", "CREDENTIAL", "STORAGE")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _is_secret(name: str, value: str) -> bool:
|
|
75
|
+
"""Whether an environment value must stay off the command line.
|
|
76
|
+
|
|
77
|
+
A presigned URL is a bearer credential whatever its variable is called. A
|
|
78
|
+
value with a line break cannot be written to an env file and stays in argv.
|
|
79
|
+
"""
|
|
80
|
+
if "\n" in value or "\r" in value:
|
|
81
|
+
return False
|
|
82
|
+
return any(hint in name.upper() for hint in _SECRET_HINTS) or "X-Amz-" in value or "Signature=" in value
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _write_env_file(values: dict[str, str]) -> str:
|
|
86
|
+
"""A private (0600) env file for the container CLI to read; the caller removes it."""
|
|
87
|
+
import tempfile
|
|
88
|
+
|
|
89
|
+
fd, path = tempfile.mkstemp(prefix="tuneplane-env-", suffix=".env")
|
|
90
|
+
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
|
91
|
+
handle.write("".join(f"{key}={value}\n" for key, value in values.items()))
|
|
92
|
+
return path
|
|
61
93
|
|
|
62
94
|
|
|
63
95
|
def container_name(run_id: str) -> str:
|
|
@@ -162,7 +194,7 @@ class ContainerRuntime(Protocol):
|
|
|
162
194
|
|
|
163
195
|
async def remove(self, name: str) -> bool: ...
|
|
164
196
|
|
|
165
|
-
async def ps(self) -> list[ContainerState]: ...
|
|
197
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]: ...
|
|
166
198
|
|
|
167
199
|
async def health(self) -> dict: ...
|
|
168
200
|
|
|
@@ -237,8 +269,21 @@ class CliRuntime:
|
|
|
237
269
|
# ttlSecondsAfterFinished.
|
|
238
270
|
|
|
239
271
|
args += self._gpu_args(spec.gpus)
|
|
240
|
-
|
|
241
|
-
|
|
272
|
+
env = dict(spec.env)
|
|
273
|
+
if not spec.gpus:
|
|
274
|
+
# Override CUDA images that default to exposing every NVIDIA device.
|
|
275
|
+
# Allocation, rather than image or user environment, owns visibility.
|
|
276
|
+
env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
|
|
277
|
+
# Secrets never ride in argv: a CLI's command line is readable through
|
|
278
|
+
# /proc by every local user for as long as the call runs. They go in a
|
|
279
|
+
# private env file the CLI reads and this process removes right after.
|
|
280
|
+
secret = {k: v for k, v in env.items() if _is_secret(k, v)}
|
|
281
|
+
for k, v in env.items():
|
|
282
|
+
if k not in secret:
|
|
283
|
+
args += ["-e", f"{k}={v}"]
|
|
284
|
+
env_file = _write_env_file(secret) if secret else None
|
|
285
|
+
if env_file is not None:
|
|
286
|
+
args += ["--env-file", env_file]
|
|
242
287
|
for k, v in spec.labels.items():
|
|
243
288
|
args += ["--label", f"{k}={v}"]
|
|
244
289
|
for host, cont, ro in spec.mounts:
|
|
@@ -281,7 +326,12 @@ class CliRuntime:
|
|
|
281
326
|
args.append(spec.image)
|
|
282
327
|
args += spec.command[1:]
|
|
283
328
|
|
|
284
|
-
|
|
329
|
+
try:
|
|
330
|
+
_, out, _ = await self._exec(*args)
|
|
331
|
+
finally:
|
|
332
|
+
if env_file is not None:
|
|
333
|
+
with contextlib.suppress(OSError):
|
|
334
|
+
os.unlink(env_file)
|
|
285
335
|
return out.strip()
|
|
286
336
|
|
|
287
337
|
async def inspect(self, name: str) -> ContainerState:
|
|
@@ -317,7 +367,7 @@ class CliRuntime:
|
|
|
317
367
|
code, _, err = await self._exec("rm", "-f", name, check=False)
|
|
318
368
|
return code == 0 or "no such container" in err.lower()
|
|
319
369
|
|
|
320
|
-
async def ps(self) -> list[ContainerState]:
|
|
370
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]:
|
|
321
371
|
"""List every container this platform owns, exited ones included.
|
|
322
372
|
|
|
323
373
|
`--filter label=` filters on the key alone, so what comes back is the platform's
|
|
@@ -330,6 +380,18 @@ class CliRuntime:
|
|
|
330
380
|
names = [n.strip() for n in out.splitlines() if n.strip()]
|
|
331
381
|
if not names:
|
|
332
382
|
return []
|
|
383
|
+
if strict:
|
|
384
|
+
async def inspect_required(name: str) -> ContainerState:
|
|
385
|
+
_, raw, _ = await self._exec("inspect", name)
|
|
386
|
+
try:
|
|
387
|
+
data = json.loads(raw)[0]
|
|
388
|
+
if not isinstance(data.get("State"), dict) or not isinstance(data.get("Config"), dict):
|
|
389
|
+
raise ValueError("missing container state or configuration")
|
|
390
|
+
return _state_from_inspect(name, data)
|
|
391
|
+
except (ValueError, IndexError, KeyError, TypeError, AttributeError) as exc:
|
|
392
|
+
raise RuntimeError_(f"cannot establish resource occupancy for container {name}") from exc
|
|
393
|
+
|
|
394
|
+
return list(await asyncio.gather(*(inspect_required(name) for name in names)))
|
|
333
395
|
states = await asyncio.gather(
|
|
334
396
|
*(self.inspect(n) for n in names), return_exceptions=True
|
|
335
397
|
)
|
|
@@ -412,6 +474,8 @@ class ProcessRuntime:
|
|
|
412
474
|
|
|
413
475
|
async def run(self, spec: ContainerSpec) -> str:
|
|
414
476
|
self.state_dir.mkdir(parents=True, exist_ok=True)
|
|
477
|
+
if spec.cpu_limit or spec.memory_limit:
|
|
478
|
+
raise RuntimeError_("the process runtime cannot enforce CPU or memory limits")
|
|
415
479
|
env = {**os.environ, **spec.env}
|
|
416
480
|
if spec.gpus:
|
|
417
481
|
env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
|
|
@@ -521,7 +585,9 @@ class ProcessRuntime:
|
|
|
521
585
|
p.unlink(missing_ok=True)
|
|
522
586
|
return True
|
|
523
587
|
|
|
524
|
-
async def ps(self) -> list[ContainerState]:
|
|
588
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]:
|
|
589
|
+
if strict:
|
|
590
|
+
raise RuntimeError_("the process runtime cannot provide enforced host-resource reservations")
|
|
525
591
|
if not self.state_dir.is_dir():
|
|
526
592
|
return []
|
|
527
593
|
out = []
|
|
@@ -17,6 +17,7 @@ import os
|
|
|
17
17
|
from pathlib import Path
|
|
18
18
|
from typing import Optional
|
|
19
19
|
|
|
20
|
+
from pydantic import Field, model_validator
|
|
20
21
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
21
22
|
from tuneplane.layout import StorageLayout
|
|
22
23
|
|
|
@@ -48,6 +49,15 @@ class NodeSettings(BaseSettings):
|
|
|
48
49
|
local_gpu_passthrough: str = ""
|
|
49
50
|
|
|
50
51
|
# ── what cards it has ───────────────────────────────────────────────────
|
|
52
|
+
local_allocatable_cpus: int = Field(default=0, ge=0)
|
|
53
|
+
local_allocatable_memory_gb: int = Field(default=0, ge=0)
|
|
54
|
+
|
|
55
|
+
@model_validator(mode="after")
|
|
56
|
+
def _host_capacity_is_paired(self):
|
|
57
|
+
if bool(self.local_allocatable_cpus) != bool(self.local_allocatable_memory_gb):
|
|
58
|
+
raise ValueError("local_allocatable_cpus and local_allocatable_memory_gb must both be positive or both zero")
|
|
59
|
+
return self
|
|
60
|
+
|
|
51
61
|
local_gpu_count: int = 0
|
|
52
62
|
local_check_gpu_health: bool = True
|
|
53
63
|
local_check_external_gpus: bool = True
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Host admission survives concurrency, restarts and uncertain container creation."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import multiprocessing
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pytest
|
|
10
|
+
|
|
11
|
+
from tuneplane_node.allocator import NoCapacity
|
|
12
|
+
from tuneplane_node.host_resources import HostResources, claim_host
|
|
13
|
+
from tuneplane_node.runtime import LABEL_GPUS, LABEL_RUN_ID, ContainerState
|
|
14
|
+
|
|
15
|
+
CAPACITY = HostResources(4, 8 * 1024**3)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Inventory:
|
|
19
|
+
def __init__(self, states=()):
|
|
20
|
+
self.states = list(states)
|
|
21
|
+
|
|
22
|
+
async def ps(self, *, strict=False):
|
|
23
|
+
assert strict
|
|
24
|
+
return list(self.states)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def state(run_id, *, status="running", labels=None):
|
|
28
|
+
return ContainerState(
|
|
29
|
+
name=f"tuneplane-{run_id}", exists=True, status=status,
|
|
30
|
+
labels={LABEL_RUN_ID: run_id, **(CAPACITY.labels() if labels is None else labels)},
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def claim(runtime, path, run_id="new", request=CAPACITY):
|
|
35
|
+
return claim_host(runtime, CAPACITY, request, run_id=run_id, lock_path=path / "host.lock")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@pytest.mark.parametrize("status", ["running", "created", "paused", "restarting", "unknown"])
|
|
39
|
+
@pytest.mark.parametrize("labels", [None, {}, {"tuneplane.host-cpus": "bad"}, {LABEL_GPUS: "0"}])
|
|
40
|
+
async def test_live_or_unknown_allocations_hold_capacity_after_restart(tmp_path, status, labels):
|
|
41
|
+
runtime = Inventory([state("old", status=status, labels=labels)])
|
|
42
|
+
with pytest.raises(NoCapacity, match="host resources unavailable"):
|
|
43
|
+
async with claim(runtime, tmp_path):
|
|
44
|
+
pytest.fail("occupied host admitted another workload")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
async def test_exit_releases_capacity_and_same_run_is_idempotent(tmp_path):
|
|
48
|
+
runtime = Inventory([state("old")])
|
|
49
|
+
async with claim(runtime, tmp_path, "old") as held:
|
|
50
|
+
assert held.existing.name == "tuneplane-old"
|
|
51
|
+
runtime.states[0].status = "exited"
|
|
52
|
+
async with claim(runtime, tmp_path) as held:
|
|
53
|
+
assert held.existing is None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
async def test_concurrent_allocators_share_the_file_lock(tmp_path):
|
|
57
|
+
runtime = Inventory()
|
|
58
|
+
started, finish = asyncio.Event(), asyncio.Event()
|
|
59
|
+
|
|
60
|
+
async def create():
|
|
61
|
+
started.set()
|
|
62
|
+
await finish.wait()
|
|
63
|
+
runtime.states.append(state("first"))
|
|
64
|
+
|
|
65
|
+
async def first():
|
|
66
|
+
async with claim(runtime, tmp_path, "first") as held:
|
|
67
|
+
await held.run(create)
|
|
68
|
+
|
|
69
|
+
async def second():
|
|
70
|
+
with pytest.raises(NoCapacity):
|
|
71
|
+
async with claim(runtime, tmp_path, "second"):
|
|
72
|
+
pytest.fail("concurrent admission oversold the host")
|
|
73
|
+
|
|
74
|
+
task = asyncio.create_task(first())
|
|
75
|
+
await started.wait()
|
|
76
|
+
waiter = asyncio.create_task(second())
|
|
77
|
+
await asyncio.sleep(0.02)
|
|
78
|
+
assert not waiter.done()
|
|
79
|
+
finish.set()
|
|
80
|
+
await asyncio.gather(task, waiter)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@pytest.mark.parametrize("failure", [TimeoutError, asyncio.CancelledError])
|
|
84
|
+
async def test_uncertain_launch_retains_durable_intent_until_observed(tmp_path, failure):
|
|
85
|
+
runtime = Inventory()
|
|
86
|
+
|
|
87
|
+
async def uncertain():
|
|
88
|
+
raise failure()
|
|
89
|
+
|
|
90
|
+
with pytest.raises(failure):
|
|
91
|
+
async with claim(runtime, tmp_path, "first") as held:
|
|
92
|
+
await held.run(uncertain)
|
|
93
|
+
with pytest.raises(NoCapacity, match="unresolved host reservation"):
|
|
94
|
+
async with claim(Inventory(), tmp_path, "first"):
|
|
95
|
+
pytest.fail("same uncertain launch was retried")
|
|
96
|
+
with pytest.raises(NoCapacity, match="host resources unavailable"):
|
|
97
|
+
async with claim(Inventory(), tmp_path, "second"):
|
|
98
|
+
pytest.fail("restart lost a pending reservation")
|
|
99
|
+
runtime.states.append(state("first", status="exited"))
|
|
100
|
+
async with claim(runtime, tmp_path, "second"):
|
|
101
|
+
pass
|
|
102
|
+
assert json.loads((tmp_path / "host.json").read_text(encoding="utf-8")) == {}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
async def test_failure_before_runtime_launch_releases_intent(tmp_path):
|
|
106
|
+
runtime = Inventory()
|
|
107
|
+
with pytest.raises(NoCapacity, match="GPU"):
|
|
108
|
+
async with claim(runtime, tmp_path):
|
|
109
|
+
raise NoCapacity("GPU shortage")
|
|
110
|
+
async with claim(runtime, tmp_path):
|
|
111
|
+
pass
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
async def test_corrupt_ledger_blocks_admission(tmp_path):
|
|
115
|
+
(tmp_path / "host.json").write_text("broken", encoding="utf-8")
|
|
116
|
+
with pytest.raises(NoCapacity, match="ledger is unreadable"):
|
|
117
|
+
async with claim(Inventory(), tmp_path):
|
|
118
|
+
pytest.fail("corrupt ledger admitted a workload")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
async def test_memory_is_admitted_independently_of_cpu(tmp_path):
|
|
122
|
+
runtime = Inventory([state("old", labels=HostResources(1, 7 * 1024**3).labels())])
|
|
123
|
+
with pytest.raises(NoCapacity):
|
|
124
|
+
async with claim(runtime, tmp_path, request=HostResources(1, 2 * 1024**3)):
|
|
125
|
+
pytest.fail("memory was oversold despite spare CPUs")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _process_claim(path, ready, outcomes, run_id):
|
|
129
|
+
"""Independent processes must see the same durable uncertain-launch intent."""
|
|
130
|
+
async def launch():
|
|
131
|
+
async with claim(Inventory(), Path(path), run_id) as held:
|
|
132
|
+
async def unobserved():
|
|
133
|
+
return "runtime accepted the launch"
|
|
134
|
+
await held.run(unobserved)
|
|
135
|
+
ready.wait(10)
|
|
136
|
+
try:
|
|
137
|
+
asyncio.run(launch())
|
|
138
|
+
outcomes.put("admitted")
|
|
139
|
+
except NoCapacity:
|
|
140
|
+
outcomes.put("blocked")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def test_separate_processes_cannot_oversell(tmp_path):
|
|
144
|
+
context = multiprocessing.get_context("spawn")
|
|
145
|
+
ready, outcomes = context.Event(), context.Queue()
|
|
146
|
+
processes = [context.Process(target=_process_claim, args=(str(tmp_path), ready, outcomes, name))
|
|
147
|
+
for name in ("first", "second")]
|
|
148
|
+
try:
|
|
149
|
+
for process in processes:
|
|
150
|
+
process.start()
|
|
151
|
+
ready.set()
|
|
152
|
+
assert sorted(outcomes.get(timeout=20) for _ in processes) == ["admitted", "blocked"]
|
|
153
|
+
for process in processes:
|
|
154
|
+
process.join(timeout=10)
|
|
155
|
+
assert process.exitcode == 0
|
|
156
|
+
finally:
|
|
157
|
+
for process in processes:
|
|
158
|
+
if process.is_alive():
|
|
159
|
+
process.terminate()
|
|
160
|
+
process.join(timeout=5)
|
|
161
|
+
outcomes.close()
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
async def test_inventory_failure_blocks_without_releasing_a_pending_launch(tmp_path):
|
|
165
|
+
class Unavailable(Inventory):
|
|
166
|
+
async def ps(self, *, strict=False):
|
|
167
|
+
raise RuntimeError("runtime unavailable")
|
|
168
|
+
|
|
169
|
+
ledger = tmp_path / "host.json"
|
|
170
|
+
original = json.dumps({"old": {"cpus": CAPACITY.cpus, "memory_bytes": CAPACITY.memory_bytes}})
|
|
171
|
+
ledger.write_text(original, encoding="utf-8")
|
|
172
|
+
with pytest.raises(RuntimeError, match="runtime unavailable"):
|
|
173
|
+
async with claim(Unavailable(), tmp_path):
|
|
174
|
+
pytest.fail("unreadable inventory admitted a job")
|
|
175
|
+
assert ledger.read_text(encoding="utf-8") == original
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
async def test_cancelling_a_lock_waiter_does_not_release_the_active_claim(tmp_path):
|
|
179
|
+
runtime = Inventory()
|
|
180
|
+
async with claim(runtime, tmp_path, "owner"):
|
|
181
|
+
async def wait():
|
|
182
|
+
async with claim(runtime, tmp_path, "waiter"):
|
|
183
|
+
pytest.fail("waiter entered a held lock")
|
|
184
|
+
waiter = asyncio.create_task(wait())
|
|
185
|
+
await asyncio.sleep(0.02)
|
|
186
|
+
waiter.cancel()
|
|
187
|
+
with pytest.raises(asyncio.CancelledError):
|
|
188
|
+
await waiter
|
|
189
|
+
ledger = json.loads((tmp_path / "host.json").read_text(encoding="utf-8"))
|
|
190
|
+
assert list(ledger) == ["owner"]
|
|
191
|
+
async with claim(runtime, tmp_path, "next"):
|
|
192
|
+
pass
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
async def test_capacity_read_includes_pending_and_does_not_rewrite_ledger(tmp_path):
|
|
196
|
+
from tuneplane_node.host_resources import read_host_capacity
|
|
197
|
+
|
|
198
|
+
path = tmp_path / "host.json"
|
|
199
|
+
original = json.dumps({
|
|
200
|
+
"visible": {"cpus": 1, "memory_bytes": 1024**3},
|
|
201
|
+
"pending": {"cpus": 2, "memory_bytes": 2 * 1024**3},
|
|
202
|
+
})
|
|
203
|
+
path.write_text(original, encoding="utf-8")
|
|
204
|
+
runtime = Inventory([state("visible", labels=HostResources(1, 1024**3).labels())])
|
|
205
|
+
capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
|
|
206
|
+
assert capacity.used_cpus == 3 and capacity.used_memory_bytes == 3 * 1024**3
|
|
207
|
+
assert capacity.free_cpus == 1
|
|
208
|
+
assert capacity.live_runs == {"visible"} and capacity.pending_runs == {"pending"}
|
|
209
|
+
assert path.read_text(encoding="utf-8") == original
|
|
210
|
+
runtime.states[0].status = "exited"
|
|
211
|
+
capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
|
|
212
|
+
assert capacity.used_cpus == 2
|
|
213
|
+
assert path.read_text(encoding="utf-8") == original
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
async def test_empty_capacity_read_does_not_create_reservation_ledger(tmp_path):
|
|
217
|
+
from tuneplane_node.host_resources import read_host_capacity
|
|
218
|
+
|
|
219
|
+
capacity = await read_host_capacity(Inventory(), CAPACITY, lock_path=tmp_path / "host.lock")
|
|
220
|
+
assert capacity.free_cpus == CAPACITY.cpus
|
|
221
|
+
assert not (tmp_path / "host.json").exists()
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Real Linux Docker acceptance; CI uses a dedicated daemon without GPU devices."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import uuid
|
|
8
|
+
|
|
9
|
+
import pytest
|
|
10
|
+
|
|
11
|
+
from tuneplane_node.allocator import NoCapacity
|
|
12
|
+
from tuneplane_node.host_resources import HostResources, claim_host
|
|
13
|
+
from tuneplane_node.runtime import LABEL_RUN_ID, CliRuntime, ContainerSpec
|
|
14
|
+
|
|
15
|
+
pytestmark = pytest.mark.skipif(
|
|
16
|
+
os.environ.get("TUNEPLANE_TEST_HOST_DOCKER") != "1",
|
|
17
|
+
reason="requires an explicitly enabled dedicated Linux Docker daemon",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
async def test_real_docker_host_admission_limits_restart_and_release(tmp_path):
|
|
22
|
+
runtime = CliRuntime("docker", timeout=30)
|
|
23
|
+
image = "alpine:3.21"
|
|
24
|
+
await runtime.pull(image, timeout=90)
|
|
25
|
+
budget = HostResources(1, 128 * 1024**2)
|
|
26
|
+
names = [f"tuneplane-host-test-{uuid.uuid4().hex}" for _ in range(2)]
|
|
27
|
+
lock_path = tmp_path / "host.lock"
|
|
28
|
+
|
|
29
|
+
async def launch(name, runtime):
|
|
30
|
+
async with claim_host(runtime, budget, budget, run_id=name, lock_path=lock_path) as held:
|
|
31
|
+
spec = ContainerSpec(
|
|
32
|
+
name=name, image=image, command=["sleep", "60"],
|
|
33
|
+
labels={LABEL_RUN_ID: name, **budget.labels()},
|
|
34
|
+
cpu_limit="1", memory_limit=str(budget.memory_bytes),
|
|
35
|
+
)
|
|
36
|
+
await held.run(lambda: runtime.run(spec))
|
|
37
|
+
return name
|
|
38
|
+
|
|
39
|
+
try:
|
|
40
|
+
outcomes = await asyncio.gather(*(launch(name, runtime) for name in names), return_exceptions=True)
|
|
41
|
+
assert sum(isinstance(outcome, NoCapacity) for outcome in outcomes) == 1
|
|
42
|
+
running = next(outcome for outcome in outcomes if isinstance(outcome, str))
|
|
43
|
+
waiting = next(name for name in names if name != running)
|
|
44
|
+
_, raw, _ = await runtime._exec("inspect", running)
|
|
45
|
+
info = json.loads(raw)[0]
|
|
46
|
+
assert info["HostConfig"]["NanoCpus"] == 1_000_000_000
|
|
47
|
+
assert info["HostConfig"]["Memory"] == budget.memory_bytes
|
|
48
|
+
assert not info["HostConfig"].get("DeviceRequests")
|
|
49
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" in info["Config"]["Env"]
|
|
50
|
+
_, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/memory.max")
|
|
51
|
+
assert int(raw.strip()) == budget.memory_bytes
|
|
52
|
+
_, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/cpu.max")
|
|
53
|
+
quota, period = map(int, raw.split())
|
|
54
|
+
assert quota == period
|
|
55
|
+
restarted = CliRuntime("docker", timeout=30)
|
|
56
|
+
with pytest.raises(NoCapacity):
|
|
57
|
+
await launch(waiting, restarted)
|
|
58
|
+
assert await runtime.stop(running, timeout=1)
|
|
59
|
+
assert await launch(waiting, restarted) == waiting
|
|
60
|
+
finally:
|
|
61
|
+
for name in names:
|
|
62
|
+
await runtime.remove(name)
|
|
@@ -21,6 +21,7 @@ from __future__ import annotations
|
|
|
21
21
|
|
|
22
22
|
import asyncio
|
|
23
23
|
import json
|
|
24
|
+
import os
|
|
24
25
|
import shutil
|
|
25
26
|
from pathlib import Path
|
|
26
27
|
|
|
@@ -87,10 +88,10 @@ def _argv(runtime: RecordingCli, verb: str) -> list[str]:
|
|
|
87
88
|
@pytest.mark.parametrize(
|
|
88
89
|
"run_id,expected",
|
|
89
90
|
[
|
|
90
|
-
("grpo_demo-alice-20260812", "
|
|
91
|
-
("a/b:c", "
|
|
92
|
-
("---", "
|
|
93
|
-
("", "
|
|
91
|
+
("grpo_demo-alice-20260812", "tp-grpo_demo-alice-20260812"),
|
|
92
|
+
("a/b:c", "tp-a-b-c"),
|
|
93
|
+
("---", "tp-job"),
|
|
94
|
+
("", "tp-job"),
|
|
94
95
|
],
|
|
95
96
|
)
|
|
96
97
|
def test_a_container_name_is_a_pure_function_of_the_run_id(run_id, expected):
|
|
@@ -166,6 +167,39 @@ def test_no_cards_means_no_gpu_flag_at_all():
|
|
|
166
167
|
# ── run ────────────────────────────────────────────────────────────────────
|
|
167
168
|
|
|
168
169
|
|
|
170
|
+
@pytest.mark.parametrize("binary", ["docker", "podman"])
|
|
171
|
+
@pytest.mark.parametrize("env", [{}, {"NVIDIA_VISIBLE_DEVICES": "all", "CUDA_VISIBLE_DEVICES": "0"}])
|
|
172
|
+
async def test_zero_gpu_containers_override_image_and_user_visibility(binary, env):
|
|
173
|
+
runtime = RecordingCli(binary=binary)
|
|
174
|
+
spec = _spec(env=env)
|
|
175
|
+
original = dict(spec.env)
|
|
176
|
+
await runtime.run(spec)
|
|
177
|
+
argv = _argv(runtime, "run")
|
|
178
|
+
assert "--gpus" not in argv and "--device" not in argv
|
|
179
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" in argv
|
|
180
|
+
assert "CUDA_VISIBLE_DEVICES=" in argv
|
|
181
|
+
assert "NVIDIA_VISIBLE_DEVICES=all" not in argv
|
|
182
|
+
assert "CUDA_VISIBLE_DEVICES=0" not in argv
|
|
183
|
+
assert spec.env == original
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
async def test_gpu_containers_preserve_their_visibility_environment():
|
|
187
|
+
runtime = RecordingCli()
|
|
188
|
+
await runtime.run(_spec(gpus=[2], env={"CUDA_VISIBLE_DEVICES": "0"}))
|
|
189
|
+
argv = _argv(runtime, "run")
|
|
190
|
+
assert "--gpus" in argv
|
|
191
|
+
assert "CUDA_VISIBLE_DEVICES=0" in argv
|
|
192
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" not in argv
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
@pytest.mark.parametrize("limits", [{"cpu_limit": "4"}, {"memory_limit": "8g"}])
|
|
196
|
+
async def test_process_runtime_refuses_limits_it_cannot_enforce(tmp_path, limits):
|
|
197
|
+
runtime = ProcessRuntime(tmp_path / "state")
|
|
198
|
+
with pytest.raises(RuntimeError_, match="cannot enforce CPU or memory"):
|
|
199
|
+
await runtime.run(_spec(**limits))
|
|
200
|
+
assert not (tmp_path / "state" / "tuneplane-run-a.json").exists()
|
|
201
|
+
|
|
202
|
+
|
|
169
203
|
async def test_the_command_replaces_the_images_entrypoint_rather_than_appending():
|
|
170
204
|
"""A Playground session on `vllm/vllm-openai` died at once with
|
|
171
205
|
`vllm: error: unrecognized arguments: -lc python3 -m vllm...` -- our argv
|
|
@@ -608,3 +642,53 @@ def test_a_container_that_finished_in_the_future_is_zero_seconds_old():
|
|
|
608
642
|
"""A clock skew between the daemon and the console must not produce a
|
|
609
643
|
negative age, which would compare as younger than every threshold."""
|
|
610
644
|
assert age_seconds("2026-09-12T10:00:00+00:00", 0.0) == 0.0
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
@pytest.mark.parametrize("payload", ["not json", "[]", '[{"State": {}}]', '[{"Config": {}}]'])
|
|
648
|
+
async def test_strict_inventory_refuses_unreadable_occupancy(payload):
|
|
649
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (0, payload, "")})
|
|
650
|
+
with pytest.raises(RuntimeError_, match="cannot establish resource occupancy"):
|
|
651
|
+
await runtime.ps(strict=True)
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
async def test_strict_inventory_propagates_inspection_failure():
|
|
655
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (1, "", "unavailable")})
|
|
656
|
+
with pytest.raises(RuntimeError_, match="inspect failed"):
|
|
657
|
+
await runtime.ps(strict=True)
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
async def test_strict_inventory_preserves_labels():
|
|
661
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""),
|
|
662
|
+
"inspect": (0, _inspect_payload(), "")})
|
|
663
|
+
states = await runtime.ps(strict=True)
|
|
664
|
+
assert len(states) == 1 and states[0].exists
|
|
665
|
+
assert states[0].labels == (await runtime.inspect("tuneplane-a")).labels
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
async def test_process_runtime_refuses_host_admission(tmp_path):
|
|
669
|
+
with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
|
|
670
|
+
await ProcessRuntime(tmp_path).ps(strict=True)
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
async def test_credentials_reach_the_container_through_a_private_env_file_never_argv():
|
|
674
|
+
seen = {}
|
|
675
|
+
|
|
676
|
+
class Reading(RecordingCli):
|
|
677
|
+
async def _exec(self, *args, check=True, timeout=None):
|
|
678
|
+
if args and args[0] == "run":
|
|
679
|
+
path = args[list(args).index("--env-file") + 1]
|
|
680
|
+
seen["mode"] = os.stat(path).st_mode & 0o777
|
|
681
|
+
with open(path, encoding="utf-8") as handle:
|
|
682
|
+
seen["text"] = handle.read()
|
|
683
|
+
seen["path"] = path
|
|
684
|
+
return await super()._exec(*args, check=check, timeout=timeout)
|
|
685
|
+
|
|
686
|
+
runtime = Reading()
|
|
687
|
+
await runtime.run(_spec(env={"TUNEPLANE_GOVERNANCE_CREDENTIAL": "tp_gw_secret", "HF_TOKEN": "hf_x",
|
|
688
|
+
"TUNEPLANE_GOVERNANCE_STORAGE": '{"secret_access_key":"s"}',
|
|
689
|
+
"TUNEPLANE_PACKAGE_URL": "https://s3/p?X-Amz-Signature=abc", "NRL_RUN_ID": "run-a"}))
|
|
690
|
+
argv = " ".join(_argv(runtime, "run"))
|
|
691
|
+
assert "tp_gw_secret" not in argv and "hf_x" not in argv and "secret_access_key" not in argv and "X-Amz" not in argv
|
|
692
|
+
assert "NRL_RUN_ID=run-a" in argv
|
|
693
|
+
assert seen["mode"] == 0o600 and "TUNEPLANE_GOVERNANCE_CREDENTIAL=tp_gw_secret\n" in seen["text"]
|
|
694
|
+
assert not os.path.exists(seen["path"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|