tuneplane-node 0.3.15__tar.gz → 0.3.16__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/.gitignore +5 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/PKG-INFO +1 -1
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/pyproject.toml +1 -1
- tuneplane_node-0.3.16/src/tuneplane_node/host_resources.py +206 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/runtime.py +25 -4
- tuneplane_node-0.3.16/tests/test_host_resources.py +221 -0
- tuneplane_node-0.3.16/tests/test_host_resources_docker.py +62 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/tests/test_runtime.py +59 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/LICENSE +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/NOTICE +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/__init__.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/allocator.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/cli.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/daemon.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/inventory.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/join.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/settings.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/src/tuneplane_node/wire.py +0 -0
- {tuneplane_node-0.3.15 → tuneplane_node-0.3.16}/tests/test_allocator.py +0 -0
|
@@ -10,6 +10,7 @@ venv/
|
|
|
10
10
|
.python-version
|
|
11
11
|
.pytest_cache/
|
|
12
12
|
.ruff_cache/
|
|
13
|
+
server/native/target/
|
|
13
14
|
|
|
14
15
|
# Frontend
|
|
15
16
|
web/node_modules/
|
|
@@ -57,3 +58,7 @@ outputs/
|
|
|
57
58
|
|
|
58
59
|
# Agent worktrees, created by the harness and never part of a commit.
|
|
59
60
|
.claude/worktrees/
|
|
61
|
+
|
|
62
|
+
# Independent public website build
|
|
63
|
+
web/dist-website/
|
|
64
|
+
:memory:.ses
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
# the local backend runs the same container runtime and allocator on its own host.
|
|
11
11
|
[project]
|
|
12
12
|
name = "tuneplane-node"
|
|
13
|
-
version = "0.3.
|
|
13
|
+
version = "0.3.16"
|
|
14
14
|
description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
|
|
15
15
|
requires-python = ">=3.12"
|
|
16
16
|
# Same licence as the control plane: this is infrastructure an organization runs,
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Host admission reconstructed from container labels under a shared file lock.
|
|
2
|
+
|
|
3
|
+
The lock file must be shared by every allocator of the same local runtime and
|
|
4
|
+
must never be unlinked. Container lifetime is reservation lifetime: a restart
|
|
5
|
+
reads the same labels, and an exited container no longer holds CPU or memory.
|
|
6
|
+
Before a container becomes observable, a durable intent covers uncertain startup.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import fcntl
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
15
|
+
import tempfile
|
|
16
|
+
from contextlib import asynccontextmanager
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from tuneplane_node.allocator import NoCapacity
|
|
21
|
+
from tuneplane_node.runtime import ContainerState
|
|
22
|
+
|
|
23
|
+
LABEL_CPUS = "tuneplane.host-cpus"
|
|
24
|
+
LABEL_MEMORY = "tuneplane.host-memory-bytes"
|
|
25
|
+
log = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class HostResources:
|
|
30
|
+
cpus: int
|
|
31
|
+
memory_bytes: int
|
|
32
|
+
|
|
33
|
+
def __post_init__(self):
|
|
34
|
+
for value in (self.cpus, self.memory_bytes):
|
|
35
|
+
if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
|
|
36
|
+
raise ValueError("host CPU and memory resources must be positive integers")
|
|
37
|
+
|
|
38
|
+
def labels(self) -> dict[str, str]:
|
|
39
|
+
return {LABEL_CPUS: str(self.cpus), LABEL_MEMORY: str(self.memory_bytes)}
|
|
40
|
+
|
|
41
|
+
@classmethod
|
|
42
|
+
def from_labels(cls, labels: dict[str, str]) -> HostResources | None:
|
|
43
|
+
try:
|
|
44
|
+
return cls(int(labels[LABEL_CPUS]), int(labels[LABEL_MEMORY]))
|
|
45
|
+
except (KeyError, ValueError, TypeError):
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class HostCapacity:
|
|
51
|
+
"""One host's observed occupancy, including unresolved launch intents."""
|
|
52
|
+
|
|
53
|
+
allocatable: HostResources
|
|
54
|
+
used_cpus: int
|
|
55
|
+
used_memory_bytes: int
|
|
56
|
+
live_runs: frozenset[str]
|
|
57
|
+
pending_runs: frozenset[str]
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def free_cpus(self) -> int:
|
|
61
|
+
return max(0, self.allocatable.cpus - self.used_cpus)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def free_memory_bytes(self) -> int:
|
|
65
|
+
return max(0, self.allocatable.memory_bytes - self.used_memory_bytes)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def single_host_request(pools) -> HostResources:
|
|
69
|
+
"""The supported local shape, shared by scheduling and container launch."""
|
|
70
|
+
if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
|
|
71
|
+
raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
|
|
72
|
+
pool = pools[0]
|
|
73
|
+
if pool.scratch_gb:
|
|
74
|
+
raise ValueError("the local backend does not yet enforce per-job scratch space")
|
|
75
|
+
return HostResources(pool.cpus, pool.memory_gb * 1024**3)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _occupancy(capacity, states, pending) -> HostCapacity:
|
|
79
|
+
visible = {state.run_id for state in states if state.exists}
|
|
80
|
+
unresolved = {key: value for key, value in pending.items() if key not in visible}
|
|
81
|
+
live = [state for state in states if state.exists and state.status != "exited"]
|
|
82
|
+
held = [HostResources.from_labels(state.labels) or capacity for state in live]
|
|
83
|
+
held.extend(unresolved.values())
|
|
84
|
+
return HostCapacity(
|
|
85
|
+
allocatable=capacity,
|
|
86
|
+
used_cpus=sum(value.cpus for value in held),
|
|
87
|
+
used_memory_bytes=sum(value.memory_bytes for value in held),
|
|
88
|
+
live_runs=frozenset(state.run_id for state in live),
|
|
89
|
+
pending_runs=frozenset(unresolved),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@asynccontextmanager
|
|
94
|
+
async def _host_lock(lock_path: Path):
|
|
95
|
+
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
96
|
+
with lock_path.open("a+b") as lock:
|
|
97
|
+
while True:
|
|
98
|
+
try:
|
|
99
|
+
fcntl.flock(lock.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
100
|
+
break
|
|
101
|
+
except BlockingIOError:
|
|
102
|
+
await asyncio.sleep(0.05)
|
|
103
|
+
try:
|
|
104
|
+
yield
|
|
105
|
+
finally:
|
|
106
|
+
fcntl.flock(lock.fileno(), fcntl.LOCK_UN)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
async def read_host_capacity(runtime, capacity: HostResources, *, lock_path: Path) -> HostCapacity:
|
|
110
|
+
"""Read under the launch lock without reconciling or rewriting the ledger."""
|
|
111
|
+
async with _host_lock(lock_path):
|
|
112
|
+
states = await runtime.ps(strict=True)
|
|
113
|
+
return _occupancy(capacity, states, _read_pending(lock_path.with_suffix(".json")))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class HostClaim:
|
|
118
|
+
existing: ContainerState | None = None
|
|
119
|
+
started: bool = False
|
|
120
|
+
|
|
121
|
+
async def run(self, start):
|
|
122
|
+
"""Mark the point after which a failed call may still create a workload."""
|
|
123
|
+
if self.existing is not None:
|
|
124
|
+
raise ValueError("an existing host allocation must not launch another workload")
|
|
125
|
+
self.started = True
|
|
126
|
+
return await start()
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _read_pending(path: Path) -> dict[str, HostResources]:
|
|
130
|
+
try:
|
|
131
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
132
|
+
return {run_id: HostResources(**value) for run_id, value in raw.items()}
|
|
133
|
+
except FileNotFoundError:
|
|
134
|
+
return {}
|
|
135
|
+
except (ValueError, TypeError, AttributeError) as exc:
|
|
136
|
+
raise NoCapacity("host reservation ledger is unreadable; restore it before admitting jobs") from exc
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _write_pending(path: Path, pending: dict[str, HostResources]) -> None:
|
|
140
|
+
payload = {key: {"cpus": value.cpus, "memory_bytes": value.memory_bytes} for key, value in pending.items()}
|
|
141
|
+
with tempfile.NamedTemporaryFile(mode="w", encoding="utf-8", dir=path.parent, delete=False) as out:
|
|
142
|
+
temporary = Path(out.name)
|
|
143
|
+
try:
|
|
144
|
+
json.dump(payload, out)
|
|
145
|
+
out.flush()
|
|
146
|
+
os.fsync(out.fileno())
|
|
147
|
+
os.replace(temporary, path)
|
|
148
|
+
directory = os.open(path.parent, os.O_RDONLY)
|
|
149
|
+
try:
|
|
150
|
+
os.fsync(directory)
|
|
151
|
+
finally:
|
|
152
|
+
os.close(directory)
|
|
153
|
+
finally:
|
|
154
|
+
temporary.unlink(missing_ok=True)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
@asynccontextmanager
|
|
158
|
+
async def claim_host(runtime, capacity: HostResources, request: HostResources, *, run_id: str, lock_path: Path):
|
|
159
|
+
"""Hold admission through container creation; yield a launch claim or an existing run.
|
|
160
|
+
|
|
161
|
+
Unknown live allocations consume the entire host allowance. Unreadable
|
|
162
|
+
inventory fails closed. The caller must put the request labels on its new
|
|
163
|
+
container and must not launch another container when an existing run is yielded.
|
|
164
|
+
"""
|
|
165
|
+
if request.cpus > capacity.cpus or request.memory_bytes > capacity.memory_bytes:
|
|
166
|
+
raise NoCapacity("the job's host-resource request exceeds this host's allocatable capacity")
|
|
167
|
+
pending_path = lock_path.with_suffix(".json")
|
|
168
|
+
async with _host_lock(lock_path):
|
|
169
|
+
states = await runtime.ps(strict=True)
|
|
170
|
+
pending = _read_pending(pending_path)
|
|
171
|
+
# A visible container takes over its durable intent, including when
|
|
172
|
+
# it already exited. An absent one is an uncertain launch, not free capacity.
|
|
173
|
+
for state in states:
|
|
174
|
+
if state.exists:
|
|
175
|
+
pending.pop(state.run_id, None)
|
|
176
|
+
_write_pending(pending_path, pending)
|
|
177
|
+
live = [state for state in states if state.exists and state.status != "exited"]
|
|
178
|
+
existing = next((state for state in live if state.run_id == run_id), None)
|
|
179
|
+
if existing is not None:
|
|
180
|
+
yield HostClaim(existing=existing)
|
|
181
|
+
return
|
|
182
|
+
if run_id in pending:
|
|
183
|
+
raise NoCapacity("this run has an unresolved host reservation; reconcile its launch before retrying")
|
|
184
|
+
occupancy = _occupancy(capacity, states, pending)
|
|
185
|
+
if request.cpus > occupancy.free_cpus or request.memory_bytes > occupancy.free_memory_bytes:
|
|
186
|
+
raise NoCapacity(
|
|
187
|
+
f"host resources unavailable: need {request.cpus} CPUs and {request.memory_bytes} memory bytes; "
|
|
188
|
+
f"free {occupancy.free_cpus} CPUs and {occupancy.free_memory_bytes} memory bytes"
|
|
189
|
+
)
|
|
190
|
+
pending[run_id] = request
|
|
191
|
+
_write_pending(pending_path, pending)
|
|
192
|
+
claim = HostClaim()
|
|
193
|
+
try:
|
|
194
|
+
yield claim
|
|
195
|
+
finally:
|
|
196
|
+
# Failed/timed-out CLI calls can still create a container later.
|
|
197
|
+
# Keep their intent across restarts until a workload is observed.
|
|
198
|
+
try:
|
|
199
|
+
if not claim.started:
|
|
200
|
+
pending.pop(run_id, None)
|
|
201
|
+
for state in await runtime.ps(strict=True):
|
|
202
|
+
if state.exists:
|
|
203
|
+
pending.pop(state.run_id, None)
|
|
204
|
+
_write_pending(pending_path, pending)
|
|
205
|
+
except Exception:
|
|
206
|
+
log.exception("host launch outcome is unknown; retaining durable reservations")
|
|
@@ -162,7 +162,7 @@ class ContainerRuntime(Protocol):
|
|
|
162
162
|
|
|
163
163
|
async def remove(self, name: str) -> bool: ...
|
|
164
164
|
|
|
165
|
-
async def ps(self) -> list[ContainerState]: ...
|
|
165
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]: ...
|
|
166
166
|
|
|
167
167
|
async def health(self) -> dict: ...
|
|
168
168
|
|
|
@@ -237,7 +237,12 @@ class CliRuntime:
|
|
|
237
237
|
# ttlSecondsAfterFinished.
|
|
238
238
|
|
|
239
239
|
args += self._gpu_args(spec.gpus)
|
|
240
|
-
|
|
240
|
+
env = dict(spec.env)
|
|
241
|
+
if not spec.gpus:
|
|
242
|
+
# Override CUDA images that default to exposing every NVIDIA device.
|
|
243
|
+
# Allocation, rather than image or user environment, owns visibility.
|
|
244
|
+
env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
|
|
245
|
+
for k, v in env.items():
|
|
241
246
|
args += ["-e", f"{k}={v}"]
|
|
242
247
|
for k, v in spec.labels.items():
|
|
243
248
|
args += ["--label", f"{k}={v}"]
|
|
@@ -317,7 +322,7 @@ class CliRuntime:
|
|
|
317
322
|
code, _, err = await self._exec("rm", "-f", name, check=False)
|
|
318
323
|
return code == 0 or "no such container" in err.lower()
|
|
319
324
|
|
|
320
|
-
async def ps(self) -> list[ContainerState]:
|
|
325
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]:
|
|
321
326
|
"""List every container this platform owns, exited ones included.
|
|
322
327
|
|
|
323
328
|
`--filter label=` filters on the key alone, so what comes back is the platform's
|
|
@@ -330,6 +335,18 @@ class CliRuntime:
|
|
|
330
335
|
names = [n.strip() for n in out.splitlines() if n.strip()]
|
|
331
336
|
if not names:
|
|
332
337
|
return []
|
|
338
|
+
if strict:
|
|
339
|
+
async def inspect_required(name: str) -> ContainerState:
|
|
340
|
+
_, raw, _ = await self._exec("inspect", name)
|
|
341
|
+
try:
|
|
342
|
+
data = json.loads(raw)[0]
|
|
343
|
+
if not isinstance(data.get("State"), dict) or not isinstance(data.get("Config"), dict):
|
|
344
|
+
raise ValueError("missing container state or configuration")
|
|
345
|
+
return _state_from_inspect(name, data)
|
|
346
|
+
except (ValueError, IndexError, KeyError, TypeError, AttributeError) as exc:
|
|
347
|
+
raise RuntimeError_(f"cannot establish resource occupancy for container {name}") from exc
|
|
348
|
+
|
|
349
|
+
return list(await asyncio.gather(*(inspect_required(name) for name in names)))
|
|
333
350
|
states = await asyncio.gather(
|
|
334
351
|
*(self.inspect(n) for n in names), return_exceptions=True
|
|
335
352
|
)
|
|
@@ -412,6 +429,8 @@ class ProcessRuntime:
|
|
|
412
429
|
|
|
413
430
|
async def run(self, spec: ContainerSpec) -> str:
|
|
414
431
|
self.state_dir.mkdir(parents=True, exist_ok=True)
|
|
432
|
+
if spec.cpu_limit or spec.memory_limit:
|
|
433
|
+
raise RuntimeError_("the process runtime cannot enforce CPU or memory limits")
|
|
415
434
|
env = {**os.environ, **spec.env}
|
|
416
435
|
if spec.gpus:
|
|
417
436
|
env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
|
|
@@ -521,7 +540,9 @@ class ProcessRuntime:
|
|
|
521
540
|
p.unlink(missing_ok=True)
|
|
522
541
|
return True
|
|
523
542
|
|
|
524
|
-
async def ps(self) -> list[ContainerState]:
|
|
543
|
+
async def ps(self, *, strict: bool = False) -> list[ContainerState]:
|
|
544
|
+
if strict:
|
|
545
|
+
raise RuntimeError_("the process runtime cannot provide enforced host-resource reservations")
|
|
525
546
|
if not self.state_dir.is_dir():
|
|
526
547
|
return []
|
|
527
548
|
out = []
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Host admission survives concurrency, restarts and uncertain container creation."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import multiprocessing
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pytest
|
|
10
|
+
|
|
11
|
+
from tuneplane_node.allocator import NoCapacity
|
|
12
|
+
from tuneplane_node.host_resources import HostResources, claim_host
|
|
13
|
+
from tuneplane_node.runtime import LABEL_GPUS, LABEL_RUN_ID, ContainerState
|
|
14
|
+
|
|
15
|
+
CAPACITY = HostResources(4, 8 * 1024**3)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Inventory:
|
|
19
|
+
def __init__(self, states=()):
|
|
20
|
+
self.states = list(states)
|
|
21
|
+
|
|
22
|
+
async def ps(self, *, strict=False):
|
|
23
|
+
assert strict
|
|
24
|
+
return list(self.states)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def state(run_id, *, status="running", labels=None):
|
|
28
|
+
return ContainerState(
|
|
29
|
+
name=f"tuneplane-{run_id}", exists=True, status=status,
|
|
30
|
+
labels={LABEL_RUN_ID: run_id, **(CAPACITY.labels() if labels is None else labels)},
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def claim(runtime, path, run_id="new", request=CAPACITY):
|
|
35
|
+
return claim_host(runtime, CAPACITY, request, run_id=run_id, lock_path=path / "host.lock")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@pytest.mark.parametrize("status", ["running", "created", "paused", "restarting", "unknown"])
|
|
39
|
+
@pytest.mark.parametrize("labels", [None, {}, {"tuneplane.host-cpus": "bad"}, {LABEL_GPUS: "0"}])
|
|
40
|
+
async def test_live_or_unknown_allocations_hold_capacity_after_restart(tmp_path, status, labels):
|
|
41
|
+
runtime = Inventory([state("old", status=status, labels=labels)])
|
|
42
|
+
with pytest.raises(NoCapacity, match="host resources unavailable"):
|
|
43
|
+
async with claim(runtime, tmp_path):
|
|
44
|
+
pytest.fail("occupied host admitted another workload")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
async def test_exit_releases_capacity_and_same_run_is_idempotent(tmp_path):
|
|
48
|
+
runtime = Inventory([state("old")])
|
|
49
|
+
async with claim(runtime, tmp_path, "old") as held:
|
|
50
|
+
assert held.existing.name == "tuneplane-old"
|
|
51
|
+
runtime.states[0].status = "exited"
|
|
52
|
+
async with claim(runtime, tmp_path) as held:
|
|
53
|
+
assert held.existing is None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
async def test_concurrent_allocators_share_the_file_lock(tmp_path):
|
|
57
|
+
runtime = Inventory()
|
|
58
|
+
started, finish = asyncio.Event(), asyncio.Event()
|
|
59
|
+
|
|
60
|
+
async def create():
|
|
61
|
+
started.set()
|
|
62
|
+
await finish.wait()
|
|
63
|
+
runtime.states.append(state("first"))
|
|
64
|
+
|
|
65
|
+
async def first():
|
|
66
|
+
async with claim(runtime, tmp_path, "first") as held:
|
|
67
|
+
await held.run(create)
|
|
68
|
+
|
|
69
|
+
async def second():
|
|
70
|
+
with pytest.raises(NoCapacity):
|
|
71
|
+
async with claim(runtime, tmp_path, "second"):
|
|
72
|
+
pytest.fail("concurrent admission oversold the host")
|
|
73
|
+
|
|
74
|
+
task = asyncio.create_task(first())
|
|
75
|
+
await started.wait()
|
|
76
|
+
waiter = asyncio.create_task(second())
|
|
77
|
+
await asyncio.sleep(0.02)
|
|
78
|
+
assert not waiter.done()
|
|
79
|
+
finish.set()
|
|
80
|
+
await asyncio.gather(task, waiter)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@pytest.mark.parametrize("failure", [TimeoutError, asyncio.CancelledError])
|
|
84
|
+
async def test_uncertain_launch_retains_durable_intent_until_observed(tmp_path, failure):
|
|
85
|
+
runtime = Inventory()
|
|
86
|
+
|
|
87
|
+
async def uncertain():
|
|
88
|
+
raise failure()
|
|
89
|
+
|
|
90
|
+
with pytest.raises(failure):
|
|
91
|
+
async with claim(runtime, tmp_path, "first") as held:
|
|
92
|
+
await held.run(uncertain)
|
|
93
|
+
with pytest.raises(NoCapacity, match="unresolved host reservation"):
|
|
94
|
+
async with claim(Inventory(), tmp_path, "first"):
|
|
95
|
+
pytest.fail("same uncertain launch was retried")
|
|
96
|
+
with pytest.raises(NoCapacity, match="host resources unavailable"):
|
|
97
|
+
async with claim(Inventory(), tmp_path, "second"):
|
|
98
|
+
pytest.fail("restart lost a pending reservation")
|
|
99
|
+
runtime.states.append(state("first", status="exited"))
|
|
100
|
+
async with claim(runtime, tmp_path, "second"):
|
|
101
|
+
pass
|
|
102
|
+
assert json.loads((tmp_path / "host.json").read_text(encoding="utf-8")) == {}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
async def test_failure_before_runtime_launch_releases_intent(tmp_path):
|
|
106
|
+
runtime = Inventory()
|
|
107
|
+
with pytest.raises(NoCapacity, match="GPU"):
|
|
108
|
+
async with claim(runtime, tmp_path):
|
|
109
|
+
raise NoCapacity("GPU shortage")
|
|
110
|
+
async with claim(runtime, tmp_path):
|
|
111
|
+
pass
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
async def test_corrupt_ledger_blocks_admission(tmp_path):
|
|
115
|
+
(tmp_path / "host.json").write_text("broken", encoding="utf-8")
|
|
116
|
+
with pytest.raises(NoCapacity, match="ledger is unreadable"):
|
|
117
|
+
async with claim(Inventory(), tmp_path):
|
|
118
|
+
pytest.fail("corrupt ledger admitted a workload")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
async def test_memory_is_admitted_independently_of_cpu(tmp_path):
|
|
122
|
+
runtime = Inventory([state("old", labels=HostResources(1, 7 * 1024**3).labels())])
|
|
123
|
+
with pytest.raises(NoCapacity):
|
|
124
|
+
async with claim(runtime, tmp_path, request=HostResources(1, 2 * 1024**3)):
|
|
125
|
+
pytest.fail("memory was oversold despite spare CPUs")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _process_claim(path, ready, outcomes, run_id):
|
|
129
|
+
"""Independent processes must see the same durable uncertain-launch intent."""
|
|
130
|
+
async def launch():
|
|
131
|
+
async with claim(Inventory(), Path(path), run_id) as held:
|
|
132
|
+
async def unobserved():
|
|
133
|
+
return "runtime accepted the launch"
|
|
134
|
+
await held.run(unobserved)
|
|
135
|
+
ready.wait(10)
|
|
136
|
+
try:
|
|
137
|
+
asyncio.run(launch())
|
|
138
|
+
outcomes.put("admitted")
|
|
139
|
+
except NoCapacity:
|
|
140
|
+
outcomes.put("blocked")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def test_separate_processes_cannot_oversell(tmp_path):
|
|
144
|
+
context = multiprocessing.get_context("spawn")
|
|
145
|
+
ready, outcomes = context.Event(), context.Queue()
|
|
146
|
+
processes = [context.Process(target=_process_claim, args=(str(tmp_path), ready, outcomes, name))
|
|
147
|
+
for name in ("first", "second")]
|
|
148
|
+
try:
|
|
149
|
+
for process in processes:
|
|
150
|
+
process.start()
|
|
151
|
+
ready.set()
|
|
152
|
+
assert sorted(outcomes.get(timeout=20) for _ in processes) == ["admitted", "blocked"]
|
|
153
|
+
for process in processes:
|
|
154
|
+
process.join(timeout=10)
|
|
155
|
+
assert process.exitcode == 0
|
|
156
|
+
finally:
|
|
157
|
+
for process in processes:
|
|
158
|
+
if process.is_alive():
|
|
159
|
+
process.terminate()
|
|
160
|
+
process.join(timeout=5)
|
|
161
|
+
outcomes.close()
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
async def test_inventory_failure_blocks_without_releasing_a_pending_launch(tmp_path):
|
|
165
|
+
class Unavailable(Inventory):
|
|
166
|
+
async def ps(self, *, strict=False):
|
|
167
|
+
raise RuntimeError("runtime unavailable")
|
|
168
|
+
|
|
169
|
+
ledger = tmp_path / "host.json"
|
|
170
|
+
original = json.dumps({"old": {"cpus": CAPACITY.cpus, "memory_bytes": CAPACITY.memory_bytes}})
|
|
171
|
+
ledger.write_text(original, encoding="utf-8")
|
|
172
|
+
with pytest.raises(RuntimeError, match="runtime unavailable"):
|
|
173
|
+
async with claim(Unavailable(), tmp_path):
|
|
174
|
+
pytest.fail("unreadable inventory admitted a job")
|
|
175
|
+
assert ledger.read_text(encoding="utf-8") == original
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
async def test_cancelling_a_lock_waiter_does_not_release_the_active_claim(tmp_path):
|
|
179
|
+
runtime = Inventory()
|
|
180
|
+
async with claim(runtime, tmp_path, "owner"):
|
|
181
|
+
async def wait():
|
|
182
|
+
async with claim(runtime, tmp_path, "waiter"):
|
|
183
|
+
pytest.fail("waiter entered a held lock")
|
|
184
|
+
waiter = asyncio.create_task(wait())
|
|
185
|
+
await asyncio.sleep(0.02)
|
|
186
|
+
waiter.cancel()
|
|
187
|
+
with pytest.raises(asyncio.CancelledError):
|
|
188
|
+
await waiter
|
|
189
|
+
ledger = json.loads((tmp_path / "host.json").read_text(encoding="utf-8"))
|
|
190
|
+
assert list(ledger) == ["owner"]
|
|
191
|
+
async with claim(runtime, tmp_path, "next"):
|
|
192
|
+
pass
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
async def test_capacity_read_includes_pending_and_does_not_rewrite_ledger(tmp_path):
|
|
196
|
+
from tuneplane_node.host_resources import read_host_capacity
|
|
197
|
+
|
|
198
|
+
path = tmp_path / "host.json"
|
|
199
|
+
original = json.dumps({
|
|
200
|
+
"visible": {"cpus": 1, "memory_bytes": 1024**3},
|
|
201
|
+
"pending": {"cpus": 2, "memory_bytes": 2 * 1024**3},
|
|
202
|
+
})
|
|
203
|
+
path.write_text(original, encoding="utf-8")
|
|
204
|
+
runtime = Inventory([state("visible", labels=HostResources(1, 1024**3).labels())])
|
|
205
|
+
capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
|
|
206
|
+
assert capacity.used_cpus == 3 and capacity.used_memory_bytes == 3 * 1024**3
|
|
207
|
+
assert capacity.free_cpus == 1
|
|
208
|
+
assert capacity.live_runs == {"visible"} and capacity.pending_runs == {"pending"}
|
|
209
|
+
assert path.read_text(encoding="utf-8") == original
|
|
210
|
+
runtime.states[0].status = "exited"
|
|
211
|
+
capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
|
|
212
|
+
assert capacity.used_cpus == 2
|
|
213
|
+
assert path.read_text(encoding="utf-8") == original
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
async def test_empty_capacity_read_does_not_create_reservation_ledger(tmp_path):
|
|
217
|
+
from tuneplane_node.host_resources import read_host_capacity
|
|
218
|
+
|
|
219
|
+
capacity = await read_host_capacity(Inventory(), CAPACITY, lock_path=tmp_path / "host.lock")
|
|
220
|
+
assert capacity.free_cpus == CAPACITY.cpus
|
|
221
|
+
assert not (tmp_path / "host.json").exists()
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Real Linux Docker acceptance; CI uses a dedicated daemon without GPU devices."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import uuid
|
|
8
|
+
|
|
9
|
+
import pytest
|
|
10
|
+
|
|
11
|
+
from tuneplane_node.allocator import NoCapacity
|
|
12
|
+
from tuneplane_node.host_resources import HostResources, claim_host
|
|
13
|
+
from tuneplane_node.runtime import LABEL_RUN_ID, CliRuntime, ContainerSpec
|
|
14
|
+
|
|
15
|
+
pytestmark = pytest.mark.skipif(
|
|
16
|
+
os.environ.get("TUNEPLANE_TEST_HOST_DOCKER") != "1",
|
|
17
|
+
reason="requires an explicitly enabled dedicated Linux Docker daemon",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
async def test_real_docker_host_admission_limits_restart_and_release(tmp_path):
|
|
22
|
+
runtime = CliRuntime("docker", timeout=30)
|
|
23
|
+
image = "alpine:3.21"
|
|
24
|
+
await runtime.pull(image, timeout=90)
|
|
25
|
+
budget = HostResources(1, 128 * 1024**2)
|
|
26
|
+
names = [f"tuneplane-host-test-{uuid.uuid4().hex}" for _ in range(2)]
|
|
27
|
+
lock_path = tmp_path / "host.lock"
|
|
28
|
+
|
|
29
|
+
async def launch(name, runtime):
|
|
30
|
+
async with claim_host(runtime, budget, budget, run_id=name, lock_path=lock_path) as held:
|
|
31
|
+
spec = ContainerSpec(
|
|
32
|
+
name=name, image=image, command=["sleep", "60"],
|
|
33
|
+
labels={LABEL_RUN_ID: name, **budget.labels()},
|
|
34
|
+
cpu_limit="1", memory_limit=str(budget.memory_bytes),
|
|
35
|
+
)
|
|
36
|
+
await held.run(lambda: runtime.run(spec))
|
|
37
|
+
return name
|
|
38
|
+
|
|
39
|
+
try:
|
|
40
|
+
outcomes = await asyncio.gather(*(launch(name, runtime) for name in names), return_exceptions=True)
|
|
41
|
+
assert sum(isinstance(outcome, NoCapacity) for outcome in outcomes) == 1
|
|
42
|
+
running = next(outcome for outcome in outcomes if isinstance(outcome, str))
|
|
43
|
+
waiting = next(name for name in names if name != running)
|
|
44
|
+
_, raw, _ = await runtime._exec("inspect", running)
|
|
45
|
+
info = json.loads(raw)[0]
|
|
46
|
+
assert info["HostConfig"]["NanoCpus"] == 1_000_000_000
|
|
47
|
+
assert info["HostConfig"]["Memory"] == budget.memory_bytes
|
|
48
|
+
assert not info["HostConfig"].get("DeviceRequests")
|
|
49
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" in info["Config"]["Env"]
|
|
50
|
+
_, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/memory.max")
|
|
51
|
+
assert int(raw.strip()) == budget.memory_bytes
|
|
52
|
+
_, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/cpu.max")
|
|
53
|
+
quota, period = map(int, raw.split())
|
|
54
|
+
assert quota == period
|
|
55
|
+
restarted = CliRuntime("docker", timeout=30)
|
|
56
|
+
with pytest.raises(NoCapacity):
|
|
57
|
+
await launch(waiting, restarted)
|
|
58
|
+
assert await runtime.stop(running, timeout=1)
|
|
59
|
+
assert await launch(waiting, restarted) == waiting
|
|
60
|
+
finally:
|
|
61
|
+
for name in names:
|
|
62
|
+
await runtime.remove(name)
|
|
@@ -166,6 +166,39 @@ def test_no_cards_means_no_gpu_flag_at_all():
|
|
|
166
166
|
# ── run ────────────────────────────────────────────────────────────────────
|
|
167
167
|
|
|
168
168
|
|
|
169
|
+
@pytest.mark.parametrize("binary", ["docker", "podman"])
|
|
170
|
+
@pytest.mark.parametrize("env", [{}, {"NVIDIA_VISIBLE_DEVICES": "all", "CUDA_VISIBLE_DEVICES": "0"}])
|
|
171
|
+
async def test_zero_gpu_containers_override_image_and_user_visibility(binary, env):
|
|
172
|
+
runtime = RecordingCli(binary=binary)
|
|
173
|
+
spec = _spec(env=env)
|
|
174
|
+
original = dict(spec.env)
|
|
175
|
+
await runtime.run(spec)
|
|
176
|
+
argv = _argv(runtime, "run")
|
|
177
|
+
assert "--gpus" not in argv and "--device" not in argv
|
|
178
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" in argv
|
|
179
|
+
assert "CUDA_VISIBLE_DEVICES=" in argv
|
|
180
|
+
assert "NVIDIA_VISIBLE_DEVICES=all" not in argv
|
|
181
|
+
assert "CUDA_VISIBLE_DEVICES=0" not in argv
|
|
182
|
+
assert spec.env == original
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
async def test_gpu_containers_preserve_their_visibility_environment():
|
|
186
|
+
runtime = RecordingCli()
|
|
187
|
+
await runtime.run(_spec(gpus=[2], env={"CUDA_VISIBLE_DEVICES": "0"}))
|
|
188
|
+
argv = _argv(runtime, "run")
|
|
189
|
+
assert "--gpus" in argv
|
|
190
|
+
assert "CUDA_VISIBLE_DEVICES=0" in argv
|
|
191
|
+
assert "NVIDIA_VISIBLE_DEVICES=void" not in argv
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
@pytest.mark.parametrize("limits", [{"cpu_limit": "4"}, {"memory_limit": "8g"}])
|
|
195
|
+
async def test_process_runtime_refuses_limits_it_cannot_enforce(tmp_path, limits):
|
|
196
|
+
runtime = ProcessRuntime(tmp_path / "state")
|
|
197
|
+
with pytest.raises(RuntimeError_, match="cannot enforce CPU or memory"):
|
|
198
|
+
await runtime.run(_spec(**limits))
|
|
199
|
+
assert not (tmp_path / "state" / "tuneplane-run-a.json").exists()
|
|
200
|
+
|
|
201
|
+
|
|
169
202
|
async def test_the_command_replaces_the_images_entrypoint_rather_than_appending():
|
|
170
203
|
"""A Playground session on `vllm/vllm-openai` died at once with
|
|
171
204
|
`vllm: error: unrecognized arguments: -lc python3 -m vllm...` -- our argv
|
|
@@ -608,3 +641,29 @@ def test_a_container_that_finished_in_the_future_is_zero_seconds_old():
|
|
|
608
641
|
"""A clock skew between the daemon and the console must not produce a
|
|
609
642
|
negative age, which would compare as younger than every threshold."""
|
|
610
643
|
assert age_seconds("2026-09-12T10:00:00+00:00", 0.0) == 0.0
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
@pytest.mark.parametrize("payload", ["not json", "[]", '[{"State": {}}]', '[{"Config": {}}]'])
|
|
647
|
+
async def test_strict_inventory_refuses_unreadable_occupancy(payload):
|
|
648
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (0, payload, "")})
|
|
649
|
+
with pytest.raises(RuntimeError_, match="cannot establish resource occupancy"):
|
|
650
|
+
await runtime.ps(strict=True)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
async def test_strict_inventory_propagates_inspection_failure():
|
|
654
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (1, "", "unavailable")})
|
|
655
|
+
with pytest.raises(RuntimeError_, match="inspect failed"):
|
|
656
|
+
await runtime.ps(strict=True)
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
async def test_strict_inventory_preserves_labels():
|
|
660
|
+
runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""),
|
|
661
|
+
"inspect": (0, _inspect_payload(), "")})
|
|
662
|
+
states = await runtime.ps(strict=True)
|
|
663
|
+
assert len(states) == 1 and states[0].exists
|
|
664
|
+
assert states[0].labels == (await runtime.inspect("tuneplane-a")).labels
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
async def test_process_runtime_refuses_host_admission(tmp_path):
|
|
668
|
+
with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
|
|
669
|
+
await ProcessRuntime(tmp_path).ps(strict=True)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|