tuneplane-node 0.3.15__tar.gz → 0.3.16__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,6 +10,7 @@ venv/
10
10
  .python-version
11
11
  .pytest_cache/
12
12
  .ruff_cache/
13
+ server/native/target/
13
14
 
14
15
  # Frontend
15
16
  web/node_modules/
@@ -57,3 +58,7 @@ outputs/
57
58
 
58
59
  # Agent worktrees, created by the harness and never part of a commit.
59
60
  .claude/worktrees/
61
+
62
+ # Independent public website build
63
+ web/dist-website/
64
+ :memory:.ses
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuneplane-node
3
- Version: 0.3.15
3
+ Version: 0.3.16
4
4
  Summary: TunePlane node daemon: runs job containers on one machine and reports it to a console.
5
5
  License-Expression: AGPL-3.0-only
6
6
  License-File: LICENSE
@@ -10,7 +10,7 @@
10
10
  # the local backend runs the same container runtime and allocator on its own host.
11
11
  [project]
12
12
  name = "tuneplane-node"
13
- version = "0.3.15"
13
+ version = "0.3.16"
14
14
  description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
15
15
  requires-python = ">=3.12"
16
16
  # Same licence as the control plane: this is infrastructure an organization runs,
@@ -0,0 +1,206 @@
1
+ """Host admission reconstructed from container labels under a shared file lock.
2
+
3
+ The lock file must be shared by every allocator of the same local runtime and
4
+ must never be unlinked. Container lifetime is reservation lifetime: a restart
5
+ reads the same labels, and an exited container no longer holds CPU or memory.
6
+ Before a container becomes observable, a durable intent covers uncertain startup.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ import fcntl
12
+ import json
13
+ import logging
14
+ import os
15
+ import tempfile
16
+ from contextlib import asynccontextmanager
17
+ from dataclasses import dataclass
18
+ from pathlib import Path
19
+
20
+ from tuneplane_node.allocator import NoCapacity
21
+ from tuneplane_node.runtime import ContainerState
22
+
23
+ LABEL_CPUS = "tuneplane.host-cpus"
24
+ LABEL_MEMORY = "tuneplane.host-memory-bytes"
25
+ log = logging.getLogger(__name__)
26
+
27
+
28
+ @dataclass(frozen=True)
29
+ class HostResources:
30
+ cpus: int
31
+ memory_bytes: int
32
+
33
+ def __post_init__(self):
34
+ for value in (self.cpus, self.memory_bytes):
35
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
36
+ raise ValueError("host CPU and memory resources must be positive integers")
37
+
38
+ def labels(self) -> dict[str, str]:
39
+ return {LABEL_CPUS: str(self.cpus), LABEL_MEMORY: str(self.memory_bytes)}
40
+
41
+ @classmethod
42
+ def from_labels(cls, labels: dict[str, str]) -> HostResources | None:
43
+ try:
44
+ return cls(int(labels[LABEL_CPUS]), int(labels[LABEL_MEMORY]))
45
+ except (KeyError, ValueError, TypeError):
46
+ return None
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class HostCapacity:
51
+ """One host's observed occupancy, including unresolved launch intents."""
52
+
53
+ allocatable: HostResources
54
+ used_cpus: int
55
+ used_memory_bytes: int
56
+ live_runs: frozenset[str]
57
+ pending_runs: frozenset[str]
58
+
59
+ @property
60
+ def free_cpus(self) -> int:
61
+ return max(0, self.allocatable.cpus - self.used_cpus)
62
+
63
+ @property
64
+ def free_memory_bytes(self) -> int:
65
+ return max(0, self.allocatable.memory_bytes - self.used_memory_bytes)
66
+
67
+
68
+ def single_host_request(pools) -> HostResources:
69
+ """The supported local shape, shared by scheduling and container launch."""
70
+ if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
71
+ raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
72
+ pool = pools[0]
73
+ if pool.scratch_gb:
74
+ raise ValueError("the local backend does not yet enforce per-job scratch space")
75
+ return HostResources(pool.cpus, pool.memory_gb * 1024**3)
76
+
77
+
78
+ def _occupancy(capacity, states, pending) -> HostCapacity:
79
+ visible = {state.run_id for state in states if state.exists}
80
+ unresolved = {key: value for key, value in pending.items() if key not in visible}
81
+ live = [state for state in states if state.exists and state.status != "exited"]
82
+ held = [HostResources.from_labels(state.labels) or capacity for state in live]
83
+ held.extend(unresolved.values())
84
+ return HostCapacity(
85
+ allocatable=capacity,
86
+ used_cpus=sum(value.cpus for value in held),
87
+ used_memory_bytes=sum(value.memory_bytes for value in held),
88
+ live_runs=frozenset(state.run_id for state in live),
89
+ pending_runs=frozenset(unresolved),
90
+ )
91
+
92
+
93
+ @asynccontextmanager
94
+ async def _host_lock(lock_path: Path):
95
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
96
+ with lock_path.open("a+b") as lock:
97
+ while True:
98
+ try:
99
+ fcntl.flock(lock.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
100
+ break
101
+ except BlockingIOError:
102
+ await asyncio.sleep(0.05)
103
+ try:
104
+ yield
105
+ finally:
106
+ fcntl.flock(lock.fileno(), fcntl.LOCK_UN)
107
+
108
+
109
+ async def read_host_capacity(runtime, capacity: HostResources, *, lock_path: Path) -> HostCapacity:
110
+ """Read under the launch lock without reconciling or rewriting the ledger."""
111
+ async with _host_lock(lock_path):
112
+ states = await runtime.ps(strict=True)
113
+ return _occupancy(capacity, states, _read_pending(lock_path.with_suffix(".json")))
114
+
115
+
116
+ @dataclass
117
+ class HostClaim:
118
+ existing: ContainerState | None = None
119
+ started: bool = False
120
+
121
+ async def run(self, start):
122
+ """Mark the point after which a failed call may still create a workload."""
123
+ if self.existing is not None:
124
+ raise ValueError("an existing host allocation must not launch another workload")
125
+ self.started = True
126
+ return await start()
127
+
128
+
129
+ def _read_pending(path: Path) -> dict[str, HostResources]:
130
+ try:
131
+ raw = json.loads(path.read_text(encoding="utf-8"))
132
+ return {run_id: HostResources(**value) for run_id, value in raw.items()}
133
+ except FileNotFoundError:
134
+ return {}
135
+ except (ValueError, TypeError, AttributeError) as exc:
136
+ raise NoCapacity("host reservation ledger is unreadable; restore it before admitting jobs") from exc
137
+
138
+
139
+ def _write_pending(path: Path, pending: dict[str, HostResources]) -> None:
140
+ payload = {key: {"cpus": value.cpus, "memory_bytes": value.memory_bytes} for key, value in pending.items()}
141
+ with tempfile.NamedTemporaryFile(mode="w", encoding="utf-8", dir=path.parent, delete=False) as out:
142
+ temporary = Path(out.name)
143
+ try:
144
+ json.dump(payload, out)
145
+ out.flush()
146
+ os.fsync(out.fileno())
147
+ os.replace(temporary, path)
148
+ directory = os.open(path.parent, os.O_RDONLY)
149
+ try:
150
+ os.fsync(directory)
151
+ finally:
152
+ os.close(directory)
153
+ finally:
154
+ temporary.unlink(missing_ok=True)
155
+
156
+
157
+ @asynccontextmanager
158
+ async def claim_host(runtime, capacity: HostResources, request: HostResources, *, run_id: str, lock_path: Path):
159
+ """Hold admission through container creation; yield a launch claim or an existing run.
160
+
161
+ Unknown live allocations consume the entire host allowance. Unreadable
162
+ inventory fails closed. The caller must put the request labels on its new
163
+ container and must not launch another container when an existing run is yielded.
164
+ """
165
+ if request.cpus > capacity.cpus or request.memory_bytes > capacity.memory_bytes:
166
+ raise NoCapacity("the job's host-resource request exceeds this host's allocatable capacity")
167
+ pending_path = lock_path.with_suffix(".json")
168
+ async with _host_lock(lock_path):
169
+ states = await runtime.ps(strict=True)
170
+ pending = _read_pending(pending_path)
171
+ # A visible container takes over its durable intent, including when
172
+ # it already exited. An absent one is an uncertain launch, not free capacity.
173
+ for state in states:
174
+ if state.exists:
175
+ pending.pop(state.run_id, None)
176
+ _write_pending(pending_path, pending)
177
+ live = [state for state in states if state.exists and state.status != "exited"]
178
+ existing = next((state for state in live if state.run_id == run_id), None)
179
+ if existing is not None:
180
+ yield HostClaim(existing=existing)
181
+ return
182
+ if run_id in pending:
183
+ raise NoCapacity("this run has an unresolved host reservation; reconcile its launch before retrying")
184
+ occupancy = _occupancy(capacity, states, pending)
185
+ if request.cpus > occupancy.free_cpus or request.memory_bytes > occupancy.free_memory_bytes:
186
+ raise NoCapacity(
187
+ f"host resources unavailable: need {request.cpus} CPUs and {request.memory_bytes} memory bytes; "
188
+ f"free {occupancy.free_cpus} CPUs and {occupancy.free_memory_bytes} memory bytes"
189
+ )
190
+ pending[run_id] = request
191
+ _write_pending(pending_path, pending)
192
+ claim = HostClaim()
193
+ try:
194
+ yield claim
195
+ finally:
196
+ # Failed/timed-out CLI calls can still create a container later.
197
+ # Keep their intent across restarts until a workload is observed.
198
+ try:
199
+ if not claim.started:
200
+ pending.pop(run_id, None)
201
+ for state in await runtime.ps(strict=True):
202
+ if state.exists:
203
+ pending.pop(state.run_id, None)
204
+ _write_pending(pending_path, pending)
205
+ except Exception:
206
+ log.exception("host launch outcome is unknown; retaining durable reservations")
@@ -162,7 +162,7 @@ class ContainerRuntime(Protocol):
162
162
 
163
163
  async def remove(self, name: str) -> bool: ...
164
164
 
165
- async def ps(self) -> list[ContainerState]: ...
165
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]: ...
166
166
 
167
167
  async def health(self) -> dict: ...
168
168
 
@@ -237,7 +237,12 @@ class CliRuntime:
237
237
  # ttlSecondsAfterFinished.
238
238
 
239
239
  args += self._gpu_args(spec.gpus)
240
- for k, v in spec.env.items():
240
+ env = dict(spec.env)
241
+ if not spec.gpus:
242
+ # Override CUDA images that default to exposing every NVIDIA device.
243
+ # Allocation, rather than image or user environment, owns visibility.
244
+ env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
245
+ for k, v in env.items():
241
246
  args += ["-e", f"{k}={v}"]
242
247
  for k, v in spec.labels.items():
243
248
  args += ["--label", f"{k}={v}"]
@@ -317,7 +322,7 @@ class CliRuntime:
317
322
  code, _, err = await self._exec("rm", "-f", name, check=False)
318
323
  return code == 0 or "no such container" in err.lower()
319
324
 
320
- async def ps(self) -> list[ContainerState]:
325
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]:
321
326
  """List every container this platform owns, exited ones included.
322
327
 
323
328
  `--filter label=` filters on the key alone, so what comes back is the platform's
@@ -330,6 +335,18 @@ class CliRuntime:
330
335
  names = [n.strip() for n in out.splitlines() if n.strip()]
331
336
  if not names:
332
337
  return []
338
+ if strict:
339
+ async def inspect_required(name: str) -> ContainerState:
340
+ _, raw, _ = await self._exec("inspect", name)
341
+ try:
342
+ data = json.loads(raw)[0]
343
+ if not isinstance(data.get("State"), dict) or not isinstance(data.get("Config"), dict):
344
+ raise ValueError("missing container state or configuration")
345
+ return _state_from_inspect(name, data)
346
+ except (ValueError, IndexError, KeyError, TypeError, AttributeError) as exc:
347
+ raise RuntimeError_(f"cannot establish resource occupancy for container {name}") from exc
348
+
349
+ return list(await asyncio.gather(*(inspect_required(name) for name in names)))
333
350
  states = await asyncio.gather(
334
351
  *(self.inspect(n) for n in names), return_exceptions=True
335
352
  )
@@ -412,6 +429,8 @@ class ProcessRuntime:
412
429
 
413
430
  async def run(self, spec: ContainerSpec) -> str:
414
431
  self.state_dir.mkdir(parents=True, exist_ok=True)
432
+ if spec.cpu_limit or spec.memory_limit:
433
+ raise RuntimeError_("the process runtime cannot enforce CPU or memory limits")
415
434
  env = {**os.environ, **spec.env}
416
435
  if spec.gpus:
417
436
  env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
@@ -521,7 +540,9 @@ class ProcessRuntime:
521
540
  p.unlink(missing_ok=True)
522
541
  return True
523
542
 
524
- async def ps(self) -> list[ContainerState]:
543
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]:
544
+ if strict:
545
+ raise RuntimeError_("the process runtime cannot provide enforced host-resource reservations")
525
546
  if not self.state_dir.is_dir():
526
547
  return []
527
548
  out = []
@@ -0,0 +1,221 @@
1
+ """Host admission survives concurrency, restarts and uncertain container creation."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import json
6
+ import multiprocessing
7
+ from pathlib import Path
8
+
9
+ import pytest
10
+
11
+ from tuneplane_node.allocator import NoCapacity
12
+ from tuneplane_node.host_resources import HostResources, claim_host
13
+ from tuneplane_node.runtime import LABEL_GPUS, LABEL_RUN_ID, ContainerState
14
+
15
+ CAPACITY = HostResources(4, 8 * 1024**3)
16
+
17
+
18
+ class Inventory:
19
+ def __init__(self, states=()):
20
+ self.states = list(states)
21
+
22
+ async def ps(self, *, strict=False):
23
+ assert strict
24
+ return list(self.states)
25
+
26
+
27
+ def state(run_id, *, status="running", labels=None):
28
+ return ContainerState(
29
+ name=f"tuneplane-{run_id}", exists=True, status=status,
30
+ labels={LABEL_RUN_ID: run_id, **(CAPACITY.labels() if labels is None else labels)},
31
+ )
32
+
33
+
34
+ def claim(runtime, path, run_id="new", request=CAPACITY):
35
+ return claim_host(runtime, CAPACITY, request, run_id=run_id, lock_path=path / "host.lock")
36
+
37
+
38
+ @pytest.mark.parametrize("status", ["running", "created", "paused", "restarting", "unknown"])
39
+ @pytest.mark.parametrize("labels", [None, {}, {"tuneplane.host-cpus": "bad"}, {LABEL_GPUS: "0"}])
40
+ async def test_live_or_unknown_allocations_hold_capacity_after_restart(tmp_path, status, labels):
41
+ runtime = Inventory([state("old", status=status, labels=labels)])
42
+ with pytest.raises(NoCapacity, match="host resources unavailable"):
43
+ async with claim(runtime, tmp_path):
44
+ pytest.fail("occupied host admitted another workload")
45
+
46
+
47
+ async def test_exit_releases_capacity_and_same_run_is_idempotent(tmp_path):
48
+ runtime = Inventory([state("old")])
49
+ async with claim(runtime, tmp_path, "old") as held:
50
+ assert held.existing.name == "tuneplane-old"
51
+ runtime.states[0].status = "exited"
52
+ async with claim(runtime, tmp_path) as held:
53
+ assert held.existing is None
54
+
55
+
56
+ async def test_concurrent_allocators_share_the_file_lock(tmp_path):
57
+ runtime = Inventory()
58
+ started, finish = asyncio.Event(), asyncio.Event()
59
+
60
+ async def create():
61
+ started.set()
62
+ await finish.wait()
63
+ runtime.states.append(state("first"))
64
+
65
+ async def first():
66
+ async with claim(runtime, tmp_path, "first") as held:
67
+ await held.run(create)
68
+
69
+ async def second():
70
+ with pytest.raises(NoCapacity):
71
+ async with claim(runtime, tmp_path, "second"):
72
+ pytest.fail("concurrent admission oversold the host")
73
+
74
+ task = asyncio.create_task(first())
75
+ await started.wait()
76
+ waiter = asyncio.create_task(second())
77
+ await asyncio.sleep(0.02)
78
+ assert not waiter.done()
79
+ finish.set()
80
+ await asyncio.gather(task, waiter)
81
+
82
+
83
+ @pytest.mark.parametrize("failure", [TimeoutError, asyncio.CancelledError])
84
+ async def test_uncertain_launch_retains_durable_intent_until_observed(tmp_path, failure):
85
+ runtime = Inventory()
86
+
87
+ async def uncertain():
88
+ raise failure()
89
+
90
+ with pytest.raises(failure):
91
+ async with claim(runtime, tmp_path, "first") as held:
92
+ await held.run(uncertain)
93
+ with pytest.raises(NoCapacity, match="unresolved host reservation"):
94
+ async with claim(Inventory(), tmp_path, "first"):
95
+ pytest.fail("same uncertain launch was retried")
96
+ with pytest.raises(NoCapacity, match="host resources unavailable"):
97
+ async with claim(Inventory(), tmp_path, "second"):
98
+ pytest.fail("restart lost a pending reservation")
99
+ runtime.states.append(state("first", status="exited"))
100
+ async with claim(runtime, tmp_path, "second"):
101
+ pass
102
+ assert json.loads((tmp_path / "host.json").read_text(encoding="utf-8")) == {}
103
+
104
+
105
+ async def test_failure_before_runtime_launch_releases_intent(tmp_path):
106
+ runtime = Inventory()
107
+ with pytest.raises(NoCapacity, match="GPU"):
108
+ async with claim(runtime, tmp_path):
109
+ raise NoCapacity("GPU shortage")
110
+ async with claim(runtime, tmp_path):
111
+ pass
112
+
113
+
114
+ async def test_corrupt_ledger_blocks_admission(tmp_path):
115
+ (tmp_path / "host.json").write_text("broken", encoding="utf-8")
116
+ with pytest.raises(NoCapacity, match="ledger is unreadable"):
117
+ async with claim(Inventory(), tmp_path):
118
+ pytest.fail("corrupt ledger admitted a workload")
119
+
120
+
121
+ async def test_memory_is_admitted_independently_of_cpu(tmp_path):
122
+ runtime = Inventory([state("old", labels=HostResources(1, 7 * 1024**3).labels())])
123
+ with pytest.raises(NoCapacity):
124
+ async with claim(runtime, tmp_path, request=HostResources(1, 2 * 1024**3)):
125
+ pytest.fail("memory was oversold despite spare CPUs")
126
+
127
+
128
+ def _process_claim(path, ready, outcomes, run_id):
129
+ """Independent processes must see the same durable uncertain-launch intent."""
130
+ async def launch():
131
+ async with claim(Inventory(), Path(path), run_id) as held:
132
+ async def unobserved():
133
+ return "runtime accepted the launch"
134
+ await held.run(unobserved)
135
+ ready.wait(10)
136
+ try:
137
+ asyncio.run(launch())
138
+ outcomes.put("admitted")
139
+ except NoCapacity:
140
+ outcomes.put("blocked")
141
+
142
+
143
+ def test_separate_processes_cannot_oversell(tmp_path):
144
+ context = multiprocessing.get_context("spawn")
145
+ ready, outcomes = context.Event(), context.Queue()
146
+ processes = [context.Process(target=_process_claim, args=(str(tmp_path), ready, outcomes, name))
147
+ for name in ("first", "second")]
148
+ try:
149
+ for process in processes:
150
+ process.start()
151
+ ready.set()
152
+ assert sorted(outcomes.get(timeout=20) for _ in processes) == ["admitted", "blocked"]
153
+ for process in processes:
154
+ process.join(timeout=10)
155
+ assert process.exitcode == 0
156
+ finally:
157
+ for process in processes:
158
+ if process.is_alive():
159
+ process.terminate()
160
+ process.join(timeout=5)
161
+ outcomes.close()
162
+
163
+
164
+ async def test_inventory_failure_blocks_without_releasing_a_pending_launch(tmp_path):
165
+ class Unavailable(Inventory):
166
+ async def ps(self, *, strict=False):
167
+ raise RuntimeError("runtime unavailable")
168
+
169
+ ledger = tmp_path / "host.json"
170
+ original = json.dumps({"old": {"cpus": CAPACITY.cpus, "memory_bytes": CAPACITY.memory_bytes}})
171
+ ledger.write_text(original, encoding="utf-8")
172
+ with pytest.raises(RuntimeError, match="runtime unavailable"):
173
+ async with claim(Unavailable(), tmp_path):
174
+ pytest.fail("unreadable inventory admitted a job")
175
+ assert ledger.read_text(encoding="utf-8") == original
176
+
177
+
178
+ async def test_cancelling_a_lock_waiter_does_not_release_the_active_claim(tmp_path):
179
+ runtime = Inventory()
180
+ async with claim(runtime, tmp_path, "owner"):
181
+ async def wait():
182
+ async with claim(runtime, tmp_path, "waiter"):
183
+ pytest.fail("waiter entered a held lock")
184
+ waiter = asyncio.create_task(wait())
185
+ await asyncio.sleep(0.02)
186
+ waiter.cancel()
187
+ with pytest.raises(asyncio.CancelledError):
188
+ await waiter
189
+ ledger = json.loads((tmp_path / "host.json").read_text(encoding="utf-8"))
190
+ assert list(ledger) == ["owner"]
191
+ async with claim(runtime, tmp_path, "next"):
192
+ pass
193
+
194
+
195
+ async def test_capacity_read_includes_pending_and_does_not_rewrite_ledger(tmp_path):
196
+ from tuneplane_node.host_resources import read_host_capacity
197
+
198
+ path = tmp_path / "host.json"
199
+ original = json.dumps({
200
+ "visible": {"cpus": 1, "memory_bytes": 1024**3},
201
+ "pending": {"cpus": 2, "memory_bytes": 2 * 1024**3},
202
+ })
203
+ path.write_text(original, encoding="utf-8")
204
+ runtime = Inventory([state("visible", labels=HostResources(1, 1024**3).labels())])
205
+ capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
206
+ assert capacity.used_cpus == 3 and capacity.used_memory_bytes == 3 * 1024**3
207
+ assert capacity.free_cpus == 1
208
+ assert capacity.live_runs == {"visible"} and capacity.pending_runs == {"pending"}
209
+ assert path.read_text(encoding="utf-8") == original
210
+ runtime.states[0].status = "exited"
211
+ capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
212
+ assert capacity.used_cpus == 2
213
+ assert path.read_text(encoding="utf-8") == original
214
+
215
+
216
+ async def test_empty_capacity_read_does_not_create_reservation_ledger(tmp_path):
217
+ from tuneplane_node.host_resources import read_host_capacity
218
+
219
+ capacity = await read_host_capacity(Inventory(), CAPACITY, lock_path=tmp_path / "host.lock")
220
+ assert capacity.free_cpus == CAPACITY.cpus
221
+ assert not (tmp_path / "host.json").exists()
@@ -0,0 +1,62 @@
1
+ """Real Linux Docker acceptance; CI uses a dedicated daemon without GPU devices."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import json
6
+ import os
7
+ import uuid
8
+
9
+ import pytest
10
+
11
+ from tuneplane_node.allocator import NoCapacity
12
+ from tuneplane_node.host_resources import HostResources, claim_host
13
+ from tuneplane_node.runtime import LABEL_RUN_ID, CliRuntime, ContainerSpec
14
+
15
+ pytestmark = pytest.mark.skipif(
16
+ os.environ.get("TUNEPLANE_TEST_HOST_DOCKER") != "1",
17
+ reason="requires an explicitly enabled dedicated Linux Docker daemon",
18
+ )
19
+
20
+
21
+ async def test_real_docker_host_admission_limits_restart_and_release(tmp_path):
22
+ runtime = CliRuntime("docker", timeout=30)
23
+ image = "alpine:3.21"
24
+ await runtime.pull(image, timeout=90)
25
+ budget = HostResources(1, 128 * 1024**2)
26
+ names = [f"tuneplane-host-test-{uuid.uuid4().hex}" for _ in range(2)]
27
+ lock_path = tmp_path / "host.lock"
28
+
29
+ async def launch(name, runtime):
30
+ async with claim_host(runtime, budget, budget, run_id=name, lock_path=lock_path) as held:
31
+ spec = ContainerSpec(
32
+ name=name, image=image, command=["sleep", "60"],
33
+ labels={LABEL_RUN_ID: name, **budget.labels()},
34
+ cpu_limit="1", memory_limit=str(budget.memory_bytes),
35
+ )
36
+ await held.run(lambda: runtime.run(spec))
37
+ return name
38
+
39
+ try:
40
+ outcomes = await asyncio.gather(*(launch(name, runtime) for name in names), return_exceptions=True)
41
+ assert sum(isinstance(outcome, NoCapacity) for outcome in outcomes) == 1
42
+ running = next(outcome for outcome in outcomes if isinstance(outcome, str))
43
+ waiting = next(name for name in names if name != running)
44
+ _, raw, _ = await runtime._exec("inspect", running)
45
+ info = json.loads(raw)[0]
46
+ assert info["HostConfig"]["NanoCpus"] == 1_000_000_000
47
+ assert info["HostConfig"]["Memory"] == budget.memory_bytes
48
+ assert not info["HostConfig"].get("DeviceRequests")
49
+ assert "NVIDIA_VISIBLE_DEVICES=void" in info["Config"]["Env"]
50
+ _, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/memory.max")
51
+ assert int(raw.strip()) == budget.memory_bytes
52
+ _, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/cpu.max")
53
+ quota, period = map(int, raw.split())
54
+ assert quota == period
55
+ restarted = CliRuntime("docker", timeout=30)
56
+ with pytest.raises(NoCapacity):
57
+ await launch(waiting, restarted)
58
+ assert await runtime.stop(running, timeout=1)
59
+ assert await launch(waiting, restarted) == waiting
60
+ finally:
61
+ for name in names:
62
+ await runtime.remove(name)
@@ -166,6 +166,39 @@ def test_no_cards_means_no_gpu_flag_at_all():
166
166
  # ── run ────────────────────────────────────────────────────────────────────
167
167
 
168
168
 
169
+ @pytest.mark.parametrize("binary", ["docker", "podman"])
170
+ @pytest.mark.parametrize("env", [{}, {"NVIDIA_VISIBLE_DEVICES": "all", "CUDA_VISIBLE_DEVICES": "0"}])
171
+ async def test_zero_gpu_containers_override_image_and_user_visibility(binary, env):
172
+ runtime = RecordingCli(binary=binary)
173
+ spec = _spec(env=env)
174
+ original = dict(spec.env)
175
+ await runtime.run(spec)
176
+ argv = _argv(runtime, "run")
177
+ assert "--gpus" not in argv and "--device" not in argv
178
+ assert "NVIDIA_VISIBLE_DEVICES=void" in argv
179
+ assert "CUDA_VISIBLE_DEVICES=" in argv
180
+ assert "NVIDIA_VISIBLE_DEVICES=all" not in argv
181
+ assert "CUDA_VISIBLE_DEVICES=0" not in argv
182
+ assert spec.env == original
183
+
184
+
185
+ async def test_gpu_containers_preserve_their_visibility_environment():
186
+ runtime = RecordingCli()
187
+ await runtime.run(_spec(gpus=[2], env={"CUDA_VISIBLE_DEVICES": "0"}))
188
+ argv = _argv(runtime, "run")
189
+ assert "--gpus" in argv
190
+ assert "CUDA_VISIBLE_DEVICES=0" in argv
191
+ assert "NVIDIA_VISIBLE_DEVICES=void" not in argv
192
+
193
+
194
+ @pytest.mark.parametrize("limits", [{"cpu_limit": "4"}, {"memory_limit": "8g"}])
195
+ async def test_process_runtime_refuses_limits_it_cannot_enforce(tmp_path, limits):
196
+ runtime = ProcessRuntime(tmp_path / "state")
197
+ with pytest.raises(RuntimeError_, match="cannot enforce CPU or memory"):
198
+ await runtime.run(_spec(**limits))
199
+ assert not (tmp_path / "state" / "tuneplane-run-a.json").exists()
200
+
201
+
169
202
  async def test_the_command_replaces_the_images_entrypoint_rather_than_appending():
170
203
  """A Playground session on `vllm/vllm-openai` died at once with
171
204
  `vllm: error: unrecognized arguments: -lc python3 -m vllm...` -- our argv
@@ -608,3 +641,29 @@ def test_a_container_that_finished_in_the_future_is_zero_seconds_old():
608
641
  """A clock skew between the daemon and the console must not produce a
609
642
  negative age, which would compare as younger than every threshold."""
610
643
  assert age_seconds("2026-09-12T10:00:00+00:00", 0.0) == 0.0
644
+
645
+
646
+ @pytest.mark.parametrize("payload", ["not json", "[]", '[{"State": {}}]', '[{"Config": {}}]'])
647
+ async def test_strict_inventory_refuses_unreadable_occupancy(payload):
648
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (0, payload, "")})
649
+ with pytest.raises(RuntimeError_, match="cannot establish resource occupancy"):
650
+ await runtime.ps(strict=True)
651
+
652
+
653
+ async def test_strict_inventory_propagates_inspection_failure():
654
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (1, "", "unavailable")})
655
+ with pytest.raises(RuntimeError_, match="inspect failed"):
656
+ await runtime.ps(strict=True)
657
+
658
+
659
+ async def test_strict_inventory_preserves_labels():
660
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""),
661
+ "inspect": (0, _inspect_payload(), "")})
662
+ states = await runtime.ps(strict=True)
663
+ assert len(states) == 1 and states[0].exists
664
+ assert states[0].labels == (await runtime.inspect("tuneplane-a")).labels
665
+
666
+
667
+ async def test_process_runtime_refuses_host_admission(tmp_path):
668
+ with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
669
+ await ProcessRuntime(tmp_path).ps(strict=True)
File without changes
File without changes