tuneplane-node 0.3.15__tar.gz → 0.3.18__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,6 +10,7 @@ venv/
10
10
  .python-version
11
11
  .pytest_cache/
12
12
  .ruff_cache/
13
+ server/native/target/
13
14
 
14
15
  # Frontend
15
16
  web/node_modules/
@@ -57,3 +58,10 @@ outputs/
57
58
 
58
59
  # Agent worktrees, created by the harness and never part of a commit.
59
60
  .claude/worktrees/
61
+
62
+ # Independent public website build
63
+ web/dist-website/
64
+ :memory:.ses
65
+
66
+ # Shipped, independently packaged plugin artifacts.
67
+ !server/src/tuneplane_server/_plugins/**/*.whl
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuneplane-node
3
- Version: 0.3.15
3
+ Version: 0.3.18
4
4
  Summary: TunePlane node daemon: runs job containers on one machine and reports it to a console.
5
5
  License-Expression: AGPL-3.0-only
6
6
  License-File: LICENSE
@@ -10,7 +10,7 @@
10
10
  # the local backend runs the same container runtime and allocator on its own host.
11
11
  [project]
12
12
  name = "tuneplane-node"
13
- version = "0.3.15"
13
+ version = "0.3.18"
14
14
  description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
15
15
  requires-python = ">=3.12"
16
16
  # Same licence as the control plane: this is infrastructure an organization runs,
@@ -47,11 +47,13 @@ import secrets
47
47
  import shutil
48
48
  import time
49
49
  from dataclasses import dataclass, field
50
+ from pathlib import Path
50
51
  from weakref import WeakValueDictionary
51
52
 
52
53
  from fastapi import Depends, FastAPI, HTTPException, Request
53
54
 
54
55
  from tuneplane_node.allocator import GpuAllocator, NoCapacity
56
+ from tuneplane_node.host_resources import HostResources, claim_host, read_host_capacity
55
57
  from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
56
58
  from tuneplane_node.runtime import age_seconds as _age_seconds
57
59
  from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
@@ -244,6 +246,19 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
244
246
 
245
247
  # ── Launch ──────────────────────────────────────────────────────────────
246
248
 
249
+ def host_capacity():
250
+ cpus = getattr(settings, "local_allocatable_cpus", 0)
251
+ memory = getattr(settings, "local_allocatable_memory_gb", 0)
252
+ if not cpus and not memory:
253
+ return None
254
+ if state.runtime.name not in ("docker", "podman"):
255
+ raise ValueError("CPU admission requires Docker or Podman")
256
+ return HostResources(cpus, memory * 1024**3)
257
+
258
+ def host_lock():
259
+ # A node-local path: separate machines must never share reservation ledgers.
260
+ return Path(getattr(settings, "node_state_dir", "~/.tuneplane-node")).expanduser() / "host-allocation.lock"
261
+
247
262
  async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
248
263
  async def _run(gpus: list[int]) -> str:
249
264
  # The node writes the pick back itself: GPU passthrough follows the node's
@@ -255,6 +270,25 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
255
270
  spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
256
271
  return await state.runtime.run(spec)
257
272
 
273
+ capacity = host_capacity()
274
+ if requested_gpus == 0 or capacity is not None:
275
+ request = HostResources.from_labels(spec.labels) or (capacity if requested_gpus > 0 else None)
276
+ if capacity is None or request is None:
277
+ raise NoCapacity("CPU-only execution requires configured host capacity and explicit CPU/memory requests")
278
+ spec.labels.update(request.labels())
279
+ spec.cpu_limit = str(request.cpus)
280
+ spec.memory_limit = str(request.memory_bytes)
281
+ async with claim_host(state.runtime, capacity, request, run_id=run_id, lock_path=host_lock()) as claim:
282
+ if claim.existing is not None:
283
+ return {"phase": "running", "container_id": claim.existing.name, "gpus": claim.existing.gpus, "idempotent": True}
284
+ if requested_gpus == 0:
285
+ cid = await claim.run(lambda: _run([]))
286
+ gpus = []
287
+ else:
288
+ gpus, cid = await state.allocator.allocate_and_run(
289
+ run_id, requested_gpus, lambda ids: claim.run(lambda: _run(ids))
290
+ )
291
+ return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
258
292
  gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
259
293
  return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
260
294
 
@@ -294,7 +328,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
294
328
  # what to do next rather than queueing on the node -- queueing belongs to
295
329
  # the console's scheduler, and a second queue must not grow here.
296
330
  pending.phase = "failed"
297
- pending.error = f"the GPUs were taken during the image pull ({exc}); submit again"
331
+ pending.error = f"the resources were taken during the image pull ({exc}); submit again"
298
332
  except Exception as exc: # noqa: BLE001
299
333
  pending.phase = "failed"
300
334
  pending.error = f"image pull or launch failed: {exc}"
@@ -439,6 +473,15 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
439
473
  except Exception as exc: # noqa: BLE001
440
474
  return {"ok": False, "runtime": base, "protocol": PROTOCOL,
441
475
  "detail": f"GPU probe failed: {exc}"}
476
+ host = None
477
+ try:
478
+ capacity = host_capacity()
479
+ if capacity is not None:
480
+ usage = await read_host_capacity(state.runtime, capacity, lock_path=host_lock())
481
+ host = {"cpus": capacity.cpus, "memory_bytes": capacity.memory_bytes,
482
+ "free_cpus": usage.free_cpus, "free_memory_bytes": usage.free_memory_bytes}
483
+ except Exception as exc:
484
+ log.warning("host capacity unavailable: %s", exc)
442
485
  # The node's view of the shared storage root. The console reports its
443
486
  # own; both should be the same filesystem, and a node that disagrees is
444
487
  # exactly the misconfiguration worth surfacing here.
@@ -467,6 +510,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
467
510
  # console's own cluster_profile says, an H100 node's cards are reported as
468
511
  # H200, and every control that works per series stops working.
469
512
  "series": _node_series(settings),
513
+ "host_resources": host,
470
514
  "gpus_total": occ.total,
471
515
  "gpus_free": max(0, len(occ.free) - reserved),
472
516
  "gpus_reserved": reserved,
@@ -0,0 +1,206 @@
1
+ """Host admission reconstructed from container labels under a shared file lock.
2
+
3
+ The lock file must be shared by every allocator of the same local runtime and
4
+ must never be unlinked. Container lifetime is reservation lifetime: a restart
5
+ reads the same labels, and an exited container no longer holds CPU or memory.
6
+ Before a container becomes observable, a durable intent covers uncertain startup.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ import fcntl
12
+ import json
13
+ import logging
14
+ import os
15
+ import tempfile
16
+ from contextlib import asynccontextmanager
17
+ from dataclasses import dataclass
18
+ from pathlib import Path
19
+
20
+ from tuneplane_node.allocator import NoCapacity
21
+ from tuneplane_node.runtime import ContainerState
22
+
23
+ LABEL_CPUS = "tuneplane.host-cpus"
24
+ LABEL_MEMORY = "tuneplane.host-memory-bytes"
25
+ log = logging.getLogger(__name__)
26
+
27
+
28
+ @dataclass(frozen=True)
29
+ class HostResources:
30
+ cpus: int
31
+ memory_bytes: int
32
+
33
+ def __post_init__(self):
34
+ for value in (self.cpus, self.memory_bytes):
35
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
36
+ raise ValueError("host CPU and memory resources must be positive integers")
37
+
38
+ def labels(self) -> dict[str, str]:
39
+ return {LABEL_CPUS: str(self.cpus), LABEL_MEMORY: str(self.memory_bytes)}
40
+
41
+ @classmethod
42
+ def from_labels(cls, labels: dict[str, str]) -> HostResources | None:
43
+ try:
44
+ return cls(int(labels[LABEL_CPUS]), int(labels[LABEL_MEMORY]))
45
+ except (KeyError, ValueError, TypeError):
46
+ return None
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class HostCapacity:
51
+ """One host's observed occupancy, including unresolved launch intents."""
52
+
53
+ allocatable: HostResources
54
+ used_cpus: int
55
+ used_memory_bytes: int
56
+ live_runs: frozenset[str]
57
+ pending_runs: frozenset[str]
58
+
59
+ @property
60
+ def free_cpus(self) -> int:
61
+ return max(0, self.allocatable.cpus - self.used_cpus)
62
+
63
+ @property
64
+ def free_memory_bytes(self) -> int:
65
+ return max(0, self.allocatable.memory_bytes - self.used_memory_bytes)
66
+
67
+
68
+ def single_host_request(pools) -> HostResources:
69
+ """The direct single-host shape, shared by scheduling and all CPU executors."""
70
+ if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
71
+ raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
72
+ pool = pools[0]
73
+ if pool.scratch_gb:
74
+ raise ValueError("direct single-host jobs do not yet enforce per-job scratch space")
75
+ return HostResources(pool.cpus, pool.memory_gb * 1024**3)
76
+
77
+
78
+ def _occupancy(capacity, states, pending) -> HostCapacity:
79
+ visible = {state.run_id for state in states if state.exists}
80
+ unresolved = {key: value for key, value in pending.items() if key not in visible}
81
+ live = [state for state in states if state.exists and state.status != "exited"]
82
+ held = [HostResources.from_labels(state.labels) or capacity for state in live]
83
+ held.extend(unresolved.values())
84
+ return HostCapacity(
85
+ allocatable=capacity,
86
+ used_cpus=sum(value.cpus for value in held),
87
+ used_memory_bytes=sum(value.memory_bytes for value in held),
88
+ live_runs=frozenset(state.run_id for state in live),
89
+ pending_runs=frozenset(unresolved),
90
+ )
91
+
92
+
93
+ @asynccontextmanager
94
+ async def _host_lock(lock_path: Path):
95
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
96
+ with lock_path.open("a+b") as lock:
97
+ while True:
98
+ try:
99
+ fcntl.flock(lock.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
100
+ break
101
+ except BlockingIOError:
102
+ await asyncio.sleep(0.05)
103
+ try:
104
+ yield
105
+ finally:
106
+ fcntl.flock(lock.fileno(), fcntl.LOCK_UN)
107
+
108
+
109
+ async def read_host_capacity(runtime, capacity: HostResources, *, lock_path: Path) -> HostCapacity:
110
+ """Read under the launch lock without reconciling or rewriting the ledger."""
111
+ async with _host_lock(lock_path):
112
+ states = await runtime.ps(strict=True)
113
+ return _occupancy(capacity, states, _read_pending(lock_path.with_suffix(".json")))
114
+
115
+
116
+ @dataclass
117
+ class HostClaim:
118
+ existing: ContainerState | None = None
119
+ started: bool = False
120
+
121
+ async def run(self, start):
122
+ """Mark the point after which a failed call may still create a workload."""
123
+ if self.existing is not None:
124
+ raise ValueError("an existing host allocation must not launch another workload")
125
+ self.started = True
126
+ return await start()
127
+
128
+
129
+ def _read_pending(path: Path) -> dict[str, HostResources]:
130
+ try:
131
+ raw = json.loads(path.read_text(encoding="utf-8"))
132
+ return {run_id: HostResources(**value) for run_id, value in raw.items()}
133
+ except FileNotFoundError:
134
+ return {}
135
+ except (ValueError, TypeError, AttributeError) as exc:
136
+ raise NoCapacity("host reservation ledger is unreadable; restore it before admitting jobs") from exc
137
+
138
+
139
+ def _write_pending(path: Path, pending: dict[str, HostResources]) -> None:
140
+ payload = {key: {"cpus": value.cpus, "memory_bytes": value.memory_bytes} for key, value in pending.items()}
141
+ with tempfile.NamedTemporaryFile(mode="w", encoding="utf-8", dir=path.parent, delete=False) as out:
142
+ temporary = Path(out.name)
143
+ try:
144
+ json.dump(payload, out)
145
+ out.flush()
146
+ os.fsync(out.fileno())
147
+ os.replace(temporary, path)
148
+ directory = os.open(path.parent, os.O_RDONLY)
149
+ try:
150
+ os.fsync(directory)
151
+ finally:
152
+ os.close(directory)
153
+ finally:
154
+ temporary.unlink(missing_ok=True)
155
+
156
+
157
+ @asynccontextmanager
158
+ async def claim_host(runtime, capacity: HostResources, request: HostResources, *, run_id: str, lock_path: Path):
159
+ """Hold admission through container creation; yield a launch claim or an existing run.
160
+
161
+ Unknown live allocations consume the entire host allowance. Unreadable
162
+ inventory fails closed. The caller must put the request labels on its new
163
+ container and must not launch another container when an existing run is yielded.
164
+ """
165
+ if request.cpus > capacity.cpus or request.memory_bytes > capacity.memory_bytes:
166
+ raise NoCapacity("the job's host-resource request exceeds this host's allocatable capacity")
167
+ pending_path = lock_path.with_suffix(".json")
168
+ async with _host_lock(lock_path):
169
+ states = await runtime.ps(strict=True)
170
+ pending = _read_pending(pending_path)
171
+ # A visible container takes over its durable intent, including when
172
+ # it already exited. An absent one is an uncertain launch, not free capacity.
173
+ for state in states:
174
+ if state.exists:
175
+ pending.pop(state.run_id, None)
176
+ _write_pending(pending_path, pending)
177
+ live = [state for state in states if state.exists and state.status != "exited"]
178
+ existing = next((state for state in live if state.run_id == run_id), None)
179
+ if existing is not None:
180
+ yield HostClaim(existing=existing)
181
+ return
182
+ if run_id in pending:
183
+ raise NoCapacity("this run has an unresolved host reservation; reconcile its launch before retrying")
184
+ occupancy = _occupancy(capacity, states, pending)
185
+ if request.cpus > occupancy.free_cpus or request.memory_bytes > occupancy.free_memory_bytes:
186
+ raise NoCapacity(
187
+ f"host resources unavailable: need {request.cpus} CPUs and {request.memory_bytes} memory bytes; "
188
+ f"free {occupancy.free_cpus} CPUs and {occupancy.free_memory_bytes} memory bytes"
189
+ )
190
+ pending[run_id] = request
191
+ _write_pending(pending_path, pending)
192
+ claim = HostClaim()
193
+ try:
194
+ yield claim
195
+ finally:
196
+ # Failed/timed-out CLI calls can still create a container later.
197
+ # Keep their intent across restarts until a workload is observed.
198
+ try:
199
+ if not claim.started:
200
+ pending.pop(run_id, None)
201
+ for state in await runtime.ps(strict=True):
202
+ if state.exists:
203
+ pending.pop(state.run_id, None)
204
+ _write_pending(pending_path, pending)
205
+ except Exception:
206
+ log.exception("host launch outcome is unknown; retaining durable reservations")
@@ -30,6 +30,7 @@ resource boundary between them. Never the default.
30
30
  from __future__ import annotations
31
31
 
32
32
  import asyncio
33
+ import contextlib
33
34
  import json
34
35
  import logging
35
36
  import os
@@ -57,7 +58,38 @@ LABEL_ATTEMPT = "tuneplane.attempt"
57
58
 
58
59
  #: Container name prefix. A fixed prefix rather than a random name is what makes a
59
60
  #: repeated launch of the same run_id produce one job instead of two.
60
- NAME_PREFIX = "tuneplane-"
61
+ #:
62
+ #: Two letters because the name is read, not parsed: nothing finds a container
63
+ #: by it -- `ps` filters on `LABEL_RUN_ID` -- so all the prefix does is tell an
64
+ #: operator reading `docker ps` on a shared machine which containers are the
65
+ #: platform's. It was `tuneplane-`, ten characters in front of an id that is
66
+ #: itself eighteen, on a name the console and the CLI both print.
67
+ NAME_PREFIX = "tp-"
68
+
69
+
70
+ #: Environment names that carry credentials: tokens, keys, vended storage sessions.
71
+ _SECRET_HINTS = ("TOKEN", "SECRET", "KEY", "PASSWORD", "PASSWD", "CREDENTIAL", "STORAGE")
72
+
73
+
74
+ def _is_secret(name: str, value: str) -> bool:
75
+ """Whether an environment value must stay off the command line.
76
+
77
+ A presigned URL is a bearer credential whatever its variable is called. A
78
+ value with a line break cannot be written to an env file and stays in argv.
79
+ """
80
+ if "\n" in value or "\r" in value:
81
+ return False
82
+ return any(hint in name.upper() for hint in _SECRET_HINTS) or "X-Amz-" in value or "Signature=" in value
83
+
84
+
85
+ def _write_env_file(values: dict[str, str]) -> str:
86
+ """A private (0600) env file for the container CLI to read; the caller removes it."""
87
+ import tempfile
88
+
89
+ fd, path = tempfile.mkstemp(prefix="tuneplane-env-", suffix=".env")
90
+ with os.fdopen(fd, "w", encoding="utf-8") as handle:
91
+ handle.write("".join(f"{key}={value}\n" for key, value in values.items()))
92
+ return path
61
93
 
62
94
 
63
95
  def container_name(run_id: str) -> str:
@@ -162,7 +194,7 @@ class ContainerRuntime(Protocol):
162
194
 
163
195
  async def remove(self, name: str) -> bool: ...
164
196
 
165
- async def ps(self) -> list[ContainerState]: ...
197
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]: ...
166
198
 
167
199
  async def health(self) -> dict: ...
168
200
 
@@ -237,8 +269,21 @@ class CliRuntime:
237
269
  # ttlSecondsAfterFinished.
238
270
 
239
271
  args += self._gpu_args(spec.gpus)
240
- for k, v in spec.env.items():
241
- args += ["-e", f"{k}={v}"]
272
+ env = dict(spec.env)
273
+ if not spec.gpus:
274
+ # Override CUDA images that default to exposing every NVIDIA device.
275
+ # Allocation, rather than image or user environment, owns visibility.
276
+ env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
277
+ # Secrets never ride in argv: a CLI's command line is readable through
278
+ # /proc by every local user for as long as the call runs. They go in a
279
+ # private env file the CLI reads and this process removes right after.
280
+ secret = {k: v for k, v in env.items() if _is_secret(k, v)}
281
+ for k, v in env.items():
282
+ if k not in secret:
283
+ args += ["-e", f"{k}={v}"]
284
+ env_file = _write_env_file(secret) if secret else None
285
+ if env_file is not None:
286
+ args += ["--env-file", env_file]
242
287
  for k, v in spec.labels.items():
243
288
  args += ["--label", f"{k}={v}"]
244
289
  for host, cont, ro in spec.mounts:
@@ -281,7 +326,12 @@ class CliRuntime:
281
326
  args.append(spec.image)
282
327
  args += spec.command[1:]
283
328
 
284
- _, out, _ = await self._exec(*args)
329
+ try:
330
+ _, out, _ = await self._exec(*args)
331
+ finally:
332
+ if env_file is not None:
333
+ with contextlib.suppress(OSError):
334
+ os.unlink(env_file)
285
335
  return out.strip()
286
336
 
287
337
  async def inspect(self, name: str) -> ContainerState:
@@ -317,7 +367,7 @@ class CliRuntime:
317
367
  code, _, err = await self._exec("rm", "-f", name, check=False)
318
368
  return code == 0 or "no such container" in err.lower()
319
369
 
320
- async def ps(self) -> list[ContainerState]:
370
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]:
321
371
  """List every container this platform owns, exited ones included.
322
372
 
323
373
  `--filter label=` filters on the key alone, so what comes back is the platform's
@@ -330,6 +380,18 @@ class CliRuntime:
330
380
  names = [n.strip() for n in out.splitlines() if n.strip()]
331
381
  if not names:
332
382
  return []
383
+ if strict:
384
+ async def inspect_required(name: str) -> ContainerState:
385
+ _, raw, _ = await self._exec("inspect", name)
386
+ try:
387
+ data = json.loads(raw)[0]
388
+ if not isinstance(data.get("State"), dict) or not isinstance(data.get("Config"), dict):
389
+ raise ValueError("missing container state or configuration")
390
+ return _state_from_inspect(name, data)
391
+ except (ValueError, IndexError, KeyError, TypeError, AttributeError) as exc:
392
+ raise RuntimeError_(f"cannot establish resource occupancy for container {name}") from exc
393
+
394
+ return list(await asyncio.gather(*(inspect_required(name) for name in names)))
333
395
  states = await asyncio.gather(
334
396
  *(self.inspect(n) for n in names), return_exceptions=True
335
397
  )
@@ -412,6 +474,8 @@ class ProcessRuntime:
412
474
 
413
475
  async def run(self, spec: ContainerSpec) -> str:
414
476
  self.state_dir.mkdir(parents=True, exist_ok=True)
477
+ if spec.cpu_limit or spec.memory_limit:
478
+ raise RuntimeError_("the process runtime cannot enforce CPU or memory limits")
415
479
  env = {**os.environ, **spec.env}
416
480
  if spec.gpus:
417
481
  env["CUDA_VISIBLE_DEVICES"] = ",".join(str(i) for i in spec.gpus)
@@ -521,7 +585,9 @@ class ProcessRuntime:
521
585
  p.unlink(missing_ok=True)
522
586
  return True
523
587
 
524
- async def ps(self) -> list[ContainerState]:
588
+ async def ps(self, *, strict: bool = False) -> list[ContainerState]:
589
+ if strict:
590
+ raise RuntimeError_("the process runtime cannot provide enforced host-resource reservations")
525
591
  if not self.state_dir.is_dir():
526
592
  return []
527
593
  out = []
@@ -17,6 +17,7 @@ import os
17
17
  from pathlib import Path
18
18
  from typing import Optional
19
19
 
20
+ from pydantic import Field, model_validator
20
21
  from pydantic_settings import BaseSettings, SettingsConfigDict
21
22
  from tuneplane.layout import StorageLayout
22
23
 
@@ -48,6 +49,15 @@ class NodeSettings(BaseSettings):
48
49
  local_gpu_passthrough: str = ""
49
50
 
50
51
  # ── what cards it has ───────────────────────────────────────────────────
52
+ local_allocatable_cpus: int = Field(default=0, ge=0)
53
+ local_allocatable_memory_gb: int = Field(default=0, ge=0)
54
+
55
+ @model_validator(mode="after")
56
+ def _host_capacity_is_paired(self):
57
+ if bool(self.local_allocatable_cpus) != bool(self.local_allocatable_memory_gb):
58
+ raise ValueError("local_allocatable_cpus and local_allocatable_memory_gb must both be positive or both zero")
59
+ return self
60
+
51
61
  local_gpu_count: int = 0
52
62
  local_check_gpu_health: bool = True
53
63
  local_check_external_gpus: bool = True
@@ -0,0 +1,221 @@
1
+ """Host admission survives concurrency, restarts and uncertain container creation."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import json
6
+ import multiprocessing
7
+ from pathlib import Path
8
+
9
+ import pytest
10
+
11
+ from tuneplane_node.allocator import NoCapacity
12
+ from tuneplane_node.host_resources import HostResources, claim_host
13
+ from tuneplane_node.runtime import LABEL_GPUS, LABEL_RUN_ID, ContainerState
14
+
15
+ CAPACITY = HostResources(4, 8 * 1024**3)
16
+
17
+
18
+ class Inventory:
19
+ def __init__(self, states=()):
20
+ self.states = list(states)
21
+
22
+ async def ps(self, *, strict=False):
23
+ assert strict
24
+ return list(self.states)
25
+
26
+
27
+ def state(run_id, *, status="running", labels=None):
28
+ return ContainerState(
29
+ name=f"tuneplane-{run_id}", exists=True, status=status,
30
+ labels={LABEL_RUN_ID: run_id, **(CAPACITY.labels() if labels is None else labels)},
31
+ )
32
+
33
+
34
+ def claim(runtime, path, run_id="new", request=CAPACITY):
35
+ return claim_host(runtime, CAPACITY, request, run_id=run_id, lock_path=path / "host.lock")
36
+
37
+
38
+ @pytest.mark.parametrize("status", ["running", "created", "paused", "restarting", "unknown"])
39
+ @pytest.mark.parametrize("labels", [None, {}, {"tuneplane.host-cpus": "bad"}, {LABEL_GPUS: "0"}])
40
+ async def test_live_or_unknown_allocations_hold_capacity_after_restart(tmp_path, status, labels):
41
+ runtime = Inventory([state("old", status=status, labels=labels)])
42
+ with pytest.raises(NoCapacity, match="host resources unavailable"):
43
+ async with claim(runtime, tmp_path):
44
+ pytest.fail("occupied host admitted another workload")
45
+
46
+
47
+ async def test_exit_releases_capacity_and_same_run_is_idempotent(tmp_path):
48
+ runtime = Inventory([state("old")])
49
+ async with claim(runtime, tmp_path, "old") as held:
50
+ assert held.existing.name == "tuneplane-old"
51
+ runtime.states[0].status = "exited"
52
+ async with claim(runtime, tmp_path) as held:
53
+ assert held.existing is None
54
+
55
+
56
+ async def test_concurrent_allocators_share_the_file_lock(tmp_path):
57
+ runtime = Inventory()
58
+ started, finish = asyncio.Event(), asyncio.Event()
59
+
60
+ async def create():
61
+ started.set()
62
+ await finish.wait()
63
+ runtime.states.append(state("first"))
64
+
65
+ async def first():
66
+ async with claim(runtime, tmp_path, "first") as held:
67
+ await held.run(create)
68
+
69
+ async def second():
70
+ with pytest.raises(NoCapacity):
71
+ async with claim(runtime, tmp_path, "second"):
72
+ pytest.fail("concurrent admission oversold the host")
73
+
74
+ task = asyncio.create_task(first())
75
+ await started.wait()
76
+ waiter = asyncio.create_task(second())
77
+ await asyncio.sleep(0.02)
78
+ assert not waiter.done()
79
+ finish.set()
80
+ await asyncio.gather(task, waiter)
81
+
82
+
83
+ @pytest.mark.parametrize("failure", [TimeoutError, asyncio.CancelledError])
84
+ async def test_uncertain_launch_retains_durable_intent_until_observed(tmp_path, failure):
85
+ runtime = Inventory()
86
+
87
+ async def uncertain():
88
+ raise failure()
89
+
90
+ with pytest.raises(failure):
91
+ async with claim(runtime, tmp_path, "first") as held:
92
+ await held.run(uncertain)
93
+ with pytest.raises(NoCapacity, match="unresolved host reservation"):
94
+ async with claim(Inventory(), tmp_path, "first"):
95
+ pytest.fail("same uncertain launch was retried")
96
+ with pytest.raises(NoCapacity, match="host resources unavailable"):
97
+ async with claim(Inventory(), tmp_path, "second"):
98
+ pytest.fail("restart lost a pending reservation")
99
+ runtime.states.append(state("first", status="exited"))
100
+ async with claim(runtime, tmp_path, "second"):
101
+ pass
102
+ assert json.loads((tmp_path / "host.json").read_text(encoding="utf-8")) == {}
103
+
104
+
105
+ async def test_failure_before_runtime_launch_releases_intent(tmp_path):
106
+ runtime = Inventory()
107
+ with pytest.raises(NoCapacity, match="GPU"):
108
+ async with claim(runtime, tmp_path):
109
+ raise NoCapacity("GPU shortage")
110
+ async with claim(runtime, tmp_path):
111
+ pass
112
+
113
+
114
+ async def test_corrupt_ledger_blocks_admission(tmp_path):
115
+ (tmp_path / "host.json").write_text("broken", encoding="utf-8")
116
+ with pytest.raises(NoCapacity, match="ledger is unreadable"):
117
+ async with claim(Inventory(), tmp_path):
118
+ pytest.fail("corrupt ledger admitted a workload")
119
+
120
+
121
+ async def test_memory_is_admitted_independently_of_cpu(tmp_path):
122
+ runtime = Inventory([state("old", labels=HostResources(1, 7 * 1024**3).labels())])
123
+ with pytest.raises(NoCapacity):
124
+ async with claim(runtime, tmp_path, request=HostResources(1, 2 * 1024**3)):
125
+ pytest.fail("memory was oversold despite spare CPUs")
126
+
127
+
128
+ def _process_claim(path, ready, outcomes, run_id):
129
+ """Independent processes must see the same durable uncertain-launch intent."""
130
+ async def launch():
131
+ async with claim(Inventory(), Path(path), run_id) as held:
132
+ async def unobserved():
133
+ return "runtime accepted the launch"
134
+ await held.run(unobserved)
135
+ ready.wait(10)
136
+ try:
137
+ asyncio.run(launch())
138
+ outcomes.put("admitted")
139
+ except NoCapacity:
140
+ outcomes.put("blocked")
141
+
142
+
143
+ def test_separate_processes_cannot_oversell(tmp_path):
144
+ context = multiprocessing.get_context("spawn")
145
+ ready, outcomes = context.Event(), context.Queue()
146
+ processes = [context.Process(target=_process_claim, args=(str(tmp_path), ready, outcomes, name))
147
+ for name in ("first", "second")]
148
+ try:
149
+ for process in processes:
150
+ process.start()
151
+ ready.set()
152
+ assert sorted(outcomes.get(timeout=20) for _ in processes) == ["admitted", "blocked"]
153
+ for process in processes:
154
+ process.join(timeout=10)
155
+ assert process.exitcode == 0
156
+ finally:
157
+ for process in processes:
158
+ if process.is_alive():
159
+ process.terminate()
160
+ process.join(timeout=5)
161
+ outcomes.close()
162
+
163
+
164
+ async def test_inventory_failure_blocks_without_releasing_a_pending_launch(tmp_path):
165
+ class Unavailable(Inventory):
166
+ async def ps(self, *, strict=False):
167
+ raise RuntimeError("runtime unavailable")
168
+
169
+ ledger = tmp_path / "host.json"
170
+ original = json.dumps({"old": {"cpus": CAPACITY.cpus, "memory_bytes": CAPACITY.memory_bytes}})
171
+ ledger.write_text(original, encoding="utf-8")
172
+ with pytest.raises(RuntimeError, match="runtime unavailable"):
173
+ async with claim(Unavailable(), tmp_path):
174
+ pytest.fail("unreadable inventory admitted a job")
175
+ assert ledger.read_text(encoding="utf-8") == original
176
+
177
+
178
+ async def test_cancelling_a_lock_waiter_does_not_release_the_active_claim(tmp_path):
179
+ runtime = Inventory()
180
+ async with claim(runtime, tmp_path, "owner"):
181
+ async def wait():
182
+ async with claim(runtime, tmp_path, "waiter"):
183
+ pytest.fail("waiter entered a held lock")
184
+ waiter = asyncio.create_task(wait())
185
+ await asyncio.sleep(0.02)
186
+ waiter.cancel()
187
+ with pytest.raises(asyncio.CancelledError):
188
+ await waiter
189
+ ledger = json.loads((tmp_path / "host.json").read_text(encoding="utf-8"))
190
+ assert list(ledger) == ["owner"]
191
+ async with claim(runtime, tmp_path, "next"):
192
+ pass
193
+
194
+
195
+ async def test_capacity_read_includes_pending_and_does_not_rewrite_ledger(tmp_path):
196
+ from tuneplane_node.host_resources import read_host_capacity
197
+
198
+ path = tmp_path / "host.json"
199
+ original = json.dumps({
200
+ "visible": {"cpus": 1, "memory_bytes": 1024**3},
201
+ "pending": {"cpus": 2, "memory_bytes": 2 * 1024**3},
202
+ })
203
+ path.write_text(original, encoding="utf-8")
204
+ runtime = Inventory([state("visible", labels=HostResources(1, 1024**3).labels())])
205
+ capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
206
+ assert capacity.used_cpus == 3 and capacity.used_memory_bytes == 3 * 1024**3
207
+ assert capacity.free_cpus == 1
208
+ assert capacity.live_runs == {"visible"} and capacity.pending_runs == {"pending"}
209
+ assert path.read_text(encoding="utf-8") == original
210
+ runtime.states[0].status = "exited"
211
+ capacity = await read_host_capacity(runtime, CAPACITY, lock_path=tmp_path / "host.lock")
212
+ assert capacity.used_cpus == 2
213
+ assert path.read_text(encoding="utf-8") == original
214
+
215
+
216
+ async def test_empty_capacity_read_does_not_create_reservation_ledger(tmp_path):
217
+ from tuneplane_node.host_resources import read_host_capacity
218
+
219
+ capacity = await read_host_capacity(Inventory(), CAPACITY, lock_path=tmp_path / "host.lock")
220
+ assert capacity.free_cpus == CAPACITY.cpus
221
+ assert not (tmp_path / "host.json").exists()
@@ -0,0 +1,62 @@
1
+ """Real Linux Docker acceptance; CI uses a dedicated daemon without GPU devices."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import json
6
+ import os
7
+ import uuid
8
+
9
+ import pytest
10
+
11
+ from tuneplane_node.allocator import NoCapacity
12
+ from tuneplane_node.host_resources import HostResources, claim_host
13
+ from tuneplane_node.runtime import LABEL_RUN_ID, CliRuntime, ContainerSpec
14
+
15
+ pytestmark = pytest.mark.skipif(
16
+ os.environ.get("TUNEPLANE_TEST_HOST_DOCKER") != "1",
17
+ reason="requires an explicitly enabled dedicated Linux Docker daemon",
18
+ )
19
+
20
+
21
+ async def test_real_docker_host_admission_limits_restart_and_release(tmp_path):
22
+ runtime = CliRuntime("docker", timeout=30)
23
+ image = "alpine:3.21"
24
+ await runtime.pull(image, timeout=90)
25
+ budget = HostResources(1, 128 * 1024**2)
26
+ names = [f"tuneplane-host-test-{uuid.uuid4().hex}" for _ in range(2)]
27
+ lock_path = tmp_path / "host.lock"
28
+
29
+ async def launch(name, runtime):
30
+ async with claim_host(runtime, budget, budget, run_id=name, lock_path=lock_path) as held:
31
+ spec = ContainerSpec(
32
+ name=name, image=image, command=["sleep", "60"],
33
+ labels={LABEL_RUN_ID: name, **budget.labels()},
34
+ cpu_limit="1", memory_limit=str(budget.memory_bytes),
35
+ )
36
+ await held.run(lambda: runtime.run(spec))
37
+ return name
38
+
39
+ try:
40
+ outcomes = await asyncio.gather(*(launch(name, runtime) for name in names), return_exceptions=True)
41
+ assert sum(isinstance(outcome, NoCapacity) for outcome in outcomes) == 1
42
+ running = next(outcome for outcome in outcomes if isinstance(outcome, str))
43
+ waiting = next(name for name in names if name != running)
44
+ _, raw, _ = await runtime._exec("inspect", running)
45
+ info = json.loads(raw)[0]
46
+ assert info["HostConfig"]["NanoCpus"] == 1_000_000_000
47
+ assert info["HostConfig"]["Memory"] == budget.memory_bytes
48
+ assert not info["HostConfig"].get("DeviceRequests")
49
+ assert "NVIDIA_VISIBLE_DEVICES=void" in info["Config"]["Env"]
50
+ _, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/memory.max")
51
+ assert int(raw.strip()) == budget.memory_bytes
52
+ _, raw, _ = await runtime._exec("exec", running, "cat", "/sys/fs/cgroup/cpu.max")
53
+ quota, period = map(int, raw.split())
54
+ assert quota == period
55
+ restarted = CliRuntime("docker", timeout=30)
56
+ with pytest.raises(NoCapacity):
57
+ await launch(waiting, restarted)
58
+ assert await runtime.stop(running, timeout=1)
59
+ assert await launch(waiting, restarted) == waiting
60
+ finally:
61
+ for name in names:
62
+ await runtime.remove(name)
@@ -21,6 +21,7 @@ from __future__ import annotations
21
21
 
22
22
  import asyncio
23
23
  import json
24
+ import os
24
25
  import shutil
25
26
  from pathlib import Path
26
27
 
@@ -87,10 +88,10 @@ def _argv(runtime: RecordingCli, verb: str) -> list[str]:
87
88
  @pytest.mark.parametrize(
88
89
  "run_id,expected",
89
90
  [
90
- ("grpo_demo-alice-20260812", "tuneplane-grpo_demo-alice-20260812"),
91
- ("a/b:c", "tuneplane-a-b-c"),
92
- ("---", "tuneplane-job"),
93
- ("", "tuneplane-job"),
91
+ ("grpo_demo-alice-20260812", "tp-grpo_demo-alice-20260812"),
92
+ ("a/b:c", "tp-a-b-c"),
93
+ ("---", "tp-job"),
94
+ ("", "tp-job"),
94
95
  ],
95
96
  )
96
97
  def test_a_container_name_is_a_pure_function_of_the_run_id(run_id, expected):
@@ -166,6 +167,39 @@ def test_no_cards_means_no_gpu_flag_at_all():
166
167
  # ── run ────────────────────────────────────────────────────────────────────
167
168
 
168
169
 
170
+ @pytest.mark.parametrize("binary", ["docker", "podman"])
171
+ @pytest.mark.parametrize("env", [{}, {"NVIDIA_VISIBLE_DEVICES": "all", "CUDA_VISIBLE_DEVICES": "0"}])
172
+ async def test_zero_gpu_containers_override_image_and_user_visibility(binary, env):
173
+ runtime = RecordingCli(binary=binary)
174
+ spec = _spec(env=env)
175
+ original = dict(spec.env)
176
+ await runtime.run(spec)
177
+ argv = _argv(runtime, "run")
178
+ assert "--gpus" not in argv and "--device" not in argv
179
+ assert "NVIDIA_VISIBLE_DEVICES=void" in argv
180
+ assert "CUDA_VISIBLE_DEVICES=" in argv
181
+ assert "NVIDIA_VISIBLE_DEVICES=all" not in argv
182
+ assert "CUDA_VISIBLE_DEVICES=0" not in argv
183
+ assert spec.env == original
184
+
185
+
186
+ async def test_gpu_containers_preserve_their_visibility_environment():
187
+ runtime = RecordingCli()
188
+ await runtime.run(_spec(gpus=[2], env={"CUDA_VISIBLE_DEVICES": "0"}))
189
+ argv = _argv(runtime, "run")
190
+ assert "--gpus" in argv
191
+ assert "CUDA_VISIBLE_DEVICES=0" in argv
192
+ assert "NVIDIA_VISIBLE_DEVICES=void" not in argv
193
+
194
+
195
+ @pytest.mark.parametrize("limits", [{"cpu_limit": "4"}, {"memory_limit": "8g"}])
196
+ async def test_process_runtime_refuses_limits_it_cannot_enforce(tmp_path, limits):
197
+ runtime = ProcessRuntime(tmp_path / "state")
198
+ with pytest.raises(RuntimeError_, match="cannot enforce CPU or memory"):
199
+ await runtime.run(_spec(**limits))
200
+ assert not (tmp_path / "state" / "tuneplane-run-a.json").exists()
201
+
202
+
169
203
  async def test_the_command_replaces_the_images_entrypoint_rather_than_appending():
170
204
  """A Playground session on `vllm/vllm-openai` died at once with
171
205
  `vllm: error: unrecognized arguments: -lc python3 -m vllm...` -- our argv
@@ -608,3 +642,53 @@ def test_a_container_that_finished_in_the_future_is_zero_seconds_old():
608
642
  """A clock skew between the daemon and the console must not produce a
609
643
  negative age, which would compare as younger than every threshold."""
610
644
  assert age_seconds("2026-09-12T10:00:00+00:00", 0.0) == 0.0
645
+
646
+
647
+ @pytest.mark.parametrize("payload", ["not json", "[]", '[{"State": {}}]', '[{"Config": {}}]'])
648
+ async def test_strict_inventory_refuses_unreadable_occupancy(payload):
649
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (0, payload, "")})
650
+ with pytest.raises(RuntimeError_, match="cannot establish resource occupancy"):
651
+ await runtime.ps(strict=True)
652
+
653
+
654
+ async def test_strict_inventory_propagates_inspection_failure():
655
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""), "inspect": (1, "", "unavailable")})
656
+ with pytest.raises(RuntimeError_, match="inspect failed"):
657
+ await runtime.ps(strict=True)
658
+
659
+
660
+ async def test_strict_inventory_preserves_labels():
661
+ runtime = RecordingCli(answers={"ps": (0, "tuneplane-a\n", ""),
662
+ "inspect": (0, _inspect_payload(), "")})
663
+ states = await runtime.ps(strict=True)
664
+ assert len(states) == 1 and states[0].exists
665
+ assert states[0].labels == (await runtime.inspect("tuneplane-a")).labels
666
+
667
+
668
+ async def test_process_runtime_refuses_host_admission(tmp_path):
669
+ with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
670
+ await ProcessRuntime(tmp_path).ps(strict=True)
671
+
672
+
673
+ async def test_credentials_reach_the_container_through_a_private_env_file_never_argv():
674
+ seen = {}
675
+
676
+ class Reading(RecordingCli):
677
+ async def _exec(self, *args, check=True, timeout=None):
678
+ if args and args[0] == "run":
679
+ path = args[list(args).index("--env-file") + 1]
680
+ seen["mode"] = os.stat(path).st_mode & 0o777
681
+ with open(path, encoding="utf-8") as handle:
682
+ seen["text"] = handle.read()
683
+ seen["path"] = path
684
+ return await super()._exec(*args, check=check, timeout=timeout)
685
+
686
+ runtime = Reading()
687
+ await runtime.run(_spec(env={"TUNEPLANE_GOVERNANCE_CREDENTIAL": "tp_gw_secret", "HF_TOKEN": "hf_x",
688
+ "TUNEPLANE_GOVERNANCE_STORAGE": '{"secret_access_key":"s"}',
689
+ "TUNEPLANE_PACKAGE_URL": "https://s3/p?X-Amz-Signature=abc", "NRL_RUN_ID": "run-a"}))
690
+ argv = " ".join(_argv(runtime, "run"))
691
+ assert "tp_gw_secret" not in argv and "hf_x" not in argv and "secret_access_key" not in argv and "X-Amz" not in argv
692
+ assert "NRL_RUN_ID=run-a" in argv
693
+ assert seen["mode"] == 0o600 and "TUNEPLANE_GOVERNANCE_CREDENTIAL=tp_gw_secret\n" in seen["text"]
694
+ assert not os.path.exists(seen["path"])
File without changes
File without changes