tuneplane-node 0.3.16__tar.gz → 0.3.18__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/.gitignore +3 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/PKG-INFO +1 -1
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/pyproject.toml +1 -1
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/daemon.py +45 -1
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/host_resources.py +2 -2
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/runtime.py +48 -3
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/settings.py +10 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/tests/test_runtime.py +29 -4
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/LICENSE +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/NOTICE +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/__init__.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/allocator.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/cli.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/inventory.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/join.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/src/tuneplane_node/wire.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/tests/test_allocator.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/tests/test_host_resources.py +0 -0
- {tuneplane_node-0.3.16 → tuneplane_node-0.3.18}/tests/test_host_resources_docker.py +0 -0
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
# the local backend runs the same container runtime and allocator on its own host.
|
|
11
11
|
[project]
|
|
12
12
|
name = "tuneplane-node"
|
|
13
|
-
version = "0.3.
|
|
13
|
+
version = "0.3.18"
|
|
14
14
|
description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
|
|
15
15
|
requires-python = ">=3.12"
|
|
16
16
|
# Same licence as the control plane: this is infrastructure an organization runs,
|
|
@@ -47,11 +47,13 @@ import secrets
|
|
|
47
47
|
import shutil
|
|
48
48
|
import time
|
|
49
49
|
from dataclasses import dataclass, field
|
|
50
|
+
from pathlib import Path
|
|
50
51
|
from weakref import WeakValueDictionary
|
|
51
52
|
|
|
52
53
|
from fastapi import Depends, FastAPI, HTTPException, Request
|
|
53
54
|
|
|
54
55
|
from tuneplane_node.allocator import GpuAllocator, NoCapacity
|
|
56
|
+
from tuneplane_node.host_resources import HostResources, claim_host, read_host_capacity
|
|
55
57
|
from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
|
|
56
58
|
from tuneplane_node.runtime import age_seconds as _age_seconds
|
|
57
59
|
from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
|
|
@@ -244,6 +246,19 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
244
246
|
|
|
245
247
|
# ── Launch ──────────────────────────────────────────────────────────────
|
|
246
248
|
|
|
249
|
+
def host_capacity():
|
|
250
|
+
cpus = getattr(settings, "local_allocatable_cpus", 0)
|
|
251
|
+
memory = getattr(settings, "local_allocatable_memory_gb", 0)
|
|
252
|
+
if not cpus and not memory:
|
|
253
|
+
return None
|
|
254
|
+
if state.runtime.name not in ("docker", "podman"):
|
|
255
|
+
raise ValueError("CPU admission requires Docker or Podman")
|
|
256
|
+
return HostResources(cpus, memory * 1024**3)
|
|
257
|
+
|
|
258
|
+
def host_lock():
|
|
259
|
+
# A node-local path: separate machines must never share reservation ledgers.
|
|
260
|
+
return Path(getattr(settings, "node_state_dir", "~/.tuneplane-node")).expanduser() / "host-allocation.lock"
|
|
261
|
+
|
|
247
262
|
async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
|
|
248
263
|
async def _run(gpus: list[int]) -> str:
|
|
249
264
|
# The node writes the pick back itself: GPU passthrough follows the node's
|
|
@@ -255,6 +270,25 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
255
270
|
spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
|
|
256
271
|
return await state.runtime.run(spec)
|
|
257
272
|
|
|
273
|
+
capacity = host_capacity()
|
|
274
|
+
if requested_gpus == 0 or capacity is not None:
|
|
275
|
+
request = HostResources.from_labels(spec.labels) or (capacity if requested_gpus > 0 else None)
|
|
276
|
+
if capacity is None or request is None:
|
|
277
|
+
raise NoCapacity("CPU-only execution requires configured host capacity and explicit CPU/memory requests")
|
|
278
|
+
spec.labels.update(request.labels())
|
|
279
|
+
spec.cpu_limit = str(request.cpus)
|
|
280
|
+
spec.memory_limit = str(request.memory_bytes)
|
|
281
|
+
async with claim_host(state.runtime, capacity, request, run_id=run_id, lock_path=host_lock()) as claim:
|
|
282
|
+
if claim.existing is not None:
|
|
283
|
+
return {"phase": "running", "container_id": claim.existing.name, "gpus": claim.existing.gpus, "idempotent": True}
|
|
284
|
+
if requested_gpus == 0:
|
|
285
|
+
cid = await claim.run(lambda: _run([]))
|
|
286
|
+
gpus = []
|
|
287
|
+
else:
|
|
288
|
+
gpus, cid = await state.allocator.allocate_and_run(
|
|
289
|
+
run_id, requested_gpus, lambda ids: claim.run(lambda: _run(ids))
|
|
290
|
+
)
|
|
291
|
+
return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
|
|
258
292
|
gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
|
|
259
293
|
return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
|
|
260
294
|
|
|
@@ -294,7 +328,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
294
328
|
# what to do next rather than queueing on the node -- queueing belongs to
|
|
295
329
|
# the console's scheduler, and a second queue must not grow here.
|
|
296
330
|
pending.phase = "failed"
|
|
297
|
-
pending.error = f"the
|
|
331
|
+
pending.error = f"the resources were taken during the image pull ({exc}); submit again"
|
|
298
332
|
except Exception as exc: # noqa: BLE001
|
|
299
333
|
pending.phase = "failed"
|
|
300
334
|
pending.error = f"image pull or launch failed: {exc}"
|
|
@@ -439,6 +473,15 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
439
473
|
except Exception as exc: # noqa: BLE001
|
|
440
474
|
return {"ok": False, "runtime": base, "protocol": PROTOCOL,
|
|
441
475
|
"detail": f"GPU probe failed: {exc}"}
|
|
476
|
+
host = None
|
|
477
|
+
try:
|
|
478
|
+
capacity = host_capacity()
|
|
479
|
+
if capacity is not None:
|
|
480
|
+
usage = await read_host_capacity(state.runtime, capacity, lock_path=host_lock())
|
|
481
|
+
host = {"cpus": capacity.cpus, "memory_bytes": capacity.memory_bytes,
|
|
482
|
+
"free_cpus": usage.free_cpus, "free_memory_bytes": usage.free_memory_bytes}
|
|
483
|
+
except Exception as exc:
|
|
484
|
+
log.warning("host capacity unavailable: %s", exc)
|
|
442
485
|
# The node's view of the shared storage root. The console reports its
|
|
443
486
|
# own; both should be the same filesystem, and a node that disagrees is
|
|
444
487
|
# exactly the misconfiguration worth surfacing here.
|
|
@@ -467,6 +510,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
|
|
|
467
510
|
# console's own cluster_profile says, an H100 node's cards are reported as
|
|
468
511
|
# H200, and every control that works per series stops working.
|
|
469
512
|
"series": _node_series(settings),
|
|
513
|
+
"host_resources": host,
|
|
470
514
|
"gpus_total": occ.total,
|
|
471
515
|
"gpus_free": max(0, len(occ.free) - reserved),
|
|
472
516
|
"gpus_reserved": reserved,
|
|
@@ -66,12 +66,12 @@ class HostCapacity:
|
|
|
66
66
|
|
|
67
67
|
|
|
68
68
|
def single_host_request(pools) -> HostResources:
|
|
69
|
-
"""The
|
|
69
|
+
"""The direct single-host shape, shared by scheduling and all CPU executors."""
|
|
70
70
|
if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
|
|
71
71
|
raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
|
|
72
72
|
pool = pools[0]
|
|
73
73
|
if pool.scratch_gb:
|
|
74
|
-
raise ValueError("
|
|
74
|
+
raise ValueError("direct single-host jobs do not yet enforce per-job scratch space")
|
|
75
75
|
return HostResources(pool.cpus, pool.memory_gb * 1024**3)
|
|
76
76
|
|
|
77
77
|
|
|
@@ -30,6 +30,7 @@ resource boundary between them. Never the default.
|
|
|
30
30
|
from __future__ import annotations
|
|
31
31
|
|
|
32
32
|
import asyncio
|
|
33
|
+
import contextlib
|
|
33
34
|
import json
|
|
34
35
|
import logging
|
|
35
36
|
import os
|
|
@@ -57,7 +58,38 @@ LABEL_ATTEMPT = "tuneplane.attempt"
|
|
|
57
58
|
|
|
58
59
|
#: Container name prefix. A fixed prefix rather than a random name is what makes a
|
|
59
60
|
#: repeated launch of the same run_id produce one job instead of two.
|
|
60
|
-
|
|
61
|
+
#:
|
|
62
|
+
#: Two letters because the name is read, not parsed: nothing finds a container
|
|
63
|
+
#: by it -- `ps` filters on `LABEL_RUN_ID` -- so all the prefix does is tell an
|
|
64
|
+
#: operator reading `docker ps` on a shared machine which containers are the
|
|
65
|
+
#: platform's. It was `tuneplane-`, ten characters in front of an id that is
|
|
66
|
+
#: itself eighteen, on a name the console and the CLI both print.
|
|
67
|
+
NAME_PREFIX = "tp-"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
#: Environment names that carry credentials: tokens, keys, vended storage sessions.
|
|
71
|
+
_SECRET_HINTS = ("TOKEN", "SECRET", "KEY", "PASSWORD", "PASSWD", "CREDENTIAL", "STORAGE")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _is_secret(name: str, value: str) -> bool:
|
|
75
|
+
"""Whether an environment value must stay off the command line.
|
|
76
|
+
|
|
77
|
+
A presigned URL is a bearer credential whatever its variable is called. A
|
|
78
|
+
value with a line break cannot be written to an env file and stays in argv.
|
|
79
|
+
"""
|
|
80
|
+
if "\n" in value or "\r" in value:
|
|
81
|
+
return False
|
|
82
|
+
return any(hint in name.upper() for hint in _SECRET_HINTS) or "X-Amz-" in value or "Signature=" in value
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _write_env_file(values: dict[str, str]) -> str:
|
|
86
|
+
"""A private (0600) env file for the container CLI to read; the caller removes it."""
|
|
87
|
+
import tempfile
|
|
88
|
+
|
|
89
|
+
fd, path = tempfile.mkstemp(prefix="tuneplane-env-", suffix=".env")
|
|
90
|
+
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
|
91
|
+
handle.write("".join(f"{key}={value}\n" for key, value in values.items()))
|
|
92
|
+
return path
|
|
61
93
|
|
|
62
94
|
|
|
63
95
|
def container_name(run_id: str) -> str:
|
|
@@ -242,8 +274,16 @@ class CliRuntime:
|
|
|
242
274
|
# Override CUDA images that default to exposing every NVIDIA device.
|
|
243
275
|
# Allocation, rather than image or user environment, owns visibility.
|
|
244
276
|
env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
|
|
277
|
+
# Secrets never ride in argv: a CLI's command line is readable through
|
|
278
|
+
# /proc by every local user for as long as the call runs. They go in a
|
|
279
|
+
# private env file the CLI reads and this process removes right after.
|
|
280
|
+
secret = {k: v for k, v in env.items() if _is_secret(k, v)}
|
|
245
281
|
for k, v in env.items():
|
|
246
|
-
|
|
282
|
+
if k not in secret:
|
|
283
|
+
args += ["-e", f"{k}={v}"]
|
|
284
|
+
env_file = _write_env_file(secret) if secret else None
|
|
285
|
+
if env_file is not None:
|
|
286
|
+
args += ["--env-file", env_file]
|
|
247
287
|
for k, v in spec.labels.items():
|
|
248
288
|
args += ["--label", f"{k}={v}"]
|
|
249
289
|
for host, cont, ro in spec.mounts:
|
|
@@ -286,7 +326,12 @@ class CliRuntime:
|
|
|
286
326
|
args.append(spec.image)
|
|
287
327
|
args += spec.command[1:]
|
|
288
328
|
|
|
289
|
-
|
|
329
|
+
try:
|
|
330
|
+
_, out, _ = await self._exec(*args)
|
|
331
|
+
finally:
|
|
332
|
+
if env_file is not None:
|
|
333
|
+
with contextlib.suppress(OSError):
|
|
334
|
+
os.unlink(env_file)
|
|
290
335
|
return out.strip()
|
|
291
336
|
|
|
292
337
|
async def inspect(self, name: str) -> ContainerState:
|
|
@@ -17,6 +17,7 @@ import os
|
|
|
17
17
|
from pathlib import Path
|
|
18
18
|
from typing import Optional
|
|
19
19
|
|
|
20
|
+
from pydantic import Field, model_validator
|
|
20
21
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
21
22
|
from tuneplane.layout import StorageLayout
|
|
22
23
|
|
|
@@ -48,6 +49,15 @@ class NodeSettings(BaseSettings):
|
|
|
48
49
|
local_gpu_passthrough: str = ""
|
|
49
50
|
|
|
50
51
|
# ── what cards it has ───────────────────────────────────────────────────
|
|
52
|
+
local_allocatable_cpus: int = Field(default=0, ge=0)
|
|
53
|
+
local_allocatable_memory_gb: int = Field(default=0, ge=0)
|
|
54
|
+
|
|
55
|
+
@model_validator(mode="after")
|
|
56
|
+
def _host_capacity_is_paired(self):
|
|
57
|
+
if bool(self.local_allocatable_cpus) != bool(self.local_allocatable_memory_gb):
|
|
58
|
+
raise ValueError("local_allocatable_cpus and local_allocatable_memory_gb must both be positive or both zero")
|
|
59
|
+
return self
|
|
60
|
+
|
|
51
61
|
local_gpu_count: int = 0
|
|
52
62
|
local_check_gpu_health: bool = True
|
|
53
63
|
local_check_external_gpus: bool = True
|
|
@@ -21,6 +21,7 @@ from __future__ import annotations
|
|
|
21
21
|
|
|
22
22
|
import asyncio
|
|
23
23
|
import json
|
|
24
|
+
import os
|
|
24
25
|
import shutil
|
|
25
26
|
from pathlib import Path
|
|
26
27
|
|
|
@@ -87,10 +88,10 @@ def _argv(runtime: RecordingCli, verb: str) -> list[str]:
|
|
|
87
88
|
@pytest.mark.parametrize(
|
|
88
89
|
"run_id,expected",
|
|
89
90
|
[
|
|
90
|
-
("grpo_demo-alice-20260812", "
|
|
91
|
-
("a/b:c", "
|
|
92
|
-
("---", "
|
|
93
|
-
("", "
|
|
91
|
+
("grpo_demo-alice-20260812", "tp-grpo_demo-alice-20260812"),
|
|
92
|
+
("a/b:c", "tp-a-b-c"),
|
|
93
|
+
("---", "tp-job"),
|
|
94
|
+
("", "tp-job"),
|
|
94
95
|
],
|
|
95
96
|
)
|
|
96
97
|
def test_a_container_name_is_a_pure_function_of_the_run_id(run_id, expected):
|
|
@@ -667,3 +668,27 @@ async def test_strict_inventory_preserves_labels():
|
|
|
667
668
|
async def test_process_runtime_refuses_host_admission(tmp_path):
|
|
668
669
|
with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
|
|
669
670
|
await ProcessRuntime(tmp_path).ps(strict=True)
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
async def test_credentials_reach_the_container_through_a_private_env_file_never_argv():
|
|
674
|
+
seen = {}
|
|
675
|
+
|
|
676
|
+
class Reading(RecordingCli):
|
|
677
|
+
async def _exec(self, *args, check=True, timeout=None):
|
|
678
|
+
if args and args[0] == "run":
|
|
679
|
+
path = args[list(args).index("--env-file") + 1]
|
|
680
|
+
seen["mode"] = os.stat(path).st_mode & 0o777
|
|
681
|
+
with open(path, encoding="utf-8") as handle:
|
|
682
|
+
seen["text"] = handle.read()
|
|
683
|
+
seen["path"] = path
|
|
684
|
+
return await super()._exec(*args, check=check, timeout=timeout)
|
|
685
|
+
|
|
686
|
+
runtime = Reading()
|
|
687
|
+
await runtime.run(_spec(env={"TUNEPLANE_GOVERNANCE_CREDENTIAL": "tp_gw_secret", "HF_TOKEN": "hf_x",
|
|
688
|
+
"TUNEPLANE_GOVERNANCE_STORAGE": '{"secret_access_key":"s"}',
|
|
689
|
+
"TUNEPLANE_PACKAGE_URL": "https://s3/p?X-Amz-Signature=abc", "NRL_RUN_ID": "run-a"}))
|
|
690
|
+
argv = " ".join(_argv(runtime, "run"))
|
|
691
|
+
assert "tp_gw_secret" not in argv and "hf_x" not in argv and "secret_access_key" not in argv and "X-Amz" not in argv
|
|
692
|
+
assert "NRL_RUN_ID=run-a" in argv
|
|
693
|
+
assert seen["mode"] == 0o600 and "TUNEPLANE_GOVERNANCE_CREDENTIAL=tp_gw_secret\n" in seen["text"]
|
|
694
|
+
assert not os.path.exists(seen["path"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|