tuneplane-node 0.3.16__tar.gz → 0.3.19__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (19) hide show
  1. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/.gitignore +3 -0
  2. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/PKG-INFO +1 -1
  3. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/pyproject.toml +1 -1
  4. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/allocator.py +1 -1
  5. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/daemon.py +45 -1
  6. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/host_resources.py +2 -2
  7. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/join.py +15 -2
  8. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/runtime.py +50 -5
  9. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/settings.py +10 -0
  10. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/tests/test_runtime.py +29 -4
  11. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/LICENSE +0 -0
  12. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/NOTICE +0 -0
  13. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/__init__.py +0 -0
  14. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/cli.py +0 -0
  15. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/inventory.py +0 -0
  16. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/src/tuneplane_node/wire.py +0 -0
  17. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/tests/test_allocator.py +0 -0
  18. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/tests/test_host_resources.py +0 -0
  19. {tuneplane_node-0.3.16 → tuneplane_node-0.3.19}/tests/test_host_resources_docker.py +0 -0
@@ -62,3 +62,6 @@ outputs/
62
62
  # Independent public website build
63
63
  web/dist-website/
64
64
  :memory:.ses
65
+
66
+ # Shipped, independently packaged plugin artifacts.
67
+ !server/src/tuneplane_server/_plugins/**/*.whl
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tuneplane-node
3
- Version: 0.3.16
3
+ Version: 0.3.19
4
4
  Summary: TunePlane node daemon: runs job containers on one machine and reports it to a console.
5
5
  License-Expression: AGPL-3.0-only
6
6
  License-File: LICENSE
@@ -10,7 +10,7 @@
10
10
  # the local backend runs the same container runtime and allocator on its own host.
11
11
  [project]
12
12
  name = "tuneplane-node"
13
- version = "0.3.16"
13
+ version = "0.3.19"
14
14
  description = "TunePlane node daemon: runs job containers on one machine and reports it to a console."
15
15
  requires-python = ">=3.12"
16
16
  # Same licence as the control plane: this is infrastructure an organization runs,
@@ -2,7 +2,7 @@
2
2
 
3
3
  Who decides "which cards this job gets" when there is no Kubernetes
4
4
  ──────────────────────────────────────────────────────────────────────────────
5
- On the kuberay backend kube-scheduler and the NVIDIA device plugin do it: the Pod
5
+ On the kubernetes backend kube-scheduler and the NVIDIA device plugin do it: the Pod
6
6
  declares `nvidia.com/gpu: 4`, the scheduler guarantees no over-subscription, and the
7
7
  device plugin injects the actual cards into the container.
8
8
 
@@ -47,11 +47,13 @@ import secrets
47
47
  import shutil
48
48
  import time
49
49
  from dataclasses import dataclass, field
50
+ from pathlib import Path
50
51
  from weakref import WeakValueDictionary
51
52
 
52
53
  from fastapi import Depends, FastAPI, HTTPException, Request
53
54
 
54
55
  from tuneplane_node.allocator import GpuAllocator, NoCapacity
56
+ from tuneplane_node.host_resources import HostResources, claim_host, read_host_capacity
55
57
  from tuneplane_node.runtime import LABEL_GPUS, ContainerState, detect_runtime
56
58
  from tuneplane_node.runtime import age_seconds as _age_seconds
57
59
  from tuneplane_node.wire import PROTOCOL, spec_from_dict, state_to_dict
@@ -244,6 +246,19 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
244
246
 
245
247
  # ── Launch ──────────────────────────────────────────────────────────────
246
248
 
249
+ def host_capacity():
250
+ cpus = getattr(settings, "local_allocatable_cpus", 0)
251
+ memory = getattr(settings, "local_allocatable_memory_gb", 0)
252
+ if not cpus and not memory:
253
+ return None
254
+ if state.runtime.name not in ("docker", "podman"):
255
+ raise ValueError("CPU admission requires Docker or Podman")
256
+ return HostResources(cpus, memory * 1024**3)
257
+
258
+ def host_lock():
259
+ # A node-local path: separate machines must never share reservation ledgers.
260
+ return Path(getattr(settings, "node_state_dir", "~/.tuneplane-node")).expanduser() / "host-allocation.lock"
261
+
247
262
  async def _start_container(spec, run_id: str, requested_gpus: int) -> dict:
248
263
  async def _run(gpus: list[int]) -> str:
249
264
  # The node writes the pick back itself: GPU passthrough follows the node's
@@ -255,6 +270,25 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
255
270
  spec.labels[LABEL_GPUS] = ",".join(str(i) for i in gpus)
256
271
  return await state.runtime.run(spec)
257
272
 
273
+ capacity = host_capacity()
274
+ if requested_gpus == 0 or capacity is not None:
275
+ request = HostResources.from_labels(spec.labels) or (capacity if requested_gpus > 0 else None)
276
+ if capacity is None or request is None:
277
+ raise NoCapacity("CPU-only execution requires configured host capacity and explicit CPU/memory requests")
278
+ spec.labels.update(request.labels())
279
+ spec.cpu_limit = str(request.cpus)
280
+ spec.memory_limit = str(request.memory_bytes)
281
+ async with claim_host(state.runtime, capacity, request, run_id=run_id, lock_path=host_lock()) as claim:
282
+ if claim.existing is not None:
283
+ return {"phase": "running", "container_id": claim.existing.name, "gpus": claim.existing.gpus, "idempotent": True}
284
+ if requested_gpus == 0:
285
+ cid = await claim.run(lambda: _run([]))
286
+ gpus = []
287
+ else:
288
+ gpus, cid = await state.allocator.allocate_and_run(
289
+ run_id, requested_gpus, lambda ids: claim.run(lambda: _run(ids))
290
+ )
291
+ return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
258
292
  gpus, cid = await state.allocator.allocate_and_run(run_id, requested_gpus, _run)
259
293
  return {"phase": "running", "container_id": cid[:12], "gpus": gpus}
260
294
 
@@ -294,7 +328,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
294
328
  # what to do next rather than queueing on the node -- queueing belongs to
295
329
  # the console's scheduler, and a second queue must not grow here.
296
330
  pending.phase = "failed"
297
- pending.error = f"the GPUs were taken during the image pull ({exc}); submit again"
331
+ pending.error = f"the resources were taken during the image pull ({exc}); submit again"
298
332
  except Exception as exc: # noqa: BLE001
299
333
  pending.phase = "failed"
300
334
  pending.error = f"image pull or launch failed: {exc}"
@@ -439,6 +473,15 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
439
473
  except Exception as exc: # noqa: BLE001
440
474
  return {"ok": False, "runtime": base, "protocol": PROTOCOL,
441
475
  "detail": f"GPU probe failed: {exc}"}
476
+ host = None
477
+ try:
478
+ capacity = host_capacity()
479
+ if capacity is not None:
480
+ usage = await read_host_capacity(state.runtime, capacity, lock_path=host_lock())
481
+ host = {"cpus": capacity.cpus, "memory_bytes": capacity.memory_bytes,
482
+ "free_cpus": usage.free_cpus, "free_memory_bytes": usage.free_memory_bytes}
483
+ except Exception as exc:
484
+ log.warning("host capacity unavailable: %s", exc)
442
485
  # The node's view of the shared storage root. The console reports its
443
486
  # own; both should be the same filesystem, and a node that disagrees is
444
487
  # exactly the misconfiguration worth surfacing here.
@@ -467,6 +510,7 @@ def create_node_app(settings, *, runtime=None, allocator=None) -> FastAPI:
467
510
  # console's own cluster_profile says, an H100 node's cards are reported as
468
511
  # H200, and every control that works per series stops working.
469
512
  "series": _node_series(settings),
513
+ "host_resources": host,
470
514
  "gpus_total": occ.total,
471
515
  "gpus_free": max(0, len(occ.free) - reserved),
472
516
  "gpus_reserved": reserved,
@@ -66,12 +66,12 @@ class HostCapacity:
66
66
 
67
67
 
68
68
  def single_host_request(pools) -> HostResources:
69
- """The supported local shape, shared by scheduling and container launch."""
69
+ """The direct single-host shape, shared by scheduling and all CPU executors."""
70
70
  if len(pools) != 1 or pools[0].nodes != 1 or not pools[0].cpus or not pools[0].memory_gb:
71
71
  raise ValueError("host admission requires one single-machine pool with explicit CPU and memory requests")
72
72
  pool = pools[0]
73
73
  if pool.scratch_gb:
74
- raise ValueError("the local backend does not yet enforce per-job scratch space")
74
+ raise ValueError("direct single-host jobs do not yet enforce per-job scratch space")
75
75
  return HostResources(pool.cpus, pool.memory_gb * 1024**3)
76
76
 
77
77
 
@@ -123,7 +123,10 @@ async def join(
123
123
  "inventory": inventory,
124
124
  }
125
125
  async with httpx.AsyncClient(timeout=timeout, transport=transport) as client:
126
- response = await client.post(f"{base}/api/fleets/{fleet_id}/nodes", json=payload)
126
+ response = await client.post(
127
+ f"{base}/api/fleets/{fleet_id}/nodes", json=payload,
128
+ headers={"User-Agent": _user_agent(node_version)},
129
+ )
127
130
  if response.status_code != 201:
128
131
  raise JoinError(_detail(response))
129
132
  body = response.json()
@@ -155,7 +158,11 @@ async def beat(
155
158
  }
156
159
  async with httpx.AsyncClient(timeout=timeout, transport=transport) as client:
157
160
  response = await client.post(
158
- url, json=payload, headers={"Authorization": f"Bearer {identity.credential}"}
161
+ url, json=payload,
162
+ headers={
163
+ "Authorization": f"Bearer {identity.credential}",
164
+ "User-Agent": _user_agent(node_version),
165
+ },
159
166
  )
160
167
  if response.status_code != 200:
161
168
  raise JoinError(_detail(response))
@@ -190,6 +197,12 @@ async def heartbeat_forever(
190
197
  await asyncio.sleep(interval_s)
191
198
 
192
199
 
200
+ def _user_agent(node_version: str) -> str:
201
+ """How the daemon names itself, so a join reads as one in the console's audit trail
202
+ rather than as `python-httpx`."""
203
+ return f"tuneplane-node/{node_version}"
204
+
205
+
193
206
  def _detail(response: httpx.Response) -> str:
194
207
  try:
195
208
  body = response.json()
@@ -30,6 +30,7 @@ resource boundary between them. Never the default.
30
30
  from __future__ import annotations
31
31
 
32
32
  import asyncio
33
+ import contextlib
33
34
  import json
34
35
  import logging
35
36
  import os
@@ -57,7 +58,38 @@ LABEL_ATTEMPT = "tuneplane.attempt"
57
58
 
58
59
  #: Container name prefix. A fixed prefix rather than a random name is what makes a
59
60
  #: repeated launch of the same run_id produce one job instead of two.
60
- NAME_PREFIX = "tuneplane-"
61
+ #:
62
+ #: Two letters because the name is read, not parsed: nothing finds a container
63
+ #: by it -- `ps` filters on `LABEL_RUN_ID` -- so all the prefix does is tell an
64
+ #: operator reading `docker ps` on a shared machine which containers are the
65
+ #: platform's. It was `tuneplane-`, ten characters in front of an id that is
66
+ #: itself eighteen, on a name the console and the CLI both print.
67
+ NAME_PREFIX = "tp-"
68
+
69
+
70
+ #: Environment names that carry credentials: tokens, keys, vended storage sessions.
71
+ _SECRET_HINTS = ("TOKEN", "SECRET", "KEY", "PASSWORD", "PASSWD", "CREDENTIAL", "STORAGE")
72
+
73
+
74
+ def _is_secret(name: str, value: str) -> bool:
75
+ """Whether an environment value must stay off the command line.
76
+
77
+ A presigned URL is a bearer credential whatever its variable is called. A
78
+ value with a line break cannot be written to an env file and stays in argv.
79
+ """
80
+ if "\n" in value or "\r" in value:
81
+ return False
82
+ return any(hint in name.upper() for hint in _SECRET_HINTS) or "X-Amz-" in value or "Signature=" in value
83
+
84
+
85
+ def _write_env_file(values: dict[str, str]) -> str:
86
+ """A private (0600) env file for the container CLI to read; the caller removes it."""
87
+ import tempfile
88
+
89
+ fd, path = tempfile.mkstemp(prefix="tuneplane-env-", suffix=".env")
90
+ with os.fdopen(fd, "w", encoding="utf-8") as handle:
91
+ handle.write("".join(f"{key}={value}\n" for key, value in values.items()))
92
+ return path
61
93
 
62
94
 
63
95
  def container_name(run_id: str) -> str:
@@ -242,8 +274,16 @@ class CliRuntime:
242
274
  # Override CUDA images that default to exposing every NVIDIA device.
243
275
  # Allocation, rather than image or user environment, owns visibility.
244
276
  env.update(NVIDIA_VISIBLE_DEVICES="void", CUDA_VISIBLE_DEVICES="")
277
+ # Secrets never ride in argv: a CLI's command line is readable through
278
+ # /proc by every local user for as long as the call runs. They go in a
279
+ # private env file the CLI reads and this process removes right after.
280
+ secret = {k: v for k, v in env.items() if _is_secret(k, v)}
245
281
  for k, v in env.items():
246
- args += ["-e", f"{k}={v}"]
282
+ if k not in secret:
283
+ args += ["-e", f"{k}={v}"]
284
+ env_file = _write_env_file(secret) if secret else None
285
+ if env_file is not None:
286
+ args += ["--env-file", env_file]
247
287
  for k, v in spec.labels.items():
248
288
  args += ["--label", f"{k}={v}"]
249
289
  for host, cont, ro in spec.mounts:
@@ -272,21 +312,26 @@ class CliRuntime:
272
312
  # `vllm` command -- died at once with
273
313
  # vllm: error: unrecognized arguments: -lc python3 -m vllm...
274
314
  # which is this `bash -lc` arriving where a subcommand was expected. The
275
- # same launch works on kuberay, because a Kubernetes container's
315
+ # same launch works on kubernetes, because a Kubernetes container's
276
316
  # `command` overrides the entrypoint by definition -- so the two backends
277
317
  # disagreed about the one field that says what to run, and the process
278
318
  # runtime (which execs the argv directly) agreed with Kubernetes.
279
319
  #
280
320
  # What this gives up is an image's own entrypoint wrapper, such as
281
321
  # nvidia_entrypoint.sh on the NGC bases. That wrapper prints a banner and
282
- # `exec "$@"`, and the kuberay path has been skipping it since it was
322
+ # `exec "$@"`, and the kubernetes path has been skipping it since it was
283
323
  # written, which is the evidence that nothing depends on it.
284
324
  if spec.command:
285
325
  args += ["--entrypoint", spec.command[0]]
286
326
  args.append(spec.image)
287
327
  args += spec.command[1:]
288
328
 
289
- _, out, _ = await self._exec(*args)
329
+ try:
330
+ _, out, _ = await self._exec(*args)
331
+ finally:
332
+ if env_file is not None:
333
+ with contextlib.suppress(OSError):
334
+ os.unlink(env_file)
290
335
  return out.strip()
291
336
 
292
337
  async def inspect(self, name: str) -> ContainerState:
@@ -17,6 +17,7 @@ import os
17
17
  from pathlib import Path
18
18
  from typing import Optional
19
19
 
20
+ from pydantic import Field, model_validator
20
21
  from pydantic_settings import BaseSettings, SettingsConfigDict
21
22
  from tuneplane.layout import StorageLayout
22
23
 
@@ -48,6 +49,15 @@ class NodeSettings(BaseSettings):
48
49
  local_gpu_passthrough: str = ""
49
50
 
50
51
  # ── what cards it has ───────────────────────────────────────────────────
52
+ local_allocatable_cpus: int = Field(default=0, ge=0)
53
+ local_allocatable_memory_gb: int = Field(default=0, ge=0)
54
+
55
+ @model_validator(mode="after")
56
+ def _host_capacity_is_paired(self):
57
+ if bool(self.local_allocatable_cpus) != bool(self.local_allocatable_memory_gb):
58
+ raise ValueError("local_allocatable_cpus and local_allocatable_memory_gb must both be positive or both zero")
59
+ return self
60
+
51
61
  local_gpu_count: int = 0
52
62
  local_check_gpu_health: bool = True
53
63
  local_check_external_gpus: bool = True
@@ -21,6 +21,7 @@ from __future__ import annotations
21
21
 
22
22
  import asyncio
23
23
  import json
24
+ import os
24
25
  import shutil
25
26
  from pathlib import Path
26
27
 
@@ -87,10 +88,10 @@ def _argv(runtime: RecordingCli, verb: str) -> list[str]:
87
88
  @pytest.mark.parametrize(
88
89
  "run_id,expected",
89
90
  [
90
- ("grpo_demo-alice-20260812", "tuneplane-grpo_demo-alice-20260812"),
91
- ("a/b:c", "tuneplane-a-b-c"),
92
- ("---", "tuneplane-job"),
93
- ("", "tuneplane-job"),
91
+ ("grpo_demo-alice-20260812", "tp-grpo_demo-alice-20260812"),
92
+ ("a/b:c", "tp-a-b-c"),
93
+ ("---", "tp-job"),
94
+ ("", "tp-job"),
94
95
  ],
95
96
  )
96
97
  def test_a_container_name_is_a_pure_function_of_the_run_id(run_id, expected):
@@ -667,3 +668,27 @@ async def test_strict_inventory_preserves_labels():
667
668
  async def test_process_runtime_refuses_host_admission(tmp_path):
668
669
  with pytest.raises(RuntimeError_, match="cannot provide enforced host-resource"):
669
670
  await ProcessRuntime(tmp_path).ps(strict=True)
671
+
672
+
673
+ async def test_credentials_reach_the_container_through_a_private_env_file_never_argv():
674
+ seen = {}
675
+
676
+ class Reading(RecordingCli):
677
+ async def _exec(self, *args, check=True, timeout=None):
678
+ if args and args[0] == "run":
679
+ path = args[list(args).index("--env-file") + 1]
680
+ seen["mode"] = os.stat(path).st_mode & 0o777
681
+ with open(path, encoding="utf-8") as handle:
682
+ seen["text"] = handle.read()
683
+ seen["path"] = path
684
+ return await super()._exec(*args, check=check, timeout=timeout)
685
+
686
+ runtime = Reading()
687
+ await runtime.run(_spec(env={"TUNEPLANE_GOVERNANCE_CREDENTIAL": "tp_gw_secret", "HF_TOKEN": "hf_x",
688
+ "TUNEPLANE_GOVERNANCE_STORAGE": '{"secret_access_key":"s"}',
689
+ "TUNEPLANE_PACKAGE_URL": "https://s3/p?X-Amz-Signature=abc", "NRL_RUN_ID": "run-a"}))
690
+ argv = " ".join(_argv(runtime, "run"))
691
+ assert "tp_gw_secret" not in argv and "hf_x" not in argv and "secret_access_key" not in argv and "X-Amz" not in argv
692
+ assert "NRL_RUN_ID=run-a" in argv
693
+ assert seen["mode"] == 0o600 and "TUNEPLANE_GOVERNANCE_CREDENTIAL=tp_gw_secret\n" in seen["text"]
694
+ assert not os.path.exists(seen["path"])
File without changes
File without changes