cortexgrid 0.2.85__tar.gz → 0.2.87__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/PKG-INFO +22 -5
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/_bundle.py +161 -24
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/_serve_entry.py +9 -1
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/jobs.py +13 -3
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/model_serving.py +39 -8
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/model_storage.py +2 -1
- cortexgrid-0.2.87/cortexgrid/serve.py +43 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/docs/cortexgrid/README.md +20 -4
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/pyproject.toml +2 -1
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/.gitignore +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/LICENSE +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/__init__.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.85 → cortexgrid-0.2.87}/cortexgrid/secrets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.87
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -13,6 +13,7 @@ Requires-Dist: cloudpickle>=3.0
|
|
|
13
13
|
Requires-Dist: fabric>=3.2.3
|
|
14
14
|
Requires-Dist: haikunator>=2.1.0
|
|
15
15
|
Requires-Dist: mlflow<4,>=3.11
|
|
16
|
+
Requires-Dist: packaging>=24
|
|
16
17
|
Requires-Dist: pip>=23.0
|
|
17
18
|
Requires-Dist: pydantic-settings>=2.13.1
|
|
18
19
|
Requires-Dist: pydantic>=2.13.3
|
|
@@ -115,9 +116,10 @@ print(f"Submitted: {job_id}")
|
|
|
115
116
|
A separate service — the **jobs control plane** — polls MLflow for pending job requests, matches them against the set of Ray submissions the cluster already has, and submits anything missing. It is also responsible for retrying failed jobs and honouring user-requested stops.
|
|
116
117
|
|
|
117
118
|
Each submission captures the code and dependencies the entry function needs automatically ([_bundle.py](https://github.com/robodatalab/cortexgrid/blob/main/cortexgrid/_bundle.py)):
|
|
118
|
-
- `bundle(entry)` traces the import graph from the function's source file, resolving each import the way the interpreter does (via `sys.path`)
|
|
119
|
-
-
|
|
120
|
-
-
|
|
119
|
+
- `bundle(entry)` traces the import graph from the function's source file, resolving each import the way the interpreter does (via `sys.path`). The standard library is excluded (it ships with the interpreter)
|
|
120
|
+
- Your own modules -- anything outside site-packages / dist-packages -- ship **as source**: they are staged at their import paths and tarred into the Ray `working_dir`
|
|
121
|
+
- Third-party packages are recorded as the installed distribution that owns the imported file, pinned to its installed version (`tqdm==4.67.3`), and Ray **pip-installs** them on the worker into a per-node cached virtualenv layered on the image (`runtime_env["pip"]`). An import into site-packages that no installed distribution owns fails `cortexgrid.remote` with `UnownedDependencyError`
|
|
122
|
+
- Distributions the worker image already has are not installed again: `worker_provides()` is the dependency closure of the packages baked into the ray image (torch and its CUDA stack, ray, mlflow, ...), and is subtracted from the pip list. See [k8s/docker/ray/Dockerfile](https://github.com/robodatalab/cortexgrid/blob/main/k8s/docker/ray/Dockerfile) and `_WORKER_BAKED` in `_bundle.py`, which must list the same packages
|
|
121
123
|
- Injects MLflow/S3 credentials so task code running on the DGX can reach all services
|
|
122
124
|
|
|
123
125
|
##### Retries
|
|
@@ -153,9 +155,24 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
|
|
|
153
155
|
|
|
154
156
|
#### Model registry and serving
|
|
155
157
|
|
|
156
|
-
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application:
|
|
158
|
+
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
|
|
157
159
|
|
|
158
160
|
```python
|
|
161
|
+
from cortexgrid import serve
|
|
162
|
+
from fastapi import FastAPI
|
|
163
|
+
|
|
164
|
+
app = FastAPI()
|
|
165
|
+
|
|
166
|
+
@serve.ingress(app)
|
|
167
|
+
class MyServeApp:
|
|
168
|
+
num_gpus = 1
|
|
169
|
+
|
|
170
|
+
def __init__(self, family: str, suffix: str, run_name: str) -> None:
|
|
171
|
+
self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
|
|
172
|
+
|
|
173
|
+
@app.post("/complete")
|
|
174
|
+
async def complete(self, body: dict): ...
|
|
175
|
+
|
|
159
176
|
saved = cortexgrid.save_model(weights_dir, MyServeApp, family="qwen", suffix="instruct")
|
|
160
177
|
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True)
|
|
161
178
|
print(deployed.url)
|
|
@@ -3,27 +3,38 @@
|
|
|
3
3
|
`bundle(seed)` describes what is needed to run the module at `seed`: the local
|
|
4
4
|
files it reaches -- following its import graph and each package's __init__
|
|
5
5
|
chain, resolving imports the way the interpreter does -- and the third-party
|
|
6
|
-
|
|
7
|
-
package location (site-packages / dist-packages)
|
|
8
|
-
|
|
9
|
-
|
|
6
|
+
distributions those files import. A file is local unless it lives in an installed
|
|
7
|
+
package location (site-packages / dist-packages). An import that resolves into
|
|
8
|
+
one is recorded as the distribution that installed the file, at its installed
|
|
9
|
+
version, and is not followed: installing that distribution brings its own
|
|
10
|
+
dependencies. The standard library is excluded (it ships with the interpreter).
|
|
11
|
+
Bundles of several seeds combine with `BundleDesc.merge`.
|
|
10
12
|
|
|
11
13
|
`stage(files, dest)` lays a bundle out under `dest` at each file's import path,
|
|
12
14
|
so `dest` on sys.path (e.g. a Ray working_dir) makes every module importable.
|
|
15
|
+
|
|
16
|
+
`BundleDesc.pip_requirements(worker_provides())` pins the third-party
|
|
17
|
+
distributions the Ray worker image does not already have, for a Ray `pip`
|
|
18
|
+
runtime_env to install on the worker.
|
|
13
19
|
"""
|
|
14
20
|
|
|
15
21
|
from __future__ import annotations
|
|
16
22
|
|
|
17
23
|
import ast
|
|
18
|
-
from collections.abc import Iterator
|
|
24
|
+
from collections.abc import Iterable, Iterator
|
|
19
25
|
from dataclasses import dataclass
|
|
20
26
|
import functools
|
|
21
27
|
import importlib.machinery
|
|
28
|
+
import importlib.metadata
|
|
22
29
|
import importlib.util
|
|
30
|
+
import os
|
|
23
31
|
from pathlib import Path
|
|
24
32
|
import shutil
|
|
25
33
|
import sys
|
|
26
34
|
|
|
35
|
+
from packaging.requirements import Requirement
|
|
36
|
+
from packaging.utils import canonicalize_name
|
|
37
|
+
|
|
27
38
|
|
|
28
39
|
ThirdPartyDependencyName = str
|
|
29
40
|
ThirdPartyDependencyVersion = str
|
|
@@ -39,17 +50,43 @@ class BundleDesc:
|
|
|
39
50
|
tp_deps={**self.tp_deps, **other.tp_deps},
|
|
40
51
|
)
|
|
41
52
|
|
|
53
|
+
def pip_requirements(
|
|
54
|
+
self, provided: frozenset[ThirdPartyDependencyName]
|
|
55
|
+
) -> list[str]:
|
|
56
|
+
"""The third-party distributions, pinned (`name==version`) and sorted,
|
|
57
|
+
minus the `provided` ones."""
|
|
58
|
+
return sorted(
|
|
59
|
+
f"{name}=={version}"
|
|
60
|
+
for name, version in self.tp_deps.items()
|
|
61
|
+
if name not in provided
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class UnownedDependencyError(LookupError):
|
|
66
|
+
"""An import resolved into an installed package location, but no installed
|
|
67
|
+
distribution owns the file, so it can be neither shipped nor installed."""
|
|
68
|
+
|
|
42
69
|
|
|
43
70
|
def bundle(seed: Path) -> BundleDesc:
|
|
44
71
|
"""What is needed to run the module at `seed`: its local files, each at its
|
|
45
|
-
real path
|
|
46
|
-
|
|
72
|
+
real path, and the third-party distributions they import, keyed by
|
|
73
|
+
canonical name, each at its installed version.
|
|
74
|
+
|
|
75
|
+
Raises UnownedDependencyError for an import that resolves into an installed
|
|
76
|
+
package location no distribution owns."""
|
|
47
77
|
seed = seed.resolve()
|
|
48
78
|
files: set[Path] = set()
|
|
79
|
+
tp_deps: dict[ThirdPartyDependencyName, ThirdPartyDependencyVersion] = {}
|
|
80
|
+
visited: set[Path] = set()
|
|
49
81
|
queue: list[Path] = [seed]
|
|
50
82
|
while queue:
|
|
51
83
|
file = queue.pop()
|
|
52
|
-
if file in
|
|
84
|
+
if file in visited:
|
|
85
|
+
continue
|
|
86
|
+
visited.add(file)
|
|
87
|
+
if not _is_local(file):
|
|
88
|
+
dist = _owning_distribution(file)
|
|
89
|
+
tp_deps[canonicalize_name(dist.metadata["Name"])] = dist.version
|
|
53
90
|
continue
|
|
54
91
|
files.add(file)
|
|
55
92
|
queue.extend(_init_chain(file)) # importing a module runs its __init__ chain
|
|
@@ -58,35 +95,135 @@ def bundle(seed: Path) -> BundleDesc:
|
|
|
58
95
|
dep = _module_file(name)
|
|
59
96
|
if dep is not None:
|
|
60
97
|
queue.append(dep)
|
|
61
|
-
return BundleDesc(local_files=files, tp_deps=
|
|
98
|
+
return BundleDesc(local_files=files, tp_deps=tp_deps)
|
|
62
99
|
|
|
63
100
|
|
|
64
101
|
def stage(files: set[Path], dest: Path) -> None:
|
|
65
102
|
"""Copy `files` under `dest`, each at its import path (relative to the
|
|
66
|
-
sys.path entry it lives under), so `dest` on sys.path imports them all.
|
|
103
|
+
sys.path entry it lives under), so `dest` on sys.path imports them all.
|
|
104
|
+
`dest` is created even when `files` is empty (e.g. code that lives entirely
|
|
105
|
+
in installed distributions)."""
|
|
106
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
67
107
|
for file in files:
|
|
68
108
|
target = dest / file.relative_to(_sys_path_root(file))
|
|
69
109
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
70
110
|
shutil.copy2(file, target)
|
|
71
111
|
|
|
72
112
|
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
#
|
|
76
|
-
#
|
|
77
|
-
|
|
113
|
+
# What the Ray worker image pip-installs, as the Dockerfile spells it. They and
|
|
114
|
+
# their dependency trees are on the worker already, so they are never installed
|
|
115
|
+
# there again -- a second copy in the job's virtualenv would shadow the image's.
|
|
116
|
+
#
|
|
117
|
+
# KEEP IN SYNC with k8s/docker/ray/Dockerfile, by hand: every package it
|
|
118
|
+
# pip-installs must be listed here. A package missing here gets installed a
|
|
119
|
+
# second time on the worker; one listed here but no longer in the image is
|
|
120
|
+
# never installed at all.
|
|
121
|
+
_WORKER_BAKED = (
|
|
122
|
+
"ray[default,serve]",
|
|
123
|
+
"smart_open[s3]",
|
|
124
|
+
"mlflow",
|
|
125
|
+
"python-dotenv",
|
|
126
|
+
"psutil",
|
|
127
|
+
"nvidia-cublas-cu12",
|
|
128
|
+
"nvidia-cudnn-cu12",
|
|
129
|
+
"nvidia-cuda-nvrtc-cu12",
|
|
130
|
+
"nvidia-cuda-runtime-cu12",
|
|
131
|
+
"nvidia-cuda-cupti-cu12",
|
|
132
|
+
"nvidia-cufft-cu12",
|
|
133
|
+
"nvidia-curand-cu12",
|
|
134
|
+
"nvidia-cusolver-cu12",
|
|
135
|
+
"nvidia-cusparse-cu12",
|
|
136
|
+
"nvidia-cusparselt-cu12",
|
|
137
|
+
"nvidia-nccl-cu12",
|
|
138
|
+
"nvidia-nvshmem-cu12",
|
|
139
|
+
"nvidia-nvtx-cu12",
|
|
140
|
+
"nvidia-nvjitlink-cu12",
|
|
141
|
+
"nvidia-cufile-cu12",
|
|
142
|
+
"cuda-bindings",
|
|
143
|
+
"triton",
|
|
144
|
+
"torch",
|
|
145
|
+
"filelock",
|
|
146
|
+
"typing-extensions",
|
|
147
|
+
"sympy",
|
|
148
|
+
"networkx",
|
|
149
|
+
"jinja2",
|
|
150
|
+
"fsspec",
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@functools.lru_cache(maxsize=1)
|
|
155
|
+
def worker_provides() -> frozenset[ThirdPartyDependencyName]:
|
|
156
|
+
"""Every distribution the Ray worker image already provides:
|
|
157
|
+
`distribution_closure(_WORKER_BAKED)`."""
|
|
158
|
+
return distribution_closure(_WORKER_BAKED)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def distribution_closure(
|
|
162
|
+
requirements: Iterable[str],
|
|
163
|
+
) -> frozenset[ThirdPartyDependencyName]:
|
|
164
|
+
"""Canonical names of the distributions `requirements` name plus everything
|
|
165
|
+
they depend on, transitively, as this environment's installed metadata
|
|
166
|
+
declares it -- extras followed where requested, environment markers
|
|
167
|
+
evaluated here. A requirement not installed here contributes only its own
|
|
168
|
+
name: its dependencies are unknown."""
|
|
169
|
+
names: set[ThirdPartyDependencyName] = set()
|
|
170
|
+
seen: set[tuple[str, frozenset[str]]] = set()
|
|
171
|
+
queue = [Requirement(spec) for spec in requirements]
|
|
172
|
+
while queue:
|
|
173
|
+
requirement = queue.pop()
|
|
174
|
+
name = canonicalize_name(requirement.name)
|
|
175
|
+
key = (name, frozenset(requirement.extras))
|
|
176
|
+
if key in seen:
|
|
177
|
+
continue
|
|
178
|
+
seen.add(key)
|
|
179
|
+
names.add(name)
|
|
180
|
+
try:
|
|
181
|
+
dist = importlib.metadata.distribution(name)
|
|
182
|
+
except importlib.metadata.PackageNotFoundError:
|
|
183
|
+
continue
|
|
184
|
+
extras = requirement.extras or {""}
|
|
185
|
+
for spec in dist.requires or ():
|
|
186
|
+
dependency = Requirement(spec)
|
|
187
|
+
if dependency.marker is None or any(
|
|
188
|
+
dependency.marker.evaluate({"extra": extra}) for extra in extras
|
|
189
|
+
):
|
|
190
|
+
queue.append(dependency)
|
|
191
|
+
return frozenset(names)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _owning_distribution(file: Path) -> importlib.metadata.Distribution:
|
|
195
|
+
"""The installed distribution whose file list (RECORD) contains `file`.
|
|
196
|
+
Rebuilds the index once on a miss, in case something was installed since it
|
|
197
|
+
was built."""
|
|
198
|
+
path = tuple(sys.path)
|
|
199
|
+
dist = _distribution_index(path).get(file)
|
|
200
|
+
if dist is None:
|
|
201
|
+
_distribution_index.cache_clear()
|
|
202
|
+
dist = _distribution_index(path).get(file)
|
|
203
|
+
if dist is None:
|
|
204
|
+
raise UnownedDependencyError(
|
|
205
|
+
f"{file} is imported from an installed package location, but no "
|
|
206
|
+
"installed distribution lists it among its files, so it can be "
|
|
207
|
+
"neither shipped nor pip-installed on the worker. Install it with "
|
|
208
|
+
"pip or uv so it carries distribution metadata."
|
|
209
|
+
)
|
|
210
|
+
return dist
|
|
78
211
|
|
|
79
212
|
|
|
80
213
|
@functools.lru_cache(maxsize=1)
|
|
81
|
-
def
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
214
|
+
def _distribution_index(
|
|
215
|
+
path: tuple[str, ...],
|
|
216
|
+
) -> dict[Path, importlib.metadata.Distribution]:
|
|
217
|
+
"""Every file installed by a distribution found on `path` (sys.path, which
|
|
218
|
+
the cache is keyed on), mapped to that distribution. Each distribution's
|
|
219
|
+
root is resolved once; its files are joined onto it lexically, which keeps
|
|
220
|
+
this fast for environments with tens of thousands of files."""
|
|
221
|
+
index: dict[Path, importlib.metadata.Distribution] = {}
|
|
222
|
+
for dist in importlib.metadata.distributions(path=list(path)):
|
|
223
|
+
root = Path(dist.locate_file("")).resolve()
|
|
224
|
+
for file in dist.files or ():
|
|
225
|
+
index[Path(os.path.normpath(root / file))] = dist
|
|
226
|
+
return index
|
|
90
227
|
|
|
91
228
|
|
|
92
229
|
def _is_local(file: Path) -> bool:
|
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
Ray Serve's REST `import_path` resolves to `cortexgrid._serve_entry:build`.
|
|
4
4
|
On the cluster replica, `build` imports the serve-app class bundled at
|
|
5
|
-
`save_model` time (its import path was stored as an MLflow tag),
|
|
5
|
+
`save_model` time (its import path was stored as an MLflow tag), applies Ray's
|
|
6
|
+
ingress with the app it was marked with by `cortexgrid.serve.ingress`, reads its
|
|
6
7
|
`num_gpus`/`num_replicas` class attributes for actor placement, wraps it as a
|
|
7
8
|
Ray Serve deployment, and binds it with the (family, suffix, run_name)
|
|
8
9
|
identifiers.
|
|
@@ -27,10 +28,17 @@ from typing import Any
|
|
|
27
28
|
from ray import serve
|
|
28
29
|
from ray.serve.deployment import Application
|
|
29
30
|
|
|
31
|
+
from cortexgrid.serve import ingress_app
|
|
32
|
+
|
|
30
33
|
|
|
31
34
|
def build(args: dict[str, Any]) -> Application:
|
|
32
35
|
module_name, class_name = args["class_import_path"].split(":")
|
|
33
36
|
serve_app = getattr(importlib.import_module(module_name), class_name)
|
|
37
|
+
# Models saved with a class wrapped by ray.serve.ingress itself carry no
|
|
38
|
+
# mark and are deployed as they are.
|
|
39
|
+
app = ingress_app(serve_app)
|
|
40
|
+
if app is not None:
|
|
41
|
+
serve_app = serve.ingress(app)(serve_app)
|
|
34
42
|
num_gpus = getattr(serve_app, "num_gpus", 0)
|
|
35
43
|
num_replicas = getattr(serve_app, "num_replicas", 1)
|
|
36
44
|
return serve.deployment(serve_app).options(
|
|
@@ -63,6 +63,9 @@ class JobLifecycle:
|
|
|
63
63
|
retry: bool = False # static flag set at job creation
|
|
64
64
|
num_gpus: int = 0
|
|
65
65
|
num_cpus: int = 1
|
|
66
|
+
# static: pinned third-party requirements the worker pip-installs (the
|
|
67
|
+
# bundle's distributions the Ray image does not already provide)
|
|
68
|
+
pip_requirements: list[str] = field(default_factory=list)
|
|
66
69
|
history: list[LifecycleEvent] = field(default_factory=list)
|
|
67
70
|
|
|
68
71
|
def to_json(self) -> str:
|
|
@@ -229,11 +232,17 @@ def schedule_remote_job(
|
|
|
229
232
|
job_id = Haikunator().haikunate(token_length=2, token_chars="0123456789")
|
|
230
233
|
entry_file = Path(inspect.getfile(fn)).resolve()
|
|
231
234
|
driver_file = Path(__file__).with_name("_ray_job_driver.py")
|
|
232
|
-
|
|
235
|
+
desc = bundle(entry_file).merge(bundle(driver_file))
|
|
236
|
+
pip_requirements = desc.pip_requirements(worker_provides())
|
|
233
237
|
with tempfile.TemporaryDirectory() as tmp:
|
|
234
238
|
code_root = Path(tmp, "project_code_root")
|
|
235
|
-
stage(
|
|
236
|
-
log.info(
|
|
239
|
+
stage(desc.local_files, code_root)
|
|
240
|
+
log.info(
|
|
241
|
+
"Submitting job %s (%d files, pip: %s)",
|
|
242
|
+
job_id,
|
|
243
|
+
len(desc.local_files),
|
|
244
|
+
pip_requirements,
|
|
245
|
+
)
|
|
237
246
|
Payload(
|
|
238
247
|
experiment_name=experiment_name,
|
|
239
248
|
run_id=run_id,
|
|
@@ -252,6 +261,7 @@ def schedule_remote_job(
|
|
|
252
261
|
retry=retry,
|
|
253
262
|
num_gpus=num_gpus,
|
|
254
263
|
num_cpus=num_cpus,
|
|
264
|
+
pip_requirements=pip_requirements,
|
|
255
265
|
).save_to_mlflow()
|
|
256
266
|
return job_id
|
|
257
267
|
|
|
@@ -19,11 +19,12 @@ the end-to-end design.
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
21
|
import inspect
|
|
22
|
+
import json
|
|
22
23
|
import logging
|
|
23
24
|
import shutil
|
|
24
25
|
import tempfile
|
|
25
26
|
import time
|
|
26
|
-
from dataclasses import dataclass
|
|
27
|
+
from dataclasses import dataclass, field
|
|
27
28
|
from pathlib import Path
|
|
28
29
|
from typing import Any
|
|
29
30
|
|
|
@@ -90,6 +91,9 @@ class BundleMetadata:
|
|
|
90
91
|
|
|
91
92
|
bundle_url: str
|
|
92
93
|
class_import_path: str
|
|
94
|
+
# pinned third-party requirements the replica pip-installs (the bundle's
|
|
95
|
+
# distributions the Ray image does not already provide)
|
|
96
|
+
pip_requirements: list[str] = field(default_factory=list)
|
|
93
97
|
|
|
94
98
|
|
|
95
99
|
def bundle_class(
|
|
@@ -100,15 +104,32 @@ def bundle_class(
|
|
|
100
104
|
|
|
101
105
|
Returns the metadata `deploy_model` needs later; callers (typically
|
|
102
106
|
`save_model`) persist it on the ModelVersion so the deploy step can run
|
|
103
|
-
without holding the class object.
|
|
107
|
+
without holding the class object.
|
|
108
|
+
|
|
109
|
+
Raises ValueError for a class wrapped by `ray.serve.ingress`: that wrapper
|
|
110
|
+
is a subclass Ray defines in its own module, and on older Ray (e.g. 2.9) it
|
|
111
|
+
reports that module as its own, so the class's source and import path would
|
|
112
|
+
resolve to Ray instead of the serve-app. `cortexgrid.serve.ingress` leaves
|
|
113
|
+
the class unwrapped."""
|
|
114
|
+
if any(klass.__module__.startswith("ray.serve") for klass in cls.__mro__):
|
|
115
|
+
raise ValueError(
|
|
116
|
+
f"{cls.__name__} is wrapped by ray.serve.ingress; decorate it with "
|
|
117
|
+
"cortexgrid.serve.ingress instead (from cortexgrid import serve)"
|
|
118
|
+
)
|
|
104
119
|
entry_file = Path(inspect.getfile(cls)).resolve()
|
|
105
120
|
serve_entry = Path(__file__).with_name("_serve_entry.py")
|
|
106
|
-
|
|
121
|
+
desc = bundle(entry_file).merge(bundle(serve_entry))
|
|
122
|
+
pip_requirements = desc.pip_requirements(worker_provides())
|
|
107
123
|
with tempfile.TemporaryDirectory() as tmp:
|
|
108
124
|
code_root = Path(tmp) / "code"
|
|
109
|
-
stage(
|
|
125
|
+
stage(desc.local_files, code_root)
|
|
110
126
|
log.info(
|
|
111
|
-
"Serve bundle for %s/%s/%s: %d files
|
|
127
|
+
"Serve bundle for %s/%s/%s: %d files, pip: %s",
|
|
128
|
+
family,
|
|
129
|
+
suffix,
|
|
130
|
+
run_name,
|
|
131
|
+
len(desc.local_files),
|
|
132
|
+
pip_requirements,
|
|
112
133
|
)
|
|
113
134
|
zip_base = Path(tmp) / f"{family}__{suffix}"
|
|
114
135
|
shutil.make_archive(str(zip_base), "zip", root_dir=str(code_root))
|
|
@@ -119,6 +140,7 @@ def bundle_class(
|
|
|
119
140
|
return BundleMetadata(
|
|
120
141
|
bundle_url=bundle_url,
|
|
121
142
|
class_import_path=f"{cls.__module__}:{cls.__name__}",
|
|
143
|
+
pip_requirements=pip_requirements,
|
|
122
144
|
)
|
|
123
145
|
|
|
124
146
|
|
|
@@ -126,6 +148,13 @@ def _build_application_spec(
|
|
|
126
148
|
family: str, suffix: str, run_name: str, meta: BundleMetadata
|
|
127
149
|
) -> dict[str, Any]:
|
|
128
150
|
"""Assemble a Ray Serve application schema from pre-bundled metadata."""
|
|
151
|
+
# working_dir carries the serve-app's own source; Ray pip-installs the
|
|
152
|
+
# third-party distributions the image lacks into a per-node cached
|
|
153
|
+
# virtualenv layered on the image. No pip key when there are none, so Ray
|
|
154
|
+
# builds no virtualenv.
|
|
155
|
+
runtime_env: dict[str, Any] = {"working_dir": meta.bundle_url}
|
|
156
|
+
if meta.pip_requirements:
|
|
157
|
+
runtime_env["pip"] = meta.pip_requirements
|
|
129
158
|
return {
|
|
130
159
|
"name": _app_name(family, suffix, run_name),
|
|
131
160
|
"route_prefix": _route_prefix(family, suffix, run_name),
|
|
@@ -140,9 +169,7 @@ def _build_application_spec(
|
|
|
140
169
|
"suffix": suffix,
|
|
141
170
|
"run_name": run_name,
|
|
142
171
|
},
|
|
143
|
-
|
|
144
|
-
# alone makes the serve app importable; nothing is pip-installed.
|
|
145
|
-
"runtime_env": {"working_dir": meta.bundle_url},
|
|
172
|
+
"runtime_env": runtime_env,
|
|
146
173
|
}
|
|
147
174
|
|
|
148
175
|
|
|
@@ -150,6 +177,7 @@ def _build_application_spec(
|
|
|
150
177
|
# `deploy_model` reads back.
|
|
151
178
|
_CLASS_IMPORT_PATH_TAG = "class_import_path"
|
|
152
179
|
_BUNDLE_URL_TAG = "serve_bundle_url"
|
|
180
|
+
_PIP_REQUIREMENTS_TAG = "serve_pip_requirements"
|
|
153
181
|
|
|
154
182
|
|
|
155
183
|
def metadata_to_tags(meta: BundleMetadata) -> dict[str, str]:
|
|
@@ -159,6 +187,7 @@ def metadata_to_tags(meta: BundleMetadata) -> dict[str, str]:
|
|
|
159
187
|
return {
|
|
160
188
|
_CLASS_IMPORT_PATH_TAG: meta.class_import_path,
|
|
161
189
|
_BUNDLE_URL_TAG: meta.bundle_url,
|
|
190
|
+
_PIP_REQUIREMENTS_TAG: json.dumps(meta.pip_requirements),
|
|
162
191
|
}
|
|
163
192
|
|
|
164
193
|
|
|
@@ -180,6 +209,8 @@ def _load_bundle_metadata(
|
|
|
180
209
|
return BundleMetadata(
|
|
181
210
|
bundle_url=tags[_BUNDLE_URL_TAG],
|
|
182
211
|
class_import_path=tags[_CLASS_IMPORT_PATH_TAG],
|
|
212
|
+
# Absent on models saved before dependencies were pip-installed.
|
|
213
|
+
pip_requirements=json.loads(tags.get(_PIP_REQUIREMENTS_TAG, "[]")),
|
|
183
214
|
)
|
|
184
215
|
except KeyError as exc:
|
|
185
216
|
raise ValueError(
|
|
@@ -143,7 +143,8 @@ def save_model(
|
|
|
143
143
|
the caller's concern. That directory boundary is the open-closed extension
|
|
144
144
|
point - new model kinds need no change here.
|
|
145
145
|
|
|
146
|
-
`serve_app` is the
|
|
146
|
+
`serve_app` is the `cortexgrid.serve.ingress` class that will front these
|
|
147
|
+
weights.
|
|
147
148
|
Its code is bundled and its import path, bundle URL, and pip list are
|
|
148
149
|
stored as tags on the ModelVersion so `deploy_model` can bind it later
|
|
149
150
|
without the caller holding the class object."""
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Declare a serve-app's HTTP ingress without importing Ray.
|
|
2
|
+
|
|
3
|
+
from cortexgrid import serve
|
|
4
|
+
|
|
5
|
+
@serve.ingress(app)
|
|
6
|
+
class MyServeApp: ...
|
|
7
|
+
|
|
8
|
+
Same shape as `ray.serve.ingress`, but the class is left exactly as written: the
|
|
9
|
+
FastAPI app is only recorded on it, and `cortexgrid._serve_entry.build` applies
|
|
10
|
+
Ray's ingress when it builds the Serve application on the cluster.
|
|
11
|
+
|
|
12
|
+
Ray's decorator replaces the class with a wrapper subclass defined in
|
|
13
|
+
ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
|
|
14
|
+
that module. Everything that locates a serve-app by its module - bundling its
|
|
15
|
+
source, recording its import path - would then find Ray instead of the user's
|
|
16
|
+
code. Deferring the wrap to the one place Serve needs it keeps the class
|
|
17
|
+
locatable everywhere else (the laptop, Ray jobs, tests).
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from typing import Any, Callable, TypeVar
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
_T = TypeVar("_T", bound=type)
|
|
26
|
+
|
|
27
|
+
_INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def ingress(app: Any) -> Callable[[_T], _T]:
|
|
31
|
+
"""Mark a serve-app class as fronted by the ASGI `app` (e.g. a FastAPI
|
|
32
|
+
instance). Returns the class itself, unwrapped."""
|
|
33
|
+
|
|
34
|
+
def decorator(cls: _T) -> _T:
|
|
35
|
+
setattr(cls, _INGRESS_APP_ATTR, app)
|
|
36
|
+
return cls
|
|
37
|
+
|
|
38
|
+
return decorator
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def ingress_app(cls: type) -> Any | None:
|
|
42
|
+
"""The app `cls` was marked with by `ingress`, or None if it was not."""
|
|
43
|
+
return getattr(cls, _INGRESS_APP_ATTR, None)
|
|
@@ -88,9 +88,10 @@ print(f"Submitted: {job_id}")
|
|
|
88
88
|
A separate service — the **jobs control plane** — polls MLflow for pending job requests, matches them against the set of Ray submissions the cluster already has, and submits anything missing. It is also responsible for retrying failed jobs and honouring user-requested stops.
|
|
89
89
|
|
|
90
90
|
Each submission captures the code and dependencies the entry function needs automatically ([_bundle.py](https://github.com/robodatalab/cortexgrid/blob/main/cortexgrid/_bundle.py)):
|
|
91
|
-
- `bundle(entry)` traces the import graph from the function's source file, resolving each import the way the interpreter does (via `sys.path`)
|
|
92
|
-
-
|
|
93
|
-
-
|
|
91
|
+
- `bundle(entry)` traces the import graph from the function's source file, resolving each import the way the interpreter does (via `sys.path`). The standard library is excluded (it ships with the interpreter)
|
|
92
|
+
- Your own modules -- anything outside site-packages / dist-packages -- ship **as source**: they are staged at their import paths and tarred into the Ray `working_dir`
|
|
93
|
+
- Third-party packages are recorded as the installed distribution that owns the imported file, pinned to its installed version (`tqdm==4.67.3`), and Ray **pip-installs** them on the worker into a per-node cached virtualenv layered on the image (`runtime_env["pip"]`). An import into site-packages that no installed distribution owns fails `cortexgrid.remote` with `UnownedDependencyError`
|
|
94
|
+
- Distributions the worker image already has are not installed again: `worker_provides()` is the dependency closure of the packages baked into the ray image (torch and its CUDA stack, ray, mlflow, ...), and is subtracted from the pip list. See [k8s/docker/ray/Dockerfile](https://github.com/robodatalab/cortexgrid/blob/main/k8s/docker/ray/Dockerfile) and `_WORKER_BAKED` in `_bundle.py`, which must list the same packages
|
|
94
95
|
- Injects MLflow/S3 credentials so task code running on the DGX can reach all services
|
|
95
96
|
|
|
96
97
|
##### Retries
|
|
@@ -126,9 +127,24 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
|
|
|
126
127
|
|
|
127
128
|
#### Model registry and serving
|
|
128
129
|
|
|
129
|
-
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application:
|
|
130
|
+
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
|
|
130
131
|
|
|
131
132
|
```python
|
|
133
|
+
from cortexgrid import serve
|
|
134
|
+
from fastapi import FastAPI
|
|
135
|
+
|
|
136
|
+
app = FastAPI()
|
|
137
|
+
|
|
138
|
+
@serve.ingress(app)
|
|
139
|
+
class MyServeApp:
|
|
140
|
+
num_gpus = 1
|
|
141
|
+
|
|
142
|
+
def __init__(self, family: str, suffix: str, run_name: str) -> None:
|
|
143
|
+
self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
|
|
144
|
+
|
|
145
|
+
@app.post("/complete")
|
|
146
|
+
async def complete(self, body: dict): ...
|
|
147
|
+
|
|
132
148
|
saved = cortexgrid.save_model(weights_dir, MyServeApp, family="qwen", suffix="instruct")
|
|
133
149
|
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True)
|
|
134
150
|
print(deployed.url)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cortexgrid"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.87"
|
|
4
4
|
description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
|
|
5
5
|
readme = "docs/cortexgrid/README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -9,6 +9,7 @@ requires-python = ">=3.11"
|
|
|
9
9
|
dependencies = [
|
|
10
10
|
"ray[default]>=2.9,<3",
|
|
11
11
|
"mlflow>=3.11,<4",
|
|
12
|
+
"packaging>=24",
|
|
12
13
|
"boto3>=1.34",
|
|
13
14
|
"setuptools>=82.0.1",
|
|
14
15
|
"tqdm>=4.60",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|