cortexgrid 0.2.92__tar.gz → 0.2.94__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/PKG-INFO +4 -2
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/__init__.py +9 -1
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/_serve_entry.py +38 -2
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/model_storage.py +67 -2
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/docs/cortexgrid/README.md +2 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/pyproject.toml +2 -2
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/.gitignore +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/LICENSE +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/model_serving.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.2.92 → cortexgrid-0.2.94}/cortexgrid/serve.py +0 -0
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.94
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
7
7
|
Project-URL: Documentation, https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/README.md
|
|
8
8
|
License-Expression: Apache-2.0
|
|
9
9
|
License-File: LICENSE
|
|
10
|
-
Requires-Python:
|
|
10
|
+
Requires-Python: <3.12,>=3.11
|
|
11
11
|
Requires-Dist: boto3>=1.34
|
|
12
12
|
Requires-Dist: cloudpickle>=3.0
|
|
13
13
|
Requires-Dist: fabric>=3.2.3
|
|
@@ -178,6 +178,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
178
178
|
print(deployed.url)
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
182
|
+
|
|
181
183
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
182
184
|
|
|
183
185
|
### API reference
|
|
@@ -74,8 +74,10 @@ from cortexgrid.ray_util import (
|
|
|
74
74
|
)
|
|
75
75
|
from cortexgrid.s3_util import delete_prefix, download, get_s3_client, upload, upload_dir
|
|
76
76
|
from cortexgrid.model_storage import (
|
|
77
|
+
IMPORTED,
|
|
77
78
|
SavedModel,
|
|
78
79
|
delete_model,
|
|
80
|
+
import_model,
|
|
79
81
|
list_models,
|
|
80
82
|
load_model,
|
|
81
83
|
model_registry_status,
|
|
@@ -119,7 +121,11 @@ def save_model(
|
|
|
119
121
|
weights_dir: str | Path, serve_app: type, family: str, suffix: str
|
|
120
122
|
) -> SavedModel:
|
|
121
123
|
"""Persist a weights directory under the current Experiment's run, paired
|
|
122
|
-
with the serve-app class that will front it at deploy time.
|
|
124
|
+
with the serve-app class that will front it at deploy time.
|
|
125
|
+
|
|
126
|
+
Every run saves a new copy under its own run_name - meant for weights the
|
|
127
|
+
run produced (e.g. a fine-tune). For a model produced elsewhere that should
|
|
128
|
+
be uploaded once and reused across runs, use `import_model`."""
|
|
123
129
|
experiment = Experiment.get_instance()
|
|
124
130
|
return _save_model_storage(
|
|
125
131
|
weights_dir,
|
|
@@ -179,8 +185,10 @@ __all__ = [
|
|
|
179
185
|
"list_secrets",
|
|
180
186
|
"delete_secret",
|
|
181
187
|
# Model registry
|
|
188
|
+
"IMPORTED",
|
|
182
189
|
"SavedModel",
|
|
183
190
|
"save_model",
|
|
191
|
+
"import_model",
|
|
184
192
|
"load_model",
|
|
185
193
|
"list_models",
|
|
186
194
|
"model_registry_status",
|
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
Ray Serve's REST `import_path` resolves to `cortexgrid._serve_entry:build`.
|
|
4
4
|
On the cluster replica, `build` imports the serve-app class bundled at
|
|
5
5
|
`save_model` time (its import path was stored as an MLflow tag), applies Ray's
|
|
6
|
-
ingress with the app it was marked with by `cortexgrid.serve.ingress
|
|
6
|
+
ingress with the app it was marked with by `cortexgrid.serve.ingress` (again on
|
|
7
|
+
each replica, see `_IngressOnReplica`), reads its
|
|
7
8
|
`num_gpus`/`num_replicas` class attributes for actor placement, wraps it as a
|
|
8
9
|
Ray Serve deployment, and binds it with the (family, suffix, run_name)
|
|
9
10
|
identifiers.
|
|
@@ -31,6 +32,33 @@ from ray.serve.deployment import Application
|
|
|
31
32
|
from cortexgrid.serve import ingress_app
|
|
32
33
|
|
|
33
34
|
|
|
35
|
+
# Requests one replica handles at once before Ray queues the rest.
|
|
36
|
+
_MAX_ONGOING_REQUESTS = 100
|
|
37
|
+
|
|
38
|
+
class _IngressOnReplica:
|
|
39
|
+
"""Mixin that re-applies Ray's ingress to `_serve_app` in the replica's own
|
|
40
|
+
process, as Ray creates the replica instance.
|
|
41
|
+
|
|
42
|
+
Ray's ingress rewrites the signature of each route method, in place, so
|
|
43
|
+
FastAPI injects the replica instance as `self`. `build` applies it in the
|
|
44
|
+
build process only. The replica imports the serve-app's module afresh, and
|
|
45
|
+
its route methods carry no rewrite. FastAPI < 0.137 analysed routes once,
|
|
46
|
+
in the build process, and the replica received the result. FastAPI >= 0.137
|
|
47
|
+
analyses them in the replica on the first request, and without the rewrite
|
|
48
|
+
reads `self` as a required query parameter (HTTP 422).
|
|
49
|
+
|
|
50
|
+
Hooked on `__new__`, not `__init__`: Ray calls `__new__` alone, before the
|
|
51
|
+
serve-app's `__init__`, whether that is sync or async."""
|
|
52
|
+
|
|
53
|
+
_serve_app: type
|
|
54
|
+
|
|
55
|
+
def __new__(cls, *args: Any, **kwargs: Any) -> Any:
|
|
56
|
+
# Applied for its side effect on the route methods; the wrapper it
|
|
57
|
+
# returns is not needed.
|
|
58
|
+
serve.ingress(ingress_app(cls._serve_app))(cls._serve_app)
|
|
59
|
+
return super().__new__(cls)
|
|
60
|
+
|
|
61
|
+
|
|
34
62
|
def build(args: dict[str, Any]) -> Application:
|
|
35
63
|
module_name, class_name = args["class_import_path"].split(":")
|
|
36
64
|
serve_app = getattr(importlib.import_module(module_name), class_name)
|
|
@@ -38,10 +66,18 @@ def build(args: dict[str, Any]) -> Application:
|
|
|
38
66
|
# mark and are deployed as they are.
|
|
39
67
|
app = ingress_app(serve_app)
|
|
40
68
|
if app is not None:
|
|
41
|
-
|
|
69
|
+
on_replica = type(
|
|
70
|
+
serve_app.__name__,
|
|
71
|
+
(_IngressOnReplica, serve_app),
|
|
72
|
+
{"_serve_app": serve_app},
|
|
73
|
+
)
|
|
74
|
+
serve_app = serve.ingress(app)(on_replica)
|
|
42
75
|
num_gpus = getattr(serve_app, "num_gpus", 0)
|
|
43
76
|
num_replicas = getattr(serve_app, "num_replicas", 1)
|
|
44
77
|
return serve.deployment(serve_app).options(
|
|
45
78
|
num_replicas=num_replicas,
|
|
79
|
+
# Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
|
|
80
|
+
# had on Ray 2.9.
|
|
81
|
+
max_ongoing_requests=_MAX_ONGOING_REQUESTS,
|
|
46
82
|
ray_actor_options={"num_gpus": num_gpus},
|
|
47
83
|
).bind(args["family"], args["suffix"], args["run_name"])
|
|
@@ -6,7 +6,14 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
|
|
|
6
6
|
family, suffix -> ModelVersion.tags["family"], ["suffix"] (denormalized)
|
|
7
7
|
weights blob path -> ModelVersion.source =
|
|
8
8
|
"s3://<bucket>/models/<run_name>/<family>/<suffix>/weights/"
|
|
9
|
-
run linkage -> ModelVersion.run_id (built-in MLflow field
|
|
9
|
+
run linkage -> ModelVersion.run_id (built-in MLflow field; unset
|
|
10
|
+
for imported models)
|
|
11
|
+
|
|
12
|
+
Two ways in: `save_model` registers a fresh copy under the calling run's
|
|
13
|
+
run_name every time it runs (fine-tuned output); `import_model` registers a
|
|
14
|
+
model produced elsewhere once, under the fixed run_name IMPORTED, and is a
|
|
15
|
+
no-op after that. Both write the same layout, so every
|
|
16
|
+
(family, suffix, run_name) consumer - load_model, deploy_model - handles both.
|
|
10
17
|
|
|
11
18
|
storage.py is pure: it takes run_id/run_name as explicit args and never reads
|
|
12
19
|
the active Experiment singleton. The facade that fills those in lives in
|
|
@@ -20,7 +27,7 @@ from dataclasses import dataclass
|
|
|
20
27
|
from datetime import datetime, timedelta, timezone
|
|
21
28
|
from pathlib import Path
|
|
22
29
|
import tempfile
|
|
23
|
-
from typing import Any
|
|
30
|
+
from typing import Any, Callable
|
|
24
31
|
|
|
25
32
|
from mlflow.exceptions import MlflowException
|
|
26
33
|
from mlflow.tracking import MlflowClient
|
|
@@ -49,6 +56,11 @@ _PHASE_BROKEN = "broken"
|
|
|
49
56
|
# derived lazily on read (see `_phase_for`); nothing is written back.
|
|
50
57
|
_UPLOAD_DEADLINE = timedelta(hours=3)
|
|
51
58
|
|
|
59
|
+
# run_name under which `import_model` registers a model: imported weights belong
|
|
60
|
+
# to no run, so they share one fixed key and outlive the run that imported
|
|
61
|
+
# them. Run names are haikunator "word-word-NN", so no run can take this name.
|
|
62
|
+
IMPORTED = "imported"
|
|
63
|
+
|
|
52
64
|
|
|
53
65
|
@dataclass
|
|
54
66
|
class SavedModel:
|
|
@@ -137,6 +149,10 @@ def save_model(
|
|
|
137
149
|
"""Upload a weights directory to S3 and register a new MLflow ModelVersion
|
|
138
150
|
paired with the serve-app that fronts it.
|
|
139
151
|
|
|
152
|
+
Meant for weights the calling run produced (e.g. a fine-tune): every run
|
|
153
|
+
saves its own copy under its own run_name, so running the same code twice
|
|
154
|
+
yields two models. For weights produced elsewhere, use `import_model`.
|
|
155
|
+
|
|
140
156
|
cortexgrid stores the weights as an opaque directory: it never inspects,
|
|
141
157
|
serializes, or reconstructs their contents, so the on-disk format
|
|
142
158
|
(HuggingFace `save_pretrained`, `torch.save`, ONNX, anything) is entirely
|
|
@@ -148,6 +164,55 @@ def save_model(
|
|
|
148
164
|
Its code is bundled and its import path, bundle URL, and pip list are
|
|
149
165
|
stored as tags on the ModelVersion so `deploy_model` can bind it later
|
|
150
166
|
without the caller holding the class object."""
|
|
167
|
+
return _upload_model(weights_dir, serve_app, suffix, family, run_id, run_name)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def import_model(
|
|
171
|
+
source: str | Path | Callable[[], str | Path],
|
|
172
|
+
serve_app: type,
|
|
173
|
+
family: str,
|
|
174
|
+
suffix: str,
|
|
175
|
+
) -> SavedModel:
|
|
176
|
+
"""Register a model produced elsewhere (e.g. a pretrained base model) under
|
|
177
|
+
the fixed key (family, suffix, IMPORTED), once.
|
|
178
|
+
|
|
179
|
+
`source` is the weights directory, or a callable returning it; the callable
|
|
180
|
+
runs only when the upload actually happens, so an expensive download can be
|
|
181
|
+
skipped on every run after the first.
|
|
182
|
+
|
|
183
|
+
If a version is already registered under the key:
|
|
184
|
+
- "ready": no-op, returns it. `source` and `serve_app` are ignored; to
|
|
185
|
+
replace the weights or the serve-app, `delete_model` it first.
|
|
186
|
+
- "uploading": raises RuntimeError - another process is importing it.
|
|
187
|
+
- "upload_failed" / "broken": deleted and imported again.
|
|
188
|
+
|
|
189
|
+
The version is linked to no MLflow run, so deleting a run leaves it in
|
|
190
|
+
place. Deploy it like any saved model:
|
|
191
|
+
`deploy_model(family, suffix, IMPORTED)`."""
|
|
192
|
+
existing = model_registry_status(family, suffix, IMPORTED)
|
|
193
|
+
if existing is not None:
|
|
194
|
+
if existing.phase == _PHASE_READY:
|
|
195
|
+
return existing
|
|
196
|
+
if existing.phase == _PHASE_UPLOADING:
|
|
197
|
+
raise RuntimeError(
|
|
198
|
+
f"Model {family}/{suffix}/{IMPORTED} is being imported by "
|
|
199
|
+
"another process"
|
|
200
|
+
)
|
|
201
|
+
delete_model(family, suffix, IMPORTED)
|
|
202
|
+
weights_dir = source() if callable(source) else source
|
|
203
|
+
return _upload_model(weights_dir, serve_app, suffix, family, None, IMPORTED)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _upload_model(
|
|
207
|
+
weights_dir: str | Path,
|
|
208
|
+
serve_app: type,
|
|
209
|
+
suffix: str,
|
|
210
|
+
family: str,
|
|
211
|
+
run_id: str | None,
|
|
212
|
+
run_name: str,
|
|
213
|
+
) -> SavedModel:
|
|
214
|
+
"""Register a ModelVersion in "uploading", upload the weights and the
|
|
215
|
+
serve-app bundle, and flip it to "ready" (or "upload_failed")."""
|
|
151
216
|
bucket = get_s3_bucket()
|
|
152
217
|
prefix = f"models/{run_name}/{family}/{suffix}"
|
|
153
218
|
size_bytes = _dir_size_bytes(weights_dir)
|
|
@@ -150,6 +150,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
150
150
|
print(deployed.url)
|
|
151
151
|
```
|
|
152
152
|
|
|
153
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
154
|
+
|
|
153
155
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
154
156
|
|
|
155
157
|
### API reference
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cortexgrid"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.94"
|
|
4
4
|
description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
|
|
5
5
|
readme = "docs/cortexgrid/README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
8
|
-
requires-python = ">=3.11"
|
|
8
|
+
requires-python = ">=3.11,<3.12"
|
|
9
9
|
dependencies = [
|
|
10
10
|
"ray[default]>=2.9,<3",
|
|
11
11
|
"mlflow>=3.11,<4",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|