cortexgrid 0.2.93__tar.gz → 0.2.95__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/PKG-INFO +3 -1
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/__init__.py +30 -1
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/model_storage.py +83 -10
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/docs/cortexgrid/README.md +2 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/pyproject.toml +1 -1
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/.gitignore +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/LICENSE +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/model_serving.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.2.93 → cortexgrid-0.2.95}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.95
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -178,6 +178,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
178
178
|
print(deployed.url)
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
182
|
+
|
|
181
183
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
182
184
|
|
|
183
185
|
### API reference
|
|
@@ -74,12 +74,14 @@ from cortexgrid.ray_util import (
|
|
|
74
74
|
)
|
|
75
75
|
from cortexgrid.s3_util import delete_prefix, download, get_s3_client, upload, upload_dir
|
|
76
76
|
from cortexgrid.model_storage import (
|
|
77
|
+
IMPORTED,
|
|
77
78
|
SavedModel,
|
|
78
79
|
delete_model,
|
|
79
80
|
list_models,
|
|
80
81
|
load_model,
|
|
81
82
|
model_registry_status,
|
|
82
83
|
)
|
|
84
|
+
from cortexgrid.model_storage import import_model as _import_model_storage
|
|
83
85
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
84
86
|
from cortexgrid.model_serving import (
|
|
85
87
|
Deployment,
|
|
@@ -119,7 +121,11 @@ def save_model(
|
|
|
119
121
|
weights_dir: str | Path, serve_app: type, family: str, suffix: str
|
|
120
122
|
) -> SavedModel:
|
|
121
123
|
"""Persist a weights directory under the current Experiment's run, paired
|
|
122
|
-
with the serve-app class that will front it at deploy time.
|
|
124
|
+
with the serve-app class that will front it at deploy time.
|
|
125
|
+
|
|
126
|
+
Every run saves a new copy under its own run_name - meant for weights the
|
|
127
|
+
run produced (e.g. a fine-tune). For a model produced elsewhere that should
|
|
128
|
+
be uploaded once and reused across runs, use `import_model`."""
|
|
123
129
|
experiment = Experiment.get_instance()
|
|
124
130
|
return _save_model_storage(
|
|
125
131
|
weights_dir,
|
|
@@ -131,6 +137,27 @@ def save_model(
|
|
|
131
137
|
)
|
|
132
138
|
|
|
133
139
|
|
|
140
|
+
def import_model(
|
|
141
|
+
source: str | Path | Callable[[], str | Path],
|
|
142
|
+
serve_app: type,
|
|
143
|
+
family: str,
|
|
144
|
+
suffix: str,
|
|
145
|
+
) -> SavedModel:
|
|
146
|
+
"""Register a model produced elsewhere once, reuse it on every later call,
|
|
147
|
+
and record on the current Experiment's run which imported model it used.
|
|
148
|
+
|
|
149
|
+
The model belongs to no run (see `cortexgrid.model_storage.import_model`),
|
|
150
|
+
so the run keeps the link instead: the tag
|
|
151
|
+
`imported_model/<family>/<suffix>` holds the version's `created_at`, set
|
|
152
|
+
whether this call uploaded the model or reused it."""
|
|
153
|
+
experiment = Experiment.get_instance()
|
|
154
|
+
model = _import_model_storage(source, serve_app, family, suffix)
|
|
155
|
+
get_mlflow_client().set_tag(
|
|
156
|
+
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
157
|
+
)
|
|
158
|
+
return model
|
|
159
|
+
|
|
160
|
+
|
|
134
161
|
__all__ = [
|
|
135
162
|
"Experiment",
|
|
136
163
|
"delete_experiment",
|
|
@@ -179,8 +206,10 @@ __all__ = [
|
|
|
179
206
|
"list_secrets",
|
|
180
207
|
"delete_secret",
|
|
181
208
|
# Model registry
|
|
209
|
+
"IMPORTED",
|
|
182
210
|
"SavedModel",
|
|
183
211
|
"save_model",
|
|
212
|
+
"import_model",
|
|
184
213
|
"load_model",
|
|
185
214
|
"list_models",
|
|
186
215
|
"model_registry_status",
|
|
@@ -6,7 +6,14 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
|
|
|
6
6
|
family, suffix -> ModelVersion.tags["family"], ["suffix"] (denormalized)
|
|
7
7
|
weights blob path -> ModelVersion.source =
|
|
8
8
|
"s3://<bucket>/models/<run_name>/<family>/<suffix>/weights/"
|
|
9
|
-
run linkage -> ModelVersion.run_id (built-in MLflow field
|
|
9
|
+
run linkage -> ModelVersion.run_id (built-in MLflow field; unset
|
|
10
|
+
for imported models)
|
|
11
|
+
|
|
12
|
+
Two ways in: `save_model` registers a fresh copy under the calling run's
|
|
13
|
+
run_name every time it runs (fine-tuned output); `import_model` registers a
|
|
14
|
+
model produced elsewhere once, under the fixed run_name IMPORTED, and is a
|
|
15
|
+
no-op after that. Both write the same layout, so every
|
|
16
|
+
(family, suffix, run_name) consumer - load_model, deploy_model - handles both.
|
|
10
17
|
|
|
11
18
|
storage.py is pure: it takes run_id/run_name as explicit args and never reads
|
|
12
19
|
the active Experiment singleton. The facade that fills those in lives in
|
|
@@ -20,7 +27,7 @@ from dataclasses import dataclass
|
|
|
20
27
|
from datetime import datetime, timedelta, timezone
|
|
21
28
|
from pathlib import Path
|
|
22
29
|
import tempfile
|
|
23
|
-
from typing import Any
|
|
30
|
+
from typing import Any, Callable
|
|
24
31
|
|
|
25
32
|
from mlflow.exceptions import MlflowException
|
|
26
33
|
from mlflow.tracking import MlflowClient
|
|
@@ -43,12 +50,18 @@ _PHASE_UPLOAD_FAILED = "upload_failed"
|
|
|
43
50
|
_PHASE_BROKEN = "broken"
|
|
44
51
|
|
|
45
52
|
# An upload still marked "uploading" this long after the version was created is
|
|
46
|
-
# treated as broken:
|
|
47
|
-
#
|
|
48
|
-
# dies mid-
|
|
53
|
+
# treated as broken: the version is created before the weights are resolved
|
|
54
|
+
# (for `import_model`, before its download), so creation_timestamp is the
|
|
55
|
+
# start of the whole upload, and a process that dies mid-way never flips the
|
|
56
|
+
# tag to "ready"/"upload_failed". Expiry is
|
|
49
57
|
# derived lazily on read (see `_phase_for`); nothing is written back.
|
|
50
58
|
_UPLOAD_DEADLINE = timedelta(hours=3)
|
|
51
59
|
|
|
60
|
+
# run_name under which `import_model` registers a model: imported weights belong
|
|
61
|
+
# to no run, so they share one fixed key and outlive the run that imported
|
|
62
|
+
# them. Run names are haikunator "word-word-NN", so no run can take this name.
|
|
63
|
+
IMPORTED = "imported"
|
|
64
|
+
|
|
52
65
|
|
|
53
66
|
@dataclass
|
|
54
67
|
class SavedModel:
|
|
@@ -137,6 +150,10 @@ def save_model(
|
|
|
137
150
|
"""Upload a weights directory to S3 and register a new MLflow ModelVersion
|
|
138
151
|
paired with the serve-app that fronts it.
|
|
139
152
|
|
|
153
|
+
Meant for weights the calling run produced (e.g. a fine-tune): every run
|
|
154
|
+
saves its own copy under its own run_name, so running the same code twice
|
|
155
|
+
yields two models. For weights produced elsewhere, use `import_model`.
|
|
156
|
+
|
|
140
157
|
cortexgrid stores the weights as an opaque directory: it never inspects,
|
|
141
158
|
serializes, or reconstructs their contents, so the on-disk format
|
|
142
159
|
(HuggingFace `save_pretrained`, `torch.save`, ONNX, anything) is entirely
|
|
@@ -148,16 +165,69 @@ def save_model(
|
|
|
148
165
|
Its code is bundled and its import path, bundle URL, and pip list are
|
|
149
166
|
stored as tags on the ModelVersion so `deploy_model` can bind it later
|
|
150
167
|
without the caller holding the class object."""
|
|
168
|
+
return _upload_model(weights_dir, serve_app, suffix, family, run_id, run_name)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def import_model(
|
|
172
|
+
source: str | Path | Callable[[], str | Path],
|
|
173
|
+
serve_app: type,
|
|
174
|
+
family: str,
|
|
175
|
+
suffix: str,
|
|
176
|
+
) -> SavedModel:
|
|
177
|
+
"""Register a model produced elsewhere (e.g. a pretrained base model) under
|
|
178
|
+
the fixed key (family, suffix, IMPORTED), once.
|
|
179
|
+
|
|
180
|
+
`source` is the weights directory, or a callable returning it; the callable
|
|
181
|
+
runs only when the upload actually happens, so an expensive download can be
|
|
182
|
+
skipped on every run after the first. It runs after the version is
|
|
183
|
+
registered as "uploading", so a concurrent import sees this one in flight
|
|
184
|
+
while it downloads.
|
|
185
|
+
|
|
186
|
+
If a version is already registered under the key:
|
|
187
|
+
- "ready": no-op, returns it. `source` and `serve_app` are ignored; to
|
|
188
|
+
replace the weights or the serve-app, `delete_model` it first.
|
|
189
|
+
- "uploading": raises RuntimeError - another process is importing it.
|
|
190
|
+
- "upload_failed" / "broken": deleted and imported again.
|
|
191
|
+
|
|
192
|
+
The version is linked to no MLflow run, so deleting a run leaves it in
|
|
193
|
+
place. Deploy it like any saved model:
|
|
194
|
+
`deploy_model(family, suffix, IMPORTED)`."""
|
|
195
|
+
existing = model_registry_status(family, suffix, IMPORTED)
|
|
196
|
+
if existing is not None:
|
|
197
|
+
if existing.phase == _PHASE_READY:
|
|
198
|
+
return existing
|
|
199
|
+
if existing.phase == _PHASE_UPLOADING:
|
|
200
|
+
raise RuntimeError(
|
|
201
|
+
f"Model {family}/{suffix}/{IMPORTED} is being imported by "
|
|
202
|
+
"another process"
|
|
203
|
+
)
|
|
204
|
+
delete_model(family, suffix, IMPORTED)
|
|
205
|
+
return _upload_model(source, serve_app, suffix, family, None, IMPORTED)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _upload_model(
|
|
209
|
+
weights: str | Path | Callable[[], str | Path],
|
|
210
|
+
serve_app: type,
|
|
211
|
+
suffix: str,
|
|
212
|
+
family: str,
|
|
213
|
+
run_id: str | None,
|
|
214
|
+
run_name: str,
|
|
215
|
+
) -> SavedModel:
|
|
216
|
+
"""Register a ModelVersion in "uploading", resolve `weights` to a directory
|
|
217
|
+
(calling it when it is a callable), upload the weights and the serve-app
|
|
218
|
+
bundle, and flip it to "ready" (or "upload_failed")."""
|
|
151
219
|
bucket = get_s3_bucket()
|
|
152
220
|
prefix = f"models/{run_name}/{family}/{suffix}"
|
|
153
|
-
size_bytes = _dir_size_bytes(weights_dir)
|
|
154
221
|
source = f"s3://{bucket}/{prefix}/weights/"
|
|
155
222
|
name = f"{family}__{suffix}"
|
|
156
223
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
157
224
|
_ensure_registered_model(client, name)
|
|
158
|
-
# Register the version up front in the "uploading" phase
|
|
159
|
-
#
|
|
160
|
-
#
|
|
225
|
+
# Register the version up front in the "uploading" phase, before `weights`
|
|
226
|
+
# is resolved: the dashboard surfaces the model while it is still being
|
|
227
|
+
# fetched and uploaded, and a concurrent `import_model` sees the import in
|
|
228
|
+
# flight for the whole download instead of starting one of its own. The
|
|
229
|
+
# size is stamped once the directory exists; the bundle tags and the flip
|
|
230
|
+
# to "ready" happen only after the upload lands.
|
|
161
231
|
version = client.create_model_version(
|
|
162
232
|
name=name,
|
|
163
233
|
source=source,
|
|
@@ -166,11 +236,14 @@ def save_model(
|
|
|
166
236
|
"family": family,
|
|
167
237
|
"suffix": suffix,
|
|
168
238
|
"run_name": run_name,
|
|
169
|
-
"size_bytes": str(size_bytes),
|
|
170
239
|
_LIFECYCLE_TAG: _PHASE_UPLOADING,
|
|
171
240
|
},
|
|
172
241
|
)
|
|
173
242
|
try:
|
|
243
|
+
weights_dir = weights() if callable(weights) else weights
|
|
244
|
+
client.set_model_version_tag(
|
|
245
|
+
name, version.version, "size_bytes", str(_dir_size_bytes(weights_dir))
|
|
246
|
+
)
|
|
174
247
|
s3_util.upload_dir(str(weights_dir), dest_path=f"{prefix}/weights")
|
|
175
248
|
bundle_meta = bundle_class(serve_app, family, suffix, run_name)
|
|
176
249
|
for key, value in metadata_to_tags(bundle_meta).items():
|
|
@@ -150,6 +150,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
150
150
|
print(deployed.url)
|
|
151
151
|
```
|
|
152
152
|
|
|
153
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
154
|
+
|
|
153
155
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
154
156
|
|
|
155
157
|
### API reference
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|