cortexgrid 0.2.94__tar.gz → 0.2.96__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/PKG-INFO +2 -2
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/__init__.py +22 -1
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/_bundle.py +21 -1
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/model_serving.py +87 -27
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/model_storage.py +54 -18
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/docs/cortexgrid/README.md +1 -1
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/pyproject.toml +1 -1
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/.gitignore +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/LICENSE +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.2.94 → cortexgrid-0.2.96}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.96
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -178,7 +178,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
178
178
|
print(deployed.url)
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
-
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and
|
|
181
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
182
182
|
|
|
183
183
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
184
184
|
|
|
@@ -77,11 +77,11 @@ from cortexgrid.model_storage import (
|
|
|
77
77
|
IMPORTED,
|
|
78
78
|
SavedModel,
|
|
79
79
|
delete_model,
|
|
80
|
-
import_model,
|
|
81
80
|
list_models,
|
|
82
81
|
load_model,
|
|
83
82
|
model_registry_status,
|
|
84
83
|
)
|
|
84
|
+
from cortexgrid.model_storage import import_model as _import_model_storage
|
|
85
85
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
86
86
|
from cortexgrid.model_serving import (
|
|
87
87
|
Deployment,
|
|
@@ -137,6 +137,27 @@ def save_model(
|
|
|
137
137
|
)
|
|
138
138
|
|
|
139
139
|
|
|
140
|
+
def import_model(
|
|
141
|
+
source: str | Path | Callable[[], str | Path],
|
|
142
|
+
serve_app: type,
|
|
143
|
+
family: str,
|
|
144
|
+
suffix: str,
|
|
145
|
+
) -> SavedModel:
|
|
146
|
+
"""Register a model produced elsewhere once, reuse it on every later call,
|
|
147
|
+
and record on the current Experiment's run which imported model it used.
|
|
148
|
+
|
|
149
|
+
The model belongs to no run (see `cortexgrid.model_storage.import_model`),
|
|
150
|
+
so the run keeps the link instead: the tag
|
|
151
|
+
`imported_model/<family>/<suffix>` holds the version's `created_at`, set
|
|
152
|
+
whether this call uploaded the model or reused it."""
|
|
153
|
+
experiment = Experiment.get_instance()
|
|
154
|
+
model = _import_model_storage(source, serve_app, family, suffix)
|
|
155
|
+
get_mlflow_client().set_tag(
|
|
156
|
+
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
157
|
+
)
|
|
158
|
+
return model
|
|
159
|
+
|
|
160
|
+
|
|
140
161
|
__all__ = [
|
|
141
162
|
"Experiment",
|
|
142
163
|
"delete_experiment",
|
|
@@ -12,6 +12,8 @@ Bundles of several seeds combine with `BundleDesc.merge`.
|
|
|
12
12
|
|
|
13
13
|
`stage(files, dest)` lays a bundle out under `dest` at each file's import path,
|
|
14
14
|
so `dest` on sys.path (e.g. a Ray working_dir) makes every module importable.
|
|
15
|
+
`digest(files)` hashes that layout, so bundles that stage identically compare
|
|
16
|
+
equal wherever their files live.
|
|
15
17
|
|
|
16
18
|
`BundleDesc.pip_requirements(worker_provides())` pins the third-party
|
|
17
19
|
distributions the Ray worker image does not already have, for a Ray `pip`
|
|
@@ -24,6 +26,7 @@ import ast
|
|
|
24
26
|
from collections.abc import Iterable, Iterator
|
|
25
27
|
from dataclasses import dataclass
|
|
26
28
|
import functools
|
|
29
|
+
import hashlib
|
|
27
30
|
import importlib.machinery
|
|
28
31
|
import importlib.metadata
|
|
29
32
|
import importlib.util
|
|
@@ -105,11 +108,23 @@ def stage(files: set[Path], dest: Path) -> None:
|
|
|
105
108
|
in installed distributions)."""
|
|
106
109
|
dest.mkdir(parents=True, exist_ok=True)
|
|
107
110
|
for file in files:
|
|
108
|
-
target = dest /
|
|
111
|
+
target = dest / _import_path(file)
|
|
109
112
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
110
113
|
shutil.copy2(file, target)
|
|
111
114
|
|
|
112
115
|
|
|
116
|
+
def digest(files: set[Path]) -> str:
|
|
117
|
+
"""SHA-256 of `files` as `stage` lays them out: each file's import path and
|
|
118
|
+
contents, in import-path order. Files that stage identically digest
|
|
119
|
+
identically, wherever they live on disk."""
|
|
120
|
+
sha = hashlib.sha256()
|
|
121
|
+
for path, file in sorted((_import_path(file).as_posix(), file) for file in files):
|
|
122
|
+
content = file.read_bytes()
|
|
123
|
+
sha.update(f"{path}\0{len(content)}\0".encode())
|
|
124
|
+
sha.update(content)
|
|
125
|
+
return sha.hexdigest()
|
|
126
|
+
|
|
127
|
+
|
|
113
128
|
# What the Ray worker image pip-installs, as the Dockerfile spells it. They and
|
|
114
129
|
# their dependency trees are on the worker already, so they are never installed
|
|
115
130
|
# there again -- a second copy in the job's virtualenv would shadow the image's.
|
|
@@ -335,6 +350,11 @@ def _package(file: Path) -> str:
|
|
|
335
350
|
return ".".join(parts)
|
|
336
351
|
|
|
337
352
|
|
|
353
|
+
def _import_path(file: Path) -> Path:
|
|
354
|
+
"""`file` relative to the sys.path entry it is imported from."""
|
|
355
|
+
return file.relative_to(_sys_path_root(file))
|
|
356
|
+
|
|
357
|
+
|
|
338
358
|
def _sys_path_root(file: Path) -> Path:
|
|
339
359
|
"""The sys.path entry `file` is imported from: its first ancestor directory
|
|
340
360
|
without an __init__.py."""
|
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
Caller stays HTTP-only: deploy/undeploy/list talk to the Ray dashboard's
|
|
4
4
|
declarative `/api/serve/applications/` endpoint via [cortexgrid.ray_util],
|
|
5
5
|
never `ray.init`. The deployment class is bundled at `save_model` time, zipped,
|
|
6
|
-
uploaded to MinIO under
|
|
6
|
+
uploaded to MinIO under
|
|
7
|
+
`serve-bundles/<run_name>/<family>__<suffix>/<fingerprint>.zip`, and
|
|
7
8
|
referenced via `runtime_env.working_dir` so Ray workers fetch it from there.
|
|
8
|
-
The bundle URL, class import path,
|
|
9
|
-
on the ModelVersion so `deploy_model` can find them later without
|
|
10
|
-
holding the class object.
|
|
9
|
+
The bundle URL, class import path, pip list, and fingerprint are persisted as
|
|
10
|
+
MLflow tags on the ModelVersion so `deploy_model` can find them later without
|
|
11
|
+
the caller holding the class object.
|
|
11
12
|
|
|
12
13
|
Naming: the Ray Serve application is named "<family>__<suffix>__<run_name>".
|
|
13
14
|
This relies on family/suffix/run_name not containing the literal "__".
|
|
@@ -26,13 +27,14 @@ import shutil
|
|
|
26
27
|
import tempfile
|
|
27
28
|
import time
|
|
28
29
|
from dataclasses import dataclass, field
|
|
30
|
+
import hashlib
|
|
29
31
|
from pathlib import Path
|
|
30
32
|
from typing import Any
|
|
31
33
|
|
|
32
34
|
from mlflow.tracking import MlflowClient
|
|
33
35
|
from ray.serve.schema import ApplicationStatus
|
|
34
36
|
|
|
35
|
-
from cortexgrid._bundle import bundle, stage, worker_provides
|
|
37
|
+
from cortexgrid._bundle import bundle, digest, stage, worker_provides
|
|
36
38
|
from cortexgrid.infra import get_mlflow_tracking_uri, get_ray_serve_uri
|
|
37
39
|
from cortexgrid.ray_util import (
|
|
38
40
|
get_serve_details,
|
|
@@ -95,17 +97,28 @@ class BundleMetadata:
|
|
|
95
97
|
# pinned third-party requirements the replica pip-installs (the bundle's
|
|
96
98
|
# distributions the Ray image does not already provide)
|
|
97
99
|
pip_requirements: list[str] = field(default_factory=list)
|
|
100
|
+
# ServeBundle.fingerprint of the uploaded bundle; empty for models saved
|
|
101
|
+
# before bundles were fingerprinted
|
|
102
|
+
fingerprint: str = ""
|
|
98
103
|
|
|
99
104
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
upload to MinIO.
|
|
105
|
+
@dataclass
|
|
106
|
+
class ServeBundle:
|
|
107
|
+
"""A serve-app's bundle, resolved locally but not uploaded yet: what
|
|
108
|
+
`build_bundle` finds and `upload_bundle` ships."""
|
|
105
109
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
110
|
+
files: set[Path]
|
|
111
|
+
class_import_path: str
|
|
112
|
+
pip_requirements: list[str]
|
|
113
|
+
# Hash of everything the replica runs: the staged files, the class it
|
|
114
|
+
# imports, and the requirements it installs. Equal fingerprints mean the
|
|
115
|
+
# same code, so an uploaded bundle can be reused.
|
|
116
|
+
fingerprint: str
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def build_bundle(cls: type) -> ServeBundle:
|
|
120
|
+
"""Resolve the serve-app class's code (and the serve entrypoint) into a
|
|
121
|
+
bundle, without uploading it.
|
|
109
122
|
|
|
110
123
|
Raises ValueError for a class wrapped by `ray.serve.ingress`: that wrapper
|
|
111
124
|
is a subclass Ray defines in its own module, and on older Ray (e.g. 2.9) it
|
|
@@ -120,17 +133,51 @@ def bundle_class(
|
|
|
120
133
|
entry_file = Path(inspect.getfile(cls)).resolve()
|
|
121
134
|
serve_entry = Path(__file__).with_name("_serve_entry.py")
|
|
122
135
|
desc = bundle(entry_file).merge(bundle(serve_entry))
|
|
136
|
+
class_import_path = f"{cls.__module__}:{cls.__name__}"
|
|
123
137
|
pip_requirements = desc.pip_requirements(worker_provides())
|
|
138
|
+
fingerprint = hashlib.sha256(
|
|
139
|
+
json.dumps(
|
|
140
|
+
[digest(desc.local_files), class_import_path, pip_requirements]
|
|
141
|
+
).encode()
|
|
142
|
+
).hexdigest()
|
|
143
|
+
return ServeBundle(
|
|
144
|
+
files=desc.local_files,
|
|
145
|
+
class_import_path=class_import_path,
|
|
146
|
+
pip_requirements=pip_requirements,
|
|
147
|
+
fingerprint=fingerprint,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def bundle_class(
|
|
152
|
+
cls: type, family: str, suffix: str, run_name: str
|
|
153
|
+
) -> BundleMetadata:
|
|
154
|
+
"""Bundle the serve-app class's code (and the serve entrypoint), zip it, and
|
|
155
|
+
upload to MinIO: `build_bundle` followed by `upload_bundle`.
|
|
156
|
+
|
|
157
|
+
Returns the metadata `deploy_model` needs later; callers (typically
|
|
158
|
+
`save_model`) persist it on the ModelVersion so the deploy step can run
|
|
159
|
+
without holding the class object."""
|
|
160
|
+
return upload_bundle(build_bundle(cls), family, suffix, run_name)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def upload_bundle(
|
|
164
|
+
serve_bundle: ServeBundle, family: str, suffix: str, run_name: str
|
|
165
|
+
) -> BundleMetadata:
|
|
166
|
+
"""Zip a built bundle and upload it under its fingerprint.
|
|
167
|
+
|
|
168
|
+
The fingerprint is part of the URL because Ray keeps a remote working_dir
|
|
169
|
+
it has downloaded and reuses it for the same URL: new code at an old URL
|
|
170
|
+
would never reach a replica."""
|
|
124
171
|
with tempfile.TemporaryDirectory() as tmp:
|
|
125
172
|
code_root = Path(tmp) / "code"
|
|
126
|
-
stage(
|
|
173
|
+
stage(serve_bundle.files, code_root)
|
|
127
174
|
log.info(
|
|
128
175
|
"Serve bundle for %s/%s/%s: %d files, pip: %s",
|
|
129
176
|
family,
|
|
130
177
|
suffix,
|
|
131
178
|
run_name,
|
|
132
|
-
len(
|
|
133
|
-
pip_requirements,
|
|
179
|
+
len(serve_bundle.files),
|
|
180
|
+
serve_bundle.pip_requirements,
|
|
134
181
|
)
|
|
135
182
|
# Ray unpacks a remote (s3://) working_dir zip by stripping its
|
|
136
183
|
# top-level directory when there is exactly one, so a bundle of a single
|
|
@@ -146,12 +193,16 @@ def bundle_class(
|
|
|
146
193
|
)
|
|
147
194
|
bundle_url = upload(
|
|
148
195
|
archive,
|
|
149
|
-
dest_path=
|
|
196
|
+
dest_path=(
|
|
197
|
+
f"serve-bundles/{run_name}/{family}__{suffix}/"
|
|
198
|
+
f"{serve_bundle.fingerprint}.zip"
|
|
199
|
+
),
|
|
150
200
|
)
|
|
151
201
|
return BundleMetadata(
|
|
152
202
|
bundle_url=bundle_url,
|
|
153
|
-
class_import_path=
|
|
154
|
-
pip_requirements=pip_requirements,
|
|
203
|
+
class_import_path=serve_bundle.class_import_path,
|
|
204
|
+
pip_requirements=serve_bundle.pip_requirements,
|
|
205
|
+
fingerprint=serve_bundle.fingerprint,
|
|
155
206
|
)
|
|
156
207
|
|
|
157
208
|
|
|
@@ -189,19 +240,34 @@ def _build_application_spec(
|
|
|
189
240
|
_CLASS_IMPORT_PATH_TAG = "class_import_path"
|
|
190
241
|
_BUNDLE_URL_TAG = "serve_bundle_url"
|
|
191
242
|
_PIP_REQUIREMENTS_TAG = "serve_pip_requirements"
|
|
243
|
+
_BUNDLE_FINGERPRINT_TAG = "serve_bundle_fingerprint"
|
|
192
244
|
|
|
193
245
|
|
|
194
246
|
def metadata_to_tags(meta: BundleMetadata) -> dict[str, str]:
|
|
195
247
|
"""Serialise BundleMetadata to MLflow tags. The inverse of
|
|
196
|
-
`
|
|
248
|
+
`metadata_from_tags`; lives here next to the consumer so the tag schema
|
|
197
249
|
stays in one place."""
|
|
198
250
|
return {
|
|
199
251
|
_CLASS_IMPORT_PATH_TAG: meta.class_import_path,
|
|
200
252
|
_BUNDLE_URL_TAG: meta.bundle_url,
|
|
201
253
|
_PIP_REQUIREMENTS_TAG: json.dumps(meta.pip_requirements),
|
|
254
|
+
_BUNDLE_FINGERPRINT_TAG: meta.fingerprint,
|
|
202
255
|
}
|
|
203
256
|
|
|
204
257
|
|
|
258
|
+
def metadata_from_tags(tags: dict[str, str]) -> BundleMetadata:
|
|
259
|
+
"""Deserialise BundleMetadata from a ModelVersion's MLflow tags. Raises
|
|
260
|
+
KeyError for a missing bundle URL or class import path."""
|
|
261
|
+
return BundleMetadata(
|
|
262
|
+
bundle_url=tags[_BUNDLE_URL_TAG],
|
|
263
|
+
class_import_path=tags[_CLASS_IMPORT_PATH_TAG],
|
|
264
|
+
# Absent on models saved before dependencies were pip-installed.
|
|
265
|
+
pip_requirements=json.loads(tags.get(_PIP_REQUIREMENTS_TAG, "[]")),
|
|
266
|
+
# Absent on models saved before bundles were fingerprinted.
|
|
267
|
+
fingerprint=tags.get(_BUNDLE_FINGERPRINT_TAG, ""),
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
|
|
205
271
|
def _load_bundle_metadata(
|
|
206
272
|
family: str, suffix: str, run_name: str
|
|
207
273
|
) -> BundleMetadata:
|
|
@@ -215,14 +281,8 @@ def _load_bundle_metadata(
|
|
|
215
281
|
raise ValueError(
|
|
216
282
|
f"No saved model for {family}/{suffix}/{run_name}; cannot deploy."
|
|
217
283
|
)
|
|
218
|
-
tags = versions[0].tags or {}
|
|
219
284
|
try:
|
|
220
|
-
return
|
|
221
|
-
bundle_url=tags[_BUNDLE_URL_TAG],
|
|
222
|
-
class_import_path=tags[_CLASS_IMPORT_PATH_TAG],
|
|
223
|
-
# Absent on models saved before dependencies were pip-installed.
|
|
224
|
-
pip_requirements=json.loads(tags.get(_PIP_REQUIREMENTS_TAG, "[]")),
|
|
225
|
-
)
|
|
285
|
+
return metadata_from_tags(versions[0].tags or {})
|
|
226
286
|
except KeyError as exc:
|
|
227
287
|
raise ValueError(
|
|
228
288
|
f"Saved model {family}/{suffix}/{run_name} is missing the deployment "
|
|
@@ -11,8 +11,9 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
|
|
|
11
11
|
|
|
12
12
|
Two ways in: `save_model` registers a fresh copy under the calling run's
|
|
13
13
|
run_name every time it runs (fine-tuned output); `import_model` registers a
|
|
14
|
-
model produced elsewhere once, under the fixed run_name IMPORTED, and
|
|
15
|
-
|
|
14
|
+
model produced elsewhere once, under the fixed run_name IMPORTED, and after
|
|
15
|
+
that only re-bundles the serve-app when its code changed. Both write the same
|
|
16
|
+
layout, so every
|
|
16
17
|
(family, suffix, run_name) consumer - load_model, deploy_model - handles both.
|
|
17
18
|
|
|
18
19
|
storage.py is pure: it takes run_id/run_name as explicit args and never reads
|
|
@@ -35,8 +36,11 @@ from mlflow.tracking import MlflowClient
|
|
|
35
36
|
from cortexgrid import s3_util
|
|
36
37
|
from cortexgrid.infra import get_mlflow_tracking_uri, get_s3_bucket
|
|
37
38
|
from cortexgrid.model_serving import (
|
|
39
|
+
build_bundle,
|
|
38
40
|
bundle_class,
|
|
41
|
+
metadata_from_tags,
|
|
39
42
|
metadata_to_tags,
|
|
43
|
+
upload_bundle,
|
|
40
44
|
)
|
|
41
45
|
|
|
42
46
|
|
|
@@ -50,9 +54,10 @@ _PHASE_UPLOAD_FAILED = "upload_failed"
|
|
|
50
54
|
_PHASE_BROKEN = "broken"
|
|
51
55
|
|
|
52
56
|
# An upload still marked "uploading" this long after the version was created is
|
|
53
|
-
# treated as broken:
|
|
54
|
-
#
|
|
55
|
-
# dies mid-
|
|
57
|
+
# treated as broken: the version is created before the weights are resolved
|
|
58
|
+
# (for `import_model`, before its download), so creation_timestamp is the
|
|
59
|
+
# start of the whole upload, and a process that dies mid-way never flips the
|
|
60
|
+
# tag to "ready"/"upload_failed". Expiry is
|
|
56
61
|
# derived lazily on read (see `_phase_for`); nothing is written back.
|
|
57
62
|
_UPLOAD_DEADLINE = timedelta(hours=3)
|
|
58
63
|
|
|
@@ -178,11 +183,15 @@ def import_model(
|
|
|
178
183
|
|
|
179
184
|
`source` is the weights directory, or a callable returning it; the callable
|
|
180
185
|
runs only when the upload actually happens, so an expensive download can be
|
|
181
|
-
skipped on every run after the first.
|
|
186
|
+
skipped on every run after the first. It runs after the version is
|
|
187
|
+
registered as "uploading", so a concurrent import sees this one in flight
|
|
188
|
+
while it downloads.
|
|
182
189
|
|
|
183
190
|
If a version is already registered under the key:
|
|
184
|
-
- "ready":
|
|
185
|
-
|
|
191
|
+
- "ready": returns it without calling `source`. If `serve_app`'s code no
|
|
192
|
+
longer matches the stored bundle, it is re-bundled first and the
|
|
193
|
+
weights are kept (see `_refresh_bundle`). To replace the weights,
|
|
194
|
+
`delete_model` it first.
|
|
186
195
|
- "uploading": raises RuntimeError - another process is importing it.
|
|
187
196
|
- "upload_failed" / "broken": deleted and imported again.
|
|
188
197
|
|
|
@@ -192,6 +201,7 @@ def import_model(
|
|
|
192
201
|
existing = model_registry_status(family, suffix, IMPORTED)
|
|
193
202
|
if existing is not None:
|
|
194
203
|
if existing.phase == _PHASE_READY:
|
|
204
|
+
_refresh_bundle(serve_app, family, suffix)
|
|
195
205
|
return existing
|
|
196
206
|
if existing.phase == _PHASE_UPLOADING:
|
|
197
207
|
raise RuntimeError(
|
|
@@ -199,30 +209,53 @@ def import_model(
|
|
|
199
209
|
"another process"
|
|
200
210
|
)
|
|
201
211
|
delete_model(family, suffix, IMPORTED)
|
|
202
|
-
|
|
203
|
-
|
|
212
|
+
return _upload_model(source, serve_app, suffix, family, None, IMPORTED)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _refresh_bundle(serve_app: type, family: str, suffix: str) -> None:
|
|
216
|
+
"""Re-bundle an imported model's serve-app when its fingerprint differs
|
|
217
|
+
from the bundle stored on the version, leaving the weights in place.
|
|
218
|
+
|
|
219
|
+
The new bundle is uploaded next to the old one and the version's tags are
|
|
220
|
+
pointed at it, so the next `deploy_model` runs the new code. An app that is
|
|
221
|
+
already running keeps the code it started with until it is deployed again;
|
|
222
|
+
the old bundle stays in storage so that app can still restart."""
|
|
223
|
+
name = f"{family}__{suffix}"
|
|
224
|
+
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
225
|
+
version = client.search_model_versions(
|
|
226
|
+
f"name='{name}' and tags.run_name='{IMPORTED}'"
|
|
227
|
+
)[0]
|
|
228
|
+
serve_bundle = build_bundle(serve_app)
|
|
229
|
+
if metadata_from_tags(version.tags).fingerprint == serve_bundle.fingerprint:
|
|
230
|
+
return
|
|
231
|
+
meta = upload_bundle(serve_bundle, family, suffix, IMPORTED)
|
|
232
|
+
for key, value in metadata_to_tags(meta).items():
|
|
233
|
+
client.set_model_version_tag(name, version.version, key, value)
|
|
204
234
|
|
|
205
235
|
|
|
206
236
|
def _upload_model(
|
|
207
|
-
|
|
237
|
+
weights: str | Path | Callable[[], str | Path],
|
|
208
238
|
serve_app: type,
|
|
209
239
|
suffix: str,
|
|
210
240
|
family: str,
|
|
211
241
|
run_id: str | None,
|
|
212
242
|
run_name: str,
|
|
213
243
|
) -> SavedModel:
|
|
214
|
-
"""Register a ModelVersion in "uploading",
|
|
215
|
-
|
|
244
|
+
"""Register a ModelVersion in "uploading", resolve `weights` to a directory
|
|
245
|
+
(calling it when it is a callable), upload the weights and the serve-app
|
|
246
|
+
bundle, and flip it to "ready" (or "upload_failed")."""
|
|
216
247
|
bucket = get_s3_bucket()
|
|
217
248
|
prefix = f"models/{run_name}/{family}/{suffix}"
|
|
218
|
-
size_bytes = _dir_size_bytes(weights_dir)
|
|
219
249
|
source = f"s3://{bucket}/{prefix}/weights/"
|
|
220
250
|
name = f"{family}__{suffix}"
|
|
221
251
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
222
252
|
_ensure_registered_model(client, name)
|
|
223
|
-
# Register the version up front in the "uploading" phase
|
|
224
|
-
#
|
|
225
|
-
#
|
|
253
|
+
# Register the version up front in the "uploading" phase, before `weights`
|
|
254
|
+
# is resolved: the dashboard surfaces the model while it is still being
|
|
255
|
+
# fetched and uploaded, and a concurrent `import_model` sees the import in
|
|
256
|
+
# flight for the whole download instead of starting one of its own. The
|
|
257
|
+
# size is stamped once the directory exists; the bundle tags and the flip
|
|
258
|
+
# to "ready" happen only after the upload lands.
|
|
226
259
|
version = client.create_model_version(
|
|
227
260
|
name=name,
|
|
228
261
|
source=source,
|
|
@@ -231,11 +264,14 @@ def _upload_model(
|
|
|
231
264
|
"family": family,
|
|
232
265
|
"suffix": suffix,
|
|
233
266
|
"run_name": run_name,
|
|
234
|
-
"size_bytes": str(size_bytes),
|
|
235
267
|
_LIFECYCLE_TAG: _PHASE_UPLOADING,
|
|
236
268
|
},
|
|
237
269
|
)
|
|
238
270
|
try:
|
|
271
|
+
weights_dir = weights() if callable(weights) else weights
|
|
272
|
+
client.set_model_version_tag(
|
|
273
|
+
name, version.version, "size_bytes", str(_dir_size_bytes(weights_dir))
|
|
274
|
+
)
|
|
239
275
|
s3_util.upload_dir(str(weights_dir), dest_path=f"{prefix}/weights")
|
|
240
276
|
bundle_meta = bundle_class(serve_app, family, suffix, run_name)
|
|
241
277
|
for key, value in metadata_to_tags(bundle_meta).items():
|
|
@@ -150,7 +150,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
150
150
|
print(deployed.url)
|
|
151
151
|
```
|
|
152
152
|
|
|
153
|
-
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and
|
|
153
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
154
154
|
|
|
155
155
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
156
156
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|