cortexgrid 0.2.93__tar.gz → 0.2.95__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.2.93
3
+ Version: 0.2.95
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -178,6 +178,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
178
178
  print(deployed.url)
179
179
  ```
180
180
 
181
+ `save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
182
+
181
183
  `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
182
184
 
183
185
  ### API reference
@@ -74,12 +74,14 @@ from cortexgrid.ray_util import (
74
74
  )
75
75
  from cortexgrid.s3_util import delete_prefix, download, get_s3_client, upload, upload_dir
76
76
  from cortexgrid.model_storage import (
77
+ IMPORTED,
77
78
  SavedModel,
78
79
  delete_model,
79
80
  list_models,
80
81
  load_model,
81
82
  model_registry_status,
82
83
  )
84
+ from cortexgrid.model_storage import import_model as _import_model_storage
83
85
  from cortexgrid.model_storage import save_model as _save_model_storage
84
86
  from cortexgrid.model_serving import (
85
87
  Deployment,
@@ -119,7 +121,11 @@ def save_model(
119
121
  weights_dir: str | Path, serve_app: type, family: str, suffix: str
120
122
  ) -> SavedModel:
121
123
  """Persist a weights directory under the current Experiment's run, paired
122
- with the serve-app class that will front it at deploy time."""
124
+ with the serve-app class that will front it at deploy time.
125
+
126
+ Every run saves a new copy under its own run_name - meant for weights the
127
+ run produced (e.g. a fine-tune). For a model produced elsewhere that should
128
+ be uploaded once and reused across runs, use `import_model`."""
123
129
  experiment = Experiment.get_instance()
124
130
  return _save_model_storage(
125
131
  weights_dir,
@@ -131,6 +137,27 @@ def save_model(
131
137
  )
132
138
 
133
139
 
140
+ def import_model(
141
+ source: str | Path | Callable[[], str | Path],
142
+ serve_app: type,
143
+ family: str,
144
+ suffix: str,
145
+ ) -> SavedModel:
146
+ """Register a model produced elsewhere once, reuse it on every later call,
147
+ and record on the current Experiment's run which imported model it used.
148
+
149
+ The model belongs to no run (see `cortexgrid.model_storage.import_model`),
150
+ so the run keeps the link instead: the tag
151
+ `imported_model/<family>/<suffix>` holds the version's `created_at`, set
152
+ whether this call uploaded the model or reused it."""
153
+ experiment = Experiment.get_instance()
154
+ model = _import_model_storage(source, serve_app, family, suffix)
155
+ get_mlflow_client().set_tag(
156
+ experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
157
+ )
158
+ return model
159
+
160
+
134
161
  __all__ = [
135
162
  "Experiment",
136
163
  "delete_experiment",
@@ -179,8 +206,10 @@ __all__ = [
179
206
  "list_secrets",
180
207
  "delete_secret",
181
208
  # Model registry
209
+ "IMPORTED",
182
210
  "SavedModel",
183
211
  "save_model",
212
+ "import_model",
184
213
  "load_model",
185
214
  "list_models",
186
215
  "model_registry_status",
@@ -6,7 +6,14 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
6
6
  family, suffix -> ModelVersion.tags["family"], ["suffix"] (denormalized)
7
7
  weights blob path -> ModelVersion.source =
8
8
  "s3://<bucket>/models/<run_name>/<family>/<suffix>/weights/"
9
- run linkage -> ModelVersion.run_id (built-in MLflow field)
9
+ run linkage -> ModelVersion.run_id (built-in MLflow field; unset
10
+ for imported models)
11
+
12
+ Two ways in: `save_model` registers a fresh copy under the calling run's
13
+ run_name every time it runs (fine-tuned output); `import_model` registers a
14
+ model produced elsewhere once, under the fixed run_name IMPORTED, and is a
15
+ no-op after that. Both write the same layout, so every
16
+ (family, suffix, run_name) consumer - load_model, deploy_model - handles both.
10
17
 
11
18
  storage.py is pure: it takes run_id/run_name as explicit args and never reads
12
19
  the active Experiment singleton. The facade that fills those in lives in
@@ -20,7 +27,7 @@ from dataclasses import dataclass
20
27
  from datetime import datetime, timedelta, timezone
21
28
  from pathlib import Path
22
29
  import tempfile
23
- from typing import Any
30
+ from typing import Any, Callable
24
31
 
25
32
  from mlflow.exceptions import MlflowException
26
33
  from mlflow.tracking import MlflowClient
@@ -43,12 +50,18 @@ _PHASE_UPLOAD_FAILED = "upload_failed"
43
50
  _PHASE_BROKEN = "broken"
44
51
 
45
52
  # An upload still marked "uploading" this long after the version was created is
46
- # treated as broken: save_model creates the version immediately before the
47
- # upload begins, so creation_timestamp is the upload start, and a process that
48
- # dies mid-upload never flips the tag to "ready"/"upload_failed". Expiry is
53
+ # treated as broken: the version is created before the weights are resolved
54
+ # (for `import_model`, before its download), so creation_timestamp is the
55
+ # start of the whole upload, and a process that dies mid-way never flips the
56
+ # tag to "ready"/"upload_failed". Expiry is
49
57
  # derived lazily on read (see `_phase_for`); nothing is written back.
50
58
  _UPLOAD_DEADLINE = timedelta(hours=3)
51
59
 
60
+ # run_name under which `import_model` registers a model: imported weights belong
61
+ # to no run, so they share one fixed key and outlive the run that imported
62
+ # them. Run names are haikunator "word-word-NN", so no run can take this name.
63
+ IMPORTED = "imported"
64
+
52
65
 
53
66
  @dataclass
54
67
  class SavedModel:
@@ -137,6 +150,10 @@ def save_model(
137
150
  """Upload a weights directory to S3 and register a new MLflow ModelVersion
138
151
  paired with the serve-app that fronts it.
139
152
 
153
+ Meant for weights the calling run produced (e.g. a fine-tune): every run
154
+ saves its own copy under its own run_name, so running the same code twice
155
+ yields two models. For weights produced elsewhere, use `import_model`.
156
+
140
157
  cortexgrid stores the weights as an opaque directory: it never inspects,
141
158
  serializes, or reconstructs their contents, so the on-disk format
142
159
  (HuggingFace `save_pretrained`, `torch.save`, ONNX, anything) is entirely
@@ -148,16 +165,69 @@ def save_model(
148
165
  Its code is bundled and its import path, bundle URL, and pip list are
149
166
  stored as tags on the ModelVersion so `deploy_model` can bind it later
150
167
  without the caller holding the class object."""
168
+ return _upload_model(weights_dir, serve_app, suffix, family, run_id, run_name)
169
+
170
+
171
+ def import_model(
172
+ source: str | Path | Callable[[], str | Path],
173
+ serve_app: type,
174
+ family: str,
175
+ suffix: str,
176
+ ) -> SavedModel:
177
+ """Register a model produced elsewhere (e.g. a pretrained base model) under
178
+ the fixed key (family, suffix, IMPORTED), once.
179
+
180
+ `source` is the weights directory, or a callable returning it; the callable
181
+ runs only when the upload actually happens, so an expensive download can be
182
+ skipped on every run after the first. It runs after the version is
183
+ registered as "uploading", so a concurrent import sees this one in flight
184
+ while it downloads.
185
+
186
+ If a version is already registered under the key:
187
+ - "ready": no-op, returns it. `source` and `serve_app` are ignored; to
188
+ replace the weights or the serve-app, `delete_model` it first.
189
+ - "uploading": raises RuntimeError - another process is importing it.
190
+ - "upload_failed" / "broken": deleted and imported again.
191
+
192
+ The version is linked to no MLflow run, so deleting a run leaves it in
193
+ place. Deploy it like any saved model:
194
+ `deploy_model(family, suffix, IMPORTED)`."""
195
+ existing = model_registry_status(family, suffix, IMPORTED)
196
+ if existing is not None:
197
+ if existing.phase == _PHASE_READY:
198
+ return existing
199
+ if existing.phase == _PHASE_UPLOADING:
200
+ raise RuntimeError(
201
+ f"Model {family}/{suffix}/{IMPORTED} is being imported by "
202
+ "another process"
203
+ )
204
+ delete_model(family, suffix, IMPORTED)
205
+ return _upload_model(source, serve_app, suffix, family, None, IMPORTED)
206
+
207
+
208
+ def _upload_model(
209
+ weights: str | Path | Callable[[], str | Path],
210
+ serve_app: type,
211
+ suffix: str,
212
+ family: str,
213
+ run_id: str | None,
214
+ run_name: str,
215
+ ) -> SavedModel:
216
+ """Register a ModelVersion in "uploading", resolve `weights` to a directory
217
+ (calling it when it is a callable), upload the weights and the serve-app
218
+ bundle, and flip it to "ready" (or "upload_failed")."""
151
219
  bucket = get_s3_bucket()
152
220
  prefix = f"models/{run_name}/{family}/{suffix}"
153
- size_bytes = _dir_size_bytes(weights_dir)
154
221
  source = f"s3://{bucket}/{prefix}/weights/"
155
222
  name = f"{family}__{suffix}"
156
223
  client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
157
224
  _ensure_registered_model(client, name)
158
- # Register the version up front in the "uploading" phase so the dashboard
159
- # can surface a model while its weights are still streaming to storage. The
160
- # bundle tags and the flip to "ready" happen only after the upload lands.
225
+ # Register the version up front in the "uploading" phase, before `weights`
226
+ # is resolved: the dashboard surfaces the model while it is still being
227
+ # fetched and uploaded, and a concurrent `import_model` sees the import in
228
+ # flight for the whole download instead of starting one of its own. The
229
+ # size is stamped once the directory exists; the bundle tags and the flip
230
+ # to "ready" happen only after the upload lands.
161
231
  version = client.create_model_version(
162
232
  name=name,
163
233
  source=source,
@@ -166,11 +236,14 @@ def save_model(
166
236
  "family": family,
167
237
  "suffix": suffix,
168
238
  "run_name": run_name,
169
- "size_bytes": str(size_bytes),
170
239
  _LIFECYCLE_TAG: _PHASE_UPLOADING,
171
240
  },
172
241
  )
173
242
  try:
243
+ weights_dir = weights() if callable(weights) else weights
244
+ client.set_model_version_tag(
245
+ name, version.version, "size_bytes", str(_dir_size_bytes(weights_dir))
246
+ )
174
247
  s3_util.upload_dir(str(weights_dir), dest_path=f"{prefix}/weights")
175
248
  bundle_meta = bundle_class(serve_app, family, suffix, run_name)
176
249
  for key, value in metadata_to_tags(bundle_meta).items():
@@ -150,6 +150,8 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
150
150
  print(deployed.url)
151
151
  ```
152
152
 
153
+ `save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and is a no-op on later runs; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
154
+
153
155
  `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
154
156
 
155
157
  ### API reference
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.2.93"
3
+ version = "0.2.95"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes