cortexgrid 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/PKG-INFO +2 -2
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/__init__.py +30 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/model_serving.py +32 -5
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/model_storage.py +100 -16
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/docs/cortexgrid/README.md +1 -1
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/pyproject.toml +1 -1
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/.gitignore +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/LICENSE +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.2 → cortexgrid-0.3.4}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -234,7 +234,7 @@ The requirements are part of the model, not of the serve-app class: GPUs, RAM an
|
|
|
234
234
|
|
|
235
235
|
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
236
236
|
|
|
237
|
-
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
237
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
|
|
238
238
|
|
|
239
239
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
240
240
|
|
|
@@ -85,6 +85,7 @@ from cortexgrid.ray_util import (
|
|
|
85
85
|
from cortexgrid.s3_util import delete_prefix, download, get_s3_client, upload, upload_dir
|
|
86
86
|
from cortexgrid.model_storage import (
|
|
87
87
|
IMPORTED,
|
|
88
|
+
NO_WEIGHTS,
|
|
88
89
|
SavedModel,
|
|
89
90
|
delete_model,
|
|
90
91
|
list_models,
|
|
@@ -95,6 +96,7 @@ from cortexgrid.model_storage import (
|
|
|
95
96
|
set_model_requirements,
|
|
96
97
|
)
|
|
97
98
|
from cortexgrid.model_storage import import_model as _import_model_storage
|
|
99
|
+
from cortexgrid.model_storage import register_model as _register_model_storage
|
|
98
100
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
99
101
|
from cortexgrid.model_serving import (
|
|
100
102
|
Deployment,
|
|
@@ -199,6 +201,32 @@ def import_model(
|
|
|
199
201
|
return model
|
|
200
202
|
|
|
201
203
|
|
|
204
|
+
def register_model(
|
|
205
|
+
serve_app: type,
|
|
206
|
+
family: str,
|
|
207
|
+
suffix: str,
|
|
208
|
+
requirements: ModelRequirements | None = None,
|
|
209
|
+
config: dict[str, str] | None = None,
|
|
210
|
+
) -> SavedModel:
|
|
211
|
+
"""Register a model with no weights of its own once - a serve-app that
|
|
212
|
+
forwards to a hosted API stages nothing - reuse it on every later call, and
|
|
213
|
+
record on the current Experiment's run which one it used.
|
|
214
|
+
|
|
215
|
+
`import_model` without the import: everything it needs beyond its code goes
|
|
216
|
+
in `config`, which the serve-app reads with `model_config` at construction
|
|
217
|
+
(see `cortexgrid.model_storage.register_model`). The model belongs to no
|
|
218
|
+
run, so the run keeps the link the same way, under the same tag
|
|
219
|
+
`imported_model/<family>/<suffix>`."""
|
|
220
|
+
experiment = Experiment.get_instance()
|
|
221
|
+
model = _register_model_storage(
|
|
222
|
+
serve_app, family, suffix, requirements, config
|
|
223
|
+
)
|
|
224
|
+
get_mlflow_client().set_tag(
|
|
225
|
+
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
226
|
+
)
|
|
227
|
+
return model
|
|
228
|
+
|
|
229
|
+
|
|
202
230
|
__all__ = [
|
|
203
231
|
"Experiment",
|
|
204
232
|
"delete_experiment",
|
|
@@ -255,9 +283,11 @@ __all__ = [
|
|
|
255
283
|
"delete_secret",
|
|
256
284
|
# Model registry
|
|
257
285
|
"IMPORTED",
|
|
286
|
+
"NO_WEIGHTS",
|
|
258
287
|
"SavedModel",
|
|
259
288
|
"save_model",
|
|
260
289
|
"import_model",
|
|
290
|
+
"register_model",
|
|
261
291
|
"load_model",
|
|
262
292
|
"list_models",
|
|
263
293
|
"model_registry_status",
|
|
@@ -485,6 +485,25 @@ def _clear_failed_application(
|
|
|
485
485
|
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
486
486
|
|
|
487
487
|
|
|
488
|
+
def _spec_already_deployed(spec: dict[str, Any]) -> bool:
|
|
489
|
+
"""True when this exact spec is already the app's target, so PUTting it
|
|
490
|
+
again would only disturb the controller.
|
|
491
|
+
|
|
492
|
+
Re-PUTting is not a no-op. While an app's build task is in flight its target
|
|
493
|
+
code version is unset, so Ray cancels that build and starts a new one no
|
|
494
|
+
matter what the config says - and since every PUT re-sends the whole
|
|
495
|
+
applications list, it restarts the in-flight builds of the other apps too.
|
|
496
|
+
Our build task downloads the model bundle and may create a pip virtualenv,
|
|
497
|
+
so a caller redeploying faster than that could keep it from ever finishing.
|
|
498
|
+
|
|
499
|
+
A DEPLOY_FAILED or DELETING app never reaches here: `_clear_failed_application`
|
|
500
|
+
has already removed it, and an identical PUT over a failed app is exactly the
|
|
501
|
+
no-op that leaves it failed.
|
|
502
|
+
"""
|
|
503
|
+
app = get_serve_details().get("applications", {}).get(spec["name"])
|
|
504
|
+
return app is not None and app.get("deployed_app_config") == spec
|
|
505
|
+
|
|
506
|
+
|
|
488
507
|
def deploy_model(
|
|
489
508
|
family: str,
|
|
490
509
|
suffix: str,
|
|
@@ -507,6 +526,11 @@ def deploy_model(
|
|
|
507
526
|
or an app still DELETING, is waited out before the new spec is PUT, so the
|
|
508
527
|
deploy starts afresh instead of Ray reusing the failed deployment.
|
|
509
528
|
|
|
529
|
+
Re-deploying a model that is already live with exactly this spec skips the
|
|
530
|
+
PUT rather than restating it: see `_spec_already_deployed` for what a
|
|
531
|
+
redundant PUT costs. The call still reports the app's phase, and with
|
|
532
|
+
`wait=True` still blocks until it is RUNNING.
|
|
533
|
+
|
|
510
534
|
With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
|
|
511
535
|
controller reports the app RUNNING. `timeout` (default 300) caps the whole
|
|
512
536
|
call, clearing a failed app included; exceeding it raises TimeoutError, and
|
|
@@ -520,11 +544,14 @@ def deploy_model(
|
|
|
520
544
|
family, suffix, run_name, meta, requirements, num_replicas
|
|
521
545
|
)
|
|
522
546
|
_clear_failed_application(family, suffix, run_name, timeout, deadline)
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
547
|
+
if not _spec_already_deployed(spec):
|
|
548
|
+
existing = [
|
|
549
|
+
a for a in _current_application_specs() if a["name"] != spec["name"]
|
|
550
|
+
]
|
|
551
|
+
# The controller registers the app, sets it DEPLOYING and stamps
|
|
552
|
+
# last_deployed_time_s before the PUT returns, so the wait below neither
|
|
553
|
+
# misses the app nor reads a status left by an earlier deploy.
|
|
554
|
+
put_serve_applications([*existing, spec])
|
|
528
555
|
if wait:
|
|
529
556
|
_wait_for_application_running(spec["name"], timeout, deadline)
|
|
530
557
|
app = get_serve_details().get("applications", {}).get(spec["name"], {})
|
|
@@ -6,17 +6,21 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
|
|
|
6
6
|
family, suffix -> ModelVersion.tags["family"], ["suffix"] (denormalized)
|
|
7
7
|
weights blob path -> ModelVersion.source =
|
|
8
8
|
"s3://<bucket>/models/<run_name>/<family>/<suffix>/weights/"
|
|
9
|
+
(NO_WEIGHTS for a model registered without any)
|
|
9
10
|
run linkage -> ModelVersion.run_id (built-in MLflow field; unset
|
|
10
11
|
for imported models)
|
|
11
12
|
requirements -> ModelVersion.tags["num_gpus"], ["ram_gb"], ["vram_gb"]
|
|
12
13
|
config -> ModelVersion.tags["config"] (JSON object)
|
|
13
14
|
|
|
14
|
-
|
|
15
|
+
Three ways in: `save_model` registers a fresh copy under the calling run's
|
|
15
16
|
run_name every time it runs (fine-tuned output); `import_model` registers a
|
|
16
17
|
model produced elsewhere once, under the fixed run_name IMPORTED, and after
|
|
17
|
-
that only re-bundles the serve-app when its code changed
|
|
18
|
-
|
|
19
|
-
|
|
18
|
+
that only re-bundles the serve-app when its code changed; `register_model`
|
|
19
|
+
registers a model that stages no weights at all - a serve-app that forwards to
|
|
20
|
+
a hosted API holds none - under that same fixed key. All three write the same
|
|
21
|
+
layout, so every (family, suffix, run_name) consumer - deploy_model, the
|
|
22
|
+
dashboard - handles them alike; only `load_model` parts them, having nothing to
|
|
23
|
+
hand back for a model registered without weights.
|
|
20
24
|
|
|
21
25
|
storage.py is pure: it takes run_id/run_name as explicit args and never reads
|
|
22
26
|
the active Experiment singleton. The facade that fills those in lives in
|
|
@@ -86,6 +90,13 @@ _UPLOAD_DEADLINE = timedelta(hours=3)
|
|
|
86
90
|
# them. Run names are haikunator "word-word-NN", so no run can take this name.
|
|
87
91
|
IMPORTED = "imported"
|
|
88
92
|
|
|
93
|
+
# ModelVersion.source of a model registered with no weights of its own: nothing
|
|
94
|
+
# was staged, so there is no blob to point at. Spelled as a URI rather than left
|
|
95
|
+
# blank because MLflow rejects a source that is empty or a local path, and
|
|
96
|
+
# because every reader - `load_model`, the dashboard's storage field - then sees
|
|
97
|
+
# why there is no path instead of an empty one.
|
|
98
|
+
NO_WEIGHTS = "cortexgrid://no-weights"
|
|
99
|
+
|
|
89
100
|
|
|
90
101
|
@dataclass
|
|
91
102
|
class SavedModel:
|
|
@@ -93,7 +104,9 @@ class SavedModel:
|
|
|
93
104
|
suffix: str
|
|
94
105
|
run_name: str
|
|
95
106
|
created_at: str
|
|
107
|
+
# Where the weights live, or NO_WEIGHTS for a model registered without any.
|
|
96
108
|
data_blob_path: str
|
|
109
|
+
# Size of the weights; 0 for a model registered without any.
|
|
97
110
|
size_bytes: int
|
|
98
111
|
# Registry lifecycle phase: "uploading" while save_model streams the weights
|
|
99
112
|
# and serve bundle to storage, "ready" once that finishes, "upload_failed"
|
|
@@ -107,6 +120,13 @@ class SavedModel:
|
|
|
107
120
|
# versions stored without any.
|
|
108
121
|
config: dict[str, str] = field(default_factory=dict)
|
|
109
122
|
|
|
123
|
+
@property
|
|
124
|
+
def has_weights(self) -> bool:
|
|
125
|
+
"""Whether this model staged weights of its own. False for one
|
|
126
|
+
registered with `register_model`, whose serve-app holds no bytes to
|
|
127
|
+
store; `load_model` on it raises."""
|
|
128
|
+
return self.data_blob_path != NO_WEIGHTS
|
|
129
|
+
|
|
110
130
|
|
|
111
131
|
def _phase_for(version: Any) -> str:
|
|
112
132
|
"""Registry lifecycle phase of a version, expiring stale uploads to "broken".
|
|
@@ -269,6 +289,52 @@ def import_model(
|
|
|
269
289
|
The version is linked to no MLflow run, so deleting a run leaves it in
|
|
270
290
|
place. Deploy it like any saved model:
|
|
271
291
|
`deploy_model(family, suffix, IMPORTED)`."""
|
|
292
|
+
return _register_imported(
|
|
293
|
+
source, serve_app, family, suffix, requirements, config
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def register_model(
|
|
298
|
+
serve_app: type,
|
|
299
|
+
family: str,
|
|
300
|
+
suffix: str,
|
|
301
|
+
requirements: ModelRequirements | None = None,
|
|
302
|
+
config: dict[str, str] | None = None,
|
|
303
|
+
) -> SavedModel:
|
|
304
|
+
"""Register a model that stages no weights under the fixed key
|
|
305
|
+
(family, suffix, IMPORTED), once.
|
|
306
|
+
|
|
307
|
+
For a model whose bytes are not ours to hold: a serve-app that forwards
|
|
308
|
+
requests to a hosted API, or one that reaches for the weights itself at
|
|
309
|
+
startup. There is nothing to upload, so only `serve_app`'s bundle is
|
|
310
|
+
stored and the version's source reads NO_WEIGHTS - `load_model` on such a
|
|
311
|
+
model raises, and `SavedModel.has_weights` is False. What it needs instead
|
|
312
|
+
of weights - which model a provider should be asked for, the name of the
|
|
313
|
+
secret holding the key - belongs in `config`, which the serve-app reads
|
|
314
|
+
with `model_config` at construction.
|
|
315
|
+
|
|
316
|
+
Registration is otherwise `import_model`'s, down to the phase an already
|
|
317
|
+
registered version leaves it in ("ready" is reused and its bundle
|
|
318
|
+
refreshed, "uploading" raises, "upload_failed"/"broken" is replaced), so
|
|
319
|
+
the entry is indistinguishable from an imported one to `deploy_model`,
|
|
320
|
+
`list_models` and the dashboard."""
|
|
321
|
+
return _register_imported(
|
|
322
|
+
None, serve_app, family, suffix, requirements, config
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _register_imported(
|
|
327
|
+
weights: str | Path | Callable[[], str | Path] | None,
|
|
328
|
+
serve_app: type,
|
|
329
|
+
family: str,
|
|
330
|
+
suffix: str,
|
|
331
|
+
requirements: ModelRequirements | None,
|
|
332
|
+
config: dict[str, str] | None,
|
|
333
|
+
) -> SavedModel:
|
|
334
|
+
"""Register `weights` under (family, suffix, IMPORTED) unless a version is
|
|
335
|
+
already there - the once-only registration `import_model` and
|
|
336
|
+
`register_model` share; `weights` is None for the model that stages
|
|
337
|
+
none."""
|
|
272
338
|
existing = model_registry_status(family, suffix, IMPORTED)
|
|
273
339
|
if existing is not None:
|
|
274
340
|
if existing.phase == _PHASE_READY:
|
|
@@ -287,7 +353,7 @@ def import_model(
|
|
|
287
353
|
)
|
|
288
354
|
delete_model(family, suffix, IMPORTED)
|
|
289
355
|
return _upload_model(
|
|
290
|
-
|
|
356
|
+
weights, serve_app, suffix, family, None, IMPORTED, requirements, config
|
|
291
357
|
)
|
|
292
358
|
|
|
293
359
|
|
|
@@ -348,7 +414,7 @@ def _set_missing_config(
|
|
|
348
414
|
|
|
349
415
|
|
|
350
416
|
def _upload_model(
|
|
351
|
-
weights: str | Path | Callable[[], str | Path],
|
|
417
|
+
weights: str | Path | Callable[[], str | Path] | None,
|
|
352
418
|
serve_app: type,
|
|
353
419
|
suffix: str,
|
|
354
420
|
family: str,
|
|
@@ -359,10 +425,17 @@ def _upload_model(
|
|
|
359
425
|
) -> SavedModel:
|
|
360
426
|
"""Register a ModelVersion in "uploading", resolve `weights` to a directory
|
|
361
427
|
(calling it when it is a callable), upload the weights and the serve-app
|
|
362
|
-
bundle, and flip it to "ready" (or "upload_failed").
|
|
363
|
-
|
|
428
|
+
bundle, and flip it to "ready" (or "upload_failed").
|
|
429
|
+
|
|
430
|
+
`weights` is None for a model that stages none: its source reads
|
|
431
|
+
NO_WEIGHTS and the bundle is the only thing uploaded. The phases are the
|
|
432
|
+
same either way, so one registration path covers both."""
|
|
364
433
|
prefix = f"models/{run_name}/{family}/{suffix}"
|
|
365
|
-
source =
|
|
434
|
+
source = (
|
|
435
|
+
NO_WEIGHTS
|
|
436
|
+
if weights is None
|
|
437
|
+
else f"s3://{get_s3_bucket()}/{prefix}/weights/"
|
|
438
|
+
)
|
|
366
439
|
name = f"{family}__{suffix}"
|
|
367
440
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
368
441
|
_ensure_registered_model(client, name)
|
|
@@ -371,7 +444,9 @@ def _upload_model(
|
|
|
371
444
|
# fetched and uploaded, and a concurrent `import_model` sees the import in
|
|
372
445
|
# flight for the whole download instead of starting one of its own. The
|
|
373
446
|
# size is stamped once the directory exists; the bundle tags and the flip
|
|
374
|
-
# to "ready" happen only after the upload lands.
|
|
447
|
+
# to "ready" happen only after the upload lands. A weights-less
|
|
448
|
+
# registration has only its bundle to upload, and passes through the same
|
|
449
|
+
# phases. No requirements and no
|
|
375
450
|
# config leave their tags unset, so a later `import_model` can still store
|
|
376
451
|
# them.
|
|
377
452
|
version = client.create_model_version(
|
|
@@ -388,11 +463,12 @@ def _upload_model(
|
|
|
388
463
|
},
|
|
389
464
|
)
|
|
390
465
|
try:
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
466
|
+
if weights is not None:
|
|
467
|
+
weights_dir = weights() if callable(weights) else weights
|
|
468
|
+
client.set_model_version_tag(
|
|
469
|
+
name, version.version, "size_bytes", str(_dir_size_bytes(weights_dir))
|
|
470
|
+
)
|
|
471
|
+
s3_util.upload_dir(str(weights_dir), dest_path=f"{prefix}/weights")
|
|
396
472
|
bundle_meta = bundle_class(serve_app, family, suffix, run_name)
|
|
397
473
|
for key, value in metadata_to_tags(bundle_meta).items():
|
|
398
474
|
client.set_model_version_tag(name, version.version, key, value)
|
|
@@ -414,7 +490,10 @@ def load_model(family: str, suffix: str, run_name: str) -> Path:
|
|
|
414
490
|
contents; the serve-app reconstructs the model from it however it likes
|
|
415
491
|
(`from_pretrained`, `torch.load`, ...). The returned directory persists
|
|
416
492
|
after this call - the caller (typically a serve-app loading weights at
|
|
417
|
-
startup) owns its lifetime.
|
|
493
|
+
startup) owns its lifetime.
|
|
494
|
+
|
|
495
|
+
Raises ValueError for a model registered with `register_model`: it stages
|
|
496
|
+
no weights, so there is none to hand back."""
|
|
418
497
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
419
498
|
name = f"{family}__{suffix}"
|
|
420
499
|
versions = client.search_model_versions(
|
|
@@ -422,6 +501,11 @@ def load_model(family: str, suffix: str, run_name: str) -> Path:
|
|
|
422
501
|
)
|
|
423
502
|
if not versions or not versions[0].source:
|
|
424
503
|
raise ValueError(f"No model {family}/{suffix}/{run_name}")
|
|
504
|
+
if versions[0].source == NO_WEIGHTS:
|
|
505
|
+
raise ValueError(
|
|
506
|
+
f"Model {family}/{suffix}/{run_name} was registered without "
|
|
507
|
+
"weights; there is nothing to load"
|
|
508
|
+
)
|
|
425
509
|
return _download_s3_uri(versions[0].source, None)
|
|
426
510
|
|
|
427
511
|
|
|
@@ -206,7 +206,7 @@ The requirements are part of the model, not of the serve-app class: GPUs, RAM an
|
|
|
206
206
|
|
|
207
207
|
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
208
208
|
|
|
209
|
-
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
209
|
+
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
|
|
210
210
|
|
|
211
211
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
212
212
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|