cortexgrid 0.3.1__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/PKG-INFO +3 -1
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/__init__.py +13 -3
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/model_storage.py +127 -10
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/docs/cortexgrid/README.md +2 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/pyproject.toml +1 -1
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/.gitignore +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/LICENSE +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/model_serving.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.1 → cortexgrid-0.3.2}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -232,6 +232,8 @@ print(deployed.url)
|
|
|
232
232
|
|
|
233
233
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
234
234
|
|
|
235
|
+
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
236
|
+
|
|
235
237
|
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
236
238
|
|
|
237
239
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
@@ -89,7 +89,9 @@ from cortexgrid.model_storage import (
|
|
|
89
89
|
delete_model,
|
|
90
90
|
list_models,
|
|
91
91
|
load_model,
|
|
92
|
+
model_config,
|
|
92
93
|
model_registry_status,
|
|
94
|
+
set_model_config,
|
|
93
95
|
set_model_requirements,
|
|
94
96
|
)
|
|
95
97
|
from cortexgrid.model_storage import import_model as _import_model_storage
|
|
@@ -149,10 +151,12 @@ def save_model(
|
|
|
149
151
|
family: str,
|
|
150
152
|
suffix: str,
|
|
151
153
|
requirements: ModelRequirements | None = None,
|
|
154
|
+
config: dict[str, str] | None = None,
|
|
152
155
|
) -> SavedModel:
|
|
153
156
|
"""Persist a weights directory under the current Experiment's run, paired
|
|
154
|
-
with the serve-app class that will front it at deploy time
|
|
155
|
-
|
|
157
|
+
with the serve-app class that will front it at deploy time, the hardware
|
|
158
|
+
one replica of it needs, and any `config` the serve-app reads with
|
|
159
|
+
`model_config`.
|
|
156
160
|
|
|
157
161
|
Every run saves a new copy under its own run_name - meant for weights the
|
|
158
162
|
run produced (e.g. a fine-tune). For a model produced elsewhere that should
|
|
@@ -166,6 +170,7 @@ def save_model(
|
|
|
166
170
|
run_id=experiment.run_id,
|
|
167
171
|
run_name=experiment.run_name(),
|
|
168
172
|
requirements=requirements,
|
|
173
|
+
config=config,
|
|
169
174
|
)
|
|
170
175
|
|
|
171
176
|
|
|
@@ -175,6 +180,7 @@ def import_model(
|
|
|
175
180
|
family: str,
|
|
176
181
|
suffix: str,
|
|
177
182
|
requirements: ModelRequirements | None = None,
|
|
183
|
+
config: dict[str, str] | None = None,
|
|
178
184
|
) -> SavedModel:
|
|
179
185
|
"""Register a model produced elsewhere once, reuse it on every later call,
|
|
180
186
|
and record on the current Experiment's run which imported model it used.
|
|
@@ -184,7 +190,9 @@ def import_model(
|
|
|
184
190
|
`imported_model/<family>/<suffix>` holds the version's `created_at`, set
|
|
185
191
|
whether this call uploaded the model or reused it."""
|
|
186
192
|
experiment = Experiment.get_instance()
|
|
187
|
-
model = _import_model_storage(
|
|
193
|
+
model = _import_model_storage(
|
|
194
|
+
source, serve_app, family, suffix, requirements, config
|
|
195
|
+
)
|
|
188
196
|
get_mlflow_client().set_tag(
|
|
189
197
|
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
190
198
|
)
|
|
@@ -255,6 +263,8 @@ __all__ = [
|
|
|
255
263
|
"model_registry_status",
|
|
256
264
|
"ModelRequirements",
|
|
257
265
|
"set_model_requirements",
|
|
266
|
+
"model_config",
|
|
267
|
+
"set_model_config",
|
|
258
268
|
"delete_model",
|
|
259
269
|
# Model serving
|
|
260
270
|
"Deployment",
|
|
@@ -9,6 +9,7 @@ Mapping cortexgrid taxonomy <-> MLflow Registry:
|
|
|
9
9
|
run linkage -> ModelVersion.run_id (built-in MLflow field; unset
|
|
10
10
|
for imported models)
|
|
11
11
|
requirements -> ModelVersion.tags["num_gpus"], ["ram_gb"], ["vram_gb"]
|
|
12
|
+
config -> ModelVersion.tags["config"] (JSON object)
|
|
12
13
|
|
|
13
14
|
Two ways in: `save_model` registers a fresh copy under the calling run's
|
|
14
15
|
run_name every time it runs (fine-tuned output); `import_model` registers a
|
|
@@ -24,8 +25,9 @@ cortexgrid/__init__.py.
|
|
|
24
25
|
|
|
25
26
|
from __future__ import annotations
|
|
26
27
|
|
|
28
|
+
import json
|
|
27
29
|
import os
|
|
28
|
-
from dataclasses import dataclass
|
|
30
|
+
from dataclasses import dataclass, field
|
|
29
31
|
from datetime import datetime, timedelta, timezone
|
|
30
32
|
from pathlib import Path
|
|
31
33
|
import tempfile
|
|
@@ -58,6 +60,19 @@ _PHASE_READY = "ready"
|
|
|
58
60
|
_PHASE_UPLOAD_FAILED = "upload_failed"
|
|
59
61
|
_PHASE_BROKEN = "broken"
|
|
60
62
|
|
|
63
|
+
# MLflow ModelVersion tag holding the model's config: whatever settings its
|
|
64
|
+
# serve-app needs that are not the weights (a provider's model id, an endpoint,
|
|
65
|
+
# the name of a secret to read). cortexgrid never interprets it - it is the
|
|
66
|
+
# serve-app's own vocabulary, stored next to the model so the app reads it at
|
|
67
|
+
# construction instead of being redeployed to change a setting.
|
|
68
|
+
#
|
|
69
|
+
# One JSON object in one tag, not a tag per key: the keys are the serve-app's
|
|
70
|
+
# to choose, free of MLflow's tag-key charset, and the whole mapping is
|
|
71
|
+
# replaced in a single write, so removing a key needs no tag deletion. MLflow
|
|
72
|
+
# caps a tag value at 8000 characters, which bounds how big a config can get.
|
|
73
|
+
_CONFIG_TAG = "config"
|
|
74
|
+
|
|
75
|
+
|
|
61
76
|
# An upload still marked "uploading" this long after the version was created is
|
|
62
77
|
# treated as broken: the version is created before the weights are resolved
|
|
63
78
|
# (for `import_model`, before its download), so creation_timestamp is the
|
|
@@ -88,6 +103,9 @@ class SavedModel:
|
|
|
88
103
|
phase: str
|
|
89
104
|
# Hardware one replica needs; defaults for versions stored without it.
|
|
90
105
|
requirements: ModelRequirements
|
|
106
|
+
# Free-form settings the serve-app reads at construction; empty for
|
|
107
|
+
# versions stored without any.
|
|
108
|
+
config: dict[str, str] = field(default_factory=dict)
|
|
91
109
|
|
|
92
110
|
|
|
93
111
|
def _phase_for(version: Any) -> str:
|
|
@@ -107,6 +125,28 @@ def _phase_for(version: Any) -> str:
|
|
|
107
125
|
return phase
|
|
108
126
|
|
|
109
127
|
|
|
128
|
+
def _config_to_tag(config: dict[str, str]) -> str:
|
|
129
|
+
"""Serialise a config mapping to its MLflow tag value.
|
|
130
|
+
|
|
131
|
+
Keys and values are strings: they round-trip through a tag, and the model
|
|
132
|
+
card edits them as text. A caller with a number or a flag spells it as a
|
|
133
|
+
string and the serve-app parses it back."""
|
|
134
|
+
for key, value in config.items():
|
|
135
|
+
if not isinstance(key, str) or not isinstance(value, str):
|
|
136
|
+
raise ValueError(
|
|
137
|
+
f"Model config must map strings to strings: {key!r}: {value!r}"
|
|
138
|
+
)
|
|
139
|
+
if not key.strip():
|
|
140
|
+
raise ValueError("Model config keys cannot be blank")
|
|
141
|
+
return json.dumps(config)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _config_from_tags(tags: dict[str, str]) -> dict[str, str]:
|
|
145
|
+
"""Deserialise a config mapping from a ModelVersion's MLflow tags; a
|
|
146
|
+
missing tag reads as no config."""
|
|
147
|
+
return json.loads(tags.get(_CONFIG_TAG, "{}"))
|
|
148
|
+
|
|
149
|
+
|
|
110
150
|
def _to_saved_model(version: Any) -> SavedModel:
|
|
111
151
|
return SavedModel(
|
|
112
152
|
family=version.tags["family"],
|
|
@@ -119,6 +159,7 @@ def _to_saved_model(version: Any) -> SavedModel:
|
|
|
119
159
|
size_bytes=int(version.tags.get("size_bytes", "0")),
|
|
120
160
|
phase=_phase_for(version),
|
|
121
161
|
requirements=requirements_from_tags(version.tags),
|
|
162
|
+
config=_config_from_tags(version.tags),
|
|
122
163
|
)
|
|
123
164
|
|
|
124
165
|
|
|
@@ -159,6 +200,7 @@ def save_model(
|
|
|
159
200
|
run_id: str,
|
|
160
201
|
run_name: str,
|
|
161
202
|
requirements: ModelRequirements | None = None,
|
|
203
|
+
config: dict[str, str] | None = None,
|
|
162
204
|
) -> SavedModel:
|
|
163
205
|
"""Upload a weights directory to S3 and register a new MLflow ModelVersion
|
|
164
206
|
paired with the serve-app that fronts it.
|
|
@@ -180,9 +222,20 @@ def save_model(
|
|
|
180
222
|
without the caller holding the class object.
|
|
181
223
|
|
|
182
224
|
`requirements` is the hardware one replica needs; None stores none, which
|
|
183
|
-
reads as no requirement. Change it later with `set_model_requirements`.
|
|
225
|
+
reads as no requirement. Change it later with `set_model_requirements`.
|
|
226
|
+
|
|
227
|
+
`config` is whatever else the serve-app needs to know about this model,
|
|
228
|
+
as a string mapping it reads with `model_config` at construction; None
|
|
229
|
+
stores none. Change it later with `set_model_config`."""
|
|
184
230
|
return _upload_model(
|
|
185
|
-
weights_dir,
|
|
231
|
+
weights_dir,
|
|
232
|
+
serve_app,
|
|
233
|
+
suffix,
|
|
234
|
+
family,
|
|
235
|
+
run_id,
|
|
236
|
+
run_name,
|
|
237
|
+
requirements,
|
|
238
|
+
config,
|
|
186
239
|
)
|
|
187
240
|
|
|
188
241
|
|
|
@@ -192,6 +245,7 @@ def import_model(
|
|
|
192
245
|
family: str,
|
|
193
246
|
suffix: str,
|
|
194
247
|
requirements: ModelRequirements | None = None,
|
|
248
|
+
config: dict[str, str] | None = None,
|
|
195
249
|
) -> SavedModel:
|
|
196
250
|
"""Register a model produced elsewhere (e.g. a pretrained base model) under
|
|
197
251
|
the fixed key (family, suffix, IMPORTED), once.
|
|
@@ -205,10 +259,10 @@ def import_model(
|
|
|
205
259
|
If a version is already registered under the key:
|
|
206
260
|
- "ready": returns it without calling `source`. If `serve_app`'s code no
|
|
207
261
|
longer matches the stored bundle, it is re-bundled first and the
|
|
208
|
-
weights are kept (see `_refresh_bundle`). `requirements`
|
|
209
|
-
only if the version has none yet, so values changed since
|
|
210
|
-
`set_model_requirements` are kept. To replace
|
|
211
|
-
`delete_model` it first.
|
|
262
|
+
weights are kept (see `_refresh_bundle`). `requirements` and `config`
|
|
263
|
+
are stored only if the version has none yet, so values changed since
|
|
264
|
+
with `set_model_requirements` / `set_model_config` are kept. To replace
|
|
265
|
+
the weights, `delete_model` it first.
|
|
212
266
|
- "uploading": raises RuntimeError - another process is importing it.
|
|
213
267
|
- "upload_failed" / "broken": deleted and imported again.
|
|
214
268
|
|
|
@@ -223,6 +277,8 @@ def import_model(
|
|
|
223
277
|
existing.requirements = _set_missing_requirements(
|
|
224
278
|
family, suffix, requirements
|
|
225
279
|
)
|
|
280
|
+
if config is not None:
|
|
281
|
+
existing.config = _set_missing_config(family, suffix, config)
|
|
226
282
|
return existing
|
|
227
283
|
if existing.phase == _PHASE_UPLOADING:
|
|
228
284
|
raise RuntimeError(
|
|
@@ -231,7 +287,7 @@ def import_model(
|
|
|
231
287
|
)
|
|
232
288
|
delete_model(family, suffix, IMPORTED)
|
|
233
289
|
return _upload_model(
|
|
234
|
-
source, serve_app, suffix, family, None, IMPORTED, requirements
|
|
290
|
+
source, serve_app, suffix, family, None, IMPORTED, requirements, config
|
|
235
291
|
)
|
|
236
292
|
|
|
237
293
|
|
|
@@ -273,6 +329,24 @@ def _set_missing_requirements(
|
|
|
273
329
|
return requirements
|
|
274
330
|
|
|
275
331
|
|
|
332
|
+
def _set_missing_config(
|
|
333
|
+
family: str, suffix: str, config: dict[str, str]
|
|
334
|
+
) -> dict[str, str]:
|
|
335
|
+
"""Store `config` on an imported model whose version has none yet. Returns
|
|
336
|
+
the config the version holds afterwards."""
|
|
337
|
+
name = f"{family}__{suffix}"
|
|
338
|
+
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
339
|
+
version = client.search_model_versions(
|
|
340
|
+
f"name='{name}' and tags.run_name='{IMPORTED}'"
|
|
341
|
+
)[0]
|
|
342
|
+
if _CONFIG_TAG in version.tags:
|
|
343
|
+
return _config_from_tags(version.tags)
|
|
344
|
+
client.set_model_version_tag(
|
|
345
|
+
name, version.version, _CONFIG_TAG, _config_to_tag(config)
|
|
346
|
+
)
|
|
347
|
+
return config
|
|
348
|
+
|
|
349
|
+
|
|
276
350
|
def _upload_model(
|
|
277
351
|
weights: str | Path | Callable[[], str | Path],
|
|
278
352
|
serve_app: type,
|
|
@@ -281,6 +355,7 @@ def _upload_model(
|
|
|
281
355
|
run_id: str | None,
|
|
282
356
|
run_name: str,
|
|
283
357
|
requirements: ModelRequirements | None,
|
|
358
|
+
config: dict[str, str] | None,
|
|
284
359
|
) -> SavedModel:
|
|
285
360
|
"""Register a ModelVersion in "uploading", resolve `weights` to a directory
|
|
286
361
|
(calling it when it is a callable), upload the weights and the serve-app
|
|
@@ -296,8 +371,9 @@ def _upload_model(
|
|
|
296
371
|
# fetched and uploaded, and a concurrent `import_model` sees the import in
|
|
297
372
|
# flight for the whole download instead of starting one of its own. The
|
|
298
373
|
# size is stamped once the directory exists; the bundle tags and the flip
|
|
299
|
-
# to "ready" happen only after the upload lands. No requirements
|
|
300
|
-
# their tags unset, so a later `import_model` can still store
|
|
374
|
+
# to "ready" happen only after the upload lands. No requirements and no
|
|
375
|
+
# config leave their tags unset, so a later `import_model` can still store
|
|
376
|
+
# them.
|
|
301
377
|
version = client.create_model_version(
|
|
302
378
|
name=name,
|
|
303
379
|
source=source,
|
|
@@ -308,6 +384,7 @@ def _upload_model(
|
|
|
308
384
|
"run_name": run_name,
|
|
309
385
|
_LIFECYCLE_TAG: _PHASE_UPLOADING,
|
|
310
386
|
**(requirements_to_tags(requirements) if requirements is not None else {}),
|
|
387
|
+
**({_CONFIG_TAG: _config_to_tag(config)} if config is not None else {}),
|
|
311
388
|
},
|
|
312
389
|
)
|
|
313
390
|
try:
|
|
@@ -393,6 +470,46 @@ def set_model_requirements(
|
|
|
393
470
|
client.set_model_version_tag(name, versions[0].version, key, value)
|
|
394
471
|
|
|
395
472
|
|
|
473
|
+
def model_config(family: str, suffix: str, run_name: str) -> dict[str, str]:
|
|
474
|
+
"""The config mapping stored on a model, empty if it has none.
|
|
475
|
+
|
|
476
|
+
Meant for the serve-app to call in `__init__` with the
|
|
477
|
+
(family, suffix, run_name) it was constructed with: the settings that are
|
|
478
|
+
not the weights - a provider's model id, an endpoint, the name of a secret
|
|
479
|
+
to read - travel with the registry entry instead of the bundled code, so
|
|
480
|
+
changing one is an edit on the model card rather than a re-save.
|
|
481
|
+
|
|
482
|
+
Read at construction, so a replica keeps the values it started with until
|
|
483
|
+
it is deployed again. Raises ValueError if the model was never
|
|
484
|
+
registered."""
|
|
485
|
+
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
486
|
+
versions = client.search_model_versions(
|
|
487
|
+
f"name='{family}__{suffix}' and tags.run_name='{run_name}'"
|
|
488
|
+
)
|
|
489
|
+
if not versions:
|
|
490
|
+
raise ValueError(f"No model {family}/{suffix}/{run_name}")
|
|
491
|
+
return _config_from_tags(versions[0].tags)
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def set_model_config(
|
|
495
|
+
family: str, suffix: str, run_name: str, config: dict[str, str]
|
|
496
|
+
) -> None:
|
|
497
|
+
"""Replace the config mapping stored on a model - the whole mapping, so a
|
|
498
|
+
key left out of `config` is gone. Takes effect on its next `deploy_model`;
|
|
499
|
+
a replica already running keeps the values it read at construction.
|
|
500
|
+
Raises ValueError if the model was never registered."""
|
|
501
|
+
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
502
|
+
name = f"{family}__{suffix}"
|
|
503
|
+
versions = client.search_model_versions(
|
|
504
|
+
f"name='{name}' and tags.run_name='{run_name}'"
|
|
505
|
+
)
|
|
506
|
+
if not versions:
|
|
507
|
+
raise ValueError(f"No model {family}/{suffix}/{run_name}")
|
|
508
|
+
client.set_model_version_tag(
|
|
509
|
+
name, versions[0].version, _CONFIG_TAG, _config_to_tag(config)
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
|
|
396
513
|
def delete_model(family: str, suffix: str, run_name: str) -> None:
|
|
397
514
|
"""Delete the ModelVersion in MLflow, its weights blob, and its serve bundle."""
|
|
398
515
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
@@ -204,6 +204,8 @@ print(deployed.url)
|
|
|
204
204
|
|
|
205
205
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
206
206
|
|
|
207
|
+
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
208
|
+
|
|
207
209
|
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`.
|
|
208
210
|
|
|
209
211
|
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|