cortexgrid 0.2.88__tar.gz → 0.2.89__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/PKG-INFO +2 -2
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/experiment.py +40 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/docs/cortexgrid/README.md +1 -1
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/pyproject.toml +1 -1
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/.gitignore +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/LICENSE +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/__init__.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/model_serving.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.2.88 → cortexgrid-0.2.89}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.89
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -55,7 +55,7 @@ import cortexgrid
|
|
|
55
55
|
cortexgrid.init(experiment="weather-forecast")
|
|
56
56
|
```
|
|
57
57
|
|
|
58
|
-
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named MLflow experiment and starts a new run inside it. Omit `experiment=` to auto-generate a unique name like `funky-koval-12`.
|
|
58
|
+
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named MLflow experiment and starts a new run inside it. If an experiment of that name was deleted (e.g. from the UI), a new experiment is created under the name: MLflow keeps a deleted experiment's name reserved, so the deleted one is renamed to `<name>__deleted__<id>` first (`delete_experiment` does that rename at deletion). Omit `experiment=` to auto-generate a unique name like `funky-koval-12`.
|
|
59
59
|
|
|
60
60
|
**One experiment per binary run.** `cortexgrid.init()` may only be called once per process. Every subsequent `cortexgrid.log_metric`, `cortexgrid.log_artifact`, checkpoint, and `cortexgrid.remote()` submission is scoped to that experiment+run. Remote jobs dispatched by the control plane inherit the experiment+run via the pickled payload, so their logging flows into the same MLflow run as the parent binary.
|
|
61
61
|
|
|
@@ -10,6 +10,7 @@ from cortexgrid.jobs import stop_experiment_run_jobs
|
|
|
10
10
|
from cortexgrid.ray_util import list_ray_jobs_with_submission_id, stop_ray_job
|
|
11
11
|
from cortexgrid.model_storage import delete_models_for_run
|
|
12
12
|
from haikunator import Haikunator # type: ignore
|
|
13
|
+
from mlflow.entities import Experiment as MlflowExperiment
|
|
13
14
|
from mlflow.tracking import MlflowClient
|
|
14
15
|
|
|
15
16
|
|
|
@@ -122,7 +123,12 @@ def _try_create_experiment_and_run(
|
|
|
122
123
|
experiment = name_gen.haikunate(token_length=2, token_chars="0123456789")
|
|
123
124
|
|
|
124
125
|
client = MlflowClient(tracking_uri=mlflow_tracking_uri)
|
|
126
|
+
# get_experiment_by_name also returns deleted experiments. A deleted one
|
|
127
|
+
# cannot take new runs, so a fresh experiment is created under its name.
|
|
125
128
|
experiment_obj = client.get_experiment_by_name(name=experiment)
|
|
129
|
+
if experiment_obj is not None and experiment_obj.lifecycle_stage != "active":
|
|
130
|
+
_release_deleted_experiment_name(client, experiment_obj)
|
|
131
|
+
experiment_obj = None
|
|
126
132
|
if experiment_obj:
|
|
127
133
|
experiment_id = experiment_obj.experiment_id
|
|
128
134
|
else:
|
|
@@ -134,6 +140,33 @@ def _try_create_experiment_and_run(
|
|
|
134
140
|
return (experiment, run.info.run_id)
|
|
135
141
|
|
|
136
142
|
|
|
143
|
+
def _deleted_experiment_name(name: str, experiment_id: str) -> str:
|
|
144
|
+
"""The name a deleted experiment is moved to, freeing `name` for reuse.
|
|
145
|
+
Unique because experiment ids are."""
|
|
146
|
+
return f"{name}__deleted__{experiment_id}"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _release_deleted_experiment_name(
|
|
150
|
+
client: MlflowClient, experiment: MlflowExperiment
|
|
151
|
+
) -> None:
|
|
152
|
+
"""Move a deleted experiment off its name, so a new experiment can take it.
|
|
153
|
+
|
|
154
|
+
MLflow keeps a deleted experiment's name reserved (experiment names are
|
|
155
|
+
unique across every lifecycle stage) and refuses to rename a deleted
|
|
156
|
+
experiment, so it is restored only for the rename and deleted again."""
|
|
157
|
+
log.info(
|
|
158
|
+
"Experiment %r (id %s) is deleted; renaming it to release the name",
|
|
159
|
+
experiment.name,
|
|
160
|
+
experiment.experiment_id,
|
|
161
|
+
)
|
|
162
|
+
client.restore_experiment(experiment.experiment_id)
|
|
163
|
+
client.rename_experiment(
|
|
164
|
+
experiment.experiment_id,
|
|
165
|
+
_deleted_experiment_name(experiment.name, experiment.experiment_id),
|
|
166
|
+
)
|
|
167
|
+
client.delete_experiment(experiment.experiment_id)
|
|
168
|
+
|
|
169
|
+
|
|
137
170
|
def delete_run(run_id: str) -> None:
|
|
138
171
|
"""Soft-delete a run in MLflow, cancel its Ray attempts, and wipe its
|
|
139
172
|
S3 job packages so it cannot be relaunched or re-read.
|
|
@@ -174,6 +207,10 @@ def list_run_ids_in_experiment(name: str) -> list[str]:
|
|
|
174
207
|
def delete_experiment(name: str) -> None:
|
|
175
208
|
"""Soft-delete every run in the experiment, then the experiment itself.
|
|
176
209
|
|
|
210
|
+
The experiment is renamed (`<name>__deleted__<id>`) before it is deleted:
|
|
211
|
+
MLflow keeps a deleted experiment's name reserved, and the rename frees it
|
|
212
|
+
so `Experiment.init(name)` can create a new experiment under it.
|
|
213
|
+
|
|
177
214
|
Idempotent: already-deleted experiments are treated as success."""
|
|
178
215
|
log.info("delete_experiment(%r): start", name)
|
|
179
216
|
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
@@ -190,6 +227,9 @@ def delete_experiment(name: str) -> None:
|
|
|
190
227
|
log.info("delete_experiment(%r): %d active run(s) to delete", name, len(runs))
|
|
191
228
|
for run in runs:
|
|
192
229
|
delete_run(run.info.run_id)
|
|
230
|
+
client.rename_experiment(
|
|
231
|
+
exp.experiment_id, _deleted_experiment_name(name, exp.experiment_id)
|
|
232
|
+
)
|
|
193
233
|
client.delete_experiment(exp.experiment_id)
|
|
194
234
|
log.info("delete_experiment(%r): done", name)
|
|
195
235
|
|
|
@@ -27,7 +27,7 @@ import cortexgrid
|
|
|
27
27
|
cortexgrid.init(experiment="weather-forecast")
|
|
28
28
|
```
|
|
29
29
|
|
|
30
|
-
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named MLflow experiment and starts a new run inside it. Omit `experiment=` to auto-generate a unique name like `funky-koval-12`.
|
|
30
|
+
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named MLflow experiment and starts a new run inside it. If an experiment of that name was deleted (e.g. from the UI), a new experiment is created under the name: MLflow keeps a deleted experiment's name reserved, so the deleted one is renamed to `<name>__deleted__<id>` first (`delete_experiment` does that rename at deletion). Omit `experiment=` to auto-generate a unique name like `funky-koval-12`.
|
|
31
31
|
|
|
32
32
|
**One experiment per binary run.** `cortexgrid.init()` may only be called once per process. Every subsequent `cortexgrid.log_metric`, `cortexgrid.log_artifact`, checkpoint, and `cortexgrid.remote()` submission is scoped to that experiment+run. Remote jobs dispatched by the control plane inherit the experiment+run via the pickled payload, so their logging flows into the same MLflow run as the parent binary.
|
|
33
33
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|