cortexgrid 0.2.90__tar.gz → 0.2.92__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/PKG-INFO +2 -2
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/__init__.py +4 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/model_serving.py +138 -34
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/docs/cortexgrid/README.md +1 -1
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/pyproject.toml +1 -1
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/.gitignore +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/LICENSE +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.2.90 → cortexgrid-0.2.92}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.92
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -178,7 +178,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
178
178
|
print(deployed.url)
|
|
179
179
|
```
|
|
180
180
|
|
|
181
|
-
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
181
|
+
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
182
182
|
|
|
183
183
|
### API reference
|
|
184
184
|
|
|
@@ -83,11 +83,13 @@ from cortexgrid.model_storage import (
|
|
|
83
83
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
84
84
|
from cortexgrid.model_serving import (
|
|
85
85
|
Deployment,
|
|
86
|
+
ModelDeployFailed,
|
|
86
87
|
ServingStatus,
|
|
87
88
|
deploy_model,
|
|
88
89
|
list_deployed_models,
|
|
89
90
|
model_serving_status,
|
|
90
91
|
undeploy_model,
|
|
92
|
+
wait_for_model_serving,
|
|
91
93
|
)
|
|
92
94
|
|
|
93
95
|
|
|
@@ -185,8 +187,10 @@ __all__ = [
|
|
|
185
187
|
"delete_model",
|
|
186
188
|
# Model serving
|
|
187
189
|
"Deployment",
|
|
190
|
+
"ModelDeployFailed",
|
|
188
191
|
"ServingStatus",
|
|
189
192
|
"deploy_model",
|
|
193
|
+
"wait_for_model_serving",
|
|
190
194
|
"model_serving_status",
|
|
191
195
|
"undeploy_model",
|
|
192
196
|
"list_deployed_models",
|
|
@@ -21,6 +21,7 @@ from __future__ import annotations
|
|
|
21
21
|
import inspect
|
|
22
22
|
import json
|
|
23
23
|
import logging
|
|
24
|
+
import re
|
|
24
25
|
import shutil
|
|
25
26
|
import tempfile
|
|
26
27
|
import time
|
|
@@ -247,38 +248,92 @@ def _current_application_specs() -> list[dict[str, Any]]:
|
|
|
247
248
|
return specs
|
|
248
249
|
|
|
249
250
|
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
251
|
+
class ModelDeployFailed(RuntimeError):
|
|
252
|
+
"""A model's Serve app cannot reach RUNNING: the controller reported
|
|
253
|
+
DEPLOY_FAILED, or no app exists for the model. Subclasses RuntimeError, which
|
|
254
|
+
`deploy_model(wait=True)` raised before this type existed."""
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
_SERVING_POLL_INTERVAL_S = 2.0
|
|
258
|
+
|
|
254
259
|
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
+
def _deadline(timeout: float | None) -> float | None:
|
|
261
|
+
return None if timeout is None else time.monotonic() + timeout
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _past(deadline: float | None) -> bool:
|
|
265
|
+
return deadline is not None and time.monotonic() >= deadline
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def wait_for_model_serving(
|
|
269
|
+
family: str, suffix: str, run_name: str, timeout: float | None = None
|
|
270
|
+
) -> None:
|
|
271
|
+
"""Block until the model's Serve app is RUNNING.
|
|
272
|
+
|
|
273
|
+
Raises ModelDeployFailed on DEPLOY_FAILED, carrying the controller's message,
|
|
274
|
+
and as soon as no app exists for the model: never deployed, undeployed, or
|
|
275
|
+
dropped by a concurrent `deploy_model` (each one GETs the applications list,
|
|
276
|
+
splices its own app in and PUTs the whole list back, so a later PUT can drop
|
|
277
|
+
an app an earlier one added). NOT_STARTED, DEPLOYING, UNHEALTHY and DELETING
|
|
278
|
+
are transient; a DELETING app ends up missing. Exceeding a finite `timeout`
|
|
279
|
+
raises TimeoutError; with `timeout=None` there is no deadline.
|
|
260
280
|
"""
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
last_message: str = ""
|
|
264
|
-
while deadline is None or time.monotonic() < deadline:
|
|
265
|
-
app = get_serve_details().get("applications", {}).get(name)
|
|
266
|
-
if app is not None:
|
|
267
|
-
last_status = str(app.get("status", "(missing)"))
|
|
268
|
-
last_message = str(app.get("message", ""))
|
|
269
|
-
if last_status == ApplicationStatus.RUNNING.value:
|
|
270
|
-
return
|
|
271
|
-
if last_status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
272
|
-
raise RuntimeError(
|
|
273
|
-
f"Serve app {name!r} DEPLOY_FAILED: {last_message}"
|
|
274
|
-
)
|
|
275
|
-
time.sleep(interval_s)
|
|
276
|
-
raise TimeoutError(
|
|
277
|
-
f"Serve app {name!r} did not reach RUNNING within {timeout_s}s "
|
|
278
|
-
f"(last status={last_status!r}, message={last_message!r})"
|
|
281
|
+
_wait_for_application_running(
|
|
282
|
+
_app_name(family, suffix, run_name), timeout, _deadline(timeout)
|
|
279
283
|
)
|
|
280
284
|
|
|
281
285
|
|
|
286
|
+
def _wait_for_application_running(
|
|
287
|
+
name: str, timeout: float | None, deadline: float | None
|
|
288
|
+
) -> None:
|
|
289
|
+
"""`wait_for_model_serving` against a deadline already running. `timeout`
|
|
290
|
+
only labels the TimeoutError."""
|
|
291
|
+
while True:
|
|
292
|
+
app = get_serve_details().get("applications", {}).get(name)
|
|
293
|
+
if app is None:
|
|
294
|
+
raise ModelDeployFailed(f"Serve app {name!r} does not exist")
|
|
295
|
+
status = str(app.get("status", "(missing)"))
|
|
296
|
+
message = str(app.get("message", ""))
|
|
297
|
+
if status == ApplicationStatus.RUNNING.value:
|
|
298
|
+
return
|
|
299
|
+
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
300
|
+
raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
|
|
301
|
+
if _past(deadline):
|
|
302
|
+
raise TimeoutError(
|
|
303
|
+
f"Serve app {name!r} did not reach RUNNING within {timeout}s "
|
|
304
|
+
f"(last status={status!r}, message={message!r})"
|
|
305
|
+
)
|
|
306
|
+
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _clear_failed_application(
|
|
310
|
+
family: str, suffix: str, run_name: str, timeout: float | None, deadline: float | None
|
|
311
|
+
) -> None:
|
|
312
|
+
"""Remove the model's DEPLOY_FAILED Serve app, and wait until it, or an app
|
|
313
|
+
already DELETING, is gone.
|
|
314
|
+
|
|
315
|
+
Ray resets a failed deployment only when a deploy arrives after the
|
|
316
|
+
deployment is marked for deletion or its version changes. Re-PUTting an
|
|
317
|
+
identical spec over a failed app, or PUTting it back before the controller's
|
|
318
|
+
next tick has processed an undeploy, leaves the failed deployment in place,
|
|
319
|
+
and the app reports DEPLOY_FAILED again without retrying."""
|
|
320
|
+
name = _app_name(family, suffix, run_name)
|
|
321
|
+
app = get_serve_details().get("applications", {}).get(name)
|
|
322
|
+
if app is None:
|
|
323
|
+
return
|
|
324
|
+
status = app.get("status")
|
|
325
|
+
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
326
|
+
undeploy_model(family, suffix, run_name)
|
|
327
|
+
elif status != ApplicationStatus.DELETING.value:
|
|
328
|
+
return
|
|
329
|
+
while name in get_serve_details().get("applications", {}):
|
|
330
|
+
if _past(deadline):
|
|
331
|
+
raise TimeoutError(
|
|
332
|
+
f"Serve app {name!r} was not removed within {timeout}s"
|
|
333
|
+
)
|
|
334
|
+
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
335
|
+
|
|
336
|
+
|
|
282
337
|
def deploy_model(
|
|
283
338
|
family: str,
|
|
284
339
|
suffix: str,
|
|
@@ -294,19 +349,28 @@ def deploy_model(
|
|
|
294
349
|
The serve-app class is pulled from the MLflow ModelVersion tags `save_model`
|
|
295
350
|
wrote at save time; the caller does not need to hold the class object.
|
|
296
351
|
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
352
|
+
A DEPLOY_FAILED app left by an earlier attempt is undeployed first, and it,
|
|
353
|
+
or an app still DELETING, is waited out before the new spec is PUT, so the
|
|
354
|
+
deploy starts afresh instead of Ray reusing the failed deployment.
|
|
355
|
+
|
|
356
|
+
With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
|
|
357
|
+
controller reports the app RUNNING. `timeout` (default 300) caps the whole
|
|
358
|
+
call, clearing a failed app included; exceeding it raises TimeoutError, and
|
|
359
|
+
DEPLOY_FAILED raises ModelDeployFailed. With `timeout=None` there is no cap.
|
|
360
|
+
Tradeoff: an app that never reaches a terminal state (e.g. GPU-starved,
|
|
361
|
+
stuck in DEPLOYING) will hang forever.
|
|
303
362
|
"""
|
|
363
|
+
deadline = _deadline(timeout)
|
|
304
364
|
meta = _load_bundle_metadata(family, suffix, run_name)
|
|
305
365
|
spec = _build_application_spec(family, suffix, run_name, meta)
|
|
366
|
+
_clear_failed_application(family, suffix, run_name, timeout, deadline)
|
|
306
367
|
existing = [a for a in _current_application_specs() if a["name"] != spec["name"]]
|
|
368
|
+
# The controller registers the app, sets it DEPLOYING and stamps
|
|
369
|
+
# last_deployed_time_s before the PUT returns, so the wait below neither
|
|
370
|
+
# misses the app nor reads a status left by an earlier deploy.
|
|
307
371
|
put_serve_applications([*existing, spec])
|
|
308
372
|
if wait:
|
|
309
|
-
_wait_for_application_running(spec["name"],
|
|
373
|
+
_wait_for_application_running(spec["name"], timeout, deadline)
|
|
310
374
|
app = get_serve_details().get("applications", {}).get(spec["name"], {})
|
|
311
375
|
return Deployment(
|
|
312
376
|
family=family,
|
|
@@ -404,3 +468,43 @@ def model_serving_status(
|
|
|
404
468
|
message=str(app.get("message", "")) or raw,
|
|
405
469
|
url=f"{get_ray_serve_uri()}{_route_prefix(family, suffix, run_name)}",
|
|
406
470
|
)
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
@dataclass
|
|
474
|
+
class ServingMessage:
|
|
475
|
+
"""One message the Ray Serve controller reports for a model's app. `source`
|
|
476
|
+
is "application" for the app-level message (e.g. the app failed to build)
|
|
477
|
+
or a deployment name for that deployment's message (e.g. its replicas
|
|
478
|
+
failed to start); `status` is the raw Ray Serve status of that source."""
|
|
479
|
+
|
|
480
|
+
source: str
|
|
481
|
+
status: str
|
|
482
|
+
message: str
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
# Ray colors parts of its messages (e.g. the serialization checker's "!!! FAIL")
|
|
486
|
+
# with ANSI escapes, which are noise outside a terminal.
|
|
487
|
+
_ANSI_ESCAPE = re.compile(r"\x1b\[[0-9;]*m")
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def model_serving_messages(
|
|
491
|
+
family: str, suffix: str, run_name: str
|
|
492
|
+
) -> list[ServingMessage]:
|
|
493
|
+
"""Return the non-empty controller messages for a model's Serve app: the
|
|
494
|
+
app-level message first, then each deployment's. Empty when no app exists.
|
|
495
|
+
This is where Ray explains a DEPLOY_FAILED or UNHEALTHY app."""
|
|
496
|
+
app = get_serve_details().get("applications", {}).get(
|
|
497
|
+
_app_name(family, suffix, run_name)
|
|
498
|
+
)
|
|
499
|
+
if app is None:
|
|
500
|
+
return []
|
|
501
|
+
sources = [("application", app)] + list(app.get("deployments", {}).items())
|
|
502
|
+
return [
|
|
503
|
+
ServingMessage(
|
|
504
|
+
source=source,
|
|
505
|
+
status=str(details.get("status", "")),
|
|
506
|
+
message=_ANSI_ESCAPE.sub("", str(details["message"])),
|
|
507
|
+
)
|
|
508
|
+
for source, details in sources
|
|
509
|
+
if details.get("message")
|
|
510
|
+
]
|
|
@@ -150,7 +150,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
150
150
|
print(deployed.url)
|
|
151
151
|
```
|
|
152
152
|
|
|
153
|
-
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
153
|
+
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
154
154
|
|
|
155
155
|
### API reference
|
|
156
156
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|