cortexgrid 0.2.90__tar.gz → 0.2.92__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.2.90
3
+ Version: 0.2.92
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -178,7 +178,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
178
178
  print(deployed.url)
179
179
  ```
180
180
 
181
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
181
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
182
182
 
183
183
  ### API reference
184
184
 
@@ -83,11 +83,13 @@ from cortexgrid.model_storage import (
83
83
  from cortexgrid.model_storage import save_model as _save_model_storage
84
84
  from cortexgrid.model_serving import (
85
85
  Deployment,
86
+ ModelDeployFailed,
86
87
  ServingStatus,
87
88
  deploy_model,
88
89
  list_deployed_models,
89
90
  model_serving_status,
90
91
  undeploy_model,
92
+ wait_for_model_serving,
91
93
  )
92
94
 
93
95
 
@@ -185,8 +187,10 @@ __all__ = [
185
187
  "delete_model",
186
188
  # Model serving
187
189
  "Deployment",
190
+ "ModelDeployFailed",
188
191
  "ServingStatus",
189
192
  "deploy_model",
193
+ "wait_for_model_serving",
190
194
  "model_serving_status",
191
195
  "undeploy_model",
192
196
  "list_deployed_models",
@@ -21,6 +21,7 @@ from __future__ import annotations
21
21
  import inspect
22
22
  import json
23
23
  import logging
24
+ import re
24
25
  import shutil
25
26
  import tempfile
26
27
  import time
@@ -247,38 +248,92 @@ def _current_application_specs() -> list[dict[str, Any]]:
247
248
  return specs
248
249
 
249
250
 
250
- def _wait_for_application_running(
251
- name: str, timeout_s: float | None = 300.0, interval_s: float = 2.0
252
- ) -> None:
253
- """Poll the Serve controller until the named application is RUNNING.
251
+ class ModelDeployFailed(RuntimeError):
252
+ """A model's Serve app cannot reach RUNNING: the controller reported
253
+ DEPLOY_FAILED, or no app exists for the model. Subclasses RuntimeError, which
254
+ `deploy_model(wait=True)` raised before this type existed."""
255
+
256
+
257
+ _SERVING_POLL_INTERVAL_S = 2.0
258
+
254
259
 
255
- Raises immediately on DEPLOY_FAILED with the controller's message. Other
256
- non-RUNNING statuses (NOT_STARTED, DEPLOYING, UNHEALTHY) are treated as
257
- transient until the timeout fires. With `timeout_s=None` there is no
258
- deadline: the loop blocks until a terminal status (RUNNING or
259
- DEPLOY_FAILED) is reached.
260
+ def _deadline(timeout: float | None) -> float | None:
261
+ return None if timeout is None else time.monotonic() + timeout
262
+
263
+
264
+ def _past(deadline: float | None) -> bool:
265
+ return deadline is not None and time.monotonic() >= deadline
266
+
267
+
268
+ def wait_for_model_serving(
269
+ family: str, suffix: str, run_name: str, timeout: float | None = None
270
+ ) -> None:
271
+ """Block until the model's Serve app is RUNNING.
272
+
273
+ Raises ModelDeployFailed on DEPLOY_FAILED, carrying the controller's message,
274
+ and as soon as no app exists for the model: never deployed, undeployed, or
275
+ dropped by a concurrent `deploy_model` (each one GETs the applications list,
276
+ splices its own app in and PUTs the whole list back, so a later PUT can drop
277
+ an app an earlier one added). NOT_STARTED, DEPLOYING, UNHEALTHY and DELETING
278
+ are transient; a DELETING app ends up missing. Exceeding a finite `timeout`
279
+ raises TimeoutError; with `timeout=None` there is no deadline.
260
280
  """
261
- deadline = None if timeout_s is None else time.monotonic() + timeout_s
262
- last_status: str = "(missing)"
263
- last_message: str = ""
264
- while deadline is None or time.monotonic() < deadline:
265
- app = get_serve_details().get("applications", {}).get(name)
266
- if app is not None:
267
- last_status = str(app.get("status", "(missing)"))
268
- last_message = str(app.get("message", ""))
269
- if last_status == ApplicationStatus.RUNNING.value:
270
- return
271
- if last_status == ApplicationStatus.DEPLOY_FAILED.value:
272
- raise RuntimeError(
273
- f"Serve app {name!r} DEPLOY_FAILED: {last_message}"
274
- )
275
- time.sleep(interval_s)
276
- raise TimeoutError(
277
- f"Serve app {name!r} did not reach RUNNING within {timeout_s}s "
278
- f"(last status={last_status!r}, message={last_message!r})"
281
+ _wait_for_application_running(
282
+ _app_name(family, suffix, run_name), timeout, _deadline(timeout)
279
283
  )
280
284
 
281
285
 
286
+ def _wait_for_application_running(
287
+ name: str, timeout: float | None, deadline: float | None
288
+ ) -> None:
289
+ """`wait_for_model_serving` against a deadline already running. `timeout`
290
+ only labels the TimeoutError."""
291
+ while True:
292
+ app = get_serve_details().get("applications", {}).get(name)
293
+ if app is None:
294
+ raise ModelDeployFailed(f"Serve app {name!r} does not exist")
295
+ status = str(app.get("status", "(missing)"))
296
+ message = str(app.get("message", ""))
297
+ if status == ApplicationStatus.RUNNING.value:
298
+ return
299
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
300
+ raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
301
+ if _past(deadline):
302
+ raise TimeoutError(
303
+ f"Serve app {name!r} did not reach RUNNING within {timeout}s "
304
+ f"(last status={status!r}, message={message!r})"
305
+ )
306
+ time.sleep(_SERVING_POLL_INTERVAL_S)
307
+
308
+
309
+ def _clear_failed_application(
310
+ family: str, suffix: str, run_name: str, timeout: float | None, deadline: float | None
311
+ ) -> None:
312
+ """Remove the model's DEPLOY_FAILED Serve app, and wait until it, or an app
313
+ already DELETING, is gone.
314
+
315
+ Ray resets a failed deployment only when a deploy arrives after the
316
+ deployment is marked for deletion or its version changes. Re-PUTting an
317
+ identical spec over a failed app, or PUTting it back before the controller's
318
+ next tick has processed an undeploy, leaves the failed deployment in place,
319
+ and the app reports DEPLOY_FAILED again without retrying."""
320
+ name = _app_name(family, suffix, run_name)
321
+ app = get_serve_details().get("applications", {}).get(name)
322
+ if app is None:
323
+ return
324
+ status = app.get("status")
325
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
326
+ undeploy_model(family, suffix, run_name)
327
+ elif status != ApplicationStatus.DELETING.value:
328
+ return
329
+ while name in get_serve_details().get("applications", {}):
330
+ if _past(deadline):
331
+ raise TimeoutError(
332
+ f"Serve app {name!r} was not removed within {timeout}s"
333
+ )
334
+ time.sleep(_SERVING_POLL_INTERVAL_S)
335
+
336
+
282
337
  def deploy_model(
283
338
  family: str,
284
339
  suffix: str,
@@ -294,19 +349,28 @@ def deploy_model(
294
349
  The serve-app class is pulled from the MLflow ModelVersion tags `save_model`
295
350
  wrote at save time; the caller does not need to hold the class object.
296
351
 
297
- With `wait=True`, blocks until the Serve controller reports the app
298
- RUNNING, capped at `timeout` seconds (default 300). DEPLOY_FAILED raises;
299
- exceeding a finite `timeout` raises TimeoutError. With `timeout=None` the
300
- wait is unbounded: it blocks until a terminal status (RUNNING or
301
- DEPLOY_FAILED) is reached. Tradeoff: an app that never reaches a terminal
302
- state (e.g. GPU-starved, stuck in DEPLOYING) will hang forever.
352
+ A DEPLOY_FAILED app left by an earlier attempt is undeployed first, and it,
353
+ or an app still DELETING, is waited out before the new spec is PUT, so the
354
+ deploy starts afresh instead of Ray reusing the failed deployment.
355
+
356
+ With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
357
+ controller reports the app RUNNING. `timeout` (default 300) caps the whole
358
+ call, clearing a failed app included; exceeding it raises TimeoutError, and
359
+ DEPLOY_FAILED raises ModelDeployFailed. With `timeout=None` there is no cap.
360
+ Tradeoff: an app that never reaches a terminal state (e.g. GPU-starved,
361
+ stuck in DEPLOYING) will hang forever.
303
362
  """
363
+ deadline = _deadline(timeout)
304
364
  meta = _load_bundle_metadata(family, suffix, run_name)
305
365
  spec = _build_application_spec(family, suffix, run_name, meta)
366
+ _clear_failed_application(family, suffix, run_name, timeout, deadline)
306
367
  existing = [a for a in _current_application_specs() if a["name"] != spec["name"]]
368
+ # The controller registers the app, sets it DEPLOYING and stamps
369
+ # last_deployed_time_s before the PUT returns, so the wait below neither
370
+ # misses the app nor reads a status left by an earlier deploy.
307
371
  put_serve_applications([*existing, spec])
308
372
  if wait:
309
- _wait_for_application_running(spec["name"], timeout_s=timeout)
373
+ _wait_for_application_running(spec["name"], timeout, deadline)
310
374
  app = get_serve_details().get("applications", {}).get(spec["name"], {})
311
375
  return Deployment(
312
376
  family=family,
@@ -404,3 +468,43 @@ def model_serving_status(
404
468
  message=str(app.get("message", "")) or raw,
405
469
  url=f"{get_ray_serve_uri()}{_route_prefix(family, suffix, run_name)}",
406
470
  )
471
+
472
+
473
+ @dataclass
474
+ class ServingMessage:
475
+ """One message the Ray Serve controller reports for a model's app. `source`
476
+ is "application" for the app-level message (e.g. the app failed to build)
477
+ or a deployment name for that deployment's message (e.g. its replicas
478
+ failed to start); `status` is the raw Ray Serve status of that source."""
479
+
480
+ source: str
481
+ status: str
482
+ message: str
483
+
484
+
485
+ # Ray colors parts of its messages (e.g. the serialization checker's "!!! FAIL")
486
+ # with ANSI escapes, which are noise outside a terminal.
487
+ _ANSI_ESCAPE = re.compile(r"\x1b\[[0-9;]*m")
488
+
489
+
490
+ def model_serving_messages(
491
+ family: str, suffix: str, run_name: str
492
+ ) -> list[ServingMessage]:
493
+ """Return the non-empty controller messages for a model's Serve app: the
494
+ app-level message first, then each deployment's. Empty when no app exists.
495
+ This is where Ray explains a DEPLOY_FAILED or UNHEALTHY app."""
496
+ app = get_serve_details().get("applications", {}).get(
497
+ _app_name(family, suffix, run_name)
498
+ )
499
+ if app is None:
500
+ return []
501
+ sources = [("application", app)] + list(app.get("deployments", {}).items())
502
+ return [
503
+ ServingMessage(
504
+ source=source,
505
+ status=str(details.get("status", "")),
506
+ message=_ANSI_ESCAPE.sub("", str(details["message"])),
507
+ )
508
+ for source, details in sources
509
+ if details.get("message")
510
+ ]
@@ -150,7 +150,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
150
150
  print(deployed.url)
151
151
  ```
152
152
 
153
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
153
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
154
154
 
155
155
  ### API reference
156
156
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.2.90"
3
+ version = "0.2.92"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes