cortexgrid 0.2.91__tar.gz → 0.2.92__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.2.91
3
+ Version: 0.2.92
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -178,7 +178,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
178
178
  print(deployed.url)
179
179
  ```
180
180
 
181
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
181
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
182
182
 
183
183
  ### API reference
184
184
 
@@ -83,11 +83,13 @@ from cortexgrid.model_storage import (
83
83
  from cortexgrid.model_storage import save_model as _save_model_storage
84
84
  from cortexgrid.model_serving import (
85
85
  Deployment,
86
+ ModelDeployFailed,
86
87
  ServingStatus,
87
88
  deploy_model,
88
89
  list_deployed_models,
89
90
  model_serving_status,
90
91
  undeploy_model,
92
+ wait_for_model_serving,
91
93
  )
92
94
 
93
95
 
@@ -185,8 +187,10 @@ __all__ = [
185
187
  "delete_model",
186
188
  # Model serving
187
189
  "Deployment",
190
+ "ModelDeployFailed",
188
191
  "ServingStatus",
189
192
  "deploy_model",
193
+ "wait_for_model_serving",
190
194
  "model_serving_status",
191
195
  "undeploy_model",
192
196
  "list_deployed_models",
@@ -248,38 +248,92 @@ def _current_application_specs() -> list[dict[str, Any]]:
248
248
  return specs
249
249
 
250
250
 
251
- def _wait_for_application_running(
252
- name: str, timeout_s: float | None = 300.0, interval_s: float = 2.0
253
- ) -> None:
254
- """Poll the Serve controller until the named application is RUNNING.
251
+ class ModelDeployFailed(RuntimeError):
252
+ """A model's Serve app cannot reach RUNNING: the controller reported
253
+ DEPLOY_FAILED, or no app exists for the model. Subclasses RuntimeError, which
254
+ `deploy_model(wait=True)` raised before this type existed."""
255
+
256
+
257
+ _SERVING_POLL_INTERVAL_S = 2.0
258
+
259
+
260
+ def _deadline(timeout: float | None) -> float | None:
261
+ return None if timeout is None else time.monotonic() + timeout
255
262
 
256
- Raises immediately on DEPLOY_FAILED with the controller's message. Other
257
- non-RUNNING statuses (NOT_STARTED, DEPLOYING, UNHEALTHY) are treated as
258
- transient until the timeout fires. With `timeout_s=None` there is no
259
- deadline: the loop blocks until a terminal status (RUNNING or
260
- DEPLOY_FAILED) is reached.
263
+
264
+ def _past(deadline: float | None) -> bool:
265
+ return deadline is not None and time.monotonic() >= deadline
266
+
267
+
268
+ def wait_for_model_serving(
269
+ family: str, suffix: str, run_name: str, timeout: float | None = None
270
+ ) -> None:
271
+ """Block until the model's Serve app is RUNNING.
272
+
273
+ Raises ModelDeployFailed on DEPLOY_FAILED, carrying the controller's message,
274
+ and as soon as no app exists for the model: never deployed, undeployed, or
275
+ dropped by a concurrent `deploy_model` (each one GETs the applications list,
276
+ splices its own app in and PUTs the whole list back, so a later PUT can drop
277
+ an app an earlier one added). NOT_STARTED, DEPLOYING, UNHEALTHY and DELETING
278
+ are transient; a DELETING app ends up missing. Exceeding a finite `timeout`
279
+ raises TimeoutError; with `timeout=None` there is no deadline.
261
280
  """
262
- deadline = None if timeout_s is None else time.monotonic() + timeout_s
263
- last_status: str = "(missing)"
264
- last_message: str = ""
265
- while deadline is None or time.monotonic() < deadline:
266
- app = get_serve_details().get("applications", {}).get(name)
267
- if app is not None:
268
- last_status = str(app.get("status", "(missing)"))
269
- last_message = str(app.get("message", ""))
270
- if last_status == ApplicationStatus.RUNNING.value:
271
- return
272
- if last_status == ApplicationStatus.DEPLOY_FAILED.value:
273
- raise RuntimeError(
274
- f"Serve app {name!r} DEPLOY_FAILED: {last_message}"
275
- )
276
- time.sleep(interval_s)
277
- raise TimeoutError(
278
- f"Serve app {name!r} did not reach RUNNING within {timeout_s}s "
279
- f"(last status={last_status!r}, message={last_message!r})"
281
+ _wait_for_application_running(
282
+ _app_name(family, suffix, run_name), timeout, _deadline(timeout)
280
283
  )
281
284
 
282
285
 
286
+ def _wait_for_application_running(
287
+ name: str, timeout: float | None, deadline: float | None
288
+ ) -> None:
289
+ """`wait_for_model_serving` against a deadline already running. `timeout`
290
+ only labels the TimeoutError."""
291
+ while True:
292
+ app = get_serve_details().get("applications", {}).get(name)
293
+ if app is None:
294
+ raise ModelDeployFailed(f"Serve app {name!r} does not exist")
295
+ status = str(app.get("status", "(missing)"))
296
+ message = str(app.get("message", ""))
297
+ if status == ApplicationStatus.RUNNING.value:
298
+ return
299
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
300
+ raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
301
+ if _past(deadline):
302
+ raise TimeoutError(
303
+ f"Serve app {name!r} did not reach RUNNING within {timeout}s "
304
+ f"(last status={status!r}, message={message!r})"
305
+ )
306
+ time.sleep(_SERVING_POLL_INTERVAL_S)
307
+
308
+
309
+ def _clear_failed_application(
310
+ family: str, suffix: str, run_name: str, timeout: float | None, deadline: float | None
311
+ ) -> None:
312
+ """Remove the model's DEPLOY_FAILED Serve app, and wait until it, or an app
313
+ already DELETING, is gone.
314
+
315
+ Ray resets a failed deployment only when a deploy arrives after the
316
+ deployment is marked for deletion or its version changes. Re-PUTting an
317
+ identical spec over a failed app, or PUTting it back before the controller's
318
+ next tick has processed an undeploy, leaves the failed deployment in place,
319
+ and the app reports DEPLOY_FAILED again without retrying."""
320
+ name = _app_name(family, suffix, run_name)
321
+ app = get_serve_details().get("applications", {}).get(name)
322
+ if app is None:
323
+ return
324
+ status = app.get("status")
325
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
326
+ undeploy_model(family, suffix, run_name)
327
+ elif status != ApplicationStatus.DELETING.value:
328
+ return
329
+ while name in get_serve_details().get("applications", {}):
330
+ if _past(deadline):
331
+ raise TimeoutError(
332
+ f"Serve app {name!r} was not removed within {timeout}s"
333
+ )
334
+ time.sleep(_SERVING_POLL_INTERVAL_S)
335
+
336
+
283
337
  def deploy_model(
284
338
  family: str,
285
339
  suffix: str,
@@ -295,19 +349,28 @@ def deploy_model(
295
349
  The serve-app class is pulled from the MLflow ModelVersion tags `save_model`
296
350
  wrote at save time; the caller does not need to hold the class object.
297
351
 
298
- With `wait=True`, blocks until the Serve controller reports the app
299
- RUNNING, capped at `timeout` seconds (default 300). DEPLOY_FAILED raises;
300
- exceeding a finite `timeout` raises TimeoutError. With `timeout=None` the
301
- wait is unbounded: it blocks until a terminal status (RUNNING or
302
- DEPLOY_FAILED) is reached. Tradeoff: an app that never reaches a terminal
303
- state (e.g. GPU-starved, stuck in DEPLOYING) will hang forever.
352
+ A DEPLOY_FAILED app left by an earlier attempt is undeployed first, and it,
353
+ or an app still DELETING, is waited out before the new spec is PUT, so the
354
+ deploy starts afresh instead of Ray reusing the failed deployment.
355
+
356
+ With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
357
+ controller reports the app RUNNING. `timeout` (default 300) caps the whole
358
+ call, clearing a failed app included; exceeding it raises TimeoutError, and
359
+ DEPLOY_FAILED raises ModelDeployFailed. With `timeout=None` there is no cap.
360
+ Tradeoff: an app that never reaches a terminal state (e.g. GPU-starved,
361
+ stuck in DEPLOYING) will hang forever.
304
362
  """
363
+ deadline = _deadline(timeout)
305
364
  meta = _load_bundle_metadata(family, suffix, run_name)
306
365
  spec = _build_application_spec(family, suffix, run_name, meta)
366
+ _clear_failed_application(family, suffix, run_name, timeout, deadline)
307
367
  existing = [a for a in _current_application_specs() if a["name"] != spec["name"]]
368
+ # The controller registers the app, sets it DEPLOYING and stamps
369
+ # last_deployed_time_s before the PUT returns, so the wait below neither
370
+ # misses the app nor reads a status left by an earlier deploy.
308
371
  put_serve_applications([*existing, spec])
309
372
  if wait:
310
- _wait_for_application_running(spec["name"], timeout_s=timeout)
373
+ _wait_for_application_running(spec["name"], timeout, deadline)
311
374
  app = get_serve_details().get("applications", {}).get(spec["name"], {})
312
375
  return Deployment(
313
376
  family=family,
@@ -150,7 +150,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
150
150
  print(deployed.url)
151
151
  ```
152
152
 
153
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
153
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
154
154
 
155
155
  ### API reference
156
156
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.2.91"
3
+ version = "0.2.92"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes