cortexgrid 0.3.11__tar.gz → 0.3.12__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/PKG-INFO +7 -5
  2. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/__init__.py +8 -0
  3. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_serve_entry.py +14 -5
  4. cortexgrid-0.3.12/cortexgrid/model_serving/__init__.py +73 -0
  5. cortexgrid-0.3.12/cortexgrid/model_serving/application_spec.py +70 -0
  6. cortexgrid-0.3.12/cortexgrid/model_serving/deployment_key.py +38 -0
  7. cortexgrid-0.3.12/cortexgrid/model_serving/deployment_records.py +46 -0
  8. cortexgrid-0.3.12/cortexgrid/model_serving/lifecycle.py +338 -0
  9. cortexgrid-0.3.12/cortexgrid/model_serving/placement.py +151 -0
  10. cortexgrid-0.3.12/cortexgrid/model_serving/registry_tags.py +97 -0
  11. cortexgrid-0.3.12/cortexgrid/model_serving/serve_bundle.py +143 -0
  12. cortexgrid-0.3.12/cortexgrid/model_serving/status.py +277 -0
  13. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/model_storage.py +18 -10
  14. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/state.py +8 -4
  15. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/docs/cortexgrid/README.md +6 -4
  16. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/pyproject.toml +8 -1
  17. cortexgrid-0.3.11/cortexgrid/model_serving.py +0 -953
  18. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/.gitignore +0 -0
  19. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/LICENSE +0 -0
  20. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_bundle.py +0 -0
  21. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_model_scheduler.py +0 -0
  22. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_ray_job_driver.py +0 -0
  23. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/checkpoint.py +0 -0
  24. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/experiment.py +0 -0
  25. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/infra.py +0 -0
  26. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/jobs.py +0 -0
  27. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/mlflow_util.py +0 -0
  28. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/py.typed +0 -0
  29. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/ray_util.py +0 -0
  30. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/s3_util.py +0 -0
  31. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/secrets.py +0 -0
  32. {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/serve.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.11
3
+ Version: 0.3.12
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -215,8 +215,10 @@ app = FastAPI()
215
215
 
216
216
  @serve.ingress(app)
217
217
  class MyServeApp:
218
- def __init__(self, family: str, suffix: str, run_name: str) -> None:
219
- self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
218
+ def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
219
+ self._weights_dir = cortexgrid.load_model(
220
+ deployment.family, deployment.suffix, deployment.run_name
221
+ )
220
222
 
221
223
  @app.post("/complete")
222
224
  async def complete(self, body: dict): ...
@@ -232,11 +234,11 @@ print(deployed.url)
232
234
 
233
235
  The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
234
236
 
235
- Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
237
+ Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(deployment)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. How one deployment serves the model - thinking on or off, say - is that deployment's own: `deploy_model(..., config={"thinking": "false"})` deploys the model once per distinct config, each deployment named by the `DeploymentKey` on the returned `Deployment`, and `model_config` lays the deployment's config over the model's. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
236
238
 
237
239
  `save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
238
240
 
239
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
241
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(deployment.key, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
240
242
 
241
243
  ### API reference
242
244
 
@@ -101,12 +101,16 @@ from cortexgrid.model_storage import register_model as _register_model_storage
101
101
  from cortexgrid.model_storage import save_model as _save_model_storage
102
102
  from cortexgrid.model_serving import (
103
103
  Deployment,
104
+ DeploymentConfig,
105
+ DeploymentKey,
104
106
  ModelDeployFailed,
107
+ ModelNotDeployed,
105
108
  ModelRequirements,
106
109
  ServingStatus,
107
110
  deploy_model,
108
111
  list_deployed_models,
109
112
  model_serving_status,
113
+ redeploy_model,
110
114
  undeploy_model,
111
115
  wait_for_model_serving,
112
116
  )
@@ -308,9 +312,13 @@ __all__ = [
308
312
  "delete_model",
309
313
  # Model serving
310
314
  "Deployment",
315
+ "DeploymentConfig",
316
+ "DeploymentKey",
311
317
  "ModelDeployFailed",
318
+ "ModelNotDeployed",
312
319
  "ServingStatus",
313
320
  "deploy_model",
321
+ "redeploy_model",
314
322
  "wait_for_model_serving",
315
323
  "model_serving_status",
316
324
  "undeploy_model",
@@ -6,13 +6,14 @@ On the cluster replica, `build` imports the serve-app class bundled at
6
6
  ingress with the app it was marked with by `cortexgrid.serve.ingress` (again on
7
7
  each replica, see `_IngressOnReplica`), wraps it as a Ray Serve deployment
8
8
  with the replica count and Ray resource requests `deploy_model` derived from
9
- the model's requirements, and binds it with the (family, suffix, run_name)
10
- identifiers.
9
+ the model's requirements, and binds it with its deployment's
10
+ `cortexgrid.DeploymentKey`.
11
11
 
12
12
  The serve-app owns everything about traffic: its own routes, request schemas,
13
13
  streaming, and timeouts. cortexgrid does not interpose a request/response
14
- contract - it only schedules the app and hands it the identifiers it needs to
15
- fetch its own weights via `cortexgrid.load_model`.
14
+ contract - it only schedules the app and hands it the key it needs to fetch
15
+ its own weights via `cortexgrid.load_model` and its settings via
16
+ `cortexgrid.model_config`.
16
17
 
17
18
  The serve-app declares no resources: the hardware a replica needs belongs to
18
19
  the model and is stored in the registry (`cortexgrid.ModelRequirements`), and
@@ -28,6 +29,7 @@ from ray import serve
28
29
  from ray.serve.deployment import Application
29
30
 
30
31
  from cortexgrid._model_scheduler import model_autoscaling_config
32
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
31
33
  from cortexgrid.serve import ingress_app
32
34
 
33
35
 
@@ -80,4 +82,11 @@ def build(args: dict[str, Any]) -> Application:
80
82
  # had on Ray 2.9.
81
83
  max_ongoing_requests=_MAX_ONGOING_REQUESTS,
82
84
  ray_actor_options=args.get("ray_actor_options", {}),
83
- ).bind(args["family"], args["suffix"], args["run_name"])
85
+ ).bind(
86
+ DeploymentKey(
87
+ args["family"],
88
+ args["suffix"],
89
+ args["run_name"],
90
+ args.get("config_fingerprint", ""),
91
+ )
92
+ )
@@ -0,0 +1,73 @@
1
+ """Cortexgrid wrappers around Ray Serve.
2
+
3
+ Caller stays HTTP-only: deploy/undeploy/list talk to the Ray dashboard's
4
+ declarative `/api/serve/applications/` endpoint via [cortexgrid.ray_util],
5
+ never `ray.init`. The deployment class is bundled at `save_model` time, zipped,
6
+ uploaded to MinIO under
7
+ `serve-bundles/<run_name>/<family>__<suffix>/<fingerprint>.zip`, and
8
+ referenced via `runtime_env.working_dir` so Ray workers fetch it from there.
9
+ The bundle URL, class import path, pip list, and fingerprint are persisted as
10
+ tags on the model's registry entry so `deploy_model` can find them later
11
+ without the caller holding the class object. So are the model's
12
+ `ModelRequirements`, which `deploy_model` turns into the replica's Ray resource
13
+ requests.
14
+
15
+ Every deployment `deploy_model` puts on Ray Serve gets a record with the jobs
16
+ control plane, keyed by its `DeploymentKey` - the model's (family, suffix,
17
+ run_name) plus a fingerprint of the config it was deployed with: that config,
18
+ the spec it PUT, plus the phase, message and replica
19
+ placements the control plane last observed (`observe_deployments`, run on
20
+ every poll cycle). Listings and status reads come from those records; waits
21
+ ask the Serve controller directly.
22
+
23
+ Naming: the Ray Serve application is named "<family>__<suffix>__<run_name>",
24
+ followed by "__<config_fingerprint>" for a deployment given a config.
25
+ This relies on family/suffix/run_name not containing the literal "__".
26
+
27
+ See [docs/cortexgrid/model-serving.md](../../docs/cortexgrid/model-serving.md)
28
+ for the end-to-end design.
29
+ """
30
+
31
+ from cortexgrid.model_serving.application_spec import app_name
32
+ from cortexgrid.model_serving.deployment_key import (
33
+ DeploymentConfig,
34
+ DeploymentKey,
35
+ deployment_key,
36
+ )
37
+ from cortexgrid.model_serving.lifecycle import (
38
+ ModelDeployFailed,
39
+ ModelNotDeployed,
40
+ deploy_model,
41
+ redeploy_model,
42
+ undeploy_model,
43
+ wait_for_model_serving,
44
+ )
45
+ from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
46
+ from cortexgrid.model_serving.registry_tags import (
47
+ bundle_fingerprint_from_tags,
48
+ has_requirement_tags,
49
+ metadata_from_tags,
50
+ metadata_to_tags,
51
+ requirements_from_tags,
52
+ requirements_to_tags,
53
+ )
54
+ from cortexgrid.model_serving.serve_bundle import (
55
+ BundleMetadata,
56
+ ServeBundle,
57
+ build_bundle,
58
+ bundle_class,
59
+ bundle_fingerprint_from_url,
60
+ upload_bundle,
61
+ )
62
+ from cortexgrid.model_serving.status import (
63
+ Deployment,
64
+ ReplicaPlacement,
65
+ ServingMessage,
66
+ ServingStatus,
67
+ deployment_config,
68
+ list_deployed_models,
69
+ model_replica_placements,
70
+ model_serving_messages,
71
+ model_serving_status,
72
+ observe_deployments,
73
+ )
@@ -0,0 +1,70 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
6
+ from cortexgrid.model_serving.placement import ModelRequirements, ray_actor_options
7
+ from cortexgrid.model_serving.serve_bundle import (
8
+ BundleMetadata,
9
+ bundle_fingerprint_from_url,
10
+ )
11
+
12
+
13
+ def app_name(key: DeploymentKey) -> str:
14
+ return "__".join(_name_segments(key))
15
+
16
+
17
+ def route_prefix(key: DeploymentKey) -> str:
18
+ return "/r/" + "/".join(_name_segments(key))
19
+
20
+
21
+ def _name_segments(key: DeploymentKey) -> list[str]:
22
+ model_segments = [key.family, key.suffix, key.run_name]
23
+ if not key.config_fingerprint:
24
+ return model_segments
25
+ return [*model_segments, key.config_fingerprint]
26
+
27
+
28
+ def build_application_spec(
29
+ key: DeploymentKey,
30
+ meta: BundleMetadata,
31
+ requirements: ModelRequirements,
32
+ num_replicas: int,
33
+ tiers: list[int],
34
+ ) -> dict[str, Any]:
35
+ """Assemble a Ray Serve application schema from pre-bundled metadata, the
36
+ model's requirements, and the cluster's GPU size classes."""
37
+ # working_dir carries the serve-app's own source; Ray pip-installs the
38
+ # third-party distributions the image lacks into a per-node cached
39
+ # virtualenv layered on the image. No pip key when there are none, so Ray
40
+ # builds no virtualenv.
41
+ runtime_env: dict[str, Any] = {"working_dir": meta.bundle_url}
42
+ if meta.pip_requirements:
43
+ runtime_env["pip"] = meta.pip_requirements
44
+ return {
45
+ "name": app_name(key),
46
+ "route_prefix": route_prefix(key),
47
+ # Ray Serve REST requires import_path to point at an Application builder
48
+ # (callable returning a bound node) or an already-bound node. A bare
49
+ # Deployment class is rejected, so cortexgrid.deploy_model goes through
50
+ # a generic builder that re-imports the user's class and binds it.
51
+ "import_path": "cortexgrid._serve_entry:build",
52
+ "args": {
53
+ "class_import_path": meta.class_import_path,
54
+ "family": key.family,
55
+ "suffix": key.suffix,
56
+ "run_name": key.run_name,
57
+ "config_fingerprint": key.config_fingerprint,
58
+ "num_replicas": num_replicas,
59
+ "ray_actor_options": ray_actor_options(requirements, tiers),
60
+ },
61
+ "runtime_env": runtime_env,
62
+ }
63
+
64
+
65
+ def bundle_fingerprint_in_spec(spec: dict[str, Any]) -> str:
66
+ return bundle_fingerprint_from_url(spec["runtime_env"]["working_dir"])
67
+
68
+
69
+ def replica_count_in_spec(spec: dict[str, Any]) -> int:
70
+ return spec["args"]["num_replicas"]
@@ -0,0 +1,38 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from dataclasses import dataclass
6
+
7
+
8
+ DeploymentConfig = dict[str, str]
9
+
10
+ _CONFIG_FINGERPRINT_LENGTH = 12
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class DeploymentKey:
15
+ family: str
16
+ suffix: str
17
+ run_name: str
18
+ config_fingerprint: str = ""
19
+
20
+
21
+ def deployment_key(
22
+ family: str, suffix: str, run_name: str, config: DeploymentConfig
23
+ ) -> DeploymentKey:
24
+ return DeploymentKey(family, suffix, run_name, _config_fingerprint(config))
25
+
26
+
27
+ def _config_fingerprint(config: DeploymentConfig) -> str:
28
+ for name, value in config.items():
29
+ if not isinstance(name, str) or not isinstance(value, str):
30
+ raise ValueError(
31
+ f"Deployment config must map strings to strings: {name!r}: {value!r}"
32
+ )
33
+ if not config:
34
+ return ""
35
+ canonical_config = json.dumps(config, sort_keys=True)
36
+ return hashlib.sha256(canonical_config.encode()).hexdigest()[
37
+ :_CONFIG_FINGERPRINT_LENGTH
38
+ ]
@@ -0,0 +1,46 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ from cortexgrid import state
6
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
7
+
8
+
9
+ DeploymentRecord = dict[str, Any]
10
+
11
+
12
+ def get_deployment_record(key: DeploymentKey) -> DeploymentRecord | None:
13
+ return state.get(*_path(key), params=_params(key))
14
+
15
+
16
+ def put_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> None:
17
+ state.put(*_path(key), body=body, params=_params(key))
18
+
19
+
20
+ def patch_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> bool:
21
+ return state.patch(*_path(key), body=body, params=_params(key))
22
+
23
+
24
+ def delete_deployment_record(key: DeploymentKey) -> None:
25
+ state.delete(*_path(key), params=_params(key))
26
+
27
+
28
+ def list_deployment_records() -> list[DeploymentRecord]:
29
+ return state.get("deployments")
30
+
31
+
32
+ def key_of_record(record: DeploymentRecord) -> DeploymentKey:
33
+ return DeploymentKey(
34
+ record["family"],
35
+ record["suffix"],
36
+ record["run_name"],
37
+ record["config_fingerprint"],
38
+ )
39
+
40
+
41
+ def _path(key: DeploymentKey) -> tuple[str, ...]:
42
+ return ("deployments", key.family, key.suffix, key.run_name)
43
+
44
+
45
+ def _params(key: DeploymentKey) -> dict[str, str]:
46
+ return {"config_fingerprint": key.config_fingerprint}
@@ -0,0 +1,338 @@
1
+ from __future__ import annotations
2
+
3
+ import time
4
+ from typing import Any
5
+
6
+ from ray.serve.schema import ApplicationStatus
7
+
8
+ from cortexgrid.infra import get_ray_serve_uri
9
+ from cortexgrid.model_serving.application_spec import (
10
+ app_name,
11
+ build_application_spec,
12
+ bundle_fingerprint_in_spec,
13
+ replica_count_in_spec,
14
+ route_prefix,
15
+ )
16
+ from cortexgrid.model_serving.deployment_key import (
17
+ DeploymentConfig,
18
+ DeploymentKey,
19
+ deployment_key,
20
+ )
21
+ from cortexgrid.model_serving.deployment_records import (
22
+ DeploymentRecord,
23
+ delete_deployment_record,
24
+ get_deployment_record,
25
+ patch_deployment_record,
26
+ put_deployment_record,
27
+ )
28
+ from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
29
+ from cortexgrid.model_serving.registry_tags import load_deploy_metadata
30
+ from cortexgrid.model_serving.serve_bundle import BundleMetadata
31
+ from cortexgrid.model_serving.status import (
32
+ PHASE_PAUSED,
33
+ Deployment,
34
+ deployment_of_record,
35
+ observed,
36
+ replaced_bundle_fingerprint_until_rolled_out,
37
+ )
38
+ from cortexgrid.ray_util import get_serve_details, put_serve_applications
39
+
40
+
41
+ def _current_application_specs() -> list[dict[str, Any]]:
42
+ """Reconstruct the most recently PUT applications list from GET output.
43
+
44
+ Ray stores the originally-deployed `ServeApplicationSchema` for each app
45
+ under `applications[<name>].deployed_app_config`, which is what we need to
46
+ PUT-round-trip. Apps without a `deployed_app_config` (e.g. created via
47
+ `serve.run` in-cluster) are skipped: we can't faithfully reproduce them
48
+ from the read-only view.
49
+ """
50
+ details = get_serve_details()
51
+ specs: list[dict[str, Any]] = []
52
+ for app in details.get("applications", {}).values():
53
+ cfg = app.get("deployed_app_config")
54
+ if cfg is not None:
55
+ specs.append(cfg)
56
+ return specs
57
+
58
+
59
+ class ModelDeployFailed(RuntimeError):
60
+ """A model's Serve app cannot reach RUNNING: the controller reported
61
+ DEPLOY_FAILED, or no app exists for the model. Subclasses RuntimeError, which
62
+ `deploy_model(wait=True)` raised before this type existed."""
63
+
64
+
65
+ class ModelNotDeployed(LookupError):
66
+ pass
67
+
68
+
69
+ _SERVING_POLL_INTERVAL_S = 2.0
70
+
71
+
72
+ def _deadline(timeout: float | None) -> float | None:
73
+ return None if timeout is None else time.monotonic() + timeout
74
+
75
+
76
+ def _past(deadline: float | None) -> bool:
77
+ return deadline is not None and time.monotonic() >= deadline
78
+
79
+
80
+ def wait_for_model_serving(key: DeploymentKey, timeout: float | None = None) -> None:
81
+ """Block until the model's Serve app is RUNNING.
82
+
83
+ Raises ModelDeployFailed on DEPLOY_FAILED, carrying the controller's message,
84
+ and as soon as no app exists for the model: never deployed, undeployed, or
85
+ dropped by a concurrent `deploy_model` (each one GETs the applications list,
86
+ splices its own app in and PUTs the whole list back, so a later PUT can drop
87
+ an app an earlier one added). NOT_STARTED, DEPLOYING, UNHEALTHY and DELETING
88
+ are transient; a DELETING app ends up missing. Exceeding a finite `timeout`
89
+ raises TimeoutError; with `timeout=None` there is no deadline.
90
+ """
91
+ _wait_for_application_running(app_name(key), timeout, _deadline(timeout))
92
+
93
+
94
+ def _wait_for_application_running(
95
+ name: str, timeout: float | None, deadline: float | None
96
+ ) -> None:
97
+ """`wait_for_model_serving` against a deadline already running. `timeout`
98
+ only labels the TimeoutError."""
99
+ while True:
100
+ app = get_serve_details().get("applications", {}).get(name)
101
+ if app is None:
102
+ raise ModelDeployFailed(f"Serve app {name!r} does not exist")
103
+ status = str(app.get("status", "(missing)"))
104
+ message = str(app.get("message", ""))
105
+ if status == ApplicationStatus.RUNNING.value:
106
+ return
107
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
108
+ raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
109
+ if _past(deadline):
110
+ raise TimeoutError(
111
+ f"Serve app {name!r} did not reach RUNNING within {timeout}s "
112
+ f"(last status={status!r}, message={message!r})"
113
+ )
114
+ time.sleep(_SERVING_POLL_INTERVAL_S)
115
+
116
+
117
+ def _clear_failed_application(
118
+ key: DeploymentKey, timeout: float | None, deadline: float | None
119
+ ) -> None:
120
+ """Remove the model's DEPLOY_FAILED Serve app, and wait until it, or an app
121
+ already DELETING, is gone.
122
+
123
+ Ray resets a failed deployment only when a deploy arrives after the
124
+ deployment is marked for deletion or its version changes. Re-PUTting an
125
+ identical spec over a failed app, or PUTting it back before the controller's
126
+ next tick has processed an undeploy, leaves the failed deployment in place,
127
+ and the app reports DEPLOY_FAILED again without retrying."""
128
+ name = app_name(key)
129
+ app = get_serve_details().get("applications", {}).get(name)
130
+ if app is None:
131
+ return
132
+ status = app.get("status")
133
+ if status == ApplicationStatus.DEPLOY_FAILED.value:
134
+ undeploy_model(key)
135
+ elif status != ApplicationStatus.DELETING.value:
136
+ return
137
+ while name in get_serve_details().get("applications", {}):
138
+ if _past(deadline):
139
+ raise TimeoutError(
140
+ f"Serve app {name!r} was not removed within {timeout}s"
141
+ )
142
+ time.sleep(_SERVING_POLL_INTERVAL_S)
143
+
144
+
145
+ def _spec_already_deployed(spec: dict[str, Any]) -> bool:
146
+ """True when this exact spec is already the app's target, so PUTting it
147
+ again would only disturb the controller.
148
+
149
+ Re-PUTting is not a no-op. While an app's build task is in flight its target
150
+ code version is unset, so Ray cancels that build and starts a new one no
151
+ matter what the config says - and since every PUT re-sends the whole
152
+ applications list, it restarts the in-flight builds of the other apps too.
153
+ Our build task downloads the model bundle and may create a pip virtualenv,
154
+ so a caller redeploying faster than that could keep it from ever finishing.
155
+
156
+ A DEPLOY_FAILED or DELETING app never reaches here: `_clear_failed_application`
157
+ has already removed it, and an identical PUT over a failed app is exactly the
158
+ no-op that leaves it failed.
159
+ """
160
+ app = get_serve_details().get("applications", {}).get(spec["name"])
161
+ return app is not None and app.get("deployed_app_config") == spec
162
+
163
+
164
+ # Phases in which a deployment record stands for a live app: one a redeploy
165
+ # with the same spec may leave alone. A failed app has to be cleared and
166
+ # re-PUT, a deleting one waited out, and a missing one PUT again.
167
+ _LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy", PHASE_PAUSED)
168
+
169
+
170
+ def _record_is_current(
171
+ record: DeploymentRecord,
172
+ key: DeploymentKey,
173
+ meta: BundleMetadata,
174
+ requirements: ModelRequirements,
175
+ num_replicas: int,
176
+ ) -> bool:
177
+ """True when the deployment record shows the model live with exactly the
178
+ spec this deploy would PUT.
179
+
180
+ The spec is rebuilt against the GPU tiers the record was deployed with,
181
+ so the check asks neither Ray's state API nor the Serve controller: a
182
+ redeploy that changes nothing costs one lookup. A tier added to or gone
183
+ from the cluster since reaches the spec on the next deploy that changes
184
+ anything else; until then the placement fallback keeps larger new GPUs
185
+ usable (see `_placement_preferences`)."""
186
+ if record["phase"] not in _LIVE_PHASES:
187
+ return False
188
+ spec = build_application_spec(key, meta, requirements, num_replicas, record["tiers"])
189
+ return spec == record["spec"]
190
+
191
+
192
+ def _bundle_fingerprint_replaced_by(
193
+ record: DeploymentRecord | None, meta: BundleMetadata
194
+ ) -> str:
195
+ if record is None or record["phase"] not in _LIVE_PHASES:
196
+ return ""
197
+ rollout_origin = record["replaced_bundle_fingerprint"] or bundle_fingerprint_in_spec(
198
+ record["spec"]
199
+ )
200
+ return "" if rollout_origin == meta.fingerprint else rollout_origin
201
+
202
+
203
+ def _observation_of_application(
204
+ name: str, replaced_bundle_fingerprint: str
205
+ ) -> DeploymentRecord:
206
+ observation = observed(get_serve_details().get("applications", {}).get(name, {}))
207
+ observation["replaced_bundle_fingerprint"] = (
208
+ replaced_bundle_fingerprint_until_rolled_out(
209
+ replaced_bundle_fingerprint, observation["phase"]
210
+ )
211
+ )
212
+ return observation
213
+
214
+
215
+ def deploy_model(
216
+ family: str,
217
+ suffix: str,
218
+ run_name: str,
219
+ num_replicas: int = 1,
220
+ wait: bool = False,
221
+ timeout: float | None = 300.0,
222
+ config: DeploymentConfig | None = None,
223
+ ) -> Deployment:
224
+ """Schedule a Ray Serve app for a previously-saved model and return a
225
+ handle carrying its base URL. The caller (e.g. model-gateway) builds
226
+ whatever client the app's routes need - streaming, long timeouts, custom
227
+ request schemas - against that URL; cortexgrid imposes no traffic contract.
228
+
229
+ The serve-app class is pulled from the registry entry's tags `save_model`
230
+ wrote at save time; the caller does not need to hold the class object.
231
+ Each of the `num_replicas` replicas requests the model's `ModelRequirements`
232
+ from Ray, so it is placed only on a node that has them free.
233
+
234
+ A DEPLOY_FAILED app left by an earlier attempt is undeployed first, and it,
235
+ or an app still DELETING, is waited out before the new spec is PUT, so the
236
+ deploy starts afresh instead of Ray reusing the failed deployment.
237
+
238
+ Re-deploying a model that is already live with exactly this spec skips the
239
+ PUT rather than restating it: see `_spec_already_deployed` for what a
240
+ redundant PUT costs. When the model's deployment record already shows it
241
+ live with this spec, the call returns from the record without asking Ray
242
+ at all (see `_record_is_current`). The call still reports the app's phase,
243
+ and with `wait=True` still blocks until it is RUNNING.
244
+
245
+ `config` holds this deployment's own settings. A model gets one deployment
246
+ per distinct config, told apart by the `DeploymentKey` on the returned
247
+ `Deployment`; its replicas read the config, laid over the model's own, with
248
+ `model_config`.
249
+
250
+ Records the deployment - its config, the spec it PUT, where the app is
251
+ served, and its phase - with the jobs control plane before waiting.
252
+
253
+ With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
254
+ controller reports the app RUNNING. `timeout` (default 300) caps the whole
255
+ call, clearing a failed app included; exceeding it raises TimeoutError, and
256
+ DEPLOY_FAILED raises ModelDeployFailed. With `timeout=None` there is no cap.
257
+ Tradeoff: an app that never reaches a terminal state (e.g. GPU-starved,
258
+ stuck in DEPLOYING) will hang forever.
259
+ """
260
+ deadline = _deadline(timeout)
261
+ deployment_config = config or {}
262
+ key = deployment_key(family, suffix, run_name, deployment_config)
263
+ meta, requirements = load_deploy_metadata(family, suffix, run_name)
264
+ record = get_deployment_record(key)
265
+ if record is not None and _record_is_current(
266
+ record, key, meta, requirements, num_replicas
267
+ ):
268
+ if wait:
269
+ _wait_for_application_running(record["spec"]["name"], timeout, deadline)
270
+ record = {
271
+ **record,
272
+ "phase": observed(
273
+ get_serve_details().get("applications", {}).get(record["spec"]["name"])
274
+ )["phase"],
275
+ }
276
+ return deployment_of_record(record)
277
+ # Read afresh on every deploy that reaches Ray: the tiers are what the
278
+ # model is placed against, so a GPU joining or leaving the cluster has to
279
+ # change the spec (and therefore re-PUT it).
280
+ replaced_bundle_fingerprint = _bundle_fingerprint_replaced_by(record, meta)
281
+ tiers = vram_tiers()
282
+ spec = build_application_spec(key, meta, requirements, num_replicas, tiers)
283
+ _clear_failed_application(key, timeout, deadline)
284
+ if not _spec_already_deployed(spec):
285
+ existing = [
286
+ a for a in _current_application_specs() if a["name"] != spec["name"]
287
+ ]
288
+ # The controller registers the app, sets it DEPLOYING and stamps
289
+ # last_deployed_time_s before the PUT returns, so the wait below neither
290
+ # misses the app nor reads a status left by an earlier deploy.
291
+ put_serve_applications([*existing, spec])
292
+ url = f"{get_ray_serve_uri()}{route_prefix(key)}"
293
+ observation = _observation_of_application(spec["name"], replaced_bundle_fingerprint)
294
+ put_deployment_record(
295
+ key,
296
+ {
297
+ "config": deployment_config,
298
+ "spec": spec,
299
+ "tiers": tiers,
300
+ "url": url,
301
+ **observation,
302
+ },
303
+ )
304
+ if wait:
305
+ _wait_for_application_running(spec["name"], timeout, deadline)
306
+ observation = _observation_of_application(
307
+ spec["name"], replaced_bundle_fingerprint
308
+ )
309
+ patch_deployment_record(key, observation)
310
+ return Deployment(
311
+ key=key,
312
+ config=deployment_config,
313
+ url=url,
314
+ phase=observation["phase"],
315
+ bundle_fingerprint=meta.fingerprint,
316
+ replaced_bundle_fingerprint=observation["replaced_bundle_fingerprint"],
317
+ )
318
+
319
+
320
+ def redeploy_model(key: DeploymentKey) -> Deployment:
321
+ record = get_deployment_record(key)
322
+ if record is None:
323
+ raise ModelNotDeployed(f"{key} is not deployed")
324
+ return deploy_model(
325
+ key.family,
326
+ key.suffix,
327
+ key.run_name,
328
+ num_replicas=replica_count_in_spec(record["spec"]),
329
+ config=record["config"],
330
+ )
331
+
332
+
333
+ def undeploy_model(key: DeploymentKey) -> None:
334
+ """Tear down the Ray Serve app of this deployment and drop its record."""
335
+ name = app_name(key)
336
+ remaining = [a for a in _current_application_specs() if a["name"] != name]
337
+ put_serve_applications(remaining)
338
+ delete_deployment_record(key)