cortexgrid 0.3.11__tar.gz → 0.3.12__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/PKG-INFO +7 -5
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/__init__.py +8 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_serve_entry.py +14 -5
- cortexgrid-0.3.12/cortexgrid/model_serving/__init__.py +73 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/application_spec.py +70 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/deployment_key.py +38 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/deployment_records.py +46 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/lifecycle.py +338 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/placement.py +151 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/registry_tags.py +97 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/serve_bundle.py +143 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/status.py +277 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/model_storage.py +18 -10
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/state.py +8 -4
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/docs/cortexgrid/README.md +6 -4
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/pyproject.toml +8 -1
- cortexgrid-0.3.11/cortexgrid/model_serving.py +0 -953
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/.gitignore +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/LICENSE +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_model_scheduler.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.11 → cortexgrid-0.3.12}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.12
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -215,8 +215,10 @@ app = FastAPI()
|
|
|
215
215
|
|
|
216
216
|
@serve.ingress(app)
|
|
217
217
|
class MyServeApp:
|
|
218
|
-
def __init__(self,
|
|
219
|
-
self._weights_dir = cortexgrid.load_model(
|
|
218
|
+
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
219
|
+
self._weights_dir = cortexgrid.load_model(
|
|
220
|
+
deployment.family, deployment.suffix, deployment.run_name
|
|
221
|
+
)
|
|
220
222
|
|
|
221
223
|
@app.post("/complete")
|
|
222
224
|
async def complete(self, body: dict): ...
|
|
@@ -232,11 +234,11 @@ print(deployed.url)
|
|
|
232
234
|
|
|
233
235
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
234
236
|
|
|
235
|
-
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(
|
|
237
|
+
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(deployment)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. How one deployment serves the model - thinking on or off, say - is that deployment's own: `deploy_model(..., config={"thinking": "false"})` deploys the model once per distinct config, each deployment named by the `DeploymentKey` on the returned `Deployment`, and `model_config` lays the deployment's config over the model's. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
236
238
|
|
|
237
239
|
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
|
|
238
240
|
|
|
239
|
-
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(
|
|
241
|
+
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(deployment.key, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
240
242
|
|
|
241
243
|
### API reference
|
|
242
244
|
|
|
@@ -101,12 +101,16 @@ from cortexgrid.model_storage import register_model as _register_model_storage
|
|
|
101
101
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
102
102
|
from cortexgrid.model_serving import (
|
|
103
103
|
Deployment,
|
|
104
|
+
DeploymentConfig,
|
|
105
|
+
DeploymentKey,
|
|
104
106
|
ModelDeployFailed,
|
|
107
|
+
ModelNotDeployed,
|
|
105
108
|
ModelRequirements,
|
|
106
109
|
ServingStatus,
|
|
107
110
|
deploy_model,
|
|
108
111
|
list_deployed_models,
|
|
109
112
|
model_serving_status,
|
|
113
|
+
redeploy_model,
|
|
110
114
|
undeploy_model,
|
|
111
115
|
wait_for_model_serving,
|
|
112
116
|
)
|
|
@@ -308,9 +312,13 @@ __all__ = [
|
|
|
308
312
|
"delete_model",
|
|
309
313
|
# Model serving
|
|
310
314
|
"Deployment",
|
|
315
|
+
"DeploymentConfig",
|
|
316
|
+
"DeploymentKey",
|
|
311
317
|
"ModelDeployFailed",
|
|
318
|
+
"ModelNotDeployed",
|
|
312
319
|
"ServingStatus",
|
|
313
320
|
"deploy_model",
|
|
321
|
+
"redeploy_model",
|
|
314
322
|
"wait_for_model_serving",
|
|
315
323
|
"model_serving_status",
|
|
316
324
|
"undeploy_model",
|
|
@@ -6,13 +6,14 @@ On the cluster replica, `build` imports the serve-app class bundled at
|
|
|
6
6
|
ingress with the app it was marked with by `cortexgrid.serve.ingress` (again on
|
|
7
7
|
each replica, see `_IngressOnReplica`), wraps it as a Ray Serve deployment
|
|
8
8
|
with the replica count and Ray resource requests `deploy_model` derived from
|
|
9
|
-
the model's requirements, and binds it with
|
|
10
|
-
|
|
9
|
+
the model's requirements, and binds it with its deployment's
|
|
10
|
+
`cortexgrid.DeploymentKey`.
|
|
11
11
|
|
|
12
12
|
The serve-app owns everything about traffic: its own routes, request schemas,
|
|
13
13
|
streaming, and timeouts. cortexgrid does not interpose a request/response
|
|
14
|
-
contract - it only schedules the app and hands it the
|
|
15
|
-
|
|
14
|
+
contract - it only schedules the app and hands it the key it needs to fetch
|
|
15
|
+
its own weights via `cortexgrid.load_model` and its settings via
|
|
16
|
+
`cortexgrid.model_config`.
|
|
16
17
|
|
|
17
18
|
The serve-app declares no resources: the hardware a replica needs belongs to
|
|
18
19
|
the model and is stored in the registry (`cortexgrid.ModelRequirements`), and
|
|
@@ -28,6 +29,7 @@ from ray import serve
|
|
|
28
29
|
from ray.serve.deployment import Application
|
|
29
30
|
|
|
30
31
|
from cortexgrid._model_scheduler import model_autoscaling_config
|
|
32
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
31
33
|
from cortexgrid.serve import ingress_app
|
|
32
34
|
|
|
33
35
|
|
|
@@ -80,4 +82,11 @@ def build(args: dict[str, Any]) -> Application:
|
|
|
80
82
|
# had on Ray 2.9.
|
|
81
83
|
max_ongoing_requests=_MAX_ONGOING_REQUESTS,
|
|
82
84
|
ray_actor_options=args.get("ray_actor_options", {}),
|
|
83
|
-
).bind(
|
|
85
|
+
).bind(
|
|
86
|
+
DeploymentKey(
|
|
87
|
+
args["family"],
|
|
88
|
+
args["suffix"],
|
|
89
|
+
args["run_name"],
|
|
90
|
+
args.get("config_fingerprint", ""),
|
|
91
|
+
)
|
|
92
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Cortexgrid wrappers around Ray Serve.
|
|
2
|
+
|
|
3
|
+
Caller stays HTTP-only: deploy/undeploy/list talk to the Ray dashboard's
|
|
4
|
+
declarative `/api/serve/applications/` endpoint via [cortexgrid.ray_util],
|
|
5
|
+
never `ray.init`. The deployment class is bundled at `save_model` time, zipped,
|
|
6
|
+
uploaded to MinIO under
|
|
7
|
+
`serve-bundles/<run_name>/<family>__<suffix>/<fingerprint>.zip`, and
|
|
8
|
+
referenced via `runtime_env.working_dir` so Ray workers fetch it from there.
|
|
9
|
+
The bundle URL, class import path, pip list, and fingerprint are persisted as
|
|
10
|
+
tags on the model's registry entry so `deploy_model` can find them later
|
|
11
|
+
without the caller holding the class object. So are the model's
|
|
12
|
+
`ModelRequirements`, which `deploy_model` turns into the replica's Ray resource
|
|
13
|
+
requests.
|
|
14
|
+
|
|
15
|
+
Every deployment `deploy_model` puts on Ray Serve gets a record with the jobs
|
|
16
|
+
control plane, keyed by its `DeploymentKey` - the model's (family, suffix,
|
|
17
|
+
run_name) plus a fingerprint of the config it was deployed with: that config,
|
|
18
|
+
the spec it PUT, plus the phase, message and replica
|
|
19
|
+
placements the control plane last observed (`observe_deployments`, run on
|
|
20
|
+
every poll cycle). Listings and status reads come from those records; waits
|
|
21
|
+
ask the Serve controller directly.
|
|
22
|
+
|
|
23
|
+
Naming: the Ray Serve application is named "<family>__<suffix>__<run_name>",
|
|
24
|
+
followed by "__<config_fingerprint>" for a deployment given a config.
|
|
25
|
+
This relies on family/suffix/run_name not containing the literal "__".
|
|
26
|
+
|
|
27
|
+
See [docs/cortexgrid/model-serving.md](../../docs/cortexgrid/model-serving.md)
|
|
28
|
+
for the end-to-end design.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from cortexgrid.model_serving.application_spec import app_name
|
|
32
|
+
from cortexgrid.model_serving.deployment_key import (
|
|
33
|
+
DeploymentConfig,
|
|
34
|
+
DeploymentKey,
|
|
35
|
+
deployment_key,
|
|
36
|
+
)
|
|
37
|
+
from cortexgrid.model_serving.lifecycle import (
|
|
38
|
+
ModelDeployFailed,
|
|
39
|
+
ModelNotDeployed,
|
|
40
|
+
deploy_model,
|
|
41
|
+
redeploy_model,
|
|
42
|
+
undeploy_model,
|
|
43
|
+
wait_for_model_serving,
|
|
44
|
+
)
|
|
45
|
+
from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
|
|
46
|
+
from cortexgrid.model_serving.registry_tags import (
|
|
47
|
+
bundle_fingerprint_from_tags,
|
|
48
|
+
has_requirement_tags,
|
|
49
|
+
metadata_from_tags,
|
|
50
|
+
metadata_to_tags,
|
|
51
|
+
requirements_from_tags,
|
|
52
|
+
requirements_to_tags,
|
|
53
|
+
)
|
|
54
|
+
from cortexgrid.model_serving.serve_bundle import (
|
|
55
|
+
BundleMetadata,
|
|
56
|
+
ServeBundle,
|
|
57
|
+
build_bundle,
|
|
58
|
+
bundle_class,
|
|
59
|
+
bundle_fingerprint_from_url,
|
|
60
|
+
upload_bundle,
|
|
61
|
+
)
|
|
62
|
+
from cortexgrid.model_serving.status import (
|
|
63
|
+
Deployment,
|
|
64
|
+
ReplicaPlacement,
|
|
65
|
+
ServingMessage,
|
|
66
|
+
ServingStatus,
|
|
67
|
+
deployment_config,
|
|
68
|
+
list_deployed_models,
|
|
69
|
+
model_replica_placements,
|
|
70
|
+
model_serving_messages,
|
|
71
|
+
model_serving_status,
|
|
72
|
+
observe_deployments,
|
|
73
|
+
)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
6
|
+
from cortexgrid.model_serving.placement import ModelRequirements, ray_actor_options
|
|
7
|
+
from cortexgrid.model_serving.serve_bundle import (
|
|
8
|
+
BundleMetadata,
|
|
9
|
+
bundle_fingerprint_from_url,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def app_name(key: DeploymentKey) -> str:
|
|
14
|
+
return "__".join(_name_segments(key))
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def route_prefix(key: DeploymentKey) -> str:
|
|
18
|
+
return "/r/" + "/".join(_name_segments(key))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _name_segments(key: DeploymentKey) -> list[str]:
|
|
22
|
+
model_segments = [key.family, key.suffix, key.run_name]
|
|
23
|
+
if not key.config_fingerprint:
|
|
24
|
+
return model_segments
|
|
25
|
+
return [*model_segments, key.config_fingerprint]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def build_application_spec(
|
|
29
|
+
key: DeploymentKey,
|
|
30
|
+
meta: BundleMetadata,
|
|
31
|
+
requirements: ModelRequirements,
|
|
32
|
+
num_replicas: int,
|
|
33
|
+
tiers: list[int],
|
|
34
|
+
) -> dict[str, Any]:
|
|
35
|
+
"""Assemble a Ray Serve application schema from pre-bundled metadata, the
|
|
36
|
+
model's requirements, and the cluster's GPU size classes."""
|
|
37
|
+
# working_dir carries the serve-app's own source; Ray pip-installs the
|
|
38
|
+
# third-party distributions the image lacks into a per-node cached
|
|
39
|
+
# virtualenv layered on the image. No pip key when there are none, so Ray
|
|
40
|
+
# builds no virtualenv.
|
|
41
|
+
runtime_env: dict[str, Any] = {"working_dir": meta.bundle_url}
|
|
42
|
+
if meta.pip_requirements:
|
|
43
|
+
runtime_env["pip"] = meta.pip_requirements
|
|
44
|
+
return {
|
|
45
|
+
"name": app_name(key),
|
|
46
|
+
"route_prefix": route_prefix(key),
|
|
47
|
+
# Ray Serve REST requires import_path to point at an Application builder
|
|
48
|
+
# (callable returning a bound node) or an already-bound node. A bare
|
|
49
|
+
# Deployment class is rejected, so cortexgrid.deploy_model goes through
|
|
50
|
+
# a generic builder that re-imports the user's class and binds it.
|
|
51
|
+
"import_path": "cortexgrid._serve_entry:build",
|
|
52
|
+
"args": {
|
|
53
|
+
"class_import_path": meta.class_import_path,
|
|
54
|
+
"family": key.family,
|
|
55
|
+
"suffix": key.suffix,
|
|
56
|
+
"run_name": key.run_name,
|
|
57
|
+
"config_fingerprint": key.config_fingerprint,
|
|
58
|
+
"num_replicas": num_replicas,
|
|
59
|
+
"ray_actor_options": ray_actor_options(requirements, tiers),
|
|
60
|
+
},
|
|
61
|
+
"runtime_env": runtime_env,
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def bundle_fingerprint_in_spec(spec: dict[str, Any]) -> str:
|
|
66
|
+
return bundle_fingerprint_from_url(spec["runtime_env"]["working_dir"])
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def replica_count_in_spec(spec: dict[str, Any]) -> int:
|
|
70
|
+
return spec["args"]["num_replicas"]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
DeploymentConfig = dict[str, str]
|
|
9
|
+
|
|
10
|
+
_CONFIG_FINGERPRINT_LENGTH = 12
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class DeploymentKey:
|
|
15
|
+
family: str
|
|
16
|
+
suffix: str
|
|
17
|
+
run_name: str
|
|
18
|
+
config_fingerprint: str = ""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def deployment_key(
|
|
22
|
+
family: str, suffix: str, run_name: str, config: DeploymentConfig
|
|
23
|
+
) -> DeploymentKey:
|
|
24
|
+
return DeploymentKey(family, suffix, run_name, _config_fingerprint(config))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _config_fingerprint(config: DeploymentConfig) -> str:
|
|
28
|
+
for name, value in config.items():
|
|
29
|
+
if not isinstance(name, str) or not isinstance(value, str):
|
|
30
|
+
raise ValueError(
|
|
31
|
+
f"Deployment config must map strings to strings: {name!r}: {value!r}"
|
|
32
|
+
)
|
|
33
|
+
if not config:
|
|
34
|
+
return ""
|
|
35
|
+
canonical_config = json.dumps(config, sort_keys=True)
|
|
36
|
+
return hashlib.sha256(canonical_config.encode()).hexdigest()[
|
|
37
|
+
:_CONFIG_FINGERPRINT_LENGTH
|
|
38
|
+
]
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from cortexgrid import state
|
|
6
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
DeploymentRecord = dict[str, Any]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def get_deployment_record(key: DeploymentKey) -> DeploymentRecord | None:
|
|
13
|
+
return state.get(*_path(key), params=_params(key))
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def put_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> None:
|
|
17
|
+
state.put(*_path(key), body=body, params=_params(key))
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def patch_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> bool:
|
|
21
|
+
return state.patch(*_path(key), body=body, params=_params(key))
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def delete_deployment_record(key: DeploymentKey) -> None:
|
|
25
|
+
state.delete(*_path(key), params=_params(key))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def list_deployment_records() -> list[DeploymentRecord]:
|
|
29
|
+
return state.get("deployments")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def key_of_record(record: DeploymentRecord) -> DeploymentKey:
|
|
33
|
+
return DeploymentKey(
|
|
34
|
+
record["family"],
|
|
35
|
+
record["suffix"],
|
|
36
|
+
record["run_name"],
|
|
37
|
+
record["config_fingerprint"],
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _path(key: DeploymentKey) -> tuple[str, ...]:
|
|
42
|
+
return ("deployments", key.family, key.suffix, key.run_name)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _params(key: DeploymentKey) -> dict[str, str]:
|
|
46
|
+
return {"config_fingerprint": key.config_fingerprint}
|
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from ray.serve.schema import ApplicationStatus
|
|
7
|
+
|
|
8
|
+
from cortexgrid.infra import get_ray_serve_uri
|
|
9
|
+
from cortexgrid.model_serving.application_spec import (
|
|
10
|
+
app_name,
|
|
11
|
+
build_application_spec,
|
|
12
|
+
bundle_fingerprint_in_spec,
|
|
13
|
+
replica_count_in_spec,
|
|
14
|
+
route_prefix,
|
|
15
|
+
)
|
|
16
|
+
from cortexgrid.model_serving.deployment_key import (
|
|
17
|
+
DeploymentConfig,
|
|
18
|
+
DeploymentKey,
|
|
19
|
+
deployment_key,
|
|
20
|
+
)
|
|
21
|
+
from cortexgrid.model_serving.deployment_records import (
|
|
22
|
+
DeploymentRecord,
|
|
23
|
+
delete_deployment_record,
|
|
24
|
+
get_deployment_record,
|
|
25
|
+
patch_deployment_record,
|
|
26
|
+
put_deployment_record,
|
|
27
|
+
)
|
|
28
|
+
from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
|
|
29
|
+
from cortexgrid.model_serving.registry_tags import load_deploy_metadata
|
|
30
|
+
from cortexgrid.model_serving.serve_bundle import BundleMetadata
|
|
31
|
+
from cortexgrid.model_serving.status import (
|
|
32
|
+
PHASE_PAUSED,
|
|
33
|
+
Deployment,
|
|
34
|
+
deployment_of_record,
|
|
35
|
+
observed,
|
|
36
|
+
replaced_bundle_fingerprint_until_rolled_out,
|
|
37
|
+
)
|
|
38
|
+
from cortexgrid.ray_util import get_serve_details, put_serve_applications
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _current_application_specs() -> list[dict[str, Any]]:
|
|
42
|
+
"""Reconstruct the most recently PUT applications list from GET output.
|
|
43
|
+
|
|
44
|
+
Ray stores the originally-deployed `ServeApplicationSchema` for each app
|
|
45
|
+
under `applications[<name>].deployed_app_config`, which is what we need to
|
|
46
|
+
PUT-round-trip. Apps without a `deployed_app_config` (e.g. created via
|
|
47
|
+
`serve.run` in-cluster) are skipped: we can't faithfully reproduce them
|
|
48
|
+
from the read-only view.
|
|
49
|
+
"""
|
|
50
|
+
details = get_serve_details()
|
|
51
|
+
specs: list[dict[str, Any]] = []
|
|
52
|
+
for app in details.get("applications", {}).values():
|
|
53
|
+
cfg = app.get("deployed_app_config")
|
|
54
|
+
if cfg is not None:
|
|
55
|
+
specs.append(cfg)
|
|
56
|
+
return specs
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class ModelDeployFailed(RuntimeError):
|
|
60
|
+
"""A model's Serve app cannot reach RUNNING: the controller reported
|
|
61
|
+
DEPLOY_FAILED, or no app exists for the model. Subclasses RuntimeError, which
|
|
62
|
+
`deploy_model(wait=True)` raised before this type existed."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class ModelNotDeployed(LookupError):
|
|
66
|
+
pass
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
_SERVING_POLL_INTERVAL_S = 2.0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _deadline(timeout: float | None) -> float | None:
|
|
73
|
+
return None if timeout is None else time.monotonic() + timeout
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _past(deadline: float | None) -> bool:
|
|
77
|
+
return deadline is not None and time.monotonic() >= deadline
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def wait_for_model_serving(key: DeploymentKey, timeout: float | None = None) -> None:
|
|
81
|
+
"""Block until the model's Serve app is RUNNING.
|
|
82
|
+
|
|
83
|
+
Raises ModelDeployFailed on DEPLOY_FAILED, carrying the controller's message,
|
|
84
|
+
and as soon as no app exists for the model: never deployed, undeployed, or
|
|
85
|
+
dropped by a concurrent `deploy_model` (each one GETs the applications list,
|
|
86
|
+
splices its own app in and PUTs the whole list back, so a later PUT can drop
|
|
87
|
+
an app an earlier one added). NOT_STARTED, DEPLOYING, UNHEALTHY and DELETING
|
|
88
|
+
are transient; a DELETING app ends up missing. Exceeding a finite `timeout`
|
|
89
|
+
raises TimeoutError; with `timeout=None` there is no deadline.
|
|
90
|
+
"""
|
|
91
|
+
_wait_for_application_running(app_name(key), timeout, _deadline(timeout))
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _wait_for_application_running(
|
|
95
|
+
name: str, timeout: float | None, deadline: float | None
|
|
96
|
+
) -> None:
|
|
97
|
+
"""`wait_for_model_serving` against a deadline already running. `timeout`
|
|
98
|
+
only labels the TimeoutError."""
|
|
99
|
+
while True:
|
|
100
|
+
app = get_serve_details().get("applications", {}).get(name)
|
|
101
|
+
if app is None:
|
|
102
|
+
raise ModelDeployFailed(f"Serve app {name!r} does not exist")
|
|
103
|
+
status = str(app.get("status", "(missing)"))
|
|
104
|
+
message = str(app.get("message", ""))
|
|
105
|
+
if status == ApplicationStatus.RUNNING.value:
|
|
106
|
+
return
|
|
107
|
+
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
108
|
+
raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
|
|
109
|
+
if _past(deadline):
|
|
110
|
+
raise TimeoutError(
|
|
111
|
+
f"Serve app {name!r} did not reach RUNNING within {timeout}s "
|
|
112
|
+
f"(last status={status!r}, message={message!r})"
|
|
113
|
+
)
|
|
114
|
+
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _clear_failed_application(
|
|
118
|
+
key: DeploymentKey, timeout: float | None, deadline: float | None
|
|
119
|
+
) -> None:
|
|
120
|
+
"""Remove the model's DEPLOY_FAILED Serve app, and wait until it, or an app
|
|
121
|
+
already DELETING, is gone.
|
|
122
|
+
|
|
123
|
+
Ray resets a failed deployment only when a deploy arrives after the
|
|
124
|
+
deployment is marked for deletion or its version changes. Re-PUTting an
|
|
125
|
+
identical spec over a failed app, or PUTting it back before the controller's
|
|
126
|
+
next tick has processed an undeploy, leaves the failed deployment in place,
|
|
127
|
+
and the app reports DEPLOY_FAILED again without retrying."""
|
|
128
|
+
name = app_name(key)
|
|
129
|
+
app = get_serve_details().get("applications", {}).get(name)
|
|
130
|
+
if app is None:
|
|
131
|
+
return
|
|
132
|
+
status = app.get("status")
|
|
133
|
+
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
134
|
+
undeploy_model(key)
|
|
135
|
+
elif status != ApplicationStatus.DELETING.value:
|
|
136
|
+
return
|
|
137
|
+
while name in get_serve_details().get("applications", {}):
|
|
138
|
+
if _past(deadline):
|
|
139
|
+
raise TimeoutError(
|
|
140
|
+
f"Serve app {name!r} was not removed within {timeout}s"
|
|
141
|
+
)
|
|
142
|
+
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _spec_already_deployed(spec: dict[str, Any]) -> bool:
|
|
146
|
+
"""True when this exact spec is already the app's target, so PUTting it
|
|
147
|
+
again would only disturb the controller.
|
|
148
|
+
|
|
149
|
+
Re-PUTting is not a no-op. While an app's build task is in flight its target
|
|
150
|
+
code version is unset, so Ray cancels that build and starts a new one no
|
|
151
|
+
matter what the config says - and since every PUT re-sends the whole
|
|
152
|
+
applications list, it restarts the in-flight builds of the other apps too.
|
|
153
|
+
Our build task downloads the model bundle and may create a pip virtualenv,
|
|
154
|
+
so a caller redeploying faster than that could keep it from ever finishing.
|
|
155
|
+
|
|
156
|
+
A DEPLOY_FAILED or DELETING app never reaches here: `_clear_failed_application`
|
|
157
|
+
has already removed it, and an identical PUT over a failed app is exactly the
|
|
158
|
+
no-op that leaves it failed.
|
|
159
|
+
"""
|
|
160
|
+
app = get_serve_details().get("applications", {}).get(spec["name"])
|
|
161
|
+
return app is not None and app.get("deployed_app_config") == spec
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# Phases in which a deployment record stands for a live app: one a redeploy
|
|
165
|
+
# with the same spec may leave alone. A failed app has to be cleared and
|
|
166
|
+
# re-PUT, a deleting one waited out, and a missing one PUT again.
|
|
167
|
+
_LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy", PHASE_PAUSED)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _record_is_current(
|
|
171
|
+
record: DeploymentRecord,
|
|
172
|
+
key: DeploymentKey,
|
|
173
|
+
meta: BundleMetadata,
|
|
174
|
+
requirements: ModelRequirements,
|
|
175
|
+
num_replicas: int,
|
|
176
|
+
) -> bool:
|
|
177
|
+
"""True when the deployment record shows the model live with exactly the
|
|
178
|
+
spec this deploy would PUT.
|
|
179
|
+
|
|
180
|
+
The spec is rebuilt against the GPU tiers the record was deployed with,
|
|
181
|
+
so the check asks neither Ray's state API nor the Serve controller: a
|
|
182
|
+
redeploy that changes nothing costs one lookup. A tier added to or gone
|
|
183
|
+
from the cluster since reaches the spec on the next deploy that changes
|
|
184
|
+
anything else; until then the placement fallback keeps larger new GPUs
|
|
185
|
+
usable (see `_placement_preferences`)."""
|
|
186
|
+
if record["phase"] not in _LIVE_PHASES:
|
|
187
|
+
return False
|
|
188
|
+
spec = build_application_spec(key, meta, requirements, num_replicas, record["tiers"])
|
|
189
|
+
return spec == record["spec"]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _bundle_fingerprint_replaced_by(
|
|
193
|
+
record: DeploymentRecord | None, meta: BundleMetadata
|
|
194
|
+
) -> str:
|
|
195
|
+
if record is None or record["phase"] not in _LIVE_PHASES:
|
|
196
|
+
return ""
|
|
197
|
+
rollout_origin = record["replaced_bundle_fingerprint"] or bundle_fingerprint_in_spec(
|
|
198
|
+
record["spec"]
|
|
199
|
+
)
|
|
200
|
+
return "" if rollout_origin == meta.fingerprint else rollout_origin
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _observation_of_application(
|
|
204
|
+
name: str, replaced_bundle_fingerprint: str
|
|
205
|
+
) -> DeploymentRecord:
|
|
206
|
+
observation = observed(get_serve_details().get("applications", {}).get(name, {}))
|
|
207
|
+
observation["replaced_bundle_fingerprint"] = (
|
|
208
|
+
replaced_bundle_fingerprint_until_rolled_out(
|
|
209
|
+
replaced_bundle_fingerprint, observation["phase"]
|
|
210
|
+
)
|
|
211
|
+
)
|
|
212
|
+
return observation
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def deploy_model(
|
|
216
|
+
family: str,
|
|
217
|
+
suffix: str,
|
|
218
|
+
run_name: str,
|
|
219
|
+
num_replicas: int = 1,
|
|
220
|
+
wait: bool = False,
|
|
221
|
+
timeout: float | None = 300.0,
|
|
222
|
+
config: DeploymentConfig | None = None,
|
|
223
|
+
) -> Deployment:
|
|
224
|
+
"""Schedule a Ray Serve app for a previously-saved model and return a
|
|
225
|
+
handle carrying its base URL. The caller (e.g. model-gateway) builds
|
|
226
|
+
whatever client the app's routes need - streaming, long timeouts, custom
|
|
227
|
+
request schemas - against that URL; cortexgrid imposes no traffic contract.
|
|
228
|
+
|
|
229
|
+
The serve-app class is pulled from the registry entry's tags `save_model`
|
|
230
|
+
wrote at save time; the caller does not need to hold the class object.
|
|
231
|
+
Each of the `num_replicas` replicas requests the model's `ModelRequirements`
|
|
232
|
+
from Ray, so it is placed only on a node that has them free.
|
|
233
|
+
|
|
234
|
+
A DEPLOY_FAILED app left by an earlier attempt is undeployed first, and it,
|
|
235
|
+
or an app still DELETING, is waited out before the new spec is PUT, so the
|
|
236
|
+
deploy starts afresh instead of Ray reusing the failed deployment.
|
|
237
|
+
|
|
238
|
+
Re-deploying a model that is already live with exactly this spec skips the
|
|
239
|
+
PUT rather than restating it: see `_spec_already_deployed` for what a
|
|
240
|
+
redundant PUT costs. When the model's deployment record already shows it
|
|
241
|
+
live with this spec, the call returns from the record without asking Ray
|
|
242
|
+
at all (see `_record_is_current`). The call still reports the app's phase,
|
|
243
|
+
and with `wait=True` still blocks until it is RUNNING.
|
|
244
|
+
|
|
245
|
+
`config` holds this deployment's own settings. A model gets one deployment
|
|
246
|
+
per distinct config, told apart by the `DeploymentKey` on the returned
|
|
247
|
+
`Deployment`; its replicas read the config, laid over the model's own, with
|
|
248
|
+
`model_config`.
|
|
249
|
+
|
|
250
|
+
Records the deployment - its config, the spec it PUT, where the app is
|
|
251
|
+
served, and its phase - with the jobs control plane before waiting.
|
|
252
|
+
|
|
253
|
+
With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
|
|
254
|
+
controller reports the app RUNNING. `timeout` (default 300) caps the whole
|
|
255
|
+
call, clearing a failed app included; exceeding it raises TimeoutError, and
|
|
256
|
+
DEPLOY_FAILED raises ModelDeployFailed. With `timeout=None` there is no cap.
|
|
257
|
+
Tradeoff: an app that never reaches a terminal state (e.g. GPU-starved,
|
|
258
|
+
stuck in DEPLOYING) will hang forever.
|
|
259
|
+
"""
|
|
260
|
+
deadline = _deadline(timeout)
|
|
261
|
+
deployment_config = config or {}
|
|
262
|
+
key = deployment_key(family, suffix, run_name, deployment_config)
|
|
263
|
+
meta, requirements = load_deploy_metadata(family, suffix, run_name)
|
|
264
|
+
record = get_deployment_record(key)
|
|
265
|
+
if record is not None and _record_is_current(
|
|
266
|
+
record, key, meta, requirements, num_replicas
|
|
267
|
+
):
|
|
268
|
+
if wait:
|
|
269
|
+
_wait_for_application_running(record["spec"]["name"], timeout, deadline)
|
|
270
|
+
record = {
|
|
271
|
+
**record,
|
|
272
|
+
"phase": observed(
|
|
273
|
+
get_serve_details().get("applications", {}).get(record["spec"]["name"])
|
|
274
|
+
)["phase"],
|
|
275
|
+
}
|
|
276
|
+
return deployment_of_record(record)
|
|
277
|
+
# Read afresh on every deploy that reaches Ray: the tiers are what the
|
|
278
|
+
# model is placed against, so a GPU joining or leaving the cluster has to
|
|
279
|
+
# change the spec (and therefore re-PUT it).
|
|
280
|
+
replaced_bundle_fingerprint = _bundle_fingerprint_replaced_by(record, meta)
|
|
281
|
+
tiers = vram_tiers()
|
|
282
|
+
spec = build_application_spec(key, meta, requirements, num_replicas, tiers)
|
|
283
|
+
_clear_failed_application(key, timeout, deadline)
|
|
284
|
+
if not _spec_already_deployed(spec):
|
|
285
|
+
existing = [
|
|
286
|
+
a for a in _current_application_specs() if a["name"] != spec["name"]
|
|
287
|
+
]
|
|
288
|
+
# The controller registers the app, sets it DEPLOYING and stamps
|
|
289
|
+
# last_deployed_time_s before the PUT returns, so the wait below neither
|
|
290
|
+
# misses the app nor reads a status left by an earlier deploy.
|
|
291
|
+
put_serve_applications([*existing, spec])
|
|
292
|
+
url = f"{get_ray_serve_uri()}{route_prefix(key)}"
|
|
293
|
+
observation = _observation_of_application(spec["name"], replaced_bundle_fingerprint)
|
|
294
|
+
put_deployment_record(
|
|
295
|
+
key,
|
|
296
|
+
{
|
|
297
|
+
"config": deployment_config,
|
|
298
|
+
"spec": spec,
|
|
299
|
+
"tiers": tiers,
|
|
300
|
+
"url": url,
|
|
301
|
+
**observation,
|
|
302
|
+
},
|
|
303
|
+
)
|
|
304
|
+
if wait:
|
|
305
|
+
_wait_for_application_running(spec["name"], timeout, deadline)
|
|
306
|
+
observation = _observation_of_application(
|
|
307
|
+
spec["name"], replaced_bundle_fingerprint
|
|
308
|
+
)
|
|
309
|
+
patch_deployment_record(key, observation)
|
|
310
|
+
return Deployment(
|
|
311
|
+
key=key,
|
|
312
|
+
config=deployment_config,
|
|
313
|
+
url=url,
|
|
314
|
+
phase=observation["phase"],
|
|
315
|
+
bundle_fingerprint=meta.fingerprint,
|
|
316
|
+
replaced_bundle_fingerprint=observation["replaced_bundle_fingerprint"],
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def redeploy_model(key: DeploymentKey) -> Deployment:
|
|
321
|
+
record = get_deployment_record(key)
|
|
322
|
+
if record is None:
|
|
323
|
+
raise ModelNotDeployed(f"{key} is not deployed")
|
|
324
|
+
return deploy_model(
|
|
325
|
+
key.family,
|
|
326
|
+
key.suffix,
|
|
327
|
+
key.run_name,
|
|
328
|
+
num_replicas=replica_count_in_spec(record["spec"]),
|
|
329
|
+
config=record["config"],
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def undeploy_model(key: DeploymentKey) -> None:
|
|
334
|
+
"""Tear down the Ray Serve app of this deployment and drop its record."""
|
|
335
|
+
name = app_name(key)
|
|
336
|
+
remaining = [a for a in _current_application_specs() if a["name"] != name]
|
|
337
|
+
put_serve_applications(remaining)
|
|
338
|
+
delete_deployment_record(key)
|