cortexgrid 0.3.10__tar.gz → 0.3.12__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/PKG-INFO +7 -5
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/__init__.py +8 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/_model_scheduler.py +84 -89
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/_serve_entry.py +15 -8
- cortexgrid-0.3.12/cortexgrid/model_serving/__init__.py +73 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/application_spec.py +70 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/deployment_key.py +38 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/deployment_records.py +46 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/lifecycle.py +338 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/placement.py +151 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/registry_tags.py +97 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/serve_bundle.py +143 -0
- cortexgrid-0.3.12/cortexgrid/model_serving/status.py +277 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/model_storage.py +18 -10
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/state.py +8 -4
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/docs/cortexgrid/README.md +6 -4
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/pyproject.toml +8 -1
- cortexgrid-0.3.10/cortexgrid/model_serving.py +0 -953
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/.gitignore +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/LICENSE +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.12}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.12
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -215,8 +215,10 @@ app = FastAPI()
|
|
|
215
215
|
|
|
216
216
|
@serve.ingress(app)
|
|
217
217
|
class MyServeApp:
|
|
218
|
-
def __init__(self,
|
|
219
|
-
self._weights_dir = cortexgrid.load_model(
|
|
218
|
+
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
219
|
+
self._weights_dir = cortexgrid.load_model(
|
|
220
|
+
deployment.family, deployment.suffix, deployment.run_name
|
|
221
|
+
)
|
|
220
222
|
|
|
221
223
|
@app.post("/complete")
|
|
222
224
|
async def complete(self, body: dict): ...
|
|
@@ -232,11 +234,11 @@ print(deployed.url)
|
|
|
232
234
|
|
|
233
235
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
234
236
|
|
|
235
|
-
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(
|
|
237
|
+
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(deployment)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. How one deployment serves the model - thinking on or off, say - is that deployment's own: `deploy_model(..., config={"thinking": "false"})` deploys the model once per distinct config, each deployment named by the `DeploymentKey` on the returned `Deployment`, and `model_config` lays the deployment's config over the model's. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
236
238
|
|
|
237
239
|
`save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
|
|
238
240
|
|
|
239
|
-
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(
|
|
241
|
+
`save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(deployment.key, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
|
|
240
242
|
|
|
241
243
|
### API reference
|
|
242
244
|
|
|
@@ -101,12 +101,16 @@ from cortexgrid.model_storage import register_model as _register_model_storage
|
|
|
101
101
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
102
102
|
from cortexgrid.model_serving import (
|
|
103
103
|
Deployment,
|
|
104
|
+
DeploymentConfig,
|
|
105
|
+
DeploymentKey,
|
|
104
106
|
ModelDeployFailed,
|
|
107
|
+
ModelNotDeployed,
|
|
105
108
|
ModelRequirements,
|
|
106
109
|
ServingStatus,
|
|
107
110
|
deploy_model,
|
|
108
111
|
list_deployed_models,
|
|
109
112
|
model_serving_status,
|
|
113
|
+
redeploy_model,
|
|
110
114
|
undeploy_model,
|
|
111
115
|
wait_for_model_serving,
|
|
112
116
|
)
|
|
@@ -308,9 +312,13 @@ __all__ = [
|
|
|
308
312
|
"delete_model",
|
|
309
313
|
# Model serving
|
|
310
314
|
"Deployment",
|
|
315
|
+
"DeploymentConfig",
|
|
316
|
+
"DeploymentKey",
|
|
311
317
|
"ModelDeployFailed",
|
|
318
|
+
"ModelNotDeployed",
|
|
312
319
|
"ServingStatus",
|
|
313
320
|
"deploy_model",
|
|
321
|
+
"redeploy_model",
|
|
314
322
|
"wait_for_model_serving",
|
|
315
323
|
"model_serving_status",
|
|
316
324
|
"undeploy_model",
|
|
@@ -18,7 +18,8 @@ log = logging.getLogger(__name__)
|
|
|
18
18
|
_RayResources = dict[str, float]
|
|
19
19
|
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
_SCHEDULER_PROTOCOL_VERSION = 2
|
|
22
|
+
_SCHEDULER_ACTOR_NAME = f"cortexgrid-model-scheduler-v{_SCHEDULER_PROTOCOL_VERSION}"
|
|
22
23
|
_SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
|
|
23
24
|
|
|
24
25
|
_PAUSE_DECISION_INTERVAL_S = 1.0
|
|
@@ -27,13 +28,12 @@ _FORGET_MODELS_SILENT_FOR_S = 30.0
|
|
|
27
28
|
_REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
|
|
28
29
|
_REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
|
|
29
30
|
|
|
31
|
+
_SERVE_REPLICA_CLASS_PREFIX = "ServeReplica:"
|
|
30
32
|
_STATE_API_RESULT_LIMIT = 10_000
|
|
31
33
|
_RESOURCE_FLOAT_TOLERANCE = 1e-6
|
|
32
34
|
|
|
33
35
|
|
|
34
|
-
def model_autoscaling_config(
|
|
35
|
-
max_replicas: int, ray_actor_options: dict[str, Any]
|
|
36
|
-
) -> dict[str, Any]:
|
|
36
|
+
def model_autoscaling_config(max_replicas: int) -> dict[str, Any]:
|
|
37
37
|
return {
|
|
38
38
|
"min_replicas": 0,
|
|
39
39
|
"initial_replicas": max_replicas,
|
|
@@ -48,47 +48,22 @@ def model_autoscaling_config(
|
|
|
48
48
|
f"{ModelAutoscalingPolicy.__module__}:"
|
|
49
49
|
f"{ModelAutoscalingPolicy.__qualname__}"
|
|
50
50
|
),
|
|
51
|
-
"policy_kwargs": {
|
|
52
|
-
"replica_resources": _resources_requested_by_replica(
|
|
53
|
-
ray_actor_options
|
|
54
|
-
)
|
|
55
|
-
},
|
|
56
51
|
},
|
|
57
52
|
}
|
|
58
53
|
|
|
59
54
|
|
|
60
|
-
def _resources_requested_by_replica(
|
|
61
|
-
ray_actor_options: dict[str, Any],
|
|
62
|
-
) -> _RayResources:
|
|
63
|
-
resources = {
|
|
64
|
-
name: float(amount)
|
|
65
|
-
for name, amount in ray_actor_options.get("resources", {}).items()
|
|
66
|
-
}
|
|
67
|
-
if ray_actor_options.get("num_gpus"):
|
|
68
|
-
resources["GPU"] = float(ray_actor_options["num_gpus"])
|
|
69
|
-
if ray_actor_options.get("memory"):
|
|
70
|
-
resources["memory"] = float(ray_actor_options["memory"])
|
|
71
|
-
return resources
|
|
72
|
-
|
|
73
|
-
|
|
74
55
|
class ModelAutoscalingPolicy:
|
|
75
|
-
def __init__(self
|
|
76
|
-
self._replica_resources = replica_resources
|
|
56
|
+
def __init__(self) -> None:
|
|
77
57
|
self._scheduler: ActorProxy[_ModelScheduler] | None = None
|
|
78
58
|
self._pending_pause_answer: Future[bool] | None = None
|
|
79
59
|
self._pause_requested_by_scheduler = False
|
|
80
60
|
|
|
81
61
|
def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
|
|
82
62
|
has_requests = context.total_num_requests > 0
|
|
83
|
-
waiting_for_replica = (
|
|
84
|
-
has_requests or context.target_num_replicas > 0
|
|
85
|
-
) and not context.running_replicas
|
|
86
63
|
if not self._awaiting_pause_answer():
|
|
87
64
|
self._pause_requested_by_scheduler = self._collect_pause_answer()
|
|
88
65
|
self._send_activity_report(
|
|
89
|
-
context.deployment_id.to_replica_actor_class_name(),
|
|
90
|
-
has_requests,
|
|
91
|
-
waiting_for_replica,
|
|
66
|
+
context.deployment_id.to_replica_actor_class_name(), has_requests
|
|
92
67
|
)
|
|
93
68
|
return self._replica_count(context, has_requests), context.policy_state
|
|
94
69
|
|
|
@@ -115,17 +90,12 @@ class ModelAutoscalingPolicy:
|
|
|
115
90
|
self._scheduler = None
|
|
116
91
|
return False
|
|
117
92
|
|
|
118
|
-
def _send_activity_report(
|
|
119
|
-
self, replica_class_name: str, has_requests: bool, waiting_for_replica: bool
|
|
120
|
-
) -> None:
|
|
93
|
+
def _send_activity_report(self, replica_class_name: str, has_requests: bool) -> None:
|
|
121
94
|
if self._scheduler is None:
|
|
122
95
|
self._scheduler = _get_or_create_scheduler_actor()
|
|
123
96
|
self._pending_pause_answer = (
|
|
124
97
|
self._scheduler.report_activity_and_check_pause.remote(
|
|
125
|
-
replica_class_name,
|
|
126
|
-
self._replica_resources,
|
|
127
|
-
has_requests,
|
|
128
|
-
waiting_for_replica,
|
|
98
|
+
replica_class_name, has_requests
|
|
129
99
|
).future()
|
|
130
100
|
)
|
|
131
101
|
|
|
@@ -147,10 +117,7 @@ def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
|
|
|
147
117
|
|
|
148
118
|
@dataclass
|
|
149
119
|
class _ScheduledModel:
|
|
150
|
-
replica_resources: _RayResources
|
|
151
120
|
has_requests: bool = False
|
|
152
|
-
waiting_for_replica: bool = False
|
|
153
|
-
waiting_since: float = 0.0
|
|
154
121
|
last_request_at: float = 0.0
|
|
155
122
|
last_report_at: float = 0.0
|
|
156
123
|
|
|
@@ -161,48 +128,50 @@ class _NodeOccupancy:
|
|
|
161
128
|
resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
|
|
162
129
|
|
|
163
130
|
|
|
131
|
+
@dataclass
|
|
132
|
+
class _ReplicaWaitingForRoom:
|
|
133
|
+
actor_id: str
|
|
134
|
+
required_resources: _RayResources
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass
|
|
138
|
+
class _ClusterOccupancy:
|
|
139
|
+
nodes: list[_NodeOccupancy]
|
|
140
|
+
replicas_waiting_for_room: list[_ReplicaWaitingForRoom]
|
|
141
|
+
|
|
142
|
+
|
|
164
143
|
class _ModelScheduler:
|
|
165
144
|
def __init__(self) -> None:
|
|
166
145
|
self._models: dict[str, _ScheduledModel] = {}
|
|
167
146
|
self._models_to_pause: set[str] = set()
|
|
168
147
|
self._last_pause_decision_at = float("-inf")
|
|
148
|
+
self._first_seen_waiting_at: dict[str, float] = {}
|
|
169
149
|
|
|
170
150
|
@ray.method
|
|
171
151
|
def report_activity_and_check_pause(
|
|
172
|
-
self,
|
|
173
|
-
replica_class_name: str,
|
|
174
|
-
replica_resources: _RayResources,
|
|
175
|
-
has_requests: bool,
|
|
176
|
-
waiting_for_replica: bool,
|
|
152
|
+
self, replica_class_name: str, has_requests: bool
|
|
177
153
|
) -> bool:
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
replica_class_name, replica_resources, has_requests, waiting_for_replica, now
|
|
154
|
+
return self._record_activity_and_check_pause(
|
|
155
|
+
replica_class_name, has_requests, time.monotonic()
|
|
181
156
|
)
|
|
157
|
+
|
|
158
|
+
def _record_activity_and_check_pause(
|
|
159
|
+
self, replica_class_name: str, has_requests: bool, now: float
|
|
160
|
+
) -> bool:
|
|
161
|
+
self._record_activity(replica_class_name, has_requests, now)
|
|
182
162
|
if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
|
|
183
163
|
self._last_pause_decision_at = now
|
|
184
164
|
self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
|
|
185
|
-
self._models_to_pause = self._select_models_to_pause()
|
|
165
|
+
self._models_to_pause = self._select_models_to_pause(now)
|
|
186
166
|
return replica_class_name in self._models_to_pause
|
|
187
167
|
|
|
188
168
|
def _record_activity(
|
|
189
|
-
self,
|
|
190
|
-
replica_class_name: str,
|
|
191
|
-
replica_resources: _RayResources,
|
|
192
|
-
has_requests: bool,
|
|
193
|
-
waiting_for_replica: bool,
|
|
194
|
-
now: float,
|
|
169
|
+
self, replica_class_name: str, has_requests: bool, now: float
|
|
195
170
|
) -> None:
|
|
196
|
-
model = self._models.setdefault(
|
|
197
|
-
replica_class_name, _ScheduledModel(replica_resources)
|
|
198
|
-
)
|
|
199
|
-
model.replica_resources = replica_resources
|
|
200
|
-
if waiting_for_replica and not model.waiting_for_replica:
|
|
201
|
-
model.waiting_since = now
|
|
171
|
+
model = self._models.setdefault(replica_class_name, _ScheduledModel())
|
|
202
172
|
if has_requests:
|
|
203
173
|
model.last_request_at = now
|
|
204
174
|
model.has_requests = has_requests
|
|
205
|
-
model.waiting_for_replica = waiting_for_replica
|
|
206
175
|
model.last_report_at = now
|
|
207
176
|
|
|
208
177
|
def _forget_models_silent_since(self, cutoff: float) -> None:
|
|
@@ -212,26 +181,23 @@ class _ModelScheduler:
|
|
|
212
181
|
if model.last_report_at >= cutoff
|
|
213
182
|
}
|
|
214
183
|
|
|
215
|
-
def _select_models_to_pause(self) -> set[str]:
|
|
216
|
-
if not
|
|
184
|
+
def _select_models_to_pause(self, now: float) -> set[str]:
|
|
185
|
+
if not _any_serve_replica_waiting_for_room():
|
|
186
|
+
self._first_seen_waiting_at = {}
|
|
217
187
|
return set()
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
188
|
+
occupancy = _read_cluster_occupancy(set(self._models))
|
|
189
|
+
self._first_seen_waiting_at = {
|
|
190
|
+
replica.actor_id: self._first_seen_waiting_at.get(replica.actor_id, now)
|
|
191
|
+
for replica in occupancy.replicas_waiting_for_room
|
|
221
192
|
}
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
for name, model in self._models.items()
|
|
226
|
-
if model.waiting_for_replica
|
|
227
|
-
and name not in models_with_placed_replicas
|
|
228
|
-
),
|
|
229
|
-
key=lambda name: self._models[name].waiting_since,
|
|
193
|
+
waiting_longest_first = sorted(
|
|
194
|
+
occupancy.replicas_waiting_for_room,
|
|
195
|
+
key=lambda replica: self._first_seen_waiting_at[replica.actor_id],
|
|
230
196
|
)
|
|
231
197
|
models_to_pause: set[str] = set()
|
|
232
|
-
for
|
|
198
|
+
for replica in waiting_longest_first:
|
|
233
199
|
models_to_pause |= self._fewest_idle_models_to_pause_for(
|
|
234
|
-
|
|
200
|
+
replica.required_resources, occupancy.nodes, models_to_pause
|
|
235
201
|
)
|
|
236
202
|
return models_to_pause
|
|
237
203
|
|
|
@@ -281,25 +247,34 @@ class _ModelScheduler:
|
|
|
281
247
|
return None
|
|
282
248
|
|
|
283
249
|
|
|
284
|
-
def
|
|
285
|
-
|
|
286
|
-
)
|
|
250
|
+
def _any_serve_replica_waiting_for_room() -> bool:
|
|
251
|
+
return any(
|
|
252
|
+
_is_serve_replica_waiting_for_room(vars(actor))
|
|
253
|
+
for actor in _list_actors_in_state("PENDING_CREATION", detail=False)
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _read_cluster_occupancy(scheduled_replica_class_names: set[str]) -> _ClusterOccupancy:
|
|
287
258
|
occupancy_by_node_id = {
|
|
288
259
|
node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
|
|
289
260
|
for node in ray.nodes()
|
|
290
261
|
if node["Alive"]
|
|
291
262
|
}
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
detail=True,
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
):
|
|
263
|
+
replicas_waiting_for_room: list[_ReplicaWaitingForRoom] = []
|
|
264
|
+
for actor in [
|
|
265
|
+
*_list_actors_in_state("ALIVE", detail=True),
|
|
266
|
+
*_list_actors_in_state("PENDING_CREATION", detail=True),
|
|
267
|
+
]:
|
|
298
268
|
actor_fields = vars(actor)
|
|
269
|
+
reserved_resources = actor_fields["required_resources"] or {}
|
|
270
|
+
if _is_serve_replica_waiting_for_room(actor_fields):
|
|
271
|
+
replicas_waiting_for_room.append(
|
|
272
|
+
_ReplicaWaitingForRoom(actor_fields["actor_id"], reserved_resources)
|
|
273
|
+
)
|
|
274
|
+
continue
|
|
299
275
|
occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
|
|
300
276
|
if occupancy is None:
|
|
301
277
|
continue
|
|
302
|
-
reserved_resources = actor_fields["required_resources"] or {}
|
|
303
278
|
_subtract_resources(occupancy.free_resources, reserved_resources)
|
|
304
279
|
replica_class_name = actor_fields["class_name"]
|
|
305
280
|
if replica_class_name in scheduled_replica_class_names:
|
|
@@ -307,7 +282,27 @@ def _read_node_occupancy(
|
|
|
307
282
|
occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
|
|
308
283
|
reserved_resources,
|
|
309
284
|
)
|
|
310
|
-
return
|
|
285
|
+
return _ClusterOccupancy(
|
|
286
|
+
nodes=list(occupancy_by_node_id.values()),
|
|
287
|
+
replicas_waiting_for_room=replicas_waiting_for_room,
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _list_actors_in_state(state: str, detail: bool) -> list[Any]:
|
|
292
|
+
return list_actors(
|
|
293
|
+
filters=[("state", "=", state)],
|
|
294
|
+
detail=detail,
|
|
295
|
+
limit=_STATE_API_RESULT_LIMIT,
|
|
296
|
+
raise_on_missing_output=False,
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _is_serve_replica_waiting_for_room(actor_fields: dict[str, Any]) -> bool:
|
|
301
|
+
return (
|
|
302
|
+
actor_fields["state"] == "PENDING_CREATION"
|
|
303
|
+
and actor_fields["node_id"] is None
|
|
304
|
+
and actor_fields["class_name"].startswith(_SERVE_REPLICA_CLASS_PREFIX)
|
|
305
|
+
)
|
|
311
306
|
|
|
312
307
|
|
|
313
308
|
def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
|
|
@@ -6,13 +6,14 @@ On the cluster replica, `build` imports the serve-app class bundled at
|
|
|
6
6
|
ingress with the app it was marked with by `cortexgrid.serve.ingress` (again on
|
|
7
7
|
each replica, see `_IngressOnReplica`), wraps it as a Ray Serve deployment
|
|
8
8
|
with the replica count and Ray resource requests `deploy_model` derived from
|
|
9
|
-
the model's requirements, and binds it with
|
|
10
|
-
|
|
9
|
+
the model's requirements, and binds it with its deployment's
|
|
10
|
+
`cortexgrid.DeploymentKey`.
|
|
11
11
|
|
|
12
12
|
The serve-app owns everything about traffic: its own routes, request schemas,
|
|
13
13
|
streaming, and timeouts. cortexgrid does not interpose a request/response
|
|
14
|
-
contract - it only schedules the app and hands it the
|
|
15
|
-
|
|
14
|
+
contract - it only schedules the app and hands it the key it needs to fetch
|
|
15
|
+
its own weights via `cortexgrid.load_model` and its settings via
|
|
16
|
+
`cortexgrid.model_config`.
|
|
16
17
|
|
|
17
18
|
The serve-app declares no resources: the hardware a replica needs belongs to
|
|
18
19
|
the model and is stored in the registry (`cortexgrid.ModelRequirements`), and
|
|
@@ -28,6 +29,7 @@ from ray import serve
|
|
|
28
29
|
from ray.serve.deployment import Application
|
|
29
30
|
|
|
30
31
|
from cortexgrid._model_scheduler import model_autoscaling_config
|
|
32
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
31
33
|
from cortexgrid.serve import ingress_app
|
|
32
34
|
|
|
33
35
|
|
|
@@ -75,11 +77,16 @@ def build(args: dict[str, Any]) -> Application:
|
|
|
75
77
|
# This builder ships in the bundle, frozen at save time, while `args`
|
|
76
78
|
# come from the cortexgrid that deploys it; one older than the bundle
|
|
77
79
|
# sends neither key.
|
|
78
|
-
autoscaling_config=model_autoscaling_config(
|
|
79
|
-
args.get("num_replicas", 1), args.get("ray_actor_options", {})
|
|
80
|
-
),
|
|
80
|
+
autoscaling_config=model_autoscaling_config(args.get("num_replicas", 1)),
|
|
81
81
|
# Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
|
|
82
82
|
# had on Ray 2.9.
|
|
83
83
|
max_ongoing_requests=_MAX_ONGOING_REQUESTS,
|
|
84
84
|
ray_actor_options=args.get("ray_actor_options", {}),
|
|
85
|
-
).bind(
|
|
85
|
+
).bind(
|
|
86
|
+
DeploymentKey(
|
|
87
|
+
args["family"],
|
|
88
|
+
args["suffix"],
|
|
89
|
+
args["run_name"],
|
|
90
|
+
args.get("config_fingerprint", ""),
|
|
91
|
+
)
|
|
92
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Cortexgrid wrappers around Ray Serve.
|
|
2
|
+
|
|
3
|
+
Caller stays HTTP-only: deploy/undeploy/list talk to the Ray dashboard's
|
|
4
|
+
declarative `/api/serve/applications/` endpoint via [cortexgrid.ray_util],
|
|
5
|
+
never `ray.init`. The deployment class is bundled at `save_model` time, zipped,
|
|
6
|
+
uploaded to MinIO under
|
|
7
|
+
`serve-bundles/<run_name>/<family>__<suffix>/<fingerprint>.zip`, and
|
|
8
|
+
referenced via `runtime_env.working_dir` so Ray workers fetch it from there.
|
|
9
|
+
The bundle URL, class import path, pip list, and fingerprint are persisted as
|
|
10
|
+
tags on the model's registry entry so `deploy_model` can find them later
|
|
11
|
+
without the caller holding the class object. So are the model's
|
|
12
|
+
`ModelRequirements`, which `deploy_model` turns into the replica's Ray resource
|
|
13
|
+
requests.
|
|
14
|
+
|
|
15
|
+
Every deployment `deploy_model` puts on Ray Serve gets a record with the jobs
|
|
16
|
+
control plane, keyed by its `DeploymentKey` - the model's (family, suffix,
|
|
17
|
+
run_name) plus a fingerprint of the config it was deployed with: that config,
|
|
18
|
+
the spec it PUT, plus the phase, message and replica
|
|
19
|
+
placements the control plane last observed (`observe_deployments`, run on
|
|
20
|
+
every poll cycle). Listings and status reads come from those records; waits
|
|
21
|
+
ask the Serve controller directly.
|
|
22
|
+
|
|
23
|
+
Naming: the Ray Serve application is named "<family>__<suffix>__<run_name>",
|
|
24
|
+
followed by "__<config_fingerprint>" for a deployment given a config.
|
|
25
|
+
This relies on family/suffix/run_name not containing the literal "__".
|
|
26
|
+
|
|
27
|
+
See [docs/cortexgrid/model-serving.md](../../docs/cortexgrid/model-serving.md)
|
|
28
|
+
for the end-to-end design.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from cortexgrid.model_serving.application_spec import app_name
|
|
32
|
+
from cortexgrid.model_serving.deployment_key import (
|
|
33
|
+
DeploymentConfig,
|
|
34
|
+
DeploymentKey,
|
|
35
|
+
deployment_key,
|
|
36
|
+
)
|
|
37
|
+
from cortexgrid.model_serving.lifecycle import (
|
|
38
|
+
ModelDeployFailed,
|
|
39
|
+
ModelNotDeployed,
|
|
40
|
+
deploy_model,
|
|
41
|
+
redeploy_model,
|
|
42
|
+
undeploy_model,
|
|
43
|
+
wait_for_model_serving,
|
|
44
|
+
)
|
|
45
|
+
from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
|
|
46
|
+
from cortexgrid.model_serving.registry_tags import (
|
|
47
|
+
bundle_fingerprint_from_tags,
|
|
48
|
+
has_requirement_tags,
|
|
49
|
+
metadata_from_tags,
|
|
50
|
+
metadata_to_tags,
|
|
51
|
+
requirements_from_tags,
|
|
52
|
+
requirements_to_tags,
|
|
53
|
+
)
|
|
54
|
+
from cortexgrid.model_serving.serve_bundle import (
|
|
55
|
+
BundleMetadata,
|
|
56
|
+
ServeBundle,
|
|
57
|
+
build_bundle,
|
|
58
|
+
bundle_class,
|
|
59
|
+
bundle_fingerprint_from_url,
|
|
60
|
+
upload_bundle,
|
|
61
|
+
)
|
|
62
|
+
from cortexgrid.model_serving.status import (
|
|
63
|
+
Deployment,
|
|
64
|
+
ReplicaPlacement,
|
|
65
|
+
ServingMessage,
|
|
66
|
+
ServingStatus,
|
|
67
|
+
deployment_config,
|
|
68
|
+
list_deployed_models,
|
|
69
|
+
model_replica_placements,
|
|
70
|
+
model_serving_messages,
|
|
71
|
+
model_serving_status,
|
|
72
|
+
observe_deployments,
|
|
73
|
+
)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
6
|
+
from cortexgrid.model_serving.placement import ModelRequirements, ray_actor_options
|
|
7
|
+
from cortexgrid.model_serving.serve_bundle import (
|
|
8
|
+
BundleMetadata,
|
|
9
|
+
bundle_fingerprint_from_url,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def app_name(key: DeploymentKey) -> str:
|
|
14
|
+
return "__".join(_name_segments(key))
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def route_prefix(key: DeploymentKey) -> str:
|
|
18
|
+
return "/r/" + "/".join(_name_segments(key))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _name_segments(key: DeploymentKey) -> list[str]:
|
|
22
|
+
model_segments = [key.family, key.suffix, key.run_name]
|
|
23
|
+
if not key.config_fingerprint:
|
|
24
|
+
return model_segments
|
|
25
|
+
return [*model_segments, key.config_fingerprint]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def build_application_spec(
|
|
29
|
+
key: DeploymentKey,
|
|
30
|
+
meta: BundleMetadata,
|
|
31
|
+
requirements: ModelRequirements,
|
|
32
|
+
num_replicas: int,
|
|
33
|
+
tiers: list[int],
|
|
34
|
+
) -> dict[str, Any]:
|
|
35
|
+
"""Assemble a Ray Serve application schema from pre-bundled metadata, the
|
|
36
|
+
model's requirements, and the cluster's GPU size classes."""
|
|
37
|
+
# working_dir carries the serve-app's own source; Ray pip-installs the
|
|
38
|
+
# third-party distributions the image lacks into a per-node cached
|
|
39
|
+
# virtualenv layered on the image. No pip key when there are none, so Ray
|
|
40
|
+
# builds no virtualenv.
|
|
41
|
+
runtime_env: dict[str, Any] = {"working_dir": meta.bundle_url}
|
|
42
|
+
if meta.pip_requirements:
|
|
43
|
+
runtime_env["pip"] = meta.pip_requirements
|
|
44
|
+
return {
|
|
45
|
+
"name": app_name(key),
|
|
46
|
+
"route_prefix": route_prefix(key),
|
|
47
|
+
# Ray Serve REST requires import_path to point at an Application builder
|
|
48
|
+
# (callable returning a bound node) or an already-bound node. A bare
|
|
49
|
+
# Deployment class is rejected, so cortexgrid.deploy_model goes through
|
|
50
|
+
# a generic builder that re-imports the user's class and binds it.
|
|
51
|
+
"import_path": "cortexgrid._serve_entry:build",
|
|
52
|
+
"args": {
|
|
53
|
+
"class_import_path": meta.class_import_path,
|
|
54
|
+
"family": key.family,
|
|
55
|
+
"suffix": key.suffix,
|
|
56
|
+
"run_name": key.run_name,
|
|
57
|
+
"config_fingerprint": key.config_fingerprint,
|
|
58
|
+
"num_replicas": num_replicas,
|
|
59
|
+
"ray_actor_options": ray_actor_options(requirements, tiers),
|
|
60
|
+
},
|
|
61
|
+
"runtime_env": runtime_env,
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def bundle_fingerprint_in_spec(spec: dict[str, Any]) -> str:
|
|
66
|
+
return bundle_fingerprint_from_url(spec["runtime_env"]["working_dir"])
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def replica_count_in_spec(spec: dict[str, Any]) -> int:
|
|
70
|
+
return spec["args"]["num_replicas"]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
DeploymentConfig = dict[str, str]
|
|
9
|
+
|
|
10
|
+
_CONFIG_FINGERPRINT_LENGTH = 12
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class DeploymentKey:
|
|
15
|
+
family: str
|
|
16
|
+
suffix: str
|
|
17
|
+
run_name: str
|
|
18
|
+
config_fingerprint: str = ""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def deployment_key(
|
|
22
|
+
family: str, suffix: str, run_name: str, config: DeploymentConfig
|
|
23
|
+
) -> DeploymentKey:
|
|
24
|
+
return DeploymentKey(family, suffix, run_name, _config_fingerprint(config))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _config_fingerprint(config: DeploymentConfig) -> str:
|
|
28
|
+
for name, value in config.items():
|
|
29
|
+
if not isinstance(name, str) or not isinstance(value, str):
|
|
30
|
+
raise ValueError(
|
|
31
|
+
f"Deployment config must map strings to strings: {name!r}: {value!r}"
|
|
32
|
+
)
|
|
33
|
+
if not config:
|
|
34
|
+
return ""
|
|
35
|
+
canonical_config = json.dumps(config, sort_keys=True)
|
|
36
|
+
return hashlib.sha256(canonical_config.encode()).hexdigest()[
|
|
37
|
+
:_CONFIG_FINGERPRINT_LENGTH
|
|
38
|
+
]
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from cortexgrid import state
|
|
6
|
+
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
DeploymentRecord = dict[str, Any]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def get_deployment_record(key: DeploymentKey) -> DeploymentRecord | None:
|
|
13
|
+
return state.get(*_path(key), params=_params(key))
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def put_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> None:
|
|
17
|
+
state.put(*_path(key), body=body, params=_params(key))
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def patch_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> bool:
|
|
21
|
+
return state.patch(*_path(key), body=body, params=_params(key))
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def delete_deployment_record(key: DeploymentKey) -> None:
|
|
25
|
+
state.delete(*_path(key), params=_params(key))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def list_deployment_records() -> list[DeploymentRecord]:
|
|
29
|
+
return state.get("deployments")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def key_of_record(record: DeploymentRecord) -> DeploymentKey:
|
|
33
|
+
return DeploymentKey(
|
|
34
|
+
record["family"],
|
|
35
|
+
record["suffix"],
|
|
36
|
+
record["run_name"],
|
|
37
|
+
record["config_fingerprint"],
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _path(key: DeploymentKey) -> tuple[str, ...]:
|
|
42
|
+
return ("deployments", key.family, key.suffix, key.run_name)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _params(key: DeploymentKey) -> dict[str, str]:
|
|
46
|
+
return {"config_fingerprint": key.config_fingerprint}
|