cortexgrid 0.3.9__tar.gz → 0.3.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/PKG-INFO +1 -1
  2. cortexgrid-0.3.10/cortexgrid/_model_scheduler.py +327 -0
  3. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/_serve_entry.py +4 -1
  4. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/model_serving.py +31 -4
  5. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/pyproject.toml +1 -1
  6. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/.gitignore +0 -0
  7. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/LICENSE +0 -0
  8. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/__init__.py +0 -0
  9. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/_bundle.py +0 -0
  10. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/_ray_job_driver.py +0 -0
  11. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/checkpoint.py +0 -0
  12. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/experiment.py +0 -0
  13. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/infra.py +0 -0
  14. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/jobs.py +0 -0
  15. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/mlflow_util.py +0 -0
  16. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/model_storage.py +0 -0
  17. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/py.typed +0 -0
  18. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/ray_util.py +0 -0
  19. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/s3_util.py +0 -0
  20. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/secrets.py +0 -0
  21. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/serve.py +0 -0
  22. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/cortexgrid/state.py +0 -0
  23. {cortexgrid-0.3.9 → cortexgrid-0.3.10}/docs/cortexgrid/README.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.9
3
+ Version: 0.3.10
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -0,0 +1,327 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import time
5
+ from concurrent.futures import Future
6
+ from dataclasses import dataclass, field
7
+ from typing import Any
8
+
9
+ import ray
10
+ from ray.actor import ActorProxy
11
+ from ray.serve.config import AutoscalingContext
12
+ from ray.util.state import list_actors
13
+
14
+
15
+ log = logging.getLogger(__name__)
16
+
17
+
18
+ _RayResources = dict[str, float]
19
+
20
+
21
+ _SCHEDULER_ACTOR_NAME = "cortexgrid-model-scheduler"
22
+ _SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
23
+
24
+ _PAUSE_DECISION_INTERVAL_S = 1.0
25
+ _FORGET_MODELS_SILENT_FOR_S = 30.0
26
+
27
+ _REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
28
+ _REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
29
+
30
+ _STATE_API_RESULT_LIMIT = 10_000
31
+ _RESOURCE_FLOAT_TOLERANCE = 1e-6
32
+
33
+
34
+ def model_autoscaling_config(
35
+ max_replicas: int, ray_actor_options: dict[str, Any]
36
+ ) -> dict[str, Any]:
37
+ return {
38
+ "min_replicas": 0,
39
+ "initial_replicas": max_replicas,
40
+ "max_replicas": max_replicas,
41
+ "upscale_delay_s": 0,
42
+ "downscale_delay_s": 0,
43
+ "downscale_to_zero_delay_s": 0,
44
+ "metrics_interval_s": _REQUEST_METRICS_PUSH_INTERVAL_S,
45
+ "look_back_period_s": _REQUEST_METRICS_AVERAGING_WINDOW_S,
46
+ "policy": {
47
+ "policy_function": (
48
+ f"{ModelAutoscalingPolicy.__module__}:"
49
+ f"{ModelAutoscalingPolicy.__qualname__}"
50
+ ),
51
+ "policy_kwargs": {
52
+ "replica_resources": _resources_requested_by_replica(
53
+ ray_actor_options
54
+ )
55
+ },
56
+ },
57
+ }
58
+
59
+
60
+ def _resources_requested_by_replica(
61
+ ray_actor_options: dict[str, Any],
62
+ ) -> _RayResources:
63
+ resources = {
64
+ name: float(amount)
65
+ for name, amount in ray_actor_options.get("resources", {}).items()
66
+ }
67
+ if ray_actor_options.get("num_gpus"):
68
+ resources["GPU"] = float(ray_actor_options["num_gpus"])
69
+ if ray_actor_options.get("memory"):
70
+ resources["memory"] = float(ray_actor_options["memory"])
71
+ return resources
72
+
73
+
74
+ class ModelAutoscalingPolicy:
75
+ def __init__(self, replica_resources: _RayResources) -> None:
76
+ self._replica_resources = replica_resources
77
+ self._scheduler: ActorProxy[_ModelScheduler] | None = None
78
+ self._pending_pause_answer: Future[bool] | None = None
79
+ self._pause_requested_by_scheduler = False
80
+
81
+ def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
82
+ has_requests = context.total_num_requests > 0
83
+ waiting_for_replica = (
84
+ has_requests or context.target_num_replicas > 0
85
+ ) and not context.running_replicas
86
+ if not self._awaiting_pause_answer():
87
+ self._pause_requested_by_scheduler = self._collect_pause_answer()
88
+ self._send_activity_report(
89
+ context.deployment_id.to_replica_actor_class_name(),
90
+ has_requests,
91
+ waiting_for_replica,
92
+ )
93
+ return self._replica_count(context, has_requests), context.policy_state
94
+
95
+ def _replica_count(self, context: AutoscalingContext, has_requests: bool) -> int:
96
+ if has_requests:
97
+ return context.capacity_adjusted_max_replicas
98
+ if self._pause_requested_by_scheduler:
99
+ return 0
100
+ return context.target_num_replicas
101
+
102
+ def _awaiting_pause_answer(self) -> bool:
103
+ return (
104
+ self._pending_pause_answer is not None
105
+ and not self._pending_pause_answer.done()
106
+ )
107
+
108
+ def _collect_pause_answer(self) -> bool:
109
+ if self._pending_pause_answer is None:
110
+ return False
111
+ try:
112
+ return self._pending_pause_answer.result()
113
+ except Exception:
114
+ log.exception("Model scheduler unreachable")
115
+ self._scheduler = None
116
+ return False
117
+
118
+ def _send_activity_report(
119
+ self, replica_class_name: str, has_requests: bool, waiting_for_replica: bool
120
+ ) -> None:
121
+ if self._scheduler is None:
122
+ self._scheduler = _get_or_create_scheduler_actor()
123
+ self._pending_pause_answer = (
124
+ self._scheduler.report_activity_and_check_pause.remote(
125
+ replica_class_name,
126
+ self._replica_resources,
127
+ has_requests,
128
+ waiting_for_replica,
129
+ ).future()
130
+ )
131
+
132
+
133
+ def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
134
+ return (
135
+ ray.remote(_ModelScheduler)
136
+ .options(
137
+ name=_SCHEDULER_ACTOR_NAME,
138
+ namespace=_SCHEDULER_ACTOR_NAMESPACE,
139
+ get_if_exists=True,
140
+ lifetime="detached",
141
+ num_cpus=0,
142
+ max_restarts=-1,
143
+ )
144
+ .remote()
145
+ )
146
+
147
+
148
+ @dataclass
149
+ class _ScheduledModel:
150
+ replica_resources: _RayResources
151
+ has_requests: bool = False
152
+ waiting_for_replica: bool = False
153
+ waiting_since: float = 0.0
154
+ last_request_at: float = 0.0
155
+ last_report_at: float = 0.0
156
+
157
+
158
+ @dataclass
159
+ class _NodeOccupancy:
160
+ free_resources: _RayResources
161
+ resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
162
+
163
+
164
+ class _ModelScheduler:
165
+ def __init__(self) -> None:
166
+ self._models: dict[str, _ScheduledModel] = {}
167
+ self._models_to_pause: set[str] = set()
168
+ self._last_pause_decision_at = float("-inf")
169
+
170
+ @ray.method
171
+ def report_activity_and_check_pause(
172
+ self,
173
+ replica_class_name: str,
174
+ replica_resources: _RayResources,
175
+ has_requests: bool,
176
+ waiting_for_replica: bool,
177
+ ) -> bool:
178
+ now = time.monotonic()
179
+ self._record_activity(
180
+ replica_class_name, replica_resources, has_requests, waiting_for_replica, now
181
+ )
182
+ if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
183
+ self._last_pause_decision_at = now
184
+ self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
185
+ self._models_to_pause = self._select_models_to_pause()
186
+ return replica_class_name in self._models_to_pause
187
+
188
+ def _record_activity(
189
+ self,
190
+ replica_class_name: str,
191
+ replica_resources: _RayResources,
192
+ has_requests: bool,
193
+ waiting_for_replica: bool,
194
+ now: float,
195
+ ) -> None:
196
+ model = self._models.setdefault(
197
+ replica_class_name, _ScheduledModel(replica_resources)
198
+ )
199
+ model.replica_resources = replica_resources
200
+ if waiting_for_replica and not model.waiting_for_replica:
201
+ model.waiting_since = now
202
+ if has_requests:
203
+ model.last_request_at = now
204
+ model.has_requests = has_requests
205
+ model.waiting_for_replica = waiting_for_replica
206
+ model.last_report_at = now
207
+
208
+ def _forget_models_silent_since(self, cutoff: float) -> None:
209
+ self._models = {
210
+ name: model
211
+ for name, model in self._models.items()
212
+ if model.last_report_at >= cutoff
213
+ }
214
+
215
+ def _select_models_to_pause(self) -> set[str]:
216
+ if not any(model.waiting_for_replica for model in self._models.values()):
217
+ return set()
218
+ nodes = _read_node_occupancy(set(self._models))
219
+ models_with_placed_replicas = {
220
+ name for node in nodes for name in node.resources_held_by_model
221
+ }
222
+ models_waiting_for_room = sorted(
223
+ (
224
+ name
225
+ for name, model in self._models.items()
226
+ if model.waiting_for_replica
227
+ and name not in models_with_placed_replicas
228
+ ),
229
+ key=lambda name: self._models[name].waiting_since,
230
+ )
231
+ models_to_pause: set[str] = set()
232
+ for name in models_waiting_for_room:
233
+ models_to_pause |= self._fewest_idle_models_to_pause_for(
234
+ self._models[name].replica_resources, nodes, models_to_pause
235
+ )
236
+ return models_to_pause
237
+
238
+ def _fewest_idle_models_to_pause_for(
239
+ self,
240
+ required_resources: _RayResources,
241
+ nodes: list[_NodeOccupancy],
242
+ already_pausing: set[str],
243
+ ) -> set[str]:
244
+ fewest_models_to_pause: list[str] | None = None
245
+ for node in nodes:
246
+ if _has_room_for(required_resources, node.free_resources):
247
+ return set()
248
+ models_to_pause = self._least_recently_used_idle_models_freeing(
249
+ required_resources, node, already_pausing
250
+ )
251
+ if models_to_pause is not None and (
252
+ fewest_models_to_pause is None
253
+ or len(models_to_pause) < len(fewest_models_to_pause)
254
+ ):
255
+ fewest_models_to_pause = models_to_pause
256
+ return set(fewest_models_to_pause or ())
257
+
258
+ def _least_recently_used_idle_models_freeing(
259
+ self,
260
+ required_resources: _RayResources,
261
+ node: _NodeOccupancy,
262
+ already_pausing: set[str],
263
+ ) -> list[str] | None:
264
+ idle_models_least_recently_used_first = sorted(
265
+ (
266
+ name
267
+ for name in node.resources_held_by_model
268
+ if name not in already_pausing and not self._models[name].has_requests
269
+ ),
270
+ key=lambda name: self._models[name].last_request_at,
271
+ )
272
+ resources_free_after_pause = dict(node.free_resources)
273
+ models_to_pause: list[str] = []
274
+ for name in idle_models_least_recently_used_first:
275
+ models_to_pause.append(name)
276
+ _add_resources(
277
+ resources_free_after_pause, node.resources_held_by_model[name]
278
+ )
279
+ if _has_room_for(required_resources, resources_free_after_pause):
280
+ return models_to_pause
281
+ return None
282
+
283
+
284
+ def _read_node_occupancy(
285
+ scheduled_replica_class_names: set[str],
286
+ ) -> list[_NodeOccupancy]:
287
+ occupancy_by_node_id = {
288
+ node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
289
+ for node in ray.nodes()
290
+ if node["Alive"]
291
+ }
292
+ for actor in list_actors(
293
+ filters=[("state", "=", "ALIVE")],
294
+ detail=True,
295
+ limit=_STATE_API_RESULT_LIMIT,
296
+ raise_on_missing_output=False,
297
+ ):
298
+ actor_fields = vars(actor)
299
+ occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
300
+ if occupancy is None:
301
+ continue
302
+ reserved_resources = actor_fields["required_resources"] or {}
303
+ _subtract_resources(occupancy.free_resources, reserved_resources)
304
+ replica_class_name = actor_fields["class_name"]
305
+ if replica_class_name in scheduled_replica_class_names:
306
+ _add_resources(
307
+ occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
308
+ reserved_resources,
309
+ )
310
+ return list(occupancy_by_node_id.values())
311
+
312
+
313
+ def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
314
+ for name, amount in amounts.items():
315
+ target[name] = target.get(name, 0.0) + amount
316
+
317
+
318
+ def _subtract_resources(target: _RayResources, amounts: _RayResources) -> None:
319
+ for name, amount in amounts.items():
320
+ target[name] = target.get(name, 0.0) - amount
321
+
322
+
323
+ def _has_room_for(required: _RayResources, available: _RayResources) -> bool:
324
+ return all(
325
+ available.get(name, 0.0) + _RESOURCE_FLOAT_TOLERANCE >= amount
326
+ for name, amount in required.items()
327
+ )
@@ -27,6 +27,7 @@ from typing import Any
27
27
  from ray import serve
28
28
  from ray.serve.deployment import Application
29
29
 
30
+ from cortexgrid._model_scheduler import model_autoscaling_config
30
31
  from cortexgrid.serve import ingress_app
31
32
 
32
33
 
@@ -74,7 +75,9 @@ def build(args: dict[str, Any]) -> Application:
74
75
  # This builder ships in the bundle, frozen at save time, while `args`
75
76
  # come from the cortexgrid that deploys it; one older than the bundle
76
77
  # sends neither key.
77
- num_replicas=args.get("num_replicas", 1),
78
+ autoscaling_config=model_autoscaling_config(
79
+ args.get("num_replicas", 1), args.get("ray_actor_options", {})
80
+ ),
78
81
  # Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
79
82
  # had on Ray 2.9.
80
83
  max_ongoing_requests=_MAX_ONGOING_REQUESTS,
@@ -39,7 +39,7 @@ import hashlib
39
39
  from pathlib import Path
40
40
  from typing import Any
41
41
 
42
- from ray.serve.schema import ApplicationStatus
42
+ from ray.serve.schema import ApplicationStatus, ReplicaState
43
43
 
44
44
  from cortexgrid import state
45
45
  from cortexgrid._bundle import bundle, digest, stage, worker_provides
@@ -77,6 +77,7 @@ def _route_prefix(family: str, suffix: str, run_name: str) -> str:
77
77
 
78
78
 
79
79
  _PHASE_NOT_DEPLOYED = "not_deployed"
80
+ _PHASE_PAUSED = "paused"
80
81
 
81
82
  # Ray Serve ApplicationStatus -> normalized serving phase. The single source of
82
83
  # the serving vocabulary, shared by Deployment, list_deployed_models, and
@@ -610,7 +611,7 @@ def _spec_already_deployed(spec: dict[str, Any]) -> bool:
610
611
  # Phases in which a deployment record stands for a live app: one a redeploy
611
612
  # with the same spec may leave alone. A failed app has to be cleared and
612
613
  # re-PUT, a deleting one waited out, and a missing one PUT again.
613
- _LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy")
614
+ _LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy", _PHASE_PAUSED)
614
615
 
615
616
 
616
617
  def _record_is_current(
@@ -687,7 +688,9 @@ def deploy_model(
687
688
  phase = record["phase"]
688
689
  if wait:
689
690
  _wait_for_application_running(record["spec"]["name"], timeout, deadline)
690
- phase = "running"
691
+ phase = _observed(
692
+ get_serve_details().get("applications", {}).get(record["spec"]["name"])
693
+ )["phase"]
691
694
  return Deployment(
692
695
  family=family,
693
696
  suffix=suffix,
@@ -771,6 +774,7 @@ class ServingStatus:
771
774
  - "deploying" replicas starting; the replica pulls the weights and
772
775
  builds the model on the worker (DEPLOYING)
773
776
  - "running" serving traffic (RUNNING)
777
+ - "paused" scaled to zero replicas by the model scheduler
774
778
  - "unhealthy" Serve app reports UNHEALTHY
775
779
  - "failed" Serve app DEPLOY_FAILED
776
780
  - "deleting" Serve app being torn down (DELETING)
@@ -867,12 +871,35 @@ def _observed(app: dict[str, Any] | None) -> dict[str, Any]:
867
871
  return {"phase": _PHASE_NOT_DEPLOYED, "message": "", "replicas": []}
868
872
  raw = str(app.get("status", ""))
869
873
  return {
870
- "phase": _serve_phase(raw),
874
+ "phase": _phase(app),
871
875
  "message": str(app.get("message", "")) or raw,
872
876
  "replicas": [asdict(placement) for placement in _placements(app)],
873
877
  }
874
878
 
875
879
 
880
+ def _phase(app: dict[str, Any]) -> str:
881
+ serve_phase = _serve_phase(str(app.get("status", "")))
882
+ if serve_phase != "running":
883
+ return serve_phase
884
+ deployments = list(app.get("deployments", {}).values())
885
+ target_replica_counts = [
886
+ deployment.get("target_num_replicas") for deployment in deployments
887
+ ]
888
+ if target_replica_counts and all(count == 0 for count in target_replica_counts):
889
+ return _PHASE_PAUSED
890
+ if any(target_replica_counts) and not _has_running_replica(deployments):
891
+ return "deploying"
892
+ return serve_phase
893
+
894
+
895
+ def _has_running_replica(deployments: list[dict[str, Any]]) -> bool:
896
+ return any(
897
+ replica.get("state") == ReplicaState.RUNNING.value
898
+ for deployment in deployments
899
+ for replica in deployment.get("replicas", [])
900
+ )
901
+
902
+
876
903
  def observe_deployments() -> None:
877
904
  """Bring every deployment record up to date with one read of the Serve
878
905
  controller: its phase, message, and where its replicas run. The jobs
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.3.9"
3
+ version = "0.3.10"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes