cortexgrid 0.3.9__tar.gz → 0.3.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/PKG-INFO +1 -1
  2. cortexgrid-0.3.11/cortexgrid/_model_scheduler.py +322 -0
  3. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/_serve_entry.py +2 -1
  4. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/model_serving.py +31 -4
  5. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/pyproject.toml +1 -1
  6. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/.gitignore +0 -0
  7. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/LICENSE +0 -0
  8. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/__init__.py +0 -0
  9. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/_bundle.py +0 -0
  10. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/_ray_job_driver.py +0 -0
  11. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/checkpoint.py +0 -0
  12. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/experiment.py +0 -0
  13. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/infra.py +0 -0
  14. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/jobs.py +0 -0
  15. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/mlflow_util.py +0 -0
  16. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/model_storage.py +0 -0
  17. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/py.typed +0 -0
  18. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/ray_util.py +0 -0
  19. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/s3_util.py +0 -0
  20. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/secrets.py +0 -0
  21. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/serve.py +0 -0
  22. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/cortexgrid/state.py +0 -0
  23. {cortexgrid-0.3.9 → cortexgrid-0.3.11}/docs/cortexgrid/README.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.9
3
+ Version: 0.3.11
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -0,0 +1,322 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import time
5
+ from concurrent.futures import Future
6
+ from dataclasses import dataclass, field
7
+ from typing import Any
8
+
9
+ import ray
10
+ from ray.actor import ActorProxy
11
+ from ray.serve.config import AutoscalingContext
12
+ from ray.util.state import list_actors
13
+
14
+
15
+ log = logging.getLogger(__name__)
16
+
17
+
18
+ _RayResources = dict[str, float]
19
+
20
+
21
+ _SCHEDULER_PROTOCOL_VERSION = 2
22
+ _SCHEDULER_ACTOR_NAME = f"cortexgrid-model-scheduler-v{_SCHEDULER_PROTOCOL_VERSION}"
23
+ _SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
24
+
25
+ _PAUSE_DECISION_INTERVAL_S = 1.0
26
+ _FORGET_MODELS_SILENT_FOR_S = 30.0
27
+
28
+ _REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
29
+ _REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
30
+
31
+ _SERVE_REPLICA_CLASS_PREFIX = "ServeReplica:"
32
+ _STATE_API_RESULT_LIMIT = 10_000
33
+ _RESOURCE_FLOAT_TOLERANCE = 1e-6
34
+
35
+
36
+ def model_autoscaling_config(max_replicas: int) -> dict[str, Any]:
37
+ return {
38
+ "min_replicas": 0,
39
+ "initial_replicas": max_replicas,
40
+ "max_replicas": max_replicas,
41
+ "upscale_delay_s": 0,
42
+ "downscale_delay_s": 0,
43
+ "downscale_to_zero_delay_s": 0,
44
+ "metrics_interval_s": _REQUEST_METRICS_PUSH_INTERVAL_S,
45
+ "look_back_period_s": _REQUEST_METRICS_AVERAGING_WINDOW_S,
46
+ "policy": {
47
+ "policy_function": (
48
+ f"{ModelAutoscalingPolicy.__module__}:"
49
+ f"{ModelAutoscalingPolicy.__qualname__}"
50
+ ),
51
+ },
52
+ }
53
+
54
+
55
+ class ModelAutoscalingPolicy:
56
+ def __init__(self) -> None:
57
+ self._scheduler: ActorProxy[_ModelScheduler] | None = None
58
+ self._pending_pause_answer: Future[bool] | None = None
59
+ self._pause_requested_by_scheduler = False
60
+
61
+ def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
62
+ has_requests = context.total_num_requests > 0
63
+ if not self._awaiting_pause_answer():
64
+ self._pause_requested_by_scheduler = self._collect_pause_answer()
65
+ self._send_activity_report(
66
+ context.deployment_id.to_replica_actor_class_name(), has_requests
67
+ )
68
+ return self._replica_count(context, has_requests), context.policy_state
69
+
70
+ def _replica_count(self, context: AutoscalingContext, has_requests: bool) -> int:
71
+ if has_requests:
72
+ return context.capacity_adjusted_max_replicas
73
+ if self._pause_requested_by_scheduler:
74
+ return 0
75
+ return context.target_num_replicas
76
+
77
+ def _awaiting_pause_answer(self) -> bool:
78
+ return (
79
+ self._pending_pause_answer is not None
80
+ and not self._pending_pause_answer.done()
81
+ )
82
+
83
+ def _collect_pause_answer(self) -> bool:
84
+ if self._pending_pause_answer is None:
85
+ return False
86
+ try:
87
+ return self._pending_pause_answer.result()
88
+ except Exception:
89
+ log.exception("Model scheduler unreachable")
90
+ self._scheduler = None
91
+ return False
92
+
93
+ def _send_activity_report(self, replica_class_name: str, has_requests: bool) -> None:
94
+ if self._scheduler is None:
95
+ self._scheduler = _get_or_create_scheduler_actor()
96
+ self._pending_pause_answer = (
97
+ self._scheduler.report_activity_and_check_pause.remote(
98
+ replica_class_name, has_requests
99
+ ).future()
100
+ )
101
+
102
+
103
+ def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
104
+ return (
105
+ ray.remote(_ModelScheduler)
106
+ .options(
107
+ name=_SCHEDULER_ACTOR_NAME,
108
+ namespace=_SCHEDULER_ACTOR_NAMESPACE,
109
+ get_if_exists=True,
110
+ lifetime="detached",
111
+ num_cpus=0,
112
+ max_restarts=-1,
113
+ )
114
+ .remote()
115
+ )
116
+
117
+
118
+ @dataclass
119
+ class _ScheduledModel:
120
+ has_requests: bool = False
121
+ last_request_at: float = 0.0
122
+ last_report_at: float = 0.0
123
+
124
+
125
+ @dataclass
126
+ class _NodeOccupancy:
127
+ free_resources: _RayResources
128
+ resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
129
+
130
+
131
+ @dataclass
132
+ class _ReplicaWaitingForRoom:
133
+ actor_id: str
134
+ required_resources: _RayResources
135
+
136
+
137
+ @dataclass
138
+ class _ClusterOccupancy:
139
+ nodes: list[_NodeOccupancy]
140
+ replicas_waiting_for_room: list[_ReplicaWaitingForRoom]
141
+
142
+
143
+ class _ModelScheduler:
144
+ def __init__(self) -> None:
145
+ self._models: dict[str, _ScheduledModel] = {}
146
+ self._models_to_pause: set[str] = set()
147
+ self._last_pause_decision_at = float("-inf")
148
+ self._first_seen_waiting_at: dict[str, float] = {}
149
+
150
+ @ray.method
151
+ def report_activity_and_check_pause(
152
+ self, replica_class_name: str, has_requests: bool
153
+ ) -> bool:
154
+ return self._record_activity_and_check_pause(
155
+ replica_class_name, has_requests, time.monotonic()
156
+ )
157
+
158
+ def _record_activity_and_check_pause(
159
+ self, replica_class_name: str, has_requests: bool, now: float
160
+ ) -> bool:
161
+ self._record_activity(replica_class_name, has_requests, now)
162
+ if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
163
+ self._last_pause_decision_at = now
164
+ self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
165
+ self._models_to_pause = self._select_models_to_pause(now)
166
+ return replica_class_name in self._models_to_pause
167
+
168
+ def _record_activity(
169
+ self, replica_class_name: str, has_requests: bool, now: float
170
+ ) -> None:
171
+ model = self._models.setdefault(replica_class_name, _ScheduledModel())
172
+ if has_requests:
173
+ model.last_request_at = now
174
+ model.has_requests = has_requests
175
+ model.last_report_at = now
176
+
177
+ def _forget_models_silent_since(self, cutoff: float) -> None:
178
+ self._models = {
179
+ name: model
180
+ for name, model in self._models.items()
181
+ if model.last_report_at >= cutoff
182
+ }
183
+
184
+ def _select_models_to_pause(self, now: float) -> set[str]:
185
+ if not _any_serve_replica_waiting_for_room():
186
+ self._first_seen_waiting_at = {}
187
+ return set()
188
+ occupancy = _read_cluster_occupancy(set(self._models))
189
+ self._first_seen_waiting_at = {
190
+ replica.actor_id: self._first_seen_waiting_at.get(replica.actor_id, now)
191
+ for replica in occupancy.replicas_waiting_for_room
192
+ }
193
+ waiting_longest_first = sorted(
194
+ occupancy.replicas_waiting_for_room,
195
+ key=lambda replica: self._first_seen_waiting_at[replica.actor_id],
196
+ )
197
+ models_to_pause: set[str] = set()
198
+ for replica in waiting_longest_first:
199
+ models_to_pause |= self._fewest_idle_models_to_pause_for(
200
+ replica.required_resources, occupancy.nodes, models_to_pause
201
+ )
202
+ return models_to_pause
203
+
204
+ def _fewest_idle_models_to_pause_for(
205
+ self,
206
+ required_resources: _RayResources,
207
+ nodes: list[_NodeOccupancy],
208
+ already_pausing: set[str],
209
+ ) -> set[str]:
210
+ fewest_models_to_pause: list[str] | None = None
211
+ for node in nodes:
212
+ if _has_room_for(required_resources, node.free_resources):
213
+ return set()
214
+ models_to_pause = self._least_recently_used_idle_models_freeing(
215
+ required_resources, node, already_pausing
216
+ )
217
+ if models_to_pause is not None and (
218
+ fewest_models_to_pause is None
219
+ or len(models_to_pause) < len(fewest_models_to_pause)
220
+ ):
221
+ fewest_models_to_pause = models_to_pause
222
+ return set(fewest_models_to_pause or ())
223
+
224
+ def _least_recently_used_idle_models_freeing(
225
+ self,
226
+ required_resources: _RayResources,
227
+ node: _NodeOccupancy,
228
+ already_pausing: set[str],
229
+ ) -> list[str] | None:
230
+ idle_models_least_recently_used_first = sorted(
231
+ (
232
+ name
233
+ for name in node.resources_held_by_model
234
+ if name not in already_pausing and not self._models[name].has_requests
235
+ ),
236
+ key=lambda name: self._models[name].last_request_at,
237
+ )
238
+ resources_free_after_pause = dict(node.free_resources)
239
+ models_to_pause: list[str] = []
240
+ for name in idle_models_least_recently_used_first:
241
+ models_to_pause.append(name)
242
+ _add_resources(
243
+ resources_free_after_pause, node.resources_held_by_model[name]
244
+ )
245
+ if _has_room_for(required_resources, resources_free_after_pause):
246
+ return models_to_pause
247
+ return None
248
+
249
+
250
+ def _any_serve_replica_waiting_for_room() -> bool:
251
+ return any(
252
+ _is_serve_replica_waiting_for_room(vars(actor))
253
+ for actor in _list_actors_in_state("PENDING_CREATION", detail=False)
254
+ )
255
+
256
+
257
+ def _read_cluster_occupancy(scheduled_replica_class_names: set[str]) -> _ClusterOccupancy:
258
+ occupancy_by_node_id = {
259
+ node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
260
+ for node in ray.nodes()
261
+ if node["Alive"]
262
+ }
263
+ replicas_waiting_for_room: list[_ReplicaWaitingForRoom] = []
264
+ for actor in [
265
+ *_list_actors_in_state("ALIVE", detail=True),
266
+ *_list_actors_in_state("PENDING_CREATION", detail=True),
267
+ ]:
268
+ actor_fields = vars(actor)
269
+ reserved_resources = actor_fields["required_resources"] or {}
270
+ if _is_serve_replica_waiting_for_room(actor_fields):
271
+ replicas_waiting_for_room.append(
272
+ _ReplicaWaitingForRoom(actor_fields["actor_id"], reserved_resources)
273
+ )
274
+ continue
275
+ occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
276
+ if occupancy is None:
277
+ continue
278
+ _subtract_resources(occupancy.free_resources, reserved_resources)
279
+ replica_class_name = actor_fields["class_name"]
280
+ if replica_class_name in scheduled_replica_class_names:
281
+ _add_resources(
282
+ occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
283
+ reserved_resources,
284
+ )
285
+ return _ClusterOccupancy(
286
+ nodes=list(occupancy_by_node_id.values()),
287
+ replicas_waiting_for_room=replicas_waiting_for_room,
288
+ )
289
+
290
+
291
+ def _list_actors_in_state(state: str, detail: bool) -> list[Any]:
292
+ return list_actors(
293
+ filters=[("state", "=", state)],
294
+ detail=detail,
295
+ limit=_STATE_API_RESULT_LIMIT,
296
+ raise_on_missing_output=False,
297
+ )
298
+
299
+
300
+ def _is_serve_replica_waiting_for_room(actor_fields: dict[str, Any]) -> bool:
301
+ return (
302
+ actor_fields["state"] == "PENDING_CREATION"
303
+ and actor_fields["node_id"] is None
304
+ and actor_fields["class_name"].startswith(_SERVE_REPLICA_CLASS_PREFIX)
305
+ )
306
+
307
+
308
+ def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
309
+ for name, amount in amounts.items():
310
+ target[name] = target.get(name, 0.0) + amount
311
+
312
+
313
+ def _subtract_resources(target: _RayResources, amounts: _RayResources) -> None:
314
+ for name, amount in amounts.items():
315
+ target[name] = target.get(name, 0.0) - amount
316
+
317
+
318
+ def _has_room_for(required: _RayResources, available: _RayResources) -> bool:
319
+ return all(
320
+ available.get(name, 0.0) + _RESOURCE_FLOAT_TOLERANCE >= amount
321
+ for name, amount in required.items()
322
+ )
@@ -27,6 +27,7 @@ from typing import Any
27
27
  from ray import serve
28
28
  from ray.serve.deployment import Application
29
29
 
30
+ from cortexgrid._model_scheduler import model_autoscaling_config
30
31
  from cortexgrid.serve import ingress_app
31
32
 
32
33
 
@@ -74,7 +75,7 @@ def build(args: dict[str, Any]) -> Application:
74
75
  # This builder ships in the bundle, frozen at save time, while `args`
75
76
  # come from the cortexgrid that deploys it; one older than the bundle
76
77
  # sends neither key.
77
- num_replicas=args.get("num_replicas", 1),
78
+ autoscaling_config=model_autoscaling_config(args.get("num_replicas", 1)),
78
79
  # Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
79
80
  # had on Ray 2.9.
80
81
  max_ongoing_requests=_MAX_ONGOING_REQUESTS,
@@ -39,7 +39,7 @@ import hashlib
39
39
  from pathlib import Path
40
40
  from typing import Any
41
41
 
42
- from ray.serve.schema import ApplicationStatus
42
+ from ray.serve.schema import ApplicationStatus, ReplicaState
43
43
 
44
44
  from cortexgrid import state
45
45
  from cortexgrid._bundle import bundle, digest, stage, worker_provides
@@ -77,6 +77,7 @@ def _route_prefix(family: str, suffix: str, run_name: str) -> str:
77
77
 
78
78
 
79
79
  _PHASE_NOT_DEPLOYED = "not_deployed"
80
+ _PHASE_PAUSED = "paused"
80
81
 
81
82
  # Ray Serve ApplicationStatus -> normalized serving phase. The single source of
82
83
  # the serving vocabulary, shared by Deployment, list_deployed_models, and
@@ -610,7 +611,7 @@ def _spec_already_deployed(spec: dict[str, Any]) -> bool:
610
611
  # Phases in which a deployment record stands for a live app: one a redeploy
611
612
  # with the same spec may leave alone. A failed app has to be cleared and
612
613
  # re-PUT, a deleting one waited out, and a missing one PUT again.
613
- _LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy")
614
+ _LIVE_PHASES = ("running", "deploying", "not_started", "unhealthy", _PHASE_PAUSED)
614
615
 
615
616
 
616
617
  def _record_is_current(
@@ -687,7 +688,9 @@ def deploy_model(
687
688
  phase = record["phase"]
688
689
  if wait:
689
690
  _wait_for_application_running(record["spec"]["name"], timeout, deadline)
690
- phase = "running"
691
+ phase = _observed(
692
+ get_serve_details().get("applications", {}).get(record["spec"]["name"])
693
+ )["phase"]
691
694
  return Deployment(
692
695
  family=family,
693
696
  suffix=suffix,
@@ -771,6 +774,7 @@ class ServingStatus:
771
774
  - "deploying" replicas starting; the replica pulls the weights and
772
775
  builds the model on the worker (DEPLOYING)
773
776
  - "running" serving traffic (RUNNING)
777
+ - "paused" scaled to zero replicas by the model scheduler
774
778
  - "unhealthy" Serve app reports UNHEALTHY
775
779
  - "failed" Serve app DEPLOY_FAILED
776
780
  - "deleting" Serve app being torn down (DELETING)
@@ -867,12 +871,35 @@ def _observed(app: dict[str, Any] | None) -> dict[str, Any]:
867
871
  return {"phase": _PHASE_NOT_DEPLOYED, "message": "", "replicas": []}
868
872
  raw = str(app.get("status", ""))
869
873
  return {
870
- "phase": _serve_phase(raw),
874
+ "phase": _phase(app),
871
875
  "message": str(app.get("message", "")) or raw,
872
876
  "replicas": [asdict(placement) for placement in _placements(app)],
873
877
  }
874
878
 
875
879
 
880
+ def _phase(app: dict[str, Any]) -> str:
881
+ serve_phase = _serve_phase(str(app.get("status", "")))
882
+ if serve_phase != "running":
883
+ return serve_phase
884
+ deployments = list(app.get("deployments", {}).values())
885
+ target_replica_counts = [
886
+ deployment.get("target_num_replicas") for deployment in deployments
887
+ ]
888
+ if target_replica_counts and all(count == 0 for count in target_replica_counts):
889
+ return _PHASE_PAUSED
890
+ if any(target_replica_counts) and not _has_running_replica(deployments):
891
+ return "deploying"
892
+ return serve_phase
893
+
894
+
895
+ def _has_running_replica(deployments: list[dict[str, Any]]) -> bool:
896
+ return any(
897
+ replica.get("state") == ReplicaState.RUNNING.value
898
+ for deployment in deployments
899
+ for replica in deployment.get("replicas", [])
900
+ )
901
+
902
+
876
903
  def observe_deployments() -> None:
877
904
  """Bring every deployment record up to date with one read of the Serve
878
905
  controller: its phase, message, and where its replicas run. The jobs
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.3.9"
3
+ version = "0.3.11"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes