cortexgrid 0.3.10__tar.gz → 0.3.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/PKG-INFO +1 -1
  2. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_model_scheduler.py +84 -89
  3. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_serve_entry.py +1 -3
  4. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/pyproject.toml +1 -1
  5. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/.gitignore +0 -0
  6. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/LICENSE +0 -0
  7. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/__init__.py +0 -0
  8. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_bundle.py +0 -0
  9. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_ray_job_driver.py +0 -0
  10. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/checkpoint.py +0 -0
  11. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/experiment.py +0 -0
  12. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/infra.py +0 -0
  13. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/jobs.py +0 -0
  14. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/mlflow_util.py +0 -0
  15. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/model_serving.py +0 -0
  16. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/model_storage.py +0 -0
  17. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/py.typed +0 -0
  18. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/ray_util.py +0 -0
  19. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/s3_util.py +0 -0
  20. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/secrets.py +0 -0
  21. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/serve.py +0 -0
  22. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/state.py +0 -0
  23. {cortexgrid-0.3.10 → cortexgrid-0.3.11}/docs/cortexgrid/README.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.10
3
+ Version: 0.3.11
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -18,7 +18,8 @@ log = logging.getLogger(__name__)
18
18
  _RayResources = dict[str, float]
19
19
 
20
20
 
21
- _SCHEDULER_ACTOR_NAME = "cortexgrid-model-scheduler"
21
+ _SCHEDULER_PROTOCOL_VERSION = 2
22
+ _SCHEDULER_ACTOR_NAME = f"cortexgrid-model-scheduler-v{_SCHEDULER_PROTOCOL_VERSION}"
22
23
  _SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
23
24
 
24
25
  _PAUSE_DECISION_INTERVAL_S = 1.0
@@ -27,13 +28,12 @@ _FORGET_MODELS_SILENT_FOR_S = 30.0
27
28
  _REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
28
29
  _REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
29
30
 
31
+ _SERVE_REPLICA_CLASS_PREFIX = "ServeReplica:"
30
32
  _STATE_API_RESULT_LIMIT = 10_000
31
33
  _RESOURCE_FLOAT_TOLERANCE = 1e-6
32
34
 
33
35
 
34
- def model_autoscaling_config(
35
- max_replicas: int, ray_actor_options: dict[str, Any]
36
- ) -> dict[str, Any]:
36
+ def model_autoscaling_config(max_replicas: int) -> dict[str, Any]:
37
37
  return {
38
38
  "min_replicas": 0,
39
39
  "initial_replicas": max_replicas,
@@ -48,47 +48,22 @@ def model_autoscaling_config(
48
48
  f"{ModelAutoscalingPolicy.__module__}:"
49
49
  f"{ModelAutoscalingPolicy.__qualname__}"
50
50
  ),
51
- "policy_kwargs": {
52
- "replica_resources": _resources_requested_by_replica(
53
- ray_actor_options
54
- )
55
- },
56
51
  },
57
52
  }
58
53
 
59
54
 
60
- def _resources_requested_by_replica(
61
- ray_actor_options: dict[str, Any],
62
- ) -> _RayResources:
63
- resources = {
64
- name: float(amount)
65
- for name, amount in ray_actor_options.get("resources", {}).items()
66
- }
67
- if ray_actor_options.get("num_gpus"):
68
- resources["GPU"] = float(ray_actor_options["num_gpus"])
69
- if ray_actor_options.get("memory"):
70
- resources["memory"] = float(ray_actor_options["memory"])
71
- return resources
72
-
73
-
74
55
  class ModelAutoscalingPolicy:
75
- def __init__(self, replica_resources: _RayResources) -> None:
76
- self._replica_resources = replica_resources
56
+ def __init__(self) -> None:
77
57
  self._scheduler: ActorProxy[_ModelScheduler] | None = None
78
58
  self._pending_pause_answer: Future[bool] | None = None
79
59
  self._pause_requested_by_scheduler = False
80
60
 
81
61
  def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
82
62
  has_requests = context.total_num_requests > 0
83
- waiting_for_replica = (
84
- has_requests or context.target_num_replicas > 0
85
- ) and not context.running_replicas
86
63
  if not self._awaiting_pause_answer():
87
64
  self._pause_requested_by_scheduler = self._collect_pause_answer()
88
65
  self._send_activity_report(
89
- context.deployment_id.to_replica_actor_class_name(),
90
- has_requests,
91
- waiting_for_replica,
66
+ context.deployment_id.to_replica_actor_class_name(), has_requests
92
67
  )
93
68
  return self._replica_count(context, has_requests), context.policy_state
94
69
 
@@ -115,17 +90,12 @@ class ModelAutoscalingPolicy:
115
90
  self._scheduler = None
116
91
  return False
117
92
 
118
- def _send_activity_report(
119
- self, replica_class_name: str, has_requests: bool, waiting_for_replica: bool
120
- ) -> None:
93
+ def _send_activity_report(self, replica_class_name: str, has_requests: bool) -> None:
121
94
  if self._scheduler is None:
122
95
  self._scheduler = _get_or_create_scheduler_actor()
123
96
  self._pending_pause_answer = (
124
97
  self._scheduler.report_activity_and_check_pause.remote(
125
- replica_class_name,
126
- self._replica_resources,
127
- has_requests,
128
- waiting_for_replica,
98
+ replica_class_name, has_requests
129
99
  ).future()
130
100
  )
131
101
 
@@ -147,10 +117,7 @@ def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
147
117
 
148
118
  @dataclass
149
119
  class _ScheduledModel:
150
- replica_resources: _RayResources
151
120
  has_requests: bool = False
152
- waiting_for_replica: bool = False
153
- waiting_since: float = 0.0
154
121
  last_request_at: float = 0.0
155
122
  last_report_at: float = 0.0
156
123
 
@@ -161,48 +128,50 @@ class _NodeOccupancy:
161
128
  resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
162
129
 
163
130
 
131
+ @dataclass
132
+ class _ReplicaWaitingForRoom:
133
+ actor_id: str
134
+ required_resources: _RayResources
135
+
136
+
137
+ @dataclass
138
+ class _ClusterOccupancy:
139
+ nodes: list[_NodeOccupancy]
140
+ replicas_waiting_for_room: list[_ReplicaWaitingForRoom]
141
+
142
+
164
143
  class _ModelScheduler:
165
144
  def __init__(self) -> None:
166
145
  self._models: dict[str, _ScheduledModel] = {}
167
146
  self._models_to_pause: set[str] = set()
168
147
  self._last_pause_decision_at = float("-inf")
148
+ self._first_seen_waiting_at: dict[str, float] = {}
169
149
 
170
150
  @ray.method
171
151
  def report_activity_and_check_pause(
172
- self,
173
- replica_class_name: str,
174
- replica_resources: _RayResources,
175
- has_requests: bool,
176
- waiting_for_replica: bool,
152
+ self, replica_class_name: str, has_requests: bool
177
153
  ) -> bool:
178
- now = time.monotonic()
179
- self._record_activity(
180
- replica_class_name, replica_resources, has_requests, waiting_for_replica, now
154
+ return self._record_activity_and_check_pause(
155
+ replica_class_name, has_requests, time.monotonic()
181
156
  )
157
+
158
+ def _record_activity_and_check_pause(
159
+ self, replica_class_name: str, has_requests: bool, now: float
160
+ ) -> bool:
161
+ self._record_activity(replica_class_name, has_requests, now)
182
162
  if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
183
163
  self._last_pause_decision_at = now
184
164
  self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
185
- self._models_to_pause = self._select_models_to_pause()
165
+ self._models_to_pause = self._select_models_to_pause(now)
186
166
  return replica_class_name in self._models_to_pause
187
167
 
188
168
  def _record_activity(
189
- self,
190
- replica_class_name: str,
191
- replica_resources: _RayResources,
192
- has_requests: bool,
193
- waiting_for_replica: bool,
194
- now: float,
169
+ self, replica_class_name: str, has_requests: bool, now: float
195
170
  ) -> None:
196
- model = self._models.setdefault(
197
- replica_class_name, _ScheduledModel(replica_resources)
198
- )
199
- model.replica_resources = replica_resources
200
- if waiting_for_replica and not model.waiting_for_replica:
201
- model.waiting_since = now
171
+ model = self._models.setdefault(replica_class_name, _ScheduledModel())
202
172
  if has_requests:
203
173
  model.last_request_at = now
204
174
  model.has_requests = has_requests
205
- model.waiting_for_replica = waiting_for_replica
206
175
  model.last_report_at = now
207
176
 
208
177
  def _forget_models_silent_since(self, cutoff: float) -> None:
@@ -212,26 +181,23 @@ class _ModelScheduler:
212
181
  if model.last_report_at >= cutoff
213
182
  }
214
183
 
215
- def _select_models_to_pause(self) -> set[str]:
216
- if not any(model.waiting_for_replica for model in self._models.values()):
184
+ def _select_models_to_pause(self, now: float) -> set[str]:
185
+ if not _any_serve_replica_waiting_for_room():
186
+ self._first_seen_waiting_at = {}
217
187
  return set()
218
- nodes = _read_node_occupancy(set(self._models))
219
- models_with_placed_replicas = {
220
- name for node in nodes for name in node.resources_held_by_model
188
+ occupancy = _read_cluster_occupancy(set(self._models))
189
+ self._first_seen_waiting_at = {
190
+ replica.actor_id: self._first_seen_waiting_at.get(replica.actor_id, now)
191
+ for replica in occupancy.replicas_waiting_for_room
221
192
  }
222
- models_waiting_for_room = sorted(
223
- (
224
- name
225
- for name, model in self._models.items()
226
- if model.waiting_for_replica
227
- and name not in models_with_placed_replicas
228
- ),
229
- key=lambda name: self._models[name].waiting_since,
193
+ waiting_longest_first = sorted(
194
+ occupancy.replicas_waiting_for_room,
195
+ key=lambda replica: self._first_seen_waiting_at[replica.actor_id],
230
196
  )
231
197
  models_to_pause: set[str] = set()
232
- for name in models_waiting_for_room:
198
+ for replica in waiting_longest_first:
233
199
  models_to_pause |= self._fewest_idle_models_to_pause_for(
234
- self._models[name].replica_resources, nodes, models_to_pause
200
+ replica.required_resources, occupancy.nodes, models_to_pause
235
201
  )
236
202
  return models_to_pause
237
203
 
@@ -281,25 +247,34 @@ class _ModelScheduler:
281
247
  return None
282
248
 
283
249
 
284
- def _read_node_occupancy(
285
- scheduled_replica_class_names: set[str],
286
- ) -> list[_NodeOccupancy]:
250
+ def _any_serve_replica_waiting_for_room() -> bool:
251
+ return any(
252
+ _is_serve_replica_waiting_for_room(vars(actor))
253
+ for actor in _list_actors_in_state("PENDING_CREATION", detail=False)
254
+ )
255
+
256
+
257
+ def _read_cluster_occupancy(scheduled_replica_class_names: set[str]) -> _ClusterOccupancy:
287
258
  occupancy_by_node_id = {
288
259
  node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
289
260
  for node in ray.nodes()
290
261
  if node["Alive"]
291
262
  }
292
- for actor in list_actors(
293
- filters=[("state", "=", "ALIVE")],
294
- detail=True,
295
- limit=_STATE_API_RESULT_LIMIT,
296
- raise_on_missing_output=False,
297
- ):
263
+ replicas_waiting_for_room: list[_ReplicaWaitingForRoom] = []
264
+ for actor in [
265
+ *_list_actors_in_state("ALIVE", detail=True),
266
+ *_list_actors_in_state("PENDING_CREATION", detail=True),
267
+ ]:
298
268
  actor_fields = vars(actor)
269
+ reserved_resources = actor_fields["required_resources"] or {}
270
+ if _is_serve_replica_waiting_for_room(actor_fields):
271
+ replicas_waiting_for_room.append(
272
+ _ReplicaWaitingForRoom(actor_fields["actor_id"], reserved_resources)
273
+ )
274
+ continue
299
275
  occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
300
276
  if occupancy is None:
301
277
  continue
302
- reserved_resources = actor_fields["required_resources"] or {}
303
278
  _subtract_resources(occupancy.free_resources, reserved_resources)
304
279
  replica_class_name = actor_fields["class_name"]
305
280
  if replica_class_name in scheduled_replica_class_names:
@@ -307,7 +282,27 @@ def _read_node_occupancy(
307
282
  occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
308
283
  reserved_resources,
309
284
  )
310
- return list(occupancy_by_node_id.values())
285
+ return _ClusterOccupancy(
286
+ nodes=list(occupancy_by_node_id.values()),
287
+ replicas_waiting_for_room=replicas_waiting_for_room,
288
+ )
289
+
290
+
291
+ def _list_actors_in_state(state: str, detail: bool) -> list[Any]:
292
+ return list_actors(
293
+ filters=[("state", "=", state)],
294
+ detail=detail,
295
+ limit=_STATE_API_RESULT_LIMIT,
296
+ raise_on_missing_output=False,
297
+ )
298
+
299
+
300
+ def _is_serve_replica_waiting_for_room(actor_fields: dict[str, Any]) -> bool:
301
+ return (
302
+ actor_fields["state"] == "PENDING_CREATION"
303
+ and actor_fields["node_id"] is None
304
+ and actor_fields["class_name"].startswith(_SERVE_REPLICA_CLASS_PREFIX)
305
+ )
311
306
 
312
307
 
313
308
  def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
@@ -75,9 +75,7 @@ def build(args: dict[str, Any]) -> Application:
75
75
  # This builder ships in the bundle, frozen at save time, while `args`
76
76
  # come from the cortexgrid that deploys it; one older than the bundle
77
77
  # sends neither key.
78
- autoscaling_config=model_autoscaling_config(
79
- args.get("num_replicas", 1), args.get("ray_actor_options", {})
80
- ),
78
+ autoscaling_config=model_autoscaling_config(args.get("num_replicas", 1)),
81
79
  # Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
82
80
  # had on Ray 2.9.
83
81
  max_ongoing_requests=_MAX_ONGOING_REQUESTS,
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.3.10"
3
+ version = "0.3.11"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes