cortexgrid 0.3.10__tar.gz → 0.3.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/PKG-INFO +1 -1
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_model_scheduler.py +84 -89
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_serve_entry.py +1 -3
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/pyproject.toml +1 -1
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/.gitignore +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/LICENSE +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/__init__.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/model_serving.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/serve.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/cortexgrid/state.py +0 -0
- {cortexgrid-0.3.10 → cortexgrid-0.3.11}/docs/cortexgrid/README.md +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.11
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -18,7 +18,8 @@ log = logging.getLogger(__name__)
|
|
|
18
18
|
_RayResources = dict[str, float]
|
|
19
19
|
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
_SCHEDULER_PROTOCOL_VERSION = 2
|
|
22
|
+
_SCHEDULER_ACTOR_NAME = f"cortexgrid-model-scheduler-v{_SCHEDULER_PROTOCOL_VERSION}"
|
|
22
23
|
_SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
|
|
23
24
|
|
|
24
25
|
_PAUSE_DECISION_INTERVAL_S = 1.0
|
|
@@ -27,13 +28,12 @@ _FORGET_MODELS_SILENT_FOR_S = 30.0
|
|
|
27
28
|
_REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
|
|
28
29
|
_REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
|
|
29
30
|
|
|
31
|
+
_SERVE_REPLICA_CLASS_PREFIX = "ServeReplica:"
|
|
30
32
|
_STATE_API_RESULT_LIMIT = 10_000
|
|
31
33
|
_RESOURCE_FLOAT_TOLERANCE = 1e-6
|
|
32
34
|
|
|
33
35
|
|
|
34
|
-
def model_autoscaling_config(
|
|
35
|
-
max_replicas: int, ray_actor_options: dict[str, Any]
|
|
36
|
-
) -> dict[str, Any]:
|
|
36
|
+
def model_autoscaling_config(max_replicas: int) -> dict[str, Any]:
|
|
37
37
|
return {
|
|
38
38
|
"min_replicas": 0,
|
|
39
39
|
"initial_replicas": max_replicas,
|
|
@@ -48,47 +48,22 @@ def model_autoscaling_config(
|
|
|
48
48
|
f"{ModelAutoscalingPolicy.__module__}:"
|
|
49
49
|
f"{ModelAutoscalingPolicy.__qualname__}"
|
|
50
50
|
),
|
|
51
|
-
"policy_kwargs": {
|
|
52
|
-
"replica_resources": _resources_requested_by_replica(
|
|
53
|
-
ray_actor_options
|
|
54
|
-
)
|
|
55
|
-
},
|
|
56
51
|
},
|
|
57
52
|
}
|
|
58
53
|
|
|
59
54
|
|
|
60
|
-
def _resources_requested_by_replica(
|
|
61
|
-
ray_actor_options: dict[str, Any],
|
|
62
|
-
) -> _RayResources:
|
|
63
|
-
resources = {
|
|
64
|
-
name: float(amount)
|
|
65
|
-
for name, amount in ray_actor_options.get("resources", {}).items()
|
|
66
|
-
}
|
|
67
|
-
if ray_actor_options.get("num_gpus"):
|
|
68
|
-
resources["GPU"] = float(ray_actor_options["num_gpus"])
|
|
69
|
-
if ray_actor_options.get("memory"):
|
|
70
|
-
resources["memory"] = float(ray_actor_options["memory"])
|
|
71
|
-
return resources
|
|
72
|
-
|
|
73
|
-
|
|
74
55
|
class ModelAutoscalingPolicy:
|
|
75
|
-
def __init__(self
|
|
76
|
-
self._replica_resources = replica_resources
|
|
56
|
+
def __init__(self) -> None:
|
|
77
57
|
self._scheduler: ActorProxy[_ModelScheduler] | None = None
|
|
78
58
|
self._pending_pause_answer: Future[bool] | None = None
|
|
79
59
|
self._pause_requested_by_scheduler = False
|
|
80
60
|
|
|
81
61
|
def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
|
|
82
62
|
has_requests = context.total_num_requests > 0
|
|
83
|
-
waiting_for_replica = (
|
|
84
|
-
has_requests or context.target_num_replicas > 0
|
|
85
|
-
) and not context.running_replicas
|
|
86
63
|
if not self._awaiting_pause_answer():
|
|
87
64
|
self._pause_requested_by_scheduler = self._collect_pause_answer()
|
|
88
65
|
self._send_activity_report(
|
|
89
|
-
context.deployment_id.to_replica_actor_class_name(),
|
|
90
|
-
has_requests,
|
|
91
|
-
waiting_for_replica,
|
|
66
|
+
context.deployment_id.to_replica_actor_class_name(), has_requests
|
|
92
67
|
)
|
|
93
68
|
return self._replica_count(context, has_requests), context.policy_state
|
|
94
69
|
|
|
@@ -115,17 +90,12 @@ class ModelAutoscalingPolicy:
|
|
|
115
90
|
self._scheduler = None
|
|
116
91
|
return False
|
|
117
92
|
|
|
118
|
-
def _send_activity_report(
|
|
119
|
-
self, replica_class_name: str, has_requests: bool, waiting_for_replica: bool
|
|
120
|
-
) -> None:
|
|
93
|
+
def _send_activity_report(self, replica_class_name: str, has_requests: bool) -> None:
|
|
121
94
|
if self._scheduler is None:
|
|
122
95
|
self._scheduler = _get_or_create_scheduler_actor()
|
|
123
96
|
self._pending_pause_answer = (
|
|
124
97
|
self._scheduler.report_activity_and_check_pause.remote(
|
|
125
|
-
replica_class_name,
|
|
126
|
-
self._replica_resources,
|
|
127
|
-
has_requests,
|
|
128
|
-
waiting_for_replica,
|
|
98
|
+
replica_class_name, has_requests
|
|
129
99
|
).future()
|
|
130
100
|
)
|
|
131
101
|
|
|
@@ -147,10 +117,7 @@ def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
|
|
|
147
117
|
|
|
148
118
|
@dataclass
|
|
149
119
|
class _ScheduledModel:
|
|
150
|
-
replica_resources: _RayResources
|
|
151
120
|
has_requests: bool = False
|
|
152
|
-
waiting_for_replica: bool = False
|
|
153
|
-
waiting_since: float = 0.0
|
|
154
121
|
last_request_at: float = 0.0
|
|
155
122
|
last_report_at: float = 0.0
|
|
156
123
|
|
|
@@ -161,48 +128,50 @@ class _NodeOccupancy:
|
|
|
161
128
|
resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
|
|
162
129
|
|
|
163
130
|
|
|
131
|
+
@dataclass
|
|
132
|
+
class _ReplicaWaitingForRoom:
|
|
133
|
+
actor_id: str
|
|
134
|
+
required_resources: _RayResources
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass
|
|
138
|
+
class _ClusterOccupancy:
|
|
139
|
+
nodes: list[_NodeOccupancy]
|
|
140
|
+
replicas_waiting_for_room: list[_ReplicaWaitingForRoom]
|
|
141
|
+
|
|
142
|
+
|
|
164
143
|
class _ModelScheduler:
|
|
165
144
|
def __init__(self) -> None:
|
|
166
145
|
self._models: dict[str, _ScheduledModel] = {}
|
|
167
146
|
self._models_to_pause: set[str] = set()
|
|
168
147
|
self._last_pause_decision_at = float("-inf")
|
|
148
|
+
self._first_seen_waiting_at: dict[str, float] = {}
|
|
169
149
|
|
|
170
150
|
@ray.method
|
|
171
151
|
def report_activity_and_check_pause(
|
|
172
|
-
self,
|
|
173
|
-
replica_class_name: str,
|
|
174
|
-
replica_resources: _RayResources,
|
|
175
|
-
has_requests: bool,
|
|
176
|
-
waiting_for_replica: bool,
|
|
152
|
+
self, replica_class_name: str, has_requests: bool
|
|
177
153
|
) -> bool:
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
replica_class_name, replica_resources, has_requests, waiting_for_replica, now
|
|
154
|
+
return self._record_activity_and_check_pause(
|
|
155
|
+
replica_class_name, has_requests, time.monotonic()
|
|
181
156
|
)
|
|
157
|
+
|
|
158
|
+
def _record_activity_and_check_pause(
|
|
159
|
+
self, replica_class_name: str, has_requests: bool, now: float
|
|
160
|
+
) -> bool:
|
|
161
|
+
self._record_activity(replica_class_name, has_requests, now)
|
|
182
162
|
if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
|
|
183
163
|
self._last_pause_decision_at = now
|
|
184
164
|
self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
|
|
185
|
-
self._models_to_pause = self._select_models_to_pause()
|
|
165
|
+
self._models_to_pause = self._select_models_to_pause(now)
|
|
186
166
|
return replica_class_name in self._models_to_pause
|
|
187
167
|
|
|
188
168
|
def _record_activity(
|
|
189
|
-
self,
|
|
190
|
-
replica_class_name: str,
|
|
191
|
-
replica_resources: _RayResources,
|
|
192
|
-
has_requests: bool,
|
|
193
|
-
waiting_for_replica: bool,
|
|
194
|
-
now: float,
|
|
169
|
+
self, replica_class_name: str, has_requests: bool, now: float
|
|
195
170
|
) -> None:
|
|
196
|
-
model = self._models.setdefault(
|
|
197
|
-
replica_class_name, _ScheduledModel(replica_resources)
|
|
198
|
-
)
|
|
199
|
-
model.replica_resources = replica_resources
|
|
200
|
-
if waiting_for_replica and not model.waiting_for_replica:
|
|
201
|
-
model.waiting_since = now
|
|
171
|
+
model = self._models.setdefault(replica_class_name, _ScheduledModel())
|
|
202
172
|
if has_requests:
|
|
203
173
|
model.last_request_at = now
|
|
204
174
|
model.has_requests = has_requests
|
|
205
|
-
model.waiting_for_replica = waiting_for_replica
|
|
206
175
|
model.last_report_at = now
|
|
207
176
|
|
|
208
177
|
def _forget_models_silent_since(self, cutoff: float) -> None:
|
|
@@ -212,26 +181,23 @@ class _ModelScheduler:
|
|
|
212
181
|
if model.last_report_at >= cutoff
|
|
213
182
|
}
|
|
214
183
|
|
|
215
|
-
def _select_models_to_pause(self) -> set[str]:
|
|
216
|
-
if not
|
|
184
|
+
def _select_models_to_pause(self, now: float) -> set[str]:
|
|
185
|
+
if not _any_serve_replica_waiting_for_room():
|
|
186
|
+
self._first_seen_waiting_at = {}
|
|
217
187
|
return set()
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
188
|
+
occupancy = _read_cluster_occupancy(set(self._models))
|
|
189
|
+
self._first_seen_waiting_at = {
|
|
190
|
+
replica.actor_id: self._first_seen_waiting_at.get(replica.actor_id, now)
|
|
191
|
+
for replica in occupancy.replicas_waiting_for_room
|
|
221
192
|
}
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
for name, model in self._models.items()
|
|
226
|
-
if model.waiting_for_replica
|
|
227
|
-
and name not in models_with_placed_replicas
|
|
228
|
-
),
|
|
229
|
-
key=lambda name: self._models[name].waiting_since,
|
|
193
|
+
waiting_longest_first = sorted(
|
|
194
|
+
occupancy.replicas_waiting_for_room,
|
|
195
|
+
key=lambda replica: self._first_seen_waiting_at[replica.actor_id],
|
|
230
196
|
)
|
|
231
197
|
models_to_pause: set[str] = set()
|
|
232
|
-
for
|
|
198
|
+
for replica in waiting_longest_first:
|
|
233
199
|
models_to_pause |= self._fewest_idle_models_to_pause_for(
|
|
234
|
-
|
|
200
|
+
replica.required_resources, occupancy.nodes, models_to_pause
|
|
235
201
|
)
|
|
236
202
|
return models_to_pause
|
|
237
203
|
|
|
@@ -281,25 +247,34 @@ class _ModelScheduler:
|
|
|
281
247
|
return None
|
|
282
248
|
|
|
283
249
|
|
|
284
|
-
def
|
|
285
|
-
|
|
286
|
-
)
|
|
250
|
+
def _any_serve_replica_waiting_for_room() -> bool:
|
|
251
|
+
return any(
|
|
252
|
+
_is_serve_replica_waiting_for_room(vars(actor))
|
|
253
|
+
for actor in _list_actors_in_state("PENDING_CREATION", detail=False)
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _read_cluster_occupancy(scheduled_replica_class_names: set[str]) -> _ClusterOccupancy:
|
|
287
258
|
occupancy_by_node_id = {
|
|
288
259
|
node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
|
|
289
260
|
for node in ray.nodes()
|
|
290
261
|
if node["Alive"]
|
|
291
262
|
}
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
detail=True,
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
):
|
|
263
|
+
replicas_waiting_for_room: list[_ReplicaWaitingForRoom] = []
|
|
264
|
+
for actor in [
|
|
265
|
+
*_list_actors_in_state("ALIVE", detail=True),
|
|
266
|
+
*_list_actors_in_state("PENDING_CREATION", detail=True),
|
|
267
|
+
]:
|
|
298
268
|
actor_fields = vars(actor)
|
|
269
|
+
reserved_resources = actor_fields["required_resources"] or {}
|
|
270
|
+
if _is_serve_replica_waiting_for_room(actor_fields):
|
|
271
|
+
replicas_waiting_for_room.append(
|
|
272
|
+
_ReplicaWaitingForRoom(actor_fields["actor_id"], reserved_resources)
|
|
273
|
+
)
|
|
274
|
+
continue
|
|
299
275
|
occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
|
|
300
276
|
if occupancy is None:
|
|
301
277
|
continue
|
|
302
|
-
reserved_resources = actor_fields["required_resources"] or {}
|
|
303
278
|
_subtract_resources(occupancy.free_resources, reserved_resources)
|
|
304
279
|
replica_class_name = actor_fields["class_name"]
|
|
305
280
|
if replica_class_name in scheduled_replica_class_names:
|
|
@@ -307,7 +282,27 @@ def _read_node_occupancy(
|
|
|
307
282
|
occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
|
|
308
283
|
reserved_resources,
|
|
309
284
|
)
|
|
310
|
-
return
|
|
285
|
+
return _ClusterOccupancy(
|
|
286
|
+
nodes=list(occupancy_by_node_id.values()),
|
|
287
|
+
replicas_waiting_for_room=replicas_waiting_for_room,
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _list_actors_in_state(state: str, detail: bool) -> list[Any]:
|
|
292
|
+
return list_actors(
|
|
293
|
+
filters=[("state", "=", state)],
|
|
294
|
+
detail=detail,
|
|
295
|
+
limit=_STATE_API_RESULT_LIMIT,
|
|
296
|
+
raise_on_missing_output=False,
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _is_serve_replica_waiting_for_room(actor_fields: dict[str, Any]) -> bool:
|
|
301
|
+
return (
|
|
302
|
+
actor_fields["state"] == "PENDING_CREATION"
|
|
303
|
+
and actor_fields["node_id"] is None
|
|
304
|
+
and actor_fields["class_name"].startswith(_SERVE_REPLICA_CLASS_PREFIX)
|
|
305
|
+
)
|
|
311
306
|
|
|
312
307
|
|
|
313
308
|
def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
|
|
@@ -75,9 +75,7 @@ def build(args: dict[str, Any]) -> Application:
|
|
|
75
75
|
# This builder ships in the bundle, frozen at save time, while `args`
|
|
76
76
|
# come from the cortexgrid that deploys it; one older than the bundle
|
|
77
77
|
# sends neither key.
|
|
78
|
-
autoscaling_config=model_autoscaling_config(
|
|
79
|
-
args.get("num_replicas", 1), args.get("ray_actor_options", {})
|
|
80
|
-
),
|
|
78
|
+
autoscaling_config=model_autoscaling_config(args.get("num_replicas", 1)),
|
|
81
79
|
# Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
|
|
82
80
|
# had on Ray 2.9.
|
|
83
81
|
max_ongoing_requests=_MAX_ONGOING_REQUESTS,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|