cortexgrid 0.3.8__tar.gz → 0.3.10__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/PKG-INFO +11 -10
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/__init__.py +21 -11
- cortexgrid-0.3.10/cortexgrid/_model_scheduler.py +327 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/_serve_entry.py +4 -1
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/checkpoint.py +13 -26
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/experiment.py +81 -106
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/infra.py +10 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/jobs.py +53 -85
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/model_serving.py +183 -68
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/model_storage.py +98 -142
- cortexgrid-0.3.10/cortexgrid/state.py +80 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/docs/cortexgrid/README.md +10 -9
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/pyproject.toml +6 -1
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/.gitignore +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/LICENSE +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.8 → cortexgrid-0.3.10}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.10
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -55,7 +55,7 @@ import cortexgrid
|
|
|
55
55
|
cortexgrid.init(experiment="weather-forecast")
|
|
56
56
|
```
|
|
57
57
|
|
|
58
|
-
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named
|
|
58
|
+
That single call reads the service URLs from the head's secrets server at `$CORTEXGRID_HEAD_URL` and connects to all services through them. It also creates (or finds) the named experiment and starts a new run inside it. Experiments and runs are cortexgrid's own records, kept by the jobs control plane; each maps onto an MLflow experiment and run, which exist so MLflow can display the run's metrics. MLflow never frees the name of a deleted experiment, so an experiment re-created under the name of a deleted one gets its MLflow experiment under `<name>__<8 hex>`. Omit `experiment=` to auto-generate a unique name like `funky-koval-12`.
|
|
59
59
|
|
|
60
60
|
**One experiment per binary run.** `cortexgrid.init()` may only be called once per process. Every subsequent `cortexgrid.log_metric`, `cortexgrid.log_artifact`, checkpoint, and `cortexgrid.remote()` submission is scoped to that experiment+run. Remote jobs dispatched by the control plane inherit the experiment+run via the pickled payload, so their logging flows into the same MLflow run as the parent binary.
|
|
61
61
|
|
|
@@ -80,7 +80,7 @@ No run-scoping context manager — `init()` starts the run, and every subsequent
|
|
|
80
80
|
|
|
81
81
|
#### Checkpointing and resuming
|
|
82
82
|
|
|
83
|
-
Inside a cortexgrid job, `cortexgrid.checkpoint()` returns an attribute-based checkpoint object that persists to
|
|
83
|
+
Inside a cortexgrid job, `cortexgrid.checkpoint()` returns an attribute-based checkpoint object that persists to S3, with its manifest recorded by the jobs control plane, when its `with` block exits. On job restart (either manual retry or `retry=True`), `cortexgrid.resume()` returns the last checkpoint for the same job ID, or `None` if there isn't one.
|
|
84
84
|
|
|
85
85
|
```python
|
|
86
86
|
ckpt = cortexgrid.resume()
|
|
@@ -111,7 +111,7 @@ job = cortexgrid.remote(train_step, batch, num_gpus=1, retry=True)
|
|
|
111
111
|
print(f"Submitted: {job.job_id}")
|
|
112
112
|
```
|
|
113
113
|
|
|
114
|
-
`cortexgrid.remote` submits a job *request* (a pickled payload plus a `JobLifecycle` record) to
|
|
114
|
+
`cortexgrid.remote` submits a job *request* (a pickled payload plus a `JobLifecycle` record) to the jobs control plane and returns a `JobFuture` immediately. It does not wait for the job to run or finish — use the UI at `http://<DGX_IP>:8000`, `job.status()`, or poll `cortexgrid.list_experiment_run_jobs(run_id)`, to observe status.
|
|
115
115
|
|
|
116
116
|
##### Blocking on the result
|
|
117
117
|
|
|
@@ -163,7 +163,7 @@ The driver cloudpickles the outcome to `job/{job_id}/result.pkl`, beside the pay
|
|
|
163
163
|
|
|
164
164
|
Waiting on a `retry=True` job raises `ValueError`: retries are unbounded by design (see below), so the wait would have no end. Fire-and-forget submission is the form training uses — submit, then watch the UI.
|
|
165
165
|
|
|
166
|
-
A separate service — the **jobs control plane** —
|
|
166
|
+
A separate service — the **jobs control plane** — keeps cortexgrid's records (experiments, runs, jobs, the model registry, deployments) in Postgres and serves them over HTTP. Its poll loop reads the jobs it still has to act on, matches them against the set of Ray submissions the cluster already has, and submits anything missing. It is also responsible for retrying failed jobs and honouring user-requested stops.
|
|
167
167
|
|
|
168
168
|
Each submission captures the code and dependencies the entry function needs automatically ([_bundle.py](https://github.com/robodatalab/cortexgrid/blob/main/cortexgrid/_bundle.py)):
|
|
169
169
|
- `bundle(entry)` traces the import graph from the function's source file, resolving each import the way the interpreter does (via `sys.path`). The standard library is excluded (it ships with the interpreter)
|
|
@@ -182,7 +182,7 @@ Pass `retry=True` and the control plane will resubmit the job whenever Ray repor
|
|
|
182
182
|
cortexgrid.stop_experiment_run_jobs(run_id) # stops every job in the run
|
|
183
183
|
```
|
|
184
184
|
|
|
185
|
-
`stop_experiment_run_jobs` never touches Ray directly. It only flips `stop_requested` on each job's lifecycle record
|
|
185
|
+
`stop_experiment_run_jobs` never touches Ray directly. It only flips `stop_requested` on each job's lifecycle record, which the control plane keeps. The control plane observes the flag on its next poll and calls `ray.stop_job` for any attempt that has reached Ray. For jobs that have not yet been submitted, the same flag short-circuits the submission path inside the worker.
|
|
186
186
|
|
|
187
187
|
#### Object storage (S3/MinIO)
|
|
188
188
|
|
|
@@ -242,12 +242,12 @@ Anything else the serve-app has to know about the model - which model a provider
|
|
|
242
242
|
|
|
243
243
|
| Function | Description |
|
|
244
244
|
|----------|-------------|
|
|
245
|
-
| `cortexgrid.init(experiment=None)` | Configure connections + start a new
|
|
245
|
+
| `cortexgrid.init(experiment=None)` | Configure connections + start a new run inside the named experiment. One call per binary. |
|
|
246
246
|
| `cortexgrid.log_metric(key, value, step)` | Log a metric |
|
|
247
247
|
| `cortexgrid.log_metrics(metrics, step)` | Log multiple metrics |
|
|
248
248
|
| `cortexgrid.log_params(params)` | Log parameters |
|
|
249
249
|
| `cortexgrid.log_artifact(path, artifact_path)` | Log a file as an artifact |
|
|
250
|
-
| `cortexgrid.checkpoint()` | Context manager returning an attribute-based checkpoint saved
|
|
250
|
+
| `cortexgrid.checkpoint()` | Context manager returning an attribute-based checkpoint saved on exit |
|
|
251
251
|
| `cortexgrid.resume()` | Load the latest checkpoint for the current job, or `None` |
|
|
252
252
|
| `cortexgrid.remote(fn, *args, num_gpus=0, num_cpus=1, retry=False, **kwargs)` | Submit a function to the jobs control plane; returns a `JobFuture` |
|
|
253
253
|
| `JobFuture.status()` / `.done()` / `.result(timeout=None)` | Live status of a submitted job, and its function's return value (blocking) |
|
|
@@ -269,9 +269,10 @@ The DGX Spark runs the following services as k8s workloads managed by Argo CD (s
|
|
|
269
269
|
| Service | Port | Purpose |
|
|
270
270
|
|---------|------|---------|
|
|
271
271
|
| Ray | 8265 | Dashboard + job submission (NodePort 30265) |
|
|
272
|
-
|
|
|
272
|
+
| Jobs control plane | 8000 | cortexgrid's records (experiments, runs, jobs, model registry, deployments); schedules jobs on Ray (NodePort 30700) |
|
|
273
|
+
| MLflow | 5000 | Metrics and params of each run |
|
|
273
274
|
| MinIO | 9000/9001 | S3-compatible artifact storage |
|
|
274
|
-
| PostgreSQL | 5432 | MLflow metadata
|
|
275
|
+
| PostgreSQL | 5432 | `cortexgrid` database (the control plane's records), MLflow metadata, UI notes |
|
|
275
276
|
| Prometheus | 9090 | Metrics collection |
|
|
276
277
|
| Grafana | 3000 | Dashboards (GPU, jobs, system) |
|
|
277
278
|
|
|
@@ -31,6 +31,7 @@ from __future__ import annotations
|
|
|
31
31
|
from pathlib import Path
|
|
32
32
|
from typing import Any, Callable
|
|
33
33
|
|
|
34
|
+
from cortexgrid import state
|
|
34
35
|
from cortexgrid.checkpoint import checkpoint, resume
|
|
35
36
|
from cortexgrid.experiment import (
|
|
36
37
|
Experiment,
|
|
@@ -188,16 +189,14 @@ def import_model(
|
|
|
188
189
|
and record on the current Experiment's run which imported model it used.
|
|
189
190
|
|
|
190
191
|
The model belongs to no run (see `cortexgrid.model_storage.import_model`),
|
|
191
|
-
so the run keeps the link instead: the
|
|
192
|
-
`
|
|
193
|
-
|
|
192
|
+
so the run keeps the link instead: the run's record notes the model's
|
|
193
|
+
`created_at` under (family, suffix), whether this call uploaded the model
|
|
194
|
+
or reused it."""
|
|
194
195
|
experiment = Experiment.get_instance()
|
|
195
196
|
model = _import_model_storage(
|
|
196
197
|
source, serve_app, family, suffix, requirements, config
|
|
197
198
|
)
|
|
198
|
-
|
|
199
|
-
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
200
|
-
)
|
|
199
|
+
_record_imported_model(experiment.run_id, family, suffix, model)
|
|
201
200
|
return model
|
|
202
201
|
|
|
203
202
|
|
|
@@ -215,18 +214,29 @@ def register_model(
|
|
|
215
214
|
`import_model` without the import: everything it needs beyond its code goes
|
|
216
215
|
in `config`, which the serve-app reads with `model_config` at construction
|
|
217
216
|
(see `cortexgrid.model_storage.register_model`). The model belongs to no
|
|
218
|
-
run, so the run keeps the link the same way
|
|
219
|
-
`imported_model/<family>/<suffix>`."""
|
|
217
|
+
run, so the run keeps the link the same way."""
|
|
220
218
|
experiment = Experiment.get_instance()
|
|
221
219
|
model = _register_model_storage(
|
|
222
220
|
serve_app, family, suffix, requirements, config
|
|
223
221
|
)
|
|
224
|
-
|
|
225
|
-
experiment.run_id, f"imported_model/{family}/{suffix}", model.created_at
|
|
226
|
-
)
|
|
222
|
+
_record_imported_model(experiment.run_id, family, suffix, model)
|
|
227
223
|
return model
|
|
228
224
|
|
|
229
225
|
|
|
226
|
+
def _record_imported_model(
|
|
227
|
+
run_id: str, family: str, suffix: str, model: SavedModel
|
|
228
|
+
) -> None:
|
|
229
|
+
"""Note on the run's record which imported model it used."""
|
|
230
|
+
state.put(
|
|
231
|
+
"runs",
|
|
232
|
+
run_id,
|
|
233
|
+
"imported-models",
|
|
234
|
+
family,
|
|
235
|
+
suffix,
|
|
236
|
+
body={"created_at": model.created_at},
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
|
|
230
240
|
__all__ = [
|
|
231
241
|
"Experiment",
|
|
232
242
|
"delete_experiment",
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import time
|
|
5
|
+
from concurrent.futures import Future
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import ray
|
|
10
|
+
from ray.actor import ActorProxy
|
|
11
|
+
from ray.serve.config import AutoscalingContext
|
|
12
|
+
from ray.util.state import list_actors
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
log = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_RayResources = dict[str, float]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
_SCHEDULER_ACTOR_NAME = "cortexgrid-model-scheduler"
|
|
22
|
+
_SCHEDULER_ACTOR_NAMESPACE = "cortexgrid"
|
|
23
|
+
|
|
24
|
+
_PAUSE_DECISION_INTERVAL_S = 1.0
|
|
25
|
+
_FORGET_MODELS_SILENT_FOR_S = 30.0
|
|
26
|
+
|
|
27
|
+
_REQUEST_METRICS_PUSH_INTERVAL_S = 0.5
|
|
28
|
+
_REQUEST_METRICS_AVERAGING_WINDOW_S = 1.0
|
|
29
|
+
|
|
30
|
+
_STATE_API_RESULT_LIMIT = 10_000
|
|
31
|
+
_RESOURCE_FLOAT_TOLERANCE = 1e-6
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def model_autoscaling_config(
|
|
35
|
+
max_replicas: int, ray_actor_options: dict[str, Any]
|
|
36
|
+
) -> dict[str, Any]:
|
|
37
|
+
return {
|
|
38
|
+
"min_replicas": 0,
|
|
39
|
+
"initial_replicas": max_replicas,
|
|
40
|
+
"max_replicas": max_replicas,
|
|
41
|
+
"upscale_delay_s": 0,
|
|
42
|
+
"downscale_delay_s": 0,
|
|
43
|
+
"downscale_to_zero_delay_s": 0,
|
|
44
|
+
"metrics_interval_s": _REQUEST_METRICS_PUSH_INTERVAL_S,
|
|
45
|
+
"look_back_period_s": _REQUEST_METRICS_AVERAGING_WINDOW_S,
|
|
46
|
+
"policy": {
|
|
47
|
+
"policy_function": (
|
|
48
|
+
f"{ModelAutoscalingPolicy.__module__}:"
|
|
49
|
+
f"{ModelAutoscalingPolicy.__qualname__}"
|
|
50
|
+
),
|
|
51
|
+
"policy_kwargs": {
|
|
52
|
+
"replica_resources": _resources_requested_by_replica(
|
|
53
|
+
ray_actor_options
|
|
54
|
+
)
|
|
55
|
+
},
|
|
56
|
+
},
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _resources_requested_by_replica(
|
|
61
|
+
ray_actor_options: dict[str, Any],
|
|
62
|
+
) -> _RayResources:
|
|
63
|
+
resources = {
|
|
64
|
+
name: float(amount)
|
|
65
|
+
for name, amount in ray_actor_options.get("resources", {}).items()
|
|
66
|
+
}
|
|
67
|
+
if ray_actor_options.get("num_gpus"):
|
|
68
|
+
resources["GPU"] = float(ray_actor_options["num_gpus"])
|
|
69
|
+
if ray_actor_options.get("memory"):
|
|
70
|
+
resources["memory"] = float(ray_actor_options["memory"])
|
|
71
|
+
return resources
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class ModelAutoscalingPolicy:
|
|
75
|
+
def __init__(self, replica_resources: _RayResources) -> None:
|
|
76
|
+
self._replica_resources = replica_resources
|
|
77
|
+
self._scheduler: ActorProxy[_ModelScheduler] | None = None
|
|
78
|
+
self._pending_pause_answer: Future[bool] | None = None
|
|
79
|
+
self._pause_requested_by_scheduler = False
|
|
80
|
+
|
|
81
|
+
def __call__(self, context: AutoscalingContext) -> tuple[int, dict[str, Any]]:
|
|
82
|
+
has_requests = context.total_num_requests > 0
|
|
83
|
+
waiting_for_replica = (
|
|
84
|
+
has_requests or context.target_num_replicas > 0
|
|
85
|
+
) and not context.running_replicas
|
|
86
|
+
if not self._awaiting_pause_answer():
|
|
87
|
+
self._pause_requested_by_scheduler = self._collect_pause_answer()
|
|
88
|
+
self._send_activity_report(
|
|
89
|
+
context.deployment_id.to_replica_actor_class_name(),
|
|
90
|
+
has_requests,
|
|
91
|
+
waiting_for_replica,
|
|
92
|
+
)
|
|
93
|
+
return self._replica_count(context, has_requests), context.policy_state
|
|
94
|
+
|
|
95
|
+
def _replica_count(self, context: AutoscalingContext, has_requests: bool) -> int:
|
|
96
|
+
if has_requests:
|
|
97
|
+
return context.capacity_adjusted_max_replicas
|
|
98
|
+
if self._pause_requested_by_scheduler:
|
|
99
|
+
return 0
|
|
100
|
+
return context.target_num_replicas
|
|
101
|
+
|
|
102
|
+
def _awaiting_pause_answer(self) -> bool:
|
|
103
|
+
return (
|
|
104
|
+
self._pending_pause_answer is not None
|
|
105
|
+
and not self._pending_pause_answer.done()
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
def _collect_pause_answer(self) -> bool:
|
|
109
|
+
if self._pending_pause_answer is None:
|
|
110
|
+
return False
|
|
111
|
+
try:
|
|
112
|
+
return self._pending_pause_answer.result()
|
|
113
|
+
except Exception:
|
|
114
|
+
log.exception("Model scheduler unreachable")
|
|
115
|
+
self._scheduler = None
|
|
116
|
+
return False
|
|
117
|
+
|
|
118
|
+
def _send_activity_report(
|
|
119
|
+
self, replica_class_name: str, has_requests: bool, waiting_for_replica: bool
|
|
120
|
+
) -> None:
|
|
121
|
+
if self._scheduler is None:
|
|
122
|
+
self._scheduler = _get_or_create_scheduler_actor()
|
|
123
|
+
self._pending_pause_answer = (
|
|
124
|
+
self._scheduler.report_activity_and_check_pause.remote(
|
|
125
|
+
replica_class_name,
|
|
126
|
+
self._replica_resources,
|
|
127
|
+
has_requests,
|
|
128
|
+
waiting_for_replica,
|
|
129
|
+
).future()
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _get_or_create_scheduler_actor() -> ActorProxy[_ModelScheduler]:
|
|
134
|
+
return (
|
|
135
|
+
ray.remote(_ModelScheduler)
|
|
136
|
+
.options(
|
|
137
|
+
name=_SCHEDULER_ACTOR_NAME,
|
|
138
|
+
namespace=_SCHEDULER_ACTOR_NAMESPACE,
|
|
139
|
+
get_if_exists=True,
|
|
140
|
+
lifetime="detached",
|
|
141
|
+
num_cpus=0,
|
|
142
|
+
max_restarts=-1,
|
|
143
|
+
)
|
|
144
|
+
.remote()
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
@dataclass
|
|
149
|
+
class _ScheduledModel:
|
|
150
|
+
replica_resources: _RayResources
|
|
151
|
+
has_requests: bool = False
|
|
152
|
+
waiting_for_replica: bool = False
|
|
153
|
+
waiting_since: float = 0.0
|
|
154
|
+
last_request_at: float = 0.0
|
|
155
|
+
last_report_at: float = 0.0
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@dataclass
|
|
159
|
+
class _NodeOccupancy:
|
|
160
|
+
free_resources: _RayResources
|
|
161
|
+
resources_held_by_model: dict[str, _RayResources] = field(default_factory=dict)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class _ModelScheduler:
|
|
165
|
+
def __init__(self) -> None:
|
|
166
|
+
self._models: dict[str, _ScheduledModel] = {}
|
|
167
|
+
self._models_to_pause: set[str] = set()
|
|
168
|
+
self._last_pause_decision_at = float("-inf")
|
|
169
|
+
|
|
170
|
+
@ray.method
|
|
171
|
+
def report_activity_and_check_pause(
|
|
172
|
+
self,
|
|
173
|
+
replica_class_name: str,
|
|
174
|
+
replica_resources: _RayResources,
|
|
175
|
+
has_requests: bool,
|
|
176
|
+
waiting_for_replica: bool,
|
|
177
|
+
) -> bool:
|
|
178
|
+
now = time.monotonic()
|
|
179
|
+
self._record_activity(
|
|
180
|
+
replica_class_name, replica_resources, has_requests, waiting_for_replica, now
|
|
181
|
+
)
|
|
182
|
+
if now - self._last_pause_decision_at >= _PAUSE_DECISION_INTERVAL_S:
|
|
183
|
+
self._last_pause_decision_at = now
|
|
184
|
+
self._forget_models_silent_since(now - _FORGET_MODELS_SILENT_FOR_S)
|
|
185
|
+
self._models_to_pause = self._select_models_to_pause()
|
|
186
|
+
return replica_class_name in self._models_to_pause
|
|
187
|
+
|
|
188
|
+
def _record_activity(
|
|
189
|
+
self,
|
|
190
|
+
replica_class_name: str,
|
|
191
|
+
replica_resources: _RayResources,
|
|
192
|
+
has_requests: bool,
|
|
193
|
+
waiting_for_replica: bool,
|
|
194
|
+
now: float,
|
|
195
|
+
) -> None:
|
|
196
|
+
model = self._models.setdefault(
|
|
197
|
+
replica_class_name, _ScheduledModel(replica_resources)
|
|
198
|
+
)
|
|
199
|
+
model.replica_resources = replica_resources
|
|
200
|
+
if waiting_for_replica and not model.waiting_for_replica:
|
|
201
|
+
model.waiting_since = now
|
|
202
|
+
if has_requests:
|
|
203
|
+
model.last_request_at = now
|
|
204
|
+
model.has_requests = has_requests
|
|
205
|
+
model.waiting_for_replica = waiting_for_replica
|
|
206
|
+
model.last_report_at = now
|
|
207
|
+
|
|
208
|
+
def _forget_models_silent_since(self, cutoff: float) -> None:
|
|
209
|
+
self._models = {
|
|
210
|
+
name: model
|
|
211
|
+
for name, model in self._models.items()
|
|
212
|
+
if model.last_report_at >= cutoff
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
def _select_models_to_pause(self) -> set[str]:
|
|
216
|
+
if not any(model.waiting_for_replica for model in self._models.values()):
|
|
217
|
+
return set()
|
|
218
|
+
nodes = _read_node_occupancy(set(self._models))
|
|
219
|
+
models_with_placed_replicas = {
|
|
220
|
+
name for node in nodes for name in node.resources_held_by_model
|
|
221
|
+
}
|
|
222
|
+
models_waiting_for_room = sorted(
|
|
223
|
+
(
|
|
224
|
+
name
|
|
225
|
+
for name, model in self._models.items()
|
|
226
|
+
if model.waiting_for_replica
|
|
227
|
+
and name not in models_with_placed_replicas
|
|
228
|
+
),
|
|
229
|
+
key=lambda name: self._models[name].waiting_since,
|
|
230
|
+
)
|
|
231
|
+
models_to_pause: set[str] = set()
|
|
232
|
+
for name in models_waiting_for_room:
|
|
233
|
+
models_to_pause |= self._fewest_idle_models_to_pause_for(
|
|
234
|
+
self._models[name].replica_resources, nodes, models_to_pause
|
|
235
|
+
)
|
|
236
|
+
return models_to_pause
|
|
237
|
+
|
|
238
|
+
def _fewest_idle_models_to_pause_for(
|
|
239
|
+
self,
|
|
240
|
+
required_resources: _RayResources,
|
|
241
|
+
nodes: list[_NodeOccupancy],
|
|
242
|
+
already_pausing: set[str],
|
|
243
|
+
) -> set[str]:
|
|
244
|
+
fewest_models_to_pause: list[str] | None = None
|
|
245
|
+
for node in nodes:
|
|
246
|
+
if _has_room_for(required_resources, node.free_resources):
|
|
247
|
+
return set()
|
|
248
|
+
models_to_pause = self._least_recently_used_idle_models_freeing(
|
|
249
|
+
required_resources, node, already_pausing
|
|
250
|
+
)
|
|
251
|
+
if models_to_pause is not None and (
|
|
252
|
+
fewest_models_to_pause is None
|
|
253
|
+
or len(models_to_pause) < len(fewest_models_to_pause)
|
|
254
|
+
):
|
|
255
|
+
fewest_models_to_pause = models_to_pause
|
|
256
|
+
return set(fewest_models_to_pause or ())
|
|
257
|
+
|
|
258
|
+
def _least_recently_used_idle_models_freeing(
|
|
259
|
+
self,
|
|
260
|
+
required_resources: _RayResources,
|
|
261
|
+
node: _NodeOccupancy,
|
|
262
|
+
already_pausing: set[str],
|
|
263
|
+
) -> list[str] | None:
|
|
264
|
+
idle_models_least_recently_used_first = sorted(
|
|
265
|
+
(
|
|
266
|
+
name
|
|
267
|
+
for name in node.resources_held_by_model
|
|
268
|
+
if name not in already_pausing and not self._models[name].has_requests
|
|
269
|
+
),
|
|
270
|
+
key=lambda name: self._models[name].last_request_at,
|
|
271
|
+
)
|
|
272
|
+
resources_free_after_pause = dict(node.free_resources)
|
|
273
|
+
models_to_pause: list[str] = []
|
|
274
|
+
for name in idle_models_least_recently_used_first:
|
|
275
|
+
models_to_pause.append(name)
|
|
276
|
+
_add_resources(
|
|
277
|
+
resources_free_after_pause, node.resources_held_by_model[name]
|
|
278
|
+
)
|
|
279
|
+
if _has_room_for(required_resources, resources_free_after_pause):
|
|
280
|
+
return models_to_pause
|
|
281
|
+
return None
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _read_node_occupancy(
|
|
285
|
+
scheduled_replica_class_names: set[str],
|
|
286
|
+
) -> list[_NodeOccupancy]:
|
|
287
|
+
occupancy_by_node_id = {
|
|
288
|
+
node["NodeID"]: _NodeOccupancy(free_resources=dict(node["Resources"]))
|
|
289
|
+
for node in ray.nodes()
|
|
290
|
+
if node["Alive"]
|
|
291
|
+
}
|
|
292
|
+
for actor in list_actors(
|
|
293
|
+
filters=[("state", "=", "ALIVE")],
|
|
294
|
+
detail=True,
|
|
295
|
+
limit=_STATE_API_RESULT_LIMIT,
|
|
296
|
+
raise_on_missing_output=False,
|
|
297
|
+
):
|
|
298
|
+
actor_fields = vars(actor)
|
|
299
|
+
occupancy = occupancy_by_node_id.get(actor_fields["node_id"])
|
|
300
|
+
if occupancy is None:
|
|
301
|
+
continue
|
|
302
|
+
reserved_resources = actor_fields["required_resources"] or {}
|
|
303
|
+
_subtract_resources(occupancy.free_resources, reserved_resources)
|
|
304
|
+
replica_class_name = actor_fields["class_name"]
|
|
305
|
+
if replica_class_name in scheduled_replica_class_names:
|
|
306
|
+
_add_resources(
|
|
307
|
+
occupancy.resources_held_by_model.setdefault(replica_class_name, {}),
|
|
308
|
+
reserved_resources,
|
|
309
|
+
)
|
|
310
|
+
return list(occupancy_by_node_id.values())
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _add_resources(target: _RayResources, amounts: _RayResources) -> None:
|
|
314
|
+
for name, amount in amounts.items():
|
|
315
|
+
target[name] = target.get(name, 0.0) + amount
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _subtract_resources(target: _RayResources, amounts: _RayResources) -> None:
|
|
319
|
+
for name, amount in amounts.items():
|
|
320
|
+
target[name] = target.get(name, 0.0) - amount
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _has_room_for(required: _RayResources, available: _RayResources) -> bool:
|
|
324
|
+
return all(
|
|
325
|
+
available.get(name, 0.0) + _RESOURCE_FLOAT_TOLERANCE >= amount
|
|
326
|
+
for name, amount in required.items()
|
|
327
|
+
)
|
|
@@ -27,6 +27,7 @@ from typing import Any
|
|
|
27
27
|
from ray import serve
|
|
28
28
|
from ray.serve.deployment import Application
|
|
29
29
|
|
|
30
|
+
from cortexgrid._model_scheduler import model_autoscaling_config
|
|
30
31
|
from cortexgrid.serve import ingress_app
|
|
31
32
|
|
|
32
33
|
|
|
@@ -74,7 +75,9 @@ def build(args: dict[str, Any]) -> Application:
|
|
|
74
75
|
# This builder ships in the bundle, frozen at save time, while `args`
|
|
75
76
|
# come from the cortexgrid that deploys it; one older than the bundle
|
|
76
77
|
# sends neither key.
|
|
77
|
-
|
|
78
|
+
autoscaling_config=model_autoscaling_config(
|
|
79
|
+
args.get("num_replicas", 1), args.get("ray_actor_options", {})
|
|
80
|
+
),
|
|
78
81
|
# Ray 2.32 lowered the default from 100 to 5; keep what serve-apps
|
|
79
82
|
# had on Ray 2.9.
|
|
80
83
|
max_ongoing_requests=_MAX_ONGOING_REQUESTS,
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
"""Durable checkpointing for cortexgrid jobs.
|
|
2
2
|
|
|
3
|
-
Save arbitrary state (primitives, torch tensors, state_dicts)
|
|
4
|
-
|
|
3
|
+
Save arbitrary state (primitives, torch tensors, state_dicts) to S3, with a
|
|
4
|
+
manifest recorded by the jobs control plane, and resume from the latest
|
|
5
|
+
checkpoint on retry.
|
|
5
6
|
|
|
6
7
|
Usage (save)::
|
|
7
8
|
|
|
@@ -20,24 +21,22 @@ Usage (resume)::
|
|
|
20
21
|
|
|
21
22
|
from __future__ import annotations
|
|
22
23
|
|
|
23
|
-
import json
|
|
24
24
|
import logging
|
|
25
25
|
import tempfile
|
|
26
26
|
from pathlib import Path
|
|
27
27
|
from typing import Any
|
|
28
28
|
|
|
29
29
|
import cloudpickle # type: ignore
|
|
30
|
-
from mlflow.tracking import MlflowClient
|
|
31
30
|
|
|
32
|
-
from cortexgrid import s3_util
|
|
33
|
-
from cortexgrid.experiment import Experiment
|
|
31
|
+
from cortexgrid import s3_util, state
|
|
32
|
+
from cortexgrid.experiment import Experiment
|
|
34
33
|
|
|
35
34
|
log = logging.getLogger(__name__)
|
|
36
35
|
_CORTEXGRID_JOB_ID: str | None = None
|
|
37
36
|
|
|
38
37
|
|
|
39
38
|
class Checkpoint:
|
|
40
|
-
"""Attribute-based checkpoint persisted
|
|
39
|
+
"""Attribute-based checkpoint persisted to S3.
|
|
41
40
|
|
|
42
41
|
Assign any cloudpickle-compatible value to an attribute and it will be
|
|
43
42
|
saved when the context manager exits::
|
|
@@ -112,9 +111,8 @@ class Checkpoint:
|
|
|
112
111
|
self._persist()
|
|
113
112
|
|
|
114
113
|
def _persist(self) -> None:
|
|
115
|
-
"""Upload attr blobs to MinIO;
|
|
114
|
+
"""Upload attr blobs to MinIO; record the manifest with the control plane."""
|
|
116
115
|
exp = Experiment.get_instance()
|
|
117
|
-
client = MlflowClient(tracking_uri=get_mlflow_tracking_uri())
|
|
118
116
|
tmpdir = Path(tempfile.mkdtemp())
|
|
119
117
|
|
|
120
118
|
manifest: dict[str, Any] = {"attrs": {}}
|
|
@@ -127,28 +125,17 @@ class Checkpoint:
|
|
|
127
125
|
)
|
|
128
126
|
manifest["attrs"][name] = {"uri": uri}
|
|
129
127
|
|
|
130
|
-
(
|
|
131
|
-
client.log_artifact(
|
|
132
|
-
exp.run_id, str(tmpdir / "manifest.json"), artifact_path=self._prefix
|
|
133
|
-
)
|
|
128
|
+
state.put("runs", exp.run_id, "checkpoints", self._prefix, body=manifest)
|
|
134
129
|
log.info("Checkpoint saved: %s (%d attrs)", self._prefix, len(self._data))
|
|
135
130
|
|
|
136
131
|
@classmethod
|
|
137
132
|
def _load(cls, prefix: str) -> Checkpoint | None:
|
|
138
|
-
"""
|
|
133
|
+
"""Read the manifest from the control plane; download attr blobs from MinIO."""
|
|
139
134
|
exp = Experiment.get_instance()
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
if
|
|
144
|
-
a.path == manifest_rel for a in client.list_artifacts(exp.run_id, prefix)
|
|
145
|
-
):
|
|
146
|
-
return None
|
|
147
|
-
|
|
148
|
-
try:
|
|
149
|
-
manifest_path = client.download_artifacts(exp.run_id, manifest_rel)
|
|
150
|
-
manifest = json.loads(Path(manifest_path).read_text())
|
|
151
|
-
except Exception:
|
|
135
|
+
# An unreachable control plane raises rather than reading as "no
|
|
136
|
+
# checkpoint", which would restart a retried job from scratch.
|
|
137
|
+
manifest = state.get("runs", exp.run_id, "checkpoints", prefix)
|
|
138
|
+
if manifest is None:
|
|
152
139
|
return None
|
|
153
140
|
|
|
154
141
|
data: dict[str, Any] = {}
|