cortexgrid 0.3.15__tar.gz → 0.3.16__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/PKG-INFO +10 -3
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/__init__.py +3 -1
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/__init__.py +3 -2
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/lifecycle.py +100 -16
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/serve_bundle.py +8 -1
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/status.py +5 -46
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/docs/cortexgrid/README.md +9 -2
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/pyproject.toml +1 -1
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/.gitignore +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/LICENSE +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/_model_scheduler.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/application_spec.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/deployment_key.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/deployment_records.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/placement.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_serving/registry_tags.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/serve.py +0 -0
- {cortexgrid-0.3.15 → cortexgrid-0.3.16}/cortexgrid/state.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.16
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -213,8 +213,15 @@ from fastapi import FastAPI
|
|
|
213
213
|
|
|
214
214
|
app = FastAPI()
|
|
215
215
|
|
|
216
|
+
class MyClient(cortexgrid.DeploymentClient):
|
|
217
|
+
def complete(self, prompt: str) -> str: ...
|
|
218
|
+
|
|
216
219
|
@serve.ingress(app)
|
|
217
220
|
class MyServeApp:
|
|
221
|
+
@classmethod
|
|
222
|
+
def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
|
|
223
|
+
return MyClient(key=deployment.key, url=deployment.url)
|
|
224
|
+
|
|
218
225
|
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
219
226
|
self._weights_dir = cortexgrid.load_model(
|
|
220
227
|
deployment.family, deployment.suffix, deployment.run_name
|
|
@@ -228,8 +235,8 @@ saved = cortexgrid.save_model(
|
|
|
228
235
|
# What one replica needs; the model is deployed only on a host that has it.
|
|
229
236
|
requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
|
|
230
237
|
)
|
|
231
|
-
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name
|
|
232
|
-
|
|
238
|
+
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
|
|
239
|
+
model = deployed.client() # MyServeApp's client, once the app serves
|
|
233
240
|
```
|
|
234
241
|
|
|
235
242
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
@@ -102,6 +102,7 @@ from cortexgrid.model_storage import register_model as _register_model_storage
|
|
|
102
102
|
from cortexgrid.model_storage import save_model as _save_model_storage
|
|
103
103
|
from cortexgrid.model_serving import (
|
|
104
104
|
Deployment,
|
|
105
|
+
DeploymentClient,
|
|
105
106
|
DeploymentConfig,
|
|
106
107
|
DeploymentKey,
|
|
107
108
|
ModelDeployFailed,
|
|
@@ -240,7 +241,7 @@ def deploy_model(
|
|
|
240
241
|
wait: bool = False,
|
|
241
242
|
timeout: float | None = 300.0,
|
|
242
243
|
config: dict[str, str] | None = None,
|
|
243
|
-
) -> Deployment:
|
|
244
|
+
) -> Deployment[Any]:
|
|
244
245
|
experiment = active_experiment()
|
|
245
246
|
return _deploy_model_serving(
|
|
246
247
|
family,
|
|
@@ -349,6 +350,7 @@ __all__ = [
|
|
|
349
350
|
"delete_model",
|
|
350
351
|
# Model serving
|
|
351
352
|
"Deployment",
|
|
353
|
+
"DeploymentClient",
|
|
352
354
|
"DeploymentConfig",
|
|
353
355
|
"DeploymentKey",
|
|
354
356
|
"ModelDeployFailed",
|
|
@@ -35,9 +35,12 @@ from cortexgrid.model_serving.deployment_key import (
|
|
|
35
35
|
deployment_key,
|
|
36
36
|
)
|
|
37
37
|
from cortexgrid.model_serving.lifecycle import (
|
|
38
|
+
Deployment,
|
|
39
|
+
DeploymentClient,
|
|
38
40
|
ModelDeployFailed,
|
|
39
41
|
ModelNotDeployed,
|
|
40
42
|
deploy_model,
|
|
43
|
+
list_deployed_models,
|
|
41
44
|
redeploy_model,
|
|
42
45
|
required_models,
|
|
43
46
|
undeploy_model,
|
|
@@ -62,12 +65,10 @@ from cortexgrid.model_serving.serve_bundle import (
|
|
|
62
65
|
upload_bundle,
|
|
63
66
|
)
|
|
64
67
|
from cortexgrid.model_serving.status import (
|
|
65
|
-
Deployment,
|
|
66
68
|
ReplicaPlacement,
|
|
67
69
|
ServingMessage,
|
|
68
70
|
ServingStatus,
|
|
69
71
|
deployment_config,
|
|
70
|
-
list_deployed_models,
|
|
71
72
|
model_replica_placements,
|
|
72
73
|
model_serving_messages,
|
|
73
74
|
model_serving_status,
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import asyncio
|
|
4
|
+
import importlib
|
|
3
5
|
import time
|
|
4
|
-
from
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any, Generic, TypeVar
|
|
5
8
|
|
|
6
9
|
from ray.serve.schema import ApplicationStatus
|
|
7
10
|
|
|
@@ -22,6 +25,8 @@ from cortexgrid.model_serving.deployment_records import (
|
|
|
22
25
|
DeploymentRecord,
|
|
23
26
|
delete_deployment_record,
|
|
24
27
|
get_deployment_record,
|
|
28
|
+
key_of_record,
|
|
29
|
+
list_deployment_records,
|
|
25
30
|
patch_deployment_record,
|
|
26
31
|
put_deployment_record,
|
|
27
32
|
)
|
|
@@ -29,9 +34,8 @@ from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
|
|
|
29
34
|
from cortexgrid.model_serving.registry_tags import load_deploy_metadata
|
|
30
35
|
from cortexgrid.model_serving.serve_bundle import BundleMetadata
|
|
31
36
|
from cortexgrid.model_serving.status import (
|
|
37
|
+
PHASE_NOT_DEPLOYED,
|
|
32
38
|
PHASE_PAUSED,
|
|
33
|
-
Deployment,
|
|
34
|
-
deployment_of_record,
|
|
35
39
|
observed,
|
|
36
40
|
replaced_bundle_fingerprint_until_rolled_out,
|
|
37
41
|
)
|
|
@@ -101,16 +105,12 @@ def _wait_for_application_running(
|
|
|
101
105
|
"""`wait_for_model_serving` against a deadline already running. `timeout`
|
|
102
106
|
only labels the TimeoutError."""
|
|
103
107
|
while True:
|
|
104
|
-
app =
|
|
105
|
-
if app is None:
|
|
106
|
-
raise ModelDeployFailed(f"Serve app {name!r} does not exist")
|
|
108
|
+
app = _application_still_able_to_serve(name)
|
|
107
109
|
status = str(app.get("status", "(missing)"))
|
|
108
|
-
message = str(app.get("message", ""))
|
|
109
110
|
if status == ApplicationStatus.RUNNING.value:
|
|
110
111
|
return
|
|
111
|
-
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
112
|
-
raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
|
|
113
112
|
if _past(deadline):
|
|
113
|
+
message = str(app.get("message", ""))
|
|
114
114
|
raise TimeoutError(
|
|
115
115
|
f"Serve app {name!r} did not reach RUNNING within {timeout}s "
|
|
116
116
|
f"(last status={status!r}, message={message!r})"
|
|
@@ -118,6 +118,90 @@ def _wait_for_application_running(
|
|
|
118
118
|
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
119
119
|
|
|
120
120
|
|
|
121
|
+
def _application_still_able_to_serve(name: str) -> dict[str, Any]:
|
|
122
|
+
serve_details = get_serve_details()
|
|
123
|
+
applications = serve_details.get("applications", {})
|
|
124
|
+
app: dict[str, Any] | None = applications.get(name)
|
|
125
|
+
if app is None:
|
|
126
|
+
raise ModelDeployFailed(f"Serve app {name!r} does not exist")
|
|
127
|
+
status = app.get("status")
|
|
128
|
+
if status == ApplicationStatus.DEPLOY_FAILED.value:
|
|
129
|
+
message = app.get("message", "")
|
|
130
|
+
raise ModelDeployFailed(f"Serve app {name!r} DEPLOY_FAILED: {message}")
|
|
131
|
+
return app
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _model_is_serving(key: DeploymentKey) -> bool:
|
|
135
|
+
name = app_name(key)
|
|
136
|
+
app = _application_still_able_to_serve(name)
|
|
137
|
+
status = app.get("status")
|
|
138
|
+
return status == ApplicationStatus.RUNNING.value
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@dataclass
|
|
142
|
+
class DeploymentClient:
|
|
143
|
+
key: DeploymentKey
|
|
144
|
+
url: str
|
|
145
|
+
|
|
146
|
+
async def is_ready(self) -> bool:
|
|
147
|
+
serving = await asyncio.to_thread(_model_is_serving, self.key)
|
|
148
|
+
return serving
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
DeploymentClientT = TypeVar("DeploymentClientT", bound=DeploymentClient)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass
|
|
155
|
+
class Deployment(Generic[DeploymentClientT]):
|
|
156
|
+
"""A scheduled Ray Serve app fronting a model. `phase` is the normalized
|
|
157
|
+
serving lifecycle phase (see `ServingStatus`); an app that appears in a
|
|
158
|
+
listing always exists, so its phase is never "not_deployed"."""
|
|
159
|
+
|
|
160
|
+
key: DeploymentKey
|
|
161
|
+
config: dict[str, str]
|
|
162
|
+
url: str
|
|
163
|
+
phase: str
|
|
164
|
+
bundle_fingerprint: str
|
|
165
|
+
replaced_bundle_fingerprint: str
|
|
166
|
+
experiment_name: str
|
|
167
|
+
class_import_path: str
|
|
168
|
+
|
|
169
|
+
def client(self) -> DeploymentClientT:
|
|
170
|
+
wait_for_model_serving(self.key)
|
|
171
|
+
deployment_client = self.client_async()
|
|
172
|
+
return deployment_client
|
|
173
|
+
|
|
174
|
+
def client_async(self) -> DeploymentClientT:
|
|
175
|
+
module_name, class_name = self.class_import_path.split(":")
|
|
176
|
+
serve_app_module = importlib.import_module(module_name)
|
|
177
|
+
serve_app = getattr(serve_app_module, class_name)
|
|
178
|
+
deployment_client: DeploymentClientT = serve_app.client(self)
|
|
179
|
+
return deployment_client
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def list_deployed_models() -> list[Deployment[Any]]:
|
|
183
|
+
"""Return a Deployment for every model `deploy_model` put on Ray Serve
|
|
184
|
+
whose app the control plane last saw existing."""
|
|
185
|
+
return [
|
|
186
|
+
deployment_of_record(record)
|
|
187
|
+
for record in list_deployment_records()
|
|
188
|
+
if record["phase"] != PHASE_NOT_DEPLOYED
|
|
189
|
+
]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def deployment_of_record(record: DeploymentRecord) -> Deployment[Any]:
|
|
193
|
+
return Deployment(
|
|
194
|
+
key=key_of_record(record),
|
|
195
|
+
config=record["config"],
|
|
196
|
+
url=record["url"],
|
|
197
|
+
phase=record["phase"],
|
|
198
|
+
bundle_fingerprint=bundle_fingerprint_in_spec(record["spec"]),
|
|
199
|
+
replaced_bundle_fingerprint=record["replaced_bundle_fingerprint"],
|
|
200
|
+
experiment_name=record["experiment_name"],
|
|
201
|
+
class_import_path=record["spec"]["args"]["class_import_path"],
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
121
205
|
def _clear_failed_application(
|
|
122
206
|
key: DeploymentKey, timeout: float | None, deadline: float | None
|
|
123
207
|
) -> None:
|
|
@@ -225,11 +309,10 @@ def deploy_model(
|
|
|
225
309
|
timeout: float | None = 300.0,
|
|
226
310
|
config: dict[str, str] | None = None,
|
|
227
311
|
experiment_name: str = "",
|
|
228
|
-
) -> Deployment:
|
|
312
|
+
) -> Deployment[Any]:
|
|
229
313
|
"""Schedule a Ray Serve app for a previously-saved model and return a
|
|
230
|
-
handle
|
|
231
|
-
|
|
232
|
-
request schemas - against that URL; cortexgrid imposes no traffic contract.
|
|
314
|
+
handle on it. The handle's `client()` is the client the serve-app's own
|
|
315
|
+
`client` factory makes for its routes; cortexgrid imposes no traffic contract.
|
|
233
316
|
|
|
234
317
|
The serve-app class is pulled from the registry entry's tags `save_model`
|
|
235
318
|
wrote at save time; the caller does not need to hold the class object.
|
|
@@ -332,10 +415,11 @@ def deploy_model(
|
|
|
332
415
|
bundle_fingerprint=meta.fingerprint,
|
|
333
416
|
replaced_bundle_fingerprint=observation["replaced_bundle_fingerprint"],
|
|
334
417
|
experiment_name=experiment_name,
|
|
418
|
+
class_import_path=meta.class_import_path,
|
|
335
419
|
)
|
|
336
420
|
|
|
337
421
|
|
|
338
|
-
def redeploy_model(key: DeploymentKey) -> Deployment:
|
|
422
|
+
def redeploy_model(key: DeploymentKey) -> Deployment[Any]:
|
|
339
423
|
record = get_deployment_record(key)
|
|
340
424
|
if record is None:
|
|
341
425
|
raise ModelNotDeployed(f"{key} is not deployed")
|
|
@@ -349,14 +433,14 @@ def redeploy_model(key: DeploymentKey) -> Deployment:
|
|
|
349
433
|
)
|
|
350
434
|
|
|
351
435
|
|
|
352
|
-
def required_models(deployment: DeploymentKey) -> list[Deployment]:
|
|
436
|
+
def required_models(deployment: DeploymentKey) -> list[Deployment[Any]]:
|
|
353
437
|
_, requirements = load_deploy_metadata(
|
|
354
438
|
deployment.family, deployment.suffix, deployment.run_name
|
|
355
439
|
)
|
|
356
440
|
return [_deployment_of(model) for model in requirements.models]
|
|
357
441
|
|
|
358
442
|
|
|
359
|
-
def _deployment_of(model: DeploymentConfig) -> Deployment:
|
|
443
|
+
def _deployment_of(model: DeploymentConfig) -> Deployment[Any]:
|
|
360
444
|
key = deployment_key(model.family, model.suffix, model.run_name, model.config)
|
|
361
445
|
record = get_deployment_record(key)
|
|
362
446
|
if record is None:
|
|
@@ -55,12 +55,19 @@ def build_bundle(cls: type) -> ServeBundle:
|
|
|
55
55
|
is a subclass Ray defines in its own module, and on older Ray (e.g. 2.9) it
|
|
56
56
|
reports that module as its own, so the class's source and import path would
|
|
57
57
|
resolve to Ray instead of the serve-app. `cortexgrid.serve.ingress` leaves
|
|
58
|
-
the class unwrapped.
|
|
58
|
+
the class unwrapped. Raises ValueError, too, for a class with no `client`
|
|
59
|
+
factory, which every `Deployment` of it makes its client with."""
|
|
59
60
|
if any(klass.__module__.startswith("ray.serve") for klass in cls.__mro__):
|
|
60
61
|
raise ValueError(
|
|
61
62
|
f"{cls.__name__} is wrapped by ray.serve.ingress; decorate it with "
|
|
62
63
|
"cortexgrid.serve.ingress instead (from cortexgrid import serve)"
|
|
63
64
|
)
|
|
65
|
+
if not callable(getattr(cls, "client", None)):
|
|
66
|
+
raise ValueError(
|
|
67
|
+
f"{cls.__name__} has no client factory; give it a classmethod "
|
|
68
|
+
"client(cls, deployment: cortexgrid.Deployment) returning its "
|
|
69
|
+
"cortexgrid.DeploymentClient"
|
|
70
|
+
)
|
|
64
71
|
entry_file = Path(inspect.getfile(cls)).resolve()
|
|
65
72
|
serve_entry = Path(__file__).parent.with_name("_serve_entry.py")
|
|
66
73
|
desc = bundle(entry_file).merge(bundle(serve_entry))
|
|
@@ -6,13 +6,9 @@ from typing import Any
|
|
|
6
6
|
|
|
7
7
|
from ray.serve.schema import ApplicationStatus, ReplicaState
|
|
8
8
|
|
|
9
|
-
from cortexgrid.model_serving.application_spec import
|
|
10
|
-
app_name,
|
|
11
|
-
bundle_fingerprint_in_spec,
|
|
12
|
-
)
|
|
9
|
+
from cortexgrid.model_serving.application_spec import app_name
|
|
13
10
|
from cortexgrid.model_serving.deployment_key import DeploymentKey
|
|
14
11
|
from cortexgrid.model_serving.deployment_records import (
|
|
15
|
-
DeploymentRecord,
|
|
16
12
|
get_deployment_record,
|
|
17
13
|
key_of_record,
|
|
18
14
|
list_deployment_records,
|
|
@@ -21,7 +17,7 @@ from cortexgrid.model_serving.deployment_records import (
|
|
|
21
17
|
from cortexgrid.ray_util import get_serve_details
|
|
22
18
|
|
|
23
19
|
|
|
24
|
-
|
|
20
|
+
PHASE_NOT_DEPLOYED = "not_deployed"
|
|
25
21
|
PHASE_PAUSED = "paused"
|
|
26
22
|
_ROLLED_OUT_PHASES = ("running", PHASE_PAUSED)
|
|
27
23
|
|
|
@@ -44,49 +40,12 @@ def _serve_phase(raw_status: str) -> str:
|
|
|
44
40
|
return _PHASE_BY_SERVE_STATUS.get(raw_status, "deploying")
|
|
45
41
|
|
|
46
42
|
|
|
47
|
-
@dataclass
|
|
48
|
-
class Deployment:
|
|
49
|
-
"""A scheduled Ray Serve app fronting a model. `phase` is the normalized
|
|
50
|
-
serving lifecycle phase (see `ServingStatus`); an app that appears in a
|
|
51
|
-
listing always exists, so its phase is never "not_deployed"."""
|
|
52
|
-
|
|
53
|
-
key: DeploymentKey
|
|
54
|
-
config: dict[str, str]
|
|
55
|
-
url: str
|
|
56
|
-
phase: str
|
|
57
|
-
bundle_fingerprint: str
|
|
58
|
-
replaced_bundle_fingerprint: str
|
|
59
|
-
experiment_name: str
|
|
60
|
-
|
|
61
|
-
|
|
62
43
|
def replaced_bundle_fingerprint_until_rolled_out(
|
|
63
44
|
replaced_bundle_fingerprint: str, phase: str
|
|
64
45
|
) -> str:
|
|
65
46
|
return "" if phase in _ROLLED_OUT_PHASES else replaced_bundle_fingerprint
|
|
66
47
|
|
|
67
48
|
|
|
68
|
-
def list_deployed_models() -> list[Deployment]:
|
|
69
|
-
"""Return a Deployment for every model `deploy_model` put on Ray Serve
|
|
70
|
-
whose app the control plane last saw existing."""
|
|
71
|
-
return [
|
|
72
|
-
deployment_of_record(record)
|
|
73
|
-
for record in list_deployment_records()
|
|
74
|
-
if record["phase"] != _PHASE_NOT_DEPLOYED
|
|
75
|
-
]
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
def deployment_of_record(record: DeploymentRecord) -> Deployment:
|
|
79
|
-
return Deployment(
|
|
80
|
-
key=key_of_record(record),
|
|
81
|
-
config=record["config"],
|
|
82
|
-
url=record["url"],
|
|
83
|
-
phase=record["phase"],
|
|
84
|
-
bundle_fingerprint=bundle_fingerprint_in_spec(record["spec"]),
|
|
85
|
-
replaced_bundle_fingerprint=record["replaced_bundle_fingerprint"],
|
|
86
|
-
experiment_name=record["experiment_name"],
|
|
87
|
-
)
|
|
88
|
-
|
|
89
|
-
|
|
90
49
|
def deployment_config(key: DeploymentKey) -> dict[str, str]:
|
|
91
50
|
record = get_deployment_record(key)
|
|
92
51
|
if record is None:
|
|
@@ -134,8 +93,8 @@ def model_serving_status(key: DeploymentKey) -> ServingStatus:
|
|
|
134
93
|
`cortexgrid.model_storage.model_registry_status`.
|
|
135
94
|
"""
|
|
136
95
|
record = get_deployment_record(key)
|
|
137
|
-
if record is None or record["phase"] ==
|
|
138
|
-
return ServingStatus(key,
|
|
96
|
+
if record is None or record["phase"] == PHASE_NOT_DEPLOYED:
|
|
97
|
+
return ServingStatus(key, PHASE_NOT_DEPLOYED, "", None)
|
|
139
98
|
return ServingStatus(
|
|
140
99
|
key=key,
|
|
141
100
|
phase=record["phase"],
|
|
@@ -193,7 +152,7 @@ def observed(app: dict[str, Any] | None) -> dict[str, Any]:
|
|
|
193
152
|
deployment record keeps; an app that does not exist reads as
|
|
194
153
|
"not_deployed"."""
|
|
195
154
|
if app is None:
|
|
196
|
-
return {"phase":
|
|
155
|
+
return {"phase": PHASE_NOT_DEPLOYED, "message": "", "replicas": []}
|
|
197
156
|
raw = str(app.get("status", ""))
|
|
198
157
|
return {
|
|
199
158
|
"phase": _phase(app),
|
|
@@ -185,8 +185,15 @@ from fastapi import FastAPI
|
|
|
185
185
|
|
|
186
186
|
app = FastAPI()
|
|
187
187
|
|
|
188
|
+
class MyClient(cortexgrid.DeploymentClient):
|
|
189
|
+
def complete(self, prompt: str) -> str: ...
|
|
190
|
+
|
|
188
191
|
@serve.ingress(app)
|
|
189
192
|
class MyServeApp:
|
|
193
|
+
@classmethod
|
|
194
|
+
def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
|
|
195
|
+
return MyClient(key=deployment.key, url=deployment.url)
|
|
196
|
+
|
|
190
197
|
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
191
198
|
self._weights_dir = cortexgrid.load_model(
|
|
192
199
|
deployment.family, deployment.suffix, deployment.run_name
|
|
@@ -200,8 +207,8 @@ saved = cortexgrid.save_model(
|
|
|
200
207
|
# What one replica needs; the model is deployed only on a host that has it.
|
|
201
208
|
requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
|
|
202
209
|
)
|
|
203
|
-
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name
|
|
204
|
-
|
|
210
|
+
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
|
|
211
|
+
model = deployed.client() # MyServeApp's client, once the app serves
|
|
205
212
|
```
|
|
206
213
|
|
|
207
214
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|