cortexgrid 0.3.16__tar.gz → 0.3.17__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/PKG-INFO +8 -16
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/lifecycle.py +5 -5
- cortexgrid-0.3.17/cortexgrid/serve.py +167 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/docs/cortexgrid/README.md +5 -15
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/pyproject.toml +3 -1
- cortexgrid-0.3.16/cortexgrid/serve.py +0 -43
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/.gitignore +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/LICENSE +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/__init__.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_model_scheduler.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/infra.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/__init__.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/application_spec.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/deployment_key.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/deployment_records.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/placement.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/registry_tags.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/serve_bundle.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/status.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/ray_util.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/state.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.17
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -11,7 +11,9 @@ Requires-Python: <3.12,>=3.11
|
|
|
11
11
|
Requires-Dist: boto3>=1.34
|
|
12
12
|
Requires-Dist: cloudpickle>=3.0
|
|
13
13
|
Requires-Dist: fabric>=3.2.3
|
|
14
|
+
Requires-Dist: fastapi<1,>=0.115
|
|
14
15
|
Requires-Dist: haikunator>=2.1.0
|
|
16
|
+
Requires-Dist: httpx>=0.27
|
|
15
17
|
Requires-Dist: mlflow<4,>=3.11
|
|
16
18
|
Requires-Dist: packaging>=24
|
|
17
19
|
Requires-Dist: pip>=23.0
|
|
@@ -205,30 +207,20 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
|
|
|
205
207
|
|
|
206
208
|
#### Model registry and serving
|
|
207
209
|
|
|
208
|
-
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class
|
|
210
|
+
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class marked with cortexgrid's `serve.ingress` (not Ray's), whose `serve.endpoint` methods it serves and whose client it generates:
|
|
209
211
|
|
|
210
212
|
```python
|
|
211
213
|
from cortexgrid import serve
|
|
212
|
-
from fastapi import FastAPI
|
|
213
214
|
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
class MyClient(cortexgrid.DeploymentClient):
|
|
217
|
-
def complete(self, prompt: str) -> str: ...
|
|
218
|
-
|
|
219
|
-
@serve.ingress(app)
|
|
215
|
+
@serve.ingress
|
|
220
216
|
class MyServeApp:
|
|
221
|
-
@classmethod
|
|
222
|
-
def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
|
|
223
|
-
return MyClient(key=deployment.key, url=deployment.url)
|
|
224
|
-
|
|
225
217
|
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
226
218
|
self._weights_dir = cortexgrid.load_model(
|
|
227
219
|
deployment.family, deployment.suffix, deployment.run_name
|
|
228
220
|
)
|
|
229
221
|
|
|
230
|
-
@
|
|
231
|
-
async def complete(self,
|
|
222
|
+
@serve.endpoint
|
|
223
|
+
async def complete(self, prompt: str) -> str: ...
|
|
232
224
|
|
|
233
225
|
saved = cortexgrid.save_model(
|
|
234
226
|
weights_dir, MyServeApp, family="qwen", suffix="instruct",
|
|
@@ -236,7 +228,7 @@ saved = cortexgrid.save_model(
|
|
|
236
228
|
requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
|
|
237
229
|
)
|
|
238
230
|
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
|
|
239
|
-
model = deployed.client() # MyServeApp's client, once the app serves
|
|
231
|
+
model: MyServeApp = deployed.client() # MyServeApp's client, once the app serves
|
|
240
232
|
```
|
|
241
233
|
|
|
242
234
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
@@ -148,11 +148,11 @@ class DeploymentClient:
|
|
|
148
148
|
return serving
|
|
149
149
|
|
|
150
150
|
|
|
151
|
-
|
|
151
|
+
ServeAppT = TypeVar("ServeAppT")
|
|
152
152
|
|
|
153
153
|
|
|
154
154
|
@dataclass
|
|
155
|
-
class Deployment(Generic[
|
|
155
|
+
class Deployment(Generic[ServeAppT]):
|
|
156
156
|
"""A scheduled Ray Serve app fronting a model. `phase` is the normalized
|
|
157
157
|
serving lifecycle phase (see `ServingStatus`); an app that appears in a
|
|
158
158
|
listing always exists, so its phase is never "not_deployed"."""
|
|
@@ -166,16 +166,16 @@ class Deployment(Generic[DeploymentClientT]):
|
|
|
166
166
|
experiment_name: str
|
|
167
167
|
class_import_path: str
|
|
168
168
|
|
|
169
|
-
def client(self) ->
|
|
169
|
+
def client(self) -> ServeAppT:
|
|
170
170
|
wait_for_model_serving(self.key)
|
|
171
171
|
deployment_client = self.client_async()
|
|
172
172
|
return deployment_client
|
|
173
173
|
|
|
174
|
-
def client_async(self) ->
|
|
174
|
+
def client_async(self) -> ServeAppT:
|
|
175
175
|
module_name, class_name = self.class_import_path.split(":")
|
|
176
176
|
serve_app_module = importlib.import_module(module_name)
|
|
177
177
|
serve_app = getattr(serve_app_module, class_name)
|
|
178
|
-
deployment_client:
|
|
178
|
+
deployment_client: ServeAppT = serve_app.client(self)
|
|
179
179
|
return deployment_client
|
|
180
180
|
|
|
181
181
|
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Declare a serve-app's HTTP ingress without importing Ray.
|
|
2
|
+
|
|
3
|
+
from cortexgrid import serve
|
|
4
|
+
|
|
5
|
+
@serve.ingress
|
|
6
|
+
class MyServeApp:
|
|
7
|
+
@serve.endpoint
|
|
8
|
+
async def predict(self, xs: list[float]) -> list[float]: ...
|
|
9
|
+
|
|
10
|
+
Unlike `ray.serve.ingress`, it builds the FastAPI app from the class's
|
|
11
|
+
`serve.endpoint` methods, and the class is left unwrapped: the FastAPI app
|
|
12
|
+
and the client generated for it are only recorded on it, and
|
|
13
|
+
`cortexgrid._serve_entry.build` applies Ray's ingress when it builds the Serve
|
|
14
|
+
application on the cluster.
|
|
15
|
+
|
|
16
|
+
Ray's decorator replaces the class with a wrapper subclass defined in
|
|
17
|
+
ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
|
|
18
|
+
that module. Everything that locates a serve-app by its module - bundling its
|
|
19
|
+
source, recording its import path - would then find Ray instead of the user's
|
|
20
|
+
code. Deferring the wrap to the one place Serve needs it keeps the class
|
|
21
|
+
locatable everywhere else (the laptop, Ray jobs, tests).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import inspect
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from typing import Any, Callable, TypeVar, get_type_hints
|
|
29
|
+
|
|
30
|
+
import httpx
|
|
31
|
+
from fastapi import FastAPI
|
|
32
|
+
from pydantic import TypeAdapter
|
|
33
|
+
|
|
34
|
+
from cortexgrid.model_serving.lifecycle import Deployment, DeploymentClient
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_T = TypeVar("_T", bound=type)
|
|
38
|
+
_Method = TypeVar("_Method", bound=Callable[..., Any])
|
|
39
|
+
|
|
40
|
+
_INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
|
|
41
|
+
_ENDPOINT_ATTR = "__cortexgrid_endpoint__"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def endpoint(method: _Method) -> _Method:
|
|
45
|
+
setattr(method, _ENDPOINT_ATTR, True)
|
|
46
|
+
return method
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def ingress(cls: _T) -> _T:
|
|
50
|
+
"""Mark a serve-app class as fronted by a FastAPI app serving its
|
|
51
|
+
endpoints, and give it the client that calls them. Returns the class
|
|
52
|
+
itself, unwrapped."""
|
|
53
|
+
app = FastAPI()
|
|
54
|
+
calls: dict[str, Any] = {"__module__": cls.__module__}
|
|
55
|
+
methods = inspect.getmembers(cls, inspect.isfunction)
|
|
56
|
+
for name, method in methods:
|
|
57
|
+
if getattr(method, _ENDPOINT_ATTR, False):
|
|
58
|
+
marshalling = _EndpointMarshalling.of(method)
|
|
59
|
+
route = _route_of(method, marshalling)
|
|
60
|
+
app.add_api_route(f"/{name}", route, methods=["POST"])
|
|
61
|
+
calls[name] = _call_of(name, method, marshalling)
|
|
62
|
+
client = type(f"{cls.__name__}Client", (_EndpointsClient,), calls)
|
|
63
|
+
setattr(cls, _INGRESS_APP_ATTR, app)
|
|
64
|
+
setattr(cls, "client", client)
|
|
65
|
+
return cls
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def ingress_app(cls: type) -> Any | None:
|
|
69
|
+
"""The app `cls` was marked with by `ingress`, or None if it was not."""
|
|
70
|
+
return getattr(cls, _INGRESS_APP_ATTR, None)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class _EndpointsClient(DeploymentClient):
|
|
74
|
+
def __init__(self, deployment: Deployment[Any]) -> None:
|
|
75
|
+
super().__init__(key=deployment.key, url=deployment.url)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass(frozen=True)
|
|
79
|
+
class _EndpointMarshalling:
|
|
80
|
+
signature_without_self: inspect.Signature
|
|
81
|
+
parameters: dict[str, TypeAdapter[Any]]
|
|
82
|
+
answer: TypeAdapter[Any]
|
|
83
|
+
|
|
84
|
+
@classmethod
|
|
85
|
+
def of(cls, method: Callable[..., Any]) -> _EndpointMarshalling:
|
|
86
|
+
signature = inspect.signature(method)
|
|
87
|
+
parameter_values = signature.parameters.values()
|
|
88
|
+
parameters_in_order = list(parameter_values)
|
|
89
|
+
signature_without_self = signature.replace(parameters=parameters_in_order[1:])
|
|
90
|
+
hints = get_type_hints(method)
|
|
91
|
+
answer_hint = hints.pop("return")
|
|
92
|
+
answer = TypeAdapter(answer_hint)
|
|
93
|
+
parameters = {name: TypeAdapter(hint) for name, hint in hints.items()}
|
|
94
|
+
return cls(signature_without_self, parameters, answer)
|
|
95
|
+
|
|
96
|
+
def arguments_to_json(self, *args: Any, **kwargs: Any) -> dict[str, Any]:
|
|
97
|
+
bound = self.signature_without_self.bind(*args, **kwargs)
|
|
98
|
+
return {
|
|
99
|
+
name: self.parameters[name].dump_python(value, mode="json")
|
|
100
|
+
for name, value in bound.arguments.items()
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
def arguments_from_json(self, body: dict[str, Any]) -> dict[str, Any]:
|
|
104
|
+
return {
|
|
105
|
+
name: parameter.validate_python(body[name])
|
|
106
|
+
for name, parameter in self.parameters.items()
|
|
107
|
+
if name in body
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
def answer_to_json(self, answer: Any) -> Any:
|
|
111
|
+
return self.answer.dump_python(answer, mode="json")
|
|
112
|
+
|
|
113
|
+
def answer_from_json(self, answered: Any) -> Any:
|
|
114
|
+
return self.answer.validate_python(answered)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _route_of(
|
|
118
|
+
method: Callable[..., Any], marshalling: _EndpointMarshalling
|
|
119
|
+
) -> Callable[..., Any]:
|
|
120
|
+
if inspect.iscoroutinefunction(method):
|
|
121
|
+
|
|
122
|
+
async def route(self: Any, body: dict[str, Any]) -> Any:
|
|
123
|
+
arguments = marshalling.arguments_from_json(body)
|
|
124
|
+
answer = await method(self, **arguments)
|
|
125
|
+
answered = marshalling.answer_to_json(answer)
|
|
126
|
+
return answered
|
|
127
|
+
|
|
128
|
+
else:
|
|
129
|
+
|
|
130
|
+
def route(self: Any, body: dict[str, Any]) -> Any:
|
|
131
|
+
arguments = marshalling.arguments_from_json(body)
|
|
132
|
+
answer = method(self, **arguments)
|
|
133
|
+
answered = marshalling.answer_to_json(answer)
|
|
134
|
+
return answered
|
|
135
|
+
|
|
136
|
+
route.__name__ = method.__name__
|
|
137
|
+
route.__qualname__ = method.__qualname__
|
|
138
|
+
return route
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _call_of(
|
|
142
|
+
name: str, method: Callable[..., Any], marshalling: _EndpointMarshalling
|
|
143
|
+
) -> Callable[..., Any]:
|
|
144
|
+
if inspect.iscoroutinefunction(method):
|
|
145
|
+
|
|
146
|
+
async def call(self: _EndpointsClient, *args: Any, **kwargs: Any) -> Any:
|
|
147
|
+
body = marshalling.arguments_to_json(*args, **kwargs)
|
|
148
|
+
async with httpx.AsyncClient(timeout=None) as client:
|
|
149
|
+
response = await client.post(f"{self.url}/{name}", json=body)
|
|
150
|
+
response.raise_for_status()
|
|
151
|
+
answered = response.json()
|
|
152
|
+
answer = marshalling.answer_from_json(answered)
|
|
153
|
+
return answer
|
|
154
|
+
|
|
155
|
+
else:
|
|
156
|
+
|
|
157
|
+
def call(self: _EndpointsClient, *args: Any, **kwargs: Any) -> Any:
|
|
158
|
+
body = marshalling.arguments_to_json(*args, **kwargs)
|
|
159
|
+
with httpx.Client(timeout=None) as client:
|
|
160
|
+
response = client.post(f"{self.url}/{name}", json=body)
|
|
161
|
+
response.raise_for_status()
|
|
162
|
+
answered = response.json()
|
|
163
|
+
answer = marshalling.answer_from_json(answered)
|
|
164
|
+
return answer
|
|
165
|
+
|
|
166
|
+
call.__name__ = name
|
|
167
|
+
return call
|
|
@@ -177,30 +177,20 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
|
|
|
177
177
|
|
|
178
178
|
#### Model registry and serving
|
|
179
179
|
|
|
180
|
-
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class
|
|
180
|
+
Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class marked with cortexgrid's `serve.ingress` (not Ray's), whose `serve.endpoint` methods it serves and whose client it generates:
|
|
181
181
|
|
|
182
182
|
```python
|
|
183
183
|
from cortexgrid import serve
|
|
184
|
-
from fastapi import FastAPI
|
|
185
184
|
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
class MyClient(cortexgrid.DeploymentClient):
|
|
189
|
-
def complete(self, prompt: str) -> str: ...
|
|
190
|
-
|
|
191
|
-
@serve.ingress(app)
|
|
185
|
+
@serve.ingress
|
|
192
186
|
class MyServeApp:
|
|
193
|
-
@classmethod
|
|
194
|
-
def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
|
|
195
|
-
return MyClient(key=deployment.key, url=deployment.url)
|
|
196
|
-
|
|
197
187
|
def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
|
|
198
188
|
self._weights_dir = cortexgrid.load_model(
|
|
199
189
|
deployment.family, deployment.suffix, deployment.run_name
|
|
200
190
|
)
|
|
201
191
|
|
|
202
|
-
@
|
|
203
|
-
async def complete(self,
|
|
192
|
+
@serve.endpoint
|
|
193
|
+
async def complete(self, prompt: str) -> str: ...
|
|
204
194
|
|
|
205
195
|
saved = cortexgrid.save_model(
|
|
206
196
|
weights_dir, MyServeApp, family="qwen", suffix="instruct",
|
|
@@ -208,7 +198,7 @@ saved = cortexgrid.save_model(
|
|
|
208
198
|
requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
|
|
209
199
|
)
|
|
210
200
|
deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
|
|
211
|
-
model = deployed.client() # MyServeApp's client, once the app serves
|
|
201
|
+
model: MyServeApp = deployed.client() # MyServeApp's client, once the app serves
|
|
212
202
|
```
|
|
213
203
|
|
|
214
204
|
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cortexgrid"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.17"
|
|
4
4
|
description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
|
|
5
5
|
readme = "docs/cortexgrid/README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -26,6 +26,8 @@ dependencies = [
|
|
|
26
26
|
"pydotenv>=0.0.7",
|
|
27
27
|
"pydantic>=2.13.3",
|
|
28
28
|
"requests>=2.31",
|
|
29
|
+
"fastapi>=0.115,<1",
|
|
30
|
+
"httpx>=0.27",
|
|
29
31
|
]
|
|
30
32
|
|
|
31
33
|
[project.urls]
|
|
@@ -1,43 +0,0 @@
|
|
|
1
|
-
"""Declare a serve-app's HTTP ingress without importing Ray.
|
|
2
|
-
|
|
3
|
-
from cortexgrid import serve
|
|
4
|
-
|
|
5
|
-
@serve.ingress(app)
|
|
6
|
-
class MyServeApp: ...
|
|
7
|
-
|
|
8
|
-
Same shape as `ray.serve.ingress`, but the class is left exactly as written: the
|
|
9
|
-
FastAPI app is only recorded on it, and `cortexgrid._serve_entry.build` applies
|
|
10
|
-
Ray's ingress when it builds the Serve application on the cluster.
|
|
11
|
-
|
|
12
|
-
Ray's decorator replaces the class with a wrapper subclass defined in
|
|
13
|
-
ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
|
|
14
|
-
that module. Everything that locates a serve-app by its module - bundling its
|
|
15
|
-
source, recording its import path - would then find Ray instead of the user's
|
|
16
|
-
code. Deferring the wrap to the one place Serve needs it keeps the class
|
|
17
|
-
locatable everywhere else (the laptop, Ray jobs, tests).
|
|
18
|
-
"""
|
|
19
|
-
|
|
20
|
-
from __future__ import annotations
|
|
21
|
-
|
|
22
|
-
from typing import Any, Callable, TypeVar
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
_T = TypeVar("_T", bound=type)
|
|
26
|
-
|
|
27
|
-
_INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def ingress(app: Any) -> Callable[[_T], _T]:
|
|
31
|
-
"""Mark a serve-app class as fronted by the ASGI `app` (e.g. a FastAPI
|
|
32
|
-
instance). Returns the class itself, unwrapped."""
|
|
33
|
-
|
|
34
|
-
def decorator(cls: _T) -> _T:
|
|
35
|
-
setattr(cls, _INGRESS_APP_ATTR, app)
|
|
36
|
-
return cls
|
|
37
|
-
|
|
38
|
-
return decorator
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
def ingress_app(cls: type) -> Any | None:
|
|
42
|
-
"""The app `cls` was marked with by `ingress`, or None if it was not."""
|
|
43
|
-
return getattr(cls, _INGRESS_APP_ATTR, None)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|