cortexgrid 0.3.16__tar.gz → 0.3.17__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/PKG-INFO +8 -16
  2. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/lifecycle.py +5 -5
  3. cortexgrid-0.3.17/cortexgrid/serve.py +167 -0
  4. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/docs/cortexgrid/README.md +5 -15
  5. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/pyproject.toml +3 -1
  6. cortexgrid-0.3.16/cortexgrid/serve.py +0 -43
  7. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/.gitignore +0 -0
  8. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/LICENSE +0 -0
  9. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/__init__.py +0 -0
  10. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_bundle.py +0 -0
  11. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_model_scheduler.py +0 -0
  12. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_ray_job_driver.py +0 -0
  13. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/_serve_entry.py +0 -0
  14. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/checkpoint.py +0 -0
  15. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/experiment.py +0 -0
  16. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/infra.py +0 -0
  17. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/jobs.py +0 -0
  18. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/mlflow_util.py +0 -0
  19. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/__init__.py +0 -0
  20. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/application_spec.py +0 -0
  21. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/deployment_key.py +0 -0
  22. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/deployment_records.py +0 -0
  23. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/placement.py +0 -0
  24. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/registry_tags.py +0 -0
  25. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/serve_bundle.py +0 -0
  26. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_serving/status.py +0 -0
  27. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/model_storage.py +0 -0
  28. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/py.typed +0 -0
  29. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/ray_util.py +0 -0
  30. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/s3_util.py +0 -0
  31. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/secrets.py +0 -0
  32. {cortexgrid-0.3.16 → cortexgrid-0.3.17}/cortexgrid/state.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.16
3
+ Version: 0.3.17
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -11,7 +11,9 @@ Requires-Python: <3.12,>=3.11
11
11
  Requires-Dist: boto3>=1.34
12
12
  Requires-Dist: cloudpickle>=3.0
13
13
  Requires-Dist: fabric>=3.2.3
14
+ Requires-Dist: fastapi<1,>=0.115
14
15
  Requires-Dist: haikunator>=2.1.0
16
+ Requires-Dist: httpx>=0.27
15
17
  Requires-Dist: mlflow<4,>=3.11
16
18
  Requires-Dist: packaging>=24
17
19
  Requires-Dist: pip>=23.0
@@ -205,30 +207,20 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
205
207
 
206
208
  #### Model registry and serving
207
209
 
208
- Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
210
+ Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class marked with cortexgrid's `serve.ingress` (not Ray's), whose `serve.endpoint` methods it serves and whose client it generates:
209
211
 
210
212
  ```python
211
213
  from cortexgrid import serve
212
- from fastapi import FastAPI
213
214
 
214
- app = FastAPI()
215
-
216
- class MyClient(cortexgrid.DeploymentClient):
217
- def complete(self, prompt: str) -> str: ...
218
-
219
- @serve.ingress(app)
215
+ @serve.ingress
220
216
  class MyServeApp:
221
- @classmethod
222
- def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
223
- return MyClient(key=deployment.key, url=deployment.url)
224
-
225
217
  def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
226
218
  self._weights_dir = cortexgrid.load_model(
227
219
  deployment.family, deployment.suffix, deployment.run_name
228
220
  )
229
221
 
230
- @app.post("/complete")
231
- async def complete(self, body: dict): ...
222
+ @serve.endpoint
223
+ async def complete(self, prompt: str) -> str: ...
232
224
 
233
225
  saved = cortexgrid.save_model(
234
226
  weights_dir, MyServeApp, family="qwen", suffix="instruct",
@@ -236,7 +228,7 @@ saved = cortexgrid.save_model(
236
228
  requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
237
229
  )
238
230
  deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
239
- model = deployed.client() # MyServeApp's client, once the app serves
231
+ model: MyServeApp = deployed.client() # MyServeApp's client, once the app serves
240
232
  ```
241
233
 
242
234
  The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
@@ -148,11 +148,11 @@ class DeploymentClient:
148
148
  return serving
149
149
 
150
150
 
151
- DeploymentClientT = TypeVar("DeploymentClientT", bound=DeploymentClient)
151
+ ServeAppT = TypeVar("ServeAppT")
152
152
 
153
153
 
154
154
  @dataclass
155
- class Deployment(Generic[DeploymentClientT]):
155
+ class Deployment(Generic[ServeAppT]):
156
156
  """A scheduled Ray Serve app fronting a model. `phase` is the normalized
157
157
  serving lifecycle phase (see `ServingStatus`); an app that appears in a
158
158
  listing always exists, so its phase is never "not_deployed"."""
@@ -166,16 +166,16 @@ class Deployment(Generic[DeploymentClientT]):
166
166
  experiment_name: str
167
167
  class_import_path: str
168
168
 
169
- def client(self) -> DeploymentClientT:
169
+ def client(self) -> ServeAppT:
170
170
  wait_for_model_serving(self.key)
171
171
  deployment_client = self.client_async()
172
172
  return deployment_client
173
173
 
174
- def client_async(self) -> DeploymentClientT:
174
+ def client_async(self) -> ServeAppT:
175
175
  module_name, class_name = self.class_import_path.split(":")
176
176
  serve_app_module = importlib.import_module(module_name)
177
177
  serve_app = getattr(serve_app_module, class_name)
178
- deployment_client: DeploymentClientT = serve_app.client(self)
178
+ deployment_client: ServeAppT = serve_app.client(self)
179
179
  return deployment_client
180
180
 
181
181
 
@@ -0,0 +1,167 @@
1
+ """Declare a serve-app's HTTP ingress without importing Ray.
2
+
3
+ from cortexgrid import serve
4
+
5
+ @serve.ingress
6
+ class MyServeApp:
7
+ @serve.endpoint
8
+ async def predict(self, xs: list[float]) -> list[float]: ...
9
+
10
+ Unlike `ray.serve.ingress`, it builds the FastAPI app from the class's
11
+ `serve.endpoint` methods, and the class is left unwrapped: the FastAPI app
12
+ and the client generated for it are only recorded on it, and
13
+ `cortexgrid._serve_entry.build` applies Ray's ingress when it builds the Serve
14
+ application on the cluster.
15
+
16
+ Ray's decorator replaces the class with a wrapper subclass defined in
17
+ ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
18
+ that module. Everything that locates a serve-app by its module - bundling its
19
+ source, recording its import path - would then find Ray instead of the user's
20
+ code. Deferring the wrap to the one place Serve needs it keeps the class
21
+ locatable everywhere else (the laptop, Ray jobs, tests).
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import inspect
27
+ from dataclasses import dataclass
28
+ from typing import Any, Callable, TypeVar, get_type_hints
29
+
30
+ import httpx
31
+ from fastapi import FastAPI
32
+ from pydantic import TypeAdapter
33
+
34
+ from cortexgrid.model_serving.lifecycle import Deployment, DeploymentClient
35
+
36
+
37
+ _T = TypeVar("_T", bound=type)
38
+ _Method = TypeVar("_Method", bound=Callable[..., Any])
39
+
40
+ _INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
41
+ _ENDPOINT_ATTR = "__cortexgrid_endpoint__"
42
+
43
+
44
+ def endpoint(method: _Method) -> _Method:
45
+ setattr(method, _ENDPOINT_ATTR, True)
46
+ return method
47
+
48
+
49
+ def ingress(cls: _T) -> _T:
50
+ """Mark a serve-app class as fronted by a FastAPI app serving its
51
+ endpoints, and give it the client that calls them. Returns the class
52
+ itself, unwrapped."""
53
+ app = FastAPI()
54
+ calls: dict[str, Any] = {"__module__": cls.__module__}
55
+ methods = inspect.getmembers(cls, inspect.isfunction)
56
+ for name, method in methods:
57
+ if getattr(method, _ENDPOINT_ATTR, False):
58
+ marshalling = _EndpointMarshalling.of(method)
59
+ route = _route_of(method, marshalling)
60
+ app.add_api_route(f"/{name}", route, methods=["POST"])
61
+ calls[name] = _call_of(name, method, marshalling)
62
+ client = type(f"{cls.__name__}Client", (_EndpointsClient,), calls)
63
+ setattr(cls, _INGRESS_APP_ATTR, app)
64
+ setattr(cls, "client", client)
65
+ return cls
66
+
67
+
68
+ def ingress_app(cls: type) -> Any | None:
69
+ """The app `cls` was marked with by `ingress`, or None if it was not."""
70
+ return getattr(cls, _INGRESS_APP_ATTR, None)
71
+
72
+
73
+ class _EndpointsClient(DeploymentClient):
74
+ def __init__(self, deployment: Deployment[Any]) -> None:
75
+ super().__init__(key=deployment.key, url=deployment.url)
76
+
77
+
78
+ @dataclass(frozen=True)
79
+ class _EndpointMarshalling:
80
+ signature_without_self: inspect.Signature
81
+ parameters: dict[str, TypeAdapter[Any]]
82
+ answer: TypeAdapter[Any]
83
+
84
+ @classmethod
85
+ def of(cls, method: Callable[..., Any]) -> _EndpointMarshalling:
86
+ signature = inspect.signature(method)
87
+ parameter_values = signature.parameters.values()
88
+ parameters_in_order = list(parameter_values)
89
+ signature_without_self = signature.replace(parameters=parameters_in_order[1:])
90
+ hints = get_type_hints(method)
91
+ answer_hint = hints.pop("return")
92
+ answer = TypeAdapter(answer_hint)
93
+ parameters = {name: TypeAdapter(hint) for name, hint in hints.items()}
94
+ return cls(signature_without_self, parameters, answer)
95
+
96
+ def arguments_to_json(self, *args: Any, **kwargs: Any) -> dict[str, Any]:
97
+ bound = self.signature_without_self.bind(*args, **kwargs)
98
+ return {
99
+ name: self.parameters[name].dump_python(value, mode="json")
100
+ for name, value in bound.arguments.items()
101
+ }
102
+
103
+ def arguments_from_json(self, body: dict[str, Any]) -> dict[str, Any]:
104
+ return {
105
+ name: parameter.validate_python(body[name])
106
+ for name, parameter in self.parameters.items()
107
+ if name in body
108
+ }
109
+
110
+ def answer_to_json(self, answer: Any) -> Any:
111
+ return self.answer.dump_python(answer, mode="json")
112
+
113
+ def answer_from_json(self, answered: Any) -> Any:
114
+ return self.answer.validate_python(answered)
115
+
116
+
117
+ def _route_of(
118
+ method: Callable[..., Any], marshalling: _EndpointMarshalling
119
+ ) -> Callable[..., Any]:
120
+ if inspect.iscoroutinefunction(method):
121
+
122
+ async def route(self: Any, body: dict[str, Any]) -> Any:
123
+ arguments = marshalling.arguments_from_json(body)
124
+ answer = await method(self, **arguments)
125
+ answered = marshalling.answer_to_json(answer)
126
+ return answered
127
+
128
+ else:
129
+
130
+ def route(self: Any, body: dict[str, Any]) -> Any:
131
+ arguments = marshalling.arguments_from_json(body)
132
+ answer = method(self, **arguments)
133
+ answered = marshalling.answer_to_json(answer)
134
+ return answered
135
+
136
+ route.__name__ = method.__name__
137
+ route.__qualname__ = method.__qualname__
138
+ return route
139
+
140
+
141
+ def _call_of(
142
+ name: str, method: Callable[..., Any], marshalling: _EndpointMarshalling
143
+ ) -> Callable[..., Any]:
144
+ if inspect.iscoroutinefunction(method):
145
+
146
+ async def call(self: _EndpointsClient, *args: Any, **kwargs: Any) -> Any:
147
+ body = marshalling.arguments_to_json(*args, **kwargs)
148
+ async with httpx.AsyncClient(timeout=None) as client:
149
+ response = await client.post(f"{self.url}/{name}", json=body)
150
+ response.raise_for_status()
151
+ answered = response.json()
152
+ answer = marshalling.answer_from_json(answered)
153
+ return answer
154
+
155
+ else:
156
+
157
+ def call(self: _EndpointsClient, *args: Any, **kwargs: Any) -> Any:
158
+ body = marshalling.arguments_to_json(*args, **kwargs)
159
+ with httpx.Client(timeout=None) as client:
160
+ response = client.post(f"{self.url}/{name}", json=body)
161
+ response.raise_for_status()
162
+ answered = response.json()
163
+ answer = marshalling.answer_from_json(answered)
164
+ return answer
165
+
166
+ call.__name__ = name
167
+ return call
@@ -177,30 +177,20 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
177
177
 
178
178
  #### Model registry and serving
179
179
 
180
- Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
180
+ Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class marked with cortexgrid's `serve.ingress` (not Ray's), whose `serve.endpoint` methods it serves and whose client it generates:
181
181
 
182
182
  ```python
183
183
  from cortexgrid import serve
184
- from fastapi import FastAPI
185
184
 
186
- app = FastAPI()
187
-
188
- class MyClient(cortexgrid.DeploymentClient):
189
- def complete(self, prompt: str) -> str: ...
190
-
191
- @serve.ingress(app)
185
+ @serve.ingress
192
186
  class MyServeApp:
193
- @classmethod
194
- def client(cls, deployment: cortexgrid.Deployment[MyClient]) -> MyClient:
195
- return MyClient(key=deployment.key, url=deployment.url)
196
-
197
187
  def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
198
188
  self._weights_dir = cortexgrid.load_model(
199
189
  deployment.family, deployment.suffix, deployment.run_name
200
190
  )
201
191
 
202
- @app.post("/complete")
203
- async def complete(self, body: dict): ...
192
+ @serve.endpoint
193
+ async def complete(self, prompt: str) -> str: ...
204
194
 
205
195
  saved = cortexgrid.save_model(
206
196
  weights_dir, MyServeApp, family="qwen", suffix="instruct",
@@ -208,7 +198,7 @@ saved = cortexgrid.save_model(
208
198
  requirements=cortexgrid.ModelRequirements(num_gpus=1, ram_gb=8, vram_gb=16),
209
199
  )
210
200
  deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name)
211
- model = deployed.client() # MyServeApp's client, once the app serves
201
+ model: MyServeApp = deployed.client() # MyServeApp's client, once the app serves
212
202
  ```
213
203
 
214
204
  The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.3.16"
3
+ version = "0.3.17"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
@@ -26,6 +26,8 @@ dependencies = [
26
26
  "pydotenv>=0.0.7",
27
27
  "pydantic>=2.13.3",
28
28
  "requests>=2.31",
29
+ "fastapi>=0.115,<1",
30
+ "httpx>=0.27",
29
31
  ]
30
32
 
31
33
  [project.urls]
@@ -1,43 +0,0 @@
1
- """Declare a serve-app's HTTP ingress without importing Ray.
2
-
3
- from cortexgrid import serve
4
-
5
- @serve.ingress(app)
6
- class MyServeApp: ...
7
-
8
- Same shape as `ray.serve.ingress`, but the class is left exactly as written: the
9
- FastAPI app is only recorded on it, and `cortexgrid._serve_entry.build` applies
10
- Ray's ingress when it builds the Serve application on the cluster.
11
-
12
- Ray's decorator replaces the class with a wrapper subclass defined in
13
- ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
14
- that module. Everything that locates a serve-app by its module - bundling its
15
- source, recording its import path - would then find Ray instead of the user's
16
- code. Deferring the wrap to the one place Serve needs it keeps the class
17
- locatable everywhere else (the laptop, Ray jobs, tests).
18
- """
19
-
20
- from __future__ import annotations
21
-
22
- from typing import Any, Callable, TypeVar
23
-
24
-
25
- _T = TypeVar("_T", bound=type)
26
-
27
- _INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
28
-
29
-
30
- def ingress(app: Any) -> Callable[[_T], _T]:
31
- """Mark a serve-app class as fronted by the ASGI `app` (e.g. a FastAPI
32
- instance). Returns the class itself, unwrapped."""
33
-
34
- def decorator(cls: _T) -> _T:
35
- setattr(cls, _INGRESS_APP_ATTR, app)
36
- return cls
37
-
38
- return decorator
39
-
40
-
41
- def ingress_app(cls: type) -> Any | None:
42
- """The app `cls` was marked with by `ingress`, or None if it was not."""
43
- return getattr(cls, _INGRESS_APP_ATTR, None)
File without changes
File without changes