cortexgrid 0.2.86__tar.gz → 0.2.88__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.2.86
3
+ Version: 0.2.88
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -155,9 +155,24 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
155
155
 
156
156
  #### Model registry and serving
157
157
 
158
- Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application:
158
+ Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
159
159
 
160
160
  ```python
161
+ from cortexgrid import serve
162
+ from fastapi import FastAPI
163
+
164
+ app = FastAPI()
165
+
166
+ @serve.ingress(app)
167
+ class MyServeApp:
168
+ num_gpus = 1
169
+
170
+ def __init__(self, family: str, suffix: str, run_name: str) -> None:
171
+ self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
172
+
173
+ @app.post("/complete")
174
+ async def complete(self, body: dict): ...
175
+
161
176
  saved = cortexgrid.save_model(weights_dir, MyServeApp, family="qwen", suffix="instruct")
162
177
  deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True)
163
178
  print(deployed.url)
@@ -100,7 +100,10 @@ def bundle(seed: Path) -> BundleDesc:
100
100
 
101
101
  def stage(files: set[Path], dest: Path) -> None:
102
102
  """Copy `files` under `dest`, each at its import path (relative to the
103
- sys.path entry it lives under), so `dest` on sys.path imports them all."""
103
+ sys.path entry it lives under), so `dest` on sys.path imports them all.
104
+ `dest` is created even when `files` is empty (e.g. code that lives entirely
105
+ in installed distributions)."""
106
+ dest.mkdir(parents=True, exist_ok=True)
104
107
  for file in files:
105
108
  target = dest / file.relative_to(_sys_path_root(file))
106
109
  target.parent.mkdir(parents=True, exist_ok=True)
@@ -2,7 +2,8 @@
2
2
 
3
3
  Ray Serve's REST `import_path` resolves to `cortexgrid._serve_entry:build`.
4
4
  On the cluster replica, `build` imports the serve-app class bundled at
5
- `save_model` time (its import path was stored as an MLflow tag), reads its
5
+ `save_model` time (its import path was stored as an MLflow tag), applies Ray's
6
+ ingress with the app it was marked with by `cortexgrid.serve.ingress`, reads its
6
7
  `num_gpus`/`num_replicas` class attributes for actor placement, wraps it as a
7
8
  Ray Serve deployment, and binds it with the (family, suffix, run_name)
8
9
  identifiers.
@@ -27,10 +28,17 @@ from typing import Any
27
28
  from ray import serve
28
29
  from ray.serve.deployment import Application
29
30
 
31
+ from cortexgrid.serve import ingress_app
32
+
30
33
 
31
34
  def build(args: dict[str, Any]) -> Application:
32
35
  module_name, class_name = args["class_import_path"].split(":")
33
36
  serve_app = getattr(importlib.import_module(module_name), class_name)
37
+ # Models saved with a class wrapped by ray.serve.ingress itself carry no
38
+ # mark and are deployed as they are.
39
+ app = ingress_app(serve_app)
40
+ if app is not None:
41
+ serve_app = serve.ingress(app)(serve_app)
34
42
  num_gpus = getattr(serve_app, "num_gpus", 0)
35
43
  num_replicas = getattr(serve_app, "num_replicas", 1)
36
44
  return serve.deployment(serve_app).options(
@@ -104,7 +104,18 @@ def bundle_class(
104
104
 
105
105
  Returns the metadata `deploy_model` needs later; callers (typically
106
106
  `save_model`) persist it on the ModelVersion so the deploy step can run
107
- without holding the class object."""
107
+ without holding the class object.
108
+
109
+ Raises ValueError for a class wrapped by `ray.serve.ingress`: that wrapper
110
+ is a subclass Ray defines in its own module, and on older Ray (e.g. 2.9) it
111
+ reports that module as its own, so the class's source and import path would
112
+ resolve to Ray instead of the serve-app. `cortexgrid.serve.ingress` leaves
113
+ the class unwrapped."""
114
+ if any(klass.__module__.startswith("ray.serve") for klass in cls.__mro__):
115
+ raise ValueError(
116
+ f"{cls.__name__} is wrapped by ray.serve.ingress; decorate it with "
117
+ "cortexgrid.serve.ingress instead (from cortexgrid import serve)"
118
+ )
108
119
  entry_file = Path(inspect.getfile(cls)).resolve()
109
120
  serve_entry = Path(__file__).with_name("_serve_entry.py")
110
121
  desc = bundle(entry_file).merge(bundle(serve_entry))
@@ -120,10 +131,13 @@ def bundle_class(
120
131
  len(desc.local_files),
121
132
  pip_requirements,
122
133
  )
123
- zip_base = Path(tmp) / f"{family}__{suffix}"
124
- shutil.make_archive(str(zip_base), "zip", root_dir=str(code_root))
134
+ # make_archive returns the path it wrote. Rebuilding it with
135
+ # Path.with_suffix would cut dotted names ("Qwen2.5-0.5B" -> "Qwen2.zip").
136
+ archive = shutil.make_archive(
137
+ str(Path(tmp) / f"{family}__{suffix}"), "zip", root_dir=str(code_root)
138
+ )
125
139
  bundle_url = upload(
126
- str(zip_base.with_suffix(".zip")),
140
+ archive,
127
141
  dest_path=f"serve-bundles/{run_name}/{family}__{suffix}.zip",
128
142
  )
129
143
  return BundleMetadata(
@@ -143,7 +143,8 @@ def save_model(
143
143
  the caller's concern. That directory boundary is the open-closed extension
144
144
  point - new model kinds need no change here.
145
145
 
146
- `serve_app` is the Ray Serve ingress class that will front these weights.
146
+ `serve_app` is the `cortexgrid.serve.ingress` class that will front these
147
+ weights.
147
148
  Its code is bundled and its import path, bundle URL, and pip list are
148
149
  stored as tags on the ModelVersion so `deploy_model` can bind it later
149
150
  without the caller holding the class object."""
@@ -0,0 +1,43 @@
1
+ """Declare a serve-app's HTTP ingress without importing Ray.
2
+
3
+ from cortexgrid import serve
4
+
5
+ @serve.ingress(app)
6
+ class MyServeApp: ...
7
+
8
+ Same shape as `ray.serve.ingress`, but the class is left exactly as written: the
9
+ FastAPI app is only recorded on it, and `cortexgrid._serve_entry.build` applies
10
+ Ray's ingress when it builds the Serve application on the cluster.
11
+
12
+ Ray's decorator replaces the class with a wrapper subclass defined in
13
+ ray/serve/api.py; older Ray (e.g. 2.9) leaves the wrapper's __module__ naming
14
+ that module. Everything that locates a serve-app by its module - bundling its
15
+ source, recording its import path - would then find Ray instead of the user's
16
+ code. Deferring the wrap to the one place Serve needs it keeps the class
17
+ locatable everywhere else (the laptop, Ray jobs, tests).
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from typing import Any, Callable, TypeVar
23
+
24
+
25
+ _T = TypeVar("_T", bound=type)
26
+
27
+ _INGRESS_APP_ATTR = "__cortexgrid_ingress_app__"
28
+
29
+
30
+ def ingress(app: Any) -> Callable[[_T], _T]:
31
+ """Mark a serve-app class as fronted by the ASGI `app` (e.g. a FastAPI
32
+ instance). Returns the class itself, unwrapped."""
33
+
34
+ def decorator(cls: _T) -> _T:
35
+ setattr(cls, _INGRESS_APP_ATTR, app)
36
+ return cls
37
+
38
+ return decorator
39
+
40
+
41
+ def ingress_app(cls: type) -> Any | None:
42
+ """The app `cls` was marked with by `ingress`, or None if it was not."""
43
+ return getattr(cls, _INGRESS_APP_ATTR, None)
@@ -127,9 +127,24 @@ s3_client = cortexgrid.get_s3_client() # boto3 S3 client
127
127
 
128
128
  #### Model registry and serving
129
129
 
130
- Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application:
130
+ Save a trained model's weights together with the serve-app that fronts it, then deploy it as a Ray Serve application. A serve-app is a class fronted by a FastAPI app, marked with cortexgrid's `serve.ingress` (not Ray's):
131
131
 
132
132
  ```python
133
+ from cortexgrid import serve
134
+ from fastapi import FastAPI
135
+
136
+ app = FastAPI()
137
+
138
+ @serve.ingress(app)
139
+ class MyServeApp:
140
+ num_gpus = 1
141
+
142
+ def __init__(self, family: str, suffix: str, run_name: str) -> None:
143
+ self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
144
+
145
+ @app.post("/complete")
146
+ async def complete(self, body: dict): ...
147
+
133
148
  saved = cortexgrid.save_model(weights_dir, MyServeApp, family="qwen", suffix="instruct")
134
149
  deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True)
135
150
  print(deployed.url)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cortexgrid"
3
- version = "0.2.86"
3
+ version = "0.2.88"
4
4
  description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
5
5
  readme = "docs/cortexgrid/README.md"
6
6
  license = "Apache-2.0"
File without changes
File without changes