cortexgrid 0.3.11__tar.gz → 0.3.13__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/PKG-INFO +7 -5
  2. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/__init__.py +8 -0
  3. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/_serve_entry.py +14 -5
  4. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/jobs.py +3 -3
  5. cortexgrid-0.3.13/cortexgrid/model_serving/__init__.py +73 -0
  6. cortexgrid-0.3.13/cortexgrid/model_serving/application_spec.py +70 -0
  7. cortexgrid-0.3.13/cortexgrid/model_serving/deployment_key.py +38 -0
  8. cortexgrid-0.3.13/cortexgrid/model_serving/deployment_records.py +46 -0
  9. cortexgrid-0.3.13/cortexgrid/model_serving/lifecycle.py +338 -0
  10. cortexgrid-0.3.13/cortexgrid/model_serving/placement.py +151 -0
  11. cortexgrid-0.3.13/cortexgrid/model_serving/registry_tags.py +97 -0
  12. cortexgrid-0.3.13/cortexgrid/model_serving/serve_bundle.py +143 -0
  13. cortexgrid-0.3.13/cortexgrid/model_serving/status.py +277 -0
  14. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/model_storage.py +18 -10
  15. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/state.py +8 -4
  16. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/docs/cortexgrid/README.md +6 -4
  17. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/pyproject.toml +8 -1
  18. cortexgrid-0.3.11/cortexgrid/model_serving.py +0 -953
  19. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/.gitignore +0 -0
  20. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/LICENSE +0 -0
  21. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/_bundle.py +0 -0
  22. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/_model_scheduler.py +0 -0
  23. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/_ray_job_driver.py +0 -0
  24. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/checkpoint.py +0 -0
  25. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/experiment.py +0 -0
  26. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/infra.py +0 -0
  27. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/mlflow_util.py +0 -0
  28. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/py.typed +0 -0
  29. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/ray_util.py +0 -0
  30. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/s3_util.py +0 -0
  31. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/secrets.py +0 -0
  32. {cortexgrid-0.3.11 → cortexgrid-0.3.13}/cortexgrid/serve.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cortexgrid
3
- Version: 0.3.11
3
+ Version: 0.3.13
4
4
  Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
5
5
  Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
6
6
  Project-URL: Repository, https://github.com/robodatalab/cortexgrid
@@ -215,8 +215,10 @@ app = FastAPI()
215
215
 
216
216
  @serve.ingress(app)
217
217
  class MyServeApp:
218
- def __init__(self, family: str, suffix: str, run_name: str) -> None:
219
- self._weights_dir = cortexgrid.load_model(family, suffix, run_name)
218
+ def __init__(self, deployment: cortexgrid.DeploymentKey) -> None:
219
+ self._weights_dir = cortexgrid.load_model(
220
+ deployment.family, deployment.suffix, deployment.run_name
221
+ )
220
222
 
221
223
  @app.post("/complete")
222
224
  async def complete(self, body: dict): ...
@@ -232,11 +234,11 @@ print(deployed.url)
232
234
 
233
235
  The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
234
236
 
235
- Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
237
+ Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(deployment)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. How one deployment serves the model - thinking on or off, say - is that deployment's own: `deploy_model(..., config={"thinking": "false"})` deploys the model once per distinct config, each deployment named by the `DeploymentKey` on the returned `Deployment`, and `model_config` lays the deployment's config over the model's. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
236
238
 
237
239
  `save_model` saves a new copy under every run - meant for weights the run produced (e.g. a fine-tune). For a model produced elsewhere (e.g. a pretrained base model), `cortexgrid.import_model(source, MyServeApp, family, suffix)` uploads it once under `run_name=cortexgrid.IMPORTED` and on later runs only re-bundles `MyServeApp` if its code changed; deploy it with `deploy_model(family, suffix, cortexgrid.IMPORTED)`. A model with no weights to stage - one behind a provider's API, e.g. Gemini or OpenAI - is registered the same way by `cortexgrid.register_model(MyServeApp, family, suffix, config=...)`: same key, same reuse, only the bundle stored. [Serving a hosted-API model](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md#serving-a-hosted-api-model) walks through one end to end - serve-app, API key, registration, deploy.
238
240
 
239
- `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(family, suffix, run_name, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
241
+ `save_model` is synchronous (registry lifecycle: `uploading` -> `ready`); `deploy_model` schedules the serving lifecycle (`deploying` -> `running`). With `wait=True` a failed deploy raises `cortexgrid.ModelDeployFailed`; `cortexgrid.wait_for_model_serving(deployment.key, timeout=...)` waits on a deploy started elsewhere, and re-deploying a failed model retries it from scratch. See [model-serving.md](https://github.com/robodatalab/cortexgrid/blob/main/docs/cortexgrid/model-serving.md) for both lifecycles end to end - upload/deploy/undeploy/delete, status queries (`model_registry_status`, `model_serving_status`), and error handling.
240
242
 
241
243
  ### API reference
242
244
 
@@ -101,12 +101,16 @@ from cortexgrid.model_storage import register_model as _register_model_storage
101
101
  from cortexgrid.model_storage import save_model as _save_model_storage
102
102
  from cortexgrid.model_serving import (
103
103
  Deployment,
104
+ DeploymentConfig,
105
+ DeploymentKey,
104
106
  ModelDeployFailed,
107
+ ModelNotDeployed,
105
108
  ModelRequirements,
106
109
  ServingStatus,
107
110
  deploy_model,
108
111
  list_deployed_models,
109
112
  model_serving_status,
113
+ redeploy_model,
110
114
  undeploy_model,
111
115
  wait_for_model_serving,
112
116
  )
@@ -308,9 +312,13 @@ __all__ = [
308
312
  "delete_model",
309
313
  # Model serving
310
314
  "Deployment",
315
+ "DeploymentConfig",
316
+ "DeploymentKey",
311
317
  "ModelDeployFailed",
318
+ "ModelNotDeployed",
312
319
  "ServingStatus",
313
320
  "deploy_model",
321
+ "redeploy_model",
314
322
  "wait_for_model_serving",
315
323
  "model_serving_status",
316
324
  "undeploy_model",
@@ -6,13 +6,14 @@ On the cluster replica, `build` imports the serve-app class bundled at
6
6
  ingress with the app it was marked with by `cortexgrid.serve.ingress` (again on
7
7
  each replica, see `_IngressOnReplica`), wraps it as a Ray Serve deployment
8
8
  with the replica count and Ray resource requests `deploy_model` derived from
9
- the model's requirements, and binds it with the (family, suffix, run_name)
10
- identifiers.
9
+ the model's requirements, and binds it with its deployment's
10
+ `cortexgrid.DeploymentKey`.
11
11
 
12
12
  The serve-app owns everything about traffic: its own routes, request schemas,
13
13
  streaming, and timeouts. cortexgrid does not interpose a request/response
14
- contract - it only schedules the app and hands it the identifiers it needs to
15
- fetch its own weights via `cortexgrid.load_model`.
14
+ contract - it only schedules the app and hands it the key it needs to fetch
15
+ its own weights via `cortexgrid.load_model` and its settings via
16
+ `cortexgrid.model_config`.
16
17
 
17
18
  The serve-app declares no resources: the hardware a replica needs belongs to
18
19
  the model and is stored in the registry (`cortexgrid.ModelRequirements`), and
@@ -28,6 +29,7 @@ from ray import serve
28
29
  from ray.serve.deployment import Application
29
30
 
30
31
  from cortexgrid._model_scheduler import model_autoscaling_config
32
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
31
33
  from cortexgrid.serve import ingress_app
32
34
 
33
35
 
@@ -80,4 +82,11 @@ def build(args: dict[str, Any]) -> Application:
80
82
  # had on Ray 2.9.
81
83
  max_ongoing_requests=_MAX_ONGOING_REQUESTS,
82
84
  ray_actor_options=args.get("ray_actor_options", {}),
83
- ).bind(args["family"], args["suffix"], args["run_name"])
85
+ ).bind(
86
+ DeploymentKey(
87
+ args["family"],
88
+ args["suffix"],
89
+ args["run_name"],
90
+ args.get("config_fingerprint", ""),
91
+ )
92
+ )
@@ -29,7 +29,7 @@ from pydantic import BaseModel, ConfigDict
29
29
  log = logging.getLogger(__name__)
30
30
 
31
31
  _JOB_POLL_INTERVAL_S = 5.0
32
- _TERMINAL_JOB_STATES = (JobStatus.FINISHED, JobStatus.FAILED, JobStatus.STOPPED)
32
+ TERMINAL_JOB_STATES = (JobStatus.FINISHED, JobStatus.FAILED, JobStatus.STOPPED)
33
33
 
34
34
 
35
35
  class JobFailed(RuntimeError):
@@ -379,7 +379,7 @@ def wait_for_job_result(run_id: str, job_id: str, timeout: float | None = None)
379
379
  )
380
380
  ray_job_id = lifecycle.get_ray_job_id()
381
381
  status = get_ray_job_status(ray_job_id)
382
- if status in _TERMINAL_JOB_STATES:
382
+ if status in TERMINAL_JOB_STATES:
383
383
  try:
384
384
  result = JobResult.load_from_mlflow(run_id, job_id)
385
385
  except FileNotFoundError:
@@ -413,7 +413,7 @@ class JobFuture:
413
413
 
414
414
  def done(self) -> bool:
415
415
  """True once Ray reports the job finished, failed or stopped."""
416
- return self.status() in _TERMINAL_JOB_STATES
416
+ return self.status() in TERMINAL_JOB_STATES
417
417
 
418
418
  def result(self, timeout: float | None = None) -> Any:
419
419
  """Block until the job finishes and return its function's value.
@@ -0,0 +1,73 @@
1
+ """Cortexgrid wrappers around Ray Serve.
2
+
3
+ Caller stays HTTP-only: deploy/undeploy/list talk to the Ray dashboard's
4
+ declarative `/api/serve/applications/` endpoint via [cortexgrid.ray_util],
5
+ never `ray.init`. The deployment class is bundled at `save_model` time, zipped,
6
+ uploaded to MinIO under
7
+ `serve-bundles/<run_name>/<family>__<suffix>/<fingerprint>.zip`, and
8
+ referenced via `runtime_env.working_dir` so Ray workers fetch it from there.
9
+ The bundle URL, class import path, pip list, and fingerprint are persisted as
10
+ tags on the model's registry entry so `deploy_model` can find them later
11
+ without the caller holding the class object. So are the model's
12
+ `ModelRequirements`, which `deploy_model` turns into the replica's Ray resource
13
+ requests.
14
+
15
+ Every deployment `deploy_model` puts on Ray Serve gets a record with the jobs
16
+ control plane, keyed by its `DeploymentKey` - the model's (family, suffix,
17
+ run_name) plus a fingerprint of the config it was deployed with: that config,
18
+ the spec it PUT, plus the phase, message and replica
19
+ placements the control plane last observed (`observe_deployments`, run on
20
+ every poll cycle). Listings and status reads come from those records; waits
21
+ ask the Serve controller directly.
22
+
23
+ Naming: the Ray Serve application is named "<family>__<suffix>__<run_name>",
24
+ followed by "__<config_fingerprint>" for a deployment given a config.
25
+ This relies on family/suffix/run_name not containing the literal "__".
26
+
27
+ See [docs/cortexgrid/model-serving.md](../../docs/cortexgrid/model-serving.md)
28
+ for the end-to-end design.
29
+ """
30
+
31
+ from cortexgrid.model_serving.application_spec import app_name
32
+ from cortexgrid.model_serving.deployment_key import (
33
+ DeploymentConfig,
34
+ DeploymentKey,
35
+ deployment_key,
36
+ )
37
+ from cortexgrid.model_serving.lifecycle import (
38
+ ModelDeployFailed,
39
+ ModelNotDeployed,
40
+ deploy_model,
41
+ redeploy_model,
42
+ undeploy_model,
43
+ wait_for_model_serving,
44
+ )
45
+ from cortexgrid.model_serving.placement import ModelRequirements, vram_tiers
46
+ from cortexgrid.model_serving.registry_tags import (
47
+ bundle_fingerprint_from_tags,
48
+ has_requirement_tags,
49
+ metadata_from_tags,
50
+ metadata_to_tags,
51
+ requirements_from_tags,
52
+ requirements_to_tags,
53
+ )
54
+ from cortexgrid.model_serving.serve_bundle import (
55
+ BundleMetadata,
56
+ ServeBundle,
57
+ build_bundle,
58
+ bundle_class,
59
+ bundle_fingerprint_from_url,
60
+ upload_bundle,
61
+ )
62
+ from cortexgrid.model_serving.status import (
63
+ Deployment,
64
+ ReplicaPlacement,
65
+ ServingMessage,
66
+ ServingStatus,
67
+ deployment_config,
68
+ list_deployed_models,
69
+ model_replica_placements,
70
+ model_serving_messages,
71
+ model_serving_status,
72
+ observe_deployments,
73
+ )
@@ -0,0 +1,70 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
6
+ from cortexgrid.model_serving.placement import ModelRequirements, ray_actor_options
7
+ from cortexgrid.model_serving.serve_bundle import (
8
+ BundleMetadata,
9
+ bundle_fingerprint_from_url,
10
+ )
11
+
12
+
13
+ def app_name(key: DeploymentKey) -> str:
14
+ return "__".join(_name_segments(key))
15
+
16
+
17
+ def route_prefix(key: DeploymentKey) -> str:
18
+ return "/r/" + "/".join(_name_segments(key))
19
+
20
+
21
+ def _name_segments(key: DeploymentKey) -> list[str]:
22
+ model_segments = [key.family, key.suffix, key.run_name]
23
+ if not key.config_fingerprint:
24
+ return model_segments
25
+ return [*model_segments, key.config_fingerprint]
26
+
27
+
28
+ def build_application_spec(
29
+ key: DeploymentKey,
30
+ meta: BundleMetadata,
31
+ requirements: ModelRequirements,
32
+ num_replicas: int,
33
+ tiers: list[int],
34
+ ) -> dict[str, Any]:
35
+ """Assemble a Ray Serve application schema from pre-bundled metadata, the
36
+ model's requirements, and the cluster's GPU size classes."""
37
+ # working_dir carries the serve-app's own source; Ray pip-installs the
38
+ # third-party distributions the image lacks into a per-node cached
39
+ # virtualenv layered on the image. No pip key when there are none, so Ray
40
+ # builds no virtualenv.
41
+ runtime_env: dict[str, Any] = {"working_dir": meta.bundle_url}
42
+ if meta.pip_requirements:
43
+ runtime_env["pip"] = meta.pip_requirements
44
+ return {
45
+ "name": app_name(key),
46
+ "route_prefix": route_prefix(key),
47
+ # Ray Serve REST requires import_path to point at an Application builder
48
+ # (callable returning a bound node) or an already-bound node. A bare
49
+ # Deployment class is rejected, so cortexgrid.deploy_model goes through
50
+ # a generic builder that re-imports the user's class and binds it.
51
+ "import_path": "cortexgrid._serve_entry:build",
52
+ "args": {
53
+ "class_import_path": meta.class_import_path,
54
+ "family": key.family,
55
+ "suffix": key.suffix,
56
+ "run_name": key.run_name,
57
+ "config_fingerprint": key.config_fingerprint,
58
+ "num_replicas": num_replicas,
59
+ "ray_actor_options": ray_actor_options(requirements, tiers),
60
+ },
61
+ "runtime_env": runtime_env,
62
+ }
63
+
64
+
65
+ def bundle_fingerprint_in_spec(spec: dict[str, Any]) -> str:
66
+ return bundle_fingerprint_from_url(spec["runtime_env"]["working_dir"])
67
+
68
+
69
+ def replica_count_in_spec(spec: dict[str, Any]) -> int:
70
+ return spec["args"]["num_replicas"]
@@ -0,0 +1,38 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from dataclasses import dataclass
6
+
7
+
8
+ DeploymentConfig = dict[str, str]
9
+
10
+ _CONFIG_FINGERPRINT_LENGTH = 12
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class DeploymentKey:
15
+ family: str
16
+ suffix: str
17
+ run_name: str
18
+ config_fingerprint: str = ""
19
+
20
+
21
+ def deployment_key(
22
+ family: str, suffix: str, run_name: str, config: DeploymentConfig
23
+ ) -> DeploymentKey:
24
+ return DeploymentKey(family, suffix, run_name, _config_fingerprint(config))
25
+
26
+
27
+ def _config_fingerprint(config: DeploymentConfig) -> str:
28
+ for name, value in config.items():
29
+ if not isinstance(name, str) or not isinstance(value, str):
30
+ raise ValueError(
31
+ f"Deployment config must map strings to strings: {name!r}: {value!r}"
32
+ )
33
+ if not config:
34
+ return ""
35
+ canonical_config = json.dumps(config, sort_keys=True)
36
+ return hashlib.sha256(canonical_config.encode()).hexdigest()[
37
+ :_CONFIG_FINGERPRINT_LENGTH
38
+ ]
@@ -0,0 +1,46 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ from cortexgrid import state
6
+ from cortexgrid.model_serving.deployment_key import DeploymentKey
7
+
8
+
9
+ DeploymentRecord = dict[str, Any]
10
+
11
+
12
+ def get_deployment_record(key: DeploymentKey) -> DeploymentRecord | None:
13
+ return state.get(*_path(key), params=_params(key))
14
+
15
+
16
+ def put_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> None:
17
+ state.put(*_path(key), body=body, params=_params(key))
18
+
19
+
20
+ def patch_deployment_record(key: DeploymentKey, body: DeploymentRecord) -> bool:
21
+ return state.patch(*_path(key), body=body, params=_params(key))
22
+
23
+
24
+ def delete_deployment_record(key: DeploymentKey) -> None:
25
+ state.delete(*_path(key), params=_params(key))
26
+
27
+
28
+ def list_deployment_records() -> list[DeploymentRecord]:
29
+ return state.get("deployments")
30
+
31
+
32
+ def key_of_record(record: DeploymentRecord) -> DeploymentKey:
33
+ return DeploymentKey(
34
+ record["family"],
35
+ record["suffix"],
36
+ record["run_name"],
37
+ record["config_fingerprint"],
38
+ )
39
+
40
+
41
+ def _path(key: DeploymentKey) -> tuple[str, ...]:
42
+ return ("deployments", key.family, key.suffix, key.run_name)
43
+
44
+
45
+ def _params(key: DeploymentKey) -> dict[str, str]:
46
+ return {"config_fingerprint": key.config_fingerprint}