cortexgrid 0.3.3__tar.gz → 0.3.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/PKG-INFO +3 -3
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/infra.py +9 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/model_serving.py +187 -15
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/ray_util.py +17 -1
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/docs/cortexgrid/README.md +1 -1
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/pyproject.toml +5 -2
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/.gitignore +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/LICENSE +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/__init__.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/_bundle.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/_ray_job_driver.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/_serve_entry.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/checkpoint.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/experiment.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/jobs.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/mlflow_util.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/model_storage.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/py.typed +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/s3_util.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/secrets.py +0 -0
- {cortexgrid-0.3.3 → cortexgrid-0.3.5}/cortexgrid/serve.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cortexgrid
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.5
|
|
4
4
|
Summary: Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3
|
|
5
5
|
Project-URL: Homepage, https://github.com/robodatalab/cortexgrid
|
|
6
6
|
Project-URL: Repository, https://github.com/robodatalab/cortexgrid
|
|
@@ -20,7 +20,7 @@ Requires-Dist: pydantic>=2.13.3
|
|
|
20
20
|
Requires-Dist: pydotenv>=0.0.7
|
|
21
21
|
Requires-Dist: python-dotenv>=1.2.2
|
|
22
22
|
Requires-Dist: pyyaml>=6.0.3
|
|
23
|
-
Requires-Dist: ray[default]<3,>=2.
|
|
23
|
+
Requires-Dist: ray[default]<3,>=2.58
|
|
24
24
|
Requires-Dist: requests>=2.31
|
|
25
25
|
Requires-Dist: setuptools>=82.0.1
|
|
26
26
|
Requires-Dist: tqdm>=4.60
|
|
@@ -230,7 +230,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
230
230
|
print(deployed.url)
|
|
231
231
|
```
|
|
232
232
|
|
|
233
|
-
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
233
|
+
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
234
234
|
|
|
235
235
|
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
236
236
|
|
|
@@ -20,6 +20,15 @@ def get_ray_serve_applications_uri() -> str:
|
|
|
20
20
|
return f"{get_ray_job_server_uri()}/api/serve/applications/"
|
|
21
21
|
|
|
22
22
|
|
|
23
|
+
def get_ray_nodes_uri() -> str:
|
|
24
|
+
"""Dashboard state-API endpoint listing the cluster's nodes.
|
|
25
|
+
|
|
26
|
+
Unlike the older `/nodes` dashboard route, this one reports each node's
|
|
27
|
+
labels and totals in snake_case, exactly as the raylet holds them.
|
|
28
|
+
"""
|
|
29
|
+
return f"{get_ray_job_server_uri()}/api/v0/nodes"
|
|
30
|
+
|
|
31
|
+
|
|
23
32
|
def get_s3_endpoint_url() -> str:
|
|
24
33
|
# AWS profile: regional s3.amazonaws.com URL (stored as "" = no override).
|
|
25
34
|
# On-prem: tailnet-reachable MinIO NodePort URL.
|
|
@@ -38,6 +38,7 @@ from ray.serve.schema import ApplicationStatus
|
|
|
38
38
|
from cortexgrid._bundle import bundle, digest, stage, worker_provides
|
|
39
39
|
from cortexgrid.infra import get_mlflow_tracking_uri, get_ray_serve_uri
|
|
40
40
|
from cortexgrid.ray_util import (
|
|
41
|
+
get_ray_nodes,
|
|
41
42
|
get_serve_details,
|
|
42
43
|
put_serve_applications,
|
|
43
44
|
)
|
|
@@ -218,10 +219,101 @@ _MIB_PER_GIB = 1024
|
|
|
218
219
|
_GIB = 1024**3
|
|
219
220
|
|
|
220
221
|
|
|
221
|
-
|
|
222
|
+
# Node label each GPU worker sets to the same MiB it advertises as the
|
|
223
|
+
# `_VRAM_RESOURCE` (see the ray-worker DaemonSet). The resource reserves VRAM;
|
|
224
|
+
# the label names the size class of the node's GPUs, which is what lets a
|
|
225
|
+
# replica ask for the smallest card that fits. Nodes with no GPU carry neither.
|
|
226
|
+
_VRAM_LABEL = "vram_mib"
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def vram_tiers() -> list[int]:
|
|
230
|
+
"""The distinct GPU sizes in the cluster, in MiB, smallest first.
|
|
231
|
+
|
|
232
|
+
One entry per size class, not per node: two 12 GiB workers are one tier.
|
|
233
|
+
Nodes that are not ALIVE, and nodes with no `vram_mib` label (CPU workers,
|
|
234
|
+
the head), contribute none - so a cluster with no GPUs reports no tiers.
|
|
235
|
+
"""
|
|
236
|
+
tiers = set()
|
|
237
|
+
for node in get_ray_nodes():
|
|
238
|
+
if node.get("state") != "ALIVE":
|
|
239
|
+
continue
|
|
240
|
+
label = (node.get("labels") or {}).get(_VRAM_LABEL)
|
|
241
|
+
# A worker that could not size its GPUs never starts, so a malformed
|
|
242
|
+
# label means someone set it by hand; skip it rather than fail every
|
|
243
|
+
# deploy in the cluster.
|
|
244
|
+
if label is not None and label.isdigit():
|
|
245
|
+
tiers.add(int(label))
|
|
246
|
+
return sorted(tiers)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _placement_preferences(needed_mib: int, tiers: list[int]) -> list[dict[str, Any]]:
|
|
250
|
+
"""The size classes a replica needing `needed_mib` should be offered, best
|
|
251
|
+
first: every tier large enough, smallest first, then a catch-all for any
|
|
252
|
+
tier that is not too small.
|
|
253
|
+
|
|
254
|
+
The catch-all is what keeps this from going stale. It excludes the sizes
|
|
255
|
+
known to be too small rather than naming the ones that fit, so a larger
|
|
256
|
+
GPU joining the cluster after this deploy is still placeable without a
|
|
257
|
+
redeploy. It is dropped when no known tier is too small, since there is
|
|
258
|
+
then nothing left for it to say.
|
|
259
|
+
|
|
260
|
+
Excluding rather than naming also matches a node carrying no `vram_mib`
|
|
261
|
+
label at all. That is harmless: a model reaching here has VRAM to reserve,
|
|
262
|
+
so it also requests `num_gpus` and `_VRAM_RESOURCE`, neither of which a
|
|
263
|
+
CPU-only node has. The resource request, not the selector, is what keeps
|
|
264
|
+
a GPU model off a CPU node.
|
|
265
|
+
"""
|
|
266
|
+
# Sorted here rather than trusted from the caller: the whole contract is
|
|
267
|
+
# "smallest first", and it must not rest on how the tiers arrived.
|
|
268
|
+
ordered = sorted(tiers)
|
|
269
|
+
fits = [tier for tier in ordered if tier >= needed_mib]
|
|
270
|
+
too_small = [str(tier) for tier in ordered if tier < needed_mib]
|
|
271
|
+
preferences = [{_VRAM_LABEL: str(tier)} for tier in fits]
|
|
272
|
+
if too_small:
|
|
273
|
+
preferences.append({_VRAM_LABEL: f"!in({', '.join(too_small)})"})
|
|
274
|
+
return preferences
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _placement_options(
|
|
278
|
+
requirements: ModelRequirements, tiers: list[int]
|
|
279
|
+
) -> dict[str, Any]:
|
|
280
|
+
"""Ask Ray for the smallest GPU that fits, falling back to larger ones.
|
|
281
|
+
|
|
282
|
+
`label_selector` names the smallest size class the model fits on, and
|
|
283
|
+
`fallback_strategy` the larger ones in order, so Ray reaches for a bigger
|
|
284
|
+
card only once every smaller one is out of VRAM. That a fallback fires on
|
|
285
|
+
exhaustion, and not merely on a size class being absent, is what makes
|
|
286
|
+
this "smallest that is free" rather than "smallest that exists"; checked
|
|
287
|
+
against Ray 2.58, which is the floor this package pins for it.
|
|
288
|
+
|
|
289
|
+
These only order the candidates. The reservation is still the `vram_mib`
|
|
290
|
+
resource, so two replicas can no more share a card's memory than before,
|
|
291
|
+
and a selector matching nothing leaves the replica pending exactly as an
|
|
292
|
+
unsatisfiable resource request does.
|
|
293
|
+
|
|
294
|
+
A model with no VRAM requirement gets no selector at all, so it stays
|
|
295
|
+
placeable on a CPU-only node - and so does every model when the cluster
|
|
296
|
+
reports no GPU sizes, which is the pre-label behaviour.
|
|
297
|
+
"""
|
|
298
|
+
if requirements.vram_gb <= 0 or not tiers:
|
|
299
|
+
return {}
|
|
300
|
+
preferences = _placement_preferences(
|
|
301
|
+
round(requirements.vram_gb * _MIB_PER_GIB), tiers
|
|
302
|
+
)
|
|
303
|
+
first, *rest = preferences
|
|
304
|
+
options: dict[str, Any] = {"label_selector": first}
|
|
305
|
+
if rest:
|
|
306
|
+
options["fallback_strategy"] = [{"label_selector": r} for r in rest]
|
|
307
|
+
return options
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _ray_actor_options(
|
|
311
|
+
requirements: ModelRequirements, tiers: list[int]
|
|
312
|
+
) -> dict[str, Any]:
|
|
222
313
|
"""Translate ModelRequirements into a replica's Ray actor resource requests.
|
|
223
314
|
Ray places the replica only on a node with that much free and reserves it
|
|
224
|
-
there; a zero requirement requests nothing.
|
|
315
|
+
there; a zero requirement requests nothing. `tiers` are the cluster's GPU
|
|
316
|
+
size classes, which decide which node Ray prefers among those that fit."""
|
|
225
317
|
options: dict[str, Any] = {"num_gpus": requirements.num_gpus}
|
|
226
318
|
if requirements.ram_gb > 0:
|
|
227
319
|
options["memory"] = int(requirements.ram_gb * _GIB)
|
|
@@ -229,6 +321,7 @@ def _ray_actor_options(requirements: ModelRequirements) -> dict[str, Any]:
|
|
|
229
321
|
options["resources"] = {
|
|
230
322
|
_VRAM_RESOURCE: round(requirements.vram_gb * _MIB_PER_GIB)
|
|
231
323
|
}
|
|
324
|
+
options.update(_placement_options(requirements, tiers))
|
|
232
325
|
return options
|
|
233
326
|
|
|
234
327
|
|
|
@@ -239,9 +332,10 @@ def _build_application_spec(
|
|
|
239
332
|
meta: BundleMetadata,
|
|
240
333
|
requirements: ModelRequirements,
|
|
241
334
|
num_replicas: int,
|
|
335
|
+
tiers: list[int],
|
|
242
336
|
) -> dict[str, Any]:
|
|
243
|
-
"""Assemble a Ray Serve application schema from pre-bundled metadata
|
|
244
|
-
the
|
|
337
|
+
"""Assemble a Ray Serve application schema from pre-bundled metadata, the
|
|
338
|
+
model's requirements, and the cluster's GPU size classes."""
|
|
245
339
|
# working_dir carries the serve-app's own source; Ray pip-installs the
|
|
246
340
|
# third-party distributions the image lacks into a per-node cached
|
|
247
341
|
# virtualenv layered on the image. No pip key when there are none, so Ray
|
|
@@ -263,7 +357,7 @@ def _build_application_spec(
|
|
|
263
357
|
"suffix": suffix,
|
|
264
358
|
"run_name": run_name,
|
|
265
359
|
"num_replicas": num_replicas,
|
|
266
|
-
"ray_actor_options": _ray_actor_options(requirements),
|
|
360
|
+
"ray_actor_options": _ray_actor_options(requirements, tiers),
|
|
267
361
|
},
|
|
268
362
|
"runtime_env": runtime_env,
|
|
269
363
|
}
|
|
@@ -312,9 +406,13 @@ class ModelRequirements:
|
|
|
312
406
|
node, CPU-only included. Models saved before requirements existed carry
|
|
313
407
|
no tags and read as the defaults."""
|
|
314
408
|
|
|
315
|
-
|
|
409
|
+
# Fractional, so several models can share one card: 0.25 puts four
|
|
410
|
+
# replicas on a GPU. Ray does not isolate them - they all see the same
|
|
411
|
+
# device - so what keeps them from overcommitting its memory is `vram_gb`,
|
|
412
|
+
# which is reserved from the card's `vram_mib` and is the real limit.
|
|
413
|
+
num_gpus: float = 0.0
|
|
316
414
|
ram_gb: float = 0.0
|
|
317
|
-
# GPU memory across the replica's num_gpus GPUs, so it needs
|
|
415
|
+
# GPU memory across the replica's num_gpus GPUs, so it needs a GPU share.
|
|
318
416
|
vram_gb: float = 0.0
|
|
319
417
|
|
|
320
418
|
def __post_init__(self) -> None:
|
|
@@ -322,7 +420,7 @@ class ModelRequirements:
|
|
|
322
420
|
raise ValueError(f"Model requirements cannot be negative: {self}")
|
|
323
421
|
if self.vram_gb > 0 and self.num_gpus == 0:
|
|
324
422
|
raise ValueError(
|
|
325
|
-
f"vram_gb={self.vram_gb} needs a GPU; set num_gpus
|
|
423
|
+
f"vram_gb={self.vram_gb} needs a GPU; set num_gpus > 0"
|
|
326
424
|
)
|
|
327
425
|
|
|
328
426
|
|
|
@@ -351,7 +449,9 @@ def requirements_from_tags(tags: dict[str, str]) -> ModelRequirements:
|
|
|
351
449
|
"""Deserialise ModelRequirements from a ModelVersion's MLflow tags; a
|
|
352
450
|
missing tag reads as no requirement."""
|
|
353
451
|
return ModelRequirements(
|
|
354
|
-
|
|
452
|
+
# float, not int: models saved before GPUs could be shared stored a
|
|
453
|
+
# whole number, which parses as one.
|
|
454
|
+
num_gpus=float(tags.get(_NUM_GPUS_TAG, "0")),
|
|
355
455
|
ram_gb=float(tags.get(_RAM_GB_TAG, "0")),
|
|
356
456
|
vram_gb=float(tags.get(_VRAM_GB_TAG, "0")),
|
|
357
457
|
)
|
|
@@ -485,6 +585,25 @@ def _clear_failed_application(
|
|
|
485
585
|
time.sleep(_SERVING_POLL_INTERVAL_S)
|
|
486
586
|
|
|
487
587
|
|
|
588
|
+
def _spec_already_deployed(spec: dict[str, Any]) -> bool:
|
|
589
|
+
"""True when this exact spec is already the app's target, so PUTting it
|
|
590
|
+
again would only disturb the controller.
|
|
591
|
+
|
|
592
|
+
Re-PUTting is not a no-op. While an app's build task is in flight its target
|
|
593
|
+
code version is unset, so Ray cancels that build and starts a new one no
|
|
594
|
+
matter what the config says - and since every PUT re-sends the whole
|
|
595
|
+
applications list, it restarts the in-flight builds of the other apps too.
|
|
596
|
+
Our build task downloads the model bundle and may create a pip virtualenv,
|
|
597
|
+
so a caller redeploying faster than that could keep it from ever finishing.
|
|
598
|
+
|
|
599
|
+
A DEPLOY_FAILED or DELETING app never reaches here: `_clear_failed_application`
|
|
600
|
+
has already removed it, and an identical PUT over a failed app is exactly the
|
|
601
|
+
no-op that leaves it failed.
|
|
602
|
+
"""
|
|
603
|
+
app = get_serve_details().get("applications", {}).get(spec["name"])
|
|
604
|
+
return app is not None and app.get("deployed_app_config") == spec
|
|
605
|
+
|
|
606
|
+
|
|
488
607
|
def deploy_model(
|
|
489
608
|
family: str,
|
|
490
609
|
suffix: str,
|
|
@@ -507,6 +626,11 @@ def deploy_model(
|
|
|
507
626
|
or an app still DELETING, is waited out before the new spec is PUT, so the
|
|
508
627
|
deploy starts afresh instead of Ray reusing the failed deployment.
|
|
509
628
|
|
|
629
|
+
Re-deploying a model that is already live with exactly this spec skips the
|
|
630
|
+
PUT rather than restating it: see `_spec_already_deployed` for what a
|
|
631
|
+
redundant PUT costs. The call still reports the app's phase, and with
|
|
632
|
+
`wait=True` still blocks until it is RUNNING.
|
|
633
|
+
|
|
510
634
|
With `wait=True`, blocks as `wait_for_model_serving` does until the Serve
|
|
511
635
|
controller reports the app RUNNING. `timeout` (default 300) caps the whole
|
|
512
636
|
call, clearing a failed app included; exceeding it raises TimeoutError, and
|
|
@@ -516,15 +640,21 @@ def deploy_model(
|
|
|
516
640
|
"""
|
|
517
641
|
deadline = _deadline(timeout)
|
|
518
642
|
meta, requirements = _load_deploy_metadata(family, suffix, run_name)
|
|
643
|
+
# Read afresh on every deploy: the tiers are what the model is placed
|
|
644
|
+
# against, so a GPU joining or leaving the cluster has to change the spec
|
|
645
|
+
# (and therefore re-PUT it), not be remembered from an earlier call.
|
|
519
646
|
spec = _build_application_spec(
|
|
520
|
-
family, suffix, run_name, meta, requirements, num_replicas
|
|
647
|
+
family, suffix, run_name, meta, requirements, num_replicas, vram_tiers()
|
|
521
648
|
)
|
|
522
649
|
_clear_failed_application(family, suffix, run_name, timeout, deadline)
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
650
|
+
if not _spec_already_deployed(spec):
|
|
651
|
+
existing = [
|
|
652
|
+
a for a in _current_application_specs() if a["name"] != spec["name"]
|
|
653
|
+
]
|
|
654
|
+
# The controller registers the app, sets it DEPLOYING and stamps
|
|
655
|
+
# last_deployed_time_s before the PUT returns, so the wait below neither
|
|
656
|
+
# misses the app nor reads a status left by an earlier deploy.
|
|
657
|
+
put_serve_applications([*existing, spec])
|
|
528
658
|
if wait:
|
|
529
659
|
_wait_for_application_running(spec["name"], timeout, deadline)
|
|
530
660
|
app = get_serve_details().get("applications", {}).get(spec["name"], {})
|
|
@@ -626,6 +756,48 @@ def model_serving_status(
|
|
|
626
756
|
)
|
|
627
757
|
|
|
628
758
|
|
|
759
|
+
@dataclass
|
|
760
|
+
class ReplicaPlacement:
|
|
761
|
+
"""Where one replica of a model's Serve app ended up.
|
|
762
|
+
|
|
763
|
+
`node_ip` is the address of the Ray worker running it, which on the
|
|
764
|
+
cluster is the ray-worker pod's IP - the dashboard joins on it to name the
|
|
765
|
+
machine and report its health. It is None for a replica the controller has
|
|
766
|
+
accepted but not yet placed."""
|
|
767
|
+
|
|
768
|
+
replica_id: str
|
|
769
|
+
state: str
|
|
770
|
+
node_id: str | None
|
|
771
|
+
node_ip: str | None
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
def model_replica_placements(
|
|
775
|
+
family: str, suffix: str, run_name: str
|
|
776
|
+
) -> list[ReplicaPlacement]:
|
|
777
|
+
"""Report which worker each of a model's replicas is running on.
|
|
778
|
+
|
|
779
|
+
This is what the requirements and the size-class preferences actually
|
|
780
|
+
resolved to: `deploy_model` asks for the smallest GPU that fits, and this
|
|
781
|
+
is the card it got. Empty when no app exists, and while an app is
|
|
782
|
+
`deploying` it fills in as replicas are placed.
|
|
783
|
+
"""
|
|
784
|
+
app = get_serve_details().get("applications", {}).get(
|
|
785
|
+
_app_name(family, suffix, run_name)
|
|
786
|
+
)
|
|
787
|
+
if app is None:
|
|
788
|
+
return []
|
|
789
|
+
return [
|
|
790
|
+
ReplicaPlacement(
|
|
791
|
+
replica_id=str(replica.get("replica_id", "")),
|
|
792
|
+
state=str(replica.get("state", "")),
|
|
793
|
+
node_id=replica.get("node_id"),
|
|
794
|
+
node_ip=replica.get("node_ip"),
|
|
795
|
+
)
|
|
796
|
+
for deployment in app.get("deployments", {}).values()
|
|
797
|
+
for replica in deployment.get("replicas", [])
|
|
798
|
+
]
|
|
799
|
+
|
|
800
|
+
|
|
629
801
|
@dataclass
|
|
630
802
|
class ServingMessage:
|
|
631
803
|
"""One message the Ray Serve controller reports for a model's app. `source`
|
|
@@ -12,7 +12,11 @@ from typing import Any
|
|
|
12
12
|
|
|
13
13
|
import requests # type: ignore
|
|
14
14
|
|
|
15
|
-
from cortexgrid.infra import
|
|
15
|
+
from cortexgrid.infra import (
|
|
16
|
+
get_ray_job_server_uri,
|
|
17
|
+
get_ray_nodes_uri,
|
|
18
|
+
get_ray_serve_applications_uri,
|
|
19
|
+
)
|
|
16
20
|
from ray.job_submission import JobSubmissionClient
|
|
17
21
|
|
|
18
22
|
|
|
@@ -155,6 +159,18 @@ def get_serve_details() -> dict[str, Any]:
|
|
|
155
159
|
return response.json()
|
|
156
160
|
|
|
157
161
|
|
|
162
|
+
def get_ray_nodes() -> list[dict[str, Any]]:
|
|
163
|
+
"""GET the state API's view of the cluster's nodes.
|
|
164
|
+
|
|
165
|
+
One dict per node, carrying at least `node_id`, `node_ip`, `state`,
|
|
166
|
+
`labels` and `resources_total`. Raises for any non-2xx response
|
|
167
|
+
(HTTPError carries the body).
|
|
168
|
+
"""
|
|
169
|
+
response = requests.get(get_ray_nodes_uri(), timeout=30)
|
|
170
|
+
response.raise_for_status()
|
|
171
|
+
return response.json()["data"]["result"]["result"]
|
|
172
|
+
|
|
173
|
+
|
|
158
174
|
def put_serve_applications(applications: list[dict[str, Any]]) -> None:
|
|
159
175
|
"""PUT the full desired set of Serve applications.
|
|
160
176
|
|
|
@@ -202,7 +202,7 @@ deployed = cortexgrid.deploy_model("qwen", "instruct", saved.run_name, wait=True
|
|
|
202
202
|
print(deployed.url)
|
|
203
203
|
```
|
|
204
204
|
|
|
205
|
-
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
205
|
+
The requirements are part of the model, not of the serve-app class: GPUs, RAM and VRAM (GiB, 0 meaning no requirement) are stored with it and matched against what the cluster's hosts have free. Among the hosts that fit, the model goes to the **smallest GPU** that does, so a 4 GiB model does not occupy a 128 GiB card a bigger one needs; it moves up only once the smaller cards are full. `num_gpus` may be a fraction (`0.25`) to share one card between models, in which case `vram_gb` is what keeps them from overcommitting it. Correct them later with `cortexgrid.set_model_requirements(family, suffix, run_name, requirements)` or on the model card in the dashboard; `cortexgrid.deploy_model(..., num_replicas=2)` chooses how many copies to run.
|
|
206
206
|
|
|
207
207
|
Anything else the serve-app has to know about the model - which model a provider should be asked for, an endpoint, the name of a secret to read - goes in a free-form string mapping on the same entry: `save_model(..., config={"model": "claude-opus-5"})`. The serve-app reads it in `__init__` with `cortexgrid.model_config(family, suffix, run_name)`; `cortexgrid.set_model_config(family, suffix, run_name, config)` or the model card replaces it. cortexgrid stores the mapping without interpreting it, and a tag is readable by anyone with registry access, so a credential belongs in `set_secret` with only its name in the config.
|
|
208
208
|
|
|
@@ -1,13 +1,16 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "cortexgrid"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.5"
|
|
4
4
|
description = "Connect your ML code to the RoboLab compute cluster — Ray, MLflow, and S3"
|
|
5
5
|
readme = "docs/cortexgrid/README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
8
8
|
requires-python = ">=3.11,<3.12"
|
|
9
9
|
dependencies = [
|
|
10
|
-
|
|
10
|
+
# 2.58 is the floor: the smallest-device scheduler uses node labels with
|
|
11
|
+
# `label_selector` / `fallback_strategy` in a replica's ray_actor_options,
|
|
12
|
+
# which Ray Serve only accepts from 2.55 on, and it is what the cluster image runs.
|
|
13
|
+
"ray[default]>=2.58,<3",
|
|
11
14
|
"mlflow>=3.11,<4",
|
|
12
15
|
"packaging>=24",
|
|
13
16
|
"boto3>=1.34",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|