sagemaker-hyperpod 3.0.0__tar.gz → 3.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sagemaker_hyperpod-3.0.0/src/sagemaker_hyperpod.egg-info → sagemaker_hyperpod-3.0.2}/PKG-INFO +68 -92
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/README.md +66 -91
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/pyproject.toml +1 -1
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/setup.py +2 -1
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/cluster.py +79 -33
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/inference.py +148 -64
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/training.py +16 -6
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/hyp_cli.py +8 -8
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/training_utils.py +87 -78
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/config/metadata.py +3 -3
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/__init__.py +14 -0
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/constants.py +61 -0
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/telemetry_logging.py +187 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/hp_endpoint_config.py +12 -3
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/hp_jumpstart_endpoint_config.py +13 -4
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_endpoint.py +10 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +8 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_jumpstart_endpoint.py +10 -0
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/training/__init__.py +6 -0
- sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_status.py → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_unified_config.py +165 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +12 -5
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2/src/sagemaker_hyperpod.egg-info}/PKG-INFO +68 -92
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/SOURCES.txt +5 -4
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/requires.txt +1 -0
- sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/validators/__init__.py +0 -12
- sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/__init__.py +0 -9
- sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_config.py +0 -2977
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/LICENSE +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/NOTICE +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/setup.cfg +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/clients/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/clients/kubernetes_client.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/command_constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/exception_constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/hyperpod_instance_types.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/kueue_constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/pytorch_constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/inference_utils.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/cancel_training_job.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/discover_namespaces.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/exec_command.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_logs.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_namespaces.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_training_job.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/list_pods.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/list_training_jobs.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/self_subject_access_review.py +0 -0
- {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/telemetry → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/templates}/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/utils.py +0 -0
- {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/templates → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/validators}/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/cluster_validator.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/job_validator.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/validator.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common}/telemetry/user_agent.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/utils.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/MonitoringConfig.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/constants.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/utils.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/training/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/dependency_links.txt +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/entry_points.txt +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/top_level.txt +0 -0
- {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/zip-safe +0 -0
{sagemaker_hyperpod-3.0.0/src/sagemaker_hyperpod.egg-info → sagemaker_hyperpod-3.0.2}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sagemaker-hyperpod
|
|
3
|
-
Version: 3.0.
|
|
3
|
+
Version: 3.0.2
|
|
4
4
|
Summary: Amazon SageMaker HyperPod SDK and CLI
|
|
5
5
|
Home-page: https://github.com/aws/sagemaker-hyperpod-cli
|
|
6
6
|
Author: Amazon Web Services
|
|
@@ -38,6 +38,7 @@ Requires-Dist: zstandard==0.15.2
|
|
|
38
38
|
Requires-Dist: pytest==8.3.2
|
|
39
39
|
Requires-Dist: pytest-cov==5.0.0
|
|
40
40
|
Requires-Dist: pytest-order==1.3.0
|
|
41
|
+
Requires-Dist: pytest-dependency==0.6.0
|
|
41
42
|
Requires-Dist: tox==4.18.0
|
|
42
43
|
Requires-Dist: ruff==0.6.2
|
|
43
44
|
Requires-Dist: hera-workflows==5.16.3
|
|
@@ -58,6 +59,8 @@ The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool th
|
|
|
58
59
|
|
|
59
60
|
This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
|
|
60
61
|
|
|
62
|
+
Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
|
|
63
|
+
|
|
61
64
|
## Table of Contents
|
|
62
65
|
- [Overview](#overview)
|
|
63
66
|
- [Prerequisites](#prerequisites)
|
|
@@ -74,8 +77,8 @@ This documentation serves as a reference for the available HyperPod CLI commands
|
|
|
74
77
|
- [Training](#training-)
|
|
75
78
|
- [Inference](#inference-)
|
|
76
79
|
- [SDK](#sdk-)
|
|
77
|
-
- [Training](#training-)
|
|
78
|
-
- [Inference](#inference)
|
|
80
|
+
- [Training](#training-sdk)
|
|
81
|
+
- [Inference](#inference-sdk)
|
|
79
82
|
|
|
80
83
|
|
|
81
84
|
## Overview
|
|
@@ -125,27 +128,9 @@ SageMaker HyperPod CLI currently supports start training job with:
|
|
|
125
128
|
1. Verify if the installation succeeded by running the following command.
|
|
126
129
|
|
|
127
130
|
```
|
|
128
|
-
|
|
131
|
+
hyp --help
|
|
129
132
|
```
|
|
130
133
|
|
|
131
|
-
1. If you have a running HyperPod cluster, you can try to run a training job using the sample configuration file provided at ```/examples/basic-job-example-config.yaml```.
|
|
132
|
-
- Get your HyperPod clusters to show their capacities.
|
|
133
|
-
```
|
|
134
|
-
hyperpod get-clusters
|
|
135
|
-
```
|
|
136
|
-
- Get your HyperPod clusters to show their capacities and quota allocation info for a team.
|
|
137
|
-
```
|
|
138
|
-
hyperpod get-clusters -n hyperpod-ns-<team-name>
|
|
139
|
-
```
|
|
140
|
-
- Connect to one HyperPod cluster and specify a namespace you have access to.
|
|
141
|
-
```
|
|
142
|
-
hyperpod connect-cluster --cluster-name <cluster-name>
|
|
143
|
-
```
|
|
144
|
-
- Start a job in your cluster. Change the `instance_type` in the yaml file to be same as the one in your HyperPod cluster. Also change the `namespace` you want to submit a job to, the example uses kubeflow namespace. You need to have installed PyTorch in your cluster.
|
|
145
|
-
```
|
|
146
|
-
hyperpod start-job --config-file ./examples/basic-job-example-config.yaml
|
|
147
|
-
```
|
|
148
|
-
|
|
149
134
|
## Usage
|
|
150
135
|
|
|
151
136
|
The HyperPod CLI provides the following commands:
|
|
@@ -159,8 +144,8 @@ The HyperPod CLI provides the following commands:
|
|
|
159
144
|
- [Training](#training-)
|
|
160
145
|
- [Inference](#inference-)
|
|
161
146
|
- [SDK](#sdk-)
|
|
162
|
-
- [Training](#training-)
|
|
163
|
-
- [Inference](#inference)
|
|
147
|
+
- [Training](#training-sdk)
|
|
148
|
+
- [Inference](#inference-sdk)
|
|
164
149
|
|
|
165
150
|
|
|
166
151
|
### Getting Cluster information
|
|
@@ -227,8 +212,8 @@ hyp create hyp-pytorch-job \
|
|
|
227
212
|
--version 1.0 \
|
|
228
213
|
--job-name test-pytorch-job \
|
|
229
214
|
--image pytorch/pytorch:latest \
|
|
230
|
-
--command '[
|
|
231
|
-
--args '[
|
|
215
|
+
--command '[python, train.py]' \
|
|
216
|
+
--args '[--epochs=10, --batch-size=32]' \
|
|
232
217
|
--environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
|
|
233
218
|
--pull-policy "IfNotPresent" \
|
|
234
219
|
--instance-type ml.p4d.24xlarge \
|
|
@@ -239,9 +224,8 @@ hyp create hyp-pytorch-job \
|
|
|
239
224
|
--queue-name "training-queue" \
|
|
240
225
|
--priority "high" \
|
|
241
226
|
--max-retry 3 \
|
|
242
|
-
--
|
|
243
|
-
--
|
|
244
|
-
--output-s3-uri s3://my-bucket/model-artifacts
|
|
227
|
+
--volume name=model-data,type=hostPath,mount_path=/data,path=/data \
|
|
228
|
+
--volume name=training-output,type=pvc,mount_path=/data,claim_name=my-pvc,read_only=false
|
|
245
229
|
```
|
|
246
230
|
|
|
247
231
|
Key required parameters explained:
|
|
@@ -250,8 +234,6 @@ Key required parameters explained:
|
|
|
250
234
|
|
|
251
235
|
--image: Docker image containing your training environment
|
|
252
236
|
|
|
253
|
-
This command starts a training job named test-pytorch-job. The --output-s3-uri specifies where the trained model artifacts will be stored, for example, s3://my-bucket/model-artifacts. Note this location, as you’ll need it for deploying the custom model.
|
|
254
|
-
|
|
255
237
|
### Inference
|
|
256
238
|
|
|
257
239
|
#### Creating a JumpstartModel Endpoint
|
|
@@ -320,15 +302,16 @@ hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
|
320
302
|
|
|
321
303
|
Along with the CLI, we also have SDKs available that can perform the training and inference functionalities that the CLI performs
|
|
322
304
|
|
|
323
|
-
### Training
|
|
305
|
+
### Training SDK
|
|
324
306
|
|
|
325
307
|
#### Creating a Training Job
|
|
326
308
|
|
|
327
309
|
```
|
|
328
310
|
|
|
329
|
-
from sagemaker.hyperpod import HyperPodPytorchJob
|
|
330
|
-
from sagemaker.hyperpod.
|
|
331
|
-
import ReplicaSpec, Template, Spec,
|
|
311
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
312
|
+
from sagemaker.hyperpod.training
|
|
313
|
+
import ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
|
|
314
|
+
from sagemaker.hyperpod.common.config import Metadata
|
|
332
315
|
|
|
333
316
|
# Define job specifications
|
|
334
317
|
nproc_per_node = "1" # Number of processes per node
|
|
@@ -343,7 +326,7 @@ replica_specs =
|
|
|
343
326
|
(
|
|
344
327
|
containers =
|
|
345
328
|
[
|
|
346
|
-
|
|
329
|
+
Containers
|
|
347
330
|
(
|
|
348
331
|
# Container name
|
|
349
332
|
name="container-name",
|
|
@@ -384,8 +367,6 @@ pytorch_job = HyperPodPytorchJob
|
|
|
384
367
|
replica_specs = replica_specs,
|
|
385
368
|
# Run policy
|
|
386
369
|
run_policy = run_policy,
|
|
387
|
-
# S3 location for artifacts
|
|
388
|
-
output_s3_uri="s3://my-bucket/model-artifacts"
|
|
389
370
|
)
|
|
390
371
|
# Launch the job
|
|
391
372
|
pytorch_job.create()
|
|
@@ -395,7 +376,7 @@ pytorch_job.create()
|
|
|
395
376
|
|
|
396
377
|
|
|
397
378
|
|
|
398
|
-
### Inference
|
|
379
|
+
### Inference SDK
|
|
399
380
|
|
|
400
381
|
#### Creating a JumpstartModel Endpoint
|
|
401
382
|
|
|
@@ -405,24 +386,21 @@ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io
|
|
|
405
386
|
from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
|
|
406
387
|
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
407
388
|
|
|
408
|
-
model
|
|
409
|
-
model_id=
|
|
410
|
-
model_version=
|
|
389
|
+
model=Model(
|
|
390
|
+
model_id='deepseek-llm-r1-distill-qwen-1-5b',
|
|
391
|
+
model_version='2.0.4',
|
|
411
392
|
)
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
instance_type="ml.g5.8xlarge"
|
|
393
|
+
server=Server(
|
|
394
|
+
instance_type='ml.g5.8xlarge',
|
|
415
395
|
)
|
|
396
|
+
endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
|
|
397
|
+
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
|
|
416
398
|
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
|
|
420
|
-
|
|
421
|
-
js_endpoint = HPJumpStartEndpoint(
|
|
399
|
+
js_endpoint=HPJumpStartEndpoint(
|
|
422
400
|
model=model,
|
|
423
401
|
server=server,
|
|
424
402
|
sage_maker_endpoint=endpoint_name,
|
|
425
|
-
tls_config=tls_config
|
|
403
|
+
tls_config=tls_config,
|
|
426
404
|
)
|
|
427
405
|
|
|
428
406
|
js_endpoint.create()
|
|
@@ -438,51 +416,51 @@ print(response)
|
|
|
438
416
|
```
|
|
439
417
|
|
|
440
418
|
|
|
441
|
-
#### Creating a Custom Inference Endpoint
|
|
419
|
+
#### Creating a Custom Inference Endpoint (with S3)
|
|
442
420
|
|
|
443
421
|
```
|
|
444
|
-
from sagemaker.hyperpod.inference.config.
|
|
445
|
-
from sagemaker.hyperpod.inference.
|
|
422
|
+
from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
|
|
423
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
446
424
|
|
|
447
|
-
|
|
448
|
-
model_source_type=
|
|
449
|
-
model_location="
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
425
|
+
model_source_config = ModelSourceConfig(
|
|
426
|
+
model_source_type='s3',
|
|
427
|
+
model_location="<my-model-folder-in-s3>",
|
|
428
|
+
s3_storage=S3Storage(
|
|
429
|
+
bucket_name='<my-model-artifacts-bucket>',
|
|
430
|
+
region='us-east-2',
|
|
431
|
+
),
|
|
453
432
|
)
|
|
454
433
|
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
)
|
|
434
|
+
environment_variables = [
|
|
435
|
+
EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
|
|
436
|
+
EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
|
|
437
|
+
EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
|
|
438
|
+
EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
|
|
439
|
+
EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
|
|
440
|
+
]
|
|
461
441
|
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
442
|
+
worker = Worker(
|
|
443
|
+
image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
|
|
444
|
+
model_volume_mount=ModelVolumeMount(
|
|
445
|
+
name='model-weights',
|
|
446
|
+
),
|
|
447
|
+
model_invocation_port=ModelInvocationPort(container_port=8080),
|
|
448
|
+
resources=Resources(
|
|
449
|
+
requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
|
|
450
|
+
limits={"nvidia.com/gpu": 1}
|
|
451
|
+
),
|
|
452
|
+
environment_variables=environment_variables,
|
|
473
453
|
)
|
|
474
454
|
|
|
475
|
-
|
|
455
|
+
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
|
|
476
456
|
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
server=server,
|
|
482
|
-
resources=resources,
|
|
483
|
-
environment=env,
|
|
484
|
-
sage_maker_endpoint=endpoint_name,
|
|
457
|
+
custom_endpoint = HPEndpoint(
|
|
458
|
+
endpoint_name='<my-endpoint-name>',
|
|
459
|
+
instance_type='ml.g5.8xlarge',
|
|
460
|
+
model_name='deepseek15b-test-model-name',
|
|
485
461
|
tls_config=tls_config,
|
|
462
|
+
model_source_config=model_source_config,
|
|
463
|
+
worker=worker,
|
|
486
464
|
)
|
|
487
465
|
|
|
488
466
|
custom_endpoint.create()
|
|
@@ -499,19 +477,17 @@ print(response)
|
|
|
499
477
|
#### Managing an Endpoint
|
|
500
478
|
|
|
501
479
|
```
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
print(endpoint.name, endpoint.status)
|
|
480
|
+
endpoint_list = HPEndpoint.list()
|
|
481
|
+
print(endpoint_list[0])
|
|
505
482
|
|
|
506
|
-
|
|
507
|
-
print(logs)
|
|
483
|
+
print(custom_endpoint.get_operator_logs(since_hours=0.5))
|
|
508
484
|
|
|
509
485
|
```
|
|
510
486
|
|
|
511
487
|
#### Deleting an Endpoint
|
|
512
488
|
|
|
513
489
|
```
|
|
514
|
-
|
|
490
|
+
custom_endpoint.delete()
|
|
515
491
|
|
|
516
492
|
```
|
|
517
493
|
|
|
@@ -5,6 +5,8 @@ The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool th
|
|
|
5
5
|
|
|
6
6
|
This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
|
|
7
7
|
|
|
8
|
+
Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
|
|
9
|
+
|
|
8
10
|
## Table of Contents
|
|
9
11
|
- [Overview](#overview)
|
|
10
12
|
- [Prerequisites](#prerequisites)
|
|
@@ -21,8 +23,8 @@ This documentation serves as a reference for the available HyperPod CLI commands
|
|
|
21
23
|
- [Training](#training-)
|
|
22
24
|
- [Inference](#inference-)
|
|
23
25
|
- [SDK](#sdk-)
|
|
24
|
-
- [Training](#training-)
|
|
25
|
-
- [Inference](#inference)
|
|
26
|
+
- [Training](#training-sdk)
|
|
27
|
+
- [Inference](#inference-sdk)
|
|
26
28
|
|
|
27
29
|
|
|
28
30
|
## Overview
|
|
@@ -72,27 +74,9 @@ SageMaker HyperPod CLI currently supports start training job with:
|
|
|
72
74
|
1. Verify if the installation succeeded by running the following command.
|
|
73
75
|
|
|
74
76
|
```
|
|
75
|
-
|
|
77
|
+
hyp --help
|
|
76
78
|
```
|
|
77
79
|
|
|
78
|
-
1. If you have a running HyperPod cluster, you can try to run a training job using the sample configuration file provided at ```/examples/basic-job-example-config.yaml```.
|
|
79
|
-
- Get your HyperPod clusters to show their capacities.
|
|
80
|
-
```
|
|
81
|
-
hyperpod get-clusters
|
|
82
|
-
```
|
|
83
|
-
- Get your HyperPod clusters to show their capacities and quota allocation info for a team.
|
|
84
|
-
```
|
|
85
|
-
hyperpod get-clusters -n hyperpod-ns-<team-name>
|
|
86
|
-
```
|
|
87
|
-
- Connect to one HyperPod cluster and specify a namespace you have access to.
|
|
88
|
-
```
|
|
89
|
-
hyperpod connect-cluster --cluster-name <cluster-name>
|
|
90
|
-
```
|
|
91
|
-
- Start a job in your cluster. Change the `instance_type` in the yaml file to be same as the one in your HyperPod cluster. Also change the `namespace` you want to submit a job to, the example uses kubeflow namespace. You need to have installed PyTorch in your cluster.
|
|
92
|
-
```
|
|
93
|
-
hyperpod start-job --config-file ./examples/basic-job-example-config.yaml
|
|
94
|
-
```
|
|
95
|
-
|
|
96
80
|
## Usage
|
|
97
81
|
|
|
98
82
|
The HyperPod CLI provides the following commands:
|
|
@@ -106,8 +90,8 @@ The HyperPod CLI provides the following commands:
|
|
|
106
90
|
- [Training](#training-)
|
|
107
91
|
- [Inference](#inference-)
|
|
108
92
|
- [SDK](#sdk-)
|
|
109
|
-
- [Training](#training-)
|
|
110
|
-
- [Inference](#inference)
|
|
93
|
+
- [Training](#training-sdk)
|
|
94
|
+
- [Inference](#inference-sdk)
|
|
111
95
|
|
|
112
96
|
|
|
113
97
|
### Getting Cluster information
|
|
@@ -174,8 +158,8 @@ hyp create hyp-pytorch-job \
|
|
|
174
158
|
--version 1.0 \
|
|
175
159
|
--job-name test-pytorch-job \
|
|
176
160
|
--image pytorch/pytorch:latest \
|
|
177
|
-
--command '[
|
|
178
|
-
--args '[
|
|
161
|
+
--command '[python, train.py]' \
|
|
162
|
+
--args '[--epochs=10, --batch-size=32]' \
|
|
179
163
|
--environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
|
|
180
164
|
--pull-policy "IfNotPresent" \
|
|
181
165
|
--instance-type ml.p4d.24xlarge \
|
|
@@ -186,9 +170,8 @@ hyp create hyp-pytorch-job \
|
|
|
186
170
|
--queue-name "training-queue" \
|
|
187
171
|
--priority "high" \
|
|
188
172
|
--max-retry 3 \
|
|
189
|
-
--
|
|
190
|
-
--
|
|
191
|
-
--output-s3-uri s3://my-bucket/model-artifacts
|
|
173
|
+
--volume name=model-data,type=hostPath,mount_path=/data,path=/data \
|
|
174
|
+
--volume name=training-output,type=pvc,mount_path=/data,claim_name=my-pvc,read_only=false
|
|
192
175
|
```
|
|
193
176
|
|
|
194
177
|
Key required parameters explained:
|
|
@@ -197,8 +180,6 @@ Key required parameters explained:
|
|
|
197
180
|
|
|
198
181
|
--image: Docker image containing your training environment
|
|
199
182
|
|
|
200
|
-
This command starts a training job named test-pytorch-job. The --output-s3-uri specifies where the trained model artifacts will be stored, for example, s3://my-bucket/model-artifacts. Note this location, as you’ll need it for deploying the custom model.
|
|
201
|
-
|
|
202
183
|
### Inference
|
|
203
184
|
|
|
204
185
|
#### Creating a JumpstartModel Endpoint
|
|
@@ -267,15 +248,16 @@ hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
|
267
248
|
|
|
268
249
|
Along with the CLI, we also have SDKs available that can perform the training and inference functionalities that the CLI performs
|
|
269
250
|
|
|
270
|
-
### Training
|
|
251
|
+
### Training SDK
|
|
271
252
|
|
|
272
253
|
#### Creating a Training Job
|
|
273
254
|
|
|
274
255
|
```
|
|
275
256
|
|
|
276
|
-
from sagemaker.hyperpod import HyperPodPytorchJob
|
|
277
|
-
from sagemaker.hyperpod.
|
|
278
|
-
import ReplicaSpec, Template, Spec,
|
|
257
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
258
|
+
from sagemaker.hyperpod.training
|
|
259
|
+
import ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
|
|
260
|
+
from sagemaker.hyperpod.common.config import Metadata
|
|
279
261
|
|
|
280
262
|
# Define job specifications
|
|
281
263
|
nproc_per_node = "1" # Number of processes per node
|
|
@@ -290,7 +272,7 @@ replica_specs =
|
|
|
290
272
|
(
|
|
291
273
|
containers =
|
|
292
274
|
[
|
|
293
|
-
|
|
275
|
+
Containers
|
|
294
276
|
(
|
|
295
277
|
# Container name
|
|
296
278
|
name="container-name",
|
|
@@ -331,8 +313,6 @@ pytorch_job = HyperPodPytorchJob
|
|
|
331
313
|
replica_specs = replica_specs,
|
|
332
314
|
# Run policy
|
|
333
315
|
run_policy = run_policy,
|
|
334
|
-
# S3 location for artifacts
|
|
335
|
-
output_s3_uri="s3://my-bucket/model-artifacts"
|
|
336
316
|
)
|
|
337
317
|
# Launch the job
|
|
338
318
|
pytorch_job.create()
|
|
@@ -342,7 +322,7 @@ pytorch_job.create()
|
|
|
342
322
|
|
|
343
323
|
|
|
344
324
|
|
|
345
|
-
### Inference
|
|
325
|
+
### Inference SDK
|
|
346
326
|
|
|
347
327
|
#### Creating a JumpstartModel Endpoint
|
|
348
328
|
|
|
@@ -352,24 +332,21 @@ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io
|
|
|
352
332
|
from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
|
|
353
333
|
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
354
334
|
|
|
355
|
-
model
|
|
356
|
-
model_id=
|
|
357
|
-
model_version=
|
|
335
|
+
model=Model(
|
|
336
|
+
model_id='deepseek-llm-r1-distill-qwen-1-5b',
|
|
337
|
+
model_version='2.0.4',
|
|
358
338
|
)
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
instance_type="ml.g5.8xlarge"
|
|
339
|
+
server=Server(
|
|
340
|
+
instance_type='ml.g5.8xlarge',
|
|
362
341
|
)
|
|
342
|
+
endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
|
|
343
|
+
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
|
|
363
344
|
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
|
|
367
|
-
|
|
368
|
-
js_endpoint = HPJumpStartEndpoint(
|
|
345
|
+
js_endpoint=HPJumpStartEndpoint(
|
|
369
346
|
model=model,
|
|
370
347
|
server=server,
|
|
371
348
|
sage_maker_endpoint=endpoint_name,
|
|
372
|
-
tls_config=tls_config
|
|
349
|
+
tls_config=tls_config,
|
|
373
350
|
)
|
|
374
351
|
|
|
375
352
|
js_endpoint.create()
|
|
@@ -385,51 +362,51 @@ print(response)
|
|
|
385
362
|
```
|
|
386
363
|
|
|
387
364
|
|
|
388
|
-
#### Creating a Custom Inference Endpoint
|
|
365
|
+
#### Creating a Custom Inference Endpoint (with S3)
|
|
389
366
|
|
|
390
367
|
```
|
|
391
|
-
from sagemaker.hyperpod.inference.config.
|
|
392
|
-
from sagemaker.hyperpod.inference.
|
|
368
|
+
from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
|
|
369
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
393
370
|
|
|
394
|
-
|
|
395
|
-
model_source_type=
|
|
396
|
-
model_location="
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
371
|
+
model_source_config = ModelSourceConfig(
|
|
372
|
+
model_source_type='s3',
|
|
373
|
+
model_location="<my-model-folder-in-s3>",
|
|
374
|
+
s3_storage=S3Storage(
|
|
375
|
+
bucket_name='<my-model-artifacts-bucket>',
|
|
376
|
+
region='us-east-2',
|
|
377
|
+
),
|
|
400
378
|
)
|
|
401
379
|
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
)
|
|
380
|
+
environment_variables = [
|
|
381
|
+
EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
|
|
382
|
+
EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
|
|
383
|
+
EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
|
|
384
|
+
EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
|
|
385
|
+
EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
|
|
386
|
+
]
|
|
408
387
|
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
388
|
+
worker = Worker(
|
|
389
|
+
image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
|
|
390
|
+
model_volume_mount=ModelVolumeMount(
|
|
391
|
+
name='model-weights',
|
|
392
|
+
),
|
|
393
|
+
model_invocation_port=ModelInvocationPort(container_port=8080),
|
|
394
|
+
resources=Resources(
|
|
395
|
+
requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
|
|
396
|
+
limits={"nvidia.com/gpu": 1}
|
|
397
|
+
),
|
|
398
|
+
environment_variables=environment_variables,
|
|
420
399
|
)
|
|
421
400
|
|
|
422
|
-
|
|
401
|
+
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
|
|
423
402
|
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
server=server,
|
|
429
|
-
resources=resources,
|
|
430
|
-
environment=env,
|
|
431
|
-
sage_maker_endpoint=endpoint_name,
|
|
403
|
+
custom_endpoint = HPEndpoint(
|
|
404
|
+
endpoint_name='<my-endpoint-name>',
|
|
405
|
+
instance_type='ml.g5.8xlarge',
|
|
406
|
+
model_name='deepseek15b-test-model-name',
|
|
432
407
|
tls_config=tls_config,
|
|
408
|
+
model_source_config=model_source_config,
|
|
409
|
+
worker=worker,
|
|
433
410
|
)
|
|
434
411
|
|
|
435
412
|
custom_endpoint.create()
|
|
@@ -446,19 +423,17 @@ print(response)
|
|
|
446
423
|
#### Managing an Endpoint
|
|
447
424
|
|
|
448
425
|
```
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
print(endpoint.name, endpoint.status)
|
|
426
|
+
endpoint_list = HPEndpoint.list()
|
|
427
|
+
print(endpoint_list[0])
|
|
452
428
|
|
|
453
|
-
|
|
454
|
-
print(logs)
|
|
429
|
+
print(custom_endpoint.get_operator_logs(since_hours=0.5))
|
|
455
430
|
|
|
456
431
|
```
|
|
457
432
|
|
|
458
433
|
#### Deleting an Endpoint
|
|
459
434
|
|
|
460
435
|
```
|
|
461
|
-
|
|
436
|
+
custom_endpoint.delete()
|
|
462
437
|
|
|
463
438
|
```
|
|
464
439
|
|
|
@@ -47,7 +47,7 @@ for root, dirs, files in os.walk(
|
|
|
47
47
|
setup(
|
|
48
48
|
data_files=sagemaker_hyperpod_recipes,
|
|
49
49
|
name="sagemaker-hyperpod",
|
|
50
|
-
version="3.0.
|
|
50
|
+
version="3.0.2",
|
|
51
51
|
description="Amazon SageMaker HyperPod SDK and CLI",
|
|
52
52
|
long_description=open("README.md").read(),
|
|
53
53
|
long_description_content_type="text/markdown",
|
|
@@ -81,6 +81,7 @@ setup(
|
|
|
81
81
|
"pytest==8.3.2",
|
|
82
82
|
"pytest-cov==5.0.0",
|
|
83
83
|
"pytest-order==1.3.0",
|
|
84
|
+
"pytest-dependency==0.6.0",
|
|
84
85
|
"tox==4.18.0",
|
|
85
86
|
"ruff==0.6.2",
|
|
86
87
|
"hera-workflows==5.16.3",
|