sagemaker-hyperpod 3.0.2__tar.gz → 3.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sagemaker_hyperpod-3.0.2/src/sagemaker_hyperpod.egg-info → sagemaker_hyperpod-3.2.0}/PKG-INFO +16 -23
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/README.md +15 -22
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/pyproject.toml +2 -2
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/setup.py +4 -2
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/clients/kubernetes_client.py +2 -6
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/commands/cluster.py +132 -69
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/commands/cluster_stack.py +379 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/commands/inference.py +48 -13
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/commands/init.py +430 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/commands/training.py +368 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/common_utils.py +71 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/command_constants.py +1 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/constants/init_constants.py +319 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/pytorch_constants.py +1 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/hyp_cli.py +218 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/inference_utils.py +29 -48
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/init_utils.py +949 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/templates/cfn_cluster_creation.py +948 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/templates/k8s_custom_endpoint_template.py +68 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/templates/k8s_js_endpoint_template.py +17 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +68 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/training_utils.py +16 -33
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/utils.py +12 -1
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cluster_management/config/hp_cluster_stack_config.py +43 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cluster_management/hp_cluster_stack.py +545 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/common/cli_decorators.py +974 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/config/metadata.py +4 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/common/exceptions/__init__.py +10 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/common/utils.py +563 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/inference/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/hp_endpoint.py +35 -1
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +526 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/hp_jumpstart_endpoint.py +44 -4
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/inference/jumpstart_public_hub_visualization_utils.py +301 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/observability/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_unified_config.py +1 -1
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +650 -0
- sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/training/quota_allocation_util.py +278 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0/src/sagemaker_hyperpod.egg-info}/PKG-INFO +16 -23
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/SOURCES.txt +16 -0
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/commands/training.py +0 -357
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/hyp_cli.py +0 -132
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +0 -74
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/utils.py +0 -299
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +0 -231
- sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +0 -260
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/LICENSE +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/NOTICE +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/setup.cfg +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/clients/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/commands/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/exception_constants.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/hyperpod_instance_types.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/constants/kueue_constants.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/cancel_training_job.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/discover_namespaces.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/exec_command.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/get_logs.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/get_namespaces.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/get_training_job.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/list_pods.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/list_training_jobs.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/service/self_subject_access_review.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/templates/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/validators/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/validators/cluster_validator.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/validators/job_validator.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/validators/validator.py +0 -0
- {sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common → sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cluster_management}/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/inference → sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/cluster_management}/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/observability → sagemaker_hyperpod-3.2.0/src/sagemaker/hyperpod/common}/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/telemetry/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/telemetry/constants.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/telemetry/telemetry_logging.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/common/telemetry/user_agent.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/config/constants.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/config/hp_endpoint_config.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/inference/config/hp_jumpstart_endpoint_config.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/observability/MonitoringConfig.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/observability/constants.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/observability/utils.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/training/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/training/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/dependency_links.txt +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/entry_points.txt +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/requires.txt +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/top_level.txt +0 -0
- {sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker_hyperpod.egg-info/zip-safe +0 -0
{sagemaker_hyperpod-3.0.2/src/sagemaker_hyperpod.egg-info → sagemaker_hyperpod-3.2.0}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sagemaker-hyperpod
|
|
3
|
-
Version: 3.0
|
|
3
|
+
Version: 3.2.0
|
|
4
4
|
Summary: Amazon SageMaker HyperPod SDK and CLI
|
|
5
5
|
Home-page: https://github.com/aws/sagemaker-hyperpod-cli
|
|
6
6
|
Author: Amazon Web Services
|
|
@@ -108,24 +108,13 @@ SageMaker HyperPod CLI currently supports start training job with:
|
|
|
108
108
|
|
|
109
109
|
1. Make sure that your local python version is 3.8, 3.9, 3.10 or 3.11.
|
|
110
110
|
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
The SageMaker Hyperpod CLI uses Helm to start training jobs. See also the [Helm installation guide](https://helm.sh/docs/intro/install/).
|
|
114
|
-
|
|
115
|
-
```
|
|
116
|
-
curl -fsSL -o get_helm.sh https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3
|
|
117
|
-
chmod 700 get_helm.sh
|
|
118
|
-
./get_helm.sh
|
|
119
|
-
rm -f ./get_helm.sh
|
|
120
|
-
```
|
|
121
|
-
|
|
122
|
-
1. Clone and install the sagemaker-hyperpod-cli package.
|
|
111
|
+
2. Install the sagemaker-hyperpod-cli package.
|
|
123
112
|
|
|
124
113
|
```
|
|
125
114
|
pip install sagemaker-hyperpod
|
|
126
115
|
```
|
|
127
116
|
|
|
128
|
-
|
|
117
|
+
3. Verify if the installation succeeded by running the following command.
|
|
129
118
|
|
|
130
119
|
```
|
|
131
120
|
hyp --help
|
|
@@ -224,8 +213,15 @@ hyp create hyp-pytorch-job \
|
|
|
224
213
|
--queue-name "training-queue" \
|
|
225
214
|
--priority "high" \
|
|
226
215
|
--max-retry 3 \
|
|
216
|
+
--accelerators 8 \
|
|
217
|
+
--vcpu 96.0 \
|
|
218
|
+
--memory 1152.0 \
|
|
219
|
+
--accelerators-limit 8 \
|
|
220
|
+
--vcpu-limit 96.0 \
|
|
221
|
+
--memory-limit 1152.0 \
|
|
222
|
+
--preferred-topology "topology.kubernetes.io/zone=us-west-2a" \
|
|
227
223
|
--volume name=model-data,type=hostPath,mount_path=/data,path=/data \
|
|
228
|
-
--volume name=training-output,type=pvc,mount_path=/
|
|
224
|
+
--volume name=training-output,type=pvc,mount_path=/data2,claim_name=my-pvc,read_only=false
|
|
229
225
|
```
|
|
230
226
|
|
|
231
227
|
Key required parameters explained:
|
|
@@ -246,7 +242,6 @@ hyp create hyp-jumpstart-endpoint \
|
|
|
246
242
|
--model-id jumpstart-model-id\
|
|
247
243
|
--instance-type ml.g5.8xlarge \
|
|
248
244
|
--endpoint-name endpoint-jumpstart \
|
|
249
|
-
--tls-output-s3-uri s3://sample-bucket
|
|
250
245
|
```
|
|
251
246
|
|
|
252
247
|
|
|
@@ -262,7 +257,7 @@ hyp invoke hyp-jumpstart-endpoint \
|
|
|
262
257
|
|
|
263
258
|
```
|
|
264
259
|
hyp list hyp-jumpstart-endpoint
|
|
265
|
-
hyp
|
|
260
|
+
hyp describe hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
266
261
|
```
|
|
267
262
|
|
|
268
263
|
#### Creating a Custom Inference Endpoint
|
|
@@ -273,7 +268,8 @@ hyp create hyp-custom-endpoint \
|
|
|
273
268
|
--endpoint-name my-custom-endpoint \
|
|
274
269
|
--model-name my-pytorch-model \
|
|
275
270
|
--model-source-type s3 \
|
|
276
|
-
--model-location my-pytorch-training
|
|
271
|
+
--model-location my-pytorch-training \
|
|
272
|
+
--model-volume-mount-name test-volume \
|
|
277
273
|
--s3-bucket-name your-bucket \
|
|
278
274
|
--s3-region us-east-1 \
|
|
279
275
|
--instance-type ml.g5.8xlarge \
|
|
@@ -387,20 +383,17 @@ from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Mod
|
|
|
387
383
|
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
388
384
|
|
|
389
385
|
model=Model(
|
|
390
|
-
model_id='deepseek-llm-r1-distill-qwen-1-5b'
|
|
391
|
-
model_version='2.0.4',
|
|
386
|
+
model_id='deepseek-llm-r1-distill-qwen-1-5b'
|
|
392
387
|
)
|
|
393
388
|
server=Server(
|
|
394
389
|
instance_type='ml.g5.8xlarge',
|
|
395
390
|
)
|
|
396
391
|
endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
|
|
397
|
-
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
|
|
398
392
|
|
|
399
393
|
js_endpoint=HPJumpStartEndpoint(
|
|
400
394
|
model=model,
|
|
401
395
|
server=server,
|
|
402
|
-
sage_maker_endpoint=endpoint_name
|
|
403
|
-
tls_config=tls_config,
|
|
396
|
+
sage_maker_endpoint=endpoint_name
|
|
404
397
|
)
|
|
405
398
|
|
|
406
399
|
js_endpoint.create()
|
|
@@ -54,24 +54,13 @@ SageMaker HyperPod CLI currently supports start training job with:
|
|
|
54
54
|
|
|
55
55
|
1. Make sure that your local python version is 3.8, 3.9, 3.10 or 3.11.
|
|
56
56
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
The SageMaker Hyperpod CLI uses Helm to start training jobs. See also the [Helm installation guide](https://helm.sh/docs/intro/install/).
|
|
60
|
-
|
|
61
|
-
```
|
|
62
|
-
curl -fsSL -o get_helm.sh https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3
|
|
63
|
-
chmod 700 get_helm.sh
|
|
64
|
-
./get_helm.sh
|
|
65
|
-
rm -f ./get_helm.sh
|
|
66
|
-
```
|
|
67
|
-
|
|
68
|
-
1. Clone and install the sagemaker-hyperpod-cli package.
|
|
57
|
+
2. Install the sagemaker-hyperpod-cli package.
|
|
69
58
|
|
|
70
59
|
```
|
|
71
60
|
pip install sagemaker-hyperpod
|
|
72
61
|
```
|
|
73
62
|
|
|
74
|
-
|
|
63
|
+
3. Verify if the installation succeeded by running the following command.
|
|
75
64
|
|
|
76
65
|
```
|
|
77
66
|
hyp --help
|
|
@@ -170,8 +159,15 @@ hyp create hyp-pytorch-job \
|
|
|
170
159
|
--queue-name "training-queue" \
|
|
171
160
|
--priority "high" \
|
|
172
161
|
--max-retry 3 \
|
|
162
|
+
--accelerators 8 \
|
|
163
|
+
--vcpu 96.0 \
|
|
164
|
+
--memory 1152.0 \
|
|
165
|
+
--accelerators-limit 8 \
|
|
166
|
+
--vcpu-limit 96.0 \
|
|
167
|
+
--memory-limit 1152.0 \
|
|
168
|
+
--preferred-topology "topology.kubernetes.io/zone=us-west-2a" \
|
|
173
169
|
--volume name=model-data,type=hostPath,mount_path=/data,path=/data \
|
|
174
|
-
--volume name=training-output,type=pvc,mount_path=/
|
|
170
|
+
--volume name=training-output,type=pvc,mount_path=/data2,claim_name=my-pvc,read_only=false
|
|
175
171
|
```
|
|
176
172
|
|
|
177
173
|
Key required parameters explained:
|
|
@@ -192,7 +188,6 @@ hyp create hyp-jumpstart-endpoint \
|
|
|
192
188
|
--model-id jumpstart-model-id\
|
|
193
189
|
--instance-type ml.g5.8xlarge \
|
|
194
190
|
--endpoint-name endpoint-jumpstart \
|
|
195
|
-
--tls-output-s3-uri s3://sample-bucket
|
|
196
191
|
```
|
|
197
192
|
|
|
198
193
|
|
|
@@ -208,7 +203,7 @@ hyp invoke hyp-jumpstart-endpoint \
|
|
|
208
203
|
|
|
209
204
|
```
|
|
210
205
|
hyp list hyp-jumpstart-endpoint
|
|
211
|
-
hyp
|
|
206
|
+
hyp describe hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
212
207
|
```
|
|
213
208
|
|
|
214
209
|
#### Creating a Custom Inference Endpoint
|
|
@@ -219,7 +214,8 @@ hyp create hyp-custom-endpoint \
|
|
|
219
214
|
--endpoint-name my-custom-endpoint \
|
|
220
215
|
--model-name my-pytorch-model \
|
|
221
216
|
--model-source-type s3 \
|
|
222
|
-
--model-location my-pytorch-training
|
|
217
|
+
--model-location my-pytorch-training \
|
|
218
|
+
--model-volume-mount-name test-volume \
|
|
223
219
|
--s3-bucket-name your-bucket \
|
|
224
220
|
--s3-region us-east-1 \
|
|
225
221
|
--instance-type ml.g5.8xlarge \
|
|
@@ -333,20 +329,17 @@ from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Mod
|
|
|
333
329
|
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
334
330
|
|
|
335
331
|
model=Model(
|
|
336
|
-
model_id='deepseek-llm-r1-distill-qwen-1-5b'
|
|
337
|
-
model_version='2.0.4',
|
|
332
|
+
model_id='deepseek-llm-r1-distill-qwen-1-5b'
|
|
338
333
|
)
|
|
339
334
|
server=Server(
|
|
340
335
|
instance_type='ml.g5.8xlarge',
|
|
341
336
|
)
|
|
342
337
|
endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
|
|
343
|
-
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
|
|
344
338
|
|
|
345
339
|
js_endpoint=HPJumpStartEndpoint(
|
|
346
340
|
model=model,
|
|
347
341
|
server=server,
|
|
348
|
-
sage_maker_endpoint=endpoint_name
|
|
349
|
-
tls_config=tls_config,
|
|
342
|
+
sage_maker_endpoint=endpoint_name
|
|
350
343
|
)
|
|
351
344
|
|
|
352
345
|
js_endpoint.create()
|
|
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
|
|
5
5
|
[project]
|
|
6
6
|
dynamic = ["dependencies"]
|
|
7
7
|
name = "sagemaker-hyperpod"
|
|
8
|
-
version = "3.0
|
|
8
|
+
version = "3.2.0"
|
|
9
9
|
description = "Amazon SageMaker HyperPod SDK and CLI"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.8"
|
|
@@ -112,4 +112,4 @@ docstring-code-format = false
|
|
|
112
112
|
#
|
|
113
113
|
# This only has an effect when the `docstring-code-format` setting is
|
|
114
114
|
# enabled.
|
|
115
|
-
docstring-code-line-length = "dynamic"
|
|
115
|
+
docstring-code-line-length = "dynamic"
|
|
@@ -47,7 +47,7 @@ for root, dirs, files in os.walk(
|
|
|
47
47
|
setup(
|
|
48
48
|
data_files=sagemaker_hyperpod_recipes,
|
|
49
49
|
name="sagemaker-hyperpod",
|
|
50
|
-
version="3.0
|
|
50
|
+
version="3.2.0",
|
|
51
51
|
description="Amazon SageMaker HyperPod SDK and CLI",
|
|
52
52
|
long_description=open("README.md").read(),
|
|
53
53
|
long_description_content_type="text/markdown",
|
|
@@ -89,7 +89,9 @@ setup(
|
|
|
89
89
|
"pydantic>=2.10.6,<3.0.0",
|
|
90
90
|
"hyperpod-pytorch-job-template>=1.0.0, <2.0.0",
|
|
91
91
|
"hyperpod-custom-inference-template>=1.0.0, <2.0.0",
|
|
92
|
-
|
|
92
|
+
"hyperpod-jumpstart-inference-template>=1.0.0, <2.0.0",
|
|
93
|
+
# To be enabled after launch
|
|
94
|
+
#"hyperpod-cluster-stack-template>=1.0.0, <2.0.0"
|
|
93
95
|
],
|
|
94
96
|
entry_points={
|
|
95
97
|
"console_scripts": [
|
|
@@ -51,14 +51,10 @@ class KubernetesClient:
|
|
|
51
51
|
_instance = None
|
|
52
52
|
_kube_client = None
|
|
53
53
|
|
|
54
|
-
def __new__(cls,
|
|
54
|
+
def __new__(cls, config_file: Optional[str] = None) -> "KubernetesClient":
|
|
55
55
|
if cls._instance is None:
|
|
56
56
|
cls._instance = super(KubernetesClient, cls).__new__(cls)
|
|
57
|
-
config.load_kube_config(
|
|
58
|
-
config_file=KUBE_CONFIG_PATH
|
|
59
|
-
if not is_get_capacity
|
|
60
|
-
else TEMP_KUBE_CONFIG_FILE
|
|
61
|
-
) # or config.load_incluster_config() for in-cluster config
|
|
57
|
+
config.load_kube_config(config_file=config_file or KUBE_CONFIG_PATH)
|
|
62
58
|
cls._instance._kube_client = client.ApiClient()
|
|
63
59
|
return cls._instance
|
|
64
60
|
|
{sagemaker_hyperpod-3.0.2 → sagemaker_hyperpod-3.2.0}/src/sagemaker/hyperpod/cli/commands/cluster.py
RENAMED
|
@@ -14,8 +14,10 @@ import logging
|
|
|
14
14
|
import subprocess
|
|
15
15
|
import json
|
|
16
16
|
import sys
|
|
17
|
+
import signal
|
|
17
18
|
import botocore.config
|
|
18
19
|
from collections import defaultdict
|
|
20
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
19
21
|
from typing import Any, Dict, List, Optional, Tuple
|
|
20
22
|
|
|
21
23
|
import boto3
|
|
@@ -191,30 +193,33 @@ def list_cluster(
|
|
|
191
193
|
|
|
192
194
|
cluster_capacities: List[List[str]] = []
|
|
193
195
|
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
196
|
+
# Process clusters in parallel with limited concurrency
|
|
197
|
+
if cluster_names:
|
|
198
|
+
with ThreadPoolExecutor(max_workers=len(cluster_names)) as executor:
|
|
199
|
+
futures = {}
|
|
200
|
+
counter = 0
|
|
201
|
+
|
|
202
|
+
for cluster_name in cluster_names[:50]: # Limit to 50 clusters
|
|
203
|
+
future = executor.submit(
|
|
204
|
+
rate_limited_operation,
|
|
205
|
+
cluster_name=cluster_name,
|
|
206
|
+
validator=validator,
|
|
207
|
+
sm_client=sm_client,
|
|
208
|
+
region=region,
|
|
209
|
+
temp_config_file=f"{TEMP_KUBE_CONFIG_FILE}_{cluster_name}",
|
|
210
|
+
namespace=namespace,
|
|
211
|
+
)
|
|
212
|
+
futures[future] = cluster_name
|
|
213
|
+
|
|
214
|
+
for future in as_completed(futures):
|
|
215
|
+
cluster_name = futures[future]
|
|
216
|
+
try:
|
|
217
|
+
result = future.result()
|
|
218
|
+
if result: # Only add if cluster processing was successful
|
|
219
|
+
cluster_capacities.extend(result)
|
|
220
|
+
counter += 1
|
|
221
|
+
except Exception as e:
|
|
222
|
+
logger.error(f"Error processing cluster {cluster_name}: {e}")
|
|
218
223
|
|
|
219
224
|
headers = [
|
|
220
225
|
"Cluster",
|
|
@@ -245,10 +250,42 @@ def rate_limited_operation(
|
|
|
245
250
|
sm_client: BaseClient,
|
|
246
251
|
region: Optional[str],
|
|
247
252
|
temp_config_file: str,
|
|
248
|
-
cluster_capacities: List[List[str]],
|
|
249
253
|
namespace: Optional[List[str]],
|
|
250
|
-
) ->
|
|
254
|
+
) -> Optional[List[List[str]]]:
|
|
251
255
|
try:
|
|
256
|
+
cluster_capacities = [] # Initialize at the beginning
|
|
257
|
+
|
|
258
|
+
# Get cluster details to check instance count
|
|
259
|
+
cluster_response = sm_client.describe_cluster(ClusterName=cluster_name)
|
|
260
|
+
cluster_status = cluster_response.get('ClusterStatus', 'Unknown')
|
|
261
|
+
|
|
262
|
+
# Check if cluster has zero instances
|
|
263
|
+
instance_groups = cluster_response.get('InstanceGroups', [])
|
|
264
|
+
total_instances = sum(
|
|
265
|
+
group.get('CurrentCount', 0) for group in instance_groups
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
# If cluster has 0 instances, add it with 0 nodes
|
|
269
|
+
if total_instances == 0:
|
|
270
|
+
logger.info(f"Adding cluster {cluster_name} with 0 instances (status: {cluster_status})")
|
|
271
|
+
zero_instance_row = [
|
|
272
|
+
cluster_name,
|
|
273
|
+
"N/A", # InstanceType
|
|
274
|
+
0, # TotalNodes
|
|
275
|
+
0, # AcceleratorDevicesAvailable
|
|
276
|
+
0, # NodeHealthStatus=Schedulable
|
|
277
|
+
"N/A", # DeepHealthCheckStatus=Passed
|
|
278
|
+
]
|
|
279
|
+
|
|
280
|
+
# Add namespace columns with 0 values
|
|
281
|
+
if namespace:
|
|
282
|
+
for ns in namespace:
|
|
283
|
+
zero_instance_row.extend([0, 0]) # Total and Available accelerator devices
|
|
284
|
+
|
|
285
|
+
cluster_capacities.append(zero_instance_row)
|
|
286
|
+
return cluster_capacities
|
|
287
|
+
|
|
288
|
+
# Proceed with EKS validation for clusters with instances
|
|
252
289
|
eks_cluster_arn = validator.validate_cluster_and_get_eks_arn(
|
|
253
290
|
cluster_name, sm_client
|
|
254
291
|
)
|
|
@@ -256,10 +293,10 @@ def rate_limited_operation(
|
|
|
256
293
|
logger.warning(
|
|
257
294
|
f"Cannot find EKS cluster behind {cluster_name}, continue..."
|
|
258
295
|
)
|
|
259
|
-
return
|
|
296
|
+
return None
|
|
260
297
|
eks_cluster_name = get_name_from_arn(eks_cluster_arn)
|
|
261
298
|
_update_kube_config(eks_cluster_name, region, temp_config_file)
|
|
262
|
-
k8s_client = KubernetesClient(
|
|
299
|
+
k8s_client = KubernetesClient(config_file=temp_config_file)
|
|
263
300
|
nodes = k8s_client.list_node_with_temp_config(
|
|
264
301
|
temp_config_file, SAGEMAKER_HYPERPOD_NAME_LABEL
|
|
265
302
|
)
|
|
@@ -268,25 +305,27 @@ def rate_limited_operation(
|
|
|
268
305
|
ns_nominal_quota = {}
|
|
269
306
|
ns_quota_usage = {}
|
|
270
307
|
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
308
|
+
if namespace:
|
|
309
|
+
for ns in namespace:
|
|
310
|
+
sm_managed_namespace = k8s_client.get_sagemaker_managed_namespace(ns)
|
|
311
|
+
if sm_managed_namespace:
|
|
312
|
+
quota_allocation_id = sm_managed_namespace.metadata.labels[
|
|
313
|
+
SAGEMAKER_QUOTA_ALLOCATION_LABEL
|
|
314
|
+
]
|
|
315
|
+
cluster_queue_name = (
|
|
316
|
+
HYPERPOD_NAMESPACE_PREFIX
|
|
317
|
+
+ quota_allocation_id
|
|
318
|
+
+ SAGEMAKER_MANAGED_CLUSTER_QUEUE_SUFFIX
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
cluster_queue = k8s_client.get_cluster_queue(cluster_queue_name)
|
|
322
|
+
nominal_quota = _get_cluster_queue_nominal_quota(cluster_queue)
|
|
323
|
+
quota_usage = _get_cluster_queue_quota_usage(cluster_queue)
|
|
324
|
+
ns_nominal_quota[ns] = nominal_quota
|
|
325
|
+
ns_quota_usage[ns] = quota_usage
|
|
326
|
+
else:
|
|
327
|
+
ns_nominal_quota[ns] = {}
|
|
328
|
+
ns_quota_usage[ns] = {}
|
|
290
329
|
|
|
291
330
|
for instance_type, nodes_summary in nodes_info.items():
|
|
292
331
|
capacities = [
|
|
@@ -297,23 +336,26 @@ def rate_limited_operation(
|
|
|
297
336
|
nodes_summary["schedulable"],
|
|
298
337
|
nodes_summary["deep_health_check_passed"],
|
|
299
338
|
]
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
339
|
+
if namespace:
|
|
340
|
+
for ns in namespace:
|
|
341
|
+
capacities.append(
|
|
342
|
+
ns_nominal_quota.get(ns)
|
|
343
|
+
.get(instance_type, {})
|
|
344
|
+
.get(NVIDIA_GPU_RESOURCE_LIMIT_KEY, "N/A")
|
|
345
|
+
)
|
|
346
|
+
capacities.append(
|
|
347
|
+
_get_available_quota(
|
|
348
|
+
ns_nominal_quota.get(ns),
|
|
349
|
+
ns_quota_usage.get(ns),
|
|
350
|
+
instance_type,
|
|
351
|
+
NVIDIA_GPU_RESOURCE_LIMIT_KEY,
|
|
352
|
+
)
|
|
312
353
|
)
|
|
313
|
-
)
|
|
314
354
|
cluster_capacities.append(capacities)
|
|
355
|
+
return cluster_capacities
|
|
315
356
|
except Exception as e:
|
|
316
357
|
logger.error(f"Error processing cluster {cluster_name}: {e}, continue...")
|
|
358
|
+
return None
|
|
317
359
|
|
|
318
360
|
|
|
319
361
|
def _get_cluster_queue_nominal_quota(cluster_queue):
|
|
@@ -519,16 +561,26 @@ def set_cluster_context(
|
|
|
519
561
|
"""
|
|
520
562
|
if debug:
|
|
521
563
|
set_logging_level(logger, logging.DEBUG)
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
)
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
564
|
+
|
|
565
|
+
timeout = 60 # 1 minute
|
|
566
|
+
|
|
567
|
+
def timeout_handler(signum, frame):
|
|
568
|
+
raise TimeoutError(f"Operation timed out after {timeout} seconds")
|
|
569
|
+
|
|
570
|
+
# Set up timeout
|
|
571
|
+
signal.signal(signal.SIGALRM, timeout_handler)
|
|
572
|
+
signal.alarm(timeout)
|
|
573
|
+
|
|
531
574
|
try:
|
|
575
|
+
validator = ClusterValidator()
|
|
576
|
+
botocore_config = botocore.config.Config(
|
|
577
|
+
user_agent_extra=get_user_agent_extra_suffix()
|
|
578
|
+
)
|
|
579
|
+
session = boto3.Session(region_name=region) if region else boto3.Session()
|
|
580
|
+
if not validator.validate_aws_credential(session):
|
|
581
|
+
logger.error("Cannot connect to HyperPod cluster due to aws credentials error")
|
|
582
|
+
sys.exit(1)
|
|
583
|
+
|
|
532
584
|
sm_client = get_sagemaker_client(session, botocore_config)
|
|
533
585
|
hp_cluster_details = sm_client.describe_cluster(ClusterName=cluster_name)
|
|
534
586
|
logger.debug("Fetched hyperpod cluster details")
|
|
@@ -542,6 +594,14 @@ def set_cluster_context(
|
|
|
542
594
|
_update_kube_config(eks_name, region, None)
|
|
543
595
|
k8s_client = KubernetesClient()
|
|
544
596
|
k8s_client.set_context(eks_cluster_arn, namespace)
|
|
597
|
+
|
|
598
|
+
# Cancel the alarm if operation completes successfully
|
|
599
|
+
signal.alarm(0)
|
|
600
|
+
logger.info(f"Successfully connected to cluster {cluster_name}")
|
|
601
|
+
|
|
602
|
+
except TimeoutError as e:
|
|
603
|
+
logger.error("Timed out - Please check credentials, setup configurations and try again")
|
|
604
|
+
sys.exit(1)
|
|
545
605
|
except botocore.exceptions.NoRegionError:
|
|
546
606
|
logger.error(
|
|
547
607
|
f"Please ensure you configured AWS default region or use '--region' argument to specify the region"
|
|
@@ -552,6 +612,9 @@ def set_cluster_context(
|
|
|
552
612
|
f"Unexpected error happens when try to connect to cluster {cluster_name}. Error: {e}"
|
|
553
613
|
)
|
|
554
614
|
sys.exit(1)
|
|
615
|
+
finally:
|
|
616
|
+
# Ensure alarm is cancelled in all cases
|
|
617
|
+
signal.alarm(0)
|
|
555
618
|
|
|
556
619
|
|
|
557
620
|
@click.command()
|