sagemaker-hyperpod 3.0.0__tar.gz → 3.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. {sagemaker_hyperpod-3.0.0/src/sagemaker_hyperpod.egg-info → sagemaker_hyperpod-3.0.2}/PKG-INFO +68 -92
  2. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/README.md +66 -91
  3. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/pyproject.toml +1 -1
  4. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/setup.py +2 -1
  5. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/cluster.py +79 -33
  6. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/inference.py +148 -64
  7. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/training.py +16 -6
  8. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/hyp_cli.py +8 -8
  9. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/training_utils.py +87 -78
  10. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/config/metadata.py +3 -3
  11. sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/__init__.py +14 -0
  12. sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/constants.py +61 -0
  13. sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common/telemetry/telemetry_logging.py +187 -0
  14. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/hp_endpoint_config.py +12 -3
  15. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/hp_jumpstart_endpoint_config.py +13 -4
  16. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_endpoint.py +10 -0
  17. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +8 -0
  18. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/hp_jumpstart_endpoint.py +10 -0
  19. sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/training/__init__.py +6 -0
  20. sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_status.py → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_unified_config.py +165 -0
  21. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +12 -5
  22. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2/src/sagemaker_hyperpod.egg-info}/PKG-INFO +68 -92
  23. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/SOURCES.txt +5 -4
  24. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/requires.txt +1 -0
  25. sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/validators/__init__.py +0 -12
  26. sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/__init__.py +0 -9
  27. sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_config.py +0 -2977
  28. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/LICENSE +0 -0
  29. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/NOTICE +0 -0
  30. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/setup.cfg +0 -0
  31. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/__init__.py +0 -0
  32. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/__init__.py +0 -0
  33. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/__init__.py +0 -0
  34. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/clients/__init__.py +0 -0
  35. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/clients/kubernetes_client.py +0 -0
  36. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/commands/__init__.py +0 -0
  37. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/__init__.py +0 -0
  38. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/command_constants.py +0 -0
  39. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/exception_constants.py +0 -0
  40. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/hyperpod_instance_types.py +0 -0
  41. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/kueue_constants.py +0 -0
  42. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/constants/pytorch_constants.py +0 -0
  43. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/inference_utils.py +0 -0
  44. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/__init__.py +0 -0
  45. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/cancel_training_job.py +0 -0
  46. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/discover_namespaces.py +0 -0
  47. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/exec_command.py +0 -0
  48. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_logs.py +0 -0
  49. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_namespaces.py +0 -0
  50. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/get_training_job.py +0 -0
  51. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/list_pods.py +0 -0
  52. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/list_training_jobs.py +0 -0
  53. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/service/self_subject_access_review.py +0 -0
  54. {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/telemetry → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/templates}/__init__.py +0 -0
  55. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +0 -0
  56. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/utils.py +0 -0
  57. {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli/templates → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/cli/validators}/__init__.py +0 -0
  58. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/cluster_validator.py +0 -0
  59. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/job_validator.py +0 -0
  60. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/cli/validators/validator.py +0 -0
  61. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/__init__.py +0 -0
  62. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/config/__init__.py +0 -0
  63. {sagemaker_hyperpod-3.0.0/src/sagemaker/hyperpod/cli → sagemaker_hyperpod-3.0.2/src/sagemaker/hyperpod/common}/telemetry/user_agent.py +0 -0
  64. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/common/utils.py +0 -0
  65. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/__init__.py +0 -0
  66. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/__init__.py +0 -0
  67. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/inference/config/constants.py +0 -0
  68. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/MonitoringConfig.py +0 -0
  69. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/__init__.py +0 -0
  70. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/constants.py +0 -0
  71. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/observability/utils.py +0 -0
  72. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker/hyperpod/training/config/__init__.py +0 -0
  73. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/dependency_links.txt +0 -0
  74. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/entry_points.txt +0 -0
  75. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/top_level.txt +0 -0
  76. {sagemaker_hyperpod-3.0.0 → sagemaker_hyperpod-3.0.2}/src/sagemaker_hyperpod.egg-info/zip-safe +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sagemaker-hyperpod
3
- Version: 3.0.0
3
+ Version: 3.0.2
4
4
  Summary: Amazon SageMaker HyperPod SDK and CLI
5
5
  Home-page: https://github.com/aws/sagemaker-hyperpod-cli
6
6
  Author: Amazon Web Services
@@ -38,6 +38,7 @@ Requires-Dist: zstandard==0.15.2
38
38
  Requires-Dist: pytest==8.3.2
39
39
  Requires-Dist: pytest-cov==5.0.0
40
40
  Requires-Dist: pytest-order==1.3.0
41
+ Requires-Dist: pytest-dependency==0.6.0
41
42
  Requires-Dist: tox==4.18.0
42
43
  Requires-Dist: ruff==0.6.2
43
44
  Requires-Dist: hera-workflows==5.16.3
@@ -58,6 +59,8 @@ The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool th
58
59
 
59
60
  This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
60
61
 
62
+ Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
63
+
61
64
  ## Table of Contents
62
65
  - [Overview](#overview)
63
66
  - [Prerequisites](#prerequisites)
@@ -74,8 +77,8 @@ This documentation serves as a reference for the available HyperPod CLI commands
74
77
  - [Training](#training-)
75
78
  - [Inference](#inference-)
76
79
  - [SDK](#sdk-)
77
- - [Training](#training-)
78
- - [Inference](#inference)
80
+ - [Training](#training-sdk)
81
+ - [Inference](#inference-sdk)
79
82
 
80
83
 
81
84
  ## Overview
@@ -125,27 +128,9 @@ SageMaker HyperPod CLI currently supports start training job with:
125
128
  1. Verify if the installation succeeded by running the following command.
126
129
 
127
130
  ```
128
- hyperpod --help
131
+ hyp --help
129
132
  ```
130
133
 
131
- 1. If you have a running HyperPod cluster, you can try to run a training job using the sample configuration file provided at ```/examples/basic-job-example-config.yaml```.
132
- - Get your HyperPod clusters to show their capacities.
133
- ```
134
- hyperpod get-clusters
135
- ```
136
- - Get your HyperPod clusters to show their capacities and quota allocation info for a team.
137
- ```
138
- hyperpod get-clusters -n hyperpod-ns-<team-name>
139
- ```
140
- - Connect to one HyperPod cluster and specify a namespace you have access to.
141
- ```
142
- hyperpod connect-cluster --cluster-name <cluster-name>
143
- ```
144
- - Start a job in your cluster. Change the `instance_type` in the yaml file to be same as the one in your HyperPod cluster. Also change the `namespace` you want to submit a job to, the example uses kubeflow namespace. You need to have installed PyTorch in your cluster.
145
- ```
146
- hyperpod start-job --config-file ./examples/basic-job-example-config.yaml
147
- ```
148
-
149
134
  ## Usage
150
135
 
151
136
  The HyperPod CLI provides the following commands:
@@ -159,8 +144,8 @@ The HyperPod CLI provides the following commands:
159
144
  - [Training](#training-)
160
145
  - [Inference](#inference-)
161
146
  - [SDK](#sdk-)
162
- - [Training](#training-)
163
- - [Inference](#inference)
147
+ - [Training](#training-sdk)
148
+ - [Inference](#inference-sdk)
164
149
 
165
150
 
166
151
  ### Getting Cluster information
@@ -227,8 +212,8 @@ hyp create hyp-pytorch-job \
227
212
  --version 1.0 \
228
213
  --job-name test-pytorch-job \
229
214
  --image pytorch/pytorch:latest \
230
- --command '["python", "train.py"]' \
231
- --args '["--epochs", "10", "--batch-size", "32"]' \
215
+ --command '[python, train.py]' \
216
+ --args '[--epochs=10, --batch-size=32]' \
232
217
  --environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
233
218
  --pull-policy "IfNotPresent" \
234
219
  --instance-type ml.p4d.24xlarge \
@@ -239,9 +224,8 @@ hyp create hyp-pytorch-job \
239
224
  --queue-name "training-queue" \
240
225
  --priority "high" \
241
226
  --max-retry 3 \
242
- --volumes '["data-vol", "model-vol", "checkpoint-vol"]' \
243
- --persistent-volume-claims '["shared-data-pvc", "model-registry-pvc"]' \
244
- --output-s3-uri s3://my-bucket/model-artifacts
227
+ --volume name=model-data,type=hostPath,mount_path=/data,path=/data \
228
+ --volume name=training-output,type=pvc,mount_path=/data,claim_name=my-pvc,read_only=false
245
229
  ```
246
230
 
247
231
  Key required parameters explained:
@@ -250,8 +234,6 @@ Key required parameters explained:
250
234
 
251
235
  --image: Docker image containing your training environment
252
236
 
253
- This command starts a training job named test-pytorch-job. The --output-s3-uri specifies where the trained model artifacts will be stored, for example, s3://my-bucket/model-artifacts. Note this location, as you’ll need it for deploying the custom model.
254
-
255
237
  ### Inference
256
238
 
257
239
  #### Creating a JumpstartModel Endpoint
@@ -320,15 +302,16 @@ hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
320
302
 
321
303
  Along with the CLI, we also have SDKs available that can perform the training and inference functionalities that the CLI performs
322
304
 
323
- ### Training
305
+ ### Training SDK
324
306
 
325
307
  #### Creating a Training Job
326
308
 
327
309
  ```
328
310
 
329
- from sagemaker.hyperpod import HyperPodPytorchJob
330
- from sagemaker.hyperpod.job
331
- import ReplicaSpec, Template, Spec, Container, Resources, RunPolicy, Metadata
311
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
312
+ from sagemaker.hyperpod.training
313
+ import ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
314
+ from sagemaker.hyperpod.common.config import Metadata
332
315
 
333
316
  # Define job specifications
334
317
  nproc_per_node = "1" # Number of processes per node
@@ -343,7 +326,7 @@ replica_specs =
343
326
  (
344
327
  containers =
345
328
  [
346
- Container
329
+ Containers
347
330
  (
348
331
  # Container name
349
332
  name="container-name",
@@ -384,8 +367,6 @@ pytorch_job = HyperPodPytorchJob
384
367
  replica_specs = replica_specs,
385
368
  # Run policy
386
369
  run_policy = run_policy,
387
- # S3 location for artifacts
388
- output_s3_uri="s3://my-bucket/model-artifacts"
389
370
  )
390
371
  # Launch the job
391
372
  pytorch_job.create()
@@ -395,7 +376,7 @@ pytorch_job.create()
395
376
 
396
377
 
397
378
 
398
- ### Inference
379
+ ### Inference SDK
399
380
 
400
381
  #### Creating a JumpstartModel Endpoint
401
382
 
@@ -405,24 +386,21 @@ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io
405
386
  from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
406
387
  from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
407
388
 
408
- model = Model(
409
- model_id="deepseek-llm-r1-distill-qwen-1-5b",
410
- model_version="2.0.4"
389
+ model=Model(
390
+ model_id='deepseek-llm-r1-distill-qwen-1-5b',
391
+ model_version='2.0.4',
411
392
  )
412
-
413
- server = Server(
414
- instance_type="ml.g5.8xlarge"
393
+ server=Server(
394
+ instance_type='ml.g5.8xlarge',
415
395
  )
396
+ endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
397
+ tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
416
398
 
417
- endpoint_name = SageMakerEndpoint(name="endpoint-jumpstart")
418
-
419
- tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
420
-
421
- js_endpoint = HPJumpStartEndpoint(
399
+ js_endpoint=HPJumpStartEndpoint(
422
400
  model=model,
423
401
  server=server,
424
402
  sage_maker_endpoint=endpoint_name,
425
- tls_config=tls_config
403
+ tls_config=tls_config,
426
404
  )
427
405
 
428
406
  js_endpoint.create()
@@ -438,51 +416,51 @@ print(response)
438
416
  ```
439
417
 
440
418
 
441
- #### Creating a Custom Inference Endpoint
419
+ #### Creating a Custom Inference Endpoint (with S3)
442
420
 
443
421
  ```
444
- from sagemaker.hyperpod.inference.config.hp_custom_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig, EnvironmentVariables
445
- from sagemaker.hyperpod.inference.hp_custom_endpoint import HPCustomEndpoint
422
+ from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
423
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
446
424
 
447
- model = Model(
448
- model_source_type="s3",
449
- model_location="test-pytorch-job/model.tar.gz",
450
- s3_bucket_name="my-bucket",
451
- s3_region="us-east-2",
452
- prefetch_enabled=True
425
+ model_source_config = ModelSourceConfig(
426
+ model_source_type='s3',
427
+ model_location="<my-model-folder-in-s3>",
428
+ s3_storage=S3Storage(
429
+ bucket_name='<my-model-artifacts-bucket>',
430
+ region='us-east-2',
431
+ ),
453
432
  )
454
433
 
455
- server = Server(
456
- instance_type="ml.g5.8xlarge",
457
- image_uri="763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0",
458
- container_port=8080,
459
- model_volume_mount_name="model-weights"
460
- )
434
+ environment_variables = [
435
+ EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
436
+ EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
437
+ EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
438
+ EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
439
+ EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
440
+ ]
461
441
 
462
- resources = {
463
- "requests": {"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
464
- "limits": {"nvidia.com/gpu": 1}
465
- }
466
-
467
- env = EnvironmentVariables(
468
- HF_MODEL_ID="/opt/ml/model",
469
- SAGEMAKER_PROGRAM="inference.py",
470
- SAGEMAKER_SUBMIT_DIRECTORY="/opt/ml/model/code",
471
- MODEL_CACHE_ROOT="/opt/ml/model",
472
- SAGEMAKER_ENV="1"
442
+ worker = Worker(
443
+ image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
444
+ model_volume_mount=ModelVolumeMount(
445
+ name='model-weights',
446
+ ),
447
+ model_invocation_port=ModelInvocationPort(container_port=8080),
448
+ resources=Resources(
449
+ requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
450
+ limits={"nvidia.com/gpu": 1}
451
+ ),
452
+ environment_variables=environment_variables,
473
453
  )
474
454
 
475
- endpoint_name = SageMakerEndpoint(name="endpoint-custom-pytorch")
455
+ tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
476
456
 
477
- tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
478
-
479
- custom_endpoint = HPCustomEndpoint(
480
- model=model,
481
- server=server,
482
- resources=resources,
483
- environment=env,
484
- sage_maker_endpoint=endpoint_name,
457
+ custom_endpoint = HPEndpoint(
458
+ endpoint_name='<my-endpoint-name>',
459
+ instance_type='ml.g5.8xlarge',
460
+ model_name='deepseek15b-test-model-name',
485
461
  tls_config=tls_config,
462
+ model_source_config=model_source_config,
463
+ worker=worker,
486
464
  )
487
465
 
488
466
  custom_endpoint.create()
@@ -499,19 +477,17 @@ print(response)
499
477
  #### Managing an Endpoint
500
478
 
501
479
  ```
502
- endpoint_iterator = HPJumpStartEndpoint.list()
503
- for endpoint in endpoint_iterator:
504
- print(endpoint.name, endpoint.status)
480
+ endpoint_list = HPEndpoint.list()
481
+ print(endpoint_list[0])
505
482
 
506
- logs = js_endpoint.get_logs()
507
- print(logs)
483
+ print(custom_endpoint.get_operator_logs(since_hours=0.5))
508
484
 
509
485
  ```
510
486
 
511
487
  #### Deleting an Endpoint
512
488
 
513
489
  ```
514
- js_endpoint.delete()
490
+ custom_endpoint.delete()
515
491
 
516
492
  ```
517
493
 
@@ -5,6 +5,8 @@ The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool th
5
5
 
6
6
  This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
7
7
 
8
+ Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
9
+
8
10
  ## Table of Contents
9
11
  - [Overview](#overview)
10
12
  - [Prerequisites](#prerequisites)
@@ -21,8 +23,8 @@ This documentation serves as a reference for the available HyperPod CLI commands
21
23
  - [Training](#training-)
22
24
  - [Inference](#inference-)
23
25
  - [SDK](#sdk-)
24
- - [Training](#training-)
25
- - [Inference](#inference)
26
+ - [Training](#training-sdk)
27
+ - [Inference](#inference-sdk)
26
28
 
27
29
 
28
30
  ## Overview
@@ -72,27 +74,9 @@ SageMaker HyperPod CLI currently supports start training job with:
72
74
  1. Verify if the installation succeeded by running the following command.
73
75
 
74
76
  ```
75
- hyperpod --help
77
+ hyp --help
76
78
  ```
77
79
 
78
- 1. If you have a running HyperPod cluster, you can try to run a training job using the sample configuration file provided at ```/examples/basic-job-example-config.yaml```.
79
- - Get your HyperPod clusters to show their capacities.
80
- ```
81
- hyperpod get-clusters
82
- ```
83
- - Get your HyperPod clusters to show their capacities and quota allocation info for a team.
84
- ```
85
- hyperpod get-clusters -n hyperpod-ns-<team-name>
86
- ```
87
- - Connect to one HyperPod cluster and specify a namespace you have access to.
88
- ```
89
- hyperpod connect-cluster --cluster-name <cluster-name>
90
- ```
91
- - Start a job in your cluster. Change the `instance_type` in the yaml file to be same as the one in your HyperPod cluster. Also change the `namespace` you want to submit a job to, the example uses kubeflow namespace. You need to have installed PyTorch in your cluster.
92
- ```
93
- hyperpod start-job --config-file ./examples/basic-job-example-config.yaml
94
- ```
95
-
96
80
  ## Usage
97
81
 
98
82
  The HyperPod CLI provides the following commands:
@@ -106,8 +90,8 @@ The HyperPod CLI provides the following commands:
106
90
  - [Training](#training-)
107
91
  - [Inference](#inference-)
108
92
  - [SDK](#sdk-)
109
- - [Training](#training-)
110
- - [Inference](#inference)
93
+ - [Training](#training-sdk)
94
+ - [Inference](#inference-sdk)
111
95
 
112
96
 
113
97
  ### Getting Cluster information
@@ -174,8 +158,8 @@ hyp create hyp-pytorch-job \
174
158
  --version 1.0 \
175
159
  --job-name test-pytorch-job \
176
160
  --image pytorch/pytorch:latest \
177
- --command '["python", "train.py"]' \
178
- --args '["--epochs", "10", "--batch-size", "32"]' \
161
+ --command '[python, train.py]' \
162
+ --args '[--epochs=10, --batch-size=32]' \
179
163
  --environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
180
164
  --pull-policy "IfNotPresent" \
181
165
  --instance-type ml.p4d.24xlarge \
@@ -186,9 +170,8 @@ hyp create hyp-pytorch-job \
186
170
  --queue-name "training-queue" \
187
171
  --priority "high" \
188
172
  --max-retry 3 \
189
- --volumes '["data-vol", "model-vol", "checkpoint-vol"]' \
190
- --persistent-volume-claims '["shared-data-pvc", "model-registry-pvc"]' \
191
- --output-s3-uri s3://my-bucket/model-artifacts
173
+ --volume name=model-data,type=hostPath,mount_path=/data,path=/data \
174
+ --volume name=training-output,type=pvc,mount_path=/data,claim_name=my-pvc,read_only=false
192
175
  ```
193
176
 
194
177
  Key required parameters explained:
@@ -197,8 +180,6 @@ Key required parameters explained:
197
180
 
198
181
  --image: Docker image containing your training environment
199
182
 
200
- This command starts a training job named test-pytorch-job. The --output-s3-uri specifies where the trained model artifacts will be stored, for example, s3://my-bucket/model-artifacts. Note this location, as you’ll need it for deploying the custom model.
201
-
202
183
  ### Inference
203
184
 
204
185
  #### Creating a JumpstartModel Endpoint
@@ -267,15 +248,16 @@ hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
267
248
 
268
249
  Along with the CLI, we also have SDKs available that can perform the training and inference functionalities that the CLI performs
269
250
 
270
- ### Training
251
+ ### Training SDK
271
252
 
272
253
  #### Creating a Training Job
273
254
 
274
255
  ```
275
256
 
276
- from sagemaker.hyperpod import HyperPodPytorchJob
277
- from sagemaker.hyperpod.job
278
- import ReplicaSpec, Template, Spec, Container, Resources, RunPolicy, Metadata
257
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
258
+ from sagemaker.hyperpod.training
259
+ import ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
260
+ from sagemaker.hyperpod.common.config import Metadata
279
261
 
280
262
  # Define job specifications
281
263
  nproc_per_node = "1" # Number of processes per node
@@ -290,7 +272,7 @@ replica_specs =
290
272
  (
291
273
  containers =
292
274
  [
293
- Container
275
+ Containers
294
276
  (
295
277
  # Container name
296
278
  name="container-name",
@@ -331,8 +313,6 @@ pytorch_job = HyperPodPytorchJob
331
313
  replica_specs = replica_specs,
332
314
  # Run policy
333
315
  run_policy = run_policy,
334
- # S3 location for artifacts
335
- output_s3_uri="s3://my-bucket/model-artifacts"
336
316
  )
337
317
  # Launch the job
338
318
  pytorch_job.create()
@@ -342,7 +322,7 @@ pytorch_job.create()
342
322
 
343
323
 
344
324
 
345
- ### Inference
325
+ ### Inference SDK
346
326
 
347
327
  #### Creating a JumpstartModel Endpoint
348
328
 
@@ -352,24 +332,21 @@ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io
352
332
  from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
353
333
  from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
354
334
 
355
- model = Model(
356
- model_id="deepseek-llm-r1-distill-qwen-1-5b",
357
- model_version="2.0.4"
335
+ model=Model(
336
+ model_id='deepseek-llm-r1-distill-qwen-1-5b',
337
+ model_version='2.0.4',
358
338
  )
359
-
360
- server = Server(
361
- instance_type="ml.g5.8xlarge"
339
+ server=Server(
340
+ instance_type='ml.g5.8xlarge',
362
341
  )
342
+ endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
343
+ tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket>')
363
344
 
364
- endpoint_name = SageMakerEndpoint(name="endpoint-jumpstart")
365
-
366
- tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
367
-
368
- js_endpoint = HPJumpStartEndpoint(
345
+ js_endpoint=HPJumpStartEndpoint(
369
346
  model=model,
370
347
  server=server,
371
348
  sage_maker_endpoint=endpoint_name,
372
- tls_config=tls_config
349
+ tls_config=tls_config,
373
350
  )
374
351
 
375
352
  js_endpoint.create()
@@ -385,51 +362,51 @@ print(response)
385
362
  ```
386
363
 
387
364
 
388
- #### Creating a Custom Inference Endpoint
365
+ #### Creating a Custom Inference Endpoint (with S3)
389
366
 
390
367
  ```
391
- from sagemaker.hyperpod.inference.config.hp_custom_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig, EnvironmentVariables
392
- from sagemaker.hyperpod.inference.hp_custom_endpoint import HPCustomEndpoint
368
+ from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
369
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
393
370
 
394
- model = Model(
395
- model_source_type="s3",
396
- model_location="test-pytorch-job/model.tar.gz",
397
- s3_bucket_name="my-bucket",
398
- s3_region="us-east-2",
399
- prefetch_enabled=True
371
+ model_source_config = ModelSourceConfig(
372
+ model_source_type='s3',
373
+ model_location="<my-model-folder-in-s3>",
374
+ s3_storage=S3Storage(
375
+ bucket_name='<my-model-artifacts-bucket>',
376
+ region='us-east-2',
377
+ ),
400
378
  )
401
379
 
402
- server = Server(
403
- instance_type="ml.g5.8xlarge",
404
- image_uri="763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0",
405
- container_port=8080,
406
- model_volume_mount_name="model-weights"
407
- )
380
+ environment_variables = [
381
+ EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
382
+ EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
383
+ EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
384
+ EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
385
+ EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
386
+ ]
408
387
 
409
- resources = {
410
- "requests": {"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
411
- "limits": {"nvidia.com/gpu": 1}
412
- }
413
-
414
- env = EnvironmentVariables(
415
- HF_MODEL_ID="/opt/ml/model",
416
- SAGEMAKER_PROGRAM="inference.py",
417
- SAGEMAKER_SUBMIT_DIRECTORY="/opt/ml/model/code",
418
- MODEL_CACHE_ROOT="/opt/ml/model",
419
- SAGEMAKER_ENV="1"
388
+ worker = Worker(
389
+ image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
390
+ model_volume_mount=ModelVolumeMount(
391
+ name='model-weights',
392
+ ),
393
+ model_invocation_port=ModelInvocationPort(container_port=8080),
394
+ resources=Resources(
395
+ requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
396
+ limits={"nvidia.com/gpu": 1}
397
+ ),
398
+ environment_variables=environment_variables,
420
399
  )
421
400
 
422
- endpoint_name = SageMakerEndpoint(name="endpoint-custom-pytorch")
401
+ tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
423
402
 
424
- tls_config = TlsConfig(tls_certificate_output_s3_uri="s3://sample-bucket")
425
-
426
- custom_endpoint = HPCustomEndpoint(
427
- model=model,
428
- server=server,
429
- resources=resources,
430
- environment=env,
431
- sage_maker_endpoint=endpoint_name,
403
+ custom_endpoint = HPEndpoint(
404
+ endpoint_name='<my-endpoint-name>',
405
+ instance_type='ml.g5.8xlarge',
406
+ model_name='deepseek15b-test-model-name',
432
407
  tls_config=tls_config,
408
+ model_source_config=model_source_config,
409
+ worker=worker,
433
410
  )
434
411
 
435
412
  custom_endpoint.create()
@@ -446,19 +423,17 @@ print(response)
446
423
  #### Managing an Endpoint
447
424
 
448
425
  ```
449
- endpoint_iterator = HPJumpStartEndpoint.list()
450
- for endpoint in endpoint_iterator:
451
- print(endpoint.name, endpoint.status)
426
+ endpoint_list = HPEndpoint.list()
427
+ print(endpoint_list[0])
452
428
 
453
- logs = js_endpoint.get_logs()
454
- print(logs)
429
+ print(custom_endpoint.get_operator_logs(since_hours=0.5))
455
430
 
456
431
  ```
457
432
 
458
433
  #### Deleting an Endpoint
459
434
 
460
435
  ```
461
- js_endpoint.delete()
436
+ custom_endpoint.delete()
462
437
 
463
438
  ```
464
439
 
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
5
5
  [project]
6
6
  dynamic = ["dependencies"]
7
7
  name = "sagemaker-hyperpod"
8
- version = "3.0.0"
8
+ version = "3.0.2"
9
9
  description = "Amazon SageMaker HyperPod SDK and CLI"
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.8"
@@ -47,7 +47,7 @@ for root, dirs, files in os.walk(
47
47
  setup(
48
48
  data_files=sagemaker_hyperpod_recipes,
49
49
  name="sagemaker-hyperpod",
50
- version="3.0.0",
50
+ version="3.0.2",
51
51
  description="Amazon SageMaker HyperPod SDK and CLI",
52
52
  long_description=open("README.md").read(),
53
53
  long_description_content_type="text/markdown",
@@ -81,6 +81,7 @@ setup(
81
81
  "pytest==8.3.2",
82
82
  "pytest-cov==5.0.0",
83
83
  "pytest-order==1.3.0",
84
+ "pytest-dependency==0.6.0",
84
85
  "tox==4.18.0",
85
86
  "ruff==0.6.2",
86
87
  "hera-workflows==5.16.3",