sagemaker-hyperpod 3.2.1__tar.gz → 3.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. sagemaker_hyperpod-3.3.0/PKG-INFO +1096 -0
  2. sagemaker_hyperpod-3.3.0/README.md +1041 -0
  3. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/pyproject.toml +1 -1
  4. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/setup.cfg +2 -1
  5. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/setup.py +1 -1
  6. sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/__init__.py +9 -0
  7. sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/cluster_stack_utils.py +498 -0
  8. sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/cluster_utils.py +145 -0
  9. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/cluster.py +31 -0
  10. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/cluster_stack.py +64 -69
  11. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/inference.py +9 -21
  12. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/init.py +55 -101
  13. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/training.py +1 -31
  14. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/common_utils.py +52 -2
  15. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/init_constants.py +57 -29
  16. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/hyp_cli.py +2 -2
  17. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/inference_utils.py +3 -4
  18. sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/init_utils.py +536 -0
  19. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/training_utils.py +2 -2
  20. sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/type_handler_utils.py +174 -0
  21. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cluster_management/hp_cluster_stack.py +225 -31
  22. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/telemetry_logging.py +20 -1
  23. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/utils.py +6 -1
  24. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_endpoint.py +11 -6
  25. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +2 -1
  26. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_jumpstart_endpoint.py +11 -6
  27. sagemaker_hyperpod-3.3.0/src/sagemaker_hyperpod.egg-info/PKG-INFO +1096 -0
  28. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/SOURCES.txt +3 -4
  29. sagemaker_hyperpod-3.2.1/PKG-INFO +0 -504
  30. sagemaker_hyperpod-3.2.1/README.md +0 -449
  31. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/init_utils.py +0 -949
  32. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/cfn_cluster_creation.py +0 -948
  33. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_custom_endpoint_template.py +0 -68
  34. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_js_endpoint_template.py +0 -17
  35. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +0 -68
  36. sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/observability/__init__.py +0 -0
  37. sagemaker_hyperpod-3.2.1/src/sagemaker_hyperpod.egg-info/PKG-INFO +0 -504
  38. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/LICENSE +0 -0
  39. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/NOTICE +0 -0
  40. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/__init__.py +0 -0
  41. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/__init__.py +0 -0
  42. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/clients/__init__.py +0 -0
  43. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/clients/kubernetes_client.py +0 -0
  44. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/commands}/__init__.py +0 -0
  45. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/commands → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/constants}/__init__.py +0 -0
  46. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/command_constants.py +0 -0
  47. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/exception_constants.py +0 -0
  48. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/hyperpod_instance_types.py +0 -0
  49. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/kueue_constants.py +0 -0
  50. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/pytorch_constants.py +0 -0
  51. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/__init__.py +0 -0
  52. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/cancel_training_job.py +0 -0
  53. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/discover_namespaces.py +0 -0
  54. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/exec_command.py +0 -0
  55. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_logs.py +0 -0
  56. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_namespaces.py +0 -0
  57. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_training_job.py +0 -0
  58. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/list_pods.py +0 -0
  59. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/list_training_jobs.py +0 -0
  60. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/self_subject_access_review.py +0 -0
  61. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/templates/__init__.py +0 -0
  62. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/utils.py +0 -0
  63. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/__init__.py +0 -0
  64. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/cluster_validator.py +0 -0
  65. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/job_validator.py +0 -0
  66. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/validator.py +0 -0
  67. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/constants → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cluster_management}/__init__.py +0 -0
  68. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cluster_management → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cluster_management/config}/__init__.py +0 -0
  69. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cluster_management/config/hp_cluster_stack_config.py +0 -0
  70. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cluster_management/config → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/common}/__init__.py +0 -0
  71. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/cli_decorators.py +0 -0
  72. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/config/__init__.py +0 -0
  73. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/config/metadata.py +0 -0
  74. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/exceptions/__init__.py +0 -0
  75. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/__init__.py +0 -0
  76. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/constants.py +0 -0
  77. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/user_agent.py +0 -0
  78. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/__init__.py +0 -0
  79. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/common → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/inference/config}/__init__.py +0 -0
  80. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/constants.py +0 -0
  81. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/hp_endpoint_config.py +0 -0
  82. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/hp_jumpstart_endpoint_config.py +0 -0
  83. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/jumpstart_public_hub_visualization_utils.py +0 -0
  84. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/MonitoringConfig.py +0 -0
  85. {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/inference/config → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/observability}/__init__.py +0 -0
  86. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/constants.py +0 -0
  87. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/utils.py +0 -0
  88. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/__init__.py +0 -0
  89. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/config/__init__.py +0 -0
  90. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_unified_config.py +0 -0
  91. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +0 -0
  92. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/quota_allocation_util.py +0 -0
  93. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/dependency_links.txt +0 -0
  94. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/entry_points.txt +0 -0
  95. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/requires.txt +0 -0
  96. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/top_level.txt +0 -0
  97. {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/zip-safe +0 -0
@@ -0,0 +1,1096 @@
1
+ Metadata-Version: 2.4
2
+ Name: sagemaker-hyperpod
3
+ Version: 3.3.0
4
+ Summary: Amazon SageMaker HyperPod SDK and CLI
5
+ Home-page: https://github.com/aws/sagemaker-hyperpod-cli
6
+ Author: Amazon Web Services
7
+ License: Apache-2.0
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: Apache Software License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.8
13
+ Classifier: Programming Language :: Python :: 3.9
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Requires-Python: >=3.8
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ License-File: NOTICE
20
+ Requires-Dist: click==8.1.7
21
+ Requires-Dist: awscli>=1.34.9
22
+ Requires-Dist: awscli-cwlogs>=1.4.6
23
+ Requires-Dist: boto3<2.0,>=1.35.3
24
+ Requires-Dist: botocore>=1.35.6
25
+ Requires-Dist: kubernetes==33.1.0
26
+ Requires-Dist: pyyaml==6.0.2
27
+ Requires-Dist: ratelimit==2.2.1
28
+ Requires-Dist: tabulate==0.9.0
29
+ Requires-Dist: itables>=2.2.2
30
+ Requires-Dist: jinja2>=3.1.2
31
+ Requires-Dist: ipywidgets>=8.1.7
32
+ Requires-Dist: hydra-core==1.3.2
33
+ Requires-Dist: omegaconf==2.3
34
+ Requires-Dist: pynvml==11.4.1
35
+ Requires-Dist: requests==2.32.4
36
+ Requires-Dist: tqdm==4.66.5
37
+ Requires-Dist: zstandard==0.15.2
38
+ Requires-Dist: pytest==8.3.2
39
+ Requires-Dist: pytest-cov==5.0.0
40
+ Requires-Dist: pytest-order==1.3.0
41
+ Requires-Dist: pytest-dependency==0.6.0
42
+ Requires-Dist: tox==4.18.0
43
+ Requires-Dist: ruff==0.6.2
44
+ Requires-Dist: hera-workflows==5.16.3
45
+ Requires-Dist: sagemaker-core<2.0.0
46
+ Requires-Dist: pydantic<3.0.0,>=2.10.6
47
+ Requires-Dist: hyperpod-pytorch-job-template<2.0.0,>=1.0.0
48
+ Requires-Dist: hyperpod-custom-inference-template<2.0.0,>=1.0.0
49
+ Requires-Dist: hyperpod-jumpstart-inference-template<2.0.0,>=1.0.0
50
+ Requires-Dist: hyperpod-cluster-stack-template<2.0.0,>=1.0.0
51
+ Dynamic: home-page
52
+ Dynamic: license-file
53
+ Dynamic: requires-dist
54
+ Dynamic: requires-python
55
+
56
+
57
+ # SageMaker HyperPod command-line interface
58
+
59
+ The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool that helps manage clusters, training jobs, and inference endpoints on the SageMaker HyperPod clusters orchestrated by Amazon EKS.
60
+
61
+ This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
62
+
63
+ Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
64
+
65
+ ## Table of Contents
66
+ - [Overview](#overview)
67
+ - [Prerequisites](#prerequisites)
68
+ - [Platform Support](#platform-support)
69
+ - [ML Framework Support](#ml-framework-support)
70
+ - [Installation](#installation)
71
+ - [Usage](#usage)
72
+ - [Getting Started](#getting-started)
73
+ - [CLI](#cli)
74
+ - [Cluster Management](#cluster-management)
75
+ - [Training](#training)
76
+ - [Inference](#inference)
77
+ - [Jumpstart Endpoint](#jumpstart-endpoint-creation)
78
+ - [Custom Endpoint](#custom-endpoint-creation)
79
+ - [SDK](#sdk)
80
+ - [Cluster Management](#cluster-management-sdk)
81
+ - [Training](#training-sdk)
82
+ - [Inference](#inference-sdk)
83
+ - [Examples](#examples)
84
+
85
+
86
+ ## Overview
87
+
88
+ The SageMaker HyperPod CLI is a tool that helps create training jobs and inference endpoint deployments to the Amazon SageMaker HyperPod clusters orchestrated by Amazon EKS. It provides a set of commands for managing the full lifecycle of jobs, including create, describe, list, and delete operations, as well as accessing pod and operator logs where applicable. The CLI is designed to abstract away the complexity of working directly with Kubernetes for these core actions of managing jobs on SageMaker HyperPod clusters orchestrated by Amazon EKS.
89
+
90
+ ## Prerequisites
91
+
92
+ ### Region Configuration
93
+
94
+ **Important**: For commands that accept the `--region` option, if no region is explicitly provided, the command will use the default region from your AWS credentials configuration.
95
+
96
+ ### Prerequisites for Training
97
+
98
+ - HyperPod CLI currently supports starting PyTorchJobs. To start a job, you need to install Training Operator first.
99
+ - You can follow [pytorch operator doc](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-eks-operator-install.html) to install it.
100
+
101
+ ### Prerequisites for Inference
102
+
103
+ - HyperPod CLI supports creating Inference Endpoints through jumpstart and through custom Endpoint config
104
+ - You can follow [inference operator doc](https://github.com/aws/sagemaker-hyperpod-cli/tree/master/helm_chart/HyperPodHelmChart/charts/inference-operator) to install it.
105
+
106
+ ## Platform Support
107
+
108
+ SageMaker HyperPod CLI currently supports Linux and MacOS platforms. Windows platform is not supported now.
109
+
110
+ ## ML Framework Support
111
+
112
+ SageMaker HyperPod CLI currently supports start training job with:
113
+ - PyTorch ML Framework. Version requirements: PyTorch >= 1.10
114
+
115
+ ## Installation
116
+
117
+ 1. Make sure that your local python version is 3.8, 3.9, 3.10 or 3.11.
118
+
119
+ 2. Install the sagemaker-hyperpod-cli package.
120
+
121
+ ```bash
122
+ pip install sagemaker-hyperpod
123
+ ```
124
+
125
+ 3. Verify if the installation succeeded by running the following command.
126
+
127
+ ```bash
128
+ hyp --help
129
+ ```
130
+
131
+ ## Usage
132
+
133
+ The HyperPod CLI provides the following commands:
134
+
135
+ - [Getting Started](#getting-started)
136
+ - [CLI](#cli)
137
+ - [Cluster Management](#cluster-management)
138
+ - [Training](#training)
139
+ - [Inference](#inference)
140
+ - [Jumpstart Endpoint](#jumpstart-endpoint-creation)
141
+ - [Custom Endpoint](#custom-endpoint-creation)
142
+ - [SDK](#sdk)
143
+ - [Cluster Management](#cluster-management-sdk)
144
+ - [Training](#training-sdk)
145
+ - [Inference](#inference-sdk)
146
+
147
+
148
+ ### Getting Started
149
+
150
+ #### Getting Cluster information
151
+
152
+ This command lists the available SageMaker HyperPod clusters and their capacity information.
153
+
154
+ ```bash
155
+ hyp list-cluster
156
+ ```
157
+
158
+ | Option | Type | Description |
159
+ |--------|------|-------------|
160
+ | `--region <region>` | Optional | The region that the SageMaker HyperPod and EKS clusters are located. If not specified, it will be set to the region from the current AWS account credentials. |
161
+ | `--namespace <namespace>` | Optional | The namespace that users want to check the quota with. Only the SageMaker managed namespaces are supported. |
162
+ | `--output <json\|table>` | Optional | The output format. Available values are `table` and `json`. The default value is `json`. |
163
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
164
+
165
+ #### Connecting to a Cluster
166
+
167
+ This command configures the local Kubectl environment to interact with the specified SageMaker HyperPod cluster and namespace.
168
+
169
+ ```bash
170
+ hyp set-cluster-context --cluster-name <cluster-name>
171
+ ```
172
+
173
+ | Option | Type | Description |
174
+ |--------|------|-------------|
175
+ | `--cluster-name <cluster-name>` | Required | The SageMaker HyperPod cluster name to configure with. |
176
+ | `--namespace <namespace>` | Optional | The namespace that you want to connect to. If not specified, Hyperpod cli commands will auto discover the accessible namespace. |
177
+ | `--region <region>` | Optional | The AWS region where the HyperPod cluster resides. |
178
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
179
+
180
+ #### Getting Cluster Context
181
+
182
+ Get all the context related to the current set Cluster
183
+
184
+ ```bash
185
+ hyp get-cluster-context
186
+ ```
187
+
188
+ | Option | Type | Description |
189
+ |--------|------|-------------|
190
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
191
+
192
+
193
+ ## CLI
194
+
195
+ ### Cluster Management
196
+
197
+ **Important**: For commands that accept the `--region` option, if no region is explicitly provided, the command will use the default region from your AWS credentials configuration.
198
+
199
+ **Cluster stack names must be unique within each AWS region.** If you attempt to create a cluster stack with a name that already exists in the same region, the deployment will fail.
200
+
201
+ #### Initialize Cluster Configuration
202
+
203
+ Initialize a new cluster configuration in the current directory:
204
+
205
+ ```bash
206
+ hyp init cluster-stack
207
+ ```
208
+
209
+ **Important**: The `resource_name_prefix` parameter in the generated `config.yaml` file serves as the primary identifier for all AWS resources created during deployment. Each deployment must use a unique resource name prefix to avoid conflicts. This prefix is automatically appended with a unique identifier during cluster creation to ensure resource uniqueness.
210
+
211
+ #### Configure Cluster Parameters
212
+
213
+ Configure cluster parameters interactively or via command line:
214
+
215
+ ```bash
216
+ hyp configure --resource-name-prefix my-cluster --stage prod
217
+ ```
218
+
219
+ #### Validate Configuration
220
+
221
+ Validate the configuration file syntax:
222
+
223
+ ```bash
224
+ hyp validate
225
+ ```
226
+
227
+ #### Create Cluster Stack
228
+
229
+ Create the cluster stack using the configured parameters:
230
+
231
+ ```bash
232
+ hyp create --region <region>
233
+ ```
234
+
235
+ **Note**: The region flag is optional. If not provided, the command will use the default region from your AWS credentials configuration.
236
+
237
+ #### List Cluster Stacks
238
+
239
+ ```bash
240
+ hyp list cluster-stack
241
+ ```
242
+
243
+ | Option | Type | Description |
244
+ |--------|------|-------------|
245
+ | `--region <region>` | Optional | The AWS region to list stacks from. |
246
+ | `--status "['CREATE_COMPLETE', 'UPDATE_COMPLETE']"` | Optional | Filter by stack status. |
247
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
248
+
249
+ #### Describe Cluster Stack
250
+
251
+ ```bash
252
+ hyp describe cluster-stack <stack-name>
253
+ ```
254
+
255
+ | Option | Type | Description |
256
+ |--------|------|-------------|
257
+ | `--region <region>` | Optional | The AWS region where the stack exists. |
258
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
259
+
260
+ #### Delete Cluster Stack
261
+
262
+ Delete a HyperPod cluster stack. Removes the specified CloudFormation stack and all associated AWS resources. This operation cannot be undone.
263
+
264
+ ```bash
265
+ hyp delete cluster-stack <stack-name>
266
+ ```
267
+
268
+ | Option | Type | Description |
269
+ |--------|------|-------------|
270
+ | `--region <region>` | Required | The AWS region where the stack exists. |
271
+ | `--retain-resources S3Bucket-TrainingData,EFSFileSystem-Models` | Optional | Comma-separated list of logical resource IDs to retain during deletion (only works on DELETE_FAILED stacks). Resource names are shown in failed deletion output, or use AWS CLI: `aws cloudformation list-stack-resources STACK_NAME --region REGION`. |
272
+ | `--debug` | Optional | Enable debug mode for detailed logging. |
273
+
274
+
275
+ #### Update Existing Cluster
276
+
277
+ ```bash
278
+ hyp update cluster --cluster-name my-cluster \
279
+ --instance-groups '[{"InstanceCount":2,"InstanceGroupName":"worker-nodes","InstanceType":"ml.m5.large"}]' \
280
+ --node-recovery Automatic
281
+ ```
282
+
283
+ #### Reset Configuration
284
+
285
+ Reset configuration to default values:
286
+
287
+ ```bash
288
+ hyp reset
289
+ ```
290
+
291
+ ### Training
292
+
293
+ #### **Option 1**: Create Pytorch job through init experience
294
+
295
+ #### Initialize Pytorch Job Configuration
296
+
297
+ Initialize a new pytorch job configuration in the current directory:
298
+
299
+ ```bash
300
+ hyp init hyp-pytorch-job
301
+ ```
302
+
303
+ #### Configure Pytorch Job Parameters
304
+
305
+ Configure pytorch job parameters interactively or via command line:
306
+
307
+ ```bash
308
+ hyp configure --job-name my-pytorch-job
309
+ ```
310
+
311
+ #### Validate Configuration
312
+
313
+ Validate the configuration file syntax:
314
+
315
+ ```bash
316
+ hyp validate
317
+ ```
318
+
319
+ #### Create Pytorch Job
320
+
321
+ Create the pytorch job using the configured parameters:
322
+
323
+ ```bash
324
+ hyp create
325
+ ```
326
+
327
+
328
+ #### **Option 2**: Create Pytorch job through create command
329
+
330
+ ```bash
331
+ hyp create hyp-pytorch-job \
332
+ --version 1.0 \
333
+ --job-name test-pytorch-job \
334
+ --image pytorch/pytorch:latest \
335
+ --command '[python, train.py]' \
336
+ --args '[--epochs=10, --batch-size=32]' \
337
+ --environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
338
+ --pull-policy "IfNotPresent" \
339
+ --instance-type ml.p4d.24xlarge \
340
+ --tasks-per-node 8 \
341
+ --label-selector '{"accelerator": "nvidia", "network": "efa"}' \
342
+ --deep-health-check-passed-nodes-only true \
343
+ --scheduler-type "kueue" \
344
+ --queue-name "training-queue" \
345
+ --priority "high" \
346
+ --max-retry 3 \
347
+ --accelerators 8 \
348
+ --vcpu 96.0 \
349
+ --memory 1152.0 \
350
+ --accelerators-limit 8 \
351
+ --vcpu-limit 96.0 \
352
+ --memory-limit 1152.0 \
353
+ --preferred-topology "topology.kubernetes.io/zone=us-west-2a" \
354
+ --volume name=model-data,type=hostPath,mount_path=/data,path=/data \
355
+ --volume name=training-output,type=pvc,mount_path=/data2,claim_name=my-pvc,read_only=false
356
+ ```
357
+
358
+ | Parameter | Type | Required | Description |
359
+ |-----------|------|----------|-------------|
360
+ | `--job-name` | TEXT | Yes | Unique name for the training job (1-63 characters, alphanumeric with hyphens) |
361
+ | `--image` | TEXT | Yes | Docker image URI containing your training code |
362
+ | `--namespace` | TEXT | No | Kubernetes namespace |
363
+ | `--command` | ARRAY | No | Command to run in the container (array of strings) |
364
+ | `--args` | ARRAY | No | Arguments for the entry script (array of strings) |
365
+ | `--environment` | OBJECT | No | Environment variables as key-value pairs |
366
+ | `--pull-policy` | TEXT | No | Image pull policy (Always, Never, IfNotPresent) |
367
+ | `--instance-type` | TEXT | No | Instance type for training |
368
+ | `--node-count` | INTEGER | No | Number of nodes (minimum: 1) |
369
+ | `--tasks-per-node` | INTEGER | No | Number of tasks per node (minimum: 1) |
370
+ | `--label-selector` | OBJECT | No | Node label selector as key-value pairs |
371
+ | `--deep-health-check-passed-nodes-only` | BOOLEAN | No | Schedule pods only on nodes that passed deep health check (default: false) |
372
+ | `--scheduler-type` | TEXT | No | Scheduler type |
373
+ | `--queue-name` | TEXT | No | Queue name for job scheduling (1-63 characters, alphanumeric with hyphens) |
374
+ | `--priority` | TEXT | No | Priority class for job scheduling |
375
+ | `--max-retry` | INTEGER | No | Maximum number of job retries (minimum: 0) |
376
+ | `--volume` | ARRAY | No | List of volume configurations (Refer [Volume Configuration](#volume-configuration) for detailed parameter info) |
377
+ | `--service-account-name` | TEXT | No | Service account name |
378
+ | `--accelerators` | INTEGER | No | Number of accelerators a.k.a GPUs or Trainium Chips |
379
+ | `--vcpu` | FLOAT | No | Number of vCPUs |
380
+ | `--memory` | FLOAT | No | Amount of memory in GiB |
381
+ | `--accelerators-limit` | INTEGER | No | Limit for the number of accelerators a.k.a GPUs or Trainium Chips |
382
+ | `--vcpu-limit` | FLOAT | No | Limit for the number of vCPUs |
383
+ | `--memory-limit` | FLOAT | No | Limit for the amount of memory in GiB |
384
+ | `--preferred-topology` | TEXT | No | Preferred topology annotation for scheduling |
385
+ | `--required-topology` | TEXT | No | Required topology annotation for scheduling |
386
+ | `--debug` | FLAG | No | Enable debug mode (default: false) |
387
+
388
+ #### List Training Jobs
389
+
390
+ ```bash
391
+ hyp list hyp-pytorch-job
392
+ ```
393
+
394
+ #### Describe a Training Job
395
+
396
+ ```bash
397
+ hyp describe hyp-pytorch-job --job-name <job-name>
398
+ ````
399
+
400
+ #### Listing Pods
401
+
402
+ This command lists all the pods associated with a specific training job.
403
+
404
+ ```bash
405
+ hyp list-pods hyp-pytorch-job --job-name <job-name>
406
+ ```
407
+
408
+ * `job-name` (string) - Required. The name of the job to list pods for.
409
+
410
+ #### Accessing Logs
411
+
412
+ This command retrieves the logs for a specific pod within a training job.
413
+
414
+ ```bash
415
+ hyp get-logs hyp-pytorch-job --pod-name <pod-name> --job-name <job-name>
416
+ ```
417
+
418
+ | Parameter | Required | Description |
419
+ |--------|------|-------------|
420
+ | `--job-name` | Yes | The name of the job to get the log for. |
421
+ | `--pod-name` | Yes | The name of the pod to get the log from. |
422
+ | `--namespace` | No | The namespace of the job. Defaults to 'default'. |
423
+ | `--container` | No | The container name to get logs from. |
424
+
425
+ #### Get Operator Logs
426
+
427
+ ```bash
428
+ hyp get-operator-logs hyp-pytorch-job --since-hours 0.5
429
+ ```
430
+
431
+ #### Delete a Training Job
432
+
433
+ ```bash
434
+ hyp delete hyp-pytorch-job --job-name <job-name>
435
+ ```
436
+
437
+ ### Inference
438
+
439
+ ### Jumpstart Endpoint Creation
440
+
441
+ #### **Option 1**: Create jumpstart endpoint through init experience
442
+
443
+ #### Initialize Jumpstart Endpoint Configuration
444
+
445
+ Initialize a new jumpstart endpoint configuration in the current directory:
446
+
447
+ ```bash
448
+ hyp init hyp-jumpstart-endpoint
449
+ ```
450
+
451
+ #### Configure Jumpstart Endpoint Parameters
452
+
453
+ Configure jumpstart endpoint parameters interactively or via command line:
454
+
455
+ ```bash
456
+ hyp configure --endpoint-name my-jumpstart-endpoint
457
+ ```
458
+
459
+ #### Validate Configuration
460
+
461
+ Validate the configuration file syntax:
462
+
463
+ ```bash
464
+ hyp validate
465
+ ```
466
+
467
+ #### Create Jumpstart Endpoint
468
+
469
+ Create the jumpstart endpoint using the configured parameters:
470
+
471
+ ```bash
472
+ hyp create
473
+ ```
474
+
475
+
476
+ #### **Option 2**: Create jumpstart endpoint through create command
477
+ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io/en/v2.82.0/doc_utils/jumpstart.html and fed into the call for creating the endpoint
478
+
479
+ ```bash
480
+ hyp create hyp-jumpstart-endpoint \
481
+ --version 1.0 \
482
+ --model-id jumpstart-model-id\
483
+ --instance-type ml.g5.8xlarge \
484
+ --endpoint-name endpoint-jumpstart
485
+ ```
486
+
487
+ | Parameter | Type | Required | Description |
488
+ |-----------|------|----------|-------------|
489
+ | `--model-id` | TEXT | Yes | JumpStart model identifier (1-63 characters, alphanumeric with hyphens) |
490
+ | `--instance-type` | TEXT | Yes | EC2 instance type for inference (must start with "ml.") |
491
+ | `--namespace` | TEXT | No | Kubernetes namespace |
492
+ | `--metadata-name` | TEXT | No | Name of the jumpstart endpoint object |
493
+ | `--accept-eula` | BOOLEAN | No | Whether model terms of use have been accepted (default: false) |
494
+ | `--model-version` | TEXT | No | Semantic version of the model (e.g., "1.0.0", 5-14 characters) |
495
+ | `--endpoint-name` | TEXT | No | Name of SageMaker endpoint (1-63 characters, alphanumeric with hyphens) |
496
+ | `--tls-certificate-output-s3-uri` | TEXT | No | S3 URI to write the TLS certificate (optional) |
497
+ | `--debug` | FLAG | No | Enable debug mode (default: false) |
498
+
499
+
500
+ #### Invoke a JumpstartModel Endpoint
501
+
502
+ ```bash
503
+ hyp invoke hyp-jumpstart-endpoint \
504
+ --endpoint-name endpoint-jumpstart \
505
+ --body '{"inputs":"What is the capital of USA?"}'
506
+ ```
507
+
508
+
509
+ #### Managing an Endpoint
510
+
511
+ ```bash
512
+ hyp list hyp-jumpstart-endpoint
513
+ hyp describe hyp-jumpstart-endpoint --name endpoint-jumpstart
514
+ ```
515
+
516
+ #### List Pods
517
+
518
+ ```bash
519
+ hyp list-pods hyp-jumpstart-endpoint
520
+ ```
521
+
522
+ #### Get Logs
523
+
524
+ ```bash
525
+ hyp get-logs hyp-jumpstart-endpoint --pod-name <pod-name>
526
+ ```
527
+
528
+ #### Get Operator Logs
529
+
530
+ ```bash
531
+ hyp get-operator-logs hyp-jumpstart-endpoint --since-hours 0.5
532
+ ```
533
+
534
+ #### Deleting an Endpoint
535
+
536
+ ```bash
537
+ hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
538
+ ```
539
+
540
+
541
+ ### Custom Endpoint Creation
542
+ #### **Option 1**: Create custom endpoint through init experience
543
+
544
+ #### Initialize Custom Endpoint Configuration
545
+
546
+ Initialize a new custom endpoint configuration in the current directory:
547
+
548
+ ```bash
549
+ hyp init hyp-custom-endpoint
550
+ ```
551
+
552
+ #### Configure Custom Endpoint Parameters
553
+
554
+ Configure custom endpoint parameters interactively or via command line:
555
+
556
+ ```bash
557
+ hyp configure --endpoint-name my-custom-endpoint
558
+ ```
559
+
560
+ #### Validate Configuration
561
+
562
+ Validate the configuration file syntax:
563
+
564
+ ```bash
565
+ hyp validate
566
+ ```
567
+
568
+ #### Create Custom Endpoint
569
+
570
+ Create the custom endpoint using the configured parameters:
571
+
572
+ ```bash
573
+ hyp create
574
+ ```
575
+
576
+
577
+ #### **Option 2**: Create custom endpoint through create command
578
+ ```bash
579
+ hyp create hyp-custom-endpoint \
580
+ --version 1.0 \
581
+ --endpoint-name endpoint-custom \
582
+ --model-name my-pytorch-model \
583
+ --model-source-type s3 \
584
+ --model-location my-pytorch-training \
585
+ --model-volume-mount-name test-volume \
586
+ --s3-bucket-name your-bucket \
587
+ --s3-region us-east-1 \
588
+ --instance-type ml.g5.8xlarge \
589
+ --image-uri 763104351884.dkr.ecr.us-east-1.amazonaws.com/pytorch-inference:latest \
590
+ --container-port 8080
591
+ ```
592
+
593
+ | Parameter | Type | Required | Description |
594
+ |-----------|------|----------|-------------|
595
+ | `--instance-type` | TEXT | Yes | EC2 instance type for inference (must start with "ml.") |
596
+ | `--model-name` | TEXT | Yes | Name of model to create on SageMaker (1-63 characters, alphanumeric with hyphens) |
597
+ | `--model-source-type` | TEXT | Yes | Model source type ("s3" or "fsx") |
598
+ | `--image-uri` | TEXT | Yes | Docker image URI for inference |
599
+ | `--container-port` | INTEGER | Yes | Port on which model server listens (1-65535) |
600
+ | `--model-volume-mount-name` | TEXT | Yes | Name of the model volume mount |
601
+ | `--namespace` | TEXT | No | Kubernetes namespace |
602
+ | `--metadata-name` | TEXT | No | Name of the custom endpoint object |
603
+ | `--endpoint-name` | TEXT | No | Name of SageMaker endpoint (1-63 characters, alphanumeric with hyphens) |
604
+ | `--env` | OBJECT | No | Environment variables as key-value pairs |
605
+ | `--metrics-enabled` | BOOLEAN | No | Enable metrics collection (default: false) |
606
+ | `--model-version` | TEXT | No | Version of the model (semantic version format) |
607
+ | `--model-location` | TEXT | No | Specific model data location |
608
+ | `--prefetch-enabled` | BOOLEAN | No | Whether to pre-fetch model data (default: false) |
609
+ | `--tls-certificate-output-s3-uri` | TEXT | No | S3 URI for TLS certificate output |
610
+ | `--fsx-dns-name` | TEXT | No | FSx File System DNS Name |
611
+ | `--fsx-file-system-id` | TEXT | No | FSx File System ID |
612
+ | `--fsx-mount-name` | TEXT | No | FSx File System Mount Name |
613
+ | `--s3-bucket-name` | TEXT | No | S3 bucket location |
614
+ | `--s3-region` | TEXT | No | S3 bucket region |
615
+ | `--model-volume-mount-path` | TEXT | No | Path inside container for model volume (default: "/opt/ml/model") |
616
+ | `--resources-limits` | OBJECT | No | Resource limits for the worker |
617
+ | `--resources-requests` | OBJECT | No | Resource requests for the worker |
618
+ | `--dimensions` | OBJECT | No | CloudWatch Metric dimensions as key-value pairs |
619
+ | `--metric-collection-period` | INTEGER | No | Period for CloudWatch query (default: 300) |
620
+ | `--metric-collection-start-time` | INTEGER | No | StartTime for CloudWatch query (default: 300) |
621
+ | `--metric-name` | TEXT | No | Metric name to query for CloudWatch trigger |
622
+ | `--metric-stat` | TEXT | No | Statistics metric for CloudWatch (default: "Average") |
623
+ | `--metric-type` | TEXT | No | Type of metric for HPA ("Value" or "Average", default: "Average") |
624
+ | `--min-value` | NUMBER | No | Minimum metric value for empty CloudWatch response (default: 0) |
625
+ | `--cloud-watch-trigger-name` | TEXT | No | Name for the CloudWatch trigger |
626
+ | `--cloud-watch-trigger-namespace` | TEXT | No | AWS CloudWatch namespace for the metric |
627
+ | `--target-value` | NUMBER | No | Target value for the CloudWatch metric |
628
+ | `--use-cached-metrics` | BOOLEAN | No | Enable caching of metric values (default: true) |
629
+ | `--invocation-endpoint` | TEXT | No | Invocation endpoint path (default: "invocations") |
630
+ | `--debug` | FLAG | No | Enable debug mode (default: false) |
631
+
632
+
633
+ #### Invoke a Custom Inference Endpoint
634
+
635
+ ```bash
636
+ hyp invoke hyp-custom-endpoint \
637
+ --endpoint-name endpoint-custom-pytorch \
638
+ --body '{"inputs":"What is the capital of USA?"}'
639
+ ```
640
+
641
+ #### Managing an Endpoint
642
+
643
+ ```bash
644
+ hyp list hyp-custom-endpoint
645
+ hyp describe hyp-custom-endpoint --name endpoint-custom
646
+ ```
647
+
648
+ #### List Pods
649
+
650
+ ```bash
651
+ hyp list-pods hyp-custom-endpoint
652
+ ```
653
+
654
+ #### Get Logs
655
+
656
+ ```bash
657
+ hyp get-logs hyp-custom-endpoint --pod-name <pod-name>
658
+ ```
659
+
660
+ #### Get Operator Logs
661
+
662
+ ```bash
663
+ hyp get-operator-logs hyp-custom-endpoint --since-hours 0.5
664
+ ```
665
+
666
+ #### Deleting an Endpoint
667
+
668
+ ```bash
669
+ hyp delete hyp-custom-endpoint --name endpoint-custom
670
+ ```
671
+
672
+ ## SDK
673
+
674
+ Along with the CLI, we also have SDKs available that can perform the cluster management, training and inference functionalities that the CLI performs
675
+
676
+ ### Cluster Management SDK
677
+
678
+ #### Creating a Cluster Stack
679
+
680
+ ```python
681
+ from sagemaker.hyperpod.cluster_management.hp_cluster_stack import HpClusterStack
682
+
683
+ # Initialize cluster stack configuration
684
+ cluster_stack = HpClusterStack(
685
+ stage="prod",
686
+ resource_name_prefix="my-hyperpod",
687
+ hyperpod_cluster_name="my-hyperpod-cluster",
688
+ eks_cluster_name="my-hyperpod-eks",
689
+
690
+ # Infrastructure components
691
+ create_vpc_stack=True,
692
+ create_eks_cluster_stack=True,
693
+ create_hyperpod_cluster_stack=True,
694
+
695
+ # Network configuration
696
+ vpc_cidr="10.192.0.0/16",
697
+ availability_zone_ids=["use2-az1", "use2-az2"],
698
+
699
+ # Instance group configuration
700
+ instance_group_settings=[
701
+ {
702
+ "InstanceCount": 1,
703
+ "InstanceGroupName": "controller-group",
704
+ "InstanceType": "ml.t3.medium",
705
+ "TargetAvailabilityZoneId": "use2-az2"
706
+ }
707
+ ]
708
+ )
709
+
710
+ # Create the cluster stack
711
+ response = cluster_stack.create(region="us-east-2")
712
+ ```
713
+
714
+ #### Listing Cluster Stacks
715
+
716
+ ```python
717
+ # List all cluster stacks
718
+ stacks = HpClusterStack.list(region="us-east-2")
719
+ print(f"Found {len(stacks['StackSummaries'])} stacks")
720
+ ```
721
+
722
+ #### Describing a Cluster Stack
723
+
724
+ ```python
725
+ # Describe a specific cluster stack
726
+ stack_info = HpClusterStack.describe("my-stack-name", region="us-east-2")
727
+ print(f"Stack status: {stack_info['Stacks'][0]['StackStatus']}")
728
+ ```
729
+
730
+ #### Monitoring Cluster Status
731
+
732
+ ```python
733
+ from sagemaker.hyperpod.cluster_management.hp_cluster_stack import HpClusterStack
734
+
735
+ stack = HpClusterStack()
736
+ response = stack.create(region="us-west-2")
737
+ status = stack.get_status(region="us-west-2")
738
+ print(status)
739
+ ```
740
+
741
+ ### Training SDK
742
+
743
+ #### Creating a Training Job
744
+
745
+ ```python
746
+ from sagemaker.hyperpod.training.hyperpod_pytorch_job import HyperPodPytorchJob
747
+ from sagemaker.hyperpod.training.config.hyperpod_pytorch_job_unified_config import (
748
+ ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
749
+ )
750
+ from sagemaker.hyperpod.common.config.metadata import Metadata
751
+
752
+ # Define job specifications
753
+ nproc_per_node = "1" # Number of processes per node
754
+ replica_specs =
755
+ [
756
+ ReplicaSpec
757
+ (
758
+ name = "pod", # Replica name
759
+ template = Template
760
+ (
761
+ spec = Spec
762
+ (
763
+ containers =
764
+ [
765
+ Containers
766
+ (
767
+ # Container name
768
+ name="container-name",
769
+
770
+ # Training image
771
+ image="123456789012.dkr.ecr.us-west-2.amazonaws.com/my-training-image:latest",
772
+
773
+ # Always pull image
774
+ image_pull_policy="Always",
775
+ resources=Resources\
776
+ (
777
+ # No GPUs requested
778
+ requests={"nvidia.com/gpu": "0"},
779
+ # No GPU limit
780
+ limits={"nvidia.com/gpu": "0"},
781
+ ),
782
+ # Command to run
783
+ command=["python", "train.py"],
784
+ # Script arguments
785
+ args=["--epochs", "10", "--batch-size", "32"],
786
+ )
787
+ ]
788
+ )
789
+ ),
790
+ )
791
+ ]
792
+ # Keep pods after completion
793
+ run_policy = RunPolicy(clean_pod_policy="None")
794
+
795
+ # Create and start the PyTorch job
796
+ pytorch_job = HyperPodPytorchJob
797
+ (
798
+ # Job name
799
+ metadata = Metadata(name="demo"),
800
+ # Processes per node
801
+ nproc_per_node = nproc_per_node,
802
+ # Replica specifications
803
+ replica_specs = replica_specs,
804
+ # Run policy
805
+ run_policy = run_policy,
806
+ )
807
+ # Launch the job
808
+ pytorch_job.create()
809
+ ```
810
+
811
+ #### List Training Jobs
812
+ ```python
813
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
814
+ import yaml
815
+
816
+ # List all PyTorch jobs
817
+ jobs = HyperPodPytorchJob.list()
818
+ print(yaml.dump(jobs))
819
+ ```
820
+
821
+ #### Describe a Training Job
822
+ ```python
823
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
824
+
825
+ # Get an existing job
826
+ job = HyperPodPytorchJob.get(name="my-pytorch-job")
827
+
828
+ print(job)
829
+ ```
830
+
831
+ #### List Pods for a Training Job
832
+ ```python
833
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
834
+
835
+ # List Pods for an existing job
836
+ job = HyperPodPytorchJob.get(name="my-pytorch-job")
837
+ print(job.list_pods())
838
+ ```
839
+
840
+ #### Get Logs from a Pod
841
+ ```python
842
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
843
+
844
+ # Get pod logs for a job
845
+ job = HyperPodPytorchJob.get(name="my-pytorch-job")
846
+ print(job.get_logs_from_pod("pod-name"))
847
+ ```
848
+
849
+ #### Get Training Operator Logs
850
+ ```python
851
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
852
+
853
+ # Get training operator logs
854
+ job = HyperPodPytorchJob.get(name="my-pytorch-job")
855
+ print(job.get_operator_logs(since_hours=0.1))
856
+ ```
857
+
858
+ #### Delete a Training Job
859
+ ```python
860
+ from sagemaker.hyperpod.training import HyperPodPytorchJob
861
+
862
+ # Get an existing job
863
+ job = HyperPodPytorchJob.get(name="my-pytorch-job")
864
+
865
+ # Delete the job
866
+ job.delete()
867
+ ```
868
+
869
+ ### Inference SDK
870
+
871
+ #### Creating a JumpstartModel Endpoint
872
+
873
+ Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io/en/v2.82.0/doc_utils/jumpstart.html and fed into the call for creating the endpoint
874
+
875
+ ```python
876
+ from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
877
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
878
+
879
+ model=Model(
880
+ model_id='deepseek-llm-r1-distill-qwen-1-5b'
881
+ )
882
+ server=Server(
883
+ instance_type='ml.g5.8xlarge',
884
+ )
885
+ endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
886
+
887
+ js_endpoint=HPJumpStartEndpoint(
888
+ model=model,
889
+ server=server,
890
+ sage_maker_endpoint=endpoint_name
891
+ )
892
+
893
+ js_endpoint.create()
894
+ ```
895
+
896
+ #### Creating a Custom Inference Endpoint (with S3)
897
+
898
+ ```python
899
+ from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
900
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
901
+
902
+ model_source_config = ModelSourceConfig(
903
+ model_source_type='s3',
904
+ model_location="<my-model-folder-in-s3>",
905
+ s3_storage=S3Storage(
906
+ bucket_name='<my-model-artifacts-bucket>',
907
+ region='us-east-2',
908
+ ),
909
+ )
910
+
911
+ environment_variables = [
912
+ EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
913
+ EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
914
+ EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
915
+ EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
916
+ EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
917
+ ]
918
+
919
+ worker = Worker(
920
+ image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
921
+ model_volume_mount=ModelVolumeMount(
922
+ name='model-weights',
923
+ ),
924
+ model_invocation_port=ModelInvocationPort(container_port=8080),
925
+ resources=Resources(
926
+ requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
927
+ limits={"nvidia.com/gpu": 1}
928
+ ),
929
+ environment_variables=environment_variables,
930
+ )
931
+
932
+ tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
933
+
934
+ custom_endpoint = HPEndpoint(
935
+ endpoint_name='<my-endpoint-name>',
936
+ instance_type='ml.g5.8xlarge',
937
+ model_name='deepseek15b-test-model-name',
938
+ tls_config=tls_config,
939
+ model_source_config=model_source_config,
940
+ worker=worker,
941
+ )
942
+
943
+ custom_endpoint.create()
944
+ ```
945
+
946
+
947
+ #### List Endpoints
948
+
949
+ ```python
950
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
951
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
952
+
953
+ # List JumpStart endpoints
954
+ jumpstart_endpoints = HPJumpStartEndpoint.list()
955
+ print(jumpstart_endpoints)
956
+
957
+ # List custom endpoints
958
+ custom_endpoints = HPEndpoint.list()
959
+ print(custom_endpoints)
960
+ ```
961
+
962
+ #### Describe an Endpoint
963
+ ```python
964
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
965
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
966
+
967
+ # Get JumpStart endpoint details
968
+ jumpstart_endpoint = HPJumpStartEndpoint.get(name="js-endpoint-name", namespace="test")
969
+ print(jumpstart_endpoint)
970
+
971
+ # Get custom endpoint details
972
+ custom_endpoint = HPEndpoint.get(name="endpoint-custom")
973
+ print(custom_endpoint)
974
+
975
+ ```
976
+
977
+ #### Invoke an Endpoint
978
+ ```python
979
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
980
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
981
+
982
+ data = '{"inputs":"What is the capital of USA?"}'
983
+ jumpstart_endpoint = HPJumpStartEndpoint.get(name="endpoint-jumpstart")
984
+ response = jumpstart_endpoint.invoke(body=data).body.read()
985
+ print(response)
986
+
987
+ custom_endpoint = HPEndpoint.get(name="endpoint-custom")
988
+ response = custom_endpoint.invoke(body=data).body.read()
989
+ print(response)
990
+ ```
991
+
992
+ #### List Pods
993
+ ```python
994
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
995
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
996
+
997
+ # List pods
998
+ js_pods = HPJumpStartEndpoint.list_pods()
999
+ print(js_pods)
1000
+
1001
+ c_pods = HPEndpoint.list_pods()
1002
+ print(c_pods)
1003
+ ```
1004
+
1005
+ #### Get Logs
1006
+ ```python
1007
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
1008
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
1009
+
1010
+ # Get logs from pod
1011
+ js_logs = HPJumpStartEndpoint.get_logs(pod=<pod-name>)
1012
+ print(js_logs)
1013
+
1014
+ c_logs = HPEndpoint.get_logs(pod=<pod-name>)
1015
+ print(c_logs)
1016
+ ```
1017
+
1018
+ #### Get Operator Logs
1019
+ ```python
1020
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
1021
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
1022
+
1023
+ # Invoke JumpStart endpoint
1024
+ print(HPJumpStartEndpoint.get_operator_logs(since_hours=0.1))
1025
+
1026
+ # Invoke custom endpoint
1027
+ print(HPEndpoint.get_operator_logs(since_hours=0.1))
1028
+ ```
1029
+
1030
+ #### Delete an Endpoint
1031
+ ```python
1032
+ from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
1033
+ from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
1034
+
1035
+ # Delete JumpStart endpoint
1036
+ jumpstart_endpoint = HPJumpStartEndpoint.get(name="endpoint-jumpstart")
1037
+ jumpstart_endpoint.delete()
1038
+
1039
+ # Delete custom endpoint
1040
+ custom_endpoint = HPEndpoint.get(name="endpoint-custom")
1041
+ custom_endpoint.delete()
1042
+ ```
1043
+
1044
+
1045
+ #### Observability - Getting Monitoring Information
1046
+ ```python
1047
+ from sagemaker.hyperpod.observability.utils import get_monitoring_config
1048
+ monitor_config = get_monitoring_config()
1049
+ ```
1050
+
1051
+ ## Examples
1052
+ #### Cluster Management Example Notebooks
1053
+
1054
+ [CLI Cluster Management Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/cluster_management/cluster_creation_init_experience.ipynb)
1055
+
1056
+ [SDK Cluster Management Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/cluster_management/cluster_creation_sdk_experience.ipynb)
1057
+
1058
+ #### Training Example Notebooks
1059
+
1060
+ [CLI Training Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/CLI/training-init-experience.ipynb)
1061
+
1062
+ [CLI Training Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/CLI/training-e2e-cli.ipynb)
1063
+
1064
+ [SDK Training Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/SDK/training_sdk_example.ipynb)
1065
+
1066
+ #### Inference Example Notebooks
1067
+
1068
+ ##### CLI
1069
+ [CLI Inference Jumpstart Model Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-jumpstart-init-experience.ipynb)
1070
+
1071
+ [CLI Inference JumpStart Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-jumpstart-e2e-cli.ipynb)
1072
+
1073
+ [CLI Inference FSX Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-fsx-model-e2e-cli.ipynb)
1074
+
1075
+ [CLI Inference S3 Model Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-s3-model-init-experience.ipynb)
1076
+
1077
+ [CLI Inference S3 Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-s3-model-e2e-cli.ipynb)
1078
+
1079
+ ##### SDK
1080
+
1081
+ [SDK Inference JumpStart Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-jumpstart-e2e.ipynb)
1082
+
1083
+ [SDK Inference FSX Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-fsx-model-e2e.ipynb)
1084
+
1085
+ [SDK Inference S3 Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-s3-model-e2e.ipynb)
1086
+
1087
+
1088
+ ## Disclaimer
1089
+
1090
+ * This CLI and SDK requires access to the user's file system to set and get context and function properly.
1091
+ It needs to read configuration files such as kubeconfig to establish the necessary environment settings.
1092
+
1093
+
1094
+ ## Working behind a proxy server ?
1095
+ * Follow these steps from [here](https://docs.aws.amazon.com/cli/v1/userguide/cli-configure-proxy.html) to set up HTTP proxy connections
1096
+