sagemaker-hyperpod 3.2.1__tar.gz → 3.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sagemaker_hyperpod-3.3.0/PKG-INFO +1096 -0
- sagemaker_hyperpod-3.3.0/README.md +1041 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/pyproject.toml +1 -1
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/setup.cfg +2 -1
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/setup.py +1 -1
- sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/__init__.py +9 -0
- sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/cluster_stack_utils.py +498 -0
- sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/cluster_utils.py +145 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/cluster.py +31 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/cluster_stack.py +64 -69
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/inference.py +9 -21
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/init.py +55 -101
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/commands/training.py +1 -31
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/common_utils.py +52 -2
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/init_constants.py +57 -29
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/hyp_cli.py +2 -2
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/inference_utils.py +3 -4
- sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/init_utils.py +536 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/training_utils.py +2 -2
- sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/type_handler_utils.py +174 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cluster_management/hp_cluster_stack.py +225 -31
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/telemetry_logging.py +20 -1
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/utils.py +6 -1
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_endpoint.py +11 -6
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_endpoint_base.py +2 -1
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/hp_jumpstart_endpoint.py +11 -6
- sagemaker_hyperpod-3.3.0/src/sagemaker_hyperpod.egg-info/PKG-INFO +1096 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/SOURCES.txt +3 -4
- sagemaker_hyperpod-3.2.1/PKG-INFO +0 -504
- sagemaker_hyperpod-3.2.1/README.md +0 -449
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/init_utils.py +0 -949
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/cfn_cluster_creation.py +0 -948
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_custom_endpoint_template.py +0 -68
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_js_endpoint_template.py +0 -17
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/templates/k8s_pytorch_job_template.py +0 -68
- sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/observability/__init__.py +0 -0
- sagemaker_hyperpod-3.2.1/src/sagemaker_hyperpod.egg-info/PKG-INFO +0 -504
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/LICENSE +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/NOTICE +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/clients/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/clients/kubernetes_client.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/commands}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/commands → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cli/constants}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/command_constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/exception_constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/hyperpod_instance_types.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/kueue_constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/constants/pytorch_constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/cancel_training_job.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/discover_namespaces.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/exec_command.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_logs.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_namespaces.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/get_training_job.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/list_pods.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/list_training_jobs.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/service/self_subject_access_review.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/templates/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/utils.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/cluster_validator.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/job_validator.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cli/validators/validator.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cli/constants → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cluster_management}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cluster_management → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/cluster_management/config}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/cluster_management/config/hp_cluster_stack_config.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/cluster_management/config → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/common}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/cli_decorators.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/config/metadata.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/exceptions/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/common/telemetry/user_agent.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/common → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/inference/config}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/hp_endpoint_config.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/config/hp_jumpstart_endpoint_config.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/inference/jumpstart_public_hub_visualization_utils.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/MonitoringConfig.py +0 -0
- {sagemaker_hyperpod-3.2.1/src/sagemaker/hyperpod/inference/config → sagemaker_hyperpod-3.3.0/src/sagemaker/hyperpod/observability}/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/constants.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/observability/utils.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/config/__init__.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/config/hyperpod_pytorch_job_unified_config.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/hyperpod_pytorch_job.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker/hyperpod/training/quota_allocation_util.py +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/dependency_links.txt +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/entry_points.txt +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/requires.txt +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/top_level.txt +0 -0
- {sagemaker_hyperpod-3.2.1 → sagemaker_hyperpod-3.3.0}/src/sagemaker_hyperpod.egg-info/zip-safe +0 -0
|
@@ -0,0 +1,1096 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sagemaker-hyperpod
|
|
3
|
+
Version: 3.3.0
|
|
4
|
+
Summary: Amazon SageMaker HyperPod SDK and CLI
|
|
5
|
+
Home-page: https://github.com/aws/sagemaker-hyperpod-cli
|
|
6
|
+
Author: Amazon Web Services
|
|
7
|
+
License: Apache-2.0
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Requires-Python: >=3.8
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
License-File: NOTICE
|
|
20
|
+
Requires-Dist: click==8.1.7
|
|
21
|
+
Requires-Dist: awscli>=1.34.9
|
|
22
|
+
Requires-Dist: awscli-cwlogs>=1.4.6
|
|
23
|
+
Requires-Dist: boto3<2.0,>=1.35.3
|
|
24
|
+
Requires-Dist: botocore>=1.35.6
|
|
25
|
+
Requires-Dist: kubernetes==33.1.0
|
|
26
|
+
Requires-Dist: pyyaml==6.0.2
|
|
27
|
+
Requires-Dist: ratelimit==2.2.1
|
|
28
|
+
Requires-Dist: tabulate==0.9.0
|
|
29
|
+
Requires-Dist: itables>=2.2.2
|
|
30
|
+
Requires-Dist: jinja2>=3.1.2
|
|
31
|
+
Requires-Dist: ipywidgets>=8.1.7
|
|
32
|
+
Requires-Dist: hydra-core==1.3.2
|
|
33
|
+
Requires-Dist: omegaconf==2.3
|
|
34
|
+
Requires-Dist: pynvml==11.4.1
|
|
35
|
+
Requires-Dist: requests==2.32.4
|
|
36
|
+
Requires-Dist: tqdm==4.66.5
|
|
37
|
+
Requires-Dist: zstandard==0.15.2
|
|
38
|
+
Requires-Dist: pytest==8.3.2
|
|
39
|
+
Requires-Dist: pytest-cov==5.0.0
|
|
40
|
+
Requires-Dist: pytest-order==1.3.0
|
|
41
|
+
Requires-Dist: pytest-dependency==0.6.0
|
|
42
|
+
Requires-Dist: tox==4.18.0
|
|
43
|
+
Requires-Dist: ruff==0.6.2
|
|
44
|
+
Requires-Dist: hera-workflows==5.16.3
|
|
45
|
+
Requires-Dist: sagemaker-core<2.0.0
|
|
46
|
+
Requires-Dist: pydantic<3.0.0,>=2.10.6
|
|
47
|
+
Requires-Dist: hyperpod-pytorch-job-template<2.0.0,>=1.0.0
|
|
48
|
+
Requires-Dist: hyperpod-custom-inference-template<2.0.0,>=1.0.0
|
|
49
|
+
Requires-Dist: hyperpod-jumpstart-inference-template<2.0.0,>=1.0.0
|
|
50
|
+
Requires-Dist: hyperpod-cluster-stack-template<2.0.0,>=1.0.0
|
|
51
|
+
Dynamic: home-page
|
|
52
|
+
Dynamic: license-file
|
|
53
|
+
Dynamic: requires-dist
|
|
54
|
+
Dynamic: requires-python
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# SageMaker HyperPod command-line interface
|
|
58
|
+
|
|
59
|
+
The Amazon SageMaker HyperPod command-line interface (HyperPod CLI) is a tool that helps manage clusters, training jobs, and inference endpoints on the SageMaker HyperPod clusters orchestrated by Amazon EKS.
|
|
60
|
+
|
|
61
|
+
This documentation serves as a reference for the available HyperPod CLI commands. For a comprehensive user guide, see [Orchestrating SageMaker HyperPod clusters with Amazon EKS](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-hyperpod-eks.html) in the *Amazon SageMaker Developer Guide*.
|
|
62
|
+
|
|
63
|
+
Note: Old `hyperpod`CLI V2 has been moved to `release_v2` branch. Please refer [release_v2 branch](https://github.com/aws/sagemaker-hyperpod-cli/tree/release_v2) for usage.
|
|
64
|
+
|
|
65
|
+
## Table of Contents
|
|
66
|
+
- [Overview](#overview)
|
|
67
|
+
- [Prerequisites](#prerequisites)
|
|
68
|
+
- [Platform Support](#platform-support)
|
|
69
|
+
- [ML Framework Support](#ml-framework-support)
|
|
70
|
+
- [Installation](#installation)
|
|
71
|
+
- [Usage](#usage)
|
|
72
|
+
- [Getting Started](#getting-started)
|
|
73
|
+
- [CLI](#cli)
|
|
74
|
+
- [Cluster Management](#cluster-management)
|
|
75
|
+
- [Training](#training)
|
|
76
|
+
- [Inference](#inference)
|
|
77
|
+
- [Jumpstart Endpoint](#jumpstart-endpoint-creation)
|
|
78
|
+
- [Custom Endpoint](#custom-endpoint-creation)
|
|
79
|
+
- [SDK](#sdk)
|
|
80
|
+
- [Cluster Management](#cluster-management-sdk)
|
|
81
|
+
- [Training](#training-sdk)
|
|
82
|
+
- [Inference](#inference-sdk)
|
|
83
|
+
- [Examples](#examples)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
## Overview
|
|
87
|
+
|
|
88
|
+
The SageMaker HyperPod CLI is a tool that helps create training jobs and inference endpoint deployments to the Amazon SageMaker HyperPod clusters orchestrated by Amazon EKS. It provides a set of commands for managing the full lifecycle of jobs, including create, describe, list, and delete operations, as well as accessing pod and operator logs where applicable. The CLI is designed to abstract away the complexity of working directly with Kubernetes for these core actions of managing jobs on SageMaker HyperPod clusters orchestrated by Amazon EKS.
|
|
89
|
+
|
|
90
|
+
## Prerequisites
|
|
91
|
+
|
|
92
|
+
### Region Configuration
|
|
93
|
+
|
|
94
|
+
**Important**: For commands that accept the `--region` option, if no region is explicitly provided, the command will use the default region from your AWS credentials configuration.
|
|
95
|
+
|
|
96
|
+
### Prerequisites for Training
|
|
97
|
+
|
|
98
|
+
- HyperPod CLI currently supports starting PyTorchJobs. To start a job, you need to install Training Operator first.
|
|
99
|
+
- You can follow [pytorch operator doc](https://docs.aws.amazon.com/sagemaker/latest/dg/sagemaker-eks-operator-install.html) to install it.
|
|
100
|
+
|
|
101
|
+
### Prerequisites for Inference
|
|
102
|
+
|
|
103
|
+
- HyperPod CLI supports creating Inference Endpoints through jumpstart and through custom Endpoint config
|
|
104
|
+
- You can follow [inference operator doc](https://github.com/aws/sagemaker-hyperpod-cli/tree/master/helm_chart/HyperPodHelmChart/charts/inference-operator) to install it.
|
|
105
|
+
|
|
106
|
+
## Platform Support
|
|
107
|
+
|
|
108
|
+
SageMaker HyperPod CLI currently supports Linux and MacOS platforms. Windows platform is not supported now.
|
|
109
|
+
|
|
110
|
+
## ML Framework Support
|
|
111
|
+
|
|
112
|
+
SageMaker HyperPod CLI currently supports start training job with:
|
|
113
|
+
- PyTorch ML Framework. Version requirements: PyTorch >= 1.10
|
|
114
|
+
|
|
115
|
+
## Installation
|
|
116
|
+
|
|
117
|
+
1. Make sure that your local python version is 3.8, 3.9, 3.10 or 3.11.
|
|
118
|
+
|
|
119
|
+
2. Install the sagemaker-hyperpod-cli package.
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
pip install sagemaker-hyperpod
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
3. Verify if the installation succeeded by running the following command.
|
|
126
|
+
|
|
127
|
+
```bash
|
|
128
|
+
hyp --help
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
## Usage
|
|
132
|
+
|
|
133
|
+
The HyperPod CLI provides the following commands:
|
|
134
|
+
|
|
135
|
+
- [Getting Started](#getting-started)
|
|
136
|
+
- [CLI](#cli)
|
|
137
|
+
- [Cluster Management](#cluster-management)
|
|
138
|
+
- [Training](#training)
|
|
139
|
+
- [Inference](#inference)
|
|
140
|
+
- [Jumpstart Endpoint](#jumpstart-endpoint-creation)
|
|
141
|
+
- [Custom Endpoint](#custom-endpoint-creation)
|
|
142
|
+
- [SDK](#sdk)
|
|
143
|
+
- [Cluster Management](#cluster-management-sdk)
|
|
144
|
+
- [Training](#training-sdk)
|
|
145
|
+
- [Inference](#inference-sdk)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
### Getting Started
|
|
149
|
+
|
|
150
|
+
#### Getting Cluster information
|
|
151
|
+
|
|
152
|
+
This command lists the available SageMaker HyperPod clusters and their capacity information.
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
hyp list-cluster
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
| Option | Type | Description |
|
|
159
|
+
|--------|------|-------------|
|
|
160
|
+
| `--region <region>` | Optional | The region that the SageMaker HyperPod and EKS clusters are located. If not specified, it will be set to the region from the current AWS account credentials. |
|
|
161
|
+
| `--namespace <namespace>` | Optional | The namespace that users want to check the quota with. Only the SageMaker managed namespaces are supported. |
|
|
162
|
+
| `--output <json\|table>` | Optional | The output format. Available values are `table` and `json`. The default value is `json`. |
|
|
163
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
164
|
+
|
|
165
|
+
#### Connecting to a Cluster
|
|
166
|
+
|
|
167
|
+
This command configures the local Kubectl environment to interact with the specified SageMaker HyperPod cluster and namespace.
|
|
168
|
+
|
|
169
|
+
```bash
|
|
170
|
+
hyp set-cluster-context --cluster-name <cluster-name>
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
| Option | Type | Description |
|
|
174
|
+
|--------|------|-------------|
|
|
175
|
+
| `--cluster-name <cluster-name>` | Required | The SageMaker HyperPod cluster name to configure with. |
|
|
176
|
+
| `--namespace <namespace>` | Optional | The namespace that you want to connect to. If not specified, Hyperpod cli commands will auto discover the accessible namespace. |
|
|
177
|
+
| `--region <region>` | Optional | The AWS region where the HyperPod cluster resides. |
|
|
178
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
179
|
+
|
|
180
|
+
#### Getting Cluster Context
|
|
181
|
+
|
|
182
|
+
Get all the context related to the current set Cluster
|
|
183
|
+
|
|
184
|
+
```bash
|
|
185
|
+
hyp get-cluster-context
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
| Option | Type | Description |
|
|
189
|
+
|--------|------|-------------|
|
|
190
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
## CLI
|
|
194
|
+
|
|
195
|
+
### Cluster Management
|
|
196
|
+
|
|
197
|
+
**Important**: For commands that accept the `--region` option, if no region is explicitly provided, the command will use the default region from your AWS credentials configuration.
|
|
198
|
+
|
|
199
|
+
**Cluster stack names must be unique within each AWS region.** If you attempt to create a cluster stack with a name that already exists in the same region, the deployment will fail.
|
|
200
|
+
|
|
201
|
+
#### Initialize Cluster Configuration
|
|
202
|
+
|
|
203
|
+
Initialize a new cluster configuration in the current directory:
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
hyp init cluster-stack
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
**Important**: The `resource_name_prefix` parameter in the generated `config.yaml` file serves as the primary identifier for all AWS resources created during deployment. Each deployment must use a unique resource name prefix to avoid conflicts. This prefix is automatically appended with a unique identifier during cluster creation to ensure resource uniqueness.
|
|
210
|
+
|
|
211
|
+
#### Configure Cluster Parameters
|
|
212
|
+
|
|
213
|
+
Configure cluster parameters interactively or via command line:
|
|
214
|
+
|
|
215
|
+
```bash
|
|
216
|
+
hyp configure --resource-name-prefix my-cluster --stage prod
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
#### Validate Configuration
|
|
220
|
+
|
|
221
|
+
Validate the configuration file syntax:
|
|
222
|
+
|
|
223
|
+
```bash
|
|
224
|
+
hyp validate
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
#### Create Cluster Stack
|
|
228
|
+
|
|
229
|
+
Create the cluster stack using the configured parameters:
|
|
230
|
+
|
|
231
|
+
```bash
|
|
232
|
+
hyp create --region <region>
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
**Note**: The region flag is optional. If not provided, the command will use the default region from your AWS credentials configuration.
|
|
236
|
+
|
|
237
|
+
#### List Cluster Stacks
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
hyp list cluster-stack
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
| Option | Type | Description |
|
|
244
|
+
|--------|------|-------------|
|
|
245
|
+
| `--region <region>` | Optional | The AWS region to list stacks from. |
|
|
246
|
+
| `--status "['CREATE_COMPLETE', 'UPDATE_COMPLETE']"` | Optional | Filter by stack status. |
|
|
247
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
248
|
+
|
|
249
|
+
#### Describe Cluster Stack
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
hyp describe cluster-stack <stack-name>
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
| Option | Type | Description |
|
|
256
|
+
|--------|------|-------------|
|
|
257
|
+
| `--region <region>` | Optional | The AWS region where the stack exists. |
|
|
258
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
259
|
+
|
|
260
|
+
#### Delete Cluster Stack
|
|
261
|
+
|
|
262
|
+
Delete a HyperPod cluster stack. Removes the specified CloudFormation stack and all associated AWS resources. This operation cannot be undone.
|
|
263
|
+
|
|
264
|
+
```bash
|
|
265
|
+
hyp delete cluster-stack <stack-name>
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
| Option | Type | Description |
|
|
269
|
+
|--------|------|-------------|
|
|
270
|
+
| `--region <region>` | Required | The AWS region where the stack exists. |
|
|
271
|
+
| `--retain-resources S3Bucket-TrainingData,EFSFileSystem-Models` | Optional | Comma-separated list of logical resource IDs to retain during deletion (only works on DELETE_FAILED stacks). Resource names are shown in failed deletion output, or use AWS CLI: `aws cloudformation list-stack-resources STACK_NAME --region REGION`. |
|
|
272
|
+
| `--debug` | Optional | Enable debug mode for detailed logging. |
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
#### Update Existing Cluster
|
|
276
|
+
|
|
277
|
+
```bash
|
|
278
|
+
hyp update cluster --cluster-name my-cluster \
|
|
279
|
+
--instance-groups '[{"InstanceCount":2,"InstanceGroupName":"worker-nodes","InstanceType":"ml.m5.large"}]' \
|
|
280
|
+
--node-recovery Automatic
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
#### Reset Configuration
|
|
284
|
+
|
|
285
|
+
Reset configuration to default values:
|
|
286
|
+
|
|
287
|
+
```bash
|
|
288
|
+
hyp reset
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
### Training
|
|
292
|
+
|
|
293
|
+
#### **Option 1**: Create Pytorch job through init experience
|
|
294
|
+
|
|
295
|
+
#### Initialize Pytorch Job Configuration
|
|
296
|
+
|
|
297
|
+
Initialize a new pytorch job configuration in the current directory:
|
|
298
|
+
|
|
299
|
+
```bash
|
|
300
|
+
hyp init hyp-pytorch-job
|
|
301
|
+
```
|
|
302
|
+
|
|
303
|
+
#### Configure Pytorch Job Parameters
|
|
304
|
+
|
|
305
|
+
Configure pytorch job parameters interactively or via command line:
|
|
306
|
+
|
|
307
|
+
```bash
|
|
308
|
+
hyp configure --job-name my-pytorch-job
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
#### Validate Configuration
|
|
312
|
+
|
|
313
|
+
Validate the configuration file syntax:
|
|
314
|
+
|
|
315
|
+
```bash
|
|
316
|
+
hyp validate
|
|
317
|
+
```
|
|
318
|
+
|
|
319
|
+
#### Create Pytorch Job
|
|
320
|
+
|
|
321
|
+
Create the pytorch job using the configured parameters:
|
|
322
|
+
|
|
323
|
+
```bash
|
|
324
|
+
hyp create
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
#### **Option 2**: Create Pytorch job through create command
|
|
329
|
+
|
|
330
|
+
```bash
|
|
331
|
+
hyp create hyp-pytorch-job \
|
|
332
|
+
--version 1.0 \
|
|
333
|
+
--job-name test-pytorch-job \
|
|
334
|
+
--image pytorch/pytorch:latest \
|
|
335
|
+
--command '[python, train.py]' \
|
|
336
|
+
--args '[--epochs=10, --batch-size=32]' \
|
|
337
|
+
--environment '{"PYTORCH_CUDA_ALLOC_CONF": "max_split_size_mb:32"}' \
|
|
338
|
+
--pull-policy "IfNotPresent" \
|
|
339
|
+
--instance-type ml.p4d.24xlarge \
|
|
340
|
+
--tasks-per-node 8 \
|
|
341
|
+
--label-selector '{"accelerator": "nvidia", "network": "efa"}' \
|
|
342
|
+
--deep-health-check-passed-nodes-only true \
|
|
343
|
+
--scheduler-type "kueue" \
|
|
344
|
+
--queue-name "training-queue" \
|
|
345
|
+
--priority "high" \
|
|
346
|
+
--max-retry 3 \
|
|
347
|
+
--accelerators 8 \
|
|
348
|
+
--vcpu 96.0 \
|
|
349
|
+
--memory 1152.0 \
|
|
350
|
+
--accelerators-limit 8 \
|
|
351
|
+
--vcpu-limit 96.0 \
|
|
352
|
+
--memory-limit 1152.0 \
|
|
353
|
+
--preferred-topology "topology.kubernetes.io/zone=us-west-2a" \
|
|
354
|
+
--volume name=model-data,type=hostPath,mount_path=/data,path=/data \
|
|
355
|
+
--volume name=training-output,type=pvc,mount_path=/data2,claim_name=my-pvc,read_only=false
|
|
356
|
+
```
|
|
357
|
+
|
|
358
|
+
| Parameter | Type | Required | Description |
|
|
359
|
+
|-----------|------|----------|-------------|
|
|
360
|
+
| `--job-name` | TEXT | Yes | Unique name for the training job (1-63 characters, alphanumeric with hyphens) |
|
|
361
|
+
| `--image` | TEXT | Yes | Docker image URI containing your training code |
|
|
362
|
+
| `--namespace` | TEXT | No | Kubernetes namespace |
|
|
363
|
+
| `--command` | ARRAY | No | Command to run in the container (array of strings) |
|
|
364
|
+
| `--args` | ARRAY | No | Arguments for the entry script (array of strings) |
|
|
365
|
+
| `--environment` | OBJECT | No | Environment variables as key-value pairs |
|
|
366
|
+
| `--pull-policy` | TEXT | No | Image pull policy (Always, Never, IfNotPresent) |
|
|
367
|
+
| `--instance-type` | TEXT | No | Instance type for training |
|
|
368
|
+
| `--node-count` | INTEGER | No | Number of nodes (minimum: 1) |
|
|
369
|
+
| `--tasks-per-node` | INTEGER | No | Number of tasks per node (minimum: 1) |
|
|
370
|
+
| `--label-selector` | OBJECT | No | Node label selector as key-value pairs |
|
|
371
|
+
| `--deep-health-check-passed-nodes-only` | BOOLEAN | No | Schedule pods only on nodes that passed deep health check (default: false) |
|
|
372
|
+
| `--scheduler-type` | TEXT | No | Scheduler type |
|
|
373
|
+
| `--queue-name` | TEXT | No | Queue name for job scheduling (1-63 characters, alphanumeric with hyphens) |
|
|
374
|
+
| `--priority` | TEXT | No | Priority class for job scheduling |
|
|
375
|
+
| `--max-retry` | INTEGER | No | Maximum number of job retries (minimum: 0) |
|
|
376
|
+
| `--volume` | ARRAY | No | List of volume configurations (Refer [Volume Configuration](#volume-configuration) for detailed parameter info) |
|
|
377
|
+
| `--service-account-name` | TEXT | No | Service account name |
|
|
378
|
+
| `--accelerators` | INTEGER | No | Number of accelerators a.k.a GPUs or Trainium Chips |
|
|
379
|
+
| `--vcpu` | FLOAT | No | Number of vCPUs |
|
|
380
|
+
| `--memory` | FLOAT | No | Amount of memory in GiB |
|
|
381
|
+
| `--accelerators-limit` | INTEGER | No | Limit for the number of accelerators a.k.a GPUs or Trainium Chips |
|
|
382
|
+
| `--vcpu-limit` | FLOAT | No | Limit for the number of vCPUs |
|
|
383
|
+
| `--memory-limit` | FLOAT | No | Limit for the amount of memory in GiB |
|
|
384
|
+
| `--preferred-topology` | TEXT | No | Preferred topology annotation for scheduling |
|
|
385
|
+
| `--required-topology` | TEXT | No | Required topology annotation for scheduling |
|
|
386
|
+
| `--debug` | FLAG | No | Enable debug mode (default: false) |
|
|
387
|
+
|
|
388
|
+
#### List Training Jobs
|
|
389
|
+
|
|
390
|
+
```bash
|
|
391
|
+
hyp list hyp-pytorch-job
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
#### Describe a Training Job
|
|
395
|
+
|
|
396
|
+
```bash
|
|
397
|
+
hyp describe hyp-pytorch-job --job-name <job-name>
|
|
398
|
+
````
|
|
399
|
+
|
|
400
|
+
#### Listing Pods
|
|
401
|
+
|
|
402
|
+
This command lists all the pods associated with a specific training job.
|
|
403
|
+
|
|
404
|
+
```bash
|
|
405
|
+
hyp list-pods hyp-pytorch-job --job-name <job-name>
|
|
406
|
+
```
|
|
407
|
+
|
|
408
|
+
* `job-name` (string) - Required. The name of the job to list pods for.
|
|
409
|
+
|
|
410
|
+
#### Accessing Logs
|
|
411
|
+
|
|
412
|
+
This command retrieves the logs for a specific pod within a training job.
|
|
413
|
+
|
|
414
|
+
```bash
|
|
415
|
+
hyp get-logs hyp-pytorch-job --pod-name <pod-name> --job-name <job-name>
|
|
416
|
+
```
|
|
417
|
+
|
|
418
|
+
| Parameter | Required | Description |
|
|
419
|
+
|--------|------|-------------|
|
|
420
|
+
| `--job-name` | Yes | The name of the job to get the log for. |
|
|
421
|
+
| `--pod-name` | Yes | The name of the pod to get the log from. |
|
|
422
|
+
| `--namespace` | No | The namespace of the job. Defaults to 'default'. |
|
|
423
|
+
| `--container` | No | The container name to get logs from. |
|
|
424
|
+
|
|
425
|
+
#### Get Operator Logs
|
|
426
|
+
|
|
427
|
+
```bash
|
|
428
|
+
hyp get-operator-logs hyp-pytorch-job --since-hours 0.5
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
#### Delete a Training Job
|
|
432
|
+
|
|
433
|
+
```bash
|
|
434
|
+
hyp delete hyp-pytorch-job --job-name <job-name>
|
|
435
|
+
```
|
|
436
|
+
|
|
437
|
+
### Inference
|
|
438
|
+
|
|
439
|
+
### Jumpstart Endpoint Creation
|
|
440
|
+
|
|
441
|
+
#### **Option 1**: Create jumpstart endpoint through init experience
|
|
442
|
+
|
|
443
|
+
#### Initialize Jumpstart Endpoint Configuration
|
|
444
|
+
|
|
445
|
+
Initialize a new jumpstart endpoint configuration in the current directory:
|
|
446
|
+
|
|
447
|
+
```bash
|
|
448
|
+
hyp init hyp-jumpstart-endpoint
|
|
449
|
+
```
|
|
450
|
+
|
|
451
|
+
#### Configure Jumpstart Endpoint Parameters
|
|
452
|
+
|
|
453
|
+
Configure jumpstart endpoint parameters interactively or via command line:
|
|
454
|
+
|
|
455
|
+
```bash
|
|
456
|
+
hyp configure --endpoint-name my-jumpstart-endpoint
|
|
457
|
+
```
|
|
458
|
+
|
|
459
|
+
#### Validate Configuration
|
|
460
|
+
|
|
461
|
+
Validate the configuration file syntax:
|
|
462
|
+
|
|
463
|
+
```bash
|
|
464
|
+
hyp validate
|
|
465
|
+
```
|
|
466
|
+
|
|
467
|
+
#### Create Jumpstart Endpoint
|
|
468
|
+
|
|
469
|
+
Create the jumpstart endpoint using the configured parameters:
|
|
470
|
+
|
|
471
|
+
```bash
|
|
472
|
+
hyp create
|
|
473
|
+
```
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
#### **Option 2**: Create jumpstart endpoint through create command
|
|
477
|
+
Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io/en/v2.82.0/doc_utils/jumpstart.html and fed into the call for creating the endpoint
|
|
478
|
+
|
|
479
|
+
```bash
|
|
480
|
+
hyp create hyp-jumpstart-endpoint \
|
|
481
|
+
--version 1.0 \
|
|
482
|
+
--model-id jumpstart-model-id\
|
|
483
|
+
--instance-type ml.g5.8xlarge \
|
|
484
|
+
--endpoint-name endpoint-jumpstart
|
|
485
|
+
```
|
|
486
|
+
|
|
487
|
+
| Parameter | Type | Required | Description |
|
|
488
|
+
|-----------|------|----------|-------------|
|
|
489
|
+
| `--model-id` | TEXT | Yes | JumpStart model identifier (1-63 characters, alphanumeric with hyphens) |
|
|
490
|
+
| `--instance-type` | TEXT | Yes | EC2 instance type for inference (must start with "ml.") |
|
|
491
|
+
| `--namespace` | TEXT | No | Kubernetes namespace |
|
|
492
|
+
| `--metadata-name` | TEXT | No | Name of the jumpstart endpoint object |
|
|
493
|
+
| `--accept-eula` | BOOLEAN | No | Whether model terms of use have been accepted (default: false) |
|
|
494
|
+
| `--model-version` | TEXT | No | Semantic version of the model (e.g., "1.0.0", 5-14 characters) |
|
|
495
|
+
| `--endpoint-name` | TEXT | No | Name of SageMaker endpoint (1-63 characters, alphanumeric with hyphens) |
|
|
496
|
+
| `--tls-certificate-output-s3-uri` | TEXT | No | S3 URI to write the TLS certificate (optional) |
|
|
497
|
+
| `--debug` | FLAG | No | Enable debug mode (default: false) |
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
#### Invoke a JumpstartModel Endpoint
|
|
501
|
+
|
|
502
|
+
```bash
|
|
503
|
+
hyp invoke hyp-jumpstart-endpoint \
|
|
504
|
+
--endpoint-name endpoint-jumpstart \
|
|
505
|
+
--body '{"inputs":"What is the capital of USA?"}'
|
|
506
|
+
```
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
#### Managing an Endpoint
|
|
510
|
+
|
|
511
|
+
```bash
|
|
512
|
+
hyp list hyp-jumpstart-endpoint
|
|
513
|
+
hyp describe hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
514
|
+
```
|
|
515
|
+
|
|
516
|
+
#### List Pods
|
|
517
|
+
|
|
518
|
+
```bash
|
|
519
|
+
hyp list-pods hyp-jumpstart-endpoint
|
|
520
|
+
```
|
|
521
|
+
|
|
522
|
+
#### Get Logs
|
|
523
|
+
|
|
524
|
+
```bash
|
|
525
|
+
hyp get-logs hyp-jumpstart-endpoint --pod-name <pod-name>
|
|
526
|
+
```
|
|
527
|
+
|
|
528
|
+
#### Get Operator Logs
|
|
529
|
+
|
|
530
|
+
```bash
|
|
531
|
+
hyp get-operator-logs hyp-jumpstart-endpoint --since-hours 0.5
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
#### Deleting an Endpoint
|
|
535
|
+
|
|
536
|
+
```bash
|
|
537
|
+
hyp delete hyp-jumpstart-endpoint --name endpoint-jumpstart
|
|
538
|
+
```
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
### Custom Endpoint Creation
|
|
542
|
+
#### **Option 1**: Create custom endpoint through init experience
|
|
543
|
+
|
|
544
|
+
#### Initialize Custom Endpoint Configuration
|
|
545
|
+
|
|
546
|
+
Initialize a new custom endpoint configuration in the current directory:
|
|
547
|
+
|
|
548
|
+
```bash
|
|
549
|
+
hyp init hyp-custom-endpoint
|
|
550
|
+
```
|
|
551
|
+
|
|
552
|
+
#### Configure Custom Endpoint Parameters
|
|
553
|
+
|
|
554
|
+
Configure custom endpoint parameters interactively or via command line:
|
|
555
|
+
|
|
556
|
+
```bash
|
|
557
|
+
hyp configure --endpoint-name my-custom-endpoint
|
|
558
|
+
```
|
|
559
|
+
|
|
560
|
+
#### Validate Configuration
|
|
561
|
+
|
|
562
|
+
Validate the configuration file syntax:
|
|
563
|
+
|
|
564
|
+
```bash
|
|
565
|
+
hyp validate
|
|
566
|
+
```
|
|
567
|
+
|
|
568
|
+
#### Create Custom Endpoint
|
|
569
|
+
|
|
570
|
+
Create the custom endpoint using the configured parameters:
|
|
571
|
+
|
|
572
|
+
```bash
|
|
573
|
+
hyp create
|
|
574
|
+
```
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
#### **Option 2**: Create custom endpoint through create command
|
|
578
|
+
```bash
|
|
579
|
+
hyp create hyp-custom-endpoint \
|
|
580
|
+
--version 1.0 \
|
|
581
|
+
--endpoint-name endpoint-custom \
|
|
582
|
+
--model-name my-pytorch-model \
|
|
583
|
+
--model-source-type s3 \
|
|
584
|
+
--model-location my-pytorch-training \
|
|
585
|
+
--model-volume-mount-name test-volume \
|
|
586
|
+
--s3-bucket-name your-bucket \
|
|
587
|
+
--s3-region us-east-1 \
|
|
588
|
+
--instance-type ml.g5.8xlarge \
|
|
589
|
+
--image-uri 763104351884.dkr.ecr.us-east-1.amazonaws.com/pytorch-inference:latest \
|
|
590
|
+
--container-port 8080
|
|
591
|
+
```
|
|
592
|
+
|
|
593
|
+
| Parameter | Type | Required | Description |
|
|
594
|
+
|-----------|------|----------|-------------|
|
|
595
|
+
| `--instance-type` | TEXT | Yes | EC2 instance type for inference (must start with "ml.") |
|
|
596
|
+
| `--model-name` | TEXT | Yes | Name of model to create on SageMaker (1-63 characters, alphanumeric with hyphens) |
|
|
597
|
+
| `--model-source-type` | TEXT | Yes | Model source type ("s3" or "fsx") |
|
|
598
|
+
| `--image-uri` | TEXT | Yes | Docker image URI for inference |
|
|
599
|
+
| `--container-port` | INTEGER | Yes | Port on which model server listens (1-65535) |
|
|
600
|
+
| `--model-volume-mount-name` | TEXT | Yes | Name of the model volume mount |
|
|
601
|
+
| `--namespace` | TEXT | No | Kubernetes namespace |
|
|
602
|
+
| `--metadata-name` | TEXT | No | Name of the custom endpoint object |
|
|
603
|
+
| `--endpoint-name` | TEXT | No | Name of SageMaker endpoint (1-63 characters, alphanumeric with hyphens) |
|
|
604
|
+
| `--env` | OBJECT | No | Environment variables as key-value pairs |
|
|
605
|
+
| `--metrics-enabled` | BOOLEAN | No | Enable metrics collection (default: false) |
|
|
606
|
+
| `--model-version` | TEXT | No | Version of the model (semantic version format) |
|
|
607
|
+
| `--model-location` | TEXT | No | Specific model data location |
|
|
608
|
+
| `--prefetch-enabled` | BOOLEAN | No | Whether to pre-fetch model data (default: false) |
|
|
609
|
+
| `--tls-certificate-output-s3-uri` | TEXT | No | S3 URI for TLS certificate output |
|
|
610
|
+
| `--fsx-dns-name` | TEXT | No | FSx File System DNS Name |
|
|
611
|
+
| `--fsx-file-system-id` | TEXT | No | FSx File System ID |
|
|
612
|
+
| `--fsx-mount-name` | TEXT | No | FSx File System Mount Name |
|
|
613
|
+
| `--s3-bucket-name` | TEXT | No | S3 bucket location |
|
|
614
|
+
| `--s3-region` | TEXT | No | S3 bucket region |
|
|
615
|
+
| `--model-volume-mount-path` | TEXT | No | Path inside container for model volume (default: "/opt/ml/model") |
|
|
616
|
+
| `--resources-limits` | OBJECT | No | Resource limits for the worker |
|
|
617
|
+
| `--resources-requests` | OBJECT | No | Resource requests for the worker |
|
|
618
|
+
| `--dimensions` | OBJECT | No | CloudWatch Metric dimensions as key-value pairs |
|
|
619
|
+
| `--metric-collection-period` | INTEGER | No | Period for CloudWatch query (default: 300) |
|
|
620
|
+
| `--metric-collection-start-time` | INTEGER | No | StartTime for CloudWatch query (default: 300) |
|
|
621
|
+
| `--metric-name` | TEXT | No | Metric name to query for CloudWatch trigger |
|
|
622
|
+
| `--metric-stat` | TEXT | No | Statistics metric for CloudWatch (default: "Average") |
|
|
623
|
+
| `--metric-type` | TEXT | No | Type of metric for HPA ("Value" or "Average", default: "Average") |
|
|
624
|
+
| `--min-value` | NUMBER | No | Minimum metric value for empty CloudWatch response (default: 0) |
|
|
625
|
+
| `--cloud-watch-trigger-name` | TEXT | No | Name for the CloudWatch trigger |
|
|
626
|
+
| `--cloud-watch-trigger-namespace` | TEXT | No | AWS CloudWatch namespace for the metric |
|
|
627
|
+
| `--target-value` | NUMBER | No | Target value for the CloudWatch metric |
|
|
628
|
+
| `--use-cached-metrics` | BOOLEAN | No | Enable caching of metric values (default: true) |
|
|
629
|
+
| `--invocation-endpoint` | TEXT | No | Invocation endpoint path (default: "invocations") |
|
|
630
|
+
| `--debug` | FLAG | No | Enable debug mode (default: false) |
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
#### Invoke a Custom Inference Endpoint
|
|
634
|
+
|
|
635
|
+
```bash
|
|
636
|
+
hyp invoke hyp-custom-endpoint \
|
|
637
|
+
--endpoint-name endpoint-custom-pytorch \
|
|
638
|
+
--body '{"inputs":"What is the capital of USA?"}'
|
|
639
|
+
```
|
|
640
|
+
|
|
641
|
+
#### Managing an Endpoint
|
|
642
|
+
|
|
643
|
+
```bash
|
|
644
|
+
hyp list hyp-custom-endpoint
|
|
645
|
+
hyp describe hyp-custom-endpoint --name endpoint-custom
|
|
646
|
+
```
|
|
647
|
+
|
|
648
|
+
#### List Pods
|
|
649
|
+
|
|
650
|
+
```bash
|
|
651
|
+
hyp list-pods hyp-custom-endpoint
|
|
652
|
+
```
|
|
653
|
+
|
|
654
|
+
#### Get Logs
|
|
655
|
+
|
|
656
|
+
```bash
|
|
657
|
+
hyp get-logs hyp-custom-endpoint --pod-name <pod-name>
|
|
658
|
+
```
|
|
659
|
+
|
|
660
|
+
#### Get Operator Logs
|
|
661
|
+
|
|
662
|
+
```bash
|
|
663
|
+
hyp get-operator-logs hyp-custom-endpoint --since-hours 0.5
|
|
664
|
+
```
|
|
665
|
+
|
|
666
|
+
#### Deleting an Endpoint
|
|
667
|
+
|
|
668
|
+
```bash
|
|
669
|
+
hyp delete hyp-custom-endpoint --name endpoint-custom
|
|
670
|
+
```
|
|
671
|
+
|
|
672
|
+
## SDK
|
|
673
|
+
|
|
674
|
+
Along with the CLI, we also have SDKs available that can perform the cluster management, training and inference functionalities that the CLI performs
|
|
675
|
+
|
|
676
|
+
### Cluster Management SDK
|
|
677
|
+
|
|
678
|
+
#### Creating a Cluster Stack
|
|
679
|
+
|
|
680
|
+
```python
|
|
681
|
+
from sagemaker.hyperpod.cluster_management.hp_cluster_stack import HpClusterStack
|
|
682
|
+
|
|
683
|
+
# Initialize cluster stack configuration
|
|
684
|
+
cluster_stack = HpClusterStack(
|
|
685
|
+
stage="prod",
|
|
686
|
+
resource_name_prefix="my-hyperpod",
|
|
687
|
+
hyperpod_cluster_name="my-hyperpod-cluster",
|
|
688
|
+
eks_cluster_name="my-hyperpod-eks",
|
|
689
|
+
|
|
690
|
+
# Infrastructure components
|
|
691
|
+
create_vpc_stack=True,
|
|
692
|
+
create_eks_cluster_stack=True,
|
|
693
|
+
create_hyperpod_cluster_stack=True,
|
|
694
|
+
|
|
695
|
+
# Network configuration
|
|
696
|
+
vpc_cidr="10.192.0.0/16",
|
|
697
|
+
availability_zone_ids=["use2-az1", "use2-az2"],
|
|
698
|
+
|
|
699
|
+
# Instance group configuration
|
|
700
|
+
instance_group_settings=[
|
|
701
|
+
{
|
|
702
|
+
"InstanceCount": 1,
|
|
703
|
+
"InstanceGroupName": "controller-group",
|
|
704
|
+
"InstanceType": "ml.t3.medium",
|
|
705
|
+
"TargetAvailabilityZoneId": "use2-az2"
|
|
706
|
+
}
|
|
707
|
+
]
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
# Create the cluster stack
|
|
711
|
+
response = cluster_stack.create(region="us-east-2")
|
|
712
|
+
```
|
|
713
|
+
|
|
714
|
+
#### Listing Cluster Stacks
|
|
715
|
+
|
|
716
|
+
```python
|
|
717
|
+
# List all cluster stacks
|
|
718
|
+
stacks = HpClusterStack.list(region="us-east-2")
|
|
719
|
+
print(f"Found {len(stacks['StackSummaries'])} stacks")
|
|
720
|
+
```
|
|
721
|
+
|
|
722
|
+
#### Describing a Cluster Stack
|
|
723
|
+
|
|
724
|
+
```python
|
|
725
|
+
# Describe a specific cluster stack
|
|
726
|
+
stack_info = HpClusterStack.describe("my-stack-name", region="us-east-2")
|
|
727
|
+
print(f"Stack status: {stack_info['Stacks'][0]['StackStatus']}")
|
|
728
|
+
```
|
|
729
|
+
|
|
730
|
+
#### Monitoring Cluster Status
|
|
731
|
+
|
|
732
|
+
```python
|
|
733
|
+
from sagemaker.hyperpod.cluster_management.hp_cluster_stack import HpClusterStack
|
|
734
|
+
|
|
735
|
+
stack = HpClusterStack()
|
|
736
|
+
response = stack.create(region="us-west-2")
|
|
737
|
+
status = stack.get_status(region="us-west-2")
|
|
738
|
+
print(status)
|
|
739
|
+
```
|
|
740
|
+
|
|
741
|
+
### Training SDK
|
|
742
|
+
|
|
743
|
+
#### Creating a Training Job
|
|
744
|
+
|
|
745
|
+
```python
|
|
746
|
+
from sagemaker.hyperpod.training.hyperpod_pytorch_job import HyperPodPytorchJob
|
|
747
|
+
from sagemaker.hyperpod.training.config.hyperpod_pytorch_job_unified_config import (
|
|
748
|
+
ReplicaSpec, Template, Spec, Containers, Resources, RunPolicy
|
|
749
|
+
)
|
|
750
|
+
from sagemaker.hyperpod.common.config.metadata import Metadata
|
|
751
|
+
|
|
752
|
+
# Define job specifications
|
|
753
|
+
nproc_per_node = "1" # Number of processes per node
|
|
754
|
+
replica_specs =
|
|
755
|
+
[
|
|
756
|
+
ReplicaSpec
|
|
757
|
+
(
|
|
758
|
+
name = "pod", # Replica name
|
|
759
|
+
template = Template
|
|
760
|
+
(
|
|
761
|
+
spec = Spec
|
|
762
|
+
(
|
|
763
|
+
containers =
|
|
764
|
+
[
|
|
765
|
+
Containers
|
|
766
|
+
(
|
|
767
|
+
# Container name
|
|
768
|
+
name="container-name",
|
|
769
|
+
|
|
770
|
+
# Training image
|
|
771
|
+
image="123456789012.dkr.ecr.us-west-2.amazonaws.com/my-training-image:latest",
|
|
772
|
+
|
|
773
|
+
# Always pull image
|
|
774
|
+
image_pull_policy="Always",
|
|
775
|
+
resources=Resources\
|
|
776
|
+
(
|
|
777
|
+
# No GPUs requested
|
|
778
|
+
requests={"nvidia.com/gpu": "0"},
|
|
779
|
+
# No GPU limit
|
|
780
|
+
limits={"nvidia.com/gpu": "0"},
|
|
781
|
+
),
|
|
782
|
+
# Command to run
|
|
783
|
+
command=["python", "train.py"],
|
|
784
|
+
# Script arguments
|
|
785
|
+
args=["--epochs", "10", "--batch-size", "32"],
|
|
786
|
+
)
|
|
787
|
+
]
|
|
788
|
+
)
|
|
789
|
+
),
|
|
790
|
+
)
|
|
791
|
+
]
|
|
792
|
+
# Keep pods after completion
|
|
793
|
+
run_policy = RunPolicy(clean_pod_policy="None")
|
|
794
|
+
|
|
795
|
+
# Create and start the PyTorch job
|
|
796
|
+
pytorch_job = HyperPodPytorchJob
|
|
797
|
+
(
|
|
798
|
+
# Job name
|
|
799
|
+
metadata = Metadata(name="demo"),
|
|
800
|
+
# Processes per node
|
|
801
|
+
nproc_per_node = nproc_per_node,
|
|
802
|
+
# Replica specifications
|
|
803
|
+
replica_specs = replica_specs,
|
|
804
|
+
# Run policy
|
|
805
|
+
run_policy = run_policy,
|
|
806
|
+
)
|
|
807
|
+
# Launch the job
|
|
808
|
+
pytorch_job.create()
|
|
809
|
+
```
|
|
810
|
+
|
|
811
|
+
#### List Training Jobs
|
|
812
|
+
```python
|
|
813
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
814
|
+
import yaml
|
|
815
|
+
|
|
816
|
+
# List all PyTorch jobs
|
|
817
|
+
jobs = HyperPodPytorchJob.list()
|
|
818
|
+
print(yaml.dump(jobs))
|
|
819
|
+
```
|
|
820
|
+
|
|
821
|
+
#### Describe a Training Job
|
|
822
|
+
```python
|
|
823
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
824
|
+
|
|
825
|
+
# Get an existing job
|
|
826
|
+
job = HyperPodPytorchJob.get(name="my-pytorch-job")
|
|
827
|
+
|
|
828
|
+
print(job)
|
|
829
|
+
```
|
|
830
|
+
|
|
831
|
+
#### List Pods for a Training Job
|
|
832
|
+
```python
|
|
833
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
834
|
+
|
|
835
|
+
# List Pods for an existing job
|
|
836
|
+
job = HyperPodPytorchJob.get(name="my-pytorch-job")
|
|
837
|
+
print(job.list_pods())
|
|
838
|
+
```
|
|
839
|
+
|
|
840
|
+
#### Get Logs from a Pod
|
|
841
|
+
```python
|
|
842
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
843
|
+
|
|
844
|
+
# Get pod logs for a job
|
|
845
|
+
job = HyperPodPytorchJob.get(name="my-pytorch-job")
|
|
846
|
+
print(job.get_logs_from_pod("pod-name"))
|
|
847
|
+
```
|
|
848
|
+
|
|
849
|
+
#### Get Training Operator Logs
|
|
850
|
+
```python
|
|
851
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
852
|
+
|
|
853
|
+
# Get training operator logs
|
|
854
|
+
job = HyperPodPytorchJob.get(name="my-pytorch-job")
|
|
855
|
+
print(job.get_operator_logs(since_hours=0.1))
|
|
856
|
+
```
|
|
857
|
+
|
|
858
|
+
#### Delete a Training Job
|
|
859
|
+
```python
|
|
860
|
+
from sagemaker.hyperpod.training import HyperPodPytorchJob
|
|
861
|
+
|
|
862
|
+
# Get an existing job
|
|
863
|
+
job = HyperPodPytorchJob.get(name="my-pytorch-job")
|
|
864
|
+
|
|
865
|
+
# Delete the job
|
|
866
|
+
job.delete()
|
|
867
|
+
```
|
|
868
|
+
|
|
869
|
+
### Inference SDK
|
|
870
|
+
|
|
871
|
+
#### Creating a JumpstartModel Endpoint
|
|
872
|
+
|
|
873
|
+
Pre-trained Jumpstart models can be gotten from https://sagemaker.readthedocs.io/en/v2.82.0/doc_utils/jumpstart.html and fed into the call for creating the endpoint
|
|
874
|
+
|
|
875
|
+
```python
|
|
876
|
+
from sagemaker.hyperpod.inference.config.hp_jumpstart_endpoint_config import Model, Server, SageMakerEndpoint, TlsConfig
|
|
877
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
878
|
+
|
|
879
|
+
model=Model(
|
|
880
|
+
model_id='deepseek-llm-r1-distill-qwen-1-5b'
|
|
881
|
+
)
|
|
882
|
+
server=Server(
|
|
883
|
+
instance_type='ml.g5.8xlarge',
|
|
884
|
+
)
|
|
885
|
+
endpoint_name=SageMakerEndpoint(name='<my-endpoint-name>')
|
|
886
|
+
|
|
887
|
+
js_endpoint=HPJumpStartEndpoint(
|
|
888
|
+
model=model,
|
|
889
|
+
server=server,
|
|
890
|
+
sage_maker_endpoint=endpoint_name
|
|
891
|
+
)
|
|
892
|
+
|
|
893
|
+
js_endpoint.create()
|
|
894
|
+
```
|
|
895
|
+
|
|
896
|
+
#### Creating a Custom Inference Endpoint (with S3)
|
|
897
|
+
|
|
898
|
+
```python
|
|
899
|
+
from sagemaker.hyperpod.inference.config.hp_endpoint_config import CloudWatchTrigger, Dimensions, AutoScalingSpec, Metrics, S3Storage, ModelSourceConfig, TlsConfig, EnvironmentVariables, ModelInvocationPort, ModelVolumeMount, Resources, Worker
|
|
900
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
901
|
+
|
|
902
|
+
model_source_config = ModelSourceConfig(
|
|
903
|
+
model_source_type='s3',
|
|
904
|
+
model_location="<my-model-folder-in-s3>",
|
|
905
|
+
s3_storage=S3Storage(
|
|
906
|
+
bucket_name='<my-model-artifacts-bucket>',
|
|
907
|
+
region='us-east-2',
|
|
908
|
+
),
|
|
909
|
+
)
|
|
910
|
+
|
|
911
|
+
environment_variables = [
|
|
912
|
+
EnvironmentVariables(name="HF_MODEL_ID", value="/opt/ml/model"),
|
|
913
|
+
EnvironmentVariables(name="SAGEMAKER_PROGRAM", value="inference.py"),
|
|
914
|
+
EnvironmentVariables(name="SAGEMAKER_SUBMIT_DIRECTORY", value="/opt/ml/model/code"),
|
|
915
|
+
EnvironmentVariables(name="MODEL_CACHE_ROOT", value="/opt/ml/model"),
|
|
916
|
+
EnvironmentVariables(name="SAGEMAKER_ENV", value="1"),
|
|
917
|
+
]
|
|
918
|
+
|
|
919
|
+
worker = Worker(
|
|
920
|
+
image='763104351884.dkr.ecr.us-east-2.amazonaws.com/huggingface-pytorch-tgi-inference:2.4.0-tgi2.3.1-gpu-py311-cu124-ubuntu22.04-v2.0',
|
|
921
|
+
model_volume_mount=ModelVolumeMount(
|
|
922
|
+
name='model-weights',
|
|
923
|
+
),
|
|
924
|
+
model_invocation_port=ModelInvocationPort(container_port=8080),
|
|
925
|
+
resources=Resources(
|
|
926
|
+
requests={"cpu": "30000m", "nvidia.com/gpu": 1, "memory": "100Gi"},
|
|
927
|
+
limits={"nvidia.com/gpu": 1}
|
|
928
|
+
),
|
|
929
|
+
environment_variables=environment_variables,
|
|
930
|
+
)
|
|
931
|
+
|
|
932
|
+
tls_config=TlsConfig(tls_certificate_output_s3_uri='s3://<my-tls-bucket-name>')
|
|
933
|
+
|
|
934
|
+
custom_endpoint = HPEndpoint(
|
|
935
|
+
endpoint_name='<my-endpoint-name>',
|
|
936
|
+
instance_type='ml.g5.8xlarge',
|
|
937
|
+
model_name='deepseek15b-test-model-name',
|
|
938
|
+
tls_config=tls_config,
|
|
939
|
+
model_source_config=model_source_config,
|
|
940
|
+
worker=worker,
|
|
941
|
+
)
|
|
942
|
+
|
|
943
|
+
custom_endpoint.create()
|
|
944
|
+
```
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
#### List Endpoints
|
|
948
|
+
|
|
949
|
+
```python
|
|
950
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
951
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
952
|
+
|
|
953
|
+
# List JumpStart endpoints
|
|
954
|
+
jumpstart_endpoints = HPJumpStartEndpoint.list()
|
|
955
|
+
print(jumpstart_endpoints)
|
|
956
|
+
|
|
957
|
+
# List custom endpoints
|
|
958
|
+
custom_endpoints = HPEndpoint.list()
|
|
959
|
+
print(custom_endpoints)
|
|
960
|
+
```
|
|
961
|
+
|
|
962
|
+
#### Describe an Endpoint
|
|
963
|
+
```python
|
|
964
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
965
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
966
|
+
|
|
967
|
+
# Get JumpStart endpoint details
|
|
968
|
+
jumpstart_endpoint = HPJumpStartEndpoint.get(name="js-endpoint-name", namespace="test")
|
|
969
|
+
print(jumpstart_endpoint)
|
|
970
|
+
|
|
971
|
+
# Get custom endpoint details
|
|
972
|
+
custom_endpoint = HPEndpoint.get(name="endpoint-custom")
|
|
973
|
+
print(custom_endpoint)
|
|
974
|
+
|
|
975
|
+
```
|
|
976
|
+
|
|
977
|
+
#### Invoke an Endpoint
|
|
978
|
+
```python
|
|
979
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
980
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
981
|
+
|
|
982
|
+
data = '{"inputs":"What is the capital of USA?"}'
|
|
983
|
+
jumpstart_endpoint = HPJumpStartEndpoint.get(name="endpoint-jumpstart")
|
|
984
|
+
response = jumpstart_endpoint.invoke(body=data).body.read()
|
|
985
|
+
print(response)
|
|
986
|
+
|
|
987
|
+
custom_endpoint = HPEndpoint.get(name="endpoint-custom")
|
|
988
|
+
response = custom_endpoint.invoke(body=data).body.read()
|
|
989
|
+
print(response)
|
|
990
|
+
```
|
|
991
|
+
|
|
992
|
+
#### List Pods
|
|
993
|
+
```python
|
|
994
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
995
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
996
|
+
|
|
997
|
+
# List pods
|
|
998
|
+
js_pods = HPJumpStartEndpoint.list_pods()
|
|
999
|
+
print(js_pods)
|
|
1000
|
+
|
|
1001
|
+
c_pods = HPEndpoint.list_pods()
|
|
1002
|
+
print(c_pods)
|
|
1003
|
+
```
|
|
1004
|
+
|
|
1005
|
+
#### Get Logs
|
|
1006
|
+
```python
|
|
1007
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
1008
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
1009
|
+
|
|
1010
|
+
# Get logs from pod
|
|
1011
|
+
js_logs = HPJumpStartEndpoint.get_logs(pod=<pod-name>)
|
|
1012
|
+
print(js_logs)
|
|
1013
|
+
|
|
1014
|
+
c_logs = HPEndpoint.get_logs(pod=<pod-name>)
|
|
1015
|
+
print(c_logs)
|
|
1016
|
+
```
|
|
1017
|
+
|
|
1018
|
+
#### Get Operator Logs
|
|
1019
|
+
```python
|
|
1020
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
1021
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
1022
|
+
|
|
1023
|
+
# Invoke JumpStart endpoint
|
|
1024
|
+
print(HPJumpStartEndpoint.get_operator_logs(since_hours=0.1))
|
|
1025
|
+
|
|
1026
|
+
# Invoke custom endpoint
|
|
1027
|
+
print(HPEndpoint.get_operator_logs(since_hours=0.1))
|
|
1028
|
+
```
|
|
1029
|
+
|
|
1030
|
+
#### Delete an Endpoint
|
|
1031
|
+
```python
|
|
1032
|
+
from sagemaker.hyperpod.inference.hp_jumpstart_endpoint import HPJumpStartEndpoint
|
|
1033
|
+
from sagemaker.hyperpod.inference.hp_endpoint import HPEndpoint
|
|
1034
|
+
|
|
1035
|
+
# Delete JumpStart endpoint
|
|
1036
|
+
jumpstart_endpoint = HPJumpStartEndpoint.get(name="endpoint-jumpstart")
|
|
1037
|
+
jumpstart_endpoint.delete()
|
|
1038
|
+
|
|
1039
|
+
# Delete custom endpoint
|
|
1040
|
+
custom_endpoint = HPEndpoint.get(name="endpoint-custom")
|
|
1041
|
+
custom_endpoint.delete()
|
|
1042
|
+
```
|
|
1043
|
+
|
|
1044
|
+
|
|
1045
|
+
#### Observability - Getting Monitoring Information
|
|
1046
|
+
```python
|
|
1047
|
+
from sagemaker.hyperpod.observability.utils import get_monitoring_config
|
|
1048
|
+
monitor_config = get_monitoring_config()
|
|
1049
|
+
```
|
|
1050
|
+
|
|
1051
|
+
## Examples
|
|
1052
|
+
#### Cluster Management Example Notebooks
|
|
1053
|
+
|
|
1054
|
+
[CLI Cluster Management Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/cluster_management/cluster_creation_init_experience.ipynb)
|
|
1055
|
+
|
|
1056
|
+
[SDK Cluster Management Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/cluster_management/cluster_creation_sdk_experience.ipynb)
|
|
1057
|
+
|
|
1058
|
+
#### Training Example Notebooks
|
|
1059
|
+
|
|
1060
|
+
[CLI Training Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/CLI/training-init-experience.ipynb)
|
|
1061
|
+
|
|
1062
|
+
[CLI Training Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/CLI/training-e2e-cli.ipynb)
|
|
1063
|
+
|
|
1064
|
+
[SDK Training Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/training/SDK/training_sdk_example.ipynb)
|
|
1065
|
+
|
|
1066
|
+
#### Inference Example Notebooks
|
|
1067
|
+
|
|
1068
|
+
##### CLI
|
|
1069
|
+
[CLI Inference Jumpstart Model Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-jumpstart-init-experience.ipynb)
|
|
1070
|
+
|
|
1071
|
+
[CLI Inference JumpStart Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-jumpstart-e2e-cli.ipynb)
|
|
1072
|
+
|
|
1073
|
+
[CLI Inference FSX Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-fsx-model-e2e-cli.ipynb)
|
|
1074
|
+
|
|
1075
|
+
[CLI Inference S3 Model Init Experience Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-s3-model-init-experience.ipynb)
|
|
1076
|
+
|
|
1077
|
+
[CLI Inference S3 Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/CLI/inference-s3-model-e2e-cli.ipynb)
|
|
1078
|
+
|
|
1079
|
+
##### SDK
|
|
1080
|
+
|
|
1081
|
+
[SDK Inference JumpStart Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-jumpstart-e2e.ipynb)
|
|
1082
|
+
|
|
1083
|
+
[SDK Inference FSX Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-fsx-model-e2e.ipynb)
|
|
1084
|
+
|
|
1085
|
+
[SDK Inference S3 Model Example](https://github.com/aws/sagemaker-hyperpod-cli/blob/main/examples/inference/SDK/inference-s3-model-e2e.ipynb)
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
## Disclaimer
|
|
1089
|
+
|
|
1090
|
+
* This CLI and SDK requires access to the user's file system to set and get context and function properly.
|
|
1091
|
+
It needs to read configuration files such as kubeconfig to establish the necessary environment settings.
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
## Working behind a proxy server ?
|
|
1095
|
+
* Follow these steps from [here](https://docs.aws.amazon.com/cli/v1/userguide/cli-configure-proxy.html) to set up HTTP proxy connections
|
|
1096
|
+
|