gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,240 @@
1
+ """Register S3 data files with Gen3 indexd.
2
+
3
+ Scans one or more S3 prefixes, collects file metadata (name, md5, size),
4
+ registers each file with the indexd API, and writes results to Glue
5
+ Iceberg tables for downstream use by dbt silver models.
6
+
7
+ Usage
8
+ -----
9
+ python register_indexd.py \
10
+ --s3-paths s3://bucket/path1 s3://bucket/path2 \
11
+ --study edcad \
12
+ --env staging
13
+
14
+ Studies come from the g3dt.yaml marker or s3://<metadata-bucket>/config/studies.yaml.
15
+ """
16
+
17
+ import sys
18
+ import logging
19
+ import argparse
20
+ import json
21
+
22
+ import yaml
23
+ from gen3.auth import Gen3Auth
24
+ from gen3.index import Gen3Index
25
+
26
+ from g3dt.upload.metadata_submitter import (
27
+ create_boto3_session,
28
+ get_gen3_api_key_aws_secret,
29
+ infer_api_endpoint_from_jwt,
30
+ )
31
+ from g3dt.indexd.indexd_registrar import (
32
+ scan_s3_files,
33
+ register_files_with_indexd,
34
+ write_to_glue,
35
+ )
36
+
37
+
38
+ def setup_logger():
39
+ logger = logging.getLogger()
40
+ logger.setLevel(logging.INFO)
41
+ if not logger.handlers:
42
+ handler = logging.StreamHandler(sys.stdout)
43
+ formatter = logging.Formatter(
44
+ "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
45
+ )
46
+ handler.setFormatter(formatter)
47
+ logger.addHandler(handler)
48
+ return logger
49
+
50
+
51
+ # Shared config resolution (SSM-backed) — see src/g3dt/config.py
52
+ from g3dt import config as g3dt_config, resolver # noqa: E402
53
+
54
+
55
+ def main():
56
+ logger = setup_logger()
57
+
58
+ parser = argparse.ArgumentParser(
59
+ description="Register S3 files with Gen3 indexd"
60
+ )
61
+ parser.add_argument(
62
+ "--s3-paths",
63
+ nargs="+",
64
+ required=True,
65
+ help="One or more S3 prefixes to scan for files",
66
+ )
67
+ parser.add_argument(
68
+ "--study",
69
+ required=True,
70
+ help="Study key (e.g. edcad, ausdiab, caughtcad)",
71
+ )
72
+ parser.add_argument(
73
+ "--env",
74
+ required=True,
75
+ help="Environment (e.g. test, staging, prod, "
76
+ "staging_ec2, prod_ec2)",
77
+ )
78
+ parser.add_argument(
79
+ "--dry-run",
80
+ action="store_true",
81
+ help="Scan and write file_metadata only; skip indexd "
82
+ "registration",
83
+ )
84
+
85
+ args = parser.parse_args()
86
+
87
+ # Env facts + resource names from SSM; the study registry from the marker
88
+ # or s3://<metadata-bucket>/config/studies.yaml.
89
+ try:
90
+ env_cfg = g3dt_config.resolve_env(args.env)
91
+ study_cfg = g3dt_config.resolve_study(args.study, args.env)
92
+ except g3dt_config.ConfigError as exc:
93
+ logger.error(str(exc))
94
+ sys.exit(1)
95
+
96
+ project_id = study_cfg.project_id
97
+ program_id = study_cfg.program_id
98
+ authz = [f"/programs/{program_id}/projects/{project_id}"]
99
+
100
+ aws_secret_name = env_cfg.aws_secret_name
101
+ aws_profile = env_cfg.aws_profile
102
+ aws_region = env_cfg.region
103
+
104
+ rc = resolver.resolve(
105
+ g3dt_config.require_project(),
106
+ g3dt_config.env_base(args.env),
107
+ profile=aws_profile,
108
+ )
109
+ athena_s3_output = rc.athena_output_location
110
+ workgroup = rc.athena_workgroup
111
+
112
+ # Indexd tables: conventional names in the env's metadata DB/bucket
113
+ # (exactly like the CDK's `releases` table).
114
+ fm_database = rc.metadata_db
115
+ fm_table = g3dt_config.FILE_METADATA_TABLE
116
+ reg_database = rc.metadata_db
117
+ reg_table = g3dt_config.INDEXD_REGISTRY_TABLE
118
+ indexd_s3_path = f"s3://{rc.metadata_bucket}/{g3dt_config.INDEXD_PREFIX}"
119
+
120
+ # --- boto3 session ---
121
+ session = create_boto3_session(
122
+ aws_profile=aws_profile, aws_region=aws_region
123
+ )
124
+
125
+ # --- Fetch API key early so endpoint can be stamped on tables ---
126
+ if aws_secret_name.startswith("/"):
127
+ logger.info(
128
+ "Loading Gen3 API key from local file: %s",
129
+ aws_secret_name,
130
+ )
131
+ with open(aws_secret_name, "r") as f:
132
+ api_key = json.load(f)
133
+ else:
134
+ logger.info(
135
+ "Fetching Gen3 API key from Secrets Manager: %s",
136
+ aws_secret_name,
137
+ )
138
+ api_key = get_gen3_api_key_aws_secret(
139
+ secret_name=aws_secret_name,
140
+ region_name=aws_region,
141
+ session=session,
142
+ )
143
+
144
+ base_url = infer_api_endpoint_from_jwt(api_key["api_key"])
145
+ indexd_endpoint = base_url.replace("api/v0", "index/index")
146
+ logger.info("Indexd endpoint: %s", indexd_endpoint)
147
+
148
+ # --- Phase 1: scan S3 ---
149
+ logger.info(
150
+ "Scanning %d S3 path(s): %s",
151
+ len(args.s3_paths),
152
+ args.s3_paths,
153
+ )
154
+ file_df = scan_s3_files(args.s3_paths, boto3_session=session)
155
+ file_df["study_id"] = args.study
156
+ file_df["indexd_endpoint"] = indexd_endpoint
157
+ logger.info("Found %d files.", len(file_df))
158
+
159
+ if file_df.empty:
160
+ logger.warning("No files found. Exiting.")
161
+ sys.exit(0)
162
+
163
+ # --- Write file_metadata table ---
164
+ fm_location = (
165
+ f"{indexd_s3_path.rstrip('/')}/file_metadata/"
166
+ if indexd_s3_path
167
+ else None
168
+ )
169
+ logger.info(
170
+ "Writing file metadata to %s.%s", fm_database, fm_table
171
+ )
172
+ write_to_glue(
173
+ df=file_df,
174
+ database=fm_database,
175
+ table=fm_table,
176
+ athena_s3_output=athena_s3_output,
177
+ table_location=fm_location,
178
+ workgroup=workgroup,
179
+ partition_cols=["study_id", "indexd_endpoint"],
180
+ merge_cols=["file_name", "study_id", "indexd_endpoint"],
181
+ schema_evolution=True,
182
+ boto3_session=session,
183
+ )
184
+
185
+ if args.dry_run:
186
+ logger.info("Dry run — skipping indexd registration.")
187
+ sys.exit(0)
188
+
189
+ auth = Gen3Auth(refresh_token=api_key)
190
+ index = Gen3Index(auth)
191
+
192
+ logger.info(
193
+ "Registering %d files with indexd (authz=%s)",
194
+ len(file_df),
195
+ authz,
196
+ )
197
+ registry_df = register_files_with_indexd(
198
+ index, file_df, authz
199
+ )
200
+ logger.info(
201
+ "Successfully registered %d / %d files.",
202
+ len(registry_df),
203
+ len(file_df),
204
+ )
205
+
206
+ if registry_df.empty:
207
+ logger.warning(
208
+ "No new registrations. All files may already exist."
209
+ )
210
+ sys.exit(0)
211
+
212
+ # --- Write indexd_registry table ---
213
+ reg_location = (
214
+ f"{indexd_s3_path.rstrip('/')}/indexd_registry/"
215
+ if indexd_s3_path
216
+ else None
217
+ )
218
+ logger.info(
219
+ "Writing indexd registry to %s.%s",
220
+ reg_database,
221
+ reg_table,
222
+ )
223
+ write_to_glue(
224
+ df=registry_df,
225
+ database=reg_database,
226
+ table=reg_table,
227
+ athena_s3_output=athena_s3_output,
228
+ table_location=reg_location,
229
+ workgroup=workgroup,
230
+ partition_cols=["study_id", "indexd_endpoint"],
231
+ merge_cols=["did"],
232
+ schema_evolution=True,
233
+ boto3_session=session,
234
+ )
235
+
236
+ logger.info("Done.")
237
+
238
+
239
+ if __name__ == "__main__":
240
+ main()
@@ -0,0 +1,140 @@
1
+ #!/bin/bash
2
+ # Script to restart gen3 etl cronjob
3
+ usage() {
4
+ echo "Usage: $0 [-e ENVIRONMENT] [-d DOMAIN] [-a APPNAME] [-n NAMESPACE] [-c ETL_CRONJOB] [-t CONTAINER] [-l] [-s]"
5
+ echo " -e ENVIRONMENT Environment profile (e.g. test, staging, prod). Config comes from G3DT_* env vars set by the g3dt CLI."
6
+ echo " -d DOMAIN The domain for argocd login (example: cd.cad.test.biocommons.org.au)"
7
+ echo " -a APPNAME The application name (example: uatgen3)"
8
+ echo " -n NAMESPACE The namespace for the resources (example: cad)"
9
+ echo " -c ETL_CRONJOB The name of the ETL cronjob to run (default: etl-cronjob)"
10
+ echo " -t CONTAINER The name of the container to check logs from (default: tube)"
11
+ echo " -l Bypass login"
12
+ echo " -s Sync the argocd app before restarting resources"
13
+ exit 1
14
+ }
15
+
16
+ set -eo pipefail
17
+
18
+ # default values
19
+ ETL_CRONJOB="etl-cronjob"
20
+ CONTAINER_TO_CHECK="tube"
21
+ LOGIN_REQUIRED=true
22
+ SYNC_APP=false
23
+ ENVIRONMENT=""
24
+
25
+ while getopts "e:d:a:n:c:t:hls" opt; do
26
+ case ${opt} in
27
+ e ) ENVIRONMENT=$OPTARG ;;
28
+ d ) DOMAIN=$OPTARG ;;
29
+ a ) APPNAME=$OPTARG ;;
30
+ n ) NAMESPACE=$OPTARG ;;
31
+ c ) ETL_CRONJOB=$OPTARG ;;
32
+ t ) CONTAINER_TO_CHECK=$OPTARG ;;
33
+ l ) LOGIN_REQUIRED=false ;;
34
+ s ) SYNC_APP=true ;;
35
+ h ) usage ;;
36
+ \? ) usage ;;
37
+ esac
38
+ done
39
+
40
+ if ! command -v argocd &> /dev/null; then
41
+ echo "argocd CLI could not be found. Please install it to proceed."
42
+ exit 1
43
+ fi
44
+
45
+ # If an environment is specified (display-only), set up AWS profile, kubeconfig,
46
+ # and ArgoCD details from G3DT_* env vars exported by the g3dt CLI.
47
+ if [ -n "$ENVIRONMENT" ]; then
48
+ CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI}"
49
+ REGION="${G3DT_REGION:-ap-southeast-2}"
50
+
51
+ # Set DOMAIN, APPNAME, NAMESPACE from env vars if not already set via flags
52
+ DOMAIN="${DOMAIN:-${G3DT_DOMAIN:?G3DT_DOMAIN not set — run via the g3dt CLI}}"
53
+ APPNAME="${APPNAME:-${G3DT_APP_NAME:?G3DT_APP_NAME not set — run via the g3dt CLI}}"
54
+ NAMESPACE="${NAMESPACE:-${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI}}"
55
+
56
+ # Never export an empty AWS_PROFILE (empty means ambient credentials).
57
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
58
+ export AWS_PROFILE="${G3DT_AWS_PROFILE}"
59
+ echo "==== Configuring AWS PROFILE as '${AWS_PROFILE}' for environment '${ENVIRONMENT}' ===="
60
+ else
61
+ echo "==== No AWS profile set for '${ENVIRONMENT}' — using ambient AWS credentials ===="
62
+ fi
63
+
64
+ echo "Updating kubeconfig for cluster: ${CLUSTER_NAME}"
65
+ if ! aws eks update-kubeconfig --name "${CLUSTER_NAME}" --region "${REGION}"; then
66
+ echo "ERROR: Failed to update kubeconfig. Check your AWS credentials/profile."
67
+ exit 1
68
+ fi
69
+ fi
70
+
71
+ # Save the current kubectl context before argocd login (which changes it)
72
+ KUBE_CONTEXT=$(kubectl config current-context 2>/dev/null)
73
+
74
+ if [ "$LOGIN_REQUIRED" = true ]; then
75
+ echo "logging into argocd via sso"
76
+ argocd login --sso $DOMAIN
77
+ echo "login successful"
78
+ else
79
+ echo "Bypassing login as per user request"
80
+ fi
81
+
82
+ # Restore the kubectl context after argocd login
83
+ if [ -n "$KUBE_CONTEXT" ]; then
84
+ echo "Restoring kubectl context to: ${KUBE_CONTEXT}"
85
+ kubectl config use-context "$KUBE_CONTEXT"
86
+ fi
87
+
88
+ if [ "$SYNC_APP" = true ]; then
89
+ echo "Syncing ArgoCD app: $APPNAME"
90
+ argocd app sync $APPNAME
91
+ if [ $? -eq 0 ]; then
92
+ echo "App $APPNAME synced successfully."
93
+ else
94
+ echo "Failed to sync app $APPNAME."
95
+ exit 1
96
+ fi
97
+ fi
98
+
99
+ RESOURCE="${ETL_CRONJOB}-$(date +%Y%m%d%H%M)"
100
+ JOB_NAME=$(kubectl create job --from=cronjob/${ETL_CRONJOB} ${RESOURCE} --namespace ${NAMESPACE} -o name | awk -F'/' '{print $2}')
101
+
102
+ if [ -z "$JOB_NAME" ]; then
103
+ echo "Failed to create job. Exiting."
104
+ exit 1
105
+ fi
106
+
107
+ echo "Job '$JOB_NAME' created. Waiting for it to complete..."
108
+
109
+ while true; do
110
+ SUCCEEDED=$(kubectl get job $JOB_NAME -n $NAMESPACE -o jsonpath='{.status.succeeded}')
111
+ FAILED=$(kubectl get job $JOB_NAME -n $NAMESPACE -o jsonpath='{.status.failed}')
112
+
113
+ if [ "$SUCCEEDED" == "1" ]; then
114
+ echo "✅ Job '$JOB_NAME' completed successfully!"
115
+ break
116
+ elif [ "$FAILED" == "1" ]; then
117
+ echo "⚠️ Job '$JOB_NAME' failed."
118
+ break
119
+ else
120
+ echo "⏳ Job '$JOB_NAME' is still running... (Succeeded: ${SUCCEEDED:-0}, Failed: ${FAILED:-0})"
121
+ sleep 10
122
+ fi
123
+ done
124
+
125
+
126
+ # Even if pod fails, it may still have passed, need to check logs
127
+ POD_NAME=$(kubectl get pods -n $NAMESPACE -l job-name=$JOB_NAME -o jsonpath='{.items[0].metadata.name}')
128
+ LOGS=$(kubectl logs $POD_NAME -n $NAMESPACE -c $CONTAINER_TO_CHECK)
129
+
130
+
131
+ if echo "$LOGS" | grep -q "Exit code: 0"; then
132
+ echo "✅ ✅ ✅Log check passed! Found 'Exit code: 0'. The job is truly successful."
133
+ else
134
+ echo "⚠️ Job Failed, as well as the log validation! Please review output:"
135
+ echo "--------------------- LOGS START ---------------------"
136
+ echo "$LOGS"
137
+ echo "---------------------- LOGS END ----------------------"
138
+ fi
139
+
140
+ echo "Script finished."
@@ -0,0 +1,102 @@
1
+ #!/bin/bash
2
+ # Script to restart argocd microservices for schema redeployment
3
+
4
+ # Function to display help message
5
+ usage() {
6
+ echo "Usage: $0 [-d DOMAIN] [-a APPNAME] [-r RESOURCES] [-n NAMESPACE] [-k KIND] [-l] [-s]"
7
+ echo " -d DOMAIN The domain for argocd login (example: cd.cad.test.biocommons.org.au)"
8
+ echo " -a APPNAME The application name (example: uatgen3)"
9
+ echo " -r RESOURCES Comma-separated string of microservice names to restart (example: sheepdog-deployment,peregrine-deployment )"
10
+ echo " -n NAMESPACE The namespace for the resources (example: cad)"
11
+ echo " -k KIND The kind of resource to restart (default: Deployment)"
12
+ echo " -l Bypass login"
13
+ echo " -s Sync the argocd app before restarting resources"
14
+ exit 1
15
+ }
16
+
17
+ set -eo pipefail
18
+
19
+ # Parse command line arguments
20
+ LOGIN_REQUIRED=true
21
+ KIND="Deployment"
22
+ SYNC_APP=false
23
+ while getopts "d:a:r:n:k:hls" opt; do
24
+ case ${opt} in
25
+ d )
26
+ DOMAIN=$OPTARG
27
+ ;;
28
+ a )
29
+ APPNAME=$OPTARG
30
+ ;;
31
+ r )
32
+ IFS=',' read -r -a RESOURCES <<< "$OPTARG"
33
+ ;;
34
+ n )
35
+ NAMESPACE=$OPTARG
36
+ ;;
37
+ k )
38
+ KIND=$OPTARG
39
+ ;;
40
+ l )
41
+ LOGIN_REQUIRED=false
42
+ ;;
43
+ s )
44
+ SYNC_APP=true
45
+ ;;
46
+ h )
47
+ usage
48
+ ;;
49
+ \? )
50
+ usage
51
+ ;;
52
+ esac
53
+ done
54
+
55
+ # Check if argocd is installed
56
+ if ! command -v argocd &> /dev/null
57
+ then
58
+ echo "argocd CLI could not be found. Please install it to proceed."
59
+ exit 1
60
+ fi
61
+
62
+ if [ "$LOGIN_REQUIRED" = true ]; then
63
+ echo "logging into argocd via sso"
64
+ argocd login --sso $DOMAIN
65
+ echo "login successful"
66
+ else
67
+ echo "Bypassing login as per user request"
68
+ fi
69
+
70
+ if [ "$SYNC_APP" = true ]; then
71
+ echo "Syncing ArgoCD app: $APPNAME"
72
+ argocd app sync $APPNAME
73
+ if [ $? -eq 0 ]; then
74
+ echo "App $APPNAME synced successfully."
75
+ else
76
+ echo "Failed to sync app $APPNAME."
77
+ exit 1
78
+ fi
79
+ fi
80
+
81
+ # Iterate through resources
82
+ for RESOURCE in "${RESOURCES[@]}"; do
83
+ echo "Restarting resource: $RESOURCE"
84
+
85
+ # Run the restart action
86
+ argocd app actions run $APPNAME restart --kind $KIND --resource-name $RESOURCE --namespace $NAMESPACE
87
+
88
+ sleep 5
89
+
90
+ # Wait for the resource to complete its restart
91
+ echo "Waiting for $RESOURCE to finish..."
92
+ while true; do
93
+ echo "Checking $RESOURCE status..."
94
+ STATUS=$(argocd app get $APPNAME -o json | jq --arg RESOURCE "$RESOURCE" '.status.resources[] | select(.name == $RESOURCE) | .health.status')
95
+ echo "Status: $STATUS"
96
+ if [ "$STATUS" == "\"Healthy\"" ]; then
97
+ echo "$RESOURCE restarted successfully"
98
+ break
99
+ fi
100
+ sleep 5 # Check every 5 seconds
101
+ done
102
+ done
@@ -0,0 +1,106 @@
1
+ #!/bin/bash
2
+ # Script to restart argocd microservices for schema redeployment
3
+
4
+ # Function to display help message
5
+ usage() {
6
+ echo "Usage: $0 [-d DOMAIN] [-a APPNAME] [-r RESOURCES] [-n NAMESPACE] [-k KIND] [-l] [-s]"
7
+ echo " -d DOMAIN The domain for argocd login (example: cd.cad.test.biocommons.org.au)"
8
+ echo " -a APPNAME The application name (example: uatgen3)"
9
+ echo " -r RESOURCES Comma-separated string of microservice names to restart (default: \"sheepdog-deployment\" \"peregrine-deployment\" \"guppy-deployment\" \"portal-deployment\")"
10
+ echo " -n NAMESPACE The namespace for the resources (default: cad)"
11
+ echo " -k KIND The kind of resource to restart (default: Deployment)"
12
+ echo " -l Bypass login"
13
+ echo " -s Run 'argocd app sync' before restarts"
14
+ exit 1
15
+ }
16
+
17
+ set -eo pipefail
18
+
19
+ # Set default values
20
+ RESOURCES=("sheepdog-deployment" "peregrine-deployment" "guppy-deployment" "portal-deployment")
21
+ NAMESPACE="cad"
22
+ KIND="Deployment"
23
+ LOGIN_REQUIRED=true
24
+ SYNC_APP=false
25
+
26
+ # Parse command line arguments
27
+ while getopts "d:a:r:n:k:hls" opt; do
28
+ case ${opt} in
29
+ d )
30
+ DOMAIN=$OPTARG
31
+ ;;
32
+ a )
33
+ APPNAME=$OPTARG
34
+ ;;
35
+ r )
36
+ IFS=',' read -r -a RESOURCES <<< "$OPTARG"
37
+ ;;
38
+ n )
39
+ NAMESPACE=$OPTARG
40
+ ;;
41
+ k )
42
+ KIND=$OPTARG
43
+ ;;
44
+ l )
45
+ LOGIN_REQUIRED=false
46
+ ;;
47
+ s )
48
+ SYNC_APP=true
49
+ ;;
50
+ h )
51
+ usage
52
+ ;;
53
+ \? )
54
+ usage
55
+ ;;
56
+ esac
57
+ done
58
+
59
+ # Check if argocd is installed
60
+ if ! command -v argocd &> /dev/null
61
+ then
62
+ echo "argocd CLI could not be found. Please install it to proceed."
63
+ exit 1
64
+ fi
65
+
66
+ if [ "$LOGIN_REQUIRED" = true ]; then
67
+ echo "logging into argocd via sso"
68
+ argocd login --sso $DOMAIN
69
+ echo "login successful"
70
+ else
71
+ echo "Bypassing login as per user request"
72
+ fi
73
+
74
+ if [ "$SYNC_APP" = true ]; then
75
+ echo "Syncing ArgoCD app: $APPNAME"
76
+ argocd app sync $APPNAME
77
+ if [ $? -eq 0 ]; then
78
+ echo "App $APPNAME synced successfully."
79
+ else
80
+ echo "Failed to sync app $APPNAME."
81
+ exit 1
82
+ fi
83
+ fi
84
+
85
+ # Iterate through resources
86
+ for RESOURCE in "${RESOURCES[@]}"; do
87
+ echo "Restarting resource: $RESOURCE"
88
+
89
+ # Run the restart action
90
+ argocd app actions run $APPNAME restart --kind $KIND --resource-name $RESOURCE --namespace $NAMESPACE
91
+
92
+ sleep 5
93
+
94
+ # Wait for the resource to complete its restart
95
+ echo "Waiting for $RESOURCE to finish..."
96
+ while true; do
97
+ echo "Checking $RESOURCE status..."
98
+ STATUS=$(argocd app get $APPNAME -o json | jq --arg RESOURCE "$RESOURCE" '.status.resources[] | select(.name == $RESOURCE) | .health.status')
99
+ echo "Status: $STATUS"
100
+ if [ "$STATUS" == "\"Healthy\"" ]; then
101
+ echo "$RESOURCE restarted successfully"
102
+ break
103
+ fi
104
+ sleep 5 # Check every 5 seconds
105
+ done
106
+ done
@@ -0,0 +1,110 @@
1
+ #!/bin/bash
2
+
3
+ # login_testenv_pod.sh
4
+ # Script to log into a specified pod in a Kubernetes environment using a grep pattern.
5
+
6
+ set -e
7
+
8
+ show_help() {
9
+ cat << EOF
10
+ Usage: ${0##*/} [OPTIONS]
11
+
12
+ Log into a running pod on aws. Direct SSO identity is used by default.
13
+ Also make sure you have an aws SSO profile configured on AWS cli.
14
+
15
+ Defaults come from the G3DT_* environment variables exported by the g3dt CLI
16
+ when set; flags always override them.
17
+
18
+ Options:
19
+ -p PROFILE AWS CLI profile to use (default: \$G3DT_AWS_PROFILE, else ambient credentials)
20
+ -r REGION AWS region (default: \$G3DT_REGION, else ap-southeast-2)
21
+ -c CLUSTER_NAME EKS cluster name (default: \$G3DT_CLUSTER_NAME)
22
+ -a ROLE_ARN Optional AWS IAM role ARN to assume (default: \$G3DT_EKS_ARN)
23
+ -n NAMESPACE Kubernetes namespace (default: \$G3DT_NAMESPACE, else cad)
24
+ -g GREP_PATTERN Pattern to grep for pod name (default: guppy)
25
+ -x Print all resolved commands before running them
26
+ -h Show this help message and exit
27
+
28
+ Example:
29
+ ${0##*/} -p myprofile -n mynamespace -g sheepdog
30
+
31
+ EOF
32
+ }
33
+
34
+ set -eo pipefail
35
+
36
+ # Default values (taken from G3DT_* env vars when the g3dt CLI exported them)
37
+ PROFILE="${G3DT_AWS_PROFILE:-}"
38
+ AWS_REGION="${G3DT_REGION:-ap-southeast-2}"
39
+ CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI or pass -c}"
40
+ ROLE_ARN="${G3DT_EKS_ARN:-}"
41
+ NAME_SPACE="${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI or pass -n}"
42
+ GREP_PATTERN="guppy"
43
+ PRINT_COMMANDS=0
44
+
45
+ while getopts "p:r:c:a:n:g:xh" opt; do
46
+ case $opt in
47
+ p) PROFILE="$OPTARG" ;;
48
+ r) AWS_REGION="$OPTARG" ;;
49
+ c) CLUSTER_NAME="$OPTARG" ;;
50
+ a) ROLE_ARN="$OPTARG" ;;
51
+ n) NAME_SPACE="$OPTARG" ;;
52
+ g) GREP_PATTERN="$OPTARG" ;;
53
+ x) PRINT_COMMANDS=1 ;;
54
+ h)
55
+ show_help
56
+ exit 0
57
+ ;;
58
+ \?)
59
+ show_help >&2
60
+ exit 1
61
+ ;;
62
+ esac
63
+ done
64
+
65
+ # Prepare commands
66
+ CMD_AWS_SSO_LOGIN="aws sso login --profile ${PROFILE}"
67
+ CMD_AWS_EKS_UPDATE="aws eks update-kubeconfig --name ${CLUSTER_NAME} --region ${AWS_REGION} --profile ${PROFILE}"
68
+ if [[ -n "$ROLE_ARN" ]]; then
69
+ CMD_AWS_EKS_UPDATE="${CMD_AWS_EKS_UPDATE} --role-arn ${ROLE_ARN}"
70
+ fi
71
+ CMD_GET_POD="kubectl get pods -n \"${NAME_SPACE}\" | grep Running | grep \"${GREP_PATTERN}\" | awk '{print \$1}'"
72
+
73
+ if [[ $PRINT_COMMANDS -eq 1 ]]; then
74
+ echo "Resolved commands to be run:"
75
+ echo ""
76
+ echo "# 1. AWS SSO Login"
77
+ echo "$CMD_AWS_SSO_LOGIN"
78
+ echo ""
79
+ echo "# 2. Update kubeconfig"
80
+ echo "$CMD_AWS_EKS_UPDATE"
81
+ echo ""
82
+ echo "# 3. Get pod name"
83
+ echo "$CMD_GET_POD"
84
+ echo ""
85
+ fi
86
+
87
+ echo "Logging into aws sso with profile: $PROFILE"
88
+ eval "$CMD_AWS_SSO_LOGIN"
89
+
90
+ echo "Updating Kubernetes context for cluster: $CLUSTER_NAME in region: $AWS_REGION"
91
+ eval "$CMD_AWS_EKS_UPDATE"
92
+
93
+ # Get pod name using grep pattern
94
+ POD_NAME=$(kubectl get pods -n "${NAME_SPACE}" | grep Running | grep "${GREP_PATTERN}" | awk '{print $1}')
95
+
96
+ if [[ -z "$POD_NAME" ]]; then
97
+ echo "Error: No running pod matching pattern '${GREP_PATTERN}' found in namespace '${NAME_SPACE}'." >&2
98
+ exit 1
99
+ fi
100
+
101
+ CMD_KUBECTL_EXEC="kubectl exec -it \"$POD_NAME\" -n \"${NAME_SPACE}\" -- bash"
102
+
103
+ if [[ $PRINT_COMMANDS -eq 1 ]]; then
104
+ echo "# 4. Exec into pod"
105
+ echo "$CMD_KUBECTL_EXEC"
106
+ echo ""
107
+ fi
108
+
109
+ echo "Logging into pod: $POD_NAME"
110
+ kubectl exec -it "$POD_NAME" -n "${NAME_SPACE}" -- bash