gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,108 @@
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+
4
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
5
+
6
+ usage() {
7
+ cat <<EOF
8
+ Usage: $(basename "$0") --studies <comma-separated-studies> --env <environment>
9
+
10
+ Run upload_metadata.py sequentially for each study.
11
+
12
+ Arguments:
13
+ --studies Comma-separated list of study config keys (e.g. ausdiab_staging,caughtcad_staging)
14
+ --env Environment string passed to the Python script (e.g. staging_ec2)
15
+
16
+ Run via the g3dt CLI:
17
+ g3dt metadata upload-all \\
18
+ --studies ausdiab_staging,caughtcad_staging,edcad_staging \\
19
+ --env staging_ec2
20
+
21
+ Failure logs are written under ~/.g3dt/logs/.
22
+ EOF
23
+ exit 1
24
+ }
25
+
26
+ # ---------- Parse arguments ----------
27
+ STUDIES=""
28
+ ENV=""
29
+
30
+ while [[ $# -gt 0 ]]; do
31
+ case "$1" in
32
+ --studies)
33
+ STUDIES="$2"
34
+ shift 2
35
+ ;;
36
+ --env)
37
+ ENV="$2"
38
+ shift 2
39
+ ;;
40
+ *)
41
+ echo "ERROR: Unknown argument: $1"
42
+ usage
43
+ ;;
44
+ esac
45
+ done
46
+
47
+ if [[ -z "$STUDIES" || -z "$ENV" ]]; then
48
+ echo "ERROR: --studies and --env are required."
49
+ usage
50
+ fi
51
+
52
+ # ---------- Production safety check ----------
53
+ if echo "$ENV $STUDIES" | grep -qi "prod"; then
54
+ echo "ERROR: Production environment detected in arguments."
55
+ echo " This script is not intended for production use. Aborting."
56
+ exit 1
57
+ fi
58
+
59
+ # ---------- Setup ----------
60
+ # Logs go outside the installed package.
61
+ LOG_DIR="$HOME/.g3dt/logs"
62
+ TIMESTAMP="$(date +%Y%m%d_%H%M%S)"
63
+ FAILED_LOG="${LOG_DIR}/${TIMESTAMP}_bulk_upload_failed.log"
64
+ mkdir -p "${LOG_DIR}"
65
+
66
+ IFS=',' read -ra STUDY_LIST <<< "$STUDIES"
67
+ FAIL_COUNT=0
68
+
69
+ echo "============================================"
70
+ echo "Bulk upload started at $(date)"
71
+ echo "Environment : ${ENV}"
72
+ echo "Studies : ${STUDIES}"
73
+ echo "Failure log : ${FAILED_LOG}"
74
+ echo "============================================"
75
+ echo ""
76
+
77
+ # ---------- Sequential execution ----------
78
+ for study in "${STUDY_LIST[@]}"; do
79
+ echo "--------------------------------------------"
80
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] Starting upload for study: ${study}"
81
+ echo "--------------------------------------------"
82
+
83
+ if python3 "${SCRIPT_DIR}/upload_metadata.py" \
84
+ --study "$study" --env "$ENV"; then
85
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] Completed successfully: ${study}"
86
+ else
87
+ EXIT_CODE=$?
88
+ FAIL_COUNT=$((FAIL_COUNT + 1))
89
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] FAILED: ${study} (exit code ${EXIT_CODE})"
90
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] ${study} exit_code=${EXIT_CODE}" >> "$FAILED_LOG"
91
+ fi
92
+
93
+ echo ""
94
+ done
95
+
96
+ # ---------- Summary ----------
97
+ echo "============================================"
98
+ echo "Bulk upload finished at $(date)"
99
+ echo "Total studies : ${#STUDY_LIST[@]}"
100
+ echo "Failures : ${FAIL_COUNT}"
101
+
102
+ if [[ $FAIL_COUNT -gt 0 ]]; then
103
+ echo "See failure details: ${FAILED_LOG}"
104
+ exit 1
105
+ fi
106
+
107
+ echo "All studies uploaded successfully."
108
+ exit 0
@@ -0,0 +1,152 @@
1
+ import sys
2
+ import logging
3
+ import argparse
4
+ import yaml
5
+ import json
6
+ from g3dt.upload.metadata_submitter import (
7
+ list_metadata_jsons_s3,
8
+ find_data_import_order_file_s3,
9
+ create_boto3_session,
10
+ get_gen3_api_key_aws_secret,
11
+ MetadataSubmitter,
12
+ )
13
+
14
+
15
+ def setup_logger(debug=False):
16
+ logger = logging.getLogger()
17
+ logger.setLevel(logging.DEBUG if debug else logging.INFO)
18
+ if not logger.handlers:
19
+ handler = logging.StreamHandler(sys.stdout)
20
+ formatter = logging.Formatter(
21
+ '%(asctime)s - %(name)s - %(levelname)s - %(message)s'
22
+ )
23
+ handler.setFormatter(formatter)
24
+ logger.addHandler(handler)
25
+ return logger
26
+
27
+
28
+ # Shared config resolution (SSM-backed) — see src/g3dt/config.py
29
+ from g3dt import config as g3dt_config, resolver # noqa: E402
30
+
31
+
32
+ def main():
33
+ parser = argparse.ArgumentParser(description="Unified Metadata Upload Script")
34
+ parser.add_argument(
35
+ "--study", required=True, help="Study name (e.g., ausdiab, caughtcad, edcad)"
36
+ )
37
+ parser.add_argument(
38
+ "--env", required=True,
39
+ help="Environment to upload to (e.g., test, staging, prod, staging_ec2, prod_ec2)"
40
+ )
41
+ parser.add_argument("--specific-node", help="Submit only a specific node")
42
+ parser.add_argument(
43
+ "--debug", action="store_true",
44
+ help="Enable debug logging"
45
+ )
46
+
47
+ args = parser.parse_args()
48
+
49
+ logger = setup_logger(debug=args.debug)
50
+
51
+ # Env facts + resource names from SSM; the study registry from the marker
52
+ # or s3://<metadata-bucket>/config/studies.yaml.
53
+ try:
54
+ env_cfg = g3dt_config.resolve_env(args.env)
55
+ study_cfg = g3dt_config.resolve_study(args.study, args.env)
56
+ except g3dt_config.ConfigError as exc:
57
+ logger.error(str(exc))
58
+ sys.exit(1)
59
+
60
+ project_id = study_cfg.project_id
61
+ program_id = study_cfg.program_id
62
+ base_dir = study_cfg.s3_metadata_path
63
+
64
+ aws_secret_name = env_cfg.aws_secret_name
65
+ aws_profile = env_cfg.aws_profile
66
+ aws_region = env_cfg.region
67
+
68
+ rc = resolver.resolve(
69
+ g3dt_config.require_project(),
70
+ g3dt_config.env_base(args.env),
71
+ profile=aws_profile,
72
+ )
73
+ # Upload-tracking table: conventional name in the env's metadata DB/bucket
74
+ # (exactly like the CDK's `releases` table).
75
+ database = rc.metadata_db
76
+ table = g3dt_config.METADATA_UPLOAD_TABLE
77
+ table_location = f"s3://{rc.metadata_bucket}/{g3dt_config.METADATA_UPLOAD_PREFIX}"
78
+ athena_s3_output = rc.athena_output_location
79
+ workgroup = rc.athena_workgroup
80
+
81
+ logger.info(
82
+ f"Starting metadata submission for {args.study} "
83
+ f"in {args.env} environment."
84
+ )
85
+ logger.info(f"Project ID: {project_id}, Program ID: {program_id}")
86
+ logger.info(f"S3 Path: {base_dir}")
87
+
88
+ session = create_boto3_session(
89
+ aws_profile=aws_profile, aws_region=aws_region
90
+ )
91
+
92
+ logger.info("Listing metadata JSON files in S3 directory.")
93
+ metadata_files = list_metadata_jsons_s3(s3_uri=base_dir, session=session)
94
+ if not metadata_files:
95
+ raise FileNotFoundError(
96
+ f"No metadata *.json files found in {base_dir}"
97
+ )
98
+ logger.info(
99
+ f"Located {len(metadata_files)} metadata *.json files in {base_dir}"
100
+ )
101
+
102
+ logger.info("Finding DataImportOrder.txt in the S3 directory.")
103
+ import_order_file_path = find_data_import_order_file_s3(
104
+ s3_uri=base_dir, session=session
105
+ )
106
+ logger.info(f"Import order file found: {import_order_file_path}")
107
+
108
+ # Fetch Gen3 API key
109
+ if aws_secret_name.startswith('/'):
110
+ logger.info(f"Loading Gen3 API key from local file: {aws_secret_name}")
111
+ with open(aws_secret_name, 'r') as f:
112
+ api_key = json.load(f)
113
+ else:
114
+ logger.info(
115
+ f"Fetching Gen3 API key from AWS Secrets Manager: {aws_secret_name}"
116
+ )
117
+ api_key = get_gen3_api_key_aws_secret(
118
+ secret_name=aws_secret_name,
119
+ region_name=aws_region,
120
+ session=session,
121
+ )
122
+
123
+ submitter = MetadataSubmitter(
124
+ metadata_file_list=metadata_files,
125
+ api_key=api_key,
126
+ project_id=project_id,
127
+ data_import_order_path=import_order_file_path,
128
+ database=database,
129
+ table=table,
130
+ athena_s3_output=athena_s3_output,
131
+ workgroup=workgroup,
132
+ table_location=table_location,
133
+ program_id=program_id,
134
+ max_size_kb=100, # Default to 100 as seen in most scripts
135
+ max_retries=3,
136
+ aws_profile=aws_profile,
137
+ aws_region=aws_region,
138
+ )
139
+
140
+ try:
141
+ logger.info("Submitting metadata to Gen3.")
142
+ submitter.submit_metadata(specific_node=args.specific_node)
143
+ logger.info(
144
+ f"Finished submitting metadata for {args.study} ({project_id})"
145
+ )
146
+ except Exception as e:
147
+ logger.error(f"An unexpected error occurred during submission: {e}")
148
+ sys.exit(1)
149
+
150
+
151
+ if __name__ == "__main__":
152
+ main()
@@ -0,0 +1 @@
1
+ from .upload_synthdata_s3 import *
@@ -0,0 +1,265 @@
1
+ import os
2
+ import sys
3
+ import time
4
+ import logging
5
+ from gen3.submission import Gen3Submission
6
+ from typing import Optional, List
7
+ import pandas as pd
8
+ from g3dt.utils.athena_utils import AthenaQuery, AthenaConfig
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+
13
+ def query_metadata_upload_guids(
14
+ database: str,
15
+ table: str,
16
+ project_id: str,
17
+ api_endpoint: str,
18
+ version: str,
19
+ athena_s3_output: str,
20
+ workgroup: str = "primary",
21
+ aws_region: str = "ap-southeast-2",
22
+ aws_profile: Optional[str] = None,
23
+ node: Optional[str] = None,
24
+ ) -> pd.DataFrame:
25
+ """
26
+ Queries the Athena metadata upload table for records matching a given
27
+ project_id, version, and api_endpoint. Optionally filters by node.
28
+
29
+ Args:
30
+ database (str): The Athena database name.
31
+ table (str): The Athena table name.
32
+ project_id (str): The compound project ID (e.g. "program1-CDAH").
33
+ api_endpoint (str): The Gen3 API endpoint URL.
34
+ version (str): The metadata version to filter on (e.g. "0.8.1").
35
+ athena_s3_output (str): S3 URI for Athena query results.
36
+ workgroup (str, optional): Athena workgroup. Default is "primary".
37
+ aws_region (str, optional): AWS region. Default is "ap-southeast-2".
38
+ aws_profile (str, optional): AWS profile name.
39
+ node (str, optional): Node name to filter on (e.g. "subject").
40
+
41
+ Returns:
42
+ pd.DataFrame: DataFrame of matching records including gen3_guid.
43
+ """
44
+ athena_config = AthenaConfig(
45
+ aws_region=aws_region,
46
+ aws_profile=aws_profile,
47
+ athena_s3_output=athena_s3_output,
48
+ )
49
+ athena_query = AthenaQuery(athena_config)
50
+
51
+ sql = (
52
+ f'SELECT * FROM "{database}"."{table}" '
53
+ f"WHERE project_id = '{project_id}' "
54
+ f"AND version = '{version}' "
55
+ f"AND api_endpoint = '{api_endpoint}'"
56
+ )
57
+ if node:
58
+ sql += f" AND node = '{node}'"
59
+
60
+ logger.info(
61
+ "Querying Athena for metadata upload records: "
62
+ "project_id=%s, version=%s, api_endpoint=%s, node=%s",
63
+ project_id,
64
+ version,
65
+ api_endpoint,
66
+ node or "ALL",
67
+ )
68
+ df = athena_query.query_athena(
69
+ sql=sql,
70
+ athena_database=database,
71
+ ctas_approach=False,
72
+ )
73
+ logger.info("Query returned %s records.", len(df))
74
+ return df
75
+
76
+
77
+ def delete_records_by_guid(
78
+ gen3_submission: Gen3Submission,
79
+ program_id: str,
80
+ project_id: str,
81
+ uuids: List[str],
82
+ batch_size: int = 40,
83
+ batch_delay: float = 0.5,
84
+ verbose: bool = False,
85
+ ):
86
+ """
87
+ Deletes Gen3 records one at a time using the SDK's
88
+ delete_record method. UUIDs are grouped into batches
89
+ for rate-limiting only (a pause between each batch).
90
+
91
+ Errors are caught and logged per-UUID so that one
92
+ failure does not stop the rest of the deletions.
93
+
94
+ Args:
95
+ gen3_submission (Gen3Submission): An authenticated
96
+ Gen3Submission instance.
97
+ program_id (str): The Gen3 program name.
98
+ project_id (str): The Gen3 project name.
99
+ uuids (list[str]): List of gen3_guid UUIDs to delete.
100
+ batch_size (int, optional): Number of UUIDs to process
101
+ before pausing. Default is 40.
102
+ batch_delay (float, optional): Seconds to pause between
103
+ batches. Default is 0.5.
104
+ verbose (bool, optional): If True, log the full API
105
+ response JSON for each request. Default is False.
106
+ """
107
+ if not uuids:
108
+ logger.info("No UUIDs provided for deletion. Skipping.")
109
+ return
110
+
111
+ batches = [
112
+ uuids[i: i + batch_size]
113
+ for i in range(0, len(uuids), batch_size)
114
+ ]
115
+ total_batches = len(batches)
116
+ total = len(uuids)
117
+
118
+ logger.info(
119
+ "Deleting %s records in %s batches "
120
+ "(batch_size=%s)...",
121
+ total, total_batches, batch_size,
122
+ )
123
+
124
+ success_count = 0
125
+ failed_ids = []
126
+
127
+ for idx, batch in enumerate(batches, start=1):
128
+ batch_success = 0
129
+ for uuid in batch:
130
+ try:
131
+ if verbose:
132
+ resp = gen3_submission.delete_record(
133
+ program_id, project_id, uuid,
134
+ )
135
+ logger.debug(
136
+ "%s | response: %s",
137
+ uuid, resp,
138
+ )
139
+ else:
140
+ with open(os.devnull, "w") as devnull:
141
+ old_stdout = sys.stdout
142
+ sys.stdout = devnull
143
+ try:
144
+ gen3_submission.delete_record(
145
+ program_id, project_id,
146
+ uuid,
147
+ )
148
+ finally:
149
+ sys.stdout = old_stdout
150
+ batch_success += 1
151
+ except Exception as e:
152
+ failed_ids.append(uuid)
153
+ try:
154
+ body = e.response.json()
155
+ code = body.get("code", "?")
156
+ if verbose:
157
+ msg = body
158
+ else:
159
+ ents = body.get("entities", [])
160
+ errs = ents[0].get("errors", [])
161
+ msg = errs[0].get(
162
+ "message", str(e),
163
+ )
164
+ except Exception:
165
+ code = "?"
166
+ msg = str(e)
167
+ logger.warning(
168
+ "\033[91m[FAIL]\033[0m %s | %s | %s",
169
+ uuid, code, msg,
170
+ )
171
+
172
+ success_count += batch_success
173
+ if batch_success == len(batch):
174
+ logger.info(
175
+ "\033[92m[Batch %d/%d]\033[0m "
176
+ "Deleted %d/%d",
177
+ idx, total_batches,
178
+ batch_success, len(batch),
179
+ )
180
+ else:
181
+ logger.info(
182
+ "[Batch %d/%d] Deleted %d/%d",
183
+ idx, total_batches,
184
+ batch_success, len(batch),
185
+ )
186
+
187
+ if idx < total_batches:
188
+ time.sleep(batch_delay)
189
+
190
+ if failed_ids:
191
+ logger.info(
192
+ "Deletion complete. "
193
+ "\033[92mSuccessful: %s\033[0m, "
194
+ "\033[91mFailed: %s\033[0m",
195
+ success_count, len(failed_ids),
196
+ )
197
+ else:
198
+ logger.info(
199
+ "\033[92mDeletion complete. "
200
+ "Successful: %s, Failed: 0\033[0m",
201
+ success_count,
202
+ )
203
+
204
+
205
+ def delete_project_metadata(
206
+ gen3_submission: Gen3Submission,
207
+ program_id: str,
208
+ project_id: str,
209
+ nodes: List[str],
210
+ prompt_for_confirmation: bool = True,
211
+ ):
212
+ """
213
+ Deletes all metadata for a project by iterating through nodes
214
+ and calling Gen3's delete_nodes API.
215
+
216
+ Nodes should be provided in deletion order (reverse of import
217
+ order), with any excluded nodes already filtered out.
218
+
219
+ Args:
220
+ gen3_submission (Gen3Submission): An authenticated
221
+ Gen3Submission instance.
222
+ program_id (str): The Gen3 program name.
223
+ project_id (str): The Gen3 project name.
224
+ nodes (list[str]): Ordered list of node names to delete.
225
+ prompt_for_confirmation (bool): Whether to prompt for
226
+ confirmation before deletion.
227
+
228
+ Returns:
229
+ None
230
+ """
231
+ if not nodes:
232
+ logger.info("No nodes provided for deletion. Skipping.")
233
+ return
234
+
235
+ if prompt_for_confirmation:
236
+ confirm = input(
237
+ "Do you want to delete the metadata? (yes/no): "
238
+ ).strip().lower()
239
+ if confirm != "yes":
240
+ logger.info("Deletion cancelled by user.")
241
+ return
242
+
243
+ total_nodes = len(nodes)
244
+ for idx, node in enumerate(nodes, start=1):
245
+ logger.info(
246
+ "\033[94m[Node %d/%d]\033[0m | "
247
+ "Project: %-10s | Node: %-25s | Deleting...",
248
+ idx, total_nodes, project_id, node,
249
+ )
250
+ try:
251
+ gen3_submission.delete_nodes(
252
+ program_id, project_id, [node]
253
+ )
254
+ logger.info(
255
+ "\033[92m[SUCCESS]\033[0m | "
256
+ "Project: %-10s | Node: %-25s",
257
+ project_id, node,
258
+ )
259
+ except Exception as e:
260
+ logger.error(
261
+ "\033[91m[FAILED]\033[0m | "
262
+ "Project: %-10s | Node: %-25s | "
263
+ "Error: %s",
264
+ project_id, node, e,
265
+ )