gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,66 @@
1
+ import logging
2
+ import json
3
+ from pathlib import Path
4
+ from typing import Optional, Dict, Any, List
5
+ import yaml
6
+
7
+ # Set up logger for this module
8
+ logger = logging.getLogger(__name__)
9
+
10
+ # ----------------- dbt Artifact Helpers -----------------
11
+
12
+
13
+ def load_json(path: Path) -> Optional[Dict[str, Any]]:
14
+ """
15
+ Load a JSON file from the given path and return its contents as a dictionary.
16
+
17
+ Args:
18
+ path (Path): The path to the JSON file.
19
+
20
+ Returns:
21
+ Optional[Dict[str, Any]]: The loaded JSON data as a dictionary, or None if the file does not exist
22
+ or loading fails.
23
+
24
+ Logs:
25
+ - Warning if the file does not exist.
26
+ - Error if loading the JSON fails.
27
+ """
28
+ try:
29
+ if not path.exists():
30
+ logger.warning(f"JSON file does not exist: {path}")
31
+ return None
32
+ with path.open() as f:
33
+ return json.load(f)
34
+ except Exception as e:
35
+ logger.error(f"Failed to load JSON from {path}: {e}")
36
+ return None
37
+
38
+
39
+ def get_model_names(dbt_schema_path) -> list:
40
+ """
41
+ Extracts and returns a list of model names from a dbt schema YAML file.
42
+
43
+ Args:
44
+ dbt_schema_path (str or Path): Path to the dbt schema.yml file.
45
+
46
+ Returns:
47
+ list: List of model names found in the schema file. Returns an empty list if none found or on error.
48
+
49
+ Raises:
50
+ None. All exceptions are caught and logged.
51
+
52
+ Logs:
53
+ - Error if the file cannot be read or parsed, or if the expected structure is missing.
54
+ """
55
+ try:
56
+ with open(dbt_schema_path, mode='r') as f:
57
+ schema = yaml.safe_load(f)
58
+ if not schema or 'models' not in schema:
59
+ logger.error(f"'models' key not found in schema file: {dbt_schema_path}")
60
+ return []
61
+ model_names = [model.get('name') for model in schema['models'] if 'name' in model]
62
+ return model_names
63
+ except Exception as e:
64
+ logger.error(f"Failed to load model names from schema file {dbt_schema_path}: {e}")
65
+ return []
66
+
@@ -0,0 +1,188 @@
1
+ import argparse
2
+ from g3dt.utils.athena_utils import AthenaConfig, AthenaQuery, AthenaValidationWriter
3
+ from g3dt.utils.dbt_utils import get_model_names
4
+ import logging
5
+ from typing import Optional
6
+
7
+ # Setup logger for this script
8
+ logger = logging.getLogger("DBTRelease")
9
+ logging.basicConfig(
10
+ level=logging.INFO,
11
+ format="[%(asctime)s] %(levelname)s %(name)s: %(message)s"
12
+ )
13
+
14
+ def safe_sql_string(value: Optional[str]) -> str:
15
+ """
16
+ Safely escapes and prepares a string value for inclusion in SQL statements.
17
+
18
+ - If the input is None or an empty string, it returns the string 'NULL'.
19
+ - Otherwise, it escapes any single quotes by doubling them and encloses
20
+ the entire string in single quotes.
21
+
22
+ Args:
23
+ value (Optional[str]): The string value to escape.
24
+
25
+ Returns:
26
+ str: The escaped and quoted string ready for SQL insertion.
27
+ """
28
+ if value is None or value == "":
29
+ return "NULL"
30
+ return "'" + str(value).replace("'", "''") + "'"
31
+
32
+ def insert_release_row(
33
+ athena_config: AthenaConfig,
34
+ model_name: str,
35
+ db_name: Optional[str],
36
+ snapshot_id: Optional[int],
37
+ committed_at: Optional[str],
38
+ release_db: str,
39
+ release_table: str,
40
+ release_tag: str,
41
+ github_sha: str
42
+ ) -> None:
43
+ """
44
+ Insert a new row into the release tracking table for a given model and snapshot.
45
+
46
+ This function checks if a row with the same release_tag, model_name, and db_name already exists
47
+ in the release table to prevent duplicate entries. If no such row exists, it inserts a new row
48
+ with the provided snapshot_id, committed_at timestamp, and other metadata.
49
+ """
50
+ athena_query = AthenaQuery(athena_config)
51
+ if not db_name:
52
+ logger.warning(f"Skipping model '{model_name}': No database found for this model.")
53
+ return
54
+
55
+ logger.debug(
56
+ f"Checking for existing release row: release_tag={release_tag!r}, model_name={model_name!r}, db_name={db_name!r}"
57
+ )
58
+ check_sql = f"""
59
+ SELECT COUNT(*) AS cnt
60
+ FROM "{release_db}"."{release_table}"
61
+ WHERE release_tag = {safe_sql_string(release_tag)}
62
+ AND model_name = {safe_sql_string(model_name)}
63
+ AND db_name = {safe_sql_string(db_name)}
64
+ """
65
+ try:
66
+ cnt_df = athena_query.query_athena(check_sql, release_db)
67
+ cnt = cnt_df.iloc[0]['cnt'] if not cnt_df.empty else 0
68
+ except Exception as e:
69
+ logger.error(
70
+ f"Could not query existing release rows for release_tag='{release_tag}', model='{model_name}', db='{db_name}': {e}",
71
+ exc_info=True
72
+ )
73
+ raise
74
+
75
+ if cnt > 0:
76
+ logger.info(f"[SKIP] Release row already exists: {release_tag} / {db_name}.{model_name}")
77
+ return
78
+
79
+ snap_val = "NULL" if snapshot_id is None else str(snapshot_id)
80
+ commit_val = f"TIMESTAMP '{committed_at}'" if committed_at else "NULL"
81
+ sha_val = safe_sql_string(github_sha)
82
+
83
+ insert_sql = f"""
84
+ INSERT INTO "{release_db}"."{release_table}"
85
+ (release_tag, db_name, model_name, snapshot_id, committed_at, inserted_at, github_sha)
86
+ VALUES (
87
+ {safe_sql_string(release_tag)},
88
+ {safe_sql_string(db_name)},
89
+ {safe_sql_string(model_name)},
90
+ {snap_val},
91
+ {commit_val},
92
+ CURRENT_TIMESTAMP,
93
+ {sha_val}
94
+ )
95
+ """
96
+ logger.info(
97
+ f"Inserting new release row for {db_name}.{model_name} [release_tag={release_tag}, snapshot_id={snapshot_id}, committed_at={committed_at}]"
98
+ )
99
+ try:
100
+ athena_query.query_athena(insert_sql, release_db, ctas_approach=False)
101
+ logger.info(f"[OK] Inserted release row for {db_name}.{model_name}.{release_tag}")
102
+ except Exception as e:
103
+ logger.error(
104
+ f"Failed to insert release row for {db_name}.{model_name}.{release_tag}: {e}",
105
+ exc_info=True
106
+ )
107
+ raise
108
+
109
+ def parse_args():
110
+ parser = argparse.ArgumentParser(
111
+ description="Write dbt model snapshot info for all models as a release row to Athena. "
112
+ "Ensures all tracked dbt models have a row in the release iceberg table with latest snapshot/commit.",
113
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter
114
+ )
115
+ parser.add_argument("--dbt-schema-path", type=str, required=True,
116
+ help="Path to dbt schema file (usually schema.yml) containing list of models to track.")
117
+ parser.add_argument("--release-db", type=str, required=True,
118
+ help="Athena database for the release tracking table.")
119
+ parser.add_argument("--release-table", type=str, required=True,
120
+ help="Athena table for the release tracking table.")
121
+ parser.add_argument("--data-release-version", type=str, required=True,
122
+ help="The release version tag to record. (e.g., v1.2.3)")
123
+ parser.add_argument("--commit-id", type=str, required=True,
124
+ help="The git commit SHA for this release (for auditing).")
125
+ parser.add_argument("--aws-region", type=str, required=True,
126
+ help="AWS Region for Athena/S3.")
127
+ parser.add_argument("--aws-profile", type=str, required=False,
128
+ help="AWS CLI profile to use for authentication.")
129
+ parser.add_argument("--athena-s3-output", type=str, required=True,
130
+ help="S3 URI for Athena query results (e.g., 's3://athena-results-bucket/').")
131
+ parser.add_argument("--release-s3-location", type=str, required=True,
132
+ help="S3 location for the release table (e.g., 's3://<metadata-bucket>/').")
133
+ parser.add_argument("-v", "--verbose", action="store_true", default=False,
134
+ help="Enable debug logging.")
135
+ return parser.parse_args()
136
+
137
+ def main():
138
+ args = parse_args()
139
+ log_level = logging.DEBUG if args.verbose else logging.INFO
140
+ logging.getLogger().setLevel(log_level)
141
+
142
+ logger.info("==== DBT release snapshot info writer started ====")
143
+ logger.debug(f"Parsed arguments: {args}")
144
+
145
+ athena_config = AthenaConfig(
146
+ aws_region=args.aws_region,
147
+ aws_profile=args.aws_profile,
148
+ athena_s3_output=args.athena_s3_output
149
+ )
150
+
151
+ dbt_models = get_model_names(args.dbt_schema_path)
152
+ athena_query = AthenaQuery(athena_config)
153
+
154
+ logger.info(f"Creating release table: {args.release_db}.{args.release_table}")
155
+ athena_query.create_release_table(
156
+ args.release_db, args.release_table, args.release_s3_location
157
+ )
158
+
159
+ logger.info(f"Processing DBT models from schema: {dbt_models}")
160
+
161
+ for model_name in dbt_models:
162
+ logger.info(f"--- Processing model: {model_name}")
163
+ db_name = athena_query.find_db_for_model(model_name)
164
+ if not db_name:
165
+ logger.warning(f"Database not found for model '{model_name}'. Skipping...")
166
+ continue
167
+
168
+ snapshot_writer = AthenaValidationWriter(athena_config, db_name, model_name)
169
+ snapshot_id, commit_datetime = snapshot_writer._get_latest_snapshot_id(return_commit_datetime=True)
170
+ logger.debug(f"Latest snapshot for {db_name}.{model_name}: {snapshot_id} @ {commit_datetime}")
171
+
172
+ insert_release_row(
173
+ athena_config=athena_config,
174
+ model_name=model_name,
175
+ db_name=db_name,
176
+ snapshot_id=snapshot_id,
177
+ committed_at=commit_datetime,
178
+ release_db=args.release_db,
179
+ release_table=args.release_table,
180
+ release_tag=args.data_release_version,
181
+ github_sha=args.commit_id
182
+ )
183
+ logger.info(f"[SUCCESS] Release info recorded for {db_name}.{model_name}")
184
+
185
+ logger.info(f"Finished release process for DBT models: {dbt_models}")
186
+
187
+ if __name__ == "__main__":
188
+ main()