gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
g3dt/resolver.py ADDED
@@ -0,0 +1,163 @@
1
+ """Resolve a project's deployed resource names from AWS SSM Parameter Store.
2
+
3
+ The CDK app (gen3-aws-data-pipeline) writes one SSM parameter per resource it
4
+ creates, under the tree ``/{project}/{env}/...``, and mirrors the human-authored
5
+ Gen3 app facts under ``/{project}/{env}/app/*``. This module reads that tree
6
+ once per process and exposes it as a typed :class:`ResolvedConfig`, so nothing
7
+ else in the toolkit ever hard-codes an AWS resource name.
8
+
9
+ The tree's exact shape is enforced on the infrastructure side by the CDK repo's
10
+ drift-guard test (``test/ssm-publishing.test.ts``): 38 parameters from the SSM
11
+ stack plus ``ec2/instanceId`` published by the EC2 stack (39 total).
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import functools
16
+ from dataclasses import dataclass, field
17
+ from typing import Mapping, Optional
18
+
19
+ import boto3
20
+
21
+ from g3dt.config import ConfigError
22
+
23
+
24
+ def _fetch_params(project: str, env: str, *, session: boto3.Session) -> dict:
25
+ """Return ``{relative_name: value}`` for every parameter under /{project}/{env}.
26
+
27
+ ``relative_name`` is the path with the ``/{project}/{env}/`` prefix removed,
28
+ e.g. ``/etl/test/buckets/metadata`` -> ``buckets/metadata``.
29
+ """
30
+ client = session.client("ssm")
31
+ base = f"/{project}/{env}"
32
+ # get_parameters_by_path returns at most 10 params per call, so paginate.
33
+ paginator = client.get_paginator("get_parameters_by_path")
34
+ out: dict = {}
35
+ for page in paginator.paginate(Path=base, Recursive=True, WithDecryption=True):
36
+ for p in page["Parameters"]:
37
+ out[p["Name"][len(base) + 1 :]] = p["Value"]
38
+ if not out:
39
+ raise ConfigError(
40
+ f"No SSM parameters found under {base} — has CDK been deployed for "
41
+ f"this env? Run `cdk deploy --all -c project={project} -c env={env}` "
42
+ f"in gen3-aws-data-pipeline, then verify with:\n"
43
+ f" aws ssm get-parameters-by-path --path {base} --recursive"
44
+ )
45
+ return out
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class ResolvedConfig:
50
+ """All resource names CDK published for one project/env.
51
+
52
+ ``params`` is the raw ``{relative_name: value}`` map (used by ``config show``
53
+ to print the whole subtree). The properties are typed accessors for the
54
+ names the toolkit reads most often; each raises a friendly
55
+ :class:`ConfigError` if its parameter is missing, which surfaces an
56
+ incomplete CDK deploy immediately instead of a cryptic AWS error later.
57
+ """
58
+
59
+ project: str
60
+ env: str
61
+ params: Mapping[str, str] = field(repr=False)
62
+
63
+ def _req(self, key: str) -> str:
64
+ try:
65
+ return self.params[key]
66
+ except KeyError:
67
+ raise ConfigError(
68
+ f"SSM parameter /{self.project}/{self.env}/{key} is missing. "
69
+ f"Found: {', '.join(sorted(self.params)) or '(none)'}"
70
+ )
71
+
72
+ # --- OUTPUT names (CDK-created) ---
73
+ @property
74
+ def metadata_bucket(self) -> str:
75
+ return self._req("buckets/metadata")
76
+
77
+ @property
78
+ def metadata_db(self) -> str:
79
+ return self._req("glue/db/metadata")
80
+
81
+ @property
82
+ def release_db(self) -> str:
83
+ return self._req("release/db")
84
+
85
+ @property
86
+ def release_table(self) -> str:
87
+ return self._req("release/table")
88
+
89
+ @property
90
+ def athena_workgroup(self) -> str:
91
+ return self._req("athena/workgroup")
92
+
93
+ @property
94
+ def athena_output_location(self) -> str:
95
+ return self._req("athena/outputLocation")
96
+
97
+ @property
98
+ def ec2_instance_id(self) -> str:
99
+ return self._req("ec2/instanceId")
100
+
101
+ @property
102
+ def ec2_log_group(self) -> str:
103
+ return self._req("ec2/logGroup")
104
+
105
+ @property
106
+ def ec2_log_bucket(self) -> str:
107
+ return self._req("ec2/logBucket")
108
+
109
+ @property
110
+ def ec2_log_prefix(self) -> str:
111
+ return self._req("ec2/logPrefix")
112
+
113
+ @property
114
+ def region(self) -> str:
115
+ return self._req("meta/region")
116
+
117
+ @property
118
+ def toolkit_version(self) -> str:
119
+ return self._req("meta/toolkitVersion")
120
+
121
+ def app(self, key: str) -> str:
122
+ """A mirrored Gen3 app fact, e.g. ``app("domain")`` (snake_case keys)."""
123
+ return self._req(f"app/{key}")
124
+
125
+ def get(self, key: str, default: Optional[str] = None) -> Optional[str]:
126
+ """A raw leaf by relative name, or ``default`` if absent."""
127
+ return self.params.get(key, default)
128
+
129
+
130
+ @functools.lru_cache(maxsize=None)
131
+ def resolve(project: str, env: str, profile: Optional[str] = None) -> ResolvedConfig:
132
+ """Resolve all deployed names for ``project``/``env`` from SSM (cached).
133
+
134
+ Cached on ``(project, env, profile)`` for the life of the process, so a
135
+ single CLI invocation makes exactly one ``get_parameters_by_path``
136
+ round-trip no matter how many commands ask for a name. Call
137
+ ``resolve.cache_clear()`` in tests.
138
+
139
+ ``profile`` is the local AWS named profile to authenticate the read (e.g.
140
+ ``etl_test``); ``None`` uses the default credential chain — which is the
141
+ instance profile on the EC2 job box and the build role in CodeBuild.
142
+ """
143
+ session = boto3.Session(profile_name=profile) if profile else boto3.Session()
144
+ return ResolvedConfig(
145
+ project=project, env=env, params=_fetch_params(project, env, session=session)
146
+ )
147
+
148
+
149
+ def list_envs(project: str, profile: Optional[str] = None) -> list:
150
+ """Return the environment names that have an SSM tree under /{project}/.
151
+
152
+ Reads the parameter names one level down (e.g. ``/etl/test/...`` -> ``test``).
153
+ """
154
+ session = boto3.Session(profile_name=profile) if profile else boto3.Session()
155
+ client = session.client("ssm")
156
+ paginator = client.get_paginator("get_parameters_by_path")
157
+ envs = set()
158
+ for page in paginator.paginate(Path=f"/{project}", Recursive=True):
159
+ for p in page["Parameters"]:
160
+ parts = p["Name"].split("/") # ['', project, env, ...]
161
+ if len(parts) > 2:
162
+ envs.add(parts[2])
163
+ return sorted(envs)
@@ -0,0 +1,170 @@
1
+ import sys
2
+ import logging
3
+ import argparse
4
+ import yaml
5
+ from g3dt.upload.metadata_submitter import (
6
+ create_boto3_session,
7
+ get_gen3_api_key_aws_secret,
8
+ create_gen3_submission_class,
9
+ )
10
+ from g3dt.upload.metadata_deleter import (
11
+ delete_project_metadata,
12
+ )
13
+
14
+ # ANSI colour codes
15
+ GREEN = "\033[92m"
16
+ RED = "\033[91m"
17
+ YELLOW = "\033[93m"
18
+ BLUE = "\033[94m"
19
+ RESET = "\033[0m"
20
+
21
+ EXCLUDE_NODES = [
22
+ "program",
23
+ "project",
24
+ "acknowledgement",
25
+ "publication",
26
+ ]
27
+
28
+
29
+ def setup_logger():
30
+ logger = logging.getLogger()
31
+ logger.setLevel(logging.INFO)
32
+ if not logger.handlers:
33
+ handler = logging.StreamHandler(sys.stdout)
34
+ formatter = logging.Formatter(
35
+ '%(asctime)s - %(name)s - %(levelname)s - %(message)s'
36
+ )
37
+ handler.setFormatter(formatter)
38
+ logger.addHandler(handler)
39
+ return logger
40
+
41
+
42
+ # Shared config resolution (SSM-backed) — see src/g3dt/config.py
43
+ from g3dt import config as g3dt_config # noqa: E402
44
+
45
+
46
+ def load_import_order(import_order_path, exclude_nodes=None):
47
+ """
48
+ Reads the DataImportOrder.txt file and returns the node list
49
+ in deletion order (reversed, with excluded nodes removed).
50
+ """
51
+ if exclude_nodes is None:
52
+ exclude_nodes = EXCLUDE_NODES
53
+ with open(import_order_path, 'r', encoding='utf-8') as f:
54
+ nodes = [line.strip() for line in f if line.strip()]
55
+ nodes = [n for n in nodes if n not in exclude_nodes]
56
+ nodes.reverse()
57
+ return nodes
58
+
59
+
60
+ def main():
61
+ logger = setup_logger()
62
+
63
+ parser = argparse.ArgumentParser(
64
+ description=(
65
+ "Delete ALL metadata for a Gen3 project. Iterates "
66
+ "through nodes in reverse DataImportOrder and calls "
67
+ "Gen3's delete_nodes API for each."
68
+ ),
69
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter,
70
+ )
71
+ parser.add_argument(
72
+ "--study",
73
+ required=True,
74
+ help=(
75
+ "Study key (bare or env-suffixed) "
76
+ "(e.g. ausdiab, caughtcad, edcad, cdah)"
77
+ ),
78
+ )
79
+ parser.add_argument(
80
+ "--env",
81
+ required=True,
82
+ help="Environment to use (selects AWS secret, profile, etc.)",
83
+ )
84
+ parser.add_argument(
85
+ "--import-order",
86
+ default="DataImportOrder.txt",
87
+ help="Path to DataImportOrder.txt",
88
+ )
89
+ parser.add_argument(
90
+ "--node",
91
+ default=None,
92
+ help=(
93
+ "Delete only a specific node (e.g. 'subject'). "
94
+ "If omitted, all nodes are processed in reverse "
95
+ "DataImportOrder."
96
+ ),
97
+ )
98
+ parser.add_argument(
99
+ "--prompt",
100
+ action="store_true",
101
+ default=False,
102
+ help="Prompt for confirmation before deleting.",
103
+ )
104
+
105
+ args = parser.parse_args()
106
+
107
+ # Env facts from SSM; the study registry from the marker or
108
+ # s3://<metadata-bucket>/config/studies.yaml.
109
+ try:
110
+ env_cfg = g3dt_config.resolve_env(args.env)
111
+ study_cfg = g3dt_config.resolve_study(args.study, args.env)
112
+ except g3dt_config.ConfigError as exc:
113
+ logger.error(str(exc))
114
+ sys.exit(1)
115
+
116
+ project_id = study_cfg.project_id
117
+ program_id = study_cfg.program_id
118
+
119
+ aws_secret_name = env_cfg.aws_secret_name
120
+ aws_profile = env_cfg.aws_profile
121
+ aws_region = env_cfg.region
122
+
123
+ logger.info(
124
+ "Study: %s | Env: %s | Program: %s | Project: %s",
125
+ args.study, args.env, program_id, project_id,
126
+ )
127
+
128
+ # AWS and Gen3 authentication
129
+ session = create_boto3_session(aws_profile=aws_profile)
130
+ api_key = get_gen3_api_key_aws_secret(
131
+ secret_name=aws_secret_name,
132
+ region_name=aws_region,
133
+ session=session,
134
+ )
135
+ sub = create_gen3_submission_class(api_key)
136
+
137
+ # Determine node list
138
+ if args.node:
139
+ nodes_to_delete = [args.node]
140
+ logger.info(
141
+ "%s[SINGLE NODE]%s Targeting node: %s",
142
+ BLUE, RESET, args.node,
143
+ )
144
+ else:
145
+ nodes_to_delete = load_import_order(args.import_order)
146
+ logger.info(
147
+ "Loaded %s nodes from %s (deletion order, "
148
+ "excluding %s)",
149
+ len(nodes_to_delete),
150
+ args.import_order,
151
+ EXCLUDE_NODES,
152
+ )
153
+
154
+ # Delete
155
+ delete_project_metadata(
156
+ gen3_submission=sub,
157
+ program_id=program_id,
158
+ project_id=project_id,
159
+ nodes=nodes_to_delete,
160
+ prompt_for_confirmation=args.prompt,
161
+ )
162
+
163
+ logger.info(
164
+ "=========================================="
165
+ )
166
+ logger.info("Deletion complete.")
167
+
168
+
169
+ if __name__ == "__main__":
170
+ main()
@@ -0,0 +1,153 @@
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+
4
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
5
+
6
+ # Exit code the per-study worker uses to signal "no data at this version —
7
+ # skipped" (see delete_metadata_by_guid.py SKIP_EXIT_CODE, shipped alongside
8
+ # this script).
9
+ SKIP_EXIT_CODE=3
10
+
11
+ usage() {
12
+ cat <<EOF
13
+ Usage: $(basename "$0") --studies <comma-separated-studies> --env <environment> --version <version|all> [--node <node>]
14
+
15
+ Delete metadata for each study sequentially, in a single job.
16
+
17
+ Arguments:
18
+ --studies Comma-separated list of study config keys (e.g. ausdiab_staging,caughtcad_staging)
19
+ --env Environment string passed to the Python worker (e.g. staging_ec2)
20
+ --version Metadata version to delete (e.g. 0.9.8), or 'all' for every version
21
+ --node (optional) Restrict deletion to a single node type
22
+
23
+ Behaviour:
24
+ * --version all -> delete_all_metadata_for_project.py (deletes whole nodes)
25
+ * --version <x.y.z> -> delete_metadata_by_guid.py (Athena GUID lookup for that version)
26
+
27
+ A study that exists but has no data at the requested version is skipped and the
28
+ loop continues. Only genuine errors (Gen3/AWS failures) count as failures.
29
+
30
+ Run via the g3dt CLI:
31
+ g3dt delete metadata \\
32
+ --studies ausdiab_staging,caughtcad_staging \\
33
+ --env staging_ec2 \\
34
+ --version 0.9.8
35
+
36
+ Failure logs are written under ~/.g3dt/logs/.
37
+ EOF
38
+ exit 1
39
+ }
40
+
41
+ # ---------- Parse arguments ----------
42
+ STUDIES=""
43
+ ENV=""
44
+ VERSION=""
45
+ NODE=""
46
+
47
+ while [[ $# -gt 0 ]]; do
48
+ case "$1" in
49
+ --studies)
50
+ STUDIES="$2"
51
+ shift 2
52
+ ;;
53
+ --env)
54
+ ENV="$2"
55
+ shift 2
56
+ ;;
57
+ --version)
58
+ VERSION="$2"
59
+ shift 2
60
+ ;;
61
+ --node)
62
+ NODE="$2"
63
+ shift 2
64
+ ;;
65
+ *)
66
+ echo "ERROR: Unknown argument: $1"
67
+ usage
68
+ ;;
69
+ esac
70
+ done
71
+
72
+ if [[ -z "$STUDIES" || -z "$ENV" || -z "$VERSION" ]]; then
73
+ echo "ERROR: --studies, --env and --version are required."
74
+ usage
75
+ fi
76
+
77
+ # Lower-case the version so 'ALL'/'All' are treated as 'all'.
78
+ VERSION_LC="$(echo "$VERSION" | tr '[:upper:]' '[:lower:]')"
79
+
80
+ # ---------- Setup ----------
81
+ # Logs go outside the installed package.
82
+ LOG_DIR="$HOME/.g3dt/logs"
83
+ TIMESTAMP="$(date +%Y%m%d_%H%M%S)"
84
+ FAILED_LOG="${LOG_DIR}/${TIMESTAMP}_delete_failed.log"
85
+ mkdir -p "${LOG_DIR}"
86
+
87
+ IFS=',' read -ra STUDY_LIST <<< "$STUDIES"
88
+ DELETED_COUNT=0
89
+ SKIPPED_COUNT=0
90
+ FAIL_COUNT=0
91
+
92
+ echo "============================================"
93
+ echo "Metadata delete started at $(date)"
94
+ echo "Environment : ${ENV}"
95
+ echo "Studies : ${STUDIES}"
96
+ echo "Version : ${VERSION}"
97
+ [[ -n "$NODE" ]] && echo "Node : ${NODE}"
98
+ echo "Failure log : ${FAILED_LOG}"
99
+ echo "============================================"
100
+ echo ""
101
+
102
+ # ---------- Sequential execution ----------
103
+ for study in "${STUDY_LIST[@]}"; do
104
+ echo "--------------------------------------------"
105
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] Starting deletion for study: ${study} (version: ${VERSION})"
106
+ echo "--------------------------------------------"
107
+
108
+ if [[ "$VERSION_LC" == "all" ]]; then
109
+ CMD=(python3 "${SCRIPT_DIR}/delete_all_metadata_for_project.py"
110
+ --study "$study" --env "$ENV")
111
+ [[ -n "$NODE" ]] && CMD+=(--node "$NODE")
112
+ else
113
+ CMD=(python3 "${SCRIPT_DIR}/delete_metadata_by_guid.py"
114
+ --study "$study" --env "$ENV" --version "$VERSION" --skip-if-empty)
115
+ [[ -n "$NODE" ]] && CMD+=(--node "$NODE")
116
+ fi
117
+
118
+ # Run the worker without aborting the loop on a non-zero exit.
119
+ set +e
120
+ "${CMD[@]}"
121
+ EXIT_CODE=$?
122
+ set -e
123
+
124
+ if [[ $EXIT_CODE -eq 0 ]]; then
125
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] Completed: ${study}"
126
+ DELETED_COUNT=$((DELETED_COUNT + 1))
127
+ elif [[ $EXIT_CODE -eq $SKIP_EXIT_CODE ]]; then
128
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] Skipped (no data at version ${VERSION}): ${study}"
129
+ SKIPPED_COUNT=$((SKIPPED_COUNT + 1))
130
+ else
131
+ FAIL_COUNT=$((FAIL_COUNT + 1))
132
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] FAILED: ${study} (exit code ${EXIT_CODE})"
133
+ echo "[$(date +%Y-%m-%d\ %H:%M:%S)] ${study} exit_code=${EXIT_CODE}" >> "$FAILED_LOG"
134
+ fi
135
+
136
+ echo ""
137
+ done
138
+
139
+ # ---------- Summary ----------
140
+ echo "============================================"
141
+ echo "Metadata delete finished at $(date)"
142
+ echo "Total studies : ${#STUDY_LIST[@]}"
143
+ echo "Deleted : ${DELETED_COUNT}"
144
+ echo "Skipped : ${SKIPPED_COUNT}"
145
+ echo "Failures : ${FAIL_COUNT}"
146
+
147
+ if [[ $FAIL_COUNT -gt 0 ]]; then
148
+ echo "See failure details: ${FAILED_LOG}"
149
+ exit 1
150
+ fi
151
+
152
+ echo "All studies processed (deleted ${DELETED_COUNT}, skipped ${SKIPPED_COUNT})."
153
+ exit 0