gen3-dataops-toolkit 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- g3dt/__init__.py +0 -0
- g3dt/cli/__init__.py +5 -0
- g3dt/cli/_internal/__init__.py +1 -0
- g3dt/cli/_internal/dispatch.py +428 -0
- g3dt/cli/_internal/registry.py +65 -0
- g3dt/cli/_internal/resolve.py +22 -0
- g3dt/cli/_internal/runner.py +76 -0
- g3dt/cli/_internal/safety.py +110 -0
- g3dt/cli/config_cmds.py +202 -0
- g3dt/cli/delete_cmds.py +101 -0
- g3dt/cli/dict_cmds.py +102 -0
- g3dt/cli/ec2_cmds.py +114 -0
- g3dt/cli/indexd_cmds.py +57 -0
- g3dt/cli/jobs.py +83 -0
- g3dt/cli/k8s.py +54 -0
- g3dt/cli/main.py +110 -0
- g3dt/cli/metadata.py +76 -0
- g3dt/cli/synth.py +206 -0
- g3dt/config.py +393 -0
- g3dt/indexd/__init__.py +0 -0
- g3dt/indexd/indexd_registrar.py +244 -0
- g3dt/ingest/ingest.py +629 -0
- g3dt/resolver.py +163 -0
- g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
- g3dt/services/delete/delete_metadata.sh +153 -0
- g3dt/services/delete/delete_metadata_by_guid.py +338 -0
- g3dt/services/dictionary/deploy_dd.sh +65 -0
- g3dt/services/dictionary/pull_dict.sh +59 -0
- g3dt/services/dictionary/upload_dictionary.py +109 -0
- g3dt/services/indexd/register_indexd.py +240 -0
- g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
- g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
- g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
- g3dt/services/k8s_ops/login_to_pod.sh +110 -0
- g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
- g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
- g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
- g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
- g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
- g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
- g3dt/services/upload/metadata/upload_metadata.py +152 -0
- g3dt/upload/__init__.py +1 -0
- g3dt/upload/metadata_deleter.py +265 -0
- g3dt/upload/metadata_submitter.py +1093 -0
- g3dt/upload/upload_synthdata_s3.py +164 -0
- g3dt/utils/athena_utils.py +834 -0
- g3dt/utils/dbt_utils.py +66 -0
- g3dt/utils/release_writer.py +188 -0
- g3dt/validate/validate.py +609 -0
- gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
- gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
- gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
- gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
|
|
3
|
+
# Exit on any error
|
|
4
|
+
set -e
|
|
5
|
+
# Ensure pipeline failures are captured
|
|
6
|
+
set -o pipefail
|
|
7
|
+
|
|
8
|
+
# Usage: bash restart_etl_and_ms.sh <profile>
|
|
9
|
+
# <profile> is display-only; all configuration comes from G3DT_* environment
|
|
10
|
+
# variables exported by the g3dt CLI (g3dt.config.script_env).
|
|
11
|
+
PROFILE="${1:-}"
|
|
12
|
+
|
|
13
|
+
if [ -z "$PROFILE" ]; then
|
|
14
|
+
echo "Usage: $0 <profile> (test|staging|prod) — run via the g3dt CLI"
|
|
15
|
+
exit 1
|
|
16
|
+
fi
|
|
17
|
+
|
|
18
|
+
# Defining script paths
|
|
19
|
+
SCRIPT_DIR="$( cd -- "$( dirname -- "${BASH_SOURCE[0]}" )" &> /dev/null && pwd )"
|
|
20
|
+
|
|
21
|
+
# Configuration from G3DT_* env vars (fail loudly if a required one is missing)
|
|
22
|
+
CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI}"
|
|
23
|
+
DOMAIN="${G3DT_DOMAIN:?G3DT_DOMAIN not set — run via the g3dt CLI}"
|
|
24
|
+
APP_NAME="${G3DT_APP_NAME:?G3DT_APP_NAME not set — run via the g3dt CLI}"
|
|
25
|
+
NAMESPACE="${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI}"
|
|
26
|
+
REGION="${G3DT_REGION:-ap-southeast-2}"
|
|
27
|
+
# This script lives alongside the argo scripts inside the package.
|
|
28
|
+
ARGO_SCRIPT_DIR="${SCRIPT_DIR}"
|
|
29
|
+
|
|
30
|
+
# Never export an empty AWS_PROFILE (empty means ambient credentials).
|
|
31
|
+
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
32
|
+
export AWS_PROFILE="${G3DT_AWS_PROFILE}"
|
|
33
|
+
echo "==== Configuring AWS PROFILE as '${AWS_PROFILE}' for profile '${PROFILE}' ===="
|
|
34
|
+
else
|
|
35
|
+
echo "==== No AWS profile set for '${PROFILE}' — using ambient AWS credentials ===="
|
|
36
|
+
fi
|
|
37
|
+
|
|
38
|
+
echo "Updating kubeconfig for cluster: ${CLUSTER_NAME}"
|
|
39
|
+
if ! aws eks update-kubeconfig --name "${CLUSTER_NAME}" --region "${REGION}"; then
|
|
40
|
+
echo "ERROR: Failed to update kubeconfig. Check your AWS credentials/profile."
|
|
41
|
+
exit 1
|
|
42
|
+
fi
|
|
43
|
+
|
|
44
|
+
echo "==== Restarting microservices (etl) ===="
|
|
45
|
+
bash "${ARGO_SCRIPT_DIR}/argocd_restart_etl.sh" \
|
|
46
|
+
-d "${DOMAIN}" \
|
|
47
|
+
-a "${APP_NAME}" \
|
|
48
|
+
-n "${NAMESPACE}" \
|
|
49
|
+
-s
|
|
50
|
+
|
|
51
|
+
echo "==== Restarting microservices (schema) ===="
|
|
52
|
+
bash "${ARGO_SCRIPT_DIR}/argocd_restart_ms.sh" \
|
|
53
|
+
-d "${DOMAIN}" \
|
|
54
|
+
-a "${APP_NAME}" \
|
|
55
|
+
-n "${NAMESPACE}" \
|
|
56
|
+
-r "sheepdog-deployment,guppy-deployment,peregrine-deployment,portal-deployment"
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import argparse
|
|
3
|
+
import logging
|
|
4
|
+
from g3dt.upload.metadata_deleter import (
|
|
5
|
+
delete_project_metadata,
|
|
6
|
+
)
|
|
7
|
+
from g3dt.upload.metadata_submitter import (
|
|
8
|
+
create_boto3_session,
|
|
9
|
+
get_gen3_api_key_aws_secret,
|
|
10
|
+
create_gen3_submission_class,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
EXCLUDE_NODES = [
|
|
14
|
+
"program",
|
|
15
|
+
"project",
|
|
16
|
+
"acknowledgement",
|
|
17
|
+
"publication",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load_import_order(import_order_path, exclude_nodes=None):
|
|
22
|
+
"""
|
|
23
|
+
Reads the DataImportOrder.txt file and returns the node list
|
|
24
|
+
in deletion order (reversed, with excluded nodes removed).
|
|
25
|
+
"""
|
|
26
|
+
if exclude_nodes is None:
|
|
27
|
+
exclude_nodes = EXCLUDE_NODES
|
|
28
|
+
with open(import_order_path, 'r', encoding='utf-8') as f:
|
|
29
|
+
nodes = [line.strip() for line in f if line.strip()]
|
|
30
|
+
nodes = [n for n in nodes if n not in exclude_nodes]
|
|
31
|
+
nodes.reverse()
|
|
32
|
+
return nodes
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def delete_for_projects(
|
|
36
|
+
gen3_submission,
|
|
37
|
+
program_id,
|
|
38
|
+
project_ids,
|
|
39
|
+
nodes,
|
|
40
|
+
prompt_for_confirmation,
|
|
41
|
+
):
|
|
42
|
+
for project_id in project_ids:
|
|
43
|
+
logging.info(
|
|
44
|
+
"Attempting to delete metadata for project '%s'",
|
|
45
|
+
project_id,
|
|
46
|
+
)
|
|
47
|
+
try:
|
|
48
|
+
delete_project_metadata(
|
|
49
|
+
gen3_submission=gen3_submission,
|
|
50
|
+
program_id=program_id,
|
|
51
|
+
project_id=project_id,
|
|
52
|
+
nodes=nodes,
|
|
53
|
+
prompt_for_confirmation=prompt_for_confirmation,
|
|
54
|
+
)
|
|
55
|
+
logging.info(
|
|
56
|
+
"Successfully deleted metadata for project '%s'.",
|
|
57
|
+
project_id,
|
|
58
|
+
)
|
|
59
|
+
except Exception as e:
|
|
60
|
+
logging.error(
|
|
61
|
+
"Error deleting metadata for project '%s': %s",
|
|
62
|
+
project_id, e,
|
|
63
|
+
)
|
|
64
|
+
raise e
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
if __name__ == "__main__":
|
|
68
|
+
logging.basicConfig(
|
|
69
|
+
level=logging.INFO,
|
|
70
|
+
format='[%(asctime)s] [%(levelname)s] %(message)s',
|
|
71
|
+
datefmt='%Y-%m-%d %H:%M:%S',
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
script_path = os.path.dirname(os.path.abspath(__file__))
|
|
75
|
+
os.chdir(script_path)
|
|
76
|
+
logging.info("Changed working directory to %s", script_path)
|
|
77
|
+
|
|
78
|
+
default_projects = [
|
|
79
|
+
"AusDiab_Simulated",
|
|
80
|
+
"EDCAD-PMS_Simulated",
|
|
81
|
+
"Baker-Biobank_Simulated",
|
|
82
|
+
"CAUGHT-CAD_Simulated",
|
|
83
|
+
"BioHeart-CT_Simulated",
|
|
84
|
+
]
|
|
85
|
+
default_program_id = "program1"
|
|
86
|
+
default_aws_profile = "default"
|
|
87
|
+
default_aws_region = "ap-southeast-2"
|
|
88
|
+
default_prompt_for_confirmation = False
|
|
89
|
+
|
|
90
|
+
parser = argparse.ArgumentParser(
|
|
91
|
+
description=(
|
|
92
|
+
"Delete synthetic project metadata from Gen3 "
|
|
93
|
+
"Sheepdog API."
|
|
94
|
+
),
|
|
95
|
+
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
|
96
|
+
)
|
|
97
|
+
parser.add_argument(
|
|
98
|
+
'-i', '--data-import-order-file',
|
|
99
|
+
type=str,
|
|
100
|
+
required=True,
|
|
101
|
+
help="Path to the DataImportOrder.txt file.",
|
|
102
|
+
)
|
|
103
|
+
parser.add_argument(
|
|
104
|
+
'-p', '--projects',
|
|
105
|
+
type=str,
|
|
106
|
+
default=",".join(default_projects),
|
|
107
|
+
help="Comma separated list of Project IDs.",
|
|
108
|
+
)
|
|
109
|
+
parser.add_argument(
|
|
110
|
+
'--program-id',
|
|
111
|
+
type=str,
|
|
112
|
+
default=default_program_id,
|
|
113
|
+
help="Gen3 program name.",
|
|
114
|
+
)
|
|
115
|
+
parser.add_argument(
|
|
116
|
+
'-s', '--aws-secret-name',
|
|
117
|
+
type=str,
|
|
118
|
+
help="AWS secret name for Gen3 API key.",
|
|
119
|
+
)
|
|
120
|
+
parser.add_argument(
|
|
121
|
+
'-profile', '--aws-profile',
|
|
122
|
+
type=str,
|
|
123
|
+
default=default_aws_profile,
|
|
124
|
+
help="AWS profile name.",
|
|
125
|
+
)
|
|
126
|
+
parser.add_argument(
|
|
127
|
+
'-r', '--aws-region',
|
|
128
|
+
type=str,
|
|
129
|
+
default=default_aws_region,
|
|
130
|
+
help="AWS region.",
|
|
131
|
+
)
|
|
132
|
+
parser.add_argument(
|
|
133
|
+
'-confirm', '--prompt',
|
|
134
|
+
action='store_true',
|
|
135
|
+
default=default_prompt_for_confirmation,
|
|
136
|
+
help="Prompt for confirmation before deleting.",
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
args = parser.parse_args()
|
|
140
|
+
|
|
141
|
+
parsed_projects = [
|
|
142
|
+
proj.strip()
|
|
143
|
+
for proj in args.projects.split(",")
|
|
144
|
+
if proj.strip()
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
logging.info("Projects to delete: %s", parsed_projects)
|
|
148
|
+
logging.info(
|
|
149
|
+
"Data import order file: %s",
|
|
150
|
+
args.data_import_order_file,
|
|
151
|
+
)
|
|
152
|
+
logging.info("AWS Secret Name: %s", args.aws_secret_name)
|
|
153
|
+
logging.info("AWS Profile: %s", args.aws_profile)
|
|
154
|
+
logging.info("AWS Region: %s", args.aws_region)
|
|
155
|
+
logging.info(
|
|
156
|
+
"Prompt for confirmation: %s", args.prompt,
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
if not os.path.isfile(args.data_import_order_file):
|
|
160
|
+
raise FileNotFoundError(
|
|
161
|
+
f"Import order file does not exist: "
|
|
162
|
+
f"{args.data_import_order_file}"
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
nodes = load_import_order(args.data_import_order_file)
|
|
166
|
+
|
|
167
|
+
session = create_boto3_session(
|
|
168
|
+
aws_profile=args.aws_profile,
|
|
169
|
+
)
|
|
170
|
+
api_key = get_gen3_api_key_aws_secret(
|
|
171
|
+
secret_name=args.aws_secret_name,
|
|
172
|
+
region_name=args.aws_region,
|
|
173
|
+
session=session,
|
|
174
|
+
)
|
|
175
|
+
sub = create_gen3_submission_class(api_key)
|
|
176
|
+
|
|
177
|
+
delete_for_projects(
|
|
178
|
+
gen3_submission=sub,
|
|
179
|
+
program_id=args.program_id,
|
|
180
|
+
project_ids=parsed_projects,
|
|
181
|
+
nodes=nodes,
|
|
182
|
+
prompt_for_confirmation=args.prompt,
|
|
183
|
+
)
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
|
|
3
|
+
# Exit on any error
|
|
4
|
+
set -e
|
|
5
|
+
# Ensure pipeline failures are captured
|
|
6
|
+
set -o pipefail
|
|
7
|
+
|
|
8
|
+
# Usage: bash full_deploy_dd_and_synth.sh <profile>
|
|
9
|
+
# <profile> is display-only; all configuration comes from G3DT_* environment
|
|
10
|
+
# variables exported by the g3dt CLI (g3dt.config.script_env).
|
|
11
|
+
PROFILE=$1
|
|
12
|
+
|
|
13
|
+
if [ -z "$PROFILE" ]; then
|
|
14
|
+
echo "Usage: $0 <profile> (test|staging|prod) — run via the g3dt CLI"
|
|
15
|
+
exit 1
|
|
16
|
+
fi
|
|
17
|
+
|
|
18
|
+
# Any configured profile is allowed here. The production warning/confirmation is
|
|
19
|
+
# enforced by the CLI entrypoint (`g3dt synth deploy`) before this script runs,
|
|
20
|
+
# since that has a TTY for the prompt.
|
|
21
|
+
|
|
22
|
+
# Defining script paths (sibling scripts ship together inside the package)
|
|
23
|
+
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)"
|
|
24
|
+
SERVICE_DIR="${SCRIPT_DIR}/.."
|
|
25
|
+
|
|
26
|
+
# Configuration from G3DT_* env vars (fail loudly if a required one is missing)
|
|
27
|
+
VERSION="${G3DT_DICTIONARY_VERSION:?G3DT_DICTIONARY_VERSION not set — run via the g3dt CLI}"
|
|
28
|
+
AWS_SECRET_NAME="${G3DT_AWS_SECRET_NAME:?G3DT_AWS_SECRET_NAME not set — run via the g3dt CLI}"
|
|
29
|
+
SCHEMA_S3_URI="${G3DT_SCHEMA_S3_URI:?G3DT_SCHEMA_S3_URI not set — run via the g3dt CLI}"
|
|
30
|
+
DOMAIN="${G3DT_DOMAIN:?G3DT_DOMAIN not set — run via the g3dt CLI}"
|
|
31
|
+
APP_NAME="${G3DT_APP_NAME:?G3DT_APP_NAME not set — run via the g3dt CLI}"
|
|
32
|
+
NAMESPACE="${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI}"
|
|
33
|
+
CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI}"
|
|
34
|
+
SCHEMA_REPO="${G3DT_SCHEMA_REPO:?G3DT_SCHEMA_REPO not set — run via the g3dt CLI}"
|
|
35
|
+
REGION="${G3DT_REGION:-ap-southeast-2}"
|
|
36
|
+
EKS_ARN="${G3DT_EKS_ARN:-}"
|
|
37
|
+
ARGO_SCRIPT_DIR="${SERVICE_DIR}/k8s_ops"
|
|
38
|
+
|
|
39
|
+
# Writable data lives outside the installed package.
|
|
40
|
+
SCHEMA_DIR="${G3DT_SCHEMA_DIR:-$HOME/.g3dt/schemas}"
|
|
41
|
+
SYNTH_BASE="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
|
|
42
|
+
|
|
43
|
+
# Derived variables
|
|
44
|
+
PREV_VERSION="v1.0.0"
|
|
45
|
+
SYNTH_META_DIR="${SYNTH_BASE}/${VERSION}/"
|
|
46
|
+
DATA_IMPORT_ORDER_FILE="${SYNTH_BASE}/${PREV_VERSION}/AusDiab_Simulated/DataImportOrder.txt"
|
|
47
|
+
|
|
48
|
+
# Never export an empty AWS_PROFILE (empty means ambient credentials).
|
|
49
|
+
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
50
|
+
export AWS_PROFILE="${G3DT_AWS_PROFILE}"
|
|
51
|
+
echo "==== [0] Configuring AWS PROFILE as '${AWS_PROFILE}' for profile '${PROFILE}' ===="
|
|
52
|
+
else
|
|
53
|
+
echo "==== [0] No AWS profile set for '${PROFILE}' — using ambient AWS credentials ===="
|
|
54
|
+
fi
|
|
55
|
+
|
|
56
|
+
echo "Updating kubeconfig for cluster: ${CLUSTER_NAME}"
|
|
57
|
+
# eks_arn is optional; only pass --role-arn when it is set.
|
|
58
|
+
ROLE_ARG=""
|
|
59
|
+
if [ -n "${EKS_ARN}" ] && [ "${EKS_ARN}" != "null" ]; then
|
|
60
|
+
ROLE_ARG="--role-arn ${EKS_ARN}"
|
|
61
|
+
fi
|
|
62
|
+
if ! aws eks update-kubeconfig \
|
|
63
|
+
--name "${CLUSTER_NAME}" \
|
|
64
|
+
--region "${REGION}" \
|
|
65
|
+
${ROLE_ARG}; then
|
|
66
|
+
echo "ERROR: Failed to update kubeconfig. Check your AWS credentials/profile."
|
|
67
|
+
exit 1
|
|
68
|
+
fi
|
|
69
|
+
|
|
70
|
+
echo "==== [1] Pulling dictionary for version ${VERSION} ===="
|
|
71
|
+
DICT_URL="https://raw.githubusercontent.com/${SCHEMA_REPO}"
|
|
72
|
+
DICT_URL="${DICT_URL}/refs/tags/${VERSION}"
|
|
73
|
+
DICT_URL="${DICT_URL}/dictionary/prod_dict/acdc_schema.json"
|
|
74
|
+
bash "${SERVICE_DIR}/dictionary/pull_dict.sh" "${DICT_URL}"
|
|
75
|
+
|
|
76
|
+
echo "==== [2] Uploading dictionary to S3: s3://${SCHEMA_S3_URI} ===="
|
|
77
|
+
UPLOAD_DICT_ARGS=("${SCHEMA_DIR}/acdc_schema_${VERSION}.json" "s3://${SCHEMA_S3_URI}")
|
|
78
|
+
# The optional trailing positional is the AWS profile; omit it for ambient credentials.
|
|
79
|
+
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
80
|
+
UPLOAD_DICT_ARGS+=("${G3DT_AWS_PROFILE}")
|
|
81
|
+
fi
|
|
82
|
+
python3 "${SERVICE_DIR}/dictionary/upload_dictionary.py" "${UPLOAD_DICT_ARGS[@]}"
|
|
83
|
+
|
|
84
|
+
echo "==== [3] Restarting microservices (schema) ===="
|
|
85
|
+
bash "${ARGO_SCRIPT_DIR}/argocd_restart_schema.sh" \
|
|
86
|
+
-d "${DOMAIN}" \
|
|
87
|
+
-a "${APP_NAME}" \
|
|
88
|
+
-n "${NAMESPACE}"
|
|
89
|
+
|
|
90
|
+
echo "==== [4] Deleting old synthetic data for version ${PREV_VERSION} ===="
|
|
91
|
+
DELETE_SYNTH_ARGS=(
|
|
92
|
+
-p "AusDiab_Simulated,EDCAD-PMS_Simulated,PREDICT_Simulated,Baker-Biobank_Simulated,CAUGHT-CAD_Simulated,BioHeart-CT_Simulated"
|
|
93
|
+
-s "${AWS_SECRET_NAME}"
|
|
94
|
+
-i "${DATA_IMPORT_ORDER_FILE}"
|
|
95
|
+
)
|
|
96
|
+
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
97
|
+
DELETE_SYNTH_ARGS+=(-profile "${G3DT_AWS_PROFILE}")
|
|
98
|
+
fi
|
|
99
|
+
python3 "${SERVICE_DIR}/synthetic_data/delete_synth_metadata_sheepdog.py" "${DELETE_SYNTH_ARGS[@]}"
|
|
100
|
+
|
|
101
|
+
echo "==== [5] Generating new synthetic data for version ${VERSION} (LLM-realistic) ===="
|
|
102
|
+
bash "${SERVICE_DIR}/synthetic_data/generate_synth_metadata.sh" \
|
|
103
|
+
--schema "${SCHEMA_DIR}/acdc_schema_${VERSION}.json" \
|
|
104
|
+
--version "${VERSION}" \
|
|
105
|
+
--provider llm \
|
|
106
|
+
--num-records "30,60,20,55" \
|
|
107
|
+
--output-root "${SYNTH_BASE}"
|
|
108
|
+
|
|
109
|
+
echo "==== [6] Uploading new synthetic data for version ${VERSION} ===="
|
|
110
|
+
UPLOAD_SYNTH_ARGS=(
|
|
111
|
+
--base-dir "${SYNTH_META_DIR}"
|
|
112
|
+
--aws-secret-name "${AWS_SECRET_NAME}"
|
|
113
|
+
)
|
|
114
|
+
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
115
|
+
UPLOAD_SYNTH_ARGS+=(--aws-profile "${G3DT_AWS_PROFILE}")
|
|
116
|
+
fi
|
|
117
|
+
python3 "${SERVICE_DIR}/synthetic_data/upload_synth_metadata_sheepdog.py" "${UPLOAD_SYNTH_ARGS[@]}"
|
|
118
|
+
|
|
119
|
+
echo "==== [7] Restarting microservices (schema and etl) ===="
|
|
120
|
+
bash "${ARGO_SCRIPT_DIR}/argocd_restart_etl.sh" \
|
|
121
|
+
-d "${DOMAIN}" \
|
|
122
|
+
-a "${APP_NAME}" \
|
|
123
|
+
-n "${NAMESPACE}" \
|
|
124
|
+
-s
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
#
|
|
3
|
+
# Generate schema-valid synthetic Gen3 metadata using gen3-metadata-simulator.
|
|
4
|
+
# One run per study writes a self-validated folder containing <node>.json +
|
|
5
|
+
# project.json + DataImportOrder.txt — the exact layout the upload step
|
|
6
|
+
# (upload_synth_metadata_sheepdog.py) consumes.
|
|
7
|
+
#
|
|
8
|
+
# The tool takes a LOCAL bundled Gen3 schema file (pulled by pull_dict.sh into
|
|
9
|
+
# ~/.g3dt/schemas/acdc_schema_<version>.json, or $G3DT_SCHEMA_DIR if set). The
|
|
10
|
+
# default provider is keyless 'random'; pass --provider llm for LLM-realistic
|
|
11
|
+
# values, in which case LLM config is read from ~/.g3dt/.env (or $G3DT_ENV_FILE
|
|
12
|
+
# if set): LLM_PROVIDER / LLM_MODEL / LLM_API_KEY_FILE.
|
|
13
|
+
|
|
14
|
+
set -euo pipefail
|
|
15
|
+
|
|
16
|
+
# LLM provider config file (lives outside the installed package).
|
|
17
|
+
ENV_FILE="${G3DT_ENV_FILE:-$HOME/.g3dt/.env}"
|
|
18
|
+
|
|
19
|
+
usage() {
|
|
20
|
+
cat <<EOF
|
|
21
|
+
Usage: $(basename "$0") --schema <path> --version <ver> [options]
|
|
22
|
+
|
|
23
|
+
Generate synthetic Gen3 metadata (one folder per study) with gen3-metadata-simulator.
|
|
24
|
+
|
|
25
|
+
Required:
|
|
26
|
+
--schema <path> Path to the bundled Gen3 JSON schema (cad.json).
|
|
27
|
+
--version <ver> Version label for the output dir (e.g. v1.1.5).
|
|
28
|
+
|
|
29
|
+
Options:
|
|
30
|
+
--studies s1,s2 Comma-separated study/project names.
|
|
31
|
+
Default: ${DEFAULT_STUDIES}
|
|
32
|
+
--num-records N|n1,n2 Records per study: one number for all, or a comma list
|
|
33
|
+
(one per study). Default: ${DEFAULT_NUM_RECORDS}
|
|
34
|
+
--provider random|llm Value strategy. Default: ${DEFAULT_PROVIDER}
|
|
35
|
+
'random' needs no key; 'llm' reads LLM config from
|
|
36
|
+
${ENV_FILE}.
|
|
37
|
+
--seed N RNG seed for reproducible output.
|
|
38
|
+
--output-root DIR Root output dir. Default: ${DEFAULT_OUTPUT_ROOT}
|
|
39
|
+
-h, --help Show this help and exit.
|
|
40
|
+
|
|
41
|
+
Examples:
|
|
42
|
+
$(basename "$0") --schema ~/.g3dt/schemas/acdc_schema_v1.1.5.json --version v1.1.5
|
|
43
|
+
$(basename "$0") --schema schema.json --version v1.1.5 --provider random --num-records 5
|
|
44
|
+
$(basename "$0") --schema schema.json --version v1.1.5 --num-records "30,60,20,55"
|
|
45
|
+
EOF
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
# Defaults (kept in sync with full_deploy_dd_and_synth.sh's 4-study record list)
|
|
49
|
+
DEFAULT_STUDIES="AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated"
|
|
50
|
+
DEFAULT_NUM_RECORDS=30
|
|
51
|
+
DEFAULT_PROVIDER=random
|
|
52
|
+
# Generated data goes outside the installed package.
|
|
53
|
+
DEFAULT_OUTPUT_ROOT="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
|
|
54
|
+
|
|
55
|
+
SCHEMA=""
|
|
56
|
+
VERSION=""
|
|
57
|
+
STUDIES="${DEFAULT_STUDIES}"
|
|
58
|
+
NUM_RECORDS="${DEFAULT_NUM_RECORDS}"
|
|
59
|
+
PROVIDER="${DEFAULT_PROVIDER}"
|
|
60
|
+
SEED=""
|
|
61
|
+
OUTPUT_ROOT="${DEFAULT_OUTPUT_ROOT}"
|
|
62
|
+
|
|
63
|
+
while [[ $# -gt 0 ]]; do
|
|
64
|
+
case "$1" in
|
|
65
|
+
--schema) SCHEMA="$2"; shift 2 ;;
|
|
66
|
+
--version) VERSION="$2"; shift 2 ;;
|
|
67
|
+
--studies) STUDIES="$2"; shift 2 ;;
|
|
68
|
+
--num-records) NUM_RECORDS="$2"; shift 2 ;;
|
|
69
|
+
--provider) PROVIDER="$2"; shift 2 ;;
|
|
70
|
+
--seed) SEED="$2"; shift 2 ;;
|
|
71
|
+
--output-root) OUTPUT_ROOT="$2"; shift 2 ;;
|
|
72
|
+
-h|--help) usage; exit 0 ;;
|
|
73
|
+
*) echo "Unknown argument: $1" >&2; usage; exit 1 ;;
|
|
74
|
+
esac
|
|
75
|
+
done
|
|
76
|
+
|
|
77
|
+
if [[ -z "$SCHEMA" || -z "$VERSION" ]]; then
|
|
78
|
+
echo "Error: --schema and --version are required." >&2
|
|
79
|
+
usage
|
|
80
|
+
exit 1
|
|
81
|
+
fi
|
|
82
|
+
if [[ ! -f "$SCHEMA" ]]; then
|
|
83
|
+
echo "Error: schema file not found: ${SCHEMA}" >&2
|
|
84
|
+
echo "Hint: pull it first, e.g. 'g3dt dict pull'." >&2
|
|
85
|
+
exit 1
|
|
86
|
+
fi
|
|
87
|
+
if ! command -v gen3-metadata-simulator &>/dev/null; then
|
|
88
|
+
echo "Error: gen3-metadata-simulator not found. Run 'g3dt synth install-simulator'." >&2
|
|
89
|
+
exit 1
|
|
90
|
+
fi
|
|
91
|
+
|
|
92
|
+
IFS=',' read -r -a STUDY_ARRAY <<< "$STUDIES"
|
|
93
|
+
|
|
94
|
+
# --num-records is either a single count applied to every study, or a comma list
|
|
95
|
+
# with one count per study (which must line up with --studies).
|
|
96
|
+
PER_STUDY_COUNTS=0
|
|
97
|
+
if [[ "$NUM_RECORDS" == *,* ]]; then
|
|
98
|
+
PER_STUDY_COUNTS=1
|
|
99
|
+
IFS=',' read -r -a NUM_RECORDS_ARRAY <<< "$NUM_RECORDS"
|
|
100
|
+
if [[ ${#NUM_RECORDS_ARRAY[@]} -ne ${#STUDY_ARRAY[@]} ]]; then
|
|
101
|
+
echo "Error: --num-records has ${#NUM_RECORDS_ARRAY[@]} values but there are ${#STUDY_ARRAY[@]} studies." >&2
|
|
102
|
+
exit 1
|
|
103
|
+
fi
|
|
104
|
+
fi
|
|
105
|
+
|
|
106
|
+
echo "Generating synthetic metadata: provider=${PROVIDER}, version=${VERSION}, studies=${STUDIES}"
|
|
107
|
+
|
|
108
|
+
for i in "${!STUDY_ARRAY[@]}"; do
|
|
109
|
+
STUDY="${STUDY_ARRAY[$i]}"
|
|
110
|
+
if [[ "$PER_STUDY_COUNTS" -eq 1 ]]; then
|
|
111
|
+
N="${NUM_RECORDS_ARRAY[$i]}"
|
|
112
|
+
else
|
|
113
|
+
N="$NUM_RECORDS"
|
|
114
|
+
fi
|
|
115
|
+
OUT="${OUTPUT_ROOT}/${VERSION}/${STUDY}"
|
|
116
|
+
mkdir -p "$OUT"
|
|
117
|
+
echo "==== ${STUDY} (n=${N}) -> ${OUT} ===="
|
|
118
|
+
|
|
119
|
+
CMD=(gen3-metadata-simulator generate
|
|
120
|
+
--schema "$SCHEMA"
|
|
121
|
+
--output-dir "$OUT"
|
|
122
|
+
--project-code "$STUDY"
|
|
123
|
+
--num-records "$N"
|
|
124
|
+
--provider "$PROVIDER")
|
|
125
|
+
[[ -n "$SEED" ]] && CMD+=(--seed "$SEED")
|
|
126
|
+
# Point the LLM provider at the user-level env file regardless of the caller's CWD.
|
|
127
|
+
if [[ "$PROVIDER" == "llm" && -f "${ENV_FILE}" ]]; then
|
|
128
|
+
CMD+=(--env-file "${ENV_FILE}")
|
|
129
|
+
fi
|
|
130
|
+
"${CMD[@]}"
|
|
131
|
+
done
|
|
132
|
+
|
|
133
|
+
echo "Done. Output under ${OUTPUT_ROOT}/${VERSION}/"
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import argparse
|
|
3
|
+
import logging
|
|
4
|
+
from g3dt.upload.metadata_submitter import (
|
|
5
|
+
find_data_import_order_file,
|
|
6
|
+
list_metadata_jsons,
|
|
7
|
+
create_boto3_session,
|
|
8
|
+
get_gen3_api_key_aws_secret,
|
|
9
|
+
MetadataSubmitter,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
DEFAULT_REGION = "ap-southeast-2"
|
|
13
|
+
DEFAULT_PROFILE = None # ambient credentials unless --aws-profile is given
|
|
14
|
+
DEFAULT_SUBMISSION_SIZE_KB = 50
|
|
15
|
+
|
|
16
|
+
def submit_synthetic_metadata(
|
|
17
|
+
base_dir,
|
|
18
|
+
project_id,
|
|
19
|
+
aws_secret_name,
|
|
20
|
+
aws_region=DEFAULT_REGION,
|
|
21
|
+
aws_profile=DEFAULT_PROFILE,
|
|
22
|
+
max_submission_size_kb=DEFAULT_SUBMISSION_SIZE_KB,
|
|
23
|
+
):
|
|
24
|
+
"""
|
|
25
|
+
Main function to orchestrate metadata submission for multiple projects.
|
|
26
|
+
"""
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
script_path = os.path.dirname(os.path.abspath(__file__))
|
|
29
|
+
os.chdir(script_path)
|
|
30
|
+
logger.info("Changed working directory to %s", script_path)
|
|
31
|
+
|
|
32
|
+
logger.info("Using base_dir=%s", base_dir)
|
|
33
|
+
|
|
34
|
+
session = create_boto3_session(aws_profile)
|
|
35
|
+
api_key = get_gen3_api_key_aws_secret(aws_secret_name, aws_region, session)
|
|
36
|
+
|
|
37
|
+
logger.info(
|
|
38
|
+
"Preparing to submit metadata for project_id=%s from %s",
|
|
39
|
+
project_id,
|
|
40
|
+
base_dir,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
try:
|
|
44
|
+
data_import_order_path = find_data_import_order_file(base_dir)
|
|
45
|
+
file_list = list_metadata_jsons(base_dir)
|
|
46
|
+
|
|
47
|
+
logger.info("Found %d metadata files for submission.", len(file_list))
|
|
48
|
+
|
|
49
|
+
submitter = MetadataSubmitter(
|
|
50
|
+
metadata_file_list=file_list,
|
|
51
|
+
api_key=api_key,
|
|
52
|
+
project_id=project_id,
|
|
53
|
+
data_import_order_path=data_import_order_path,
|
|
54
|
+
max_size_kb=max_submission_size_kb,
|
|
55
|
+
aws_profile=aws_profile,
|
|
56
|
+
upload_to_database=False,
|
|
57
|
+
dataset_root=None,
|
|
58
|
+
database=None,
|
|
59
|
+
table=None
|
|
60
|
+
)
|
|
61
|
+
submitter.submit_metadata()
|
|
62
|
+
logger.info("Finished submitting for project_id=%s", project_id)
|
|
63
|
+
|
|
64
|
+
except FileNotFoundError as e:
|
|
65
|
+
logger.error(
|
|
66
|
+
"Could not find required files for project %s in %s. Error: %s",
|
|
67
|
+
project_id,
|
|
68
|
+
base_dir,
|
|
69
|
+
e,
|
|
70
|
+
)
|
|
71
|
+
raise e
|
|
72
|
+
except Exception as e:
|
|
73
|
+
logger.error(
|
|
74
|
+
"An unexpected error occurred during submission for project %s: %s",
|
|
75
|
+
project_id,
|
|
76
|
+
e,
|
|
77
|
+
)
|
|
78
|
+
raise e
|
|
79
|
+
|
|
80
|
+
def main():
|
|
81
|
+
parser = argparse.ArgumentParser(
|
|
82
|
+
description="Upload synthetic Gen3 metadata for all projects in a directory."
|
|
83
|
+
)
|
|
84
|
+
parser.add_argument(
|
|
85
|
+
"--base-dir",
|
|
86
|
+
required=True,
|
|
87
|
+
help="Base directory where metadata project folders are found."
|
|
88
|
+
)
|
|
89
|
+
parser.add_argument(
|
|
90
|
+
"--aws-secret-name",
|
|
91
|
+
required=True,
|
|
92
|
+
help="AWS secrets manager secret name containing the Gen3 API key."
|
|
93
|
+
)
|
|
94
|
+
parser.add_argument(
|
|
95
|
+
"--aws-region",
|
|
96
|
+
default=DEFAULT_REGION,
|
|
97
|
+
help=f"AWS region for secrets manager (default: {DEFAULT_REGION})."
|
|
98
|
+
)
|
|
99
|
+
parser.add_argument(
|
|
100
|
+
"--aws-profile",
|
|
101
|
+
default=DEFAULT_PROFILE,
|
|
102
|
+
help="AWS profile to use (default: None)."
|
|
103
|
+
)
|
|
104
|
+
parser.add_argument(
|
|
105
|
+
"--submission-size-kb",
|
|
106
|
+
type=int,
|
|
107
|
+
default=DEFAULT_SUBMISSION_SIZE_KB,
|
|
108
|
+
help="Maximum submission size in KB per chunk (default: 50)."
|
|
109
|
+
)
|
|
110
|
+
args = parser.parse_args()
|
|
111
|
+
|
|
112
|
+
logging.basicConfig(
|
|
113
|
+
format="%(asctime)s %(levelname)s:%(name)s:%(message)s", level=logging.INFO
|
|
114
|
+
)
|
|
115
|
+
logger = logging.getLogger(__name__)
|
|
116
|
+
|
|
117
|
+
logger.info("Starting metadata submission script.")
|
|
118
|
+
|
|
119
|
+
# Validate base_dir
|
|
120
|
+
if not os.path.isdir(args.base_dir):
|
|
121
|
+
logger.error(
|
|
122
|
+
"The base directory %s does not exist or is not a directory.",
|
|
123
|
+
args.base_dir
|
|
124
|
+
)
|
|
125
|
+
raise FileNotFoundError(
|
|
126
|
+
"The base directory %s does not exist or is not a directory."
|
|
127
|
+
% args.base_dir
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
# Find all projects: folders inside base_dir
|
|
131
|
+
logger.info(
|
|
132
|
+
"Looking for project subdirectories in base directory: %s",
|
|
133
|
+
args.base_dir
|
|
134
|
+
)
|
|
135
|
+
project_id_list = os.listdir(args.base_dir)
|
|
136
|
+
logger.info(
|
|
137
|
+
"Found %d project subdirectories: %s",
|
|
138
|
+
len(project_id_list),
|
|
139
|
+
project_id_list
|
|
140
|
+
)
|
|
141
|
+
if not project_id_list:
|
|
142
|
+
logger.error(
|
|
143
|
+
"No project subdirectories found in base directory: %s",
|
|
144
|
+
args.base_dir
|
|
145
|
+
)
|
|
146
|
+
raise FileNotFoundError(
|
|
147
|
+
"No project subdirectories found in base directory: %s"
|
|
148
|
+
% args.base_dir
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
for project_id in project_id_list:
|
|
152
|
+
project_base_dir = os.path.join(args.base_dir, project_id)
|
|
153
|
+
submit_synthetic_metadata(
|
|
154
|
+
base_dir=project_base_dir,
|
|
155
|
+
project_id=project_id,
|
|
156
|
+
aws_secret_name=args.aws_secret_name,
|
|
157
|
+
aws_region=args.aws_region,
|
|
158
|
+
aws_profile=args.aws_profile,
|
|
159
|
+
max_submission_size_kb=args.submission_size_kb,
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
logger.info("Script finished.")
|
|
163
|
+
|
|
164
|
+
if __name__ == "__main__":
|
|
165
|
+
main()
|