gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,56 @@
1
+ #!/bin/bash
2
+
3
+ # Exit on any error
4
+ set -e
5
+ # Ensure pipeline failures are captured
6
+ set -o pipefail
7
+
8
+ # Usage: bash restart_etl_and_ms.sh <profile>
9
+ # <profile> is display-only; all configuration comes from G3DT_* environment
10
+ # variables exported by the g3dt CLI (g3dt.config.script_env).
11
+ PROFILE="${1:-}"
12
+
13
+ if [ -z "$PROFILE" ]; then
14
+ echo "Usage: $0 <profile> (test|staging|prod) — run via the g3dt CLI"
15
+ exit 1
16
+ fi
17
+
18
+ # Defining script paths
19
+ SCRIPT_DIR="$( cd -- "$( dirname -- "${BASH_SOURCE[0]}" )" &> /dev/null && pwd )"
20
+
21
+ # Configuration from G3DT_* env vars (fail loudly if a required one is missing)
22
+ CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI}"
23
+ DOMAIN="${G3DT_DOMAIN:?G3DT_DOMAIN not set — run via the g3dt CLI}"
24
+ APP_NAME="${G3DT_APP_NAME:?G3DT_APP_NAME not set — run via the g3dt CLI}"
25
+ NAMESPACE="${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI}"
26
+ REGION="${G3DT_REGION:-ap-southeast-2}"
27
+ # This script lives alongside the argo scripts inside the package.
28
+ ARGO_SCRIPT_DIR="${SCRIPT_DIR}"
29
+
30
+ # Never export an empty AWS_PROFILE (empty means ambient credentials).
31
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
32
+ export AWS_PROFILE="${G3DT_AWS_PROFILE}"
33
+ echo "==== Configuring AWS PROFILE as '${AWS_PROFILE}' for profile '${PROFILE}' ===="
34
+ else
35
+ echo "==== No AWS profile set for '${PROFILE}' — using ambient AWS credentials ===="
36
+ fi
37
+
38
+ echo "Updating kubeconfig for cluster: ${CLUSTER_NAME}"
39
+ if ! aws eks update-kubeconfig --name "${CLUSTER_NAME}" --region "${REGION}"; then
40
+ echo "ERROR: Failed to update kubeconfig. Check your AWS credentials/profile."
41
+ exit 1
42
+ fi
43
+
44
+ echo "==== Restarting microservices (etl) ===="
45
+ bash "${ARGO_SCRIPT_DIR}/argocd_restart_etl.sh" \
46
+ -d "${DOMAIN}" \
47
+ -a "${APP_NAME}" \
48
+ -n "${NAMESPACE}" \
49
+ -s
50
+
51
+ echo "==== Restarting microservices (schema) ===="
52
+ bash "${ARGO_SCRIPT_DIR}/argocd_restart_ms.sh" \
53
+ -d "${DOMAIN}" \
54
+ -a "${APP_NAME}" \
55
+ -n "${NAMESPACE}" \
56
+ -r "sheepdog-deployment,guppy-deployment,peregrine-deployment,portal-deployment"
@@ -0,0 +1,183 @@
1
+ import os
2
+ import argparse
3
+ import logging
4
+ from g3dt.upload.metadata_deleter import (
5
+ delete_project_metadata,
6
+ )
7
+ from g3dt.upload.metadata_submitter import (
8
+ create_boto3_session,
9
+ get_gen3_api_key_aws_secret,
10
+ create_gen3_submission_class,
11
+ )
12
+
13
+ EXCLUDE_NODES = [
14
+ "program",
15
+ "project",
16
+ "acknowledgement",
17
+ "publication",
18
+ ]
19
+
20
+
21
+ def load_import_order(import_order_path, exclude_nodes=None):
22
+ """
23
+ Reads the DataImportOrder.txt file and returns the node list
24
+ in deletion order (reversed, with excluded nodes removed).
25
+ """
26
+ if exclude_nodes is None:
27
+ exclude_nodes = EXCLUDE_NODES
28
+ with open(import_order_path, 'r', encoding='utf-8') as f:
29
+ nodes = [line.strip() for line in f if line.strip()]
30
+ nodes = [n for n in nodes if n not in exclude_nodes]
31
+ nodes.reverse()
32
+ return nodes
33
+
34
+
35
+ def delete_for_projects(
36
+ gen3_submission,
37
+ program_id,
38
+ project_ids,
39
+ nodes,
40
+ prompt_for_confirmation,
41
+ ):
42
+ for project_id in project_ids:
43
+ logging.info(
44
+ "Attempting to delete metadata for project '%s'",
45
+ project_id,
46
+ )
47
+ try:
48
+ delete_project_metadata(
49
+ gen3_submission=gen3_submission,
50
+ program_id=program_id,
51
+ project_id=project_id,
52
+ nodes=nodes,
53
+ prompt_for_confirmation=prompt_for_confirmation,
54
+ )
55
+ logging.info(
56
+ "Successfully deleted metadata for project '%s'.",
57
+ project_id,
58
+ )
59
+ except Exception as e:
60
+ logging.error(
61
+ "Error deleting metadata for project '%s': %s",
62
+ project_id, e,
63
+ )
64
+ raise e
65
+
66
+
67
+ if __name__ == "__main__":
68
+ logging.basicConfig(
69
+ level=logging.INFO,
70
+ format='[%(asctime)s] [%(levelname)s] %(message)s',
71
+ datefmt='%Y-%m-%d %H:%M:%S',
72
+ )
73
+
74
+ script_path = os.path.dirname(os.path.abspath(__file__))
75
+ os.chdir(script_path)
76
+ logging.info("Changed working directory to %s", script_path)
77
+
78
+ default_projects = [
79
+ "AusDiab_Simulated",
80
+ "EDCAD-PMS_Simulated",
81
+ "Baker-Biobank_Simulated",
82
+ "CAUGHT-CAD_Simulated",
83
+ "BioHeart-CT_Simulated",
84
+ ]
85
+ default_program_id = "program1"
86
+ default_aws_profile = "default"
87
+ default_aws_region = "ap-southeast-2"
88
+ default_prompt_for_confirmation = False
89
+
90
+ parser = argparse.ArgumentParser(
91
+ description=(
92
+ "Delete synthetic project metadata from Gen3 "
93
+ "Sheepdog API."
94
+ ),
95
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter,
96
+ )
97
+ parser.add_argument(
98
+ '-i', '--data-import-order-file',
99
+ type=str,
100
+ required=True,
101
+ help="Path to the DataImportOrder.txt file.",
102
+ )
103
+ parser.add_argument(
104
+ '-p', '--projects',
105
+ type=str,
106
+ default=",".join(default_projects),
107
+ help="Comma separated list of Project IDs.",
108
+ )
109
+ parser.add_argument(
110
+ '--program-id',
111
+ type=str,
112
+ default=default_program_id,
113
+ help="Gen3 program name.",
114
+ )
115
+ parser.add_argument(
116
+ '-s', '--aws-secret-name',
117
+ type=str,
118
+ help="AWS secret name for Gen3 API key.",
119
+ )
120
+ parser.add_argument(
121
+ '-profile', '--aws-profile',
122
+ type=str,
123
+ default=default_aws_profile,
124
+ help="AWS profile name.",
125
+ )
126
+ parser.add_argument(
127
+ '-r', '--aws-region',
128
+ type=str,
129
+ default=default_aws_region,
130
+ help="AWS region.",
131
+ )
132
+ parser.add_argument(
133
+ '-confirm', '--prompt',
134
+ action='store_true',
135
+ default=default_prompt_for_confirmation,
136
+ help="Prompt for confirmation before deleting.",
137
+ )
138
+
139
+ args = parser.parse_args()
140
+
141
+ parsed_projects = [
142
+ proj.strip()
143
+ for proj in args.projects.split(",")
144
+ if proj.strip()
145
+ ]
146
+
147
+ logging.info("Projects to delete: %s", parsed_projects)
148
+ logging.info(
149
+ "Data import order file: %s",
150
+ args.data_import_order_file,
151
+ )
152
+ logging.info("AWS Secret Name: %s", args.aws_secret_name)
153
+ logging.info("AWS Profile: %s", args.aws_profile)
154
+ logging.info("AWS Region: %s", args.aws_region)
155
+ logging.info(
156
+ "Prompt for confirmation: %s", args.prompt,
157
+ )
158
+
159
+ if not os.path.isfile(args.data_import_order_file):
160
+ raise FileNotFoundError(
161
+ f"Import order file does not exist: "
162
+ f"{args.data_import_order_file}"
163
+ )
164
+
165
+ nodes = load_import_order(args.data_import_order_file)
166
+
167
+ session = create_boto3_session(
168
+ aws_profile=args.aws_profile,
169
+ )
170
+ api_key = get_gen3_api_key_aws_secret(
171
+ secret_name=args.aws_secret_name,
172
+ region_name=args.aws_region,
173
+ session=session,
174
+ )
175
+ sub = create_gen3_submission_class(api_key)
176
+
177
+ delete_for_projects(
178
+ gen3_submission=sub,
179
+ program_id=args.program_id,
180
+ project_ids=parsed_projects,
181
+ nodes=nodes,
182
+ prompt_for_confirmation=args.prompt,
183
+ )
@@ -0,0 +1,124 @@
1
+ #!/bin/bash
2
+
3
+ # Exit on any error
4
+ set -e
5
+ # Ensure pipeline failures are captured
6
+ set -o pipefail
7
+
8
+ # Usage: bash full_deploy_dd_and_synth.sh <profile>
9
+ # <profile> is display-only; all configuration comes from G3DT_* environment
10
+ # variables exported by the g3dt CLI (g3dt.config.script_env).
11
+ PROFILE=$1
12
+
13
+ if [ -z "$PROFILE" ]; then
14
+ echo "Usage: $0 <profile> (test|staging|prod) — run via the g3dt CLI"
15
+ exit 1
16
+ fi
17
+
18
+ # Any configured profile is allowed here. The production warning/confirmation is
19
+ # enforced by the CLI entrypoint (`g3dt synth deploy`) before this script runs,
20
+ # since that has a TTY for the prompt.
21
+
22
+ # Defining script paths (sibling scripts ship together inside the package)
23
+ SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)"
24
+ SERVICE_DIR="${SCRIPT_DIR}/.."
25
+
26
+ # Configuration from G3DT_* env vars (fail loudly if a required one is missing)
27
+ VERSION="${G3DT_DICTIONARY_VERSION:?G3DT_DICTIONARY_VERSION not set — run via the g3dt CLI}"
28
+ AWS_SECRET_NAME="${G3DT_AWS_SECRET_NAME:?G3DT_AWS_SECRET_NAME not set — run via the g3dt CLI}"
29
+ SCHEMA_S3_URI="${G3DT_SCHEMA_S3_URI:?G3DT_SCHEMA_S3_URI not set — run via the g3dt CLI}"
30
+ DOMAIN="${G3DT_DOMAIN:?G3DT_DOMAIN not set — run via the g3dt CLI}"
31
+ APP_NAME="${G3DT_APP_NAME:?G3DT_APP_NAME not set — run via the g3dt CLI}"
32
+ NAMESPACE="${G3DT_NAMESPACE:?G3DT_NAMESPACE not set — run via the g3dt CLI}"
33
+ CLUSTER_NAME="${G3DT_CLUSTER_NAME:?G3DT_CLUSTER_NAME not set — run via the g3dt CLI}"
34
+ SCHEMA_REPO="${G3DT_SCHEMA_REPO:?G3DT_SCHEMA_REPO not set — run via the g3dt CLI}"
35
+ REGION="${G3DT_REGION:-ap-southeast-2}"
36
+ EKS_ARN="${G3DT_EKS_ARN:-}"
37
+ ARGO_SCRIPT_DIR="${SERVICE_DIR}/k8s_ops"
38
+
39
+ # Writable data lives outside the installed package.
40
+ SCHEMA_DIR="${G3DT_SCHEMA_DIR:-$HOME/.g3dt/schemas}"
41
+ SYNTH_BASE="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
42
+
43
+ # Derived variables
44
+ PREV_VERSION="v1.0.0"
45
+ SYNTH_META_DIR="${SYNTH_BASE}/${VERSION}/"
46
+ DATA_IMPORT_ORDER_FILE="${SYNTH_BASE}/${PREV_VERSION}/AusDiab_Simulated/DataImportOrder.txt"
47
+
48
+ # Never export an empty AWS_PROFILE (empty means ambient credentials).
49
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
50
+ export AWS_PROFILE="${G3DT_AWS_PROFILE}"
51
+ echo "==== [0] Configuring AWS PROFILE as '${AWS_PROFILE}' for profile '${PROFILE}' ===="
52
+ else
53
+ echo "==== [0] No AWS profile set for '${PROFILE}' — using ambient AWS credentials ===="
54
+ fi
55
+
56
+ echo "Updating kubeconfig for cluster: ${CLUSTER_NAME}"
57
+ # eks_arn is optional; only pass --role-arn when it is set.
58
+ ROLE_ARG=""
59
+ if [ -n "${EKS_ARN}" ] && [ "${EKS_ARN}" != "null" ]; then
60
+ ROLE_ARG="--role-arn ${EKS_ARN}"
61
+ fi
62
+ if ! aws eks update-kubeconfig \
63
+ --name "${CLUSTER_NAME}" \
64
+ --region "${REGION}" \
65
+ ${ROLE_ARG}; then
66
+ echo "ERROR: Failed to update kubeconfig. Check your AWS credentials/profile."
67
+ exit 1
68
+ fi
69
+
70
+ echo "==== [1] Pulling dictionary for version ${VERSION} ===="
71
+ DICT_URL="https://raw.githubusercontent.com/${SCHEMA_REPO}"
72
+ DICT_URL="${DICT_URL}/refs/tags/${VERSION}"
73
+ DICT_URL="${DICT_URL}/dictionary/prod_dict/acdc_schema.json"
74
+ bash "${SERVICE_DIR}/dictionary/pull_dict.sh" "${DICT_URL}"
75
+
76
+ echo "==== [2] Uploading dictionary to S3: s3://${SCHEMA_S3_URI} ===="
77
+ UPLOAD_DICT_ARGS=("${SCHEMA_DIR}/acdc_schema_${VERSION}.json" "s3://${SCHEMA_S3_URI}")
78
+ # The optional trailing positional is the AWS profile; omit it for ambient credentials.
79
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
80
+ UPLOAD_DICT_ARGS+=("${G3DT_AWS_PROFILE}")
81
+ fi
82
+ python3 "${SERVICE_DIR}/dictionary/upload_dictionary.py" "${UPLOAD_DICT_ARGS[@]}"
83
+
84
+ echo "==== [3] Restarting microservices (schema) ===="
85
+ bash "${ARGO_SCRIPT_DIR}/argocd_restart_schema.sh" \
86
+ -d "${DOMAIN}" \
87
+ -a "${APP_NAME}" \
88
+ -n "${NAMESPACE}"
89
+
90
+ echo "==== [4] Deleting old synthetic data for version ${PREV_VERSION} ===="
91
+ DELETE_SYNTH_ARGS=(
92
+ -p "AusDiab_Simulated,EDCAD-PMS_Simulated,PREDICT_Simulated,Baker-Biobank_Simulated,CAUGHT-CAD_Simulated,BioHeart-CT_Simulated"
93
+ -s "${AWS_SECRET_NAME}"
94
+ -i "${DATA_IMPORT_ORDER_FILE}"
95
+ )
96
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
97
+ DELETE_SYNTH_ARGS+=(-profile "${G3DT_AWS_PROFILE}")
98
+ fi
99
+ python3 "${SERVICE_DIR}/synthetic_data/delete_synth_metadata_sheepdog.py" "${DELETE_SYNTH_ARGS[@]}"
100
+
101
+ echo "==== [5] Generating new synthetic data for version ${VERSION} (LLM-realistic) ===="
102
+ bash "${SERVICE_DIR}/synthetic_data/generate_synth_metadata.sh" \
103
+ --schema "${SCHEMA_DIR}/acdc_schema_${VERSION}.json" \
104
+ --version "${VERSION}" \
105
+ --provider llm \
106
+ --num-records "30,60,20,55" \
107
+ --output-root "${SYNTH_BASE}"
108
+
109
+ echo "==== [6] Uploading new synthetic data for version ${VERSION} ===="
110
+ UPLOAD_SYNTH_ARGS=(
111
+ --base-dir "${SYNTH_META_DIR}"
112
+ --aws-secret-name "${AWS_SECRET_NAME}"
113
+ )
114
+ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
115
+ UPLOAD_SYNTH_ARGS+=(--aws-profile "${G3DT_AWS_PROFILE}")
116
+ fi
117
+ python3 "${SERVICE_DIR}/synthetic_data/upload_synth_metadata_sheepdog.py" "${UPLOAD_SYNTH_ARGS[@]}"
118
+
119
+ echo "==== [7] Restarting microservices (schema and etl) ===="
120
+ bash "${ARGO_SCRIPT_DIR}/argocd_restart_etl.sh" \
121
+ -d "${DOMAIN}" \
122
+ -a "${APP_NAME}" \
123
+ -n "${NAMESPACE}" \
124
+ -s
@@ -0,0 +1,133 @@
1
+ #!/usr/bin/env bash
2
+ #
3
+ # Generate schema-valid synthetic Gen3 metadata using gen3-metadata-simulator.
4
+ # One run per study writes a self-validated folder containing <node>.json +
5
+ # project.json + DataImportOrder.txt — the exact layout the upload step
6
+ # (upload_synth_metadata_sheepdog.py) consumes.
7
+ #
8
+ # The tool takes a LOCAL bundled Gen3 schema file (pulled by pull_dict.sh into
9
+ # ~/.g3dt/schemas/acdc_schema_<version>.json, or $G3DT_SCHEMA_DIR if set). The
10
+ # default provider is keyless 'random'; pass --provider llm for LLM-realistic
11
+ # values, in which case LLM config is read from ~/.g3dt/.env (or $G3DT_ENV_FILE
12
+ # if set): LLM_PROVIDER / LLM_MODEL / LLM_API_KEY_FILE.
13
+
14
+ set -euo pipefail
15
+
16
+ # LLM provider config file (lives outside the installed package).
17
+ ENV_FILE="${G3DT_ENV_FILE:-$HOME/.g3dt/.env}"
18
+
19
+ usage() {
20
+ cat <<EOF
21
+ Usage: $(basename "$0") --schema <path> --version <ver> [options]
22
+
23
+ Generate synthetic Gen3 metadata (one folder per study) with gen3-metadata-simulator.
24
+
25
+ Required:
26
+ --schema <path> Path to the bundled Gen3 JSON schema (cad.json).
27
+ --version <ver> Version label for the output dir (e.g. v1.1.5).
28
+
29
+ Options:
30
+ --studies s1,s2 Comma-separated study/project names.
31
+ Default: ${DEFAULT_STUDIES}
32
+ --num-records N|n1,n2 Records per study: one number for all, or a comma list
33
+ (one per study). Default: ${DEFAULT_NUM_RECORDS}
34
+ --provider random|llm Value strategy. Default: ${DEFAULT_PROVIDER}
35
+ 'random' needs no key; 'llm' reads LLM config from
36
+ ${ENV_FILE}.
37
+ --seed N RNG seed for reproducible output.
38
+ --output-root DIR Root output dir. Default: ${DEFAULT_OUTPUT_ROOT}
39
+ -h, --help Show this help and exit.
40
+
41
+ Examples:
42
+ $(basename "$0") --schema ~/.g3dt/schemas/acdc_schema_v1.1.5.json --version v1.1.5
43
+ $(basename "$0") --schema schema.json --version v1.1.5 --provider random --num-records 5
44
+ $(basename "$0") --schema schema.json --version v1.1.5 --num-records "30,60,20,55"
45
+ EOF
46
+ }
47
+
48
+ # Defaults (kept in sync with full_deploy_dd_and_synth.sh's 4-study record list)
49
+ DEFAULT_STUDIES="AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated"
50
+ DEFAULT_NUM_RECORDS=30
51
+ DEFAULT_PROVIDER=random
52
+ # Generated data goes outside the installed package.
53
+ DEFAULT_OUTPUT_ROOT="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
54
+
55
+ SCHEMA=""
56
+ VERSION=""
57
+ STUDIES="${DEFAULT_STUDIES}"
58
+ NUM_RECORDS="${DEFAULT_NUM_RECORDS}"
59
+ PROVIDER="${DEFAULT_PROVIDER}"
60
+ SEED=""
61
+ OUTPUT_ROOT="${DEFAULT_OUTPUT_ROOT}"
62
+
63
+ while [[ $# -gt 0 ]]; do
64
+ case "$1" in
65
+ --schema) SCHEMA="$2"; shift 2 ;;
66
+ --version) VERSION="$2"; shift 2 ;;
67
+ --studies) STUDIES="$2"; shift 2 ;;
68
+ --num-records) NUM_RECORDS="$2"; shift 2 ;;
69
+ --provider) PROVIDER="$2"; shift 2 ;;
70
+ --seed) SEED="$2"; shift 2 ;;
71
+ --output-root) OUTPUT_ROOT="$2"; shift 2 ;;
72
+ -h|--help) usage; exit 0 ;;
73
+ *) echo "Unknown argument: $1" >&2; usage; exit 1 ;;
74
+ esac
75
+ done
76
+
77
+ if [[ -z "$SCHEMA" || -z "$VERSION" ]]; then
78
+ echo "Error: --schema and --version are required." >&2
79
+ usage
80
+ exit 1
81
+ fi
82
+ if [[ ! -f "$SCHEMA" ]]; then
83
+ echo "Error: schema file not found: ${SCHEMA}" >&2
84
+ echo "Hint: pull it first, e.g. 'g3dt dict pull'." >&2
85
+ exit 1
86
+ fi
87
+ if ! command -v gen3-metadata-simulator &>/dev/null; then
88
+ echo "Error: gen3-metadata-simulator not found. Run 'g3dt synth install-simulator'." >&2
89
+ exit 1
90
+ fi
91
+
92
+ IFS=',' read -r -a STUDY_ARRAY <<< "$STUDIES"
93
+
94
+ # --num-records is either a single count applied to every study, or a comma list
95
+ # with one count per study (which must line up with --studies).
96
+ PER_STUDY_COUNTS=0
97
+ if [[ "$NUM_RECORDS" == *,* ]]; then
98
+ PER_STUDY_COUNTS=1
99
+ IFS=',' read -r -a NUM_RECORDS_ARRAY <<< "$NUM_RECORDS"
100
+ if [[ ${#NUM_RECORDS_ARRAY[@]} -ne ${#STUDY_ARRAY[@]} ]]; then
101
+ echo "Error: --num-records has ${#NUM_RECORDS_ARRAY[@]} values but there are ${#STUDY_ARRAY[@]} studies." >&2
102
+ exit 1
103
+ fi
104
+ fi
105
+
106
+ echo "Generating synthetic metadata: provider=${PROVIDER}, version=${VERSION}, studies=${STUDIES}"
107
+
108
+ for i in "${!STUDY_ARRAY[@]}"; do
109
+ STUDY="${STUDY_ARRAY[$i]}"
110
+ if [[ "$PER_STUDY_COUNTS" -eq 1 ]]; then
111
+ N="${NUM_RECORDS_ARRAY[$i]}"
112
+ else
113
+ N="$NUM_RECORDS"
114
+ fi
115
+ OUT="${OUTPUT_ROOT}/${VERSION}/${STUDY}"
116
+ mkdir -p "$OUT"
117
+ echo "==== ${STUDY} (n=${N}) -> ${OUT} ===="
118
+
119
+ CMD=(gen3-metadata-simulator generate
120
+ --schema "$SCHEMA"
121
+ --output-dir "$OUT"
122
+ --project-code "$STUDY"
123
+ --num-records "$N"
124
+ --provider "$PROVIDER")
125
+ [[ -n "$SEED" ]] && CMD+=(--seed "$SEED")
126
+ # Point the LLM provider at the user-level env file regardless of the caller's CWD.
127
+ if [[ "$PROVIDER" == "llm" && -f "${ENV_FILE}" ]]; then
128
+ CMD+=(--env-file "${ENV_FILE}")
129
+ fi
130
+ "${CMD[@]}"
131
+ done
132
+
133
+ echo "Done. Output under ${OUTPUT_ROOT}/${VERSION}/"
@@ -0,0 +1,165 @@
1
+ import os
2
+ import argparse
3
+ import logging
4
+ from g3dt.upload.metadata_submitter import (
5
+ find_data_import_order_file,
6
+ list_metadata_jsons,
7
+ create_boto3_session,
8
+ get_gen3_api_key_aws_secret,
9
+ MetadataSubmitter,
10
+ )
11
+
12
+ DEFAULT_REGION = "ap-southeast-2"
13
+ DEFAULT_PROFILE = None # ambient credentials unless --aws-profile is given
14
+ DEFAULT_SUBMISSION_SIZE_KB = 50
15
+
16
+ def submit_synthetic_metadata(
17
+ base_dir,
18
+ project_id,
19
+ aws_secret_name,
20
+ aws_region=DEFAULT_REGION,
21
+ aws_profile=DEFAULT_PROFILE,
22
+ max_submission_size_kb=DEFAULT_SUBMISSION_SIZE_KB,
23
+ ):
24
+ """
25
+ Main function to orchestrate metadata submission for multiple projects.
26
+ """
27
+ logger = logging.getLogger(__name__)
28
+ script_path = os.path.dirname(os.path.abspath(__file__))
29
+ os.chdir(script_path)
30
+ logger.info("Changed working directory to %s", script_path)
31
+
32
+ logger.info("Using base_dir=%s", base_dir)
33
+
34
+ session = create_boto3_session(aws_profile)
35
+ api_key = get_gen3_api_key_aws_secret(aws_secret_name, aws_region, session)
36
+
37
+ logger.info(
38
+ "Preparing to submit metadata for project_id=%s from %s",
39
+ project_id,
40
+ base_dir,
41
+ )
42
+
43
+ try:
44
+ data_import_order_path = find_data_import_order_file(base_dir)
45
+ file_list = list_metadata_jsons(base_dir)
46
+
47
+ logger.info("Found %d metadata files for submission.", len(file_list))
48
+
49
+ submitter = MetadataSubmitter(
50
+ metadata_file_list=file_list,
51
+ api_key=api_key,
52
+ project_id=project_id,
53
+ data_import_order_path=data_import_order_path,
54
+ max_size_kb=max_submission_size_kb,
55
+ aws_profile=aws_profile,
56
+ upload_to_database=False,
57
+ dataset_root=None,
58
+ database=None,
59
+ table=None
60
+ )
61
+ submitter.submit_metadata()
62
+ logger.info("Finished submitting for project_id=%s", project_id)
63
+
64
+ except FileNotFoundError as e:
65
+ logger.error(
66
+ "Could not find required files for project %s in %s. Error: %s",
67
+ project_id,
68
+ base_dir,
69
+ e,
70
+ )
71
+ raise e
72
+ except Exception as e:
73
+ logger.error(
74
+ "An unexpected error occurred during submission for project %s: %s",
75
+ project_id,
76
+ e,
77
+ )
78
+ raise e
79
+
80
+ def main():
81
+ parser = argparse.ArgumentParser(
82
+ description="Upload synthetic Gen3 metadata for all projects in a directory."
83
+ )
84
+ parser.add_argument(
85
+ "--base-dir",
86
+ required=True,
87
+ help="Base directory where metadata project folders are found."
88
+ )
89
+ parser.add_argument(
90
+ "--aws-secret-name",
91
+ required=True,
92
+ help="AWS secrets manager secret name containing the Gen3 API key."
93
+ )
94
+ parser.add_argument(
95
+ "--aws-region",
96
+ default=DEFAULT_REGION,
97
+ help=f"AWS region for secrets manager (default: {DEFAULT_REGION})."
98
+ )
99
+ parser.add_argument(
100
+ "--aws-profile",
101
+ default=DEFAULT_PROFILE,
102
+ help="AWS profile to use (default: None)."
103
+ )
104
+ parser.add_argument(
105
+ "--submission-size-kb",
106
+ type=int,
107
+ default=DEFAULT_SUBMISSION_SIZE_KB,
108
+ help="Maximum submission size in KB per chunk (default: 50)."
109
+ )
110
+ args = parser.parse_args()
111
+
112
+ logging.basicConfig(
113
+ format="%(asctime)s %(levelname)s:%(name)s:%(message)s", level=logging.INFO
114
+ )
115
+ logger = logging.getLogger(__name__)
116
+
117
+ logger.info("Starting metadata submission script.")
118
+
119
+ # Validate base_dir
120
+ if not os.path.isdir(args.base_dir):
121
+ logger.error(
122
+ "The base directory %s does not exist or is not a directory.",
123
+ args.base_dir
124
+ )
125
+ raise FileNotFoundError(
126
+ "The base directory %s does not exist or is not a directory."
127
+ % args.base_dir
128
+ )
129
+
130
+ # Find all projects: folders inside base_dir
131
+ logger.info(
132
+ "Looking for project subdirectories in base directory: %s",
133
+ args.base_dir
134
+ )
135
+ project_id_list = os.listdir(args.base_dir)
136
+ logger.info(
137
+ "Found %d project subdirectories: %s",
138
+ len(project_id_list),
139
+ project_id_list
140
+ )
141
+ if not project_id_list:
142
+ logger.error(
143
+ "No project subdirectories found in base directory: %s",
144
+ args.base_dir
145
+ )
146
+ raise FileNotFoundError(
147
+ "No project subdirectories found in base directory: %s"
148
+ % args.base_dir
149
+ )
150
+
151
+ for project_id in project_id_list:
152
+ project_base_dir = os.path.join(args.base_dir, project_id)
153
+ submit_synthetic_metadata(
154
+ base_dir=project_base_dir,
155
+ project_id=project_id,
156
+ aws_secret_name=args.aws_secret_name,
157
+ aws_region=args.aws_region,
158
+ aws_profile=args.aws_profile,
159
+ max_submission_size_kb=args.submission_size_kb,
160
+ )
161
+
162
+ logger.info("Script finished.")
163
+
164
+ if __name__ == "__main__":
165
+ main()