gen3-dataops-toolkit 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- g3dt/__init__.py +0 -0
- g3dt/cli/__init__.py +5 -0
- g3dt/cli/_internal/__init__.py +1 -0
- g3dt/cli/_internal/dispatch.py +428 -0
- g3dt/cli/_internal/registry.py +65 -0
- g3dt/cli/_internal/resolve.py +22 -0
- g3dt/cli/_internal/runner.py +76 -0
- g3dt/cli/_internal/safety.py +110 -0
- g3dt/cli/config_cmds.py +202 -0
- g3dt/cli/delete_cmds.py +101 -0
- g3dt/cli/dict_cmds.py +102 -0
- g3dt/cli/ec2_cmds.py +114 -0
- g3dt/cli/indexd_cmds.py +57 -0
- g3dt/cli/jobs.py +83 -0
- g3dt/cli/k8s.py +54 -0
- g3dt/cli/main.py +110 -0
- g3dt/cli/metadata.py +76 -0
- g3dt/cli/synth.py +206 -0
- g3dt/config.py +393 -0
- g3dt/indexd/__init__.py +0 -0
- g3dt/indexd/indexd_registrar.py +244 -0
- g3dt/ingest/ingest.py +629 -0
- g3dt/resolver.py +163 -0
- g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
- g3dt/services/delete/delete_metadata.sh +153 -0
- g3dt/services/delete/delete_metadata_by_guid.py +338 -0
- g3dt/services/dictionary/deploy_dd.sh +65 -0
- g3dt/services/dictionary/pull_dict.sh +59 -0
- g3dt/services/dictionary/upload_dictionary.py +109 -0
- g3dt/services/indexd/register_indexd.py +240 -0
- g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
- g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
- g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
- g3dt/services/k8s_ops/login_to_pod.sh +110 -0
- g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
- g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
- g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
- g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
- g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
- g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
- g3dt/services/upload/metadata/upload_metadata.py +152 -0
- g3dt/upload/__init__.py +1 -0
- g3dt/upload/metadata_deleter.py +265 -0
- g3dt/upload/metadata_submitter.py +1093 -0
- g3dt/upload/upload_synthdata_s3.py +164 -0
- g3dt/utils/athena_utils.py +834 -0
- g3dt/utils/dbt_utils.py +66 -0
- g3dt/utils/release_writer.py +188 -0
- g3dt/validate/validate.py +609 -0
- gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
- gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
- gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
- gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,1093 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import sys
|
|
3
|
+
import time
|
|
4
|
+
import json
|
|
5
|
+
import boto3
|
|
6
|
+
from botocore.exceptions import BotoCoreError, ClientError
|
|
7
|
+
from gen3.auth import Gen3Auth
|
|
8
|
+
from gen3.submission import Gen3Submission
|
|
9
|
+
import logging
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
import jwt
|
|
12
|
+
import requests
|
|
13
|
+
from typing import Any, Dict, List, Optional
|
|
14
|
+
import re
|
|
15
|
+
import pandas as pd
|
|
16
|
+
import uuid
|
|
17
|
+
from g3dt.utils.athena_utils import write_iceberg_to_db
|
|
18
|
+
from tenacity import retry, stop_after_attempt, wait_exponential
|
|
19
|
+
|
|
20
|
+
# redefine to use local cache in /tmp
|
|
21
|
+
os.environ['XDG_CACHE_HOME'] = '/tmp/.cache'
|
|
22
|
+
|
|
23
|
+
logger = logging.getLogger(__name__)
|
|
24
|
+
|
|
25
|
+
def create_boto3_session(
|
|
26
|
+
aws_profile: Optional[str] = None,
|
|
27
|
+
aws_region: Optional[str] = None,
|
|
28
|
+
):
|
|
29
|
+
"""
|
|
30
|
+
Create and return a boto3 Session object using an optional AWS profile.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
aws_profile (str, optional): The AWS CLI named profile to use.
|
|
34
|
+
If None, uses default credentials.
|
|
35
|
+
aws_region (str, optional): The AWS region to use.
|
|
36
|
+
If None, uses the default region from the environment.
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
boto3.Session: The created session instance.
|
|
40
|
+
"""
|
|
41
|
+
logger.debug(
|
|
42
|
+
"Creating boto3 session with aws_profile=%s, aws_region=%s",
|
|
43
|
+
aws_profile, aws_region,
|
|
44
|
+
)
|
|
45
|
+
kwargs = {}
|
|
46
|
+
if aws_profile:
|
|
47
|
+
kwargs['profile_name'] = aws_profile
|
|
48
|
+
if aws_region:
|
|
49
|
+
kwargs['region_name'] = aws_region
|
|
50
|
+
return boto3.Session(**kwargs)
|
|
51
|
+
|
|
52
|
+
def is_s3_uri(s3_uri: str) -> bool:
|
|
53
|
+
"""
|
|
54
|
+
Check if the provided URI is a valid S3 URI.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
s3_uri (str): The string to check.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
bool: True if the string starts with 's3://', False otherwise.
|
|
61
|
+
"""
|
|
62
|
+
logger.debug("Checking if %s is an S3 URI.", s3_uri)
|
|
63
|
+
return s3_uri.startswith("s3://")
|
|
64
|
+
|
|
65
|
+
def get_filename(file_path: str) -> str:
|
|
66
|
+
"""
|
|
67
|
+
Extract the filename from a file path.
|
|
68
|
+
|
|
69
|
+
Args:
|
|
70
|
+
file_path (str): The full path to a file.
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
str: The filename (with extension).
|
|
74
|
+
"""
|
|
75
|
+
filename = file_path.split("/")[-1]
|
|
76
|
+
logger.debug(
|
|
77
|
+
"Extracted filename '%s' from file_path '%s'.",
|
|
78
|
+
filename,
|
|
79
|
+
file_path,
|
|
80
|
+
)
|
|
81
|
+
return filename
|
|
82
|
+
|
|
83
|
+
def get_node_from_file_path(file_path: str) -> str:
|
|
84
|
+
"""
|
|
85
|
+
Extract the node name from a file path, assuming file is named as 'node.json'.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
file_path (str): The file path.
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
str: The base node name before the extension.
|
|
92
|
+
"""
|
|
93
|
+
filename = get_filename(file_path)
|
|
94
|
+
node = filename.split(".")[0]
|
|
95
|
+
logger.debug("Extracted node '%s' from filename '%s'.", node, filename)
|
|
96
|
+
return node
|
|
97
|
+
|
|
98
|
+
def list_metadata_jsons(metadata_dir: str) -> list:
|
|
99
|
+
"""
|
|
100
|
+
List all .json files in a given directory.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
metadata_dir (str): Directory containing metadata JSON files.
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
list: List of absolute paths to all .json files in the directory.
|
|
107
|
+
|
|
108
|
+
Raises:
|
|
109
|
+
Exception: If there is an error reading the directory.
|
|
110
|
+
"""
|
|
111
|
+
try:
|
|
112
|
+
logger.info(
|
|
113
|
+
"Listing .json files in metadata directory: %s",
|
|
114
|
+
metadata_dir,
|
|
115
|
+
)
|
|
116
|
+
files = os.listdir(metadata_dir)
|
|
117
|
+
return [
|
|
118
|
+
os.path.abspath(os.path.join(metadata_dir, file_name))
|
|
119
|
+
for file_name in files
|
|
120
|
+
if file_name.endswith(".json")
|
|
121
|
+
]
|
|
122
|
+
except OSError as e:
|
|
123
|
+
logger.error("Error listing metadata JSONs in %s: %s", metadata_dir, e)
|
|
124
|
+
raise
|
|
125
|
+
|
|
126
|
+
def find_data_import_order_file(metadata_dir: str) -> str:
|
|
127
|
+
"""
|
|
128
|
+
Find the DataImportOrder.txt file within a directory.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
metadata_dir (str): Directory to search in.
|
|
132
|
+
|
|
133
|
+
Returns:
|
|
134
|
+
str: Full path to the DataImportOrder.txt file.
|
|
135
|
+
|
|
136
|
+
Raises:
|
|
137
|
+
FileNotFoundError: If no such file is found.
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
logger.info("Searching for DataImportOrder.txt in %s", metadata_dir)
|
|
141
|
+
files = [os.path.join(metadata_dir, f) for f in os.listdir(metadata_dir)]
|
|
142
|
+
order_files = [f for f in files if "DataImportOrder.txt" in f]
|
|
143
|
+
if not order_files:
|
|
144
|
+
logger.error("No DataImportOrder.txt file found in the given directory.")
|
|
145
|
+
raise FileNotFoundError(
|
|
146
|
+
"No DataImportOrder.txt file found in the given directory."
|
|
147
|
+
)
|
|
148
|
+
logger.debug("Found DataImportOrder.txt file: %s", order_files[0])
|
|
149
|
+
return order_files[0]
|
|
150
|
+
except OSError as e:
|
|
151
|
+
logger.error(
|
|
152
|
+
"Error finding DataImportOrder.txt in %s: %s",
|
|
153
|
+
metadata_dir,
|
|
154
|
+
e,
|
|
155
|
+
)
|
|
156
|
+
raise
|
|
157
|
+
|
|
158
|
+
def list_metadata_jsons_s3(s3_uri: str, session) -> list:
|
|
159
|
+
"""
|
|
160
|
+
List all .json files in an S3 "directory" (prefix).
|
|
161
|
+
|
|
162
|
+
Args:
|
|
163
|
+
s3_uri (str): S3 URI to the metadata directory
|
|
164
|
+
(e.g. "s3://my-bucket/path/to/dir").
|
|
165
|
+
session (boto3.Session): An active boto3 Session.
|
|
166
|
+
|
|
167
|
+
Returns:
|
|
168
|
+
list: List of S3 URIs for all .json files found under the prefix.
|
|
169
|
+
"""
|
|
170
|
+
logger.info("Listing .json files in S3 metadata directory: %s", s3_uri)
|
|
171
|
+
s3 = session.client('s3')
|
|
172
|
+
bucket = s3_uri.split("/")[2]
|
|
173
|
+
prefix = "/".join(s3_uri.split("/")[3:])
|
|
174
|
+
if prefix and not prefix.endswith("/"):
|
|
175
|
+
prefix += "/" # Ensure prefix ends with a slash for directories
|
|
176
|
+
|
|
177
|
+
objects = s3.list_objects(Bucket=bucket, Prefix=prefix)
|
|
178
|
+
result = [
|
|
179
|
+
f"s3://{bucket}/{obj['Key']}"
|
|
180
|
+
for obj in objects.get('Contents', [])
|
|
181
|
+
if obj['Key'].endswith(".json")
|
|
182
|
+
]
|
|
183
|
+
logger.debug("Found %s .json files in S3 at %s", len(result), s3_uri)
|
|
184
|
+
return result
|
|
185
|
+
|
|
186
|
+
def find_data_import_order_file_s3(s3_uri: str, session) -> str:
|
|
187
|
+
"""
|
|
188
|
+
Search for the DataImportOrder.txt file in an S3 directory.
|
|
189
|
+
|
|
190
|
+
Args:
|
|
191
|
+
s3_uri (str): S3 URI specifying the directory/prefix to search.
|
|
192
|
+
session (boto3.Session): An active boto3 Session.
|
|
193
|
+
|
|
194
|
+
Returns:
|
|
195
|
+
str: Full S3 URI of the found DataImportOrder.txt file.
|
|
196
|
+
|
|
197
|
+
Raises:
|
|
198
|
+
FileNotFoundError: If the file does not exist in the specified prefix.
|
|
199
|
+
"""
|
|
200
|
+
logger.info(
|
|
201
|
+
"Searching for DataImportOrder.txt in S3 metadata directory: %s",
|
|
202
|
+
s3_uri,
|
|
203
|
+
)
|
|
204
|
+
s3 = session.client('s3')
|
|
205
|
+
bucket = s3_uri.split("/")[2]
|
|
206
|
+
prefix = "/".join(s3_uri.split("/")[3:])
|
|
207
|
+
objects = s3.list_objects(Bucket=bucket, Prefix=prefix)
|
|
208
|
+
order_files = [
|
|
209
|
+
obj['Key']
|
|
210
|
+
for obj in objects.get('Contents', [])
|
|
211
|
+
if obj['Key'].endswith("DataImportOrder.txt")
|
|
212
|
+
]
|
|
213
|
+
if not order_files:
|
|
214
|
+
logger.error("No DataImportOrder.txt file found in the given S3 directory.")
|
|
215
|
+
raise FileNotFoundError(
|
|
216
|
+
"No DataImportOrder.txt file found in the given directory."
|
|
217
|
+
)
|
|
218
|
+
logger.debug(
|
|
219
|
+
"Found DataImportOrder.txt file in S3: s3://%s/%s",
|
|
220
|
+
bucket,
|
|
221
|
+
order_files[0],
|
|
222
|
+
)
|
|
223
|
+
return f"s3://{bucket}/{order_files[0]}"
|
|
224
|
+
|
|
225
|
+
def read_metadata_json(file_path: str) -> dict:
|
|
226
|
+
"""
|
|
227
|
+
Read and return a JSON file from the local file system.
|
|
228
|
+
|
|
229
|
+
Args:
|
|
230
|
+
file_path (str): Path to the .json file.
|
|
231
|
+
|
|
232
|
+
Returns:
|
|
233
|
+
dict or list: Parsed contents of the JSON file.
|
|
234
|
+
"""
|
|
235
|
+
logger.info("Reading metadata json from local file: %s", file_path)
|
|
236
|
+
with open(file_path, "r", encoding="utf-8") as f:
|
|
237
|
+
data = json.load(f)
|
|
238
|
+
logger.debug(
|
|
239
|
+
"Read %s objects from %s",
|
|
240
|
+
len(data) if isinstance(data, list) else 'object',
|
|
241
|
+
file_path,
|
|
242
|
+
)
|
|
243
|
+
return data
|
|
244
|
+
|
|
245
|
+
def read_metadata_json_s3(s3_uri: str, session) -> dict:
|
|
246
|
+
"""
|
|
247
|
+
Read and return JSON data from an S3 file.
|
|
248
|
+
|
|
249
|
+
Args:
|
|
250
|
+
s3_uri (str): Full S3 URI to the .json file.
|
|
251
|
+
session (boto3.Session): Boto3 session.
|
|
252
|
+
|
|
253
|
+
Returns:
|
|
254
|
+
dict or list: Parsed JSON object from S3 file.
|
|
255
|
+
"""
|
|
256
|
+
logger.info("Reading metadata json from S3 file: %s", s3_uri)
|
|
257
|
+
s3 = session.client('s3')
|
|
258
|
+
obj = s3.get_object(
|
|
259
|
+
Bucket=s3_uri.split("/")[2],
|
|
260
|
+
Key="/".join(s3_uri.split("/")[3:]),
|
|
261
|
+
)
|
|
262
|
+
data = json.loads(obj['Body'].read().decode('utf-8'))
|
|
263
|
+
logger.debug(
|
|
264
|
+
"Read %s objects from %s",
|
|
265
|
+
len(data) if isinstance(data, list) else 'object',
|
|
266
|
+
s3_uri,
|
|
267
|
+
)
|
|
268
|
+
return data
|
|
269
|
+
|
|
270
|
+
def read_data_import_order_txt_s3(s3_uri: str, session, exclude_nodes: list = None) -> list:
|
|
271
|
+
"""
|
|
272
|
+
Read a DataImportOrder.txt file from S3 and return node order as a list, optionally excluding some nodes.
|
|
273
|
+
|
|
274
|
+
Args:
|
|
275
|
+
s3_uri (str): S3 URI to the DataImportOrder.txt file.
|
|
276
|
+
session (boto3.Session): Boto3 session.
|
|
277
|
+
exclude_nodes (list, optional): Node names to exclude from result.
|
|
278
|
+
|
|
279
|
+
Returns:
|
|
280
|
+
list: Node names (order as listed in file), optionally excluding nodes in exclude_nodes.
|
|
281
|
+
|
|
282
|
+
Raises:
|
|
283
|
+
ValueError: If the provided S3 URI does not point to DataImportOrder.txt.
|
|
284
|
+
"""
|
|
285
|
+
filename = s3_uri.split("/")[-1]
|
|
286
|
+
if 'DataImportOrder.txt' not in filename:
|
|
287
|
+
logger.error("File %s is not a DataImportOrder.txt file", filename)
|
|
288
|
+
raise ValueError(
|
|
289
|
+
f"File {filename} is not a DataImportOrder.txt file"
|
|
290
|
+
)
|
|
291
|
+
logger.info(
|
|
292
|
+
"Reading DataImportOrder.txt from S3 file: %s",
|
|
293
|
+
s3_uri,
|
|
294
|
+
)
|
|
295
|
+
s3 = session.client('s3')
|
|
296
|
+
obj = s3.get_object(
|
|
297
|
+
Bucket=s3_uri.split("/")[2],
|
|
298
|
+
Key="/".join(s3_uri.split("/")[3:]),
|
|
299
|
+
)
|
|
300
|
+
content = obj['Body'].read().decode('utf-8')
|
|
301
|
+
import_order = [
|
|
302
|
+
line.rstrip()
|
|
303
|
+
for line in content.splitlines()
|
|
304
|
+
if line.strip()
|
|
305
|
+
]
|
|
306
|
+
logger.debug("Raw import order from S3 file: %s", import_order)
|
|
307
|
+
if exclude_nodes is not None:
|
|
308
|
+
import_order = [node for node in import_order if node not in exclude_nodes]
|
|
309
|
+
logger.debug(
|
|
310
|
+
"Import order after excluding nodes %s: %s",
|
|
311
|
+
exclude_nodes,
|
|
312
|
+
import_order,
|
|
313
|
+
)
|
|
314
|
+
logger.debug(
|
|
315
|
+
"Final import order from S3 file %s: %s",
|
|
316
|
+
s3_uri,
|
|
317
|
+
import_order,
|
|
318
|
+
)
|
|
319
|
+
return import_order
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def read_data_import_order_txt(file_path: str, exclude_nodes: list) -> list:
|
|
323
|
+
"""
|
|
324
|
+
Read DataImportOrder.txt from local file, optionally excluding some nodes.
|
|
325
|
+
|
|
326
|
+
Args:
|
|
327
|
+
file_path (str): Path to DataImportOrder.txt.
|
|
328
|
+
exclude_nodes (list): Node names to exclude from result.
|
|
329
|
+
|
|
330
|
+
Returns:
|
|
331
|
+
list: Node names, excludes specified nodes, keeps listed order.
|
|
332
|
+
|
|
333
|
+
Raises:
|
|
334
|
+
FileNotFoundError: If the file is not found.
|
|
335
|
+
"""
|
|
336
|
+
try:
|
|
337
|
+
logger.info(
|
|
338
|
+
"Reading DataImportOrder.txt from local file: %s",
|
|
339
|
+
file_path,
|
|
340
|
+
)
|
|
341
|
+
with open(file_path, "r", encoding="utf-8") as f:
|
|
342
|
+
import_order = [line.rstrip() for line in f if line.strip()]
|
|
343
|
+
logger.debug("Raw import order from file: %s", import_order)
|
|
344
|
+
if exclude_nodes is not None:
|
|
345
|
+
import_order = [
|
|
346
|
+
node for node in import_order if node not in exclude_nodes
|
|
347
|
+
]
|
|
348
|
+
logger.debug(
|
|
349
|
+
"Import order after excluding nodes %s: %s",
|
|
350
|
+
exclude_nodes,
|
|
351
|
+
import_order,
|
|
352
|
+
)
|
|
353
|
+
logger.debug("Final import order from %s: %s", file_path, import_order)
|
|
354
|
+
return import_order
|
|
355
|
+
except FileNotFoundError:
|
|
356
|
+
logger.error("Error: DataImportOrder.txt not found in %s", file_path)
|
|
357
|
+
return []
|
|
358
|
+
|
|
359
|
+
def split_json_objects(json_list, max_size_kb=50, print_results=False) -> list:
|
|
360
|
+
"""
|
|
361
|
+
Split a list of JSON-serializable objects into size-limited chunks.
|
|
362
|
+
|
|
363
|
+
Each chunk/list, when JSON-serialized, will not exceed max_size_kb kilobytes.
|
|
364
|
+
|
|
365
|
+
Args:
|
|
366
|
+
json_list (list): List of JSON serializable objects.
|
|
367
|
+
max_size_kb (int, optional): Max chunk size in KB. Default: 50.
|
|
368
|
+
print_results (bool, optional): If True, info log the size/count per chunk. Default: False.
|
|
369
|
+
|
|
370
|
+
Returns:
|
|
371
|
+
list: List of lists. Each sublist size (JSON-serialized) <= max_size_kb.
|
|
372
|
+
"""
|
|
373
|
+
logger.info(
|
|
374
|
+
"Splitting JSON objects into max %s KB chunks. Total items: %s",
|
|
375
|
+
max_size_kb,
|
|
376
|
+
len(json_list),
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
def get_size_in_kb(obj):
|
|
380
|
+
"""
|
|
381
|
+
Get the size in kilobytes of the JSON-serialized object.
|
|
382
|
+
|
|
383
|
+
Args:
|
|
384
|
+
obj: JSON-serializable object.
|
|
385
|
+
|
|
386
|
+
Returns:
|
|
387
|
+
float: Size of the object in kilobytes.
|
|
388
|
+
"""
|
|
389
|
+
size_kb = sys.getsizeof(json.dumps(obj)) / 1024
|
|
390
|
+
logger.debug("Calculated size: %.2f KB", size_kb)
|
|
391
|
+
return size_kb
|
|
392
|
+
|
|
393
|
+
def split_list(items):
|
|
394
|
+
"""
|
|
395
|
+
Recursively split the list so each chunk fits within max_size_kb.
|
|
396
|
+
|
|
397
|
+
Args:
|
|
398
|
+
json_list (list): List to split.
|
|
399
|
+
|
|
400
|
+
Returns:
|
|
401
|
+
list: List of sublists.
|
|
402
|
+
"""
|
|
403
|
+
if get_size_in_kb(items) <= max_size_kb:
|
|
404
|
+
logger.debug(
|
|
405
|
+
"Split length %s is within max size %s KB.",
|
|
406
|
+
len(items),
|
|
407
|
+
max_size_kb,
|
|
408
|
+
)
|
|
409
|
+
return [items]
|
|
410
|
+
mid = len(items) // 2
|
|
411
|
+
left_list = items[:mid]
|
|
412
|
+
right_list = items[mid:]
|
|
413
|
+
logger.debug(
|
|
414
|
+
"Splitting list at index %s: left %s, right %s",
|
|
415
|
+
mid,
|
|
416
|
+
len(left_list),
|
|
417
|
+
len(right_list),
|
|
418
|
+
)
|
|
419
|
+
return split_list(left_list) + split_list(right_list)
|
|
420
|
+
|
|
421
|
+
split_lists = split_list(json_list)
|
|
422
|
+
if print_results:
|
|
423
|
+
for i, lst in enumerate(split_lists):
|
|
424
|
+
logger.info(
|
|
425
|
+
"List %s size: %.2f KB, contains %s objects",
|
|
426
|
+
i + 1,
|
|
427
|
+
get_size_in_kb(lst),
|
|
428
|
+
len(lst),
|
|
429
|
+
)
|
|
430
|
+
logger.debug("Total splits: %s", len(split_lists))
|
|
431
|
+
return split_lists
|
|
432
|
+
|
|
433
|
+
def get_gen3_api_key_aws_secret(secret_name: str, region_name: str, session) -> dict:
|
|
434
|
+
"""
|
|
435
|
+
Retrieve a Gen3 API key stored as a secret in AWS Secrets Manager and parse it as a dict.
|
|
436
|
+
|
|
437
|
+
Args:
|
|
438
|
+
secret_name (str): Name of the AWS secret.
|
|
439
|
+
region_name (str): AWS region where the secret is located.
|
|
440
|
+
session (boto3.Session): Boto3 session.
|
|
441
|
+
|
|
442
|
+
Returns:
|
|
443
|
+
dict: Parsed Gen3 API key.
|
|
444
|
+
|
|
445
|
+
Raises:
|
|
446
|
+
Exception: On failure to retrieve or parse the secret.
|
|
447
|
+
"""
|
|
448
|
+
logger.info(
|
|
449
|
+
"Retrieving Gen3 API key from AWS Secrets Manager: "
|
|
450
|
+
"secret_name=%s, region=%s",
|
|
451
|
+
secret_name,
|
|
452
|
+
region_name,
|
|
453
|
+
)
|
|
454
|
+
client = session.client(
|
|
455
|
+
service_name='secretsmanager',
|
|
456
|
+
region_name=region_name,
|
|
457
|
+
)
|
|
458
|
+
try:
|
|
459
|
+
get_secret_value_response = client.get_secret_value(
|
|
460
|
+
SecretId=secret_name,
|
|
461
|
+
)
|
|
462
|
+
except (BotoCoreError, ClientError) as e:
|
|
463
|
+
logger.error("Error getting secret value from AWS Secrets Manager: %s", e)
|
|
464
|
+
raise
|
|
465
|
+
|
|
466
|
+
secret = get_secret_value_response['SecretString']
|
|
467
|
+
|
|
468
|
+
try:
|
|
469
|
+
secret = json.loads(secret)
|
|
470
|
+
api_key = secret
|
|
471
|
+
logger.debug("Retrieved Gen3 API key from secret %s", secret_name)
|
|
472
|
+
return api_key
|
|
473
|
+
except (json.JSONDecodeError, TypeError) as e:
|
|
474
|
+
logger.error("Error parsing Gen3 API key from AWS Secrets Manager: %s", e)
|
|
475
|
+
raise
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def infer_api_endpoint_from_jwt(
|
|
479
|
+
jwt_token: str,
|
|
480
|
+
api_version: str = 'v0',
|
|
481
|
+
) -> str:
|
|
482
|
+
"""
|
|
483
|
+
Extract the API endpoint URL from a JSON Web Token (JWT) credential.
|
|
484
|
+
|
|
485
|
+
Args:
|
|
486
|
+
jwt_token (str): The JSON Web Token (JWT) credential.
|
|
487
|
+
api_version (str, optional): API version to append to the URL
|
|
488
|
+
(default: "v0").
|
|
489
|
+
|
|
490
|
+
Returns:
|
|
491
|
+
str: The extracted API endpoint URL.
|
|
492
|
+
"""
|
|
493
|
+
logger.info("Decoding JWT to extract API URL.")
|
|
494
|
+
url = jwt.decode(
|
|
495
|
+
jwt_token,
|
|
496
|
+
options={"verify_signature": False},
|
|
497
|
+
).get('iss', '')
|
|
498
|
+
if '/user' in url:
|
|
499
|
+
url = url.split('/user')[0]
|
|
500
|
+
url = f"{url}/api/{api_version}"
|
|
501
|
+
logger.info("Extracted API URL from JWT: %s", url)
|
|
502
|
+
return url
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def create_gen3_submission_class(api_key: dict):
|
|
506
|
+
"""
|
|
507
|
+
Create and authenticate a Gen3Submission client using an
|
|
508
|
+
API key dictionary.
|
|
509
|
+
|
|
510
|
+
The API endpoint is inferred from the JWT token embedded
|
|
511
|
+
in the API key.
|
|
512
|
+
|
|
513
|
+
Args:
|
|
514
|
+
api_key (dict): The Gen3 API key as a Python dict
|
|
515
|
+
(must contain 'api_key' field).
|
|
516
|
+
|
|
517
|
+
Returns:
|
|
518
|
+
Gen3Submission: An authenticated Gen3Submission object.
|
|
519
|
+
"""
|
|
520
|
+
logger.debug("Extracting JWT token from API key dict.")
|
|
521
|
+
jwt_token = api_key['api_key']
|
|
522
|
+
logger.info("Inferring API endpoint from JWT token.")
|
|
523
|
+
api_endpoint = infer_api_endpoint_from_jwt(jwt_token)
|
|
524
|
+
logger.debug("Inferred API endpoint: %s", api_endpoint)
|
|
525
|
+
logger.info(
|
|
526
|
+
"Creating Gen3Submission class for endpoint: %s",
|
|
527
|
+
api_endpoint,
|
|
528
|
+
)
|
|
529
|
+
auth = Gen3Auth(refresh_token=api_key)
|
|
530
|
+
submit = Gen3Submission(endpoint=api_endpoint, auth_provider=auth)
|
|
531
|
+
return submit
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
class MetadataSubmitter:
|
|
535
|
+
def __init__(
|
|
536
|
+
self,
|
|
537
|
+
metadata_file_list: list,
|
|
538
|
+
api_key: dict,
|
|
539
|
+
project_id: str,
|
|
540
|
+
data_import_order_path: str,
|
|
541
|
+
database: str,
|
|
542
|
+
table: str,
|
|
543
|
+
athena_s3_output: str,
|
|
544
|
+
workgroup: str = "primary",
|
|
545
|
+
table_location: str = None,
|
|
546
|
+
program_id: str = "program1",
|
|
547
|
+
max_size_kb: int = 100,
|
|
548
|
+
exclude_nodes: Optional[List[str]] = None,
|
|
549
|
+
max_retries: int = 3,
|
|
550
|
+
aws_profile: str = None,
|
|
551
|
+
aws_region: str = None,
|
|
552
|
+
partition_cols: Optional[List[str]] = None,
|
|
553
|
+
upload_to_database: bool = True
|
|
554
|
+
):
|
|
555
|
+
"""
|
|
556
|
+
Initialise a MetadataSubmitter for submitting a set of metadata JSON
|
|
557
|
+
files to a Gen3 data commons endpoint, in order.
|
|
558
|
+
|
|
559
|
+
**Workflow Overview:**
|
|
560
|
+
1. **Node Traversal:** The submitter iterates through
|
|
561
|
+
each node defined in the `data_import_order` list.
|
|
562
|
+
2. **File Resolution:** For each node name, it locates the corresponding JSON file
|
|
563
|
+
(e.g., `node.json`) from the provided file list.
|
|
564
|
+
3. **Chunking:** The JSON file is read and split into manageable chunks based on size.
|
|
565
|
+
4. **Submission:** Each chunk is submitted to the Gen3 Sheepdog API via `gen3.submission`.
|
|
566
|
+
5. **Response Handling:** The API response, which includes the `submission_id` for
|
|
567
|
+
the records, is captured.
|
|
568
|
+
6. **Persistence:** The response data is flattened, converted into a DataFrame, and
|
|
569
|
+
written to Parquet files in S3. These records are also registered in a specific
|
|
570
|
+
upload table within the configured database for audit and tracking.
|
|
571
|
+
|
|
572
|
+
Args:
|
|
573
|
+
metadata_file_list (list): List of local file paths or S3 URIs to
|
|
574
|
+
metadata .json files, one per node type.
|
|
575
|
+
api_key (dict): Gen3 API key as a parsed dictionary.
|
|
576
|
+
project_id (str): Gen3 project ID to submit data to (e.g., "internal-project").
|
|
577
|
+
data_import_order_path (str): Path or S3 URI to DataImportOrder.txt
|
|
578
|
+
specifying node submission order.
|
|
579
|
+
database (str): Database name for storing the metadata upload.
|
|
580
|
+
Example: "etl_test_dataops_metadata_db"
|
|
581
|
+
table (str): Table name for storing the metadata upload.
|
|
582
|
+
Example: "metadata_upload"
|
|
583
|
+
athena_s3_output (str): S3 path for Athena query
|
|
584
|
+
results output.
|
|
585
|
+
workgroup (str, optional): Athena workgroup to use
|
|
586
|
+
(default: "primary").
|
|
587
|
+
table_location (str, optional): S3 location for the Iceberg table.
|
|
588
|
+
If None, uses the default location for the database/table.
|
|
589
|
+
program_id (str, optional): Gen3 program ID (default: "program1").
|
|
590
|
+
max_size_kb (int, optional): Maximum size per submission chunk,
|
|
591
|
+
in KB (default: 100).
|
|
592
|
+
exclude_nodes (list, optional): List of node names to skip during
|
|
593
|
+
submission. Defaults to ["project", "program", "acknowledgement", "publication"].
|
|
594
|
+
max_retries (int, optional): Maximum number of retry attempts per
|
|
595
|
+
node chunk (default: 3).
|
|
596
|
+
aws_profile (str, optional): AWS CLI named profile to use for boto3
|
|
597
|
+
session (default: None).
|
|
598
|
+
aws_region (str, optional): AWS region to use for boto3
|
|
599
|
+
session (default: None).
|
|
600
|
+
partition_cols (list, optional): List of column names to partition the parquet table by.
|
|
601
|
+
Defaults to ["upload_datetime"].
|
|
602
|
+
upload_to_database (bool, optional): Whether to upload
|
|
603
|
+
the metadata to a database. Defaults to True.
|
|
604
|
+
Configured via database, table, and
|
|
605
|
+
athena_s3_output.
|
|
606
|
+
"""
|
|
607
|
+
self.metadata_file_list = metadata_file_list
|
|
608
|
+
self.api_key = api_key
|
|
609
|
+
self.project_id = project_id
|
|
610
|
+
self.data_import_order_path = data_import_order_path
|
|
611
|
+
self.database = database
|
|
612
|
+
self.table = table
|
|
613
|
+
self.athena_s3_output = athena_s3_output
|
|
614
|
+
self.workgroup = workgroup
|
|
615
|
+
self.table_location = table_location
|
|
616
|
+
self.program_id = program_id
|
|
617
|
+
self.max_size_kb = max_size_kb
|
|
618
|
+
self.exclude_nodes = exclude_nodes or [
|
|
619
|
+
"project",
|
|
620
|
+
"program",
|
|
621
|
+
"acknowledgement",
|
|
622
|
+
"publication",
|
|
623
|
+
]
|
|
624
|
+
self.max_retries = max_retries
|
|
625
|
+
self.submission_results = []
|
|
626
|
+
self.aws_profile = aws_profile
|
|
627
|
+
self.aws_region = aws_region
|
|
628
|
+
self.partition_cols = partition_cols or ["upload_datetime"]
|
|
629
|
+
self.upload_to_database = upload_to_database
|
|
630
|
+
self.boto3_session = None
|
|
631
|
+
if self.upload_to_database:
|
|
632
|
+
self.boto3_session = self._create_boto3_session()
|
|
633
|
+
logger.info("MetadataSubmitter initialised.")
|
|
634
|
+
|
|
635
|
+
def _create_gen3_submission_class(self):
|
|
636
|
+
"""Helper to instantiate the Gen3Submission class using the provided API key."""
|
|
637
|
+
return create_gen3_submission_class(self.api_key)
|
|
638
|
+
|
|
639
|
+
def _create_boto3_session(self):
|
|
640
|
+
"""Helper to create a boto3 session using the provided AWS profile."""
|
|
641
|
+
return create_boto3_session(
|
|
642
|
+
self.aws_profile, self.aws_region
|
|
643
|
+
)
|
|
644
|
+
|
|
645
|
+
def _flatten_submission_results(self, submission_results: List[Dict]) -> List[Dict]:
|
|
646
|
+
"""
|
|
647
|
+
Flattens a list of Gen3 submission result dictionaries into a single
|
|
648
|
+
list of entity dictionaries.
|
|
649
|
+
|
|
650
|
+
For each submission result, this function processes its entities (if any),
|
|
651
|
+
extracting the 'project_id' and 'submitter_id' from the 'unique_keys'
|
|
652
|
+
field (if present) into the top-level entity dictionary for easy access.
|
|
653
|
+
|
|
654
|
+
Any submission result that does not have a code of 200 or lacks entities
|
|
655
|
+
is skipped, and a warning is logged.
|
|
656
|
+
|
|
657
|
+
Args:
|
|
658
|
+
submission_results (List[Dict]):
|
|
659
|
+
A list of Gen3 submission result dictionaries, each containing at
|
|
660
|
+
least a "code" and "entities" entry.
|
|
661
|
+
|
|
662
|
+
Returns:
|
|
663
|
+
List[Dict]:
|
|
664
|
+
A flat list, where each element is an entity dictionary
|
|
665
|
+
(with keys 'project_id' and 'submitter_id' added if available).
|
|
666
|
+
"""
|
|
667
|
+
flat_list_dict = []
|
|
668
|
+
total = len(submission_results)
|
|
669
|
+
logger.info("Flattening %s submission result(s)...", total)
|
|
670
|
+
|
|
671
|
+
for idx, obj in enumerate(submission_results, 1):
|
|
672
|
+
transaction_id = obj.get("transaction_id")
|
|
673
|
+
code = obj.get("code")
|
|
674
|
+
if code != 200:
|
|
675
|
+
logger.warning(
|
|
676
|
+
"Skipping submission result at index %s (code=%s)",
|
|
677
|
+
idx - 1,
|
|
678
|
+
code,
|
|
679
|
+
)
|
|
680
|
+
continue
|
|
681
|
+
|
|
682
|
+
entities = obj.get("entities")
|
|
683
|
+
|
|
684
|
+
if entities is None:
|
|
685
|
+
logger.warning("No entities found in submission result at index %s", idx - 1)
|
|
686
|
+
continue
|
|
687
|
+
|
|
688
|
+
logger.info(
|
|
689
|
+
"Processing submission result %s of %s, %s entities",
|
|
690
|
+
idx,
|
|
691
|
+
total,
|
|
692
|
+
len(entities),
|
|
693
|
+
)
|
|
694
|
+
|
|
695
|
+
for entity in entities:
|
|
696
|
+
unique_keys = entity.get("unique_keys", [{}])
|
|
697
|
+
if unique_keys and isinstance(unique_keys, list):
|
|
698
|
+
keys = unique_keys[0]
|
|
699
|
+
entity["project_id"] = keys.get("project_id")
|
|
700
|
+
entity["submitter_id"] = keys.get("submitter_id")
|
|
701
|
+
entity["transaction_id"] = transaction_id
|
|
702
|
+
entity["file_path"] = obj.get("file_path", '')
|
|
703
|
+
flat_list_dict.append(entity)
|
|
704
|
+
|
|
705
|
+
# renaming cols
|
|
706
|
+
for entity in flat_list_dict:
|
|
707
|
+
entity["gen3_guid"] = entity.pop("id", None)
|
|
708
|
+
entity["node"] = entity.pop("type", None)
|
|
709
|
+
|
|
710
|
+
logger.info("Finished flattening. Total entities: %s", len(flat_list_dict))
|
|
711
|
+
return flat_list_dict
|
|
712
|
+
|
|
713
|
+
def _find_version_from_path(self, path: str) -> Optional[str]:
|
|
714
|
+
"""
|
|
715
|
+
Extracts a semantic version string (e.g., '1.0.0' or 'v1.0.0') from a file path.
|
|
716
|
+
|
|
717
|
+
Args:
|
|
718
|
+
path (str): The file path to inspect.
|
|
719
|
+
|
|
720
|
+
Returns:
|
|
721
|
+
Optional[str]: The extracted version string if found, otherwise None.
|
|
722
|
+
"""
|
|
723
|
+
version_pattern = re.compile(r"^v?(\d+\.\d+\.\d+)$")
|
|
724
|
+
found_versions = []
|
|
725
|
+
|
|
726
|
+
for segment in path.split('/'):
|
|
727
|
+
match = version_pattern.match(segment)
|
|
728
|
+
if match:
|
|
729
|
+
found_versions.append(match.group(1))
|
|
730
|
+
|
|
731
|
+
if not found_versions:
|
|
732
|
+
return None
|
|
733
|
+
|
|
734
|
+
if len(found_versions) > 1:
|
|
735
|
+
logger.warning("more than one match found in path for version string")
|
|
736
|
+
|
|
737
|
+
return found_versions[-1]
|
|
738
|
+
|
|
739
|
+
def _collect_versions_from_metadata_file_list(self) -> str:
|
|
740
|
+
"""
|
|
741
|
+
Extract and validate version information from the internal list of metadata
|
|
742
|
+
file paths (self.metadata_file_list).
|
|
743
|
+
|
|
744
|
+
Returns:
|
|
745
|
+
str: The single version found in the file list.
|
|
746
|
+
|
|
747
|
+
Raises:
|
|
748
|
+
ValueError: If more than one version is found across the files,
|
|
749
|
+
or if no version is found at all.
|
|
750
|
+
"""
|
|
751
|
+
versions = []
|
|
752
|
+
for file_path in self.metadata_file_list:
|
|
753
|
+
version = self._find_version_from_path(file_path)
|
|
754
|
+
if version:
|
|
755
|
+
versions.append(version)
|
|
756
|
+
versions = list(set(versions))
|
|
757
|
+
if len(versions) > 1:
|
|
758
|
+
logger.error(
|
|
759
|
+
"more than one version found in metadata file list: %s",
|
|
760
|
+
self.metadata_file_list,
|
|
761
|
+
)
|
|
762
|
+
raise ValueError(
|
|
763
|
+
"More than one version found in metadata file list: %s"
|
|
764
|
+
% self.metadata_file_list
|
|
765
|
+
)
|
|
766
|
+
if not versions:
|
|
767
|
+
raise ValueError(
|
|
768
|
+
"No version found in metadata file list: %s" % self.metadata_file_list
|
|
769
|
+
)
|
|
770
|
+
return versions[0]
|
|
771
|
+
|
|
772
|
+
def _upload_submission_results(self, submission_results: list):
|
|
773
|
+
"""
|
|
774
|
+
Uploads the submission results to S3 and a Parquet table.
|
|
775
|
+
|
|
776
|
+
This function performs the final step of the pipeline:
|
|
777
|
+
1. Flattens the submission response structure.
|
|
778
|
+
2. Prepares a DataFrame with metadata (upload_id, datetime, version).
|
|
779
|
+
3. Writes the DataFrame to Parquet files in S3 and registers them in the
|
|
780
|
+
database configured via `self.database` and `self.table`.
|
|
781
|
+
|
|
782
|
+
**Retry Mechanism:**
|
|
783
|
+
Uses the `tenacity` library to retry the upload if it fails.
|
|
784
|
+
- Stop: After `self.max_retries` attempts.
|
|
785
|
+
- Wait: Exponential backoff starting at 1s, doubling up to 10s.
|
|
786
|
+
|
|
787
|
+
Args:
|
|
788
|
+
submission_results (list): List of submission results to upload.
|
|
789
|
+
|
|
790
|
+
Configuration used (from __init__):
|
|
791
|
+
database (str): e.g. "etl_test_dataops_metadata_db"
|
|
792
|
+
table (str): e.g. "metadata_upload"
|
|
793
|
+
athena_s3_output (str): S3 path for Athena query results output.
|
|
794
|
+
workgroup (str): Athena workgroup to use.
|
|
795
|
+
table_location (str): S3 location for the Iceberg table.
|
|
796
|
+
"""
|
|
797
|
+
|
|
798
|
+
@retry(
|
|
799
|
+
stop=stop_after_attempt(self.max_retries),
|
|
800
|
+
wait=wait_exponential(multiplier=1, max=10)
|
|
801
|
+
)
|
|
802
|
+
def inner_upload():
|
|
803
|
+
logger.debug("Collecting version from metadata file list.")
|
|
804
|
+
version = self._collect_versions_from_metadata_file_list()
|
|
805
|
+
logger.debug("Extracted version: %s", version)
|
|
806
|
+
|
|
807
|
+
logger.debug("Inferring API endpoint from JWT.")
|
|
808
|
+
api_endpoint = infer_api_endpoint_from_jwt(self.api_key['api_key'])
|
|
809
|
+
logger.debug("Using API endpoint: %s", api_endpoint)
|
|
810
|
+
|
|
811
|
+
upload_datetime = datetime.now().isoformat()
|
|
812
|
+
upload_id = str(uuid.uuid4())
|
|
813
|
+
logger.debug("Upload datetime: %s", upload_datetime)
|
|
814
|
+
logger.debug("Generated upload ID: %s", upload_id)
|
|
815
|
+
|
|
816
|
+
logger.debug("Flattening submission results for upload.")
|
|
817
|
+
flattened_results = self._flatten_submission_results(submission_results)
|
|
818
|
+
logger.debug(
|
|
819
|
+
"Flattened %s submission result entries.",
|
|
820
|
+
len(flattened_results),
|
|
821
|
+
)
|
|
822
|
+
|
|
823
|
+
logger.debug("Converting flattened results to DataFrame.")
|
|
824
|
+
flattened_results_df = pd.DataFrame(flattened_results)
|
|
825
|
+
flattened_results_df['upload_datetime'] = upload_datetime
|
|
826
|
+
flattened_results_df['upload_id'] = upload_id
|
|
827
|
+
flattened_results_df['api_endpoint'] = api_endpoint
|
|
828
|
+
flattened_results_df['version'] = version
|
|
829
|
+
|
|
830
|
+
logger.info(
|
|
831
|
+
"Writing DataFrame to Iceberg table: "
|
|
832
|
+
"database=%s, table=%s, athena_s3_output=%s",
|
|
833
|
+
self.database,
|
|
834
|
+
self.table,
|
|
835
|
+
self.athena_s3_output,
|
|
836
|
+
)
|
|
837
|
+
write_iceberg_to_db(
|
|
838
|
+
df=flattened_results_df,
|
|
839
|
+
database=self.database,
|
|
840
|
+
table=self.table,
|
|
841
|
+
athena_s3_output=self.athena_s3_output,
|
|
842
|
+
workgroup=self.workgroup,
|
|
843
|
+
table_location=self.table_location,
|
|
844
|
+
boto3_session=self.boto3_session,
|
|
845
|
+
)
|
|
846
|
+
logger.info(
|
|
847
|
+
"\033[94m[SUCCESS]\033[0m Metadata submission results upload complete. "
|
|
848
|
+
"Uploaded to database=%s, table=%s.",
|
|
849
|
+
self.database,
|
|
850
|
+
self.table,
|
|
851
|
+
)
|
|
852
|
+
|
|
853
|
+
# Execute the decorated inner function
|
|
854
|
+
try:
|
|
855
|
+
inner_upload()
|
|
856
|
+
except Exception as e:
|
|
857
|
+
logger.critical("Failed to upload submission results after %s attempts.", self.max_retries)
|
|
858
|
+
raise e
|
|
859
|
+
|
|
860
|
+
def _submit_data_chunks(
|
|
861
|
+
self,
|
|
862
|
+
split_json_list: list,
|
|
863
|
+
node: str,
|
|
864
|
+
gen3_submitter,
|
|
865
|
+
file_path: str,
|
|
866
|
+
upload_to_database: bool = True
|
|
867
|
+
) -> List[Dict]:
|
|
868
|
+
"""
|
|
869
|
+
Submit each chunk of data (in split_json_list) for a given node to Gen3,
|
|
870
|
+
using retry logic and logging on failures.
|
|
871
|
+
|
|
872
|
+
Upon completion of each chunk (success or failure), the response is uploaded
|
|
873
|
+
to the configured S3 Parquet table using `_upload_submission_results`.
|
|
874
|
+
|
|
875
|
+
Args:
|
|
876
|
+
split_json_list (list): List of JSON-serializable chunked data to
|
|
877
|
+
submit.
|
|
878
|
+
node (str): Name of the data node being submitted (e.g., "program").
|
|
879
|
+
gen3_submitter: A Gen3Submission instance for making submissions.
|
|
880
|
+
file_path (str): Path of the file that was submitted.
|
|
881
|
+
Used only for data capture in the result logs.
|
|
882
|
+
|
|
883
|
+
Returns:
|
|
884
|
+
List[Dict]: List of response dictionaries for each submitted chunk.
|
|
885
|
+
|
|
886
|
+
Raises:
|
|
887
|
+
RuntimeError: If submission fails after all retry attempts for any chunk.
|
|
888
|
+
"""
|
|
889
|
+
n_json_data = len(split_json_list)
|
|
890
|
+
|
|
891
|
+
for index, jsn in enumerate(split_json_list):
|
|
892
|
+
# Holds results for the current chunk
|
|
893
|
+
current_chunk_response: List[Dict[str, Any]] = []
|
|
894
|
+
progress_str = f"{index + 1}/{n_json_data}"
|
|
895
|
+
|
|
896
|
+
submission_success = False
|
|
897
|
+
last_exception: Optional[Exception] = None
|
|
898
|
+
|
|
899
|
+
attempt = 0
|
|
900
|
+
while attempt <= self.max_retries:
|
|
901
|
+
try:
|
|
902
|
+
if attempt == 0:
|
|
903
|
+
logger.info(
|
|
904
|
+
"[SUBMIT] | Project: %-10s | Node: %-12s | "
|
|
905
|
+
"Split: %-5s",
|
|
906
|
+
self.project_id,
|
|
907
|
+
node,
|
|
908
|
+
progress_str,
|
|
909
|
+
)
|
|
910
|
+
else:
|
|
911
|
+
logger.warning(
|
|
912
|
+
"[RETRY] | Project: %-10s | Node: %-12s | "
|
|
913
|
+
"Split: %-5s | "
|
|
914
|
+
"Attempt: %s/%s",
|
|
915
|
+
self.project_id,
|
|
916
|
+
node,
|
|
917
|
+
progress_str,
|
|
918
|
+
attempt,
|
|
919
|
+
self.max_retries,
|
|
920
|
+
)
|
|
921
|
+
|
|
922
|
+
res = gen3_submitter.submit_record(self.program_id, self.project_id, jsn)
|
|
923
|
+
res.update({"file_path": file_path})
|
|
924
|
+
current_chunk_response.append(res)
|
|
925
|
+
submission_success = True
|
|
926
|
+
logger.info(
|
|
927
|
+
"\033[92m[SUCCESS]\033[0m | Project: %-10s | "
|
|
928
|
+
"Node: %-12s | Split: %-5s",
|
|
929
|
+
self.project_id,
|
|
930
|
+
node,
|
|
931
|
+
progress_str,
|
|
932
|
+
)
|
|
933
|
+
break # Success
|
|
934
|
+
|
|
935
|
+
except (
|
|
936
|
+
requests.exceptions.RequestException,
|
|
937
|
+
ValueError,
|
|
938
|
+
TypeError,
|
|
939
|
+
) as e:
|
|
940
|
+
last_exception = e
|
|
941
|
+
logger.error(
|
|
942
|
+
"Error submitting chunk %s for node '%s': %s",
|
|
943
|
+
progress_str,
|
|
944
|
+
node,
|
|
945
|
+
e,
|
|
946
|
+
)
|
|
947
|
+
|
|
948
|
+
if attempt < self.max_retries:
|
|
949
|
+
time.sleep(0.2)
|
|
950
|
+
else:
|
|
951
|
+
logger.critical(
|
|
952
|
+
"\033[91m[FAILED]\033[0m | Project: %-10s | "
|
|
953
|
+
"Node: %-12s | Split: %-5s | Error: %s",
|
|
954
|
+
self.project_id,
|
|
955
|
+
node,
|
|
956
|
+
progress_str,
|
|
957
|
+
e,
|
|
958
|
+
)
|
|
959
|
+
attempt += 1
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
if upload_to_database:
|
|
963
|
+
# Also submitting data chunk response info to s3 and parquet table
|
|
964
|
+
logger.info("Submitting data chunk response info to S3 and Parquet table.")
|
|
965
|
+
self._upload_submission_results(submission_results=current_chunk_response)
|
|
966
|
+
|
|
967
|
+
if not submission_success:
|
|
968
|
+
# After retries, still failed
|
|
969
|
+
raise RuntimeError(
|
|
970
|
+
(
|
|
971
|
+
"Failed to submit chunk %s for node '%s' after %s attempts. "
|
|
972
|
+
"Last error: %s"
|
|
973
|
+
)
|
|
974
|
+
% (progress_str, node, self.max_retries + 1, last_exception)
|
|
975
|
+
) from last_exception
|
|
976
|
+
|
|
977
|
+
logger.info("Finished submitting node '%s'.", node)
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def _read_data_import_order(
|
|
981
|
+
self,
|
|
982
|
+
data_import_order_path: str,
|
|
983
|
+
exclude_nodes: List[str],
|
|
984
|
+
boto3_session=None,
|
|
985
|
+
):
|
|
986
|
+
"""Helper to read the data import order from local disk or S3."""
|
|
987
|
+
if is_s3_uri(data_import_order_path):
|
|
988
|
+
session = boto3_session or self.boto3_session
|
|
989
|
+
return read_data_import_order_txt_s3(
|
|
990
|
+
data_import_order_path,
|
|
991
|
+
session,
|
|
992
|
+
exclude_nodes,
|
|
993
|
+
)
|
|
994
|
+
else:
|
|
995
|
+
return read_data_import_order_txt(data_import_order_path, exclude_nodes)
|
|
996
|
+
|
|
997
|
+
def _prepare_json_chunks(self, metadata_file_path: str, max_size_kb: int) -> List[List[Dict]]:
|
|
998
|
+
"""
|
|
999
|
+
Read JSON data from a given file path and split it into chunks,
|
|
1000
|
+
each with a maximum size of ``max_size_kb`` kilobytes.
|
|
1001
|
+
|
|
1002
|
+
Args:
|
|
1003
|
+
metadata_file_path (str): File path (local or S3 URI) to the JSON data.
|
|
1004
|
+
max_size_kb (int): Maximum allowed size (in kilobytes) for each chunk.
|
|
1005
|
+
|
|
1006
|
+
Returns:
|
|
1007
|
+
list: A list of chunks, where each chunk is a list of dictionaries
|
|
1008
|
+
containing JSON data.
|
|
1009
|
+
"""
|
|
1010
|
+
logger.info("Reading metadata json from %s", metadata_file_path)
|
|
1011
|
+
if is_s3_uri(metadata_file_path):
|
|
1012
|
+
session = self.boto3_session
|
|
1013
|
+
data = read_metadata_json_s3(metadata_file_path, session)
|
|
1014
|
+
else:
|
|
1015
|
+
data = read_metadata_json(metadata_file_path)
|
|
1016
|
+
return split_json_objects(data, max_size_kb)
|
|
1017
|
+
|
|
1018
|
+
def _create_file_map(self):
|
|
1019
|
+
"""
|
|
1020
|
+
Generate a mapping from node names to metadata file paths.
|
|
1021
|
+
|
|
1022
|
+
This method infers the node name for each file in `self.metadata_file_list`
|
|
1023
|
+
and returns a dictionary where the keys are node names and the values
|
|
1024
|
+
are the corresponding file paths.
|
|
1025
|
+
|
|
1026
|
+
Returns:
|
|
1027
|
+
dict: Dictionary mapping node names (str) to their associated metadata file paths.
|
|
1028
|
+
"""
|
|
1029
|
+
file_map = {
|
|
1030
|
+
get_node_from_file_path(file_path): file_path
|
|
1031
|
+
for file_path in self.metadata_file_list
|
|
1032
|
+
}
|
|
1033
|
+
return file_map
|
|
1034
|
+
|
|
1035
|
+
def submit_metadata(self, specific_node: Optional[str] = None) -> List[Dict[str, Any]]:
|
|
1036
|
+
"""
|
|
1037
|
+
Submits metadata for each node defined in the data import order, except those in the exclude list.
|
|
1038
|
+
|
|
1039
|
+
Args:
|
|
1040
|
+
specific_node (Optional[str]): If provided, only submits metadata for the specified node.
|
|
1041
|
+
|
|
1042
|
+
**Detailed Process:**
|
|
1043
|
+
1. **Order Resolution:** The function reads the import order to determine the sequence of nodes.
|
|
1044
|
+
2. **File Mapping:** It finds the matching `node.json` file for each node in the order.
|
|
1045
|
+
3. **Chunk & Submit:** For every file, the JSON content is split into chunks and submitted
|
|
1046
|
+
to the Sheepdog API via `gen3.submission`.
|
|
1047
|
+
4. **Audit Logging:** The API response (containing `submission_id`) is flattened and
|
|
1048
|
+
converted to a DataFrame. This is then written to Parquet files in S3 and registered
|
|
1049
|
+
in the configured upload table.
|
|
1050
|
+
|
|
1051
|
+
Returns:
|
|
1052
|
+
List[Dict[str, Any]]: A list of response dictionaries returned from the Gen3 metadata submissions.
|
|
1053
|
+
Each dictionary contains the response from submitting a chunk of metadata for a given node.
|
|
1054
|
+
The keys in the dictionary are "node_name", "response", and "status_code".
|
|
1055
|
+
"""
|
|
1056
|
+
gen3_submitter = self._create_gen3_submission_class()
|
|
1057
|
+
data_import_order = self._read_data_import_order(
|
|
1058
|
+
self.data_import_order_path,
|
|
1059
|
+
self.exclude_nodes,
|
|
1060
|
+
self.boto3_session,
|
|
1061
|
+
)
|
|
1062
|
+
file_map = self._create_file_map()
|
|
1063
|
+
|
|
1064
|
+
logger.info("Starting metadata submission.")
|
|
1065
|
+
|
|
1066
|
+
if specific_node:
|
|
1067
|
+
if specific_node not in data_import_order:
|
|
1068
|
+
raise ValueError(f"Node '{specific_node}' not found in data import order.")
|
|
1069
|
+
data_import_order = [specific_node]
|
|
1070
|
+
|
|
1071
|
+
for node in data_import_order:
|
|
1072
|
+
if node in self.exclude_nodes:
|
|
1073
|
+
logger.info("Skipping node '%s' (in exclude list).", node)
|
|
1074
|
+
continue
|
|
1075
|
+
file_path = file_map.get(node)
|
|
1076
|
+
if not file_path:
|
|
1077
|
+
logger.info("Skipping node '%s' (not present in file list).", node)
|
|
1078
|
+
continue
|
|
1079
|
+
|
|
1080
|
+
logger.info("Processing file '%s' for node '%s'.", file_path, node)
|
|
1081
|
+
logger.info("Splitting JSON data into chunks.")
|
|
1082
|
+
json_chunks = self._prepare_json_chunks(file_path, self.max_size_kb)
|
|
1083
|
+
|
|
1084
|
+
logger.info("Submitting chunks to Gen3.")
|
|
1085
|
+
self._submit_data_chunks(
|
|
1086
|
+
split_json_list=json_chunks,
|
|
1087
|
+
node=node,
|
|
1088
|
+
gen3_submitter=gen3_submitter,
|
|
1089
|
+
file_path=file_path,
|
|
1090
|
+
upload_to_database=self.upload_to_database
|
|
1091
|
+
)
|
|
1092
|
+
|
|
1093
|
+
|