gen3-dataops-toolkit 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- g3dt/__init__.py +0 -0
- g3dt/cli/__init__.py +5 -0
- g3dt/cli/_internal/__init__.py +1 -0
- g3dt/cli/_internal/dispatch.py +428 -0
- g3dt/cli/_internal/registry.py +65 -0
- g3dt/cli/_internal/resolve.py +22 -0
- g3dt/cli/_internal/runner.py +76 -0
- g3dt/cli/_internal/safety.py +110 -0
- g3dt/cli/config_cmds.py +202 -0
- g3dt/cli/delete_cmds.py +101 -0
- g3dt/cli/dict_cmds.py +102 -0
- g3dt/cli/ec2_cmds.py +114 -0
- g3dt/cli/indexd_cmds.py +57 -0
- g3dt/cli/jobs.py +83 -0
- g3dt/cli/k8s.py +54 -0
- g3dt/cli/main.py +110 -0
- g3dt/cli/metadata.py +76 -0
- g3dt/cli/synth.py +206 -0
- g3dt/config.py +393 -0
- g3dt/indexd/__init__.py +0 -0
- g3dt/indexd/indexd_registrar.py +244 -0
- g3dt/ingest/ingest.py +629 -0
- g3dt/resolver.py +163 -0
- g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
- g3dt/services/delete/delete_metadata.sh +153 -0
- g3dt/services/delete/delete_metadata_by_guid.py +338 -0
- g3dt/services/dictionary/deploy_dd.sh +65 -0
- g3dt/services/dictionary/pull_dict.sh +59 -0
- g3dt/services/dictionary/upload_dictionary.py +109 -0
- g3dt/services/indexd/register_indexd.py +240 -0
- g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
- g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
- g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
- g3dt/services/k8s_ops/login_to_pod.sh +110 -0
- g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
- g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
- g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
- g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
- g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
- g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
- g3dt/services/upload/metadata/upload_metadata.py +152 -0
- g3dt/upload/__init__.py +1 -0
- g3dt/upload/metadata_deleter.py +265 -0
- g3dt/upload/metadata_submitter.py +1093 -0
- g3dt/upload/upload_synthdata_s3.py +164 -0
- g3dt/utils/athena_utils.py +834 -0
- g3dt/utils/dbt_utils.py +66 -0
- g3dt/utils/release_writer.py +188 -0
- g3dt/validate/validate.py +609 -0
- gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
- gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
- gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
- gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
g3dt/ingest/ingest.py
ADDED
|
@@ -0,0 +1,629 @@
|
|
|
1
|
+
import awswrangler as wr
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import boto3
|
|
4
|
+
from g3dt.utils.athena_utils import write_iceberg_to_db
|
|
5
|
+
import urllib.parse
|
|
6
|
+
import os
|
|
7
|
+
import uuid
|
|
8
|
+
import hashlib
|
|
9
|
+
from datetime import datetime
|
|
10
|
+
from botocore.exceptions import ClientError
|
|
11
|
+
import logging
|
|
12
|
+
import pytz # Replaced tzlocal with pytz
|
|
13
|
+
import s3fs
|
|
14
|
+
from typing import Dict
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
logger.setLevel(logging.INFO)
|
|
19
|
+
|
|
20
|
+
s3 = boto3.client("s3")
|
|
21
|
+
|
|
22
|
+
# ---------- helpers ----------
|
|
23
|
+
def parse_s3(uri: str):
|
|
24
|
+
p = urllib.parse.urlparse(uri)
|
|
25
|
+
return p.netloc, p.path.lstrip("/")
|
|
26
|
+
|
|
27
|
+
def get_tags(uri: str) -> dict:
|
|
28
|
+
b, k = parse_s3(uri)
|
|
29
|
+
try:
|
|
30
|
+
ts = s3.get_object_tagging(Bucket=b, Key=k)["TagSet"]
|
|
31
|
+
logger.debug(f"Retrieved tags for {uri}: {ts}")
|
|
32
|
+
return {t["Key"]: t["Value"] for t in ts}
|
|
33
|
+
except ClientError as e:
|
|
34
|
+
if e.response["Error"]["Code"] == "NoSuchKey":
|
|
35
|
+
logger.warning(f"NoSuchKey when getting tags for {uri}")
|
|
36
|
+
return {}
|
|
37
|
+
logger.error(f"ClientError getting tags for {uri}: {e}")
|
|
38
|
+
raise
|
|
39
|
+
except Exception as e:
|
|
40
|
+
logger.error(f"Failed to get tags for S3 object {uri}: {e}")
|
|
41
|
+
raise RuntimeError(f"Failed to get tags for S3 object {uri}: {e}")
|
|
42
|
+
|
|
43
|
+
def get_head_meta(uri: str) -> dict:
|
|
44
|
+
b, k = parse_s3(uri)
|
|
45
|
+
try:
|
|
46
|
+
h = s3.head_object(Bucket=b, Key=k)
|
|
47
|
+
logger.debug(f"Head metadata for {uri}: {h}")
|
|
48
|
+
except ClientError as e:
|
|
49
|
+
logger.error(f"Failed to get head metadata for S3 object {uri}: {e}")
|
|
50
|
+
raise RuntimeError(f"Failed to get head metadata for S3 object {uri}: {e}")
|
|
51
|
+
except Exception as e:
|
|
52
|
+
logger.error(f"Unexpected error getting head metadata for S3 object {uri}: {e}")
|
|
53
|
+
raise RuntimeError(f"Unexpected error getting head metadata for S3 object {uri}: {e}")
|
|
54
|
+
|
|
55
|
+
# Consistently use Australia/Melbourne timezone
|
|
56
|
+
aest_tz = pytz.timezone("Australia/Melbourne")
|
|
57
|
+
last_modified_aest = h["LastModified"].astimezone(aest_tz)
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
"ingest_file_etag": (h.get("ETag") or "").strip('"'),
|
|
61
|
+
"ingest_file_size_bytes": str(h.get("ContentLength")),
|
|
62
|
+
"ingest_file_last_modified": last_modified_aest.strftime("%Y-%m-%dT%H:%M:%S%z"),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
def normalise(name: str) -> str:
|
|
66
|
+
n = (name or "col").strip().lower().replace("/", "_").replace(" ", "_")
|
|
67
|
+
allowed = "abcdefghijklmnopqrstuvwxyz0123456789_"
|
|
68
|
+
n = "".join(ch if ch in allowed else "_" for ch in n)
|
|
69
|
+
while "__" in n: n = n.replace("__", "_")
|
|
70
|
+
if not n: n = "col"
|
|
71
|
+
if n[0].isdigit(): n = f"c_{n}"
|
|
72
|
+
return n
|
|
73
|
+
|
|
74
|
+
def read_csv_robust(uri: str) -> pd.DataFrame:
|
|
75
|
+
# delimiter sniff + encoding fallback, keep everything as string
|
|
76
|
+
last_err = None
|
|
77
|
+
for enc in ("utf-8-sig", "cp1252"):
|
|
78
|
+
try:
|
|
79
|
+
logger.debug(f"Attempting to read CSV from {uri} with encoding {enc}")
|
|
80
|
+
df = wr.s3.read_csv(
|
|
81
|
+
path=uri,
|
|
82
|
+
sep=None, engine="python",
|
|
83
|
+
dtype=str,
|
|
84
|
+
keep_default_na=False,
|
|
85
|
+
encoding=enc,
|
|
86
|
+
quoting=0, # QUOTE_MINIMAL
|
|
87
|
+
)
|
|
88
|
+
logger.debug(f"Successfully read CSV from {uri} with encoding {enc}")
|
|
89
|
+
break
|
|
90
|
+
except Exception as e:
|
|
91
|
+
logger.debug(f"Failed to read CSV from {uri} with encoding {enc}: {e}")
|
|
92
|
+
last_err = e
|
|
93
|
+
else:
|
|
94
|
+
logger.error(f"Failed to read CSV from {uri} after trying all encodings: {last_err}")
|
|
95
|
+
raise RuntimeError(f"Failed to read CSV from {uri}: {last_err}")
|
|
96
|
+
# normalise & uniquify headers
|
|
97
|
+
cols = [normalise(c) for c in df.columns]
|
|
98
|
+
seen, out = set(), []
|
|
99
|
+
for c in cols:
|
|
100
|
+
base, i, cand = c, 1, c
|
|
101
|
+
while cand in seen:
|
|
102
|
+
i += 1
|
|
103
|
+
cand = f"{base}_{i}"
|
|
104
|
+
seen.add(cand); out.append(cand)
|
|
105
|
+
df.columns = out
|
|
106
|
+
logger.debug(f"Normalised columns for {uri}: {out}")
|
|
107
|
+
return df.reset_index(drop=True)
|
|
108
|
+
|
|
109
|
+
def read_json_robust(uri: str) -> pd.DataFrame:
|
|
110
|
+
# encoding fallback, keep everything as string
|
|
111
|
+
last_err = None
|
|
112
|
+
for enc in ("utf-8-sig", "cp1252"):
|
|
113
|
+
try:
|
|
114
|
+
logger.debug(f"Attempting to read JSON from {uri} with encoding {enc}")
|
|
115
|
+
# Remove keep_default_na, which is not a valid argument for wr.s3.read_json
|
|
116
|
+
df = wr.s3.read_json(
|
|
117
|
+
path=uri,
|
|
118
|
+
orient="records",
|
|
119
|
+
dtype=str,
|
|
120
|
+
encoding=enc,
|
|
121
|
+
)
|
|
122
|
+
logger.debug(f"Successfully read JSON from {uri} with encoding {enc}")
|
|
123
|
+
break
|
|
124
|
+
except Exception as e:
|
|
125
|
+
logger.debug(f"Failed to read JSON from {uri} with encoding {enc}: {e}")
|
|
126
|
+
last_err = e
|
|
127
|
+
else:
|
|
128
|
+
logger.error(f"Failed to read JSON from {uri} after trying all encodings: {last_err}")
|
|
129
|
+
raise RuntimeError(f"Failed to read JSON from {uri}: {last_err}")
|
|
130
|
+
return df
|
|
131
|
+
|
|
132
|
+
def read_xlsx_robust(s3_uri: str) -> dict[str, pd.DataFrame]:
|
|
133
|
+
"""
|
|
134
|
+
Reads an XLSX file from the given S3 URI using pandas with the openpyxl engine.
|
|
135
|
+
Reads all values as strings and disables default NA values.
|
|
136
|
+
Returns a dictionary of dataframes, with the sheet name as the key.
|
|
137
|
+
If there is only one sheet, the key is set to the file name (minus extension) from s3_uri.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
s3_uri (str): The S3 URI of the XLSX file.
|
|
141
|
+
|
|
142
|
+
Returns:
|
|
143
|
+
dict[str, pd.DataFrame]: Dictionary mapping sheet name (or file name) to DataFrame.
|
|
144
|
+
|
|
145
|
+
Raises:
|
|
146
|
+
RuntimeError: If reading the XLSX file fails,
|
|
147
|
+
ValueError: If the S3 URI is invalid.
|
|
148
|
+
"""
|
|
149
|
+
# Check that s3_uri is a valid S3 URI
|
|
150
|
+
if not isinstance(s3_uri, str) or not s3_uri.startswith("s3://"):
|
|
151
|
+
logger.error(f"Invalid S3 URI: {s3_uri}")
|
|
152
|
+
raise ValueError(f"Invalid S3 URI: {s3_uri}")
|
|
153
|
+
try:
|
|
154
|
+
logger.debug(f"Attempting to read XLSX from {s3_uri}")
|
|
155
|
+
# Read all sheets, always returns dict
|
|
156
|
+
df_dict = pd.read_excel(
|
|
157
|
+
s3_uri,
|
|
158
|
+
sheet_name=None,
|
|
159
|
+
engine="openpyxl",
|
|
160
|
+
dtype=str,
|
|
161
|
+
keep_default_na=False
|
|
162
|
+
)
|
|
163
|
+
logger.debug(f"Successfully read XLSX from {s3_uri}: sheets={list(df_dict.keys())}")
|
|
164
|
+
|
|
165
|
+
if len(df_dict) == 1:
|
|
166
|
+
# Only one sheet, rename the key to the file name (no extension)
|
|
167
|
+
import os
|
|
168
|
+
file_name = os.path.splitext(os.path.basename(s3_uri))[0]
|
|
169
|
+
only_df = next(iter(df_dict.values()))
|
|
170
|
+
logger.debug(f"Only one sheet found, renaming key to file name: {file_name}")
|
|
171
|
+
return {file_name: only_df}
|
|
172
|
+
|
|
173
|
+
return df_dict
|
|
174
|
+
except Exception as e:
|
|
175
|
+
logger.error(f"Failed to read XLSX from {s3_uri}: {e}")
|
|
176
|
+
raise RuntimeError(f"Failed to read XLSX from {s3_uri}: {e}")
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def flatten_xlsx_dict(df_dict: dict[str, pd.DataFrame]) -> pd.DataFrame:
|
|
180
|
+
"""
|
|
181
|
+
Flattens a dictionary of DataFrames (representing Excel sheets) into a single DataFrame.
|
|
182
|
+
|
|
183
|
+
Adds a "sheet_name" column to each DataFrame, indicating the originating sheet,
|
|
184
|
+
then concatenates all DataFrames into one.
|
|
185
|
+
|
|
186
|
+
Args:
|
|
187
|
+
df_dict (dict[str, pd.DataFrame]): Dictionary mapping sheet names to DataFrames.
|
|
188
|
+
|
|
189
|
+
Returns:
|
|
190
|
+
pd.DataFrame: Concatenated DataFrame with an added "sheet_name" column.
|
|
191
|
+
|
|
192
|
+
Raises:
|
|
193
|
+
ValueError: If the provided dictionary is empty.
|
|
194
|
+
RuntimeError: If an error occurs during DataFrame concatenation.
|
|
195
|
+
"""
|
|
196
|
+
if not df_dict:
|
|
197
|
+
logger.error("The df_dict provided to flatten_xlsx_dict is empty.")
|
|
198
|
+
raise ValueError("Input dictionary of DataFrames is empty.")
|
|
199
|
+
|
|
200
|
+
dfs_with_sheet = []
|
|
201
|
+
try:
|
|
202
|
+
for sheet_name, df in df_dict.items():
|
|
203
|
+
df_with_sheet = df.copy()
|
|
204
|
+
df_with_sheet["sheet_name"] = sheet_name
|
|
205
|
+
dfs_with_sheet.append(df_with_sheet)
|
|
206
|
+
return pd.concat(dfs_with_sheet, ignore_index=True)
|
|
207
|
+
except Exception as e:
|
|
208
|
+
logger.error(f"Failed to flatten XLSX dictionary: {e}")
|
|
209
|
+
raise RuntimeError(f"Failed to flatten XLSX dictionary: {e}") from e
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def get_format(uri: str) -> str:
|
|
213
|
+
"""
|
|
214
|
+
Returns the file format (extension) of the given URI, without the leading dot.
|
|
215
|
+
"""
|
|
216
|
+
_, ext = os.path.splitext(uri)
|
|
217
|
+
fmt = ext[1:].lower()
|
|
218
|
+
logger.debug(f"File format for {uri} is {fmt}")
|
|
219
|
+
return fmt
|
|
220
|
+
|
|
221
|
+
def compute_row_hash(row: pd.Series) -> str:
|
|
222
|
+
parts = [f"{k}={row[k] if row[k] is not None else ''}" for k in sorted(row.index)]
|
|
223
|
+
return hashlib.sha256(("||".join(parts)).encode("utf-8")).hexdigest()
|
|
224
|
+
|
|
225
|
+
def sanitize_submission_date(value: str) -> str:
|
|
226
|
+
"""
|
|
227
|
+
Parse a submission_date string and return it in 'YYYY-MM-DD' format.
|
|
228
|
+
Handles various formats including 'YYYY_MM_DD', 'DD MM YYYY', 'DD-MM-YYYY',
|
|
229
|
+
and 'YYYY-MM-DD'. It expects the input to be in one of these formats.
|
|
230
|
+
|
|
231
|
+
Raises a ValueError if the input is empty or cannot be parsed as a date.
|
|
232
|
+
"""
|
|
233
|
+
if not value:
|
|
234
|
+
logger.error("submission_date tag is empty")
|
|
235
|
+
raise ValueError("submission_date tag is empty")
|
|
236
|
+
|
|
237
|
+
# Standardize separators by replacing underscores and spaces with hyphens
|
|
238
|
+
cleaned_value = value.replace('_', '-').replace(' ', '-')
|
|
239
|
+
|
|
240
|
+
# Attempt to parse the date in 'YYYY-MM-DD' format
|
|
241
|
+
try:
|
|
242
|
+
return datetime.strptime(cleaned_value, "%Y-%m-%d").strftime("%Y-%m-%d")
|
|
243
|
+
except ValueError:
|
|
244
|
+
# If the first format fails, try the 'DD-MM-YYYY' format
|
|
245
|
+
try:
|
|
246
|
+
return datetime.strptime(cleaned_value, "%d-%m-%Y").strftime("%Y-%m-%d")
|
|
247
|
+
except ValueError:
|
|
248
|
+
logger.error(f"Unable to parse submission_date format to ISO: {value!r}")
|
|
249
|
+
raise ValueError(f"Unable to parse submission_date format to ISO: {value!r}")
|
|
250
|
+
|
|
251
|
+
def prepare_ingest_metadata(
|
|
252
|
+
df: pd.DataFrame,
|
|
253
|
+
uri: str,
|
|
254
|
+
tags: dict,
|
|
255
|
+
head_meta: dict,
|
|
256
|
+
study_id: str,
|
|
257
|
+
ingest_run_id: str,
|
|
258
|
+
ingest_received_at: str,
|
|
259
|
+
ingest_timezone: str,
|
|
260
|
+
ingest_submission_id: str,
|
|
261
|
+
) -> pd.DataFrame:
|
|
262
|
+
"""
|
|
263
|
+
Attach ingest and tag metadata columns to a DataFrame for a single file.
|
|
264
|
+
|
|
265
|
+
Parameters
|
|
266
|
+
----------
|
|
267
|
+
df : pd.DataFrame
|
|
268
|
+
The DataFrame to annotate.
|
|
269
|
+
uri : str
|
|
270
|
+
S3 URI of the file.
|
|
271
|
+
tags : dict
|
|
272
|
+
S3 object tags.
|
|
273
|
+
head_meta : dict
|
|
274
|
+
S3 object head metadata.
|
|
275
|
+
study_id : str
|
|
276
|
+
Study identifier.
|
|
277
|
+
ingest_run_id : str
|
|
278
|
+
Unique ID for this ingest run.
|
|
279
|
+
ingest_received_at : str
|
|
280
|
+
ISO timestamp for when ingest was received.
|
|
281
|
+
ingest_timezone : str
|
|
282
|
+
Timezone string for ingest.
|
|
283
|
+
ingest_submission_id : str
|
|
284
|
+
Submission ID for this ingest.
|
|
285
|
+
|
|
286
|
+
Returns
|
|
287
|
+
-------
|
|
288
|
+
pd.DataFrame
|
|
289
|
+
Annotated DataFrame.
|
|
290
|
+
"""
|
|
291
|
+
file_name = os.path.basename(urllib.parse.urlparse(uri).path)
|
|
292
|
+
df = df.copy()
|
|
293
|
+
try:
|
|
294
|
+
logger.debug(f"Annotating DataFrame for file {uri} with ingest metadata.")
|
|
295
|
+
df["study_id"] = str(study_id)
|
|
296
|
+
df["submission_date"] = sanitize_submission_date(tags["submission_date"])
|
|
297
|
+
df["ingest_run_id"] = ingest_run_id
|
|
298
|
+
df["ingest_received_at"] = ingest_received_at
|
|
299
|
+
df["ingest_timezone"] = ingest_timezone
|
|
300
|
+
df["ingest_original_file_path"] = uri
|
|
301
|
+
df["ingest_file_name"] = file_name
|
|
302
|
+
df["ingest_submission_id"] = str(ingest_submission_id or "")
|
|
303
|
+
df["ingest_file_etag"] = head_meta["ingest_file_etag"]
|
|
304
|
+
df["ingest_file_size_bytes"] = head_meta["ingest_file_size_bytes"]
|
|
305
|
+
df["ingest_file_last_modified"] = head_meta["ingest_file_last_modified"]
|
|
306
|
+
except Exception as e:
|
|
307
|
+
logger.error(f"Failed to prepare ingest metadata for file {uri}: {e}")
|
|
308
|
+
raise RuntimeError(f"Failed to prepare ingest metadata for file {uri}: {e}")
|
|
309
|
+
|
|
310
|
+
# Add S3 tags as columns (namespaced)
|
|
311
|
+
for k, v in tags.items():
|
|
312
|
+
try:
|
|
313
|
+
df[f"tag_{normalise(k)}"] = str(v)
|
|
314
|
+
except Exception as e:
|
|
315
|
+
logger.error(f"Failed to add tag '{k}' as column for file {uri}: {e}")
|
|
316
|
+
raise RuntimeError(f"Failed to add tag '{k}' as column for file {uri}: {e}")
|
|
317
|
+
|
|
318
|
+
# Compute row-hash over raw columns (exclude ingest_*, tag_*, study_id, submission_date)
|
|
319
|
+
try:
|
|
320
|
+
exclude = {c for c in df.columns if c.startswith("ingest_") or c.startswith("tag_")}
|
|
321
|
+
exclude |= {"study_id", "submission_date"}
|
|
322
|
+
raw_cols = [c for c in df.columns if c not in exclude]
|
|
323
|
+
logger.debug(f"Computing row hash for file {uri} on columns: {raw_cols}")
|
|
324
|
+
df["ingest_row_hash"] = df[raw_cols].apply(compute_row_hash, axis=1)
|
|
325
|
+
except Exception as e:
|
|
326
|
+
logger.error(f"Failed to compute row hash for file {uri}: {e}")
|
|
327
|
+
raise RuntimeError(f"Failed to compute row hash for file {uri}: {e}")
|
|
328
|
+
return df
|
|
329
|
+
|
|
330
|
+
def align_and_combine_frames(frames: list[pd.DataFrame]) -> pd.DataFrame:
|
|
331
|
+
"""
|
|
332
|
+
Align columns across multiple DataFrames and concatenate them.
|
|
333
|
+
|
|
334
|
+
Parameters
|
|
335
|
+
----------
|
|
336
|
+
frames : list of pd.DataFrame
|
|
337
|
+
List of DataFrames to align and combine.
|
|
338
|
+
|
|
339
|
+
Returns
|
|
340
|
+
-------
|
|
341
|
+
pd.DataFrame
|
|
342
|
+
Combined DataFrame with aligned columns.
|
|
343
|
+
"""
|
|
344
|
+
try:
|
|
345
|
+
all_cols = sorted({c for f in frames for c in f.columns})
|
|
346
|
+
logger.debug(f"Aligning and combining {len(frames)} DataFrames with columns: {all_cols}")
|
|
347
|
+
aligned_frames = [f.reindex(columns=all_cols, fill_value="") for f in frames]
|
|
348
|
+
combined = pd.concat(aligned_frames, ignore_index=True)
|
|
349
|
+
logger.debug(f"Combined DataFrame shape: {combined.shape}")
|
|
350
|
+
return combined
|
|
351
|
+
except Exception as e:
|
|
352
|
+
logger.error(f"Failed to align and combine DataFrames: {e}")
|
|
353
|
+
raise RuntimeError(f"Failed to align and combine DataFrames: {e}")
|
|
354
|
+
|
|
355
|
+
def get_ingest_true_files(s3_uri: str, exclude_directories: list[str] = None) -> list[str]:
|
|
356
|
+
"""
|
|
357
|
+
Get a list of files from an S3 URI, excluding directories.
|
|
358
|
+
|
|
359
|
+
Parameters
|
|
360
|
+
----------
|
|
361
|
+
s3_uri : str
|
|
362
|
+
S3 URI to list files from.
|
|
363
|
+
exclude_directories : list[str], optional
|
|
364
|
+
List of directories to exclude. Defaults to None.
|
|
365
|
+
|
|
366
|
+
Returns
|
|
367
|
+
-------
|
|
368
|
+
list[str]
|
|
369
|
+
List of file paths.
|
|
370
|
+
"""
|
|
371
|
+
logger.debug(f"Getting files from {s3_uri}")
|
|
372
|
+
try:
|
|
373
|
+
file_paths = wr.s3.list_objects(path=s3_uri)
|
|
374
|
+
except Exception as e:
|
|
375
|
+
logger.error(f"Failed to list objects in S3 path {s3_uri}: {e}")
|
|
376
|
+
raise RuntimeError(f"Failed to list objects in S3 path {s3_uri}: {e}")
|
|
377
|
+
logger.debug(f"Found {len(file_paths)} files in {s3_uri}")
|
|
378
|
+
if exclude_directories:
|
|
379
|
+
file_paths = [f for f in file_paths if not any(f.startswith(d) for d in exclude_directories)]
|
|
380
|
+
logger.debug(f"Found {len(file_paths)} files after excluding directories: {exclude_directories}")
|
|
381
|
+
else:
|
|
382
|
+
logger.debug(f"No directories excluded from {s3_uri}")
|
|
383
|
+
|
|
384
|
+
ingest_files = []
|
|
385
|
+
for path in file_paths:
|
|
386
|
+
try:
|
|
387
|
+
tags = get_tags(path)
|
|
388
|
+
except Exception as e:
|
|
389
|
+
logger.warning(f"Could not get tags for {path}: {e}")
|
|
390
|
+
continue
|
|
391
|
+
if not tags or "ingest" not in tags or tags["ingest"] != "true":
|
|
392
|
+
logger.debug(f"Skipping {path}: missing or non-true 'ingest' tag")
|
|
393
|
+
continue
|
|
394
|
+
ingest_files.append(path)
|
|
395
|
+
logger.debug(f"Found {len(ingest_files)} files with 'ingest' tag set to 'true'")
|
|
396
|
+
return ingest_files
|
|
397
|
+
|
|
398
|
+
def ingest_table_to_parquet_dataset(
|
|
399
|
+
s3_uri: str,
|
|
400
|
+
database: str,
|
|
401
|
+
table_prefix: str,
|
|
402
|
+
athena_s3_output: str,
|
|
403
|
+
workgroup: str = "primary",
|
|
404
|
+
ingest_timezone: str = "Australia/Melbourne",
|
|
405
|
+
ingest_submission_id: str = None,
|
|
406
|
+
ingest_received_at: str = None,
|
|
407
|
+
ingest_run_id: str = None,
|
|
408
|
+
) -> dict:
|
|
409
|
+
"""
|
|
410
|
+
Ingest a single CSV or JSON file from S3, annotate with metadata, and write to an Iceberg table.
|
|
411
|
+
The file is written to a Glue table named "{table_prefix}_{node}", where node is taken from the S3 tags.
|
|
412
|
+
|
|
413
|
+
Steps:
|
|
414
|
+
- Reads the file and its S3 tags/head metadata.
|
|
415
|
+
- Annotates with ingest metadata and S3 tags.
|
|
416
|
+
- Computes a row hash for deduplication/auditing.
|
|
417
|
+
- Writes to an Iceberg table in Glue/Athena under table "{table_prefix}_{node}".
|
|
418
|
+
|
|
419
|
+
Parameters
|
|
420
|
+
----------
|
|
421
|
+
s3_uri : str
|
|
422
|
+
S3 URI to a CSV or JSON file.
|
|
423
|
+
database : str
|
|
424
|
+
Glue database name.
|
|
425
|
+
table_prefix : str
|
|
426
|
+
Glue table prefix.
|
|
427
|
+
athena_s3_output : str
|
|
428
|
+
S3 URI for Athena query results and temporary staging.
|
|
429
|
+
workgroup : str, optional
|
|
430
|
+
Athena workgroup to use. Defaults to "primary".
|
|
431
|
+
ingest_timezone : str, optional
|
|
432
|
+
Timezone for ingest metadata. Defaults to "Australia/Melbourne".
|
|
433
|
+
ingest_submission_id : str, optional
|
|
434
|
+
Submission ID for this ingest. Defaults to None.
|
|
435
|
+
ingest_received_at : str, optional
|
|
436
|
+
Timestamp for when the ingest was received. If None, will use current UTC time.
|
|
437
|
+
ingest_run_id : str, optional
|
|
438
|
+
Unique ID for this ingest run. If None, will generate a new one.
|
|
439
|
+
|
|
440
|
+
Returns
|
|
441
|
+
-------
|
|
442
|
+
dict
|
|
443
|
+
Summary of the ingest run, including run ID, file count, and tables/partitions written.
|
|
444
|
+
"""
|
|
445
|
+
if ingest_received_at is None:
|
|
446
|
+
aest_tz = pytz.timezone("Australia/Melbourne")
|
|
447
|
+
ingest_received_at = datetime.now(aest_tz).strftime("%Y-%m-%dT%H:%M:%S%z")
|
|
448
|
+
if ingest_run_id is None:
|
|
449
|
+
ingest_run_id = uuid.uuid4().hex[:16]
|
|
450
|
+
|
|
451
|
+
results = []
|
|
452
|
+
tables_written = set()
|
|
453
|
+
partitions_written = {}
|
|
454
|
+
|
|
455
|
+
logger.info(f"Starting ingest run {ingest_run_id} for file: {s3_uri}.")
|
|
456
|
+
uri = s3_uri
|
|
457
|
+
logger.debug(f"Processing file: {uri}")
|
|
458
|
+
try:
|
|
459
|
+
tags = get_tags(uri)
|
|
460
|
+
logger.debug(f"Tags for {uri}: {tags}")
|
|
461
|
+
except Exception as e:
|
|
462
|
+
logger.error(f"Failed to get tags for file {uri}: {e}")
|
|
463
|
+
raise RuntimeError(f"Failed to get tags for file {uri}: {e}")
|
|
464
|
+
|
|
465
|
+
if "submission_date" not in tags:
|
|
466
|
+
logger.error(f"{uri} is missing required S3 object tag 'submission_date'")
|
|
467
|
+
raise ValueError(f"{uri} is missing required S3 object tag 'submission_date'")
|
|
468
|
+
if "node" not in tags:
|
|
469
|
+
logger.error(f"{uri} is missing required S3 object tag 'node'")
|
|
470
|
+
raise ValueError(f"{uri} is missing required S3 object tag 'node'")
|
|
471
|
+
|
|
472
|
+
study_id = tags["study_id"]
|
|
473
|
+
node = tags["node"]
|
|
474
|
+
table_name = node
|
|
475
|
+
if table_prefix:
|
|
476
|
+
table_name = f"{table_prefix}_{node}"
|
|
477
|
+
tables_written.add(table_name)
|
|
478
|
+
|
|
479
|
+
file_format = get_format(uri)
|
|
480
|
+
logger.debug(f"File format for {uri}: {file_format}")
|
|
481
|
+
try:
|
|
482
|
+
if file_format == "json":
|
|
483
|
+
df = read_json_robust(uri)
|
|
484
|
+
elif file_format == "csv":
|
|
485
|
+
df = read_csv_robust(uri)
|
|
486
|
+
elif file_format == "xlsx":
|
|
487
|
+
df_dict = read_xlsx_robust(uri)
|
|
488
|
+
df = flatten_xlsx_dict(df_dict)
|
|
489
|
+
else:
|
|
490
|
+
logger.error(f"Unsupported file format: {file_format} for file {uri}")
|
|
491
|
+
raise ValueError(f"Unsupported file format: {file_format}")
|
|
492
|
+
logger.debug(f"Read {len(df)} rows from {uri} ({file_format})")
|
|
493
|
+
except Exception as e:
|
|
494
|
+
logger.error(f"Failed to read file {uri} (format: {file_format}): {e}")
|
|
495
|
+
raise RuntimeError(f"Failed to read file {uri} (format: {file_format}): {e}")
|
|
496
|
+
|
|
497
|
+
try:
|
|
498
|
+
head_meta = get_head_meta(uri)
|
|
499
|
+
logger.debug(f"Head meta for {uri}: {head_meta}")
|
|
500
|
+
except Exception as e:
|
|
501
|
+
logger.error(f"Failed to get head metadata for file {uri}: {e}")
|
|
502
|
+
raise RuntimeError(f"Failed to get head metadata for file {uri}: {e}")
|
|
503
|
+
|
|
504
|
+
try:
|
|
505
|
+
annotated_df = prepare_ingest_metadata(
|
|
506
|
+
df=df,
|
|
507
|
+
uri=uri,
|
|
508
|
+
tags=tags,
|
|
509
|
+
head_meta=head_meta,
|
|
510
|
+
study_id=study_id,
|
|
511
|
+
ingest_run_id=ingest_run_id,
|
|
512
|
+
ingest_received_at=ingest_received_at,
|
|
513
|
+
ingest_timezone=ingest_timezone,
|
|
514
|
+
ingest_submission_id=ingest_submission_id,
|
|
515
|
+
)
|
|
516
|
+
logger.debug(f"Annotated DataFrame for {uri} with {len(annotated_df)} rows")
|
|
517
|
+
except Exception as e:
|
|
518
|
+
logger.error(f"Failed to prepare ingest metadata for file {uri}: {e}")
|
|
519
|
+
raise RuntimeError(f"Failed to prepare ingest metadata for file {uri}: {e}")
|
|
520
|
+
|
|
521
|
+
# Write this file's DataFrame to its Iceberg table
|
|
522
|
+
try:
|
|
523
|
+
write_iceberg_to_db(
|
|
524
|
+
df=annotated_df,
|
|
525
|
+
database=database,
|
|
526
|
+
table=table_name,
|
|
527
|
+
athena_s3_output=athena_s3_output,
|
|
528
|
+
workgroup=workgroup,
|
|
529
|
+
)
|
|
530
|
+
logger.debug(f"Successfully wrote {len(annotated_df)} rows to Iceberg table {database}.{table_name}")
|
|
531
|
+
except Exception as e:
|
|
532
|
+
logger.error(f"Failed to write DataFrame to Iceberg table {table_name}: {e}")
|
|
533
|
+
raise RuntimeError(f"Failed to write DataFrame to Iceberg table {table_name}: {e}")
|
|
534
|
+
|
|
535
|
+
# Track partitions written for this table
|
|
536
|
+
try:
|
|
537
|
+
partitions = sorted({
|
|
538
|
+
(r["study_id"], r["submission_date"])
|
|
539
|
+
for _, r in annotated_df[["study_id", "submission_date"]]
|
|
540
|
+
.drop_duplicates().iterrows()
|
|
541
|
+
})
|
|
542
|
+
partitions_written.setdefault(table_name, []).extend(partitions)
|
|
543
|
+
logger.debug(f"Partitions written for {table_name}: {partitions}")
|
|
544
|
+
except Exception as e:
|
|
545
|
+
logger.error(f"Failed to extract partitions from DataFrame for table {table_name}: {e}")
|
|
546
|
+
raise RuntimeError(f"Failed to extract partitions from DataFrame for table {table_name}: {e}")
|
|
547
|
+
|
|
548
|
+
results.append({
|
|
549
|
+
"uri": uri,
|
|
550
|
+
"table": table_name,
|
|
551
|
+
"rows_written": len(annotated_df),
|
|
552
|
+
"partitions": partitions,
|
|
553
|
+
})
|
|
554
|
+
|
|
555
|
+
logger.info(
|
|
556
|
+
f"Ingest run {ingest_run_id} complete. "
|
|
557
|
+
f"Files processed: 1. Tables: {sorted(tables_written)}"
|
|
558
|
+
)
|
|
559
|
+
|
|
560
|
+
return {
|
|
561
|
+
"ingest_run_id": ingest_run_id,
|
|
562
|
+
"ingest_submission_id": ingest_submission_id,
|
|
563
|
+
"ingest_received_at": ingest_received_at,
|
|
564
|
+
"files_processed": 1,
|
|
565
|
+
"tables_written": sorted(tables_written),
|
|
566
|
+
"partitions_written": partitions_written,
|
|
567
|
+
"results": results
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
def ingest_files_to_parquet_dataset(
|
|
571
|
+
s3_uris: list,
|
|
572
|
+
database: str,
|
|
573
|
+
table_prefix: str,
|
|
574
|
+
athena_s3_output: str,
|
|
575
|
+
workgroup: str = "primary",
|
|
576
|
+
ingest_submission_id: str = None,
|
|
577
|
+
exclude_fn: list = ['program.json', 'project.json'],
|
|
578
|
+
):
|
|
579
|
+
"""
|
|
580
|
+
Ingest multiple files from S3, annotate with metadata, and write to Iceberg tables in Glue/Athena.
|
|
581
|
+
Each time this function is called, a single ingestion ID is created which will be attached to all the files
|
|
582
|
+
in the list of s3_uris.
|
|
583
|
+
|
|
584
|
+
Parameters
|
|
585
|
+
----------
|
|
586
|
+
s3_uris : list
|
|
587
|
+
List of S3 URIs to CSV or JSON files to ingest.
|
|
588
|
+
database : str
|
|
589
|
+
Glue database name.
|
|
590
|
+
table_prefix : str
|
|
591
|
+
Prefix for Glue table names.
|
|
592
|
+
athena_s3_output : str
|
|
593
|
+
S3 URI for Athena query results and temporary staging.
|
|
594
|
+
workgroup : str, optional
|
|
595
|
+
Athena workgroup to use. Defaults to "primary".
|
|
596
|
+
ingest_submission_id : str, optional
|
|
597
|
+
Submission ID for ingest metadata. Defaults to None.
|
|
598
|
+
exclude_fn : list, optional
|
|
599
|
+
List of file names from the s3_uris to exclude from ingestion.
|
|
600
|
+
Example: ['program.json', 'project.json'] or ['randomDatafile.csv']
|
|
601
|
+
"""
|
|
602
|
+
|
|
603
|
+
ingest_run_id = str(uuid.uuid4())
|
|
604
|
+
|
|
605
|
+
# Generate timestamp in Australia/Melbourne timezone
|
|
606
|
+
aest_tz = pytz.timezone("Australia/Melbourne")
|
|
607
|
+
ingest_received_at = datetime.now(aest_tz).strftime("%Y-%m-%dT%H:%M:%S%z")
|
|
608
|
+
|
|
609
|
+
# Hardcode the timezone for metadata
|
|
610
|
+
ingest_timezone = "Australia/Melbourne"
|
|
611
|
+
|
|
612
|
+
results = []
|
|
613
|
+
for uri in s3_uris:
|
|
614
|
+
if any(uri.endswith(exclude) for exclude in exclude_fn):
|
|
615
|
+
continue
|
|
616
|
+
resp = ingest_table_to_parquet_dataset(
|
|
617
|
+
s3_uri=uri,
|
|
618
|
+
database=database,
|
|
619
|
+
table_prefix=table_prefix,
|
|
620
|
+
athena_s3_output=athena_s3_output,
|
|
621
|
+
workgroup=workgroup,
|
|
622
|
+
ingest_timezone=ingest_timezone,
|
|
623
|
+
ingest_submission_id=ingest_submission_id,
|
|
624
|
+
ingest_run_id=ingest_run_id,
|
|
625
|
+
ingest_received_at=ingest_received_at
|
|
626
|
+
)
|
|
627
|
+
results.append(resp)
|
|
628
|
+
|
|
629
|
+
return results
|