gen3-dataops-toolkit 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. g3dt/__init__.py +0 -0
  2. g3dt/cli/__init__.py +5 -0
  3. g3dt/cli/_internal/__init__.py +1 -0
  4. g3dt/cli/_internal/dispatch.py +428 -0
  5. g3dt/cli/_internal/registry.py +65 -0
  6. g3dt/cli/_internal/resolve.py +22 -0
  7. g3dt/cli/_internal/runner.py +76 -0
  8. g3dt/cli/_internal/safety.py +110 -0
  9. g3dt/cli/config_cmds.py +202 -0
  10. g3dt/cli/delete_cmds.py +101 -0
  11. g3dt/cli/dict_cmds.py +102 -0
  12. g3dt/cli/ec2_cmds.py +114 -0
  13. g3dt/cli/indexd_cmds.py +57 -0
  14. g3dt/cli/jobs.py +83 -0
  15. g3dt/cli/k8s.py +54 -0
  16. g3dt/cli/main.py +110 -0
  17. g3dt/cli/metadata.py +76 -0
  18. g3dt/cli/synth.py +206 -0
  19. g3dt/config.py +393 -0
  20. g3dt/indexd/__init__.py +0 -0
  21. g3dt/indexd/indexd_registrar.py +244 -0
  22. g3dt/ingest/ingest.py +629 -0
  23. g3dt/resolver.py +163 -0
  24. g3dt/services/delete/delete_all_metadata_for_project.py +170 -0
  25. g3dt/services/delete/delete_metadata.sh +153 -0
  26. g3dt/services/delete/delete_metadata_by_guid.py +338 -0
  27. g3dt/services/dictionary/deploy_dd.sh +65 -0
  28. g3dt/services/dictionary/pull_dict.sh +59 -0
  29. g3dt/services/dictionary/upload_dictionary.py +109 -0
  30. g3dt/services/indexd/register_indexd.py +240 -0
  31. g3dt/services/k8s_ops/argocd_restart_etl.sh +140 -0
  32. g3dt/services/k8s_ops/argocd_restart_ms.sh +102 -0
  33. g3dt/services/k8s_ops/argocd_restart_schema.sh +106 -0
  34. g3dt/services/k8s_ops/login_to_pod.sh +110 -0
  35. g3dt/services/k8s_ops/restart_etl_and_ms.sh +56 -0
  36. g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +183 -0
  37. g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +124 -0
  38. g3dt/services/synthetic_data/generate_synth_metadata.sh +133 -0
  39. g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +165 -0
  40. g3dt/services/upload/metadata/upload_all_studies.sh +108 -0
  41. g3dt/services/upload/metadata/upload_metadata.py +152 -0
  42. g3dt/upload/__init__.py +1 -0
  43. g3dt/upload/metadata_deleter.py +265 -0
  44. g3dt/upload/metadata_submitter.py +1093 -0
  45. g3dt/upload/upload_synthdata_s3.py +164 -0
  46. g3dt/utils/athena_utils.py +834 -0
  47. g3dt/utils/dbt_utils.py +66 -0
  48. g3dt/utils/release_writer.py +188 -0
  49. g3dt/validate/validate.py +609 -0
  50. gen3_dataops_toolkit-2.0.0.dist-info/METADATA +125 -0
  51. gen3_dataops_toolkit-2.0.0.dist-info/RECORD +53 -0
  52. gen3_dataops_toolkit-2.0.0.dist-info/WHEEL +4 -0
  53. gen3_dataops_toolkit-2.0.0.dist-info/entry_points.txt +3 -0
g3dt/ingest/ingest.py ADDED
@@ -0,0 +1,629 @@
1
+ import awswrangler as wr
2
+ import pandas as pd
3
+ import boto3
4
+ from g3dt.utils.athena_utils import write_iceberg_to_db
5
+ import urllib.parse
6
+ import os
7
+ import uuid
8
+ import hashlib
9
+ from datetime import datetime
10
+ from botocore.exceptions import ClientError
11
+ import logging
12
+ import pytz # Replaced tzlocal with pytz
13
+ import s3fs
14
+ from typing import Dict
15
+
16
+
17
+ logger = logging.getLogger(__name__)
18
+ logger.setLevel(logging.INFO)
19
+
20
+ s3 = boto3.client("s3")
21
+
22
+ # ---------- helpers ----------
23
+ def parse_s3(uri: str):
24
+ p = urllib.parse.urlparse(uri)
25
+ return p.netloc, p.path.lstrip("/")
26
+
27
+ def get_tags(uri: str) -> dict:
28
+ b, k = parse_s3(uri)
29
+ try:
30
+ ts = s3.get_object_tagging(Bucket=b, Key=k)["TagSet"]
31
+ logger.debug(f"Retrieved tags for {uri}: {ts}")
32
+ return {t["Key"]: t["Value"] for t in ts}
33
+ except ClientError as e:
34
+ if e.response["Error"]["Code"] == "NoSuchKey":
35
+ logger.warning(f"NoSuchKey when getting tags for {uri}")
36
+ return {}
37
+ logger.error(f"ClientError getting tags for {uri}: {e}")
38
+ raise
39
+ except Exception as e:
40
+ logger.error(f"Failed to get tags for S3 object {uri}: {e}")
41
+ raise RuntimeError(f"Failed to get tags for S3 object {uri}: {e}")
42
+
43
+ def get_head_meta(uri: str) -> dict:
44
+ b, k = parse_s3(uri)
45
+ try:
46
+ h = s3.head_object(Bucket=b, Key=k)
47
+ logger.debug(f"Head metadata for {uri}: {h}")
48
+ except ClientError as e:
49
+ logger.error(f"Failed to get head metadata for S3 object {uri}: {e}")
50
+ raise RuntimeError(f"Failed to get head metadata for S3 object {uri}: {e}")
51
+ except Exception as e:
52
+ logger.error(f"Unexpected error getting head metadata for S3 object {uri}: {e}")
53
+ raise RuntimeError(f"Unexpected error getting head metadata for S3 object {uri}: {e}")
54
+
55
+ # Consistently use Australia/Melbourne timezone
56
+ aest_tz = pytz.timezone("Australia/Melbourne")
57
+ last_modified_aest = h["LastModified"].astimezone(aest_tz)
58
+
59
+ return {
60
+ "ingest_file_etag": (h.get("ETag") or "").strip('"'),
61
+ "ingest_file_size_bytes": str(h.get("ContentLength")),
62
+ "ingest_file_last_modified": last_modified_aest.strftime("%Y-%m-%dT%H:%M:%S%z"),
63
+ }
64
+
65
+ def normalise(name: str) -> str:
66
+ n = (name or "col").strip().lower().replace("/", "_").replace(" ", "_")
67
+ allowed = "abcdefghijklmnopqrstuvwxyz0123456789_"
68
+ n = "".join(ch if ch in allowed else "_" for ch in n)
69
+ while "__" in n: n = n.replace("__", "_")
70
+ if not n: n = "col"
71
+ if n[0].isdigit(): n = f"c_{n}"
72
+ return n
73
+
74
+ def read_csv_robust(uri: str) -> pd.DataFrame:
75
+ # delimiter sniff + encoding fallback, keep everything as string
76
+ last_err = None
77
+ for enc in ("utf-8-sig", "cp1252"):
78
+ try:
79
+ logger.debug(f"Attempting to read CSV from {uri} with encoding {enc}")
80
+ df = wr.s3.read_csv(
81
+ path=uri,
82
+ sep=None, engine="python",
83
+ dtype=str,
84
+ keep_default_na=False,
85
+ encoding=enc,
86
+ quoting=0, # QUOTE_MINIMAL
87
+ )
88
+ logger.debug(f"Successfully read CSV from {uri} with encoding {enc}")
89
+ break
90
+ except Exception as e:
91
+ logger.debug(f"Failed to read CSV from {uri} with encoding {enc}: {e}")
92
+ last_err = e
93
+ else:
94
+ logger.error(f"Failed to read CSV from {uri} after trying all encodings: {last_err}")
95
+ raise RuntimeError(f"Failed to read CSV from {uri}: {last_err}")
96
+ # normalise & uniquify headers
97
+ cols = [normalise(c) for c in df.columns]
98
+ seen, out = set(), []
99
+ for c in cols:
100
+ base, i, cand = c, 1, c
101
+ while cand in seen:
102
+ i += 1
103
+ cand = f"{base}_{i}"
104
+ seen.add(cand); out.append(cand)
105
+ df.columns = out
106
+ logger.debug(f"Normalised columns for {uri}: {out}")
107
+ return df.reset_index(drop=True)
108
+
109
+ def read_json_robust(uri: str) -> pd.DataFrame:
110
+ # encoding fallback, keep everything as string
111
+ last_err = None
112
+ for enc in ("utf-8-sig", "cp1252"):
113
+ try:
114
+ logger.debug(f"Attempting to read JSON from {uri} with encoding {enc}")
115
+ # Remove keep_default_na, which is not a valid argument for wr.s3.read_json
116
+ df = wr.s3.read_json(
117
+ path=uri,
118
+ orient="records",
119
+ dtype=str,
120
+ encoding=enc,
121
+ )
122
+ logger.debug(f"Successfully read JSON from {uri} with encoding {enc}")
123
+ break
124
+ except Exception as e:
125
+ logger.debug(f"Failed to read JSON from {uri} with encoding {enc}: {e}")
126
+ last_err = e
127
+ else:
128
+ logger.error(f"Failed to read JSON from {uri} after trying all encodings: {last_err}")
129
+ raise RuntimeError(f"Failed to read JSON from {uri}: {last_err}")
130
+ return df
131
+
132
+ def read_xlsx_robust(s3_uri: str) -> dict[str, pd.DataFrame]:
133
+ """
134
+ Reads an XLSX file from the given S3 URI using pandas with the openpyxl engine.
135
+ Reads all values as strings and disables default NA values.
136
+ Returns a dictionary of dataframes, with the sheet name as the key.
137
+ If there is only one sheet, the key is set to the file name (minus extension) from s3_uri.
138
+
139
+ Args:
140
+ s3_uri (str): The S3 URI of the XLSX file.
141
+
142
+ Returns:
143
+ dict[str, pd.DataFrame]: Dictionary mapping sheet name (or file name) to DataFrame.
144
+
145
+ Raises:
146
+ RuntimeError: If reading the XLSX file fails,
147
+ ValueError: If the S3 URI is invalid.
148
+ """
149
+ # Check that s3_uri is a valid S3 URI
150
+ if not isinstance(s3_uri, str) or not s3_uri.startswith("s3://"):
151
+ logger.error(f"Invalid S3 URI: {s3_uri}")
152
+ raise ValueError(f"Invalid S3 URI: {s3_uri}")
153
+ try:
154
+ logger.debug(f"Attempting to read XLSX from {s3_uri}")
155
+ # Read all sheets, always returns dict
156
+ df_dict = pd.read_excel(
157
+ s3_uri,
158
+ sheet_name=None,
159
+ engine="openpyxl",
160
+ dtype=str,
161
+ keep_default_na=False
162
+ )
163
+ logger.debug(f"Successfully read XLSX from {s3_uri}: sheets={list(df_dict.keys())}")
164
+
165
+ if len(df_dict) == 1:
166
+ # Only one sheet, rename the key to the file name (no extension)
167
+ import os
168
+ file_name = os.path.splitext(os.path.basename(s3_uri))[0]
169
+ only_df = next(iter(df_dict.values()))
170
+ logger.debug(f"Only one sheet found, renaming key to file name: {file_name}")
171
+ return {file_name: only_df}
172
+
173
+ return df_dict
174
+ except Exception as e:
175
+ logger.error(f"Failed to read XLSX from {s3_uri}: {e}")
176
+ raise RuntimeError(f"Failed to read XLSX from {s3_uri}: {e}")
177
+
178
+
179
+ def flatten_xlsx_dict(df_dict: dict[str, pd.DataFrame]) -> pd.DataFrame:
180
+ """
181
+ Flattens a dictionary of DataFrames (representing Excel sheets) into a single DataFrame.
182
+
183
+ Adds a "sheet_name" column to each DataFrame, indicating the originating sheet,
184
+ then concatenates all DataFrames into one.
185
+
186
+ Args:
187
+ df_dict (dict[str, pd.DataFrame]): Dictionary mapping sheet names to DataFrames.
188
+
189
+ Returns:
190
+ pd.DataFrame: Concatenated DataFrame with an added "sheet_name" column.
191
+
192
+ Raises:
193
+ ValueError: If the provided dictionary is empty.
194
+ RuntimeError: If an error occurs during DataFrame concatenation.
195
+ """
196
+ if not df_dict:
197
+ logger.error("The df_dict provided to flatten_xlsx_dict is empty.")
198
+ raise ValueError("Input dictionary of DataFrames is empty.")
199
+
200
+ dfs_with_sheet = []
201
+ try:
202
+ for sheet_name, df in df_dict.items():
203
+ df_with_sheet = df.copy()
204
+ df_with_sheet["sheet_name"] = sheet_name
205
+ dfs_with_sheet.append(df_with_sheet)
206
+ return pd.concat(dfs_with_sheet, ignore_index=True)
207
+ except Exception as e:
208
+ logger.error(f"Failed to flatten XLSX dictionary: {e}")
209
+ raise RuntimeError(f"Failed to flatten XLSX dictionary: {e}") from e
210
+
211
+
212
+ def get_format(uri: str) -> str:
213
+ """
214
+ Returns the file format (extension) of the given URI, without the leading dot.
215
+ """
216
+ _, ext = os.path.splitext(uri)
217
+ fmt = ext[1:].lower()
218
+ logger.debug(f"File format for {uri} is {fmt}")
219
+ return fmt
220
+
221
+ def compute_row_hash(row: pd.Series) -> str:
222
+ parts = [f"{k}={row[k] if row[k] is not None else ''}" for k in sorted(row.index)]
223
+ return hashlib.sha256(("||".join(parts)).encode("utf-8")).hexdigest()
224
+
225
+ def sanitize_submission_date(value: str) -> str:
226
+ """
227
+ Parse a submission_date string and return it in 'YYYY-MM-DD' format.
228
+ Handles various formats including 'YYYY_MM_DD', 'DD MM YYYY', 'DD-MM-YYYY',
229
+ and 'YYYY-MM-DD'. It expects the input to be in one of these formats.
230
+
231
+ Raises a ValueError if the input is empty or cannot be parsed as a date.
232
+ """
233
+ if not value:
234
+ logger.error("submission_date tag is empty")
235
+ raise ValueError("submission_date tag is empty")
236
+
237
+ # Standardize separators by replacing underscores and spaces with hyphens
238
+ cleaned_value = value.replace('_', '-').replace(' ', '-')
239
+
240
+ # Attempt to parse the date in 'YYYY-MM-DD' format
241
+ try:
242
+ return datetime.strptime(cleaned_value, "%Y-%m-%d").strftime("%Y-%m-%d")
243
+ except ValueError:
244
+ # If the first format fails, try the 'DD-MM-YYYY' format
245
+ try:
246
+ return datetime.strptime(cleaned_value, "%d-%m-%Y").strftime("%Y-%m-%d")
247
+ except ValueError:
248
+ logger.error(f"Unable to parse submission_date format to ISO: {value!r}")
249
+ raise ValueError(f"Unable to parse submission_date format to ISO: {value!r}")
250
+
251
+ def prepare_ingest_metadata(
252
+ df: pd.DataFrame,
253
+ uri: str,
254
+ tags: dict,
255
+ head_meta: dict,
256
+ study_id: str,
257
+ ingest_run_id: str,
258
+ ingest_received_at: str,
259
+ ingest_timezone: str,
260
+ ingest_submission_id: str,
261
+ ) -> pd.DataFrame:
262
+ """
263
+ Attach ingest and tag metadata columns to a DataFrame for a single file.
264
+
265
+ Parameters
266
+ ----------
267
+ df : pd.DataFrame
268
+ The DataFrame to annotate.
269
+ uri : str
270
+ S3 URI of the file.
271
+ tags : dict
272
+ S3 object tags.
273
+ head_meta : dict
274
+ S3 object head metadata.
275
+ study_id : str
276
+ Study identifier.
277
+ ingest_run_id : str
278
+ Unique ID for this ingest run.
279
+ ingest_received_at : str
280
+ ISO timestamp for when ingest was received.
281
+ ingest_timezone : str
282
+ Timezone string for ingest.
283
+ ingest_submission_id : str
284
+ Submission ID for this ingest.
285
+
286
+ Returns
287
+ -------
288
+ pd.DataFrame
289
+ Annotated DataFrame.
290
+ """
291
+ file_name = os.path.basename(urllib.parse.urlparse(uri).path)
292
+ df = df.copy()
293
+ try:
294
+ logger.debug(f"Annotating DataFrame for file {uri} with ingest metadata.")
295
+ df["study_id"] = str(study_id)
296
+ df["submission_date"] = sanitize_submission_date(tags["submission_date"])
297
+ df["ingest_run_id"] = ingest_run_id
298
+ df["ingest_received_at"] = ingest_received_at
299
+ df["ingest_timezone"] = ingest_timezone
300
+ df["ingest_original_file_path"] = uri
301
+ df["ingest_file_name"] = file_name
302
+ df["ingest_submission_id"] = str(ingest_submission_id or "")
303
+ df["ingest_file_etag"] = head_meta["ingest_file_etag"]
304
+ df["ingest_file_size_bytes"] = head_meta["ingest_file_size_bytes"]
305
+ df["ingest_file_last_modified"] = head_meta["ingest_file_last_modified"]
306
+ except Exception as e:
307
+ logger.error(f"Failed to prepare ingest metadata for file {uri}: {e}")
308
+ raise RuntimeError(f"Failed to prepare ingest metadata for file {uri}: {e}")
309
+
310
+ # Add S3 tags as columns (namespaced)
311
+ for k, v in tags.items():
312
+ try:
313
+ df[f"tag_{normalise(k)}"] = str(v)
314
+ except Exception as e:
315
+ logger.error(f"Failed to add tag '{k}' as column for file {uri}: {e}")
316
+ raise RuntimeError(f"Failed to add tag '{k}' as column for file {uri}: {e}")
317
+
318
+ # Compute row-hash over raw columns (exclude ingest_*, tag_*, study_id, submission_date)
319
+ try:
320
+ exclude = {c for c in df.columns if c.startswith("ingest_") or c.startswith("tag_")}
321
+ exclude |= {"study_id", "submission_date"}
322
+ raw_cols = [c for c in df.columns if c not in exclude]
323
+ logger.debug(f"Computing row hash for file {uri} on columns: {raw_cols}")
324
+ df["ingest_row_hash"] = df[raw_cols].apply(compute_row_hash, axis=1)
325
+ except Exception as e:
326
+ logger.error(f"Failed to compute row hash for file {uri}: {e}")
327
+ raise RuntimeError(f"Failed to compute row hash for file {uri}: {e}")
328
+ return df
329
+
330
+ def align_and_combine_frames(frames: list[pd.DataFrame]) -> pd.DataFrame:
331
+ """
332
+ Align columns across multiple DataFrames and concatenate them.
333
+
334
+ Parameters
335
+ ----------
336
+ frames : list of pd.DataFrame
337
+ List of DataFrames to align and combine.
338
+
339
+ Returns
340
+ -------
341
+ pd.DataFrame
342
+ Combined DataFrame with aligned columns.
343
+ """
344
+ try:
345
+ all_cols = sorted({c for f in frames for c in f.columns})
346
+ logger.debug(f"Aligning and combining {len(frames)} DataFrames with columns: {all_cols}")
347
+ aligned_frames = [f.reindex(columns=all_cols, fill_value="") for f in frames]
348
+ combined = pd.concat(aligned_frames, ignore_index=True)
349
+ logger.debug(f"Combined DataFrame shape: {combined.shape}")
350
+ return combined
351
+ except Exception as e:
352
+ logger.error(f"Failed to align and combine DataFrames: {e}")
353
+ raise RuntimeError(f"Failed to align and combine DataFrames: {e}")
354
+
355
+ def get_ingest_true_files(s3_uri: str, exclude_directories: list[str] = None) -> list[str]:
356
+ """
357
+ Get a list of files from an S3 URI, excluding directories.
358
+
359
+ Parameters
360
+ ----------
361
+ s3_uri : str
362
+ S3 URI to list files from.
363
+ exclude_directories : list[str], optional
364
+ List of directories to exclude. Defaults to None.
365
+
366
+ Returns
367
+ -------
368
+ list[str]
369
+ List of file paths.
370
+ """
371
+ logger.debug(f"Getting files from {s3_uri}")
372
+ try:
373
+ file_paths = wr.s3.list_objects(path=s3_uri)
374
+ except Exception as e:
375
+ logger.error(f"Failed to list objects in S3 path {s3_uri}: {e}")
376
+ raise RuntimeError(f"Failed to list objects in S3 path {s3_uri}: {e}")
377
+ logger.debug(f"Found {len(file_paths)} files in {s3_uri}")
378
+ if exclude_directories:
379
+ file_paths = [f for f in file_paths if not any(f.startswith(d) for d in exclude_directories)]
380
+ logger.debug(f"Found {len(file_paths)} files after excluding directories: {exclude_directories}")
381
+ else:
382
+ logger.debug(f"No directories excluded from {s3_uri}")
383
+
384
+ ingest_files = []
385
+ for path in file_paths:
386
+ try:
387
+ tags = get_tags(path)
388
+ except Exception as e:
389
+ logger.warning(f"Could not get tags for {path}: {e}")
390
+ continue
391
+ if not tags or "ingest" not in tags or tags["ingest"] != "true":
392
+ logger.debug(f"Skipping {path}: missing or non-true 'ingest' tag")
393
+ continue
394
+ ingest_files.append(path)
395
+ logger.debug(f"Found {len(ingest_files)} files with 'ingest' tag set to 'true'")
396
+ return ingest_files
397
+
398
+ def ingest_table_to_parquet_dataset(
399
+ s3_uri: str,
400
+ database: str,
401
+ table_prefix: str,
402
+ athena_s3_output: str,
403
+ workgroup: str = "primary",
404
+ ingest_timezone: str = "Australia/Melbourne",
405
+ ingest_submission_id: str = None,
406
+ ingest_received_at: str = None,
407
+ ingest_run_id: str = None,
408
+ ) -> dict:
409
+ """
410
+ Ingest a single CSV or JSON file from S3, annotate with metadata, and write to an Iceberg table.
411
+ The file is written to a Glue table named "{table_prefix}_{node}", where node is taken from the S3 tags.
412
+
413
+ Steps:
414
+ - Reads the file and its S3 tags/head metadata.
415
+ - Annotates with ingest metadata and S3 tags.
416
+ - Computes a row hash for deduplication/auditing.
417
+ - Writes to an Iceberg table in Glue/Athena under table "{table_prefix}_{node}".
418
+
419
+ Parameters
420
+ ----------
421
+ s3_uri : str
422
+ S3 URI to a CSV or JSON file.
423
+ database : str
424
+ Glue database name.
425
+ table_prefix : str
426
+ Glue table prefix.
427
+ athena_s3_output : str
428
+ S3 URI for Athena query results and temporary staging.
429
+ workgroup : str, optional
430
+ Athena workgroup to use. Defaults to "primary".
431
+ ingest_timezone : str, optional
432
+ Timezone for ingest metadata. Defaults to "Australia/Melbourne".
433
+ ingest_submission_id : str, optional
434
+ Submission ID for this ingest. Defaults to None.
435
+ ingest_received_at : str, optional
436
+ Timestamp for when the ingest was received. If None, will use current UTC time.
437
+ ingest_run_id : str, optional
438
+ Unique ID for this ingest run. If None, will generate a new one.
439
+
440
+ Returns
441
+ -------
442
+ dict
443
+ Summary of the ingest run, including run ID, file count, and tables/partitions written.
444
+ """
445
+ if ingest_received_at is None:
446
+ aest_tz = pytz.timezone("Australia/Melbourne")
447
+ ingest_received_at = datetime.now(aest_tz).strftime("%Y-%m-%dT%H:%M:%S%z")
448
+ if ingest_run_id is None:
449
+ ingest_run_id = uuid.uuid4().hex[:16]
450
+
451
+ results = []
452
+ tables_written = set()
453
+ partitions_written = {}
454
+
455
+ logger.info(f"Starting ingest run {ingest_run_id} for file: {s3_uri}.")
456
+ uri = s3_uri
457
+ logger.debug(f"Processing file: {uri}")
458
+ try:
459
+ tags = get_tags(uri)
460
+ logger.debug(f"Tags for {uri}: {tags}")
461
+ except Exception as e:
462
+ logger.error(f"Failed to get tags for file {uri}: {e}")
463
+ raise RuntimeError(f"Failed to get tags for file {uri}: {e}")
464
+
465
+ if "submission_date" not in tags:
466
+ logger.error(f"{uri} is missing required S3 object tag 'submission_date'")
467
+ raise ValueError(f"{uri} is missing required S3 object tag 'submission_date'")
468
+ if "node" not in tags:
469
+ logger.error(f"{uri} is missing required S3 object tag 'node'")
470
+ raise ValueError(f"{uri} is missing required S3 object tag 'node'")
471
+
472
+ study_id = tags["study_id"]
473
+ node = tags["node"]
474
+ table_name = node
475
+ if table_prefix:
476
+ table_name = f"{table_prefix}_{node}"
477
+ tables_written.add(table_name)
478
+
479
+ file_format = get_format(uri)
480
+ logger.debug(f"File format for {uri}: {file_format}")
481
+ try:
482
+ if file_format == "json":
483
+ df = read_json_robust(uri)
484
+ elif file_format == "csv":
485
+ df = read_csv_robust(uri)
486
+ elif file_format == "xlsx":
487
+ df_dict = read_xlsx_robust(uri)
488
+ df = flatten_xlsx_dict(df_dict)
489
+ else:
490
+ logger.error(f"Unsupported file format: {file_format} for file {uri}")
491
+ raise ValueError(f"Unsupported file format: {file_format}")
492
+ logger.debug(f"Read {len(df)} rows from {uri} ({file_format})")
493
+ except Exception as e:
494
+ logger.error(f"Failed to read file {uri} (format: {file_format}): {e}")
495
+ raise RuntimeError(f"Failed to read file {uri} (format: {file_format}): {e}")
496
+
497
+ try:
498
+ head_meta = get_head_meta(uri)
499
+ logger.debug(f"Head meta for {uri}: {head_meta}")
500
+ except Exception as e:
501
+ logger.error(f"Failed to get head metadata for file {uri}: {e}")
502
+ raise RuntimeError(f"Failed to get head metadata for file {uri}: {e}")
503
+
504
+ try:
505
+ annotated_df = prepare_ingest_metadata(
506
+ df=df,
507
+ uri=uri,
508
+ tags=tags,
509
+ head_meta=head_meta,
510
+ study_id=study_id,
511
+ ingest_run_id=ingest_run_id,
512
+ ingest_received_at=ingest_received_at,
513
+ ingest_timezone=ingest_timezone,
514
+ ingest_submission_id=ingest_submission_id,
515
+ )
516
+ logger.debug(f"Annotated DataFrame for {uri} with {len(annotated_df)} rows")
517
+ except Exception as e:
518
+ logger.error(f"Failed to prepare ingest metadata for file {uri}: {e}")
519
+ raise RuntimeError(f"Failed to prepare ingest metadata for file {uri}: {e}")
520
+
521
+ # Write this file's DataFrame to its Iceberg table
522
+ try:
523
+ write_iceberg_to_db(
524
+ df=annotated_df,
525
+ database=database,
526
+ table=table_name,
527
+ athena_s3_output=athena_s3_output,
528
+ workgroup=workgroup,
529
+ )
530
+ logger.debug(f"Successfully wrote {len(annotated_df)} rows to Iceberg table {database}.{table_name}")
531
+ except Exception as e:
532
+ logger.error(f"Failed to write DataFrame to Iceberg table {table_name}: {e}")
533
+ raise RuntimeError(f"Failed to write DataFrame to Iceberg table {table_name}: {e}")
534
+
535
+ # Track partitions written for this table
536
+ try:
537
+ partitions = sorted({
538
+ (r["study_id"], r["submission_date"])
539
+ for _, r in annotated_df[["study_id", "submission_date"]]
540
+ .drop_duplicates().iterrows()
541
+ })
542
+ partitions_written.setdefault(table_name, []).extend(partitions)
543
+ logger.debug(f"Partitions written for {table_name}: {partitions}")
544
+ except Exception as e:
545
+ logger.error(f"Failed to extract partitions from DataFrame for table {table_name}: {e}")
546
+ raise RuntimeError(f"Failed to extract partitions from DataFrame for table {table_name}: {e}")
547
+
548
+ results.append({
549
+ "uri": uri,
550
+ "table": table_name,
551
+ "rows_written": len(annotated_df),
552
+ "partitions": partitions,
553
+ })
554
+
555
+ logger.info(
556
+ f"Ingest run {ingest_run_id} complete. "
557
+ f"Files processed: 1. Tables: {sorted(tables_written)}"
558
+ )
559
+
560
+ return {
561
+ "ingest_run_id": ingest_run_id,
562
+ "ingest_submission_id": ingest_submission_id,
563
+ "ingest_received_at": ingest_received_at,
564
+ "files_processed": 1,
565
+ "tables_written": sorted(tables_written),
566
+ "partitions_written": partitions_written,
567
+ "results": results
568
+ }
569
+
570
+ def ingest_files_to_parquet_dataset(
571
+ s3_uris: list,
572
+ database: str,
573
+ table_prefix: str,
574
+ athena_s3_output: str,
575
+ workgroup: str = "primary",
576
+ ingest_submission_id: str = None,
577
+ exclude_fn: list = ['program.json', 'project.json'],
578
+ ):
579
+ """
580
+ Ingest multiple files from S3, annotate with metadata, and write to Iceberg tables in Glue/Athena.
581
+ Each time this function is called, a single ingestion ID is created which will be attached to all the files
582
+ in the list of s3_uris.
583
+
584
+ Parameters
585
+ ----------
586
+ s3_uris : list
587
+ List of S3 URIs to CSV or JSON files to ingest.
588
+ database : str
589
+ Glue database name.
590
+ table_prefix : str
591
+ Prefix for Glue table names.
592
+ athena_s3_output : str
593
+ S3 URI for Athena query results and temporary staging.
594
+ workgroup : str, optional
595
+ Athena workgroup to use. Defaults to "primary".
596
+ ingest_submission_id : str, optional
597
+ Submission ID for ingest metadata. Defaults to None.
598
+ exclude_fn : list, optional
599
+ List of file names from the s3_uris to exclude from ingestion.
600
+ Example: ['program.json', 'project.json'] or ['randomDatafile.csv']
601
+ """
602
+
603
+ ingest_run_id = str(uuid.uuid4())
604
+
605
+ # Generate timestamp in Australia/Melbourne timezone
606
+ aest_tz = pytz.timezone("Australia/Melbourne")
607
+ ingest_received_at = datetime.now(aest_tz).strftime("%Y-%m-%dT%H:%M:%S%z")
608
+
609
+ # Hardcode the timezone for metadata
610
+ ingest_timezone = "Australia/Melbourne"
611
+
612
+ results = []
613
+ for uri in s3_uris:
614
+ if any(uri.endswith(exclude) for exclude in exclude_fn):
615
+ continue
616
+ resp = ingest_table_to_parquet_dataset(
617
+ s3_uri=uri,
618
+ database=database,
619
+ table_prefix=table_prefix,
620
+ athena_s3_output=athena_s3_output,
621
+ workgroup=workgroup,
622
+ ingest_timezone=ingest_timezone,
623
+ ingest_submission_id=ingest_submission_id,
624
+ ingest_run_id=ingest_run_id,
625
+ ingest_received_at=ingest_received_at
626
+ )
627
+ results.append(resp)
628
+
629
+ return results