exadata-validator 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,61 @@
1
+ Metadata-Version: 2.4
2
+ Name: exadata-validator
3
+ Version: 0.0.1
4
+ Summary: Validate data-model folders using DuckDB.
5
+ Author: Exaflow Team
6
+ Requires-Python: >=3.10,<3.11
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Programming Language :: Python :: 3.10
9
+ Requires-Dist: click (>=8.1,<8.2)
10
+ Requires-Dist: duckdb (>=1.1,<1.2)
11
+ Description-Content-Type: text/markdown
12
+
13
+ # exadata-validator
14
+
15
+ `exadata-validator` validates data-model folders using DuckDB.
16
+
17
+ ## Install With pip
18
+
19
+ ```bash
20
+ python -m venv .venv
21
+ source .venv/bin/activate
22
+ pip install exadata-validator
23
+ ```
24
+
25
+ Validate a data model folder:
26
+
27
+ ```bash
28
+ exadata-validator validate-data-model /path/to/data_model_folder
29
+ ```
30
+
31
+ ## Develop With Poetry
32
+
33
+ From the repository root:
34
+
35
+ ```bash
36
+ cd data-validator/exaflow-data-validator
37
+ poetry install
38
+ ```
39
+
40
+ Then run the same CLI through Poetry:
41
+
42
+ ```bash
43
+ poetry run exadata-validator validate-data-model /path/to/data_model_folder
44
+ ```
45
+
46
+ Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
47
+
48
+ ## Folder Layout
49
+
50
+ ```text
51
+ /path/to/data_model_folder/
52
+ CDEsMetadata.json
53
+ dataset1.csv
54
+ dataset2.csv
55
+ ```
56
+
57
+ ## Validation Notes
58
+
59
+ - CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
60
+ - Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
61
+
@@ -0,0 +1,48 @@
1
+ # exadata-validator
2
+
3
+ `exadata-validator` validates data-model folders using DuckDB.
4
+
5
+ ## Install With pip
6
+
7
+ ```bash
8
+ python -m venv .venv
9
+ source .venv/bin/activate
10
+ pip install exadata-validator
11
+ ```
12
+
13
+ Validate a data model folder:
14
+
15
+ ```bash
16
+ exadata-validator validate-data-model /path/to/data_model_folder
17
+ ```
18
+
19
+ ## Develop With Poetry
20
+
21
+ From the repository root:
22
+
23
+ ```bash
24
+ cd data-validator/exaflow-data-validator
25
+ poetry install
26
+ ```
27
+
28
+ Then run the same CLI through Poetry:
29
+
30
+ ```bash
31
+ poetry run exadata-validator validate-data-model /path/to/data_model_folder
32
+ ```
33
+
34
+ Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
35
+
36
+ ## Folder Layout
37
+
38
+ ```text
39
+ /path/to/data_model_folder/
40
+ CDEsMetadata.json
41
+ dataset1.csv
42
+ dataset2.csv
43
+ ```
44
+
45
+ ## Validation Notes
46
+
47
+ - CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
48
+ - Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
@@ -0,0 +1 @@
1
+ """Standalone validator package."""
@@ -0,0 +1,4 @@
1
+ from data_validator.commands import entry
2
+
3
+ if __name__ == "__main__":
4
+ entry()
@@ -0,0 +1,99 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import sys
5
+ from pathlib import Path
6
+ from uuid import uuid4
7
+
8
+ import click as cl
9
+
10
+ from data_validator.exceptions import ExitCode
11
+ from data_validator.exceptions import InvalidDatasetError
12
+ from data_validator.exceptions import handle_errors
13
+ from data_validator.reporting import render_report
14
+ from data_validator.usecases import validate_data_model_folder
15
+
16
+ logging.basicConfig(
17
+ stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
18
+ )
19
+ LOGGER = logging.getLogger(__name__)
20
+
21
+
22
+ @cl.group()
23
+ def cli():
24
+ """exadata-validator command line interface."""
25
+
26
+
27
+ def _format_report_link(output_path: Path) -> str:
28
+ resolved = output_path.resolve()
29
+ uri = resolved.as_uri()
30
+ label = resolved.name
31
+ if cl.get_text_stream("stdout").isatty():
32
+ # OSC 8 hyperlink sequence for terminals that support clickable links.
33
+ return f"\033]8;;{uri}\033\\{label}\033]8;;\033\\"
34
+ return uri
35
+
36
+
37
+ def _default_html_report_path() -> Path:
38
+ return Path("/tmp") / f"exadata-validator-report-{uuid4().hex}.html"
39
+
40
+
41
+ @cli.command("validate-data-model")
42
+ @cl.argument("folder", type=cl.Path(exists=True, file_okay=False, path_type=Path))
43
+ @cl.option(
44
+ "--threads",
45
+ type=cl.IntRange(min=1),
46
+ default=None,
47
+ help="DuckDB worker threads (defaults to CPU count).",
48
+ )
49
+ @cl.option(
50
+ "--report-all/--fail-fast",
51
+ default=True,
52
+ help="Collect and report all validation errors instead of failing at the first one.",
53
+ )
54
+ @cl.option(
55
+ "--format",
56
+ "output_format",
57
+ type=cl.Choice(["text", "json", "ndjson", "html"], case_sensitive=False),
58
+ default="text",
59
+ show_default=True,
60
+ help="Output format when collecting all validation errors.",
61
+ )
62
+ @cl.option(
63
+ "--output",
64
+ "output_path",
65
+ type=cl.Path(dir_okay=False, writable=True, path_type=Path),
66
+ default=None,
67
+ help="Optional output file path when collecting all validation errors.",
68
+ )
69
+ @handle_errors
70
+ def validate_data_model(
71
+ folder: Path,
72
+ threads: int | None,
73
+ report_all: bool,
74
+ output_format: str,
75
+ output_path: Path | None,
76
+ ):
77
+ report = validate_data_model_folder(folder, threads=threads, report_all=report_all)
78
+
79
+ if report_all:
80
+ selected_format = output_format.lower()
81
+ rendered = render_report(report, selected_format)
82
+ target_output_path = output_path
83
+ if selected_format == "html" and target_output_path is None:
84
+ target_output_path = _default_html_report_path()
85
+
86
+ if target_output_path:
87
+ target_output_path.write_text(rendered + "\n", encoding="utf-8")
88
+ cl.echo(f"Report written to {_format_report_link(target_output_path)}")
89
+ else:
90
+ cl.echo(rendered)
91
+ if report.has_errors:
92
+ raise SystemExit(ExitCode.FILE_ERROR)
93
+ elif report.has_errors:
94
+ raise InvalidDatasetError(report.issues[0].message)
95
+
96
+ LOGGER.info("Validation completed successfully for %s", folder)
97
+
98
+
99
+ entry = cli
@@ -0,0 +1,175 @@
1
+ import json
2
+ from dataclasses import dataclass
3
+
4
+ from data_validator.exceptions import InvalidDataModelError
5
+ from data_validator.exceptions import UserInputError
6
+
7
+
8
+ @dataclass
9
+ class CommonDataElement:
10
+ code: str
11
+ metadata: str
12
+
13
+ @classmethod
14
+ def from_metadata(cls, metadata: dict):
15
+ code = metadata["code"]
16
+ if not code.isidentifier():
17
+ raise UserInputError(f"CDE: {code} is not a valid python identifier")
18
+
19
+ validate_metadata(code, metadata)
20
+ return cls(code=code, metadata=json.dumps(metadata))
21
+
22
+ def get_enumerations(self):
23
+ parsed = json.loads(self.metadata)
24
+ return parsed["enumerations"] if "enumerations" in parsed else {}
25
+
26
+
27
+ def flatten_cdes(schema_data):
28
+ cdes = []
29
+
30
+ if "variables" in schema_data:
31
+ for metadata in schema_data["variables"]:
32
+ metadata = reformat_metadata(metadata)
33
+ cdes.append(CommonDataElement.from_metadata(metadata))
34
+
35
+ if "groups" in schema_data:
36
+ for group_data in schema_data["groups"]:
37
+ cdes.extend(flatten_cdes(group_data))
38
+
39
+ return cdes
40
+
41
+
42
+ def get_sql_type_per_column(cdes):
43
+ return {code: json.loads(cde.metadata)["sql_type"] for code, cde in cdes.items()}
44
+
45
+
46
+ def get_cdes_with_min_max(cdes, columns):
47
+ cdes_with_min_max = {}
48
+ for code, cde in cdes.items():
49
+ if code not in columns:
50
+ continue
51
+
52
+ metadata = json.loads(cde.metadata)
53
+ min_value = metadata.get("min")
54
+ max_value = metadata.get("max")
55
+
56
+ if min_value is not None or max_value is not None:
57
+ cdes_with_min_max[code] = (min_value, max_value)
58
+
59
+ return cdes_with_min_max
60
+
61
+
62
+ def get_cdes_with_enumerations(cdes, columns):
63
+ cdes_with_enumerations = {}
64
+ for code, cde in cdes.items():
65
+ if code not in columns:
66
+ continue
67
+
68
+ metadata = json.loads(cde.metadata)
69
+ if metadata["is_categorical"]:
70
+ cdes_with_enumerations[code] = list(metadata["enumerations"].keys())
71
+
72
+ return cdes_with_enumerations
73
+
74
+
75
+ def get_dataset_enums(cdes):
76
+ return json.loads(cdes["dataset"].metadata)["enumerations"]
77
+
78
+
79
+ def validate_dataset_present_on_cdes_with_proper_format(cdes):
80
+ dataset_cde = [cde for cde in cdes if cde.code == "dataset"]
81
+ if not dataset_cde:
82
+ raise InvalidDataModelError("There is no 'dataset' CDE in the data model.")
83
+
84
+ dataset_metadata = json.loads(dataset_cde[0].metadata)
85
+ if not dataset_metadata["is_categorical"]:
86
+ raise InvalidDataModelError(
87
+ "CDE 'dataset' must have the 'isCategorical' property equal to 'true'."
88
+ )
89
+
90
+ if dataset_metadata["sql_type"] != "text":
91
+ raise InvalidDataModelError(
92
+ "CDE 'dataset' must have the 'sql_type' property equal to 'text'."
93
+ )
94
+
95
+
96
+ def validate_longitudinal_data_model(cdes):
97
+ subject_id_metadata = None
98
+ visit_id_metadata = None
99
+
100
+ for cde in cdes:
101
+ if cde.code == "subjectid":
102
+ subject_id_metadata = json.loads(cde.metadata)
103
+ elif cde.code == "visitid":
104
+ visit_id_metadata = json.loads(cde.metadata)
105
+
106
+ if not subject_id_metadata:
107
+ raise InvalidDataModelError(
108
+ "There is no 'subjectid' CDE in the longitudinal data model."
109
+ )
110
+
111
+ if not visit_id_metadata:
112
+ raise InvalidDataModelError(
113
+ "There is no 'visitid' CDE in the longitudinal data model."
114
+ )
115
+
116
+ validate_visitid_cde(visit_id_metadata)
117
+
118
+
119
+ def validate_visitid_cde(metadata):
120
+ if not metadata["is_categorical"]:
121
+ raise InvalidDataModelError(
122
+ "CDE 'visitid' must have the 'isCategorical' property equal to 'true'."
123
+ )
124
+
125
+ if metadata["sql_type"] != "text":
126
+ raise InvalidDataModelError(
127
+ "CDE 'visitid' must have the 'sql_type' property equal to 'text'."
128
+ )
129
+
130
+ if "enumerations" not in metadata:
131
+ raise InvalidDataModelError(
132
+ "CDE 'visitid' must contain the 'enumerations' property."
133
+ )
134
+
135
+
136
+ def reformat_metadata(metadata):
137
+ new_key_assign = {
138
+ "isCategorical": "is_categorical",
139
+ "minValue": "min",
140
+ "maxValue": "max",
141
+ }
142
+
143
+ for old_key, new_key in new_key_assign.items():
144
+ if old_key in metadata:
145
+ metadata[new_key] = metadata.pop(old_key)
146
+
147
+ if "enumerations" in metadata:
148
+ metadata["enumerations"] = {
149
+ enumeration["code"]: enumeration["label"]
150
+ for enumeration in metadata["enumerations"]
151
+ }
152
+
153
+ return metadata
154
+
155
+
156
+ def validate_metadata(code, metadata):
157
+ for element in ["is_categorical", "code", "sql_type", "label", "type"]:
158
+ if element not in metadata:
159
+ raise InvalidDataModelError(
160
+ f"Element: {element} is missing from the CDE {code}"
161
+ )
162
+
163
+ if metadata["is_categorical"] and "enumerations" not in metadata:
164
+ raise InvalidDataModelError(
165
+ f"The CDE {code} has 'is_categorical' set to True but there are no enumerations."
166
+ )
167
+
168
+ if {"min", "max"} <= set(metadata) and metadata["min"] >= metadata["max"]:
169
+ raise InvalidDataModelError(f"The CDE {code} has min greater than the max.")
170
+
171
+ valid_metadata_types = ["nominal", "real", "integer", "text"]
172
+ if metadata["type"] not in valid_metadata_types:
173
+ raise InvalidDataModelError(
174
+ f"The CDE {code} has an 'type' the only valid types are:{valid_metadata_types} "
175
+ )