exadata-validator 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- exadata_validator-0.0.1/PKG-INFO +61 -0
- exadata_validator-0.0.1/README.md +48 -0
- exadata_validator-0.0.1/data_validator/__init__.py +1 -0
- exadata_validator-0.0.1/data_validator/__main__.py +4 -0
- exadata_validator-0.0.1/data_validator/commands.py +99 -0
- exadata_validator-0.0.1/data_validator/dataelements.py +175 -0
- exadata_validator-0.0.1/data_validator/duckdb_validator.py +400 -0
- exadata_validator-0.0.1/data_validator/exceptions.py +80 -0
- exadata_validator-0.0.1/data_validator/reporting.py +342 -0
- exadata_validator-0.0.1/data_validator/usecases.py +159 -0
- exadata_validator-0.0.1/pyproject.toml +27 -0
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: exadata-validator
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Validate data-model folders using DuckDB.
|
|
5
|
+
Author: Exaflow Team
|
|
6
|
+
Requires-Python: >=3.10,<3.11
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Requires-Dist: click (>=8.1,<8.2)
|
|
10
|
+
Requires-Dist: duckdb (>=1.1,<1.2)
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# exadata-validator
|
|
14
|
+
|
|
15
|
+
`exadata-validator` validates data-model folders using DuckDB.
|
|
16
|
+
|
|
17
|
+
## Install With pip
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
python -m venv .venv
|
|
21
|
+
source .venv/bin/activate
|
|
22
|
+
pip install exadata-validator
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Validate a data model folder:
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
exadata-validator validate-data-model /path/to/data_model_folder
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Develop With Poetry
|
|
32
|
+
|
|
33
|
+
From the repository root:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
cd data-validator/exaflow-data-validator
|
|
37
|
+
poetry install
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Then run the same CLI through Poetry:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
poetry run exadata-validator validate-data-model /path/to/data_model_folder
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
|
|
47
|
+
|
|
48
|
+
## Folder Layout
|
|
49
|
+
|
|
50
|
+
```text
|
|
51
|
+
/path/to/data_model_folder/
|
|
52
|
+
CDEsMetadata.json
|
|
53
|
+
dataset1.csv
|
|
54
|
+
dataset2.csv
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Validation Notes
|
|
58
|
+
|
|
59
|
+
- CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
|
|
60
|
+
- Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
|
|
61
|
+
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# exadata-validator
|
|
2
|
+
|
|
3
|
+
`exadata-validator` validates data-model folders using DuckDB.
|
|
4
|
+
|
|
5
|
+
## Install With pip
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
python -m venv .venv
|
|
9
|
+
source .venv/bin/activate
|
|
10
|
+
pip install exadata-validator
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Validate a data model folder:
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
exadata-validator validate-data-model /path/to/data_model_folder
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Develop With Poetry
|
|
20
|
+
|
|
21
|
+
From the repository root:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
cd data-validator/exaflow-data-validator
|
|
25
|
+
poetry install
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Then run the same CLI through Poetry:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
poetry run exadata-validator validate-data-model /path/to/data_model_folder
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
|
|
35
|
+
|
|
36
|
+
## Folder Layout
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
/path/to/data_model_folder/
|
|
40
|
+
CDEsMetadata.json
|
|
41
|
+
dataset1.csv
|
|
42
|
+
dataset2.csv
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Validation Notes
|
|
46
|
+
|
|
47
|
+
- CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
|
|
48
|
+
- Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Standalone validator package."""
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from uuid import uuid4
|
|
7
|
+
|
|
8
|
+
import click as cl
|
|
9
|
+
|
|
10
|
+
from data_validator.exceptions import ExitCode
|
|
11
|
+
from data_validator.exceptions import InvalidDatasetError
|
|
12
|
+
from data_validator.exceptions import handle_errors
|
|
13
|
+
from data_validator.reporting import render_report
|
|
14
|
+
from data_validator.usecases import validate_data_model_folder
|
|
15
|
+
|
|
16
|
+
logging.basicConfig(
|
|
17
|
+
stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
|
|
18
|
+
)
|
|
19
|
+
LOGGER = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@cl.group()
|
|
23
|
+
def cli():
|
|
24
|
+
"""exadata-validator command line interface."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _format_report_link(output_path: Path) -> str:
|
|
28
|
+
resolved = output_path.resolve()
|
|
29
|
+
uri = resolved.as_uri()
|
|
30
|
+
label = resolved.name
|
|
31
|
+
if cl.get_text_stream("stdout").isatty():
|
|
32
|
+
# OSC 8 hyperlink sequence for terminals that support clickable links.
|
|
33
|
+
return f"\033]8;;{uri}\033\\{label}\033]8;;\033\\"
|
|
34
|
+
return uri
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _default_html_report_path() -> Path:
|
|
38
|
+
return Path("/tmp") / f"exadata-validator-report-{uuid4().hex}.html"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@cli.command("validate-data-model")
|
|
42
|
+
@cl.argument("folder", type=cl.Path(exists=True, file_okay=False, path_type=Path))
|
|
43
|
+
@cl.option(
|
|
44
|
+
"--threads",
|
|
45
|
+
type=cl.IntRange(min=1),
|
|
46
|
+
default=None,
|
|
47
|
+
help="DuckDB worker threads (defaults to CPU count).",
|
|
48
|
+
)
|
|
49
|
+
@cl.option(
|
|
50
|
+
"--report-all/--fail-fast",
|
|
51
|
+
default=True,
|
|
52
|
+
help="Collect and report all validation errors instead of failing at the first one.",
|
|
53
|
+
)
|
|
54
|
+
@cl.option(
|
|
55
|
+
"--format",
|
|
56
|
+
"output_format",
|
|
57
|
+
type=cl.Choice(["text", "json", "ndjson", "html"], case_sensitive=False),
|
|
58
|
+
default="text",
|
|
59
|
+
show_default=True,
|
|
60
|
+
help="Output format when collecting all validation errors.",
|
|
61
|
+
)
|
|
62
|
+
@cl.option(
|
|
63
|
+
"--output",
|
|
64
|
+
"output_path",
|
|
65
|
+
type=cl.Path(dir_okay=False, writable=True, path_type=Path),
|
|
66
|
+
default=None,
|
|
67
|
+
help="Optional output file path when collecting all validation errors.",
|
|
68
|
+
)
|
|
69
|
+
@handle_errors
|
|
70
|
+
def validate_data_model(
|
|
71
|
+
folder: Path,
|
|
72
|
+
threads: int | None,
|
|
73
|
+
report_all: bool,
|
|
74
|
+
output_format: str,
|
|
75
|
+
output_path: Path | None,
|
|
76
|
+
):
|
|
77
|
+
report = validate_data_model_folder(folder, threads=threads, report_all=report_all)
|
|
78
|
+
|
|
79
|
+
if report_all:
|
|
80
|
+
selected_format = output_format.lower()
|
|
81
|
+
rendered = render_report(report, selected_format)
|
|
82
|
+
target_output_path = output_path
|
|
83
|
+
if selected_format == "html" and target_output_path is None:
|
|
84
|
+
target_output_path = _default_html_report_path()
|
|
85
|
+
|
|
86
|
+
if target_output_path:
|
|
87
|
+
target_output_path.write_text(rendered + "\n", encoding="utf-8")
|
|
88
|
+
cl.echo(f"Report written to {_format_report_link(target_output_path)}")
|
|
89
|
+
else:
|
|
90
|
+
cl.echo(rendered)
|
|
91
|
+
if report.has_errors:
|
|
92
|
+
raise SystemExit(ExitCode.FILE_ERROR)
|
|
93
|
+
elif report.has_errors:
|
|
94
|
+
raise InvalidDatasetError(report.issues[0].message)
|
|
95
|
+
|
|
96
|
+
LOGGER.info("Validation completed successfully for %s", folder)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
entry = cli
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
|
|
4
|
+
from data_validator.exceptions import InvalidDataModelError
|
|
5
|
+
from data_validator.exceptions import UserInputError
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class CommonDataElement:
|
|
10
|
+
code: str
|
|
11
|
+
metadata: str
|
|
12
|
+
|
|
13
|
+
@classmethod
|
|
14
|
+
def from_metadata(cls, metadata: dict):
|
|
15
|
+
code = metadata["code"]
|
|
16
|
+
if not code.isidentifier():
|
|
17
|
+
raise UserInputError(f"CDE: {code} is not a valid python identifier")
|
|
18
|
+
|
|
19
|
+
validate_metadata(code, metadata)
|
|
20
|
+
return cls(code=code, metadata=json.dumps(metadata))
|
|
21
|
+
|
|
22
|
+
def get_enumerations(self):
|
|
23
|
+
parsed = json.loads(self.metadata)
|
|
24
|
+
return parsed["enumerations"] if "enumerations" in parsed else {}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def flatten_cdes(schema_data):
|
|
28
|
+
cdes = []
|
|
29
|
+
|
|
30
|
+
if "variables" in schema_data:
|
|
31
|
+
for metadata in schema_data["variables"]:
|
|
32
|
+
metadata = reformat_metadata(metadata)
|
|
33
|
+
cdes.append(CommonDataElement.from_metadata(metadata))
|
|
34
|
+
|
|
35
|
+
if "groups" in schema_data:
|
|
36
|
+
for group_data in schema_data["groups"]:
|
|
37
|
+
cdes.extend(flatten_cdes(group_data))
|
|
38
|
+
|
|
39
|
+
return cdes
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def get_sql_type_per_column(cdes):
|
|
43
|
+
return {code: json.loads(cde.metadata)["sql_type"] for code, cde in cdes.items()}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def get_cdes_with_min_max(cdes, columns):
|
|
47
|
+
cdes_with_min_max = {}
|
|
48
|
+
for code, cde in cdes.items():
|
|
49
|
+
if code not in columns:
|
|
50
|
+
continue
|
|
51
|
+
|
|
52
|
+
metadata = json.loads(cde.metadata)
|
|
53
|
+
min_value = metadata.get("min")
|
|
54
|
+
max_value = metadata.get("max")
|
|
55
|
+
|
|
56
|
+
if min_value is not None or max_value is not None:
|
|
57
|
+
cdes_with_min_max[code] = (min_value, max_value)
|
|
58
|
+
|
|
59
|
+
return cdes_with_min_max
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def get_cdes_with_enumerations(cdes, columns):
|
|
63
|
+
cdes_with_enumerations = {}
|
|
64
|
+
for code, cde in cdes.items():
|
|
65
|
+
if code not in columns:
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
metadata = json.loads(cde.metadata)
|
|
69
|
+
if metadata["is_categorical"]:
|
|
70
|
+
cdes_with_enumerations[code] = list(metadata["enumerations"].keys())
|
|
71
|
+
|
|
72
|
+
return cdes_with_enumerations
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def get_dataset_enums(cdes):
|
|
76
|
+
return json.loads(cdes["dataset"].metadata)["enumerations"]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def validate_dataset_present_on_cdes_with_proper_format(cdes):
|
|
80
|
+
dataset_cde = [cde for cde in cdes if cde.code == "dataset"]
|
|
81
|
+
if not dataset_cde:
|
|
82
|
+
raise InvalidDataModelError("There is no 'dataset' CDE in the data model.")
|
|
83
|
+
|
|
84
|
+
dataset_metadata = json.loads(dataset_cde[0].metadata)
|
|
85
|
+
if not dataset_metadata["is_categorical"]:
|
|
86
|
+
raise InvalidDataModelError(
|
|
87
|
+
"CDE 'dataset' must have the 'isCategorical' property equal to 'true'."
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
if dataset_metadata["sql_type"] != "text":
|
|
91
|
+
raise InvalidDataModelError(
|
|
92
|
+
"CDE 'dataset' must have the 'sql_type' property equal to 'text'."
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def validate_longitudinal_data_model(cdes):
|
|
97
|
+
subject_id_metadata = None
|
|
98
|
+
visit_id_metadata = None
|
|
99
|
+
|
|
100
|
+
for cde in cdes:
|
|
101
|
+
if cde.code == "subjectid":
|
|
102
|
+
subject_id_metadata = json.loads(cde.metadata)
|
|
103
|
+
elif cde.code == "visitid":
|
|
104
|
+
visit_id_metadata = json.loads(cde.metadata)
|
|
105
|
+
|
|
106
|
+
if not subject_id_metadata:
|
|
107
|
+
raise InvalidDataModelError(
|
|
108
|
+
"There is no 'subjectid' CDE in the longitudinal data model."
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
if not visit_id_metadata:
|
|
112
|
+
raise InvalidDataModelError(
|
|
113
|
+
"There is no 'visitid' CDE in the longitudinal data model."
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
validate_visitid_cde(visit_id_metadata)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def validate_visitid_cde(metadata):
|
|
120
|
+
if not metadata["is_categorical"]:
|
|
121
|
+
raise InvalidDataModelError(
|
|
122
|
+
"CDE 'visitid' must have the 'isCategorical' property equal to 'true'."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
if metadata["sql_type"] != "text":
|
|
126
|
+
raise InvalidDataModelError(
|
|
127
|
+
"CDE 'visitid' must have the 'sql_type' property equal to 'text'."
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if "enumerations" not in metadata:
|
|
131
|
+
raise InvalidDataModelError(
|
|
132
|
+
"CDE 'visitid' must contain the 'enumerations' property."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def reformat_metadata(metadata):
|
|
137
|
+
new_key_assign = {
|
|
138
|
+
"isCategorical": "is_categorical",
|
|
139
|
+
"minValue": "min",
|
|
140
|
+
"maxValue": "max",
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
for old_key, new_key in new_key_assign.items():
|
|
144
|
+
if old_key in metadata:
|
|
145
|
+
metadata[new_key] = metadata.pop(old_key)
|
|
146
|
+
|
|
147
|
+
if "enumerations" in metadata:
|
|
148
|
+
metadata["enumerations"] = {
|
|
149
|
+
enumeration["code"]: enumeration["label"]
|
|
150
|
+
for enumeration in metadata["enumerations"]
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return metadata
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def validate_metadata(code, metadata):
|
|
157
|
+
for element in ["is_categorical", "code", "sql_type", "label", "type"]:
|
|
158
|
+
if element not in metadata:
|
|
159
|
+
raise InvalidDataModelError(
|
|
160
|
+
f"Element: {element} is missing from the CDE {code}"
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
if metadata["is_categorical"] and "enumerations" not in metadata:
|
|
164
|
+
raise InvalidDataModelError(
|
|
165
|
+
f"The CDE {code} has 'is_categorical' set to True but there are no enumerations."
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
if {"min", "max"} <= set(metadata) and metadata["min"] >= metadata["max"]:
|
|
169
|
+
raise InvalidDataModelError(f"The CDE {code} has min greater than the max.")
|
|
170
|
+
|
|
171
|
+
valid_metadata_types = ["nominal", "real", "integer", "text"]
|
|
172
|
+
if metadata["type"] not in valid_metadata_types:
|
|
173
|
+
raise InvalidDataModelError(
|
|
174
|
+
f"The CDE {code} has an 'type' the only valid types are:{valid_metadata_types} "
|
|
175
|
+
)
|