exadata-validator 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_validator/__init__.py +1 -0
- data_validator/__main__.py +4 -0
- data_validator/commands.py +99 -0
- data_validator/dataelements.py +175 -0
- data_validator/duckdb_validator.py +400 -0
- data_validator/exceptions.py +80 -0
- data_validator/reporting.py +342 -0
- data_validator/usecases.py +159 -0
- exadata_validator-0.0.1.dist-info/METADATA +61 -0
- exadata_validator-0.0.1.dist-info/RECORD +12 -0
- exadata_validator-0.0.1.dist-info/WHEEL +4 -0
- exadata_validator-0.0.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Standalone validator package."""
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from uuid import uuid4
|
|
7
|
+
|
|
8
|
+
import click as cl
|
|
9
|
+
|
|
10
|
+
from data_validator.exceptions import ExitCode
|
|
11
|
+
from data_validator.exceptions import InvalidDatasetError
|
|
12
|
+
from data_validator.exceptions import handle_errors
|
|
13
|
+
from data_validator.reporting import render_report
|
|
14
|
+
from data_validator.usecases import validate_data_model_folder
|
|
15
|
+
|
|
16
|
+
logging.basicConfig(
|
|
17
|
+
stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
|
|
18
|
+
)
|
|
19
|
+
LOGGER = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@cl.group()
|
|
23
|
+
def cli():
|
|
24
|
+
"""exadata-validator command line interface."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _format_report_link(output_path: Path) -> str:
|
|
28
|
+
resolved = output_path.resolve()
|
|
29
|
+
uri = resolved.as_uri()
|
|
30
|
+
label = resolved.name
|
|
31
|
+
if cl.get_text_stream("stdout").isatty():
|
|
32
|
+
# OSC 8 hyperlink sequence for terminals that support clickable links.
|
|
33
|
+
return f"\033]8;;{uri}\033\\{label}\033]8;;\033\\"
|
|
34
|
+
return uri
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _default_html_report_path() -> Path:
|
|
38
|
+
return Path("/tmp") / f"exadata-validator-report-{uuid4().hex}.html"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@cli.command("validate-data-model")
|
|
42
|
+
@cl.argument("folder", type=cl.Path(exists=True, file_okay=False, path_type=Path))
|
|
43
|
+
@cl.option(
|
|
44
|
+
"--threads",
|
|
45
|
+
type=cl.IntRange(min=1),
|
|
46
|
+
default=None,
|
|
47
|
+
help="DuckDB worker threads (defaults to CPU count).",
|
|
48
|
+
)
|
|
49
|
+
@cl.option(
|
|
50
|
+
"--report-all/--fail-fast",
|
|
51
|
+
default=True,
|
|
52
|
+
help="Collect and report all validation errors instead of failing at the first one.",
|
|
53
|
+
)
|
|
54
|
+
@cl.option(
|
|
55
|
+
"--format",
|
|
56
|
+
"output_format",
|
|
57
|
+
type=cl.Choice(["text", "json", "ndjson", "html"], case_sensitive=False),
|
|
58
|
+
default="text",
|
|
59
|
+
show_default=True,
|
|
60
|
+
help="Output format when collecting all validation errors.",
|
|
61
|
+
)
|
|
62
|
+
@cl.option(
|
|
63
|
+
"--output",
|
|
64
|
+
"output_path",
|
|
65
|
+
type=cl.Path(dir_okay=False, writable=True, path_type=Path),
|
|
66
|
+
default=None,
|
|
67
|
+
help="Optional output file path when collecting all validation errors.",
|
|
68
|
+
)
|
|
69
|
+
@handle_errors
|
|
70
|
+
def validate_data_model(
|
|
71
|
+
folder: Path,
|
|
72
|
+
threads: int | None,
|
|
73
|
+
report_all: bool,
|
|
74
|
+
output_format: str,
|
|
75
|
+
output_path: Path | None,
|
|
76
|
+
):
|
|
77
|
+
report = validate_data_model_folder(folder, threads=threads, report_all=report_all)
|
|
78
|
+
|
|
79
|
+
if report_all:
|
|
80
|
+
selected_format = output_format.lower()
|
|
81
|
+
rendered = render_report(report, selected_format)
|
|
82
|
+
target_output_path = output_path
|
|
83
|
+
if selected_format == "html" and target_output_path is None:
|
|
84
|
+
target_output_path = _default_html_report_path()
|
|
85
|
+
|
|
86
|
+
if target_output_path:
|
|
87
|
+
target_output_path.write_text(rendered + "\n", encoding="utf-8")
|
|
88
|
+
cl.echo(f"Report written to {_format_report_link(target_output_path)}")
|
|
89
|
+
else:
|
|
90
|
+
cl.echo(rendered)
|
|
91
|
+
if report.has_errors:
|
|
92
|
+
raise SystemExit(ExitCode.FILE_ERROR)
|
|
93
|
+
elif report.has_errors:
|
|
94
|
+
raise InvalidDatasetError(report.issues[0].message)
|
|
95
|
+
|
|
96
|
+
LOGGER.info("Validation completed successfully for %s", folder)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
entry = cli
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
|
|
4
|
+
from data_validator.exceptions import InvalidDataModelError
|
|
5
|
+
from data_validator.exceptions import UserInputError
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class CommonDataElement:
|
|
10
|
+
code: str
|
|
11
|
+
metadata: str
|
|
12
|
+
|
|
13
|
+
@classmethod
|
|
14
|
+
def from_metadata(cls, metadata: dict):
|
|
15
|
+
code = metadata["code"]
|
|
16
|
+
if not code.isidentifier():
|
|
17
|
+
raise UserInputError(f"CDE: {code} is not a valid python identifier")
|
|
18
|
+
|
|
19
|
+
validate_metadata(code, metadata)
|
|
20
|
+
return cls(code=code, metadata=json.dumps(metadata))
|
|
21
|
+
|
|
22
|
+
def get_enumerations(self):
|
|
23
|
+
parsed = json.loads(self.metadata)
|
|
24
|
+
return parsed["enumerations"] if "enumerations" in parsed else {}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def flatten_cdes(schema_data):
|
|
28
|
+
cdes = []
|
|
29
|
+
|
|
30
|
+
if "variables" in schema_data:
|
|
31
|
+
for metadata in schema_data["variables"]:
|
|
32
|
+
metadata = reformat_metadata(metadata)
|
|
33
|
+
cdes.append(CommonDataElement.from_metadata(metadata))
|
|
34
|
+
|
|
35
|
+
if "groups" in schema_data:
|
|
36
|
+
for group_data in schema_data["groups"]:
|
|
37
|
+
cdes.extend(flatten_cdes(group_data))
|
|
38
|
+
|
|
39
|
+
return cdes
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def get_sql_type_per_column(cdes):
|
|
43
|
+
return {code: json.loads(cde.metadata)["sql_type"] for code, cde in cdes.items()}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def get_cdes_with_min_max(cdes, columns):
|
|
47
|
+
cdes_with_min_max = {}
|
|
48
|
+
for code, cde in cdes.items():
|
|
49
|
+
if code not in columns:
|
|
50
|
+
continue
|
|
51
|
+
|
|
52
|
+
metadata = json.loads(cde.metadata)
|
|
53
|
+
min_value = metadata.get("min")
|
|
54
|
+
max_value = metadata.get("max")
|
|
55
|
+
|
|
56
|
+
if min_value is not None or max_value is not None:
|
|
57
|
+
cdes_with_min_max[code] = (min_value, max_value)
|
|
58
|
+
|
|
59
|
+
return cdes_with_min_max
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def get_cdes_with_enumerations(cdes, columns):
|
|
63
|
+
cdes_with_enumerations = {}
|
|
64
|
+
for code, cde in cdes.items():
|
|
65
|
+
if code not in columns:
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
metadata = json.loads(cde.metadata)
|
|
69
|
+
if metadata["is_categorical"]:
|
|
70
|
+
cdes_with_enumerations[code] = list(metadata["enumerations"].keys())
|
|
71
|
+
|
|
72
|
+
return cdes_with_enumerations
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def get_dataset_enums(cdes):
|
|
76
|
+
return json.loads(cdes["dataset"].metadata)["enumerations"]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def validate_dataset_present_on_cdes_with_proper_format(cdes):
|
|
80
|
+
dataset_cde = [cde for cde in cdes if cde.code == "dataset"]
|
|
81
|
+
if not dataset_cde:
|
|
82
|
+
raise InvalidDataModelError("There is no 'dataset' CDE in the data model.")
|
|
83
|
+
|
|
84
|
+
dataset_metadata = json.loads(dataset_cde[0].metadata)
|
|
85
|
+
if not dataset_metadata["is_categorical"]:
|
|
86
|
+
raise InvalidDataModelError(
|
|
87
|
+
"CDE 'dataset' must have the 'isCategorical' property equal to 'true'."
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
if dataset_metadata["sql_type"] != "text":
|
|
91
|
+
raise InvalidDataModelError(
|
|
92
|
+
"CDE 'dataset' must have the 'sql_type' property equal to 'text'."
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def validate_longitudinal_data_model(cdes):
|
|
97
|
+
subject_id_metadata = None
|
|
98
|
+
visit_id_metadata = None
|
|
99
|
+
|
|
100
|
+
for cde in cdes:
|
|
101
|
+
if cde.code == "subjectid":
|
|
102
|
+
subject_id_metadata = json.loads(cde.metadata)
|
|
103
|
+
elif cde.code == "visitid":
|
|
104
|
+
visit_id_metadata = json.loads(cde.metadata)
|
|
105
|
+
|
|
106
|
+
if not subject_id_metadata:
|
|
107
|
+
raise InvalidDataModelError(
|
|
108
|
+
"There is no 'subjectid' CDE in the longitudinal data model."
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
if not visit_id_metadata:
|
|
112
|
+
raise InvalidDataModelError(
|
|
113
|
+
"There is no 'visitid' CDE in the longitudinal data model."
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
validate_visitid_cde(visit_id_metadata)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def validate_visitid_cde(metadata):
|
|
120
|
+
if not metadata["is_categorical"]:
|
|
121
|
+
raise InvalidDataModelError(
|
|
122
|
+
"CDE 'visitid' must have the 'isCategorical' property equal to 'true'."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
if metadata["sql_type"] != "text":
|
|
126
|
+
raise InvalidDataModelError(
|
|
127
|
+
"CDE 'visitid' must have the 'sql_type' property equal to 'text'."
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if "enumerations" not in metadata:
|
|
131
|
+
raise InvalidDataModelError(
|
|
132
|
+
"CDE 'visitid' must contain the 'enumerations' property."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def reformat_metadata(metadata):
|
|
137
|
+
new_key_assign = {
|
|
138
|
+
"isCategorical": "is_categorical",
|
|
139
|
+
"minValue": "min",
|
|
140
|
+
"maxValue": "max",
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
for old_key, new_key in new_key_assign.items():
|
|
144
|
+
if old_key in metadata:
|
|
145
|
+
metadata[new_key] = metadata.pop(old_key)
|
|
146
|
+
|
|
147
|
+
if "enumerations" in metadata:
|
|
148
|
+
metadata["enumerations"] = {
|
|
149
|
+
enumeration["code"]: enumeration["label"]
|
|
150
|
+
for enumeration in metadata["enumerations"]
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return metadata
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def validate_metadata(code, metadata):
|
|
157
|
+
for element in ["is_categorical", "code", "sql_type", "label", "type"]:
|
|
158
|
+
if element not in metadata:
|
|
159
|
+
raise InvalidDataModelError(
|
|
160
|
+
f"Element: {element} is missing from the CDE {code}"
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
if metadata["is_categorical"] and "enumerations" not in metadata:
|
|
164
|
+
raise InvalidDataModelError(
|
|
165
|
+
f"The CDE {code} has 'is_categorical' set to True but there are no enumerations."
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
if {"min", "max"} <= set(metadata) and metadata["min"] >= metadata["max"]:
|
|
169
|
+
raise InvalidDataModelError(f"The CDE {code} has min greater than the max.")
|
|
170
|
+
|
|
171
|
+
valid_metadata_types = ["nominal", "real", "integer", "text"]
|
|
172
|
+
if metadata["type"] not in valid_metadata_types:
|
|
173
|
+
raise InvalidDataModelError(
|
|
174
|
+
f"The CDE {code} has an 'type' the only valid types are:{valid_metadata_types} "
|
|
175
|
+
)
|
|
@@ -0,0 +1,400 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import difflib
|
|
5
|
+
import os
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Callable
|
|
9
|
+
|
|
10
|
+
import duckdb
|
|
11
|
+
|
|
12
|
+
from data_validator.dataelements import flatten_cdes
|
|
13
|
+
from data_validator.dataelements import get_cdes_with_enumerations
|
|
14
|
+
from data_validator.dataelements import get_cdes_with_min_max
|
|
15
|
+
from data_validator.dataelements import get_dataset_enums
|
|
16
|
+
from data_validator.dataelements import get_sql_type_per_column
|
|
17
|
+
from data_validator.exceptions import InvalidDatasetError
|
|
18
|
+
from data_validator.reporting import ValidationIssue
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _quote_identifier(value: str) -> str:
|
|
22
|
+
return '"' + value.replace('"', '""') + '"'
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _quote_literal(value: str) -> str:
|
|
26
|
+
return "'" + value.replace("'", "''") + "'"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _format_in_list(values: list[str]) -> str:
|
|
30
|
+
return ", ".join(_quote_literal(value) for value in values)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _format_allowed_values(values: list[str], max_items: int = 8) -> str:
|
|
34
|
+
shown = values[:max_items]
|
|
35
|
+
formatted = ", ".join(repr(value) for value in shown)
|
|
36
|
+
if len(values) > max_items:
|
|
37
|
+
return f"{formatted}, ... (total={len(values)})"
|
|
38
|
+
return formatted
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class _FusedCheck:
|
|
43
|
+
key: str
|
|
44
|
+
rule: str
|
|
45
|
+
column: str | None
|
|
46
|
+
condition_sql: str
|
|
47
|
+
sample_expression_sql: str
|
|
48
|
+
message_factory: Callable[[int, str | None], str]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class DuckDBDatasetValidator:
|
|
52
|
+
"""Validate CSV files against a data model using direct DuckDB CSV queries."""
|
|
53
|
+
|
|
54
|
+
def __init__(self, data_model_metadata: dict, threads: int | None = None) -> None:
|
|
55
|
+
cdes_list = flatten_cdes(copy.deepcopy(data_model_metadata))
|
|
56
|
+
self._cdes = {cde.code: cde for cde in cdes_list}
|
|
57
|
+
self._sql_type_per_column = get_sql_type_per_column(self._cdes)
|
|
58
|
+
self._dataset_enumerations = list(get_dataset_enums(self._cdes).keys())
|
|
59
|
+
self._longitudinal = bool(data_model_metadata.get("longitudinal", False))
|
|
60
|
+
self._threads = threads if threads is not None else (os.cpu_count() or 1)
|
|
61
|
+
|
|
62
|
+
def validate_csv(
|
|
63
|
+
self, csv_path: Path, report_all: bool = False
|
|
64
|
+
) -> list[ValidationIssue]:
|
|
65
|
+
issues: list[ValidationIssue] = []
|
|
66
|
+
conn = duckdb.connect(database=":memory:")
|
|
67
|
+
try:
|
|
68
|
+
conn.execute(f"PRAGMA threads={self._threads}")
|
|
69
|
+
|
|
70
|
+
columns = self._get_columns(conn, csv_path)
|
|
71
|
+
issues.extend(self._validate_columns(columns, csv_path))
|
|
72
|
+
|
|
73
|
+
if issues and not report_all:
|
|
74
|
+
raise InvalidDatasetError(issues[0].message)
|
|
75
|
+
|
|
76
|
+
fused_checks = self._build_fused_checks(columns)
|
|
77
|
+
issues.extend(self._run_fused_checks(conn, csv_path, fused_checks))
|
|
78
|
+
|
|
79
|
+
if self._longitudinal and {"subjectid", "visitid"} <= set(columns):
|
|
80
|
+
issues.extend(self._validate_longitudinal_pairs(conn, csv_path))
|
|
81
|
+
except duckdb.Error as exc:
|
|
82
|
+
message = f"Unable to validate csv '{csv_path.name}'. {exc}"
|
|
83
|
+
if report_all:
|
|
84
|
+
issues.append(
|
|
85
|
+
ValidationIssue(
|
|
86
|
+
rule="csv.duckdb_error",
|
|
87
|
+
message=message,
|
|
88
|
+
file=str(csv_path),
|
|
89
|
+
)
|
|
90
|
+
)
|
|
91
|
+
return issues
|
|
92
|
+
raise InvalidDatasetError(message) from exc
|
|
93
|
+
finally:
|
|
94
|
+
conn.close()
|
|
95
|
+
|
|
96
|
+
if issues and not report_all:
|
|
97
|
+
raise InvalidDatasetError(issues[0].message)
|
|
98
|
+
return issues
|
|
99
|
+
|
|
100
|
+
def _get_columns(
|
|
101
|
+
self, conn: duckdb.DuckDBPyConnection, csv_path: Path
|
|
102
|
+
) -> list[str]:
|
|
103
|
+
info = conn.execute(
|
|
104
|
+
(
|
|
105
|
+
"DESCRIBE SELECT * "
|
|
106
|
+
"FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=[''])"
|
|
107
|
+
),
|
|
108
|
+
[str(csv_path)],
|
|
109
|
+
).fetchall()
|
|
110
|
+
return [str(row[0]) for row in info]
|
|
111
|
+
|
|
112
|
+
def _validate_columns(
|
|
113
|
+
self, columns: list[str], csv_path: Path
|
|
114
|
+
) -> list[ValidationIssue]:
|
|
115
|
+
issues: list[ValidationIssue] = []
|
|
116
|
+
|
|
117
|
+
if "dataset" not in columns:
|
|
118
|
+
issues.append(
|
|
119
|
+
ValidationIssue(
|
|
120
|
+
rule="columns.dataset_required",
|
|
121
|
+
message="The 'dataset' column is required to exist in the csv.",
|
|
122
|
+
file=str(csv_path),
|
|
123
|
+
column="dataset",
|
|
124
|
+
)
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
unknown_columns = (
|
|
128
|
+
set(columns) - set(self._sql_type_per_column.keys()) - {"row_id"}
|
|
129
|
+
)
|
|
130
|
+
for column in sorted(unknown_columns):
|
|
131
|
+
suggestion = self._get_column_suggestion(column)
|
|
132
|
+
message = f"Column '{column}' is not present in the CDEs."
|
|
133
|
+
if suggestion:
|
|
134
|
+
message += f" Did you mean '{suggestion}'?"
|
|
135
|
+
issues.append(
|
|
136
|
+
ValidationIssue(
|
|
137
|
+
rule="columns.unknown",
|
|
138
|
+
message=message,
|
|
139
|
+
file=str(csv_path),
|
|
140
|
+
column=column,
|
|
141
|
+
)
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
if self._longitudinal:
|
|
145
|
+
for required in ("subjectid", "visitid"):
|
|
146
|
+
if required not in columns:
|
|
147
|
+
issues.append(
|
|
148
|
+
ValidationIssue(
|
|
149
|
+
rule="longitudinal.required_column",
|
|
150
|
+
message=(
|
|
151
|
+
"The "
|
|
152
|
+
f"'{required}' column is required for longitudinal data models."
|
|
153
|
+
),
|
|
154
|
+
file=str(csv_path),
|
|
155
|
+
column=required,
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
return issues
|
|
160
|
+
|
|
161
|
+
def _get_column_suggestion(self, unknown_column: str) -> str | None:
|
|
162
|
+
candidates = [
|
|
163
|
+
column for column in self._sql_type_per_column.keys() if column != "row_id"
|
|
164
|
+
]
|
|
165
|
+
matches = difflib.get_close_matches(unknown_column, candidates, n=1, cutoff=0.7)
|
|
166
|
+
return matches[0] if matches else None
|
|
167
|
+
|
|
168
|
+
def _build_fused_checks(self, columns: list[str]) -> list[_FusedCheck]:
|
|
169
|
+
checks: list[_FusedCheck] = []
|
|
170
|
+
|
|
171
|
+
type_mapping = {"int": "BIGINT", "real": "DOUBLE"}
|
|
172
|
+
for column in columns:
|
|
173
|
+
sql_type = self._sql_type_per_column.get(column)
|
|
174
|
+
cast_type = type_mapping.get(sql_type)
|
|
175
|
+
if not cast_type:
|
|
176
|
+
continue
|
|
177
|
+
|
|
178
|
+
quoted_column = _quote_identifier(column)
|
|
179
|
+
checks.append(
|
|
180
|
+
_FusedCheck(
|
|
181
|
+
key=f"type_{column}",
|
|
182
|
+
rule="types.invalid",
|
|
183
|
+
column=column,
|
|
184
|
+
condition_sql=(
|
|
185
|
+
f"{quoted_column} IS NOT NULL "
|
|
186
|
+
f"AND TRY_CAST({quoted_column} AS {cast_type}) IS NULL"
|
|
187
|
+
),
|
|
188
|
+
sample_expression_sql=quoted_column,
|
|
189
|
+
message_factory=lambda count, sample, c=column, t=sql_type: (
|
|
190
|
+
f"Column '{c}' has invalid {t} values "
|
|
191
|
+
f"(count={count}, example={sample!r})."
|
|
192
|
+
),
|
|
193
|
+
)
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
cdes_with_min_max = get_cdes_with_min_max(self._cdes, columns)
|
|
197
|
+
for column, (min_value, max_value) in cdes_with_min_max.items():
|
|
198
|
+
quoted_column = _quote_identifier(column)
|
|
199
|
+
if min_value is not None:
|
|
200
|
+
checks.append(
|
|
201
|
+
_FusedCheck(
|
|
202
|
+
key=f"min_{column}",
|
|
203
|
+
rule="range.min",
|
|
204
|
+
column=column,
|
|
205
|
+
condition_sql=(
|
|
206
|
+
f"{quoted_column} IS NOT NULL "
|
|
207
|
+
f"AND TRY_CAST({quoted_column} AS DOUBLE) < {min_value}"
|
|
208
|
+
),
|
|
209
|
+
sample_expression_sql=quoted_column,
|
|
210
|
+
message_factory=lambda count, sample, c=column, v=min_value: (
|
|
211
|
+
f"Column '{c}' has values below minimum {v} "
|
|
212
|
+
f"(count={count}, example={sample!r})."
|
|
213
|
+
),
|
|
214
|
+
)
|
|
215
|
+
)
|
|
216
|
+
if max_value is not None:
|
|
217
|
+
checks.append(
|
|
218
|
+
_FusedCheck(
|
|
219
|
+
key=f"max_{column}",
|
|
220
|
+
rule="range.max",
|
|
221
|
+
column=column,
|
|
222
|
+
condition_sql=(
|
|
223
|
+
f"{quoted_column} IS NOT NULL "
|
|
224
|
+
f"AND TRY_CAST({quoted_column} AS DOUBLE) > {max_value}"
|
|
225
|
+
),
|
|
226
|
+
sample_expression_sql=quoted_column,
|
|
227
|
+
message_factory=lambda count, sample, c=column, v=max_value: (
|
|
228
|
+
f"Column '{c}' has values above maximum {v} "
|
|
229
|
+
f"(count={count}, example={sample!r})."
|
|
230
|
+
),
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
cdes_with_enumerations = get_cdes_with_enumerations(self._cdes, columns)
|
|
235
|
+
for column, allowed_values in cdes_with_enumerations.items():
|
|
236
|
+
if not allowed_values:
|
|
237
|
+
continue
|
|
238
|
+
quoted_column = _quote_identifier(column)
|
|
239
|
+
in_list = _format_in_list(allowed_values)
|
|
240
|
+
allowed_display = _format_allowed_values(allowed_values)
|
|
241
|
+
checks.append(
|
|
242
|
+
_FusedCheck(
|
|
243
|
+
key=f"enum_{column}",
|
|
244
|
+
rule="enum.invalid",
|
|
245
|
+
column=column,
|
|
246
|
+
condition_sql=(
|
|
247
|
+
f"{quoted_column} IS NOT NULL AND {quoted_column} NOT IN ({in_list})"
|
|
248
|
+
),
|
|
249
|
+
sample_expression_sql=quoted_column,
|
|
250
|
+
message_factory=lambda count,
|
|
251
|
+
sample,
|
|
252
|
+
c=column,
|
|
253
|
+
allowed=allowed_display: (
|
|
254
|
+
f"Column '{c}' has invalid categorical value(s). "
|
|
255
|
+
f"Allowed values: [{allowed}]. "
|
|
256
|
+
f"Found {count} invalid row(s), example invalid value: {sample!r}."
|
|
257
|
+
),
|
|
258
|
+
)
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
if "dataset" in columns:
|
|
262
|
+
in_list = _format_in_list(self._dataset_enumerations)
|
|
263
|
+
dataset_allowed_display = _format_allowed_values(self._dataset_enumerations)
|
|
264
|
+
checks.append(
|
|
265
|
+
_FusedCheck(
|
|
266
|
+
key="dataset_enum",
|
|
267
|
+
rule="dataset.enum",
|
|
268
|
+
column="dataset",
|
|
269
|
+
condition_sql=f"dataset IS NOT NULL AND dataset NOT IN ({in_list})",
|
|
270
|
+
sample_expression_sql="dataset",
|
|
271
|
+
message_factory=lambda count,
|
|
272
|
+
sample,
|
|
273
|
+
allowed=dataset_allowed_display: (
|
|
274
|
+
"Column 'dataset' has value(s) not declared in CDEsMetadata "
|
|
275
|
+
f"enumerations. Allowed values: [{allowed}]. "
|
|
276
|
+
f"Found {count} invalid row(s), example invalid value: {sample!r}."
|
|
277
|
+
),
|
|
278
|
+
)
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
if self._longitudinal and "subjectid" in columns:
|
|
282
|
+
checks.append(
|
|
283
|
+
_FusedCheck(
|
|
284
|
+
key="longitudinal_subjectid_null",
|
|
285
|
+
rule="longitudinal.subjectid_null",
|
|
286
|
+
column="subjectid",
|
|
287
|
+
condition_sql="subjectid IS NULL",
|
|
288
|
+
sample_expression_sql=_quote_literal("NULL"),
|
|
289
|
+
message_factory=lambda count, _: (
|
|
290
|
+
"Column 'subjectid' should never contain null values "
|
|
291
|
+
f"(count={count})."
|
|
292
|
+
),
|
|
293
|
+
)
|
|
294
|
+
)
|
|
295
|
+
if self._longitudinal and "visitid" in columns:
|
|
296
|
+
checks.append(
|
|
297
|
+
_FusedCheck(
|
|
298
|
+
key="longitudinal_visitid_null",
|
|
299
|
+
rule="longitudinal.visitid_null",
|
|
300
|
+
column="visitid",
|
|
301
|
+
condition_sql="visitid IS NULL",
|
|
302
|
+
sample_expression_sql=_quote_literal("NULL"),
|
|
303
|
+
message_factory=lambda count, _: (
|
|
304
|
+
"Column 'visitid' should never contain null values "
|
|
305
|
+
f"(count={count})."
|
|
306
|
+
),
|
|
307
|
+
)
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
return checks
|
|
311
|
+
|
|
312
|
+
def _run_fused_checks(
|
|
313
|
+
self,
|
|
314
|
+
conn: duckdb.DuckDBPyConnection,
|
|
315
|
+
csv_path: Path,
|
|
316
|
+
checks: list[_FusedCheck],
|
|
317
|
+
) -> list[ValidationIssue]:
|
|
318
|
+
if not checks:
|
|
319
|
+
return []
|
|
320
|
+
|
|
321
|
+
projections = []
|
|
322
|
+
for check in checks:
|
|
323
|
+
count_alias = _quote_identifier(f"{check.key}__count")
|
|
324
|
+
sample_alias = _quote_identifier(f"{check.key}__sample")
|
|
325
|
+
row_alias = _quote_identifier(f"{check.key}__row")
|
|
326
|
+
projections.append(
|
|
327
|
+
f"SUM(CASE WHEN {check.condition_sql} THEN 1 ELSE 0 END) AS {count_alias}"
|
|
328
|
+
)
|
|
329
|
+
projections.append(
|
|
330
|
+
(
|
|
331
|
+
"MIN(CASE WHEN "
|
|
332
|
+
f"{check.condition_sql} THEN {check.sample_expression_sql} "
|
|
333
|
+
f"ELSE NULL END) AS {sample_alias}"
|
|
334
|
+
)
|
|
335
|
+
)
|
|
336
|
+
projections.append(
|
|
337
|
+
f"MIN(CASE WHEN {check.condition_sql} THEN _rownum ELSE NULL END) AS {row_alias}"
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
query = (
|
|
341
|
+
"WITH csv_data AS ("
|
|
342
|
+
" SELECT row_number() OVER () AS _rownum, * "
|
|
343
|
+
" FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=[''])"
|
|
344
|
+
") "
|
|
345
|
+
"SELECT " + ", ".join(projections) + " FROM csv_data"
|
|
346
|
+
)
|
|
347
|
+
row = conn.execute(query, [str(csv_path)]).fetchone()
|
|
348
|
+
if row is None:
|
|
349
|
+
return []
|
|
350
|
+
|
|
351
|
+
issues: list[ValidationIssue] = []
|
|
352
|
+
for index, check in enumerate(checks):
|
|
353
|
+
count = int(row[index * 3] or 0)
|
|
354
|
+
sample = row[index * 3 + 1]
|
|
355
|
+
first_data_row = row[index * 3 + 2]
|
|
356
|
+
if count <= 0:
|
|
357
|
+
continue
|
|
358
|
+
line_number = (
|
|
359
|
+
int(first_data_row) + 1 if first_data_row is not None else None
|
|
360
|
+
)
|
|
361
|
+
message = check.message_factory(count, sample)
|
|
362
|
+
if line_number is not None:
|
|
363
|
+
message += f" First seen at line {line_number}."
|
|
364
|
+
issues.append(
|
|
365
|
+
ValidationIssue(
|
|
366
|
+
rule=check.rule,
|
|
367
|
+
message=message,
|
|
368
|
+
file=str(csv_path),
|
|
369
|
+
column=check.column,
|
|
370
|
+
row=line_number,
|
|
371
|
+
)
|
|
372
|
+
)
|
|
373
|
+
return issues
|
|
374
|
+
|
|
375
|
+
def _validate_longitudinal_pairs(
|
|
376
|
+
self, conn: duckdb.DuckDBPyConnection, csv_path: Path
|
|
377
|
+
) -> list[ValidationIssue]:
|
|
378
|
+
duplicates = conn.execute(
|
|
379
|
+
(
|
|
380
|
+
"SELECT subjectid, visitid, COUNT(*) AS duplicate_count "
|
|
381
|
+
"FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=['']) "
|
|
382
|
+
"GROUP BY subjectid, visitid "
|
|
383
|
+
"HAVING COUNT(*) > 1 "
|
|
384
|
+
"LIMIT 5"
|
|
385
|
+
),
|
|
386
|
+
[str(csv_path)],
|
|
387
|
+
).fetchall()
|
|
388
|
+
if not duplicates:
|
|
389
|
+
return []
|
|
390
|
+
|
|
391
|
+
return [
|
|
392
|
+
ValidationIssue(
|
|
393
|
+
rule="longitudinal.duplicate_pair",
|
|
394
|
+
message=(
|
|
395
|
+
"Invalid csv: duplicate (visitid, subjectid) pairs detected: "
|
|
396
|
+
f"{[(subject, visit) for subject, visit, _ in duplicates]}"
|
|
397
|
+
),
|
|
398
|
+
file=str(csv_path),
|
|
399
|
+
)
|
|
400
|
+
]
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
from contextlib import contextmanager
|
|
3
|
+
from enum import IntEnum
|
|
4
|
+
from functools import wraps
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class DataBaseError(Exception):
|
|
8
|
+
"""Legacy DB error type."""
|
|
9
|
+
|
|
10
|
+
def __init__(self, message) -> None:
|
|
11
|
+
self.message = message
|
|
12
|
+
super().__init__(message)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class UserInputError(Exception):
|
|
16
|
+
def __init__(self, message) -> None:
|
|
17
|
+
self.message = message
|
|
18
|
+
super().__init__(message)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class FileContentError(Exception):
|
|
22
|
+
def __init__(self, message) -> None:
|
|
23
|
+
self.message = message
|
|
24
|
+
super().__init__(message)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class InvalidDatasetError(Exception):
|
|
28
|
+
def __init__(self, message) -> None:
|
|
29
|
+
self.message = message
|
|
30
|
+
super().__init__(message)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class InvalidDataModelError(Exception):
|
|
34
|
+
def __init__(self, message) -> None:
|
|
35
|
+
self.message = message
|
|
36
|
+
super().__init__(message)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class ForeignKeyError(Exception):
|
|
40
|
+
"""Legacy FK error type."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, message) -> None:
|
|
43
|
+
self.message = message
|
|
44
|
+
super().__init__(message)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class ExitCode(IntEnum):
|
|
48
|
+
OK = 0
|
|
49
|
+
USER_ERROR = 64
|
|
50
|
+
DB_ERROR = 65
|
|
51
|
+
FILE_ERROR = 66
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def handle_errors(func):
|
|
55
|
+
@contextmanager
|
|
56
|
+
def _handle_errors():
|
|
57
|
+
try:
|
|
58
|
+
yield
|
|
59
|
+
except UserInputError as exc:
|
|
60
|
+
print("User input error:\n")
|
|
61
|
+
print(f"\t{exc.message}")
|
|
62
|
+
sys.exit(ExitCode.USER_ERROR)
|
|
63
|
+
except DataBaseError as exc:
|
|
64
|
+
print("Database error:\n")
|
|
65
|
+
print(f"\t{exc.message}")
|
|
66
|
+
sys.exit(ExitCode.DB_ERROR)
|
|
67
|
+
except (FileContentError, InvalidDatasetError, InvalidDataModelError) as exc:
|
|
68
|
+
print(f"\nValidation error: {exc.message}")
|
|
69
|
+
sys.exit(ExitCode.FILE_ERROR)
|
|
70
|
+
except ForeignKeyError as exc:
|
|
71
|
+
print("Foreign key error:\n")
|
|
72
|
+
print(f"\t{exc.message}")
|
|
73
|
+
sys.exit(ExitCode.USER_ERROR)
|
|
74
|
+
|
|
75
|
+
@wraps(func)
|
|
76
|
+
def wrapper(*args, **kwargs):
|
|
77
|
+
with _handle_errors():
|
|
78
|
+
return func(*args, **kwargs)
|
|
79
|
+
|
|
80
|
+
return wrapper
|
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from dataclasses import field
|
|
6
|
+
from datetime import datetime
|
|
7
|
+
from datetime import timezone
|
|
8
|
+
from html import escape
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class ValidationIssue:
|
|
14
|
+
rule: str
|
|
15
|
+
message: str
|
|
16
|
+
file: str | None = None
|
|
17
|
+
column: str | None = None
|
|
18
|
+
row: int | None = None
|
|
19
|
+
severity: str = "error"
|
|
20
|
+
|
|
21
|
+
def to_dict(self) -> dict:
|
|
22
|
+
return {
|
|
23
|
+
"severity": self.severity,
|
|
24
|
+
"rule": self.rule,
|
|
25
|
+
"message": self.message,
|
|
26
|
+
"file": self.file,
|
|
27
|
+
"column": self.column,
|
|
28
|
+
"row": self.row,
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ValidationReport:
|
|
34
|
+
folder: str
|
|
35
|
+
files_checked: list[str] = field(default_factory=list)
|
|
36
|
+
issues: list[ValidationIssue] = field(default_factory=list)
|
|
37
|
+
generated_at: str = field(
|
|
38
|
+
default_factory=lambda: datetime.now(timezone.utc).isoformat()
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def has_errors(self) -> bool:
|
|
43
|
+
return bool(self.issues)
|
|
44
|
+
|
|
45
|
+
def to_dict(self) -> dict:
|
|
46
|
+
issues_by_file: dict[str, list[dict]] = {}
|
|
47
|
+
for file_key, grouped_issues in _group_issues_by_file(self):
|
|
48
|
+
key = file_key if file_key is not None else "__folder__"
|
|
49
|
+
issues_by_file[key] = [issue.to_dict() for issue in grouped_issues]
|
|
50
|
+
|
|
51
|
+
return {
|
|
52
|
+
"folder": self.folder,
|
|
53
|
+
"generated_at": self.generated_at,
|
|
54
|
+
"files_checked": self.files_checked,
|
|
55
|
+
"error_count": len(self.issues),
|
|
56
|
+
"issues": [issue.to_dict() for issue in self.issues],
|
|
57
|
+
"issues_by_file": issues_by_file,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def render_report(report: ValidationReport, output_format: str) -> str:
|
|
62
|
+
if output_format == "text":
|
|
63
|
+
return _render_text(report)
|
|
64
|
+
if output_format == "json":
|
|
65
|
+
return json.dumps(report.to_dict(), indent=2)
|
|
66
|
+
if output_format == "ndjson":
|
|
67
|
+
return _render_ndjson(report)
|
|
68
|
+
if output_format == "html":
|
|
69
|
+
return _render_html(report)
|
|
70
|
+
raise ValueError(f"Unsupported report format: {output_format}")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _group_issues_by_file(
|
|
74
|
+
report: ValidationReport,
|
|
75
|
+
) -> list[tuple[str | None, list[ValidationIssue]]]:
|
|
76
|
+
grouped: dict[str | None, list[ValidationIssue]] = {}
|
|
77
|
+
for issue in report.issues:
|
|
78
|
+
grouped.setdefault(issue.file, []).append(issue)
|
|
79
|
+
|
|
80
|
+
ordered: list[tuple[str | None, list[ValidationIssue]]] = []
|
|
81
|
+
for file_path in report.files_checked:
|
|
82
|
+
issues = grouped.pop(file_path, None)
|
|
83
|
+
if issues:
|
|
84
|
+
ordered.append((file_path, issues))
|
|
85
|
+
|
|
86
|
+
for file_path, issues in grouped.items():
|
|
87
|
+
ordered.append((file_path, issues))
|
|
88
|
+
|
|
89
|
+
return ordered
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _render_text(report: ValidationReport) -> str:
|
|
93
|
+
lines = [
|
|
94
|
+
f"Validation report for: {report.folder}",
|
|
95
|
+
f"Files checked: {len(report.files_checked)}",
|
|
96
|
+
f"Errors: {len(report.issues)}",
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
grouped_issues = _group_issues_by_file(report)
|
|
100
|
+
if not grouped_issues:
|
|
101
|
+
lines.append("No validation errors were found.")
|
|
102
|
+
return "\n".join(lines)
|
|
103
|
+
|
|
104
|
+
index = 1
|
|
105
|
+
for file_path, issues in grouped_issues:
|
|
106
|
+
if file_path is None:
|
|
107
|
+
lines.append("Folder-level issues:")
|
|
108
|
+
else:
|
|
109
|
+
lines.append(f"CSV: {file_path}")
|
|
110
|
+
for issue in issues:
|
|
111
|
+
location_parts = []
|
|
112
|
+
if issue.column:
|
|
113
|
+
location_parts.append(f"column={issue.column}")
|
|
114
|
+
if issue.row is not None:
|
|
115
|
+
location_parts.append(f"line={issue.row}")
|
|
116
|
+
location = ", ".join(location_parts) if location_parts else "global"
|
|
117
|
+
lines.append(f"{index}. [{issue.rule}] {location}: {issue.message}")
|
|
118
|
+
index += 1
|
|
119
|
+
return "\n".join(lines)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _render_ndjson(report: ValidationReport) -> str:
|
|
123
|
+
lines = [
|
|
124
|
+
json.dumps(
|
|
125
|
+
{
|
|
126
|
+
"type": "summary",
|
|
127
|
+
"folder": report.folder,
|
|
128
|
+
"generated_at": report.generated_at,
|
|
129
|
+
"files_checked": len(report.files_checked),
|
|
130
|
+
"error_count": len(report.issues),
|
|
131
|
+
}
|
|
132
|
+
)
|
|
133
|
+
]
|
|
134
|
+
for issue in report.issues:
|
|
135
|
+
lines.append(
|
|
136
|
+
json.dumps(
|
|
137
|
+
{
|
|
138
|
+
"type": "issue",
|
|
139
|
+
"folder": report.folder,
|
|
140
|
+
**issue.to_dict(),
|
|
141
|
+
}
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
return "\n".join(lines)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _render_html(report: ValidationReport) -> str:
|
|
148
|
+
grouped_sections: list[str] = []
|
|
149
|
+
grouped_issues = _group_issues_by_file(report)
|
|
150
|
+
for file_path, issues in grouped_issues:
|
|
151
|
+
section_title = "Folder-level issues"
|
|
152
|
+
title_tooltip = "Folder-level validation issues"
|
|
153
|
+
if file_path is not None:
|
|
154
|
+
file_name = Path(file_path).name
|
|
155
|
+
section_title = file_name
|
|
156
|
+
title_tooltip = file_path
|
|
157
|
+
|
|
158
|
+
rows = []
|
|
159
|
+
for issue in issues:
|
|
160
|
+
severity_class = f"severity-{escape(issue.severity).lower()}"
|
|
161
|
+
rows.append(
|
|
162
|
+
"<tr>"
|
|
163
|
+
f"<td><span class='severity {severity_class}'>{escape(issue.severity)}</span></td>"
|
|
164
|
+
f"<td>{escape(issue.rule)}</td>"
|
|
165
|
+
f"<td>{escape(issue.column or '')}</td>"
|
|
166
|
+
f"<td>{'' if issue.row is None else issue.row}</td>"
|
|
167
|
+
f"<td>{escape(issue.message)}</td>"
|
|
168
|
+
"</tr>"
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
table_rows = "\n".join(rows)
|
|
172
|
+
grouped_sections.append(
|
|
173
|
+
"<section class='file-section'>"
|
|
174
|
+
"<div class='file-header'>"
|
|
175
|
+
f"<h2 class='file-title' title='{escape(title_tooltip)}'>{escape(section_title)}</h2>"
|
|
176
|
+
f"<p class='file-count'>{len(issues)} error(s)</p>"
|
|
177
|
+
"</div>"
|
|
178
|
+
"<div class='table-wrap'>"
|
|
179
|
+
"<table>"
|
|
180
|
+
"<colgroup>"
|
|
181
|
+
"<col style='width:90px'>"
|
|
182
|
+
"<col style='width:150px'>"
|
|
183
|
+
"<col style='width:22%'>"
|
|
184
|
+
"<col style='width:70px'>"
|
|
185
|
+
"<col>"
|
|
186
|
+
"</colgroup>"
|
|
187
|
+
"<thead><tr>"
|
|
188
|
+
"<th>Severity</th><th>Rule</th><th>Column</th><th>Line</th><th>Message</th>"
|
|
189
|
+
"</tr></thead><tbody>"
|
|
190
|
+
f"{table_rows}"
|
|
191
|
+
"</tbody></table></div></section>"
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
sections_html = (
|
|
195
|
+
"".join(grouped_sections)
|
|
196
|
+
if grouped_sections
|
|
197
|
+
else (
|
|
198
|
+
"<div class='table-wrap'><table><thead><tr>"
|
|
199
|
+
"<th>Severity</th><th>Rule</th><th>Column</th><th>Line</th><th>Message</th>"
|
|
200
|
+
"</tr></thead><tbody><tr><td colspan='5'><div class='empty-state'>"
|
|
201
|
+
"No validation errors were found."
|
|
202
|
+
"</div></td></tr></tbody></table></div>"
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
return (
|
|
206
|
+
"<!doctype html>"
|
|
207
|
+
"<html><head><meta charset='utf-8'>"
|
|
208
|
+
"<meta name='viewport' content='width=device-width, initial-scale=1'>"
|
|
209
|
+
"<title>Data Validation Report</title>"
|
|
210
|
+
"<style>"
|
|
211
|
+
":root{"
|
|
212
|
+
"--bg:#f4f7fb;"
|
|
213
|
+
"--card:#ffffff;"
|
|
214
|
+
"--text:#12263a;"
|
|
215
|
+
"--muted:#5a6b7c;"
|
|
216
|
+
"--border:#d7e0ea;"
|
|
217
|
+
"--header:#e8eef6;"
|
|
218
|
+
"--accent:#0c6d9a;"
|
|
219
|
+
"--error-bg:#ffe9e8;"
|
|
220
|
+
"--error-text:#8f1e18;"
|
|
221
|
+
"}"
|
|
222
|
+
"body{"
|
|
223
|
+
"margin:0;"
|
|
224
|
+
"font-family:'Avenir Next','Trebuchet MS','Gill Sans',sans-serif;"
|
|
225
|
+
"background:linear-gradient(160deg,#f7fbff 0%,#eef4fa 100%);"
|
|
226
|
+
"color:var(--text);"
|
|
227
|
+
"}"
|
|
228
|
+
".container{"
|
|
229
|
+
"max-width:1120px;"
|
|
230
|
+
"margin:32px auto;"
|
|
231
|
+
"padding:0 20px;"
|
|
232
|
+
"}"
|
|
233
|
+
".card{"
|
|
234
|
+
"background:var(--card);"
|
|
235
|
+
"border:1px solid var(--border);"
|
|
236
|
+
"border-radius:16px;"
|
|
237
|
+
"box-shadow:0 12px 30px rgba(8,39,64,.08);"
|
|
238
|
+
"overflow:hidden;"
|
|
239
|
+
"}"
|
|
240
|
+
".hero{"
|
|
241
|
+
"padding:24px 28px 18px 28px;"
|
|
242
|
+
"background:linear-gradient(135deg,#edf4fb 0%,#e4edf7 100%);"
|
|
243
|
+
"border-bottom:1px solid var(--border);"
|
|
244
|
+
"}"
|
|
245
|
+
".title{"
|
|
246
|
+
"margin:0;"
|
|
247
|
+
"font-size:28px;"
|
|
248
|
+
"letter-spacing:.3px;"
|
|
249
|
+
"}"
|
|
250
|
+
".subtitle{"
|
|
251
|
+
"margin:6px 0 0 0;"
|
|
252
|
+
"color:var(--muted);"
|
|
253
|
+
"font-size:14px;"
|
|
254
|
+
"}"
|
|
255
|
+
".summary{"
|
|
256
|
+
"display:grid;"
|
|
257
|
+
"grid-template-columns:repeat(auto-fit,minmax(180px,1fr));"
|
|
258
|
+
"gap:12px;"
|
|
259
|
+
"padding:18px 24px 22px 24px;"
|
|
260
|
+
"}"
|
|
261
|
+
".stat{"
|
|
262
|
+
"background:#f8fbff;"
|
|
263
|
+
"border:1px solid var(--border);"
|
|
264
|
+
"border-radius:12px;"
|
|
265
|
+
"padding:12px 14px;"
|
|
266
|
+
"}"
|
|
267
|
+
".stat-label{"
|
|
268
|
+
"margin:0 0 6px 0;"
|
|
269
|
+
"font-size:12px;"
|
|
270
|
+
"text-transform:uppercase;"
|
|
271
|
+
"letter-spacing:.08em;"
|
|
272
|
+
"color:var(--muted);"
|
|
273
|
+
"}"
|
|
274
|
+
".stat-value{"
|
|
275
|
+
"margin:0;"
|
|
276
|
+
"font-size:16px;"
|
|
277
|
+
"font-weight:600;"
|
|
278
|
+
"line-height:1.3;"
|
|
279
|
+
"word-break:break-word;"
|
|
280
|
+
"}"
|
|
281
|
+
".file-section{margin:0 16px 18px 16px;border:1px solid var(--border);border-radius:12px;overflow:hidden;background:#fff;}"
|
|
282
|
+
".file-header{display:flex;justify-content:space-between;gap:12px;align-items:center;padding:12px 14px;background:#f3f8fe;border-bottom:1px solid var(--border);}"
|
|
283
|
+
".file-title{margin:0;font-size:16px;font-weight:700;cursor:help;word-break:break-word;}"
|
|
284
|
+
".file-count{margin:0;font-size:12px;color:var(--muted);font-weight:600;text-transform:uppercase;letter-spacing:.06em;}"
|
|
285
|
+
".table-wrap{padding:0;overflow-x:auto;}"
|
|
286
|
+
"table{border-collapse:collapse;width:100%;min-width:920px;background:#fff;table-layout:fixed;}"
|
|
287
|
+
"th,td{padding:10px 12px;text-align:left;vertical-align:top;border-bottom:1px solid var(--border);}"
|
|
288
|
+
"th{"
|
|
289
|
+
"background:var(--header);"
|
|
290
|
+
"font-size:12px;"
|
|
291
|
+
"text-transform:uppercase;"
|
|
292
|
+
"letter-spacing:.06em;"
|
|
293
|
+
"color:#27445d;"
|
|
294
|
+
"position:sticky;"
|
|
295
|
+
"top:0;"
|
|
296
|
+
"}"
|
|
297
|
+
"tbody tr:nth-child(even){background:#fbfdff;}"
|
|
298
|
+
"tbody tr:hover{background:#f2f8fe;}"
|
|
299
|
+
".severity{"
|
|
300
|
+
"display:inline-block;"
|
|
301
|
+
"padding:3px 8px;"
|
|
302
|
+
"border-radius:999px;"
|
|
303
|
+
"font-size:11px;"
|
|
304
|
+
"font-weight:700;"
|
|
305
|
+
"text-transform:uppercase;"
|
|
306
|
+
"letter-spacing:.05em;"
|
|
307
|
+
"}"
|
|
308
|
+
".severity-error{background:var(--error-bg);color:var(--error-text);}"
|
|
309
|
+
".empty-state{padding:18px 0;color:var(--muted);font-style:italic;}"
|
|
310
|
+
"@media (max-width:700px){"
|
|
311
|
+
".container{margin:18px auto;padding:0 10px;}"
|
|
312
|
+
".hero{padding:18px 16px;}"
|
|
313
|
+
".summary{padding:12px 14px 16px 14px;}"
|
|
314
|
+
".title{font-size:22px;}"
|
|
315
|
+
"}"
|
|
316
|
+
"</style></head><body>"
|
|
317
|
+
"<div class='container'><section class='card'>"
|
|
318
|
+
"<header class='hero'>"
|
|
319
|
+
"<h1 class='title'>Data Validation Report</h1>"
|
|
320
|
+
"<p class='subtitle'>Generated from exadata-validator</p>"
|
|
321
|
+
"</header>"
|
|
322
|
+
"<section class='summary'>"
|
|
323
|
+
"<article class='stat'>"
|
|
324
|
+
"<p class='stat-label'>Folder</p>"
|
|
325
|
+
f"<p class='stat-value'>{escape(report.folder)}</p>"
|
|
326
|
+
"</article>"
|
|
327
|
+
"<article class='stat'>"
|
|
328
|
+
"<p class='stat-label'>Generated</p>"
|
|
329
|
+
f"<p class='stat-value'>{escape(report.generated_at)}</p>"
|
|
330
|
+
"</article>"
|
|
331
|
+
"<article class='stat'>"
|
|
332
|
+
"<p class='stat-label'>Files Checked</p>"
|
|
333
|
+
f"<p class='stat-value'>{len(report.files_checked)}</p>"
|
|
334
|
+
"</article>"
|
|
335
|
+
"<article class='stat'>"
|
|
336
|
+
"<p class='stat-label'>Errors</p>"
|
|
337
|
+
f"<p class='stat-value'>{len(report.issues)}</p>"
|
|
338
|
+
"</article>"
|
|
339
|
+
"</section>"
|
|
340
|
+
f"{sections_html}"
|
|
341
|
+
"</section></div></body></html>"
|
|
342
|
+
)
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import duckdb
|
|
8
|
+
|
|
9
|
+
from data_validator.dataelements import flatten_cdes
|
|
10
|
+
from data_validator.dataelements import (
|
|
11
|
+
validate_dataset_present_on_cdes_with_proper_format,
|
|
12
|
+
)
|
|
13
|
+
from data_validator.dataelements import validate_longitudinal_data_model
|
|
14
|
+
from data_validator.duckdb_validator import DuckDBDatasetValidator
|
|
15
|
+
from data_validator.exceptions import FileContentError
|
|
16
|
+
from data_validator.exceptions import InvalidDataModelError
|
|
17
|
+
from data_validator.exceptions import InvalidDatasetError
|
|
18
|
+
from data_validator.exceptions import UserInputError
|
|
19
|
+
from data_validator.reporting import ValidationIssue
|
|
20
|
+
from data_validator.reporting import ValidationReport
|
|
21
|
+
|
|
22
|
+
LONGITUDINAL = "longitudinal"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _quote_literal(value: str) -> str:
|
|
26
|
+
return "'" + value.replace("'", "''") + "'"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _read_json_file(path: Path) -> dict:
|
|
30
|
+
with path.open("r", encoding="utf-8") as stream:
|
|
31
|
+
try:
|
|
32
|
+
return json.load(stream)
|
|
33
|
+
except json.JSONDecodeError as exc:
|
|
34
|
+
raise FileContentError(
|
|
35
|
+
f"Unable to decode json file. {exc.args[0]}"
|
|
36
|
+
) from exc
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def validate_data_model_metadata(data_model_metadata: dict) -> None:
|
|
40
|
+
if "version" not in data_model_metadata:
|
|
41
|
+
raise UserInputError("You need to include a version on the CDEsMetadata.json")
|
|
42
|
+
|
|
43
|
+
cdes = flatten_cdes(copy.deepcopy(data_model_metadata))
|
|
44
|
+
validate_dataset_present_on_cdes_with_proper_format(cdes)
|
|
45
|
+
|
|
46
|
+
if LONGITUDINAL in data_model_metadata:
|
|
47
|
+
longitudinal = data_model_metadata[LONGITUDINAL]
|
|
48
|
+
if not isinstance(longitudinal, bool):
|
|
49
|
+
raise UserInputError(
|
|
50
|
+
f"Longitudinal flag should be boolean, value given: {longitudinal}"
|
|
51
|
+
)
|
|
52
|
+
if longitudinal:
|
|
53
|
+
validate_longitudinal_data_model(cdes)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _validate_dataset_uniqueness_with_sql(
|
|
57
|
+
csv_files: list[Path],
|
|
58
|
+
) -> list[ValidationIssue]:
|
|
59
|
+
if len(csv_files) < 2:
|
|
60
|
+
return []
|
|
61
|
+
|
|
62
|
+
file_list_sql = ", ".join(_quote_literal(str(path)) for path in csv_files)
|
|
63
|
+
query = (
|
|
64
|
+
"WITH csv_rows AS ("
|
|
65
|
+
" SELECT filename, dataset "
|
|
66
|
+
f" FROM read_csv_auto([{file_list_sql}], "
|
|
67
|
+
" header=true, all_varchar=true, nullstr=[''], "
|
|
68
|
+
" filename=true, union_by_name=true)"
|
|
69
|
+
"), normalized AS ("
|
|
70
|
+
" SELECT filename, NULLIF(LOWER(TRIM(dataset)), '') AS dataset_normalized "
|
|
71
|
+
" FROM csv_rows "
|
|
72
|
+
" WHERE dataset IS NOT NULL"
|
|
73
|
+
") "
|
|
74
|
+
"SELECT dataset_normalized, list(DISTINCT filename) AS files "
|
|
75
|
+
"FROM normalized "
|
|
76
|
+
"WHERE dataset_normalized IS NOT NULL "
|
|
77
|
+
"GROUP BY dataset_normalized "
|
|
78
|
+
"HAVING COUNT(DISTINCT filename) > 1 "
|
|
79
|
+
"ORDER BY dataset_normalized"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
conn = duckdb.connect(database=":memory:")
|
|
83
|
+
try:
|
|
84
|
+
rows = conn.execute(query).fetchall()
|
|
85
|
+
except duckdb.Error as exc:
|
|
86
|
+
raise InvalidDatasetError(
|
|
87
|
+
f"Unable to validate folder-level dataset uniqueness: {exc}"
|
|
88
|
+
) from exc
|
|
89
|
+
finally:
|
|
90
|
+
conn.close()
|
|
91
|
+
|
|
92
|
+
issues: list[ValidationIssue] = []
|
|
93
|
+
for dataset_normalized, files in rows:
|
|
94
|
+
file_names = sorted(Path(str(path)).name for path in files)
|
|
95
|
+
issues.append(
|
|
96
|
+
ValidationIssue(
|
|
97
|
+
rule="folder.dataset_uniqueness",
|
|
98
|
+
message=(
|
|
99
|
+
"Dataset code collision after normalization (trim+lower) for "
|
|
100
|
+
f"'{dataset_normalized}' across files: {file_names}"
|
|
101
|
+
),
|
|
102
|
+
)
|
|
103
|
+
)
|
|
104
|
+
return issues
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def validate_data_model_folder(
|
|
108
|
+
folder: Path, *, threads: int | None = None, report_all: bool = False
|
|
109
|
+
) -> ValidationReport:
|
|
110
|
+
report = ValidationReport(folder=str(folder))
|
|
111
|
+
|
|
112
|
+
metadata_path = folder / "CDEsMetadata.json"
|
|
113
|
+
if not metadata_path.exists():
|
|
114
|
+
raise UserInputError(f"Missing metadata file: {metadata_path}")
|
|
115
|
+
|
|
116
|
+
data_model_metadata = _read_json_file(metadata_path)
|
|
117
|
+
try:
|
|
118
|
+
validate_data_model_metadata(data_model_metadata)
|
|
119
|
+
except (InvalidDataModelError, UserInputError) as exc:
|
|
120
|
+
if not report_all:
|
|
121
|
+
raise
|
|
122
|
+
report.issues.append(
|
|
123
|
+
ValidationIssue(
|
|
124
|
+
rule="metadata.invalid",
|
|
125
|
+
message=str(exc),
|
|
126
|
+
file=str(metadata_path),
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
return report
|
|
130
|
+
|
|
131
|
+
csv_files = sorted(folder.glob("*.csv"))
|
|
132
|
+
if not csv_files:
|
|
133
|
+
raise UserInputError(
|
|
134
|
+
f"No CSV files found in {folder}. Expected at least one dataset csv."
|
|
135
|
+
)
|
|
136
|
+
report.files_checked = [str(csv_file) for csv_file in csv_files]
|
|
137
|
+
|
|
138
|
+
validator = DuckDBDatasetValidator(data_model_metadata, threads=threads)
|
|
139
|
+
for csv_file in csv_files:
|
|
140
|
+
csv_issues = validator.validate_csv(csv_file, report_all=report_all)
|
|
141
|
+
report.issues.extend(csv_issues)
|
|
142
|
+
|
|
143
|
+
try:
|
|
144
|
+
duplicate_issues = _validate_dataset_uniqueness_with_sql(csv_files)
|
|
145
|
+
report.issues.extend(duplicate_issues)
|
|
146
|
+
except InvalidDatasetError as exc:
|
|
147
|
+
if not report_all:
|
|
148
|
+
raise
|
|
149
|
+
report.issues.append(
|
|
150
|
+
ValidationIssue(
|
|
151
|
+
rule="folder.dataset_uniqueness",
|
|
152
|
+
message=str(exc),
|
|
153
|
+
)
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
if report.issues and not report_all:
|
|
157
|
+
raise InvalidDatasetError(report.issues[0].message)
|
|
158
|
+
|
|
159
|
+
return report
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: exadata-validator
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Validate data-model folders using DuckDB.
|
|
5
|
+
Author: Exaflow Team
|
|
6
|
+
Requires-Python: >=3.10,<3.11
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Requires-Dist: click (>=8.1,<8.2)
|
|
10
|
+
Requires-Dist: duckdb (>=1.1,<1.2)
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# exadata-validator
|
|
14
|
+
|
|
15
|
+
`exadata-validator` validates data-model folders using DuckDB.
|
|
16
|
+
|
|
17
|
+
## Install With pip
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
python -m venv .venv
|
|
21
|
+
source .venv/bin/activate
|
|
22
|
+
pip install exadata-validator
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Validate a data model folder:
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
exadata-validator validate-data-model /path/to/data_model_folder
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Develop With Poetry
|
|
32
|
+
|
|
33
|
+
From the repository root:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
cd data-validator/exaflow-data-validator
|
|
37
|
+
poetry install
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Then run the same CLI through Poetry:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
poetry run exadata-validator validate-data-model /path/to/data_model_folder
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
|
|
47
|
+
|
|
48
|
+
## Folder Layout
|
|
49
|
+
|
|
50
|
+
```text
|
|
51
|
+
/path/to/data_model_folder/
|
|
52
|
+
CDEsMetadata.json
|
|
53
|
+
dataset1.csv
|
|
54
|
+
dataset2.csv
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Validation Notes
|
|
58
|
+
|
|
59
|
+
- CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
|
|
60
|
+
- Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
|
|
61
|
+
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
data_validator/__init__.py,sha256=Tu8Zffh4uJ1ryN2xbmt6NH3Rmtq7iCHkvurX-bWo8uY,36
|
|
2
|
+
data_validator/__main__.py,sha256=roEHUIHJaAQBK-dtqzNdaE9kUG6qaC-ZOoI4M-vuA8A,82
|
|
3
|
+
data_validator/commands.py,sha256=PHWSAfGllMCmK1xsT7oUy8v8jTcVGke6gp79kcWIt-c,2971
|
|
4
|
+
data_validator/dataelements.py,sha256=sElw3t-MvgeTwqC5G0RGFuFioTYIiiu6eKAZkCqEA-0,5392
|
|
5
|
+
data_validator/duckdb_validator.py,sha256=OBqK8ATi4zsAl7g1idv2CBlOLdDvRCtZciq9isdF8mc,15446
|
|
6
|
+
data_validator/exceptions.py,sha256=hEWv45Y_eIPJKwIuK3HJ9Cefay4PINfVcUfjgMFiODU,2030
|
|
7
|
+
data_validator/reporting.py,sha256=pdeJ-ad_fOuYKt1jZqY0iFD433UWh8I5w5m011n6Moo,11622
|
|
8
|
+
data_validator/usecases.py,sha256=NGvAmnxngFiMiiCY_3bz2zOQWAT1Oa3HO-F5ALm4H9o,5353
|
|
9
|
+
exadata_validator-0.0.1.dist-info/METADATA,sha256=xTI6S5nld1kObK9jyCjFSIVLcQ3FI8aPPVzED-UjuP8,1379
|
|
10
|
+
exadata_validator-0.0.1.dist-info/WHEEL,sha256=kJCRJT_g0adfAJzTx2GUMmS80rTJIVHRCfG0DQgLq3o,88
|
|
11
|
+
exadata_validator-0.0.1.dist-info/entry_points.txt,sha256=50_GAf8IaFjGTrft9Iy0tso4G3v-R2ITd4C8Sf7gmMI,67
|
|
12
|
+
exadata_validator-0.0.1.dist-info/RECORD,,
|