exadata-validator 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ """Standalone validator package."""
@@ -0,0 +1,4 @@
1
+ from data_validator.commands import entry
2
+
3
+ if __name__ == "__main__":
4
+ entry()
@@ -0,0 +1,99 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import sys
5
+ from pathlib import Path
6
+ from uuid import uuid4
7
+
8
+ import click as cl
9
+
10
+ from data_validator.exceptions import ExitCode
11
+ from data_validator.exceptions import InvalidDatasetError
12
+ from data_validator.exceptions import handle_errors
13
+ from data_validator.reporting import render_report
14
+ from data_validator.usecases import validate_data_model_folder
15
+
16
+ logging.basicConfig(
17
+ stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
18
+ )
19
+ LOGGER = logging.getLogger(__name__)
20
+
21
+
22
+ @cl.group()
23
+ def cli():
24
+ """exadata-validator command line interface."""
25
+
26
+
27
+ def _format_report_link(output_path: Path) -> str:
28
+ resolved = output_path.resolve()
29
+ uri = resolved.as_uri()
30
+ label = resolved.name
31
+ if cl.get_text_stream("stdout").isatty():
32
+ # OSC 8 hyperlink sequence for terminals that support clickable links.
33
+ return f"\033]8;;{uri}\033\\{label}\033]8;;\033\\"
34
+ return uri
35
+
36
+
37
+ def _default_html_report_path() -> Path:
38
+ return Path("/tmp") / f"exadata-validator-report-{uuid4().hex}.html"
39
+
40
+
41
+ @cli.command("validate-data-model")
42
+ @cl.argument("folder", type=cl.Path(exists=True, file_okay=False, path_type=Path))
43
+ @cl.option(
44
+ "--threads",
45
+ type=cl.IntRange(min=1),
46
+ default=None,
47
+ help="DuckDB worker threads (defaults to CPU count).",
48
+ )
49
+ @cl.option(
50
+ "--report-all/--fail-fast",
51
+ default=True,
52
+ help="Collect and report all validation errors instead of failing at the first one.",
53
+ )
54
+ @cl.option(
55
+ "--format",
56
+ "output_format",
57
+ type=cl.Choice(["text", "json", "ndjson", "html"], case_sensitive=False),
58
+ default="text",
59
+ show_default=True,
60
+ help="Output format when collecting all validation errors.",
61
+ )
62
+ @cl.option(
63
+ "--output",
64
+ "output_path",
65
+ type=cl.Path(dir_okay=False, writable=True, path_type=Path),
66
+ default=None,
67
+ help="Optional output file path when collecting all validation errors.",
68
+ )
69
+ @handle_errors
70
+ def validate_data_model(
71
+ folder: Path,
72
+ threads: int | None,
73
+ report_all: bool,
74
+ output_format: str,
75
+ output_path: Path | None,
76
+ ):
77
+ report = validate_data_model_folder(folder, threads=threads, report_all=report_all)
78
+
79
+ if report_all:
80
+ selected_format = output_format.lower()
81
+ rendered = render_report(report, selected_format)
82
+ target_output_path = output_path
83
+ if selected_format == "html" and target_output_path is None:
84
+ target_output_path = _default_html_report_path()
85
+
86
+ if target_output_path:
87
+ target_output_path.write_text(rendered + "\n", encoding="utf-8")
88
+ cl.echo(f"Report written to {_format_report_link(target_output_path)}")
89
+ else:
90
+ cl.echo(rendered)
91
+ if report.has_errors:
92
+ raise SystemExit(ExitCode.FILE_ERROR)
93
+ elif report.has_errors:
94
+ raise InvalidDatasetError(report.issues[0].message)
95
+
96
+ LOGGER.info("Validation completed successfully for %s", folder)
97
+
98
+
99
+ entry = cli
@@ -0,0 +1,175 @@
1
+ import json
2
+ from dataclasses import dataclass
3
+
4
+ from data_validator.exceptions import InvalidDataModelError
5
+ from data_validator.exceptions import UserInputError
6
+
7
+
8
+ @dataclass
9
+ class CommonDataElement:
10
+ code: str
11
+ metadata: str
12
+
13
+ @classmethod
14
+ def from_metadata(cls, metadata: dict):
15
+ code = metadata["code"]
16
+ if not code.isidentifier():
17
+ raise UserInputError(f"CDE: {code} is not a valid python identifier")
18
+
19
+ validate_metadata(code, metadata)
20
+ return cls(code=code, metadata=json.dumps(metadata))
21
+
22
+ def get_enumerations(self):
23
+ parsed = json.loads(self.metadata)
24
+ return parsed["enumerations"] if "enumerations" in parsed else {}
25
+
26
+
27
+ def flatten_cdes(schema_data):
28
+ cdes = []
29
+
30
+ if "variables" in schema_data:
31
+ for metadata in schema_data["variables"]:
32
+ metadata = reformat_metadata(metadata)
33
+ cdes.append(CommonDataElement.from_metadata(metadata))
34
+
35
+ if "groups" in schema_data:
36
+ for group_data in schema_data["groups"]:
37
+ cdes.extend(flatten_cdes(group_data))
38
+
39
+ return cdes
40
+
41
+
42
+ def get_sql_type_per_column(cdes):
43
+ return {code: json.loads(cde.metadata)["sql_type"] for code, cde in cdes.items()}
44
+
45
+
46
+ def get_cdes_with_min_max(cdes, columns):
47
+ cdes_with_min_max = {}
48
+ for code, cde in cdes.items():
49
+ if code not in columns:
50
+ continue
51
+
52
+ metadata = json.loads(cde.metadata)
53
+ min_value = metadata.get("min")
54
+ max_value = metadata.get("max")
55
+
56
+ if min_value is not None or max_value is not None:
57
+ cdes_with_min_max[code] = (min_value, max_value)
58
+
59
+ return cdes_with_min_max
60
+
61
+
62
+ def get_cdes_with_enumerations(cdes, columns):
63
+ cdes_with_enumerations = {}
64
+ for code, cde in cdes.items():
65
+ if code not in columns:
66
+ continue
67
+
68
+ metadata = json.loads(cde.metadata)
69
+ if metadata["is_categorical"]:
70
+ cdes_with_enumerations[code] = list(metadata["enumerations"].keys())
71
+
72
+ return cdes_with_enumerations
73
+
74
+
75
+ def get_dataset_enums(cdes):
76
+ return json.loads(cdes["dataset"].metadata)["enumerations"]
77
+
78
+
79
+ def validate_dataset_present_on_cdes_with_proper_format(cdes):
80
+ dataset_cde = [cde for cde in cdes if cde.code == "dataset"]
81
+ if not dataset_cde:
82
+ raise InvalidDataModelError("There is no 'dataset' CDE in the data model.")
83
+
84
+ dataset_metadata = json.loads(dataset_cde[0].metadata)
85
+ if not dataset_metadata["is_categorical"]:
86
+ raise InvalidDataModelError(
87
+ "CDE 'dataset' must have the 'isCategorical' property equal to 'true'."
88
+ )
89
+
90
+ if dataset_metadata["sql_type"] != "text":
91
+ raise InvalidDataModelError(
92
+ "CDE 'dataset' must have the 'sql_type' property equal to 'text'."
93
+ )
94
+
95
+
96
+ def validate_longitudinal_data_model(cdes):
97
+ subject_id_metadata = None
98
+ visit_id_metadata = None
99
+
100
+ for cde in cdes:
101
+ if cde.code == "subjectid":
102
+ subject_id_metadata = json.loads(cde.metadata)
103
+ elif cde.code == "visitid":
104
+ visit_id_metadata = json.loads(cde.metadata)
105
+
106
+ if not subject_id_metadata:
107
+ raise InvalidDataModelError(
108
+ "There is no 'subjectid' CDE in the longitudinal data model."
109
+ )
110
+
111
+ if not visit_id_metadata:
112
+ raise InvalidDataModelError(
113
+ "There is no 'visitid' CDE in the longitudinal data model."
114
+ )
115
+
116
+ validate_visitid_cde(visit_id_metadata)
117
+
118
+
119
+ def validate_visitid_cde(metadata):
120
+ if not metadata["is_categorical"]:
121
+ raise InvalidDataModelError(
122
+ "CDE 'visitid' must have the 'isCategorical' property equal to 'true'."
123
+ )
124
+
125
+ if metadata["sql_type"] != "text":
126
+ raise InvalidDataModelError(
127
+ "CDE 'visitid' must have the 'sql_type' property equal to 'text'."
128
+ )
129
+
130
+ if "enumerations" not in metadata:
131
+ raise InvalidDataModelError(
132
+ "CDE 'visitid' must contain the 'enumerations' property."
133
+ )
134
+
135
+
136
+ def reformat_metadata(metadata):
137
+ new_key_assign = {
138
+ "isCategorical": "is_categorical",
139
+ "minValue": "min",
140
+ "maxValue": "max",
141
+ }
142
+
143
+ for old_key, new_key in new_key_assign.items():
144
+ if old_key in metadata:
145
+ metadata[new_key] = metadata.pop(old_key)
146
+
147
+ if "enumerations" in metadata:
148
+ metadata["enumerations"] = {
149
+ enumeration["code"]: enumeration["label"]
150
+ for enumeration in metadata["enumerations"]
151
+ }
152
+
153
+ return metadata
154
+
155
+
156
+ def validate_metadata(code, metadata):
157
+ for element in ["is_categorical", "code", "sql_type", "label", "type"]:
158
+ if element not in metadata:
159
+ raise InvalidDataModelError(
160
+ f"Element: {element} is missing from the CDE {code}"
161
+ )
162
+
163
+ if metadata["is_categorical"] and "enumerations" not in metadata:
164
+ raise InvalidDataModelError(
165
+ f"The CDE {code} has 'is_categorical' set to True but there are no enumerations."
166
+ )
167
+
168
+ if {"min", "max"} <= set(metadata) and metadata["min"] >= metadata["max"]:
169
+ raise InvalidDataModelError(f"The CDE {code} has min greater than the max.")
170
+
171
+ valid_metadata_types = ["nominal", "real", "integer", "text"]
172
+ if metadata["type"] not in valid_metadata_types:
173
+ raise InvalidDataModelError(
174
+ f"The CDE {code} has an 'type' the only valid types are:{valid_metadata_types} "
175
+ )
@@ -0,0 +1,400 @@
1
+ from __future__ import annotations
2
+
3
+ import copy
4
+ import difflib
5
+ import os
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+ from typing import Callable
9
+
10
+ import duckdb
11
+
12
+ from data_validator.dataelements import flatten_cdes
13
+ from data_validator.dataelements import get_cdes_with_enumerations
14
+ from data_validator.dataelements import get_cdes_with_min_max
15
+ from data_validator.dataelements import get_dataset_enums
16
+ from data_validator.dataelements import get_sql_type_per_column
17
+ from data_validator.exceptions import InvalidDatasetError
18
+ from data_validator.reporting import ValidationIssue
19
+
20
+
21
+ def _quote_identifier(value: str) -> str:
22
+ return '"' + value.replace('"', '""') + '"'
23
+
24
+
25
+ def _quote_literal(value: str) -> str:
26
+ return "'" + value.replace("'", "''") + "'"
27
+
28
+
29
+ def _format_in_list(values: list[str]) -> str:
30
+ return ", ".join(_quote_literal(value) for value in values)
31
+
32
+
33
+ def _format_allowed_values(values: list[str], max_items: int = 8) -> str:
34
+ shown = values[:max_items]
35
+ formatted = ", ".join(repr(value) for value in shown)
36
+ if len(values) > max_items:
37
+ return f"{formatted}, ... (total={len(values)})"
38
+ return formatted
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class _FusedCheck:
43
+ key: str
44
+ rule: str
45
+ column: str | None
46
+ condition_sql: str
47
+ sample_expression_sql: str
48
+ message_factory: Callable[[int, str | None], str]
49
+
50
+
51
+ class DuckDBDatasetValidator:
52
+ """Validate CSV files against a data model using direct DuckDB CSV queries."""
53
+
54
+ def __init__(self, data_model_metadata: dict, threads: int | None = None) -> None:
55
+ cdes_list = flatten_cdes(copy.deepcopy(data_model_metadata))
56
+ self._cdes = {cde.code: cde for cde in cdes_list}
57
+ self._sql_type_per_column = get_sql_type_per_column(self._cdes)
58
+ self._dataset_enumerations = list(get_dataset_enums(self._cdes).keys())
59
+ self._longitudinal = bool(data_model_metadata.get("longitudinal", False))
60
+ self._threads = threads if threads is not None else (os.cpu_count() or 1)
61
+
62
+ def validate_csv(
63
+ self, csv_path: Path, report_all: bool = False
64
+ ) -> list[ValidationIssue]:
65
+ issues: list[ValidationIssue] = []
66
+ conn = duckdb.connect(database=":memory:")
67
+ try:
68
+ conn.execute(f"PRAGMA threads={self._threads}")
69
+
70
+ columns = self._get_columns(conn, csv_path)
71
+ issues.extend(self._validate_columns(columns, csv_path))
72
+
73
+ if issues and not report_all:
74
+ raise InvalidDatasetError(issues[0].message)
75
+
76
+ fused_checks = self._build_fused_checks(columns)
77
+ issues.extend(self._run_fused_checks(conn, csv_path, fused_checks))
78
+
79
+ if self._longitudinal and {"subjectid", "visitid"} <= set(columns):
80
+ issues.extend(self._validate_longitudinal_pairs(conn, csv_path))
81
+ except duckdb.Error as exc:
82
+ message = f"Unable to validate csv '{csv_path.name}'. {exc}"
83
+ if report_all:
84
+ issues.append(
85
+ ValidationIssue(
86
+ rule="csv.duckdb_error",
87
+ message=message,
88
+ file=str(csv_path),
89
+ )
90
+ )
91
+ return issues
92
+ raise InvalidDatasetError(message) from exc
93
+ finally:
94
+ conn.close()
95
+
96
+ if issues and not report_all:
97
+ raise InvalidDatasetError(issues[0].message)
98
+ return issues
99
+
100
+ def _get_columns(
101
+ self, conn: duckdb.DuckDBPyConnection, csv_path: Path
102
+ ) -> list[str]:
103
+ info = conn.execute(
104
+ (
105
+ "DESCRIBE SELECT * "
106
+ "FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=[''])"
107
+ ),
108
+ [str(csv_path)],
109
+ ).fetchall()
110
+ return [str(row[0]) for row in info]
111
+
112
+ def _validate_columns(
113
+ self, columns: list[str], csv_path: Path
114
+ ) -> list[ValidationIssue]:
115
+ issues: list[ValidationIssue] = []
116
+
117
+ if "dataset" not in columns:
118
+ issues.append(
119
+ ValidationIssue(
120
+ rule="columns.dataset_required",
121
+ message="The 'dataset' column is required to exist in the csv.",
122
+ file=str(csv_path),
123
+ column="dataset",
124
+ )
125
+ )
126
+
127
+ unknown_columns = (
128
+ set(columns) - set(self._sql_type_per_column.keys()) - {"row_id"}
129
+ )
130
+ for column in sorted(unknown_columns):
131
+ suggestion = self._get_column_suggestion(column)
132
+ message = f"Column '{column}' is not present in the CDEs."
133
+ if suggestion:
134
+ message += f" Did you mean '{suggestion}'?"
135
+ issues.append(
136
+ ValidationIssue(
137
+ rule="columns.unknown",
138
+ message=message,
139
+ file=str(csv_path),
140
+ column=column,
141
+ )
142
+ )
143
+
144
+ if self._longitudinal:
145
+ for required in ("subjectid", "visitid"):
146
+ if required not in columns:
147
+ issues.append(
148
+ ValidationIssue(
149
+ rule="longitudinal.required_column",
150
+ message=(
151
+ "The "
152
+ f"'{required}' column is required for longitudinal data models."
153
+ ),
154
+ file=str(csv_path),
155
+ column=required,
156
+ )
157
+ )
158
+
159
+ return issues
160
+
161
+ def _get_column_suggestion(self, unknown_column: str) -> str | None:
162
+ candidates = [
163
+ column for column in self._sql_type_per_column.keys() if column != "row_id"
164
+ ]
165
+ matches = difflib.get_close_matches(unknown_column, candidates, n=1, cutoff=0.7)
166
+ return matches[0] if matches else None
167
+
168
+ def _build_fused_checks(self, columns: list[str]) -> list[_FusedCheck]:
169
+ checks: list[_FusedCheck] = []
170
+
171
+ type_mapping = {"int": "BIGINT", "real": "DOUBLE"}
172
+ for column in columns:
173
+ sql_type = self._sql_type_per_column.get(column)
174
+ cast_type = type_mapping.get(sql_type)
175
+ if not cast_type:
176
+ continue
177
+
178
+ quoted_column = _quote_identifier(column)
179
+ checks.append(
180
+ _FusedCheck(
181
+ key=f"type_{column}",
182
+ rule="types.invalid",
183
+ column=column,
184
+ condition_sql=(
185
+ f"{quoted_column} IS NOT NULL "
186
+ f"AND TRY_CAST({quoted_column} AS {cast_type}) IS NULL"
187
+ ),
188
+ sample_expression_sql=quoted_column,
189
+ message_factory=lambda count, sample, c=column, t=sql_type: (
190
+ f"Column '{c}' has invalid {t} values "
191
+ f"(count={count}, example={sample!r})."
192
+ ),
193
+ )
194
+ )
195
+
196
+ cdes_with_min_max = get_cdes_with_min_max(self._cdes, columns)
197
+ for column, (min_value, max_value) in cdes_with_min_max.items():
198
+ quoted_column = _quote_identifier(column)
199
+ if min_value is not None:
200
+ checks.append(
201
+ _FusedCheck(
202
+ key=f"min_{column}",
203
+ rule="range.min",
204
+ column=column,
205
+ condition_sql=(
206
+ f"{quoted_column} IS NOT NULL "
207
+ f"AND TRY_CAST({quoted_column} AS DOUBLE) < {min_value}"
208
+ ),
209
+ sample_expression_sql=quoted_column,
210
+ message_factory=lambda count, sample, c=column, v=min_value: (
211
+ f"Column '{c}' has values below minimum {v} "
212
+ f"(count={count}, example={sample!r})."
213
+ ),
214
+ )
215
+ )
216
+ if max_value is not None:
217
+ checks.append(
218
+ _FusedCheck(
219
+ key=f"max_{column}",
220
+ rule="range.max",
221
+ column=column,
222
+ condition_sql=(
223
+ f"{quoted_column} IS NOT NULL "
224
+ f"AND TRY_CAST({quoted_column} AS DOUBLE) > {max_value}"
225
+ ),
226
+ sample_expression_sql=quoted_column,
227
+ message_factory=lambda count, sample, c=column, v=max_value: (
228
+ f"Column '{c}' has values above maximum {v} "
229
+ f"(count={count}, example={sample!r})."
230
+ ),
231
+ )
232
+ )
233
+
234
+ cdes_with_enumerations = get_cdes_with_enumerations(self._cdes, columns)
235
+ for column, allowed_values in cdes_with_enumerations.items():
236
+ if not allowed_values:
237
+ continue
238
+ quoted_column = _quote_identifier(column)
239
+ in_list = _format_in_list(allowed_values)
240
+ allowed_display = _format_allowed_values(allowed_values)
241
+ checks.append(
242
+ _FusedCheck(
243
+ key=f"enum_{column}",
244
+ rule="enum.invalid",
245
+ column=column,
246
+ condition_sql=(
247
+ f"{quoted_column} IS NOT NULL AND {quoted_column} NOT IN ({in_list})"
248
+ ),
249
+ sample_expression_sql=quoted_column,
250
+ message_factory=lambda count,
251
+ sample,
252
+ c=column,
253
+ allowed=allowed_display: (
254
+ f"Column '{c}' has invalid categorical value(s). "
255
+ f"Allowed values: [{allowed}]. "
256
+ f"Found {count} invalid row(s), example invalid value: {sample!r}."
257
+ ),
258
+ )
259
+ )
260
+
261
+ if "dataset" in columns:
262
+ in_list = _format_in_list(self._dataset_enumerations)
263
+ dataset_allowed_display = _format_allowed_values(self._dataset_enumerations)
264
+ checks.append(
265
+ _FusedCheck(
266
+ key="dataset_enum",
267
+ rule="dataset.enum",
268
+ column="dataset",
269
+ condition_sql=f"dataset IS NOT NULL AND dataset NOT IN ({in_list})",
270
+ sample_expression_sql="dataset",
271
+ message_factory=lambda count,
272
+ sample,
273
+ allowed=dataset_allowed_display: (
274
+ "Column 'dataset' has value(s) not declared in CDEsMetadata "
275
+ f"enumerations. Allowed values: [{allowed}]. "
276
+ f"Found {count} invalid row(s), example invalid value: {sample!r}."
277
+ ),
278
+ )
279
+ )
280
+
281
+ if self._longitudinal and "subjectid" in columns:
282
+ checks.append(
283
+ _FusedCheck(
284
+ key="longitudinal_subjectid_null",
285
+ rule="longitudinal.subjectid_null",
286
+ column="subjectid",
287
+ condition_sql="subjectid IS NULL",
288
+ sample_expression_sql=_quote_literal("NULL"),
289
+ message_factory=lambda count, _: (
290
+ "Column 'subjectid' should never contain null values "
291
+ f"(count={count})."
292
+ ),
293
+ )
294
+ )
295
+ if self._longitudinal and "visitid" in columns:
296
+ checks.append(
297
+ _FusedCheck(
298
+ key="longitudinal_visitid_null",
299
+ rule="longitudinal.visitid_null",
300
+ column="visitid",
301
+ condition_sql="visitid IS NULL",
302
+ sample_expression_sql=_quote_literal("NULL"),
303
+ message_factory=lambda count, _: (
304
+ "Column 'visitid' should never contain null values "
305
+ f"(count={count})."
306
+ ),
307
+ )
308
+ )
309
+
310
+ return checks
311
+
312
+ def _run_fused_checks(
313
+ self,
314
+ conn: duckdb.DuckDBPyConnection,
315
+ csv_path: Path,
316
+ checks: list[_FusedCheck],
317
+ ) -> list[ValidationIssue]:
318
+ if not checks:
319
+ return []
320
+
321
+ projections = []
322
+ for check in checks:
323
+ count_alias = _quote_identifier(f"{check.key}__count")
324
+ sample_alias = _quote_identifier(f"{check.key}__sample")
325
+ row_alias = _quote_identifier(f"{check.key}__row")
326
+ projections.append(
327
+ f"SUM(CASE WHEN {check.condition_sql} THEN 1 ELSE 0 END) AS {count_alias}"
328
+ )
329
+ projections.append(
330
+ (
331
+ "MIN(CASE WHEN "
332
+ f"{check.condition_sql} THEN {check.sample_expression_sql} "
333
+ f"ELSE NULL END) AS {sample_alias}"
334
+ )
335
+ )
336
+ projections.append(
337
+ f"MIN(CASE WHEN {check.condition_sql} THEN _rownum ELSE NULL END) AS {row_alias}"
338
+ )
339
+
340
+ query = (
341
+ "WITH csv_data AS ("
342
+ " SELECT row_number() OVER () AS _rownum, * "
343
+ " FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=[''])"
344
+ ") "
345
+ "SELECT " + ", ".join(projections) + " FROM csv_data"
346
+ )
347
+ row = conn.execute(query, [str(csv_path)]).fetchone()
348
+ if row is None:
349
+ return []
350
+
351
+ issues: list[ValidationIssue] = []
352
+ for index, check in enumerate(checks):
353
+ count = int(row[index * 3] or 0)
354
+ sample = row[index * 3 + 1]
355
+ first_data_row = row[index * 3 + 2]
356
+ if count <= 0:
357
+ continue
358
+ line_number = (
359
+ int(first_data_row) + 1 if first_data_row is not None else None
360
+ )
361
+ message = check.message_factory(count, sample)
362
+ if line_number is not None:
363
+ message += f" First seen at line {line_number}."
364
+ issues.append(
365
+ ValidationIssue(
366
+ rule=check.rule,
367
+ message=message,
368
+ file=str(csv_path),
369
+ column=check.column,
370
+ row=line_number,
371
+ )
372
+ )
373
+ return issues
374
+
375
+ def _validate_longitudinal_pairs(
376
+ self, conn: duckdb.DuckDBPyConnection, csv_path: Path
377
+ ) -> list[ValidationIssue]:
378
+ duplicates = conn.execute(
379
+ (
380
+ "SELECT subjectid, visitid, COUNT(*) AS duplicate_count "
381
+ "FROM read_csv_auto(?, header=true, all_varchar=true, nullstr=['']) "
382
+ "GROUP BY subjectid, visitid "
383
+ "HAVING COUNT(*) > 1 "
384
+ "LIMIT 5"
385
+ ),
386
+ [str(csv_path)],
387
+ ).fetchall()
388
+ if not duplicates:
389
+ return []
390
+
391
+ return [
392
+ ValidationIssue(
393
+ rule="longitudinal.duplicate_pair",
394
+ message=(
395
+ "Invalid csv: duplicate (visitid, subjectid) pairs detected: "
396
+ f"{[(subject, visit) for subject, visit, _ in duplicates]}"
397
+ ),
398
+ file=str(csv_path),
399
+ )
400
+ ]
@@ -0,0 +1,80 @@
1
+ import sys
2
+ from contextlib import contextmanager
3
+ from enum import IntEnum
4
+ from functools import wraps
5
+
6
+
7
+ class DataBaseError(Exception):
8
+ """Legacy DB error type."""
9
+
10
+ def __init__(self, message) -> None:
11
+ self.message = message
12
+ super().__init__(message)
13
+
14
+
15
+ class UserInputError(Exception):
16
+ def __init__(self, message) -> None:
17
+ self.message = message
18
+ super().__init__(message)
19
+
20
+
21
+ class FileContentError(Exception):
22
+ def __init__(self, message) -> None:
23
+ self.message = message
24
+ super().__init__(message)
25
+
26
+
27
+ class InvalidDatasetError(Exception):
28
+ def __init__(self, message) -> None:
29
+ self.message = message
30
+ super().__init__(message)
31
+
32
+
33
+ class InvalidDataModelError(Exception):
34
+ def __init__(self, message) -> None:
35
+ self.message = message
36
+ super().__init__(message)
37
+
38
+
39
+ class ForeignKeyError(Exception):
40
+ """Legacy FK error type."""
41
+
42
+ def __init__(self, message) -> None:
43
+ self.message = message
44
+ super().__init__(message)
45
+
46
+
47
+ class ExitCode(IntEnum):
48
+ OK = 0
49
+ USER_ERROR = 64
50
+ DB_ERROR = 65
51
+ FILE_ERROR = 66
52
+
53
+
54
+ def handle_errors(func):
55
+ @contextmanager
56
+ def _handle_errors():
57
+ try:
58
+ yield
59
+ except UserInputError as exc:
60
+ print("User input error:\n")
61
+ print(f"\t{exc.message}")
62
+ sys.exit(ExitCode.USER_ERROR)
63
+ except DataBaseError as exc:
64
+ print("Database error:\n")
65
+ print(f"\t{exc.message}")
66
+ sys.exit(ExitCode.DB_ERROR)
67
+ except (FileContentError, InvalidDatasetError, InvalidDataModelError) as exc:
68
+ print(f"\nValidation error: {exc.message}")
69
+ sys.exit(ExitCode.FILE_ERROR)
70
+ except ForeignKeyError as exc:
71
+ print("Foreign key error:\n")
72
+ print(f"\t{exc.message}")
73
+ sys.exit(ExitCode.USER_ERROR)
74
+
75
+ @wraps(func)
76
+ def wrapper(*args, **kwargs):
77
+ with _handle_errors():
78
+ return func(*args, **kwargs)
79
+
80
+ return wrapper
@@ -0,0 +1,342 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from dataclasses import dataclass
5
+ from dataclasses import field
6
+ from datetime import datetime
7
+ from datetime import timezone
8
+ from html import escape
9
+ from pathlib import Path
10
+
11
+
12
+ @dataclass
13
+ class ValidationIssue:
14
+ rule: str
15
+ message: str
16
+ file: str | None = None
17
+ column: str | None = None
18
+ row: int | None = None
19
+ severity: str = "error"
20
+
21
+ def to_dict(self) -> dict:
22
+ return {
23
+ "severity": self.severity,
24
+ "rule": self.rule,
25
+ "message": self.message,
26
+ "file": self.file,
27
+ "column": self.column,
28
+ "row": self.row,
29
+ }
30
+
31
+
32
+ @dataclass
33
+ class ValidationReport:
34
+ folder: str
35
+ files_checked: list[str] = field(default_factory=list)
36
+ issues: list[ValidationIssue] = field(default_factory=list)
37
+ generated_at: str = field(
38
+ default_factory=lambda: datetime.now(timezone.utc).isoformat()
39
+ )
40
+
41
+ @property
42
+ def has_errors(self) -> bool:
43
+ return bool(self.issues)
44
+
45
+ def to_dict(self) -> dict:
46
+ issues_by_file: dict[str, list[dict]] = {}
47
+ for file_key, grouped_issues in _group_issues_by_file(self):
48
+ key = file_key if file_key is not None else "__folder__"
49
+ issues_by_file[key] = [issue.to_dict() for issue in grouped_issues]
50
+
51
+ return {
52
+ "folder": self.folder,
53
+ "generated_at": self.generated_at,
54
+ "files_checked": self.files_checked,
55
+ "error_count": len(self.issues),
56
+ "issues": [issue.to_dict() for issue in self.issues],
57
+ "issues_by_file": issues_by_file,
58
+ }
59
+
60
+
61
+ def render_report(report: ValidationReport, output_format: str) -> str:
62
+ if output_format == "text":
63
+ return _render_text(report)
64
+ if output_format == "json":
65
+ return json.dumps(report.to_dict(), indent=2)
66
+ if output_format == "ndjson":
67
+ return _render_ndjson(report)
68
+ if output_format == "html":
69
+ return _render_html(report)
70
+ raise ValueError(f"Unsupported report format: {output_format}")
71
+
72
+
73
+ def _group_issues_by_file(
74
+ report: ValidationReport,
75
+ ) -> list[tuple[str | None, list[ValidationIssue]]]:
76
+ grouped: dict[str | None, list[ValidationIssue]] = {}
77
+ for issue in report.issues:
78
+ grouped.setdefault(issue.file, []).append(issue)
79
+
80
+ ordered: list[tuple[str | None, list[ValidationIssue]]] = []
81
+ for file_path in report.files_checked:
82
+ issues = grouped.pop(file_path, None)
83
+ if issues:
84
+ ordered.append((file_path, issues))
85
+
86
+ for file_path, issues in grouped.items():
87
+ ordered.append((file_path, issues))
88
+
89
+ return ordered
90
+
91
+
92
+ def _render_text(report: ValidationReport) -> str:
93
+ lines = [
94
+ f"Validation report for: {report.folder}",
95
+ f"Files checked: {len(report.files_checked)}",
96
+ f"Errors: {len(report.issues)}",
97
+ ]
98
+
99
+ grouped_issues = _group_issues_by_file(report)
100
+ if not grouped_issues:
101
+ lines.append("No validation errors were found.")
102
+ return "\n".join(lines)
103
+
104
+ index = 1
105
+ for file_path, issues in grouped_issues:
106
+ if file_path is None:
107
+ lines.append("Folder-level issues:")
108
+ else:
109
+ lines.append(f"CSV: {file_path}")
110
+ for issue in issues:
111
+ location_parts = []
112
+ if issue.column:
113
+ location_parts.append(f"column={issue.column}")
114
+ if issue.row is not None:
115
+ location_parts.append(f"line={issue.row}")
116
+ location = ", ".join(location_parts) if location_parts else "global"
117
+ lines.append(f"{index}. [{issue.rule}] {location}: {issue.message}")
118
+ index += 1
119
+ return "\n".join(lines)
120
+
121
+
122
+ def _render_ndjson(report: ValidationReport) -> str:
123
+ lines = [
124
+ json.dumps(
125
+ {
126
+ "type": "summary",
127
+ "folder": report.folder,
128
+ "generated_at": report.generated_at,
129
+ "files_checked": len(report.files_checked),
130
+ "error_count": len(report.issues),
131
+ }
132
+ )
133
+ ]
134
+ for issue in report.issues:
135
+ lines.append(
136
+ json.dumps(
137
+ {
138
+ "type": "issue",
139
+ "folder": report.folder,
140
+ **issue.to_dict(),
141
+ }
142
+ )
143
+ )
144
+ return "\n".join(lines)
145
+
146
+
147
+ def _render_html(report: ValidationReport) -> str:
148
+ grouped_sections: list[str] = []
149
+ grouped_issues = _group_issues_by_file(report)
150
+ for file_path, issues in grouped_issues:
151
+ section_title = "Folder-level issues"
152
+ title_tooltip = "Folder-level validation issues"
153
+ if file_path is not None:
154
+ file_name = Path(file_path).name
155
+ section_title = file_name
156
+ title_tooltip = file_path
157
+
158
+ rows = []
159
+ for issue in issues:
160
+ severity_class = f"severity-{escape(issue.severity).lower()}"
161
+ rows.append(
162
+ "<tr>"
163
+ f"<td><span class='severity {severity_class}'>{escape(issue.severity)}</span></td>"
164
+ f"<td>{escape(issue.rule)}</td>"
165
+ f"<td>{escape(issue.column or '')}</td>"
166
+ f"<td>{'' if issue.row is None else issue.row}</td>"
167
+ f"<td>{escape(issue.message)}</td>"
168
+ "</tr>"
169
+ )
170
+
171
+ table_rows = "\n".join(rows)
172
+ grouped_sections.append(
173
+ "<section class='file-section'>"
174
+ "<div class='file-header'>"
175
+ f"<h2 class='file-title' title='{escape(title_tooltip)}'>{escape(section_title)}</h2>"
176
+ f"<p class='file-count'>{len(issues)} error(s)</p>"
177
+ "</div>"
178
+ "<div class='table-wrap'>"
179
+ "<table>"
180
+ "<colgroup>"
181
+ "<col style='width:90px'>"
182
+ "<col style='width:150px'>"
183
+ "<col style='width:22%'>"
184
+ "<col style='width:70px'>"
185
+ "<col>"
186
+ "</colgroup>"
187
+ "<thead><tr>"
188
+ "<th>Severity</th><th>Rule</th><th>Column</th><th>Line</th><th>Message</th>"
189
+ "</tr></thead><tbody>"
190
+ f"{table_rows}"
191
+ "</tbody></table></div></section>"
192
+ )
193
+
194
+ sections_html = (
195
+ "".join(grouped_sections)
196
+ if grouped_sections
197
+ else (
198
+ "<div class='table-wrap'><table><thead><tr>"
199
+ "<th>Severity</th><th>Rule</th><th>Column</th><th>Line</th><th>Message</th>"
200
+ "</tr></thead><tbody><tr><td colspan='5'><div class='empty-state'>"
201
+ "No validation errors were found."
202
+ "</div></td></tr></tbody></table></div>"
203
+ )
204
+ )
205
+ return (
206
+ "<!doctype html>"
207
+ "<html><head><meta charset='utf-8'>"
208
+ "<meta name='viewport' content='width=device-width, initial-scale=1'>"
209
+ "<title>Data Validation Report</title>"
210
+ "<style>"
211
+ ":root{"
212
+ "--bg:#f4f7fb;"
213
+ "--card:#ffffff;"
214
+ "--text:#12263a;"
215
+ "--muted:#5a6b7c;"
216
+ "--border:#d7e0ea;"
217
+ "--header:#e8eef6;"
218
+ "--accent:#0c6d9a;"
219
+ "--error-bg:#ffe9e8;"
220
+ "--error-text:#8f1e18;"
221
+ "}"
222
+ "body{"
223
+ "margin:0;"
224
+ "font-family:'Avenir Next','Trebuchet MS','Gill Sans',sans-serif;"
225
+ "background:linear-gradient(160deg,#f7fbff 0%,#eef4fa 100%);"
226
+ "color:var(--text);"
227
+ "}"
228
+ ".container{"
229
+ "max-width:1120px;"
230
+ "margin:32px auto;"
231
+ "padding:0 20px;"
232
+ "}"
233
+ ".card{"
234
+ "background:var(--card);"
235
+ "border:1px solid var(--border);"
236
+ "border-radius:16px;"
237
+ "box-shadow:0 12px 30px rgba(8,39,64,.08);"
238
+ "overflow:hidden;"
239
+ "}"
240
+ ".hero{"
241
+ "padding:24px 28px 18px 28px;"
242
+ "background:linear-gradient(135deg,#edf4fb 0%,#e4edf7 100%);"
243
+ "border-bottom:1px solid var(--border);"
244
+ "}"
245
+ ".title{"
246
+ "margin:0;"
247
+ "font-size:28px;"
248
+ "letter-spacing:.3px;"
249
+ "}"
250
+ ".subtitle{"
251
+ "margin:6px 0 0 0;"
252
+ "color:var(--muted);"
253
+ "font-size:14px;"
254
+ "}"
255
+ ".summary{"
256
+ "display:grid;"
257
+ "grid-template-columns:repeat(auto-fit,minmax(180px,1fr));"
258
+ "gap:12px;"
259
+ "padding:18px 24px 22px 24px;"
260
+ "}"
261
+ ".stat{"
262
+ "background:#f8fbff;"
263
+ "border:1px solid var(--border);"
264
+ "border-radius:12px;"
265
+ "padding:12px 14px;"
266
+ "}"
267
+ ".stat-label{"
268
+ "margin:0 0 6px 0;"
269
+ "font-size:12px;"
270
+ "text-transform:uppercase;"
271
+ "letter-spacing:.08em;"
272
+ "color:var(--muted);"
273
+ "}"
274
+ ".stat-value{"
275
+ "margin:0;"
276
+ "font-size:16px;"
277
+ "font-weight:600;"
278
+ "line-height:1.3;"
279
+ "word-break:break-word;"
280
+ "}"
281
+ ".file-section{margin:0 16px 18px 16px;border:1px solid var(--border);border-radius:12px;overflow:hidden;background:#fff;}"
282
+ ".file-header{display:flex;justify-content:space-between;gap:12px;align-items:center;padding:12px 14px;background:#f3f8fe;border-bottom:1px solid var(--border);}"
283
+ ".file-title{margin:0;font-size:16px;font-weight:700;cursor:help;word-break:break-word;}"
284
+ ".file-count{margin:0;font-size:12px;color:var(--muted);font-weight:600;text-transform:uppercase;letter-spacing:.06em;}"
285
+ ".table-wrap{padding:0;overflow-x:auto;}"
286
+ "table{border-collapse:collapse;width:100%;min-width:920px;background:#fff;table-layout:fixed;}"
287
+ "th,td{padding:10px 12px;text-align:left;vertical-align:top;border-bottom:1px solid var(--border);}"
288
+ "th{"
289
+ "background:var(--header);"
290
+ "font-size:12px;"
291
+ "text-transform:uppercase;"
292
+ "letter-spacing:.06em;"
293
+ "color:#27445d;"
294
+ "position:sticky;"
295
+ "top:0;"
296
+ "}"
297
+ "tbody tr:nth-child(even){background:#fbfdff;}"
298
+ "tbody tr:hover{background:#f2f8fe;}"
299
+ ".severity{"
300
+ "display:inline-block;"
301
+ "padding:3px 8px;"
302
+ "border-radius:999px;"
303
+ "font-size:11px;"
304
+ "font-weight:700;"
305
+ "text-transform:uppercase;"
306
+ "letter-spacing:.05em;"
307
+ "}"
308
+ ".severity-error{background:var(--error-bg);color:var(--error-text);}"
309
+ ".empty-state{padding:18px 0;color:var(--muted);font-style:italic;}"
310
+ "@media (max-width:700px){"
311
+ ".container{margin:18px auto;padding:0 10px;}"
312
+ ".hero{padding:18px 16px;}"
313
+ ".summary{padding:12px 14px 16px 14px;}"
314
+ ".title{font-size:22px;}"
315
+ "}"
316
+ "</style></head><body>"
317
+ "<div class='container'><section class='card'>"
318
+ "<header class='hero'>"
319
+ "<h1 class='title'>Data Validation Report</h1>"
320
+ "<p class='subtitle'>Generated from exadata-validator</p>"
321
+ "</header>"
322
+ "<section class='summary'>"
323
+ "<article class='stat'>"
324
+ "<p class='stat-label'>Folder</p>"
325
+ f"<p class='stat-value'>{escape(report.folder)}</p>"
326
+ "</article>"
327
+ "<article class='stat'>"
328
+ "<p class='stat-label'>Generated</p>"
329
+ f"<p class='stat-value'>{escape(report.generated_at)}</p>"
330
+ "</article>"
331
+ "<article class='stat'>"
332
+ "<p class='stat-label'>Files Checked</p>"
333
+ f"<p class='stat-value'>{len(report.files_checked)}</p>"
334
+ "</article>"
335
+ "<article class='stat'>"
336
+ "<p class='stat-label'>Errors</p>"
337
+ f"<p class='stat-value'>{len(report.issues)}</p>"
338
+ "</article>"
339
+ "</section>"
340
+ f"{sections_html}"
341
+ "</section></div></body></html>"
342
+ )
@@ -0,0 +1,159 @@
1
+ from __future__ import annotations
2
+
3
+ import copy
4
+ import json
5
+ from pathlib import Path
6
+
7
+ import duckdb
8
+
9
+ from data_validator.dataelements import flatten_cdes
10
+ from data_validator.dataelements import (
11
+ validate_dataset_present_on_cdes_with_proper_format,
12
+ )
13
+ from data_validator.dataelements import validate_longitudinal_data_model
14
+ from data_validator.duckdb_validator import DuckDBDatasetValidator
15
+ from data_validator.exceptions import FileContentError
16
+ from data_validator.exceptions import InvalidDataModelError
17
+ from data_validator.exceptions import InvalidDatasetError
18
+ from data_validator.exceptions import UserInputError
19
+ from data_validator.reporting import ValidationIssue
20
+ from data_validator.reporting import ValidationReport
21
+
22
+ LONGITUDINAL = "longitudinal"
23
+
24
+
25
+ def _quote_literal(value: str) -> str:
26
+ return "'" + value.replace("'", "''") + "'"
27
+
28
+
29
+ def _read_json_file(path: Path) -> dict:
30
+ with path.open("r", encoding="utf-8") as stream:
31
+ try:
32
+ return json.load(stream)
33
+ except json.JSONDecodeError as exc:
34
+ raise FileContentError(
35
+ f"Unable to decode json file. {exc.args[0]}"
36
+ ) from exc
37
+
38
+
39
+ def validate_data_model_metadata(data_model_metadata: dict) -> None:
40
+ if "version" not in data_model_metadata:
41
+ raise UserInputError("You need to include a version on the CDEsMetadata.json")
42
+
43
+ cdes = flatten_cdes(copy.deepcopy(data_model_metadata))
44
+ validate_dataset_present_on_cdes_with_proper_format(cdes)
45
+
46
+ if LONGITUDINAL in data_model_metadata:
47
+ longitudinal = data_model_metadata[LONGITUDINAL]
48
+ if not isinstance(longitudinal, bool):
49
+ raise UserInputError(
50
+ f"Longitudinal flag should be boolean, value given: {longitudinal}"
51
+ )
52
+ if longitudinal:
53
+ validate_longitudinal_data_model(cdes)
54
+
55
+
56
+ def _validate_dataset_uniqueness_with_sql(
57
+ csv_files: list[Path],
58
+ ) -> list[ValidationIssue]:
59
+ if len(csv_files) < 2:
60
+ return []
61
+
62
+ file_list_sql = ", ".join(_quote_literal(str(path)) for path in csv_files)
63
+ query = (
64
+ "WITH csv_rows AS ("
65
+ " SELECT filename, dataset "
66
+ f" FROM read_csv_auto([{file_list_sql}], "
67
+ " header=true, all_varchar=true, nullstr=[''], "
68
+ " filename=true, union_by_name=true)"
69
+ "), normalized AS ("
70
+ " SELECT filename, NULLIF(LOWER(TRIM(dataset)), '') AS dataset_normalized "
71
+ " FROM csv_rows "
72
+ " WHERE dataset IS NOT NULL"
73
+ ") "
74
+ "SELECT dataset_normalized, list(DISTINCT filename) AS files "
75
+ "FROM normalized "
76
+ "WHERE dataset_normalized IS NOT NULL "
77
+ "GROUP BY dataset_normalized "
78
+ "HAVING COUNT(DISTINCT filename) > 1 "
79
+ "ORDER BY dataset_normalized"
80
+ )
81
+
82
+ conn = duckdb.connect(database=":memory:")
83
+ try:
84
+ rows = conn.execute(query).fetchall()
85
+ except duckdb.Error as exc:
86
+ raise InvalidDatasetError(
87
+ f"Unable to validate folder-level dataset uniqueness: {exc}"
88
+ ) from exc
89
+ finally:
90
+ conn.close()
91
+
92
+ issues: list[ValidationIssue] = []
93
+ for dataset_normalized, files in rows:
94
+ file_names = sorted(Path(str(path)).name for path in files)
95
+ issues.append(
96
+ ValidationIssue(
97
+ rule="folder.dataset_uniqueness",
98
+ message=(
99
+ "Dataset code collision after normalization (trim+lower) for "
100
+ f"'{dataset_normalized}' across files: {file_names}"
101
+ ),
102
+ )
103
+ )
104
+ return issues
105
+
106
+
107
+ def validate_data_model_folder(
108
+ folder: Path, *, threads: int | None = None, report_all: bool = False
109
+ ) -> ValidationReport:
110
+ report = ValidationReport(folder=str(folder))
111
+
112
+ metadata_path = folder / "CDEsMetadata.json"
113
+ if not metadata_path.exists():
114
+ raise UserInputError(f"Missing metadata file: {metadata_path}")
115
+
116
+ data_model_metadata = _read_json_file(metadata_path)
117
+ try:
118
+ validate_data_model_metadata(data_model_metadata)
119
+ except (InvalidDataModelError, UserInputError) as exc:
120
+ if not report_all:
121
+ raise
122
+ report.issues.append(
123
+ ValidationIssue(
124
+ rule="metadata.invalid",
125
+ message=str(exc),
126
+ file=str(metadata_path),
127
+ )
128
+ )
129
+ return report
130
+
131
+ csv_files = sorted(folder.glob("*.csv"))
132
+ if not csv_files:
133
+ raise UserInputError(
134
+ f"No CSV files found in {folder}. Expected at least one dataset csv."
135
+ )
136
+ report.files_checked = [str(csv_file) for csv_file in csv_files]
137
+
138
+ validator = DuckDBDatasetValidator(data_model_metadata, threads=threads)
139
+ for csv_file in csv_files:
140
+ csv_issues = validator.validate_csv(csv_file, report_all=report_all)
141
+ report.issues.extend(csv_issues)
142
+
143
+ try:
144
+ duplicate_issues = _validate_dataset_uniqueness_with_sql(csv_files)
145
+ report.issues.extend(duplicate_issues)
146
+ except InvalidDatasetError as exc:
147
+ if not report_all:
148
+ raise
149
+ report.issues.append(
150
+ ValidationIssue(
151
+ rule="folder.dataset_uniqueness",
152
+ message=str(exc),
153
+ )
154
+ )
155
+
156
+ if report.issues and not report_all:
157
+ raise InvalidDatasetError(report.issues[0].message)
158
+
159
+ return report
@@ -0,0 +1,61 @@
1
+ Metadata-Version: 2.4
2
+ Name: exadata-validator
3
+ Version: 0.0.1
4
+ Summary: Validate data-model folders using DuckDB.
5
+ Author: Exaflow Team
6
+ Requires-Python: >=3.10,<3.11
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Programming Language :: Python :: 3.10
9
+ Requires-Dist: click (>=8.1,<8.2)
10
+ Requires-Dist: duckdb (>=1.1,<1.2)
11
+ Description-Content-Type: text/markdown
12
+
13
+ # exadata-validator
14
+
15
+ `exadata-validator` validates data-model folders using DuckDB.
16
+
17
+ ## Install With pip
18
+
19
+ ```bash
20
+ python -m venv .venv
21
+ source .venv/bin/activate
22
+ pip install exadata-validator
23
+ ```
24
+
25
+ Validate a data model folder:
26
+
27
+ ```bash
28
+ exadata-validator validate-data-model /path/to/data_model_folder
29
+ ```
30
+
31
+ ## Develop With Poetry
32
+
33
+ From the repository root:
34
+
35
+ ```bash
36
+ cd data-validator/exaflow-data-validator
37
+ poetry install
38
+ ```
39
+
40
+ Then run the same CLI through Poetry:
41
+
42
+ ```bash
43
+ poetry run exadata-validator validate-data-model /path/to/data_model_folder
44
+ ```
45
+
46
+ Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
47
+
48
+ ## Folder Layout
49
+
50
+ ```text
51
+ /path/to/data_model_folder/
52
+ CDEsMetadata.json
53
+ dataset1.csv
54
+ dataset2.csv
55
+ ```
56
+
57
+ ## Validation Notes
58
+
59
+ - CSV validation queries files directly with DuckDB and uses fused aggregate checks to reduce scan overhead.
60
+ - Folder-level dataset uniqueness is enforced across all CSV files via SQL using normalized codes (`trim + lower`).
61
+
@@ -0,0 +1,12 @@
1
+ data_validator/__init__.py,sha256=Tu8Zffh4uJ1ryN2xbmt6NH3Rmtq7iCHkvurX-bWo8uY,36
2
+ data_validator/__main__.py,sha256=roEHUIHJaAQBK-dtqzNdaE9kUG6qaC-ZOoI4M-vuA8A,82
3
+ data_validator/commands.py,sha256=PHWSAfGllMCmK1xsT7oUy8v8jTcVGke6gp79kcWIt-c,2971
4
+ data_validator/dataelements.py,sha256=sElw3t-MvgeTwqC5G0RGFuFioTYIiiu6eKAZkCqEA-0,5392
5
+ data_validator/duckdb_validator.py,sha256=OBqK8ATi4zsAl7g1idv2CBlOLdDvRCtZciq9isdF8mc,15446
6
+ data_validator/exceptions.py,sha256=hEWv45Y_eIPJKwIuK3HJ9Cefay4PINfVcUfjgMFiODU,2030
7
+ data_validator/reporting.py,sha256=pdeJ-ad_fOuYKt1jZqY0iFD433UWh8I5w5m011n6Moo,11622
8
+ data_validator/usecases.py,sha256=NGvAmnxngFiMiiCY_3bz2zOQWAT1Oa3HO-F5ALm4H9o,5353
9
+ exadata_validator-0.0.1.dist-info/METADATA,sha256=xTI6S5nld1kObK9jyCjFSIVLcQ3FI8aPPVzED-UjuP8,1379
10
+ exadata_validator-0.0.1.dist-info/WHEEL,sha256=kJCRJT_g0adfAJzTx2GUMmS80rTJIVHRCfG0DQgLq3o,88
11
+ exadata_validator-0.0.1.dist-info/entry_points.txt,sha256=50_GAf8IaFjGTrft9Iy0tso4G3v-R2ITd4C8Sf7gmMI,67
12
+ exadata_validator-0.0.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.3.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ exadata-validator=data_validator.commands:entry
3
+