exadata-validator 0.0.1__tar.gz → 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {exadata_validator-0.0.1 → exadata_validator-1.1.0}/PKG-INFO +9 -9
- {exadata_validator-0.0.1 → exadata_validator-1.1.0}/README.md +8 -8
- {exadata_validator-0.0.1 → exadata_validator-1.1.0}/pyproject.toml +3 -3
- exadata_validator-1.1.0/validator/__main__.py +4 -0
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/commands.py +6 -6
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/dataelements.py +2 -2
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/duckdb_validator.py +9 -14
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/reporting.py +0 -30
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/usecases.py +10 -12
- exadata_validator-0.0.1/data_validator/__main__.py +0 -4
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/__init__.py +0 -0
- {exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/exceptions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: exadata-validator
|
|
3
|
-
Version:
|
|
3
|
+
Version: 1.1.0
|
|
4
4
|
Summary: Validate data-model folders using DuckDB.
|
|
5
5
|
Author: Exaflow Team
|
|
6
6
|
Requires-Python: >=3.10,<3.11
|
|
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
|
|
|
14
14
|
|
|
15
15
|
`exadata-validator` validates data-model folders using DuckDB.
|
|
16
16
|
|
|
17
|
-
## Install
|
|
17
|
+
## Install with pip
|
|
18
18
|
|
|
19
19
|
```bash
|
|
20
20
|
python -m venv .venv
|
|
@@ -28,21 +28,21 @@ Validate a data model folder:
|
|
|
28
28
|
exadata-validator validate-data-model /path/to/data_model_folder
|
|
29
29
|
```
|
|
30
30
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
From the repository root:
|
|
31
|
+
Run as a Python module:
|
|
34
32
|
|
|
35
33
|
```bash
|
|
36
|
-
|
|
37
|
-
poetry install
|
|
34
|
+
python -m validator validate-data-model /path/to/data_model_folder
|
|
38
35
|
```
|
|
39
36
|
|
|
40
|
-
|
|
37
|
+
Useful options:
|
|
41
38
|
|
|
42
39
|
```bash
|
|
43
|
-
|
|
40
|
+
exadata-validator validate-data-model /path/to/data_model_folder --fail-fast
|
|
41
|
+
exadata-validator validate-data-model /path/to/data_model_folder --format html --output report.html
|
|
44
42
|
```
|
|
45
43
|
|
|
44
|
+
By default the command collects and reports all validation errors in text format. If `--format html` is used without `--output`, the report is written under `/tmp`.
|
|
45
|
+
|
|
46
46
|
Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
|
|
47
47
|
|
|
48
48
|
## Folder Layout
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
`exadata-validator` validates data-model folders using DuckDB.
|
|
4
4
|
|
|
5
|
-
## Install
|
|
5
|
+
## Install with pip
|
|
6
6
|
|
|
7
7
|
```bash
|
|
8
8
|
python -m venv .venv
|
|
@@ -16,21 +16,21 @@ Validate a data model folder:
|
|
|
16
16
|
exadata-validator validate-data-model /path/to/data_model_folder
|
|
17
17
|
```
|
|
18
18
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
From the repository root:
|
|
19
|
+
Run as a Python module:
|
|
22
20
|
|
|
23
21
|
```bash
|
|
24
|
-
|
|
25
|
-
poetry install
|
|
22
|
+
python -m validator validate-data-model /path/to/data_model_folder
|
|
26
23
|
```
|
|
27
24
|
|
|
28
|
-
|
|
25
|
+
Useful options:
|
|
29
26
|
|
|
30
27
|
```bash
|
|
31
|
-
|
|
28
|
+
exadata-validator validate-data-model /path/to/data_model_folder --fail-fast
|
|
29
|
+
exadata-validator validate-data-model /path/to/data_model_folder --format html --output report.html
|
|
32
30
|
```
|
|
33
31
|
|
|
32
|
+
By default the command collects and reports all validation errors in text format. If `--format html` is used without `--output`, the report is written under `/tmp`.
|
|
33
|
+
|
|
34
34
|
Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
|
|
35
35
|
|
|
36
36
|
## Folder Layout
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "exadata-validator"
|
|
3
|
-
version = "
|
|
3
|
+
version = "1.1.0"
|
|
4
4
|
description = "Validate data-model folders using DuckDB."
|
|
5
5
|
authors = ["Exaflow Team"]
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
packages = [
|
|
8
|
-
{ include = "
|
|
8
|
+
{ include = "validator" },
|
|
9
9
|
]
|
|
10
10
|
|
|
11
11
|
[tool.poetry.dependencies]
|
|
@@ -17,7 +17,7 @@ duckdb = "~1.1"
|
|
|
17
17
|
pytest = "~8.4"
|
|
18
18
|
|
|
19
19
|
[tool.poetry.scripts]
|
|
20
|
-
exadata-validator = "
|
|
20
|
+
exadata-validator = "validator.commands:entry"
|
|
21
21
|
|
|
22
22
|
[tool.pytest.ini_options]
|
|
23
23
|
testpaths = ["tests"]
|
|
@@ -7,11 +7,11 @@ from uuid import uuid4
|
|
|
7
7
|
|
|
8
8
|
import click as cl
|
|
9
9
|
|
|
10
|
-
from
|
|
11
|
-
from
|
|
12
|
-
from
|
|
13
|
-
from
|
|
14
|
-
from
|
|
10
|
+
from validator.exceptions import ExitCode
|
|
11
|
+
from validator.exceptions import InvalidDatasetError
|
|
12
|
+
from validator.exceptions import handle_errors
|
|
13
|
+
from validator.reporting import render_report
|
|
14
|
+
from validator.usecases import validate_data_model_folder
|
|
15
15
|
|
|
16
16
|
logging.basicConfig(
|
|
17
17
|
stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
|
|
@@ -54,7 +54,7 @@ def _default_html_report_path() -> Path:
|
|
|
54
54
|
@cl.option(
|
|
55
55
|
"--format",
|
|
56
56
|
"output_format",
|
|
57
|
-
type=cl.Choice(["text", "
|
|
57
|
+
type=cl.Choice(["text", "html"], case_sensitive=False),
|
|
58
58
|
default="text",
|
|
59
59
|
show_default=True,
|
|
60
60
|
help="Output format when collecting all validation errors.",
|
{exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/dataelements.py
RENAMED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import json
|
|
2
2
|
from dataclasses import dataclass
|
|
3
3
|
|
|
4
|
-
from
|
|
5
|
-
from
|
|
4
|
+
from validator.exceptions import InvalidDataModelError
|
|
5
|
+
from validator.exceptions import UserInputError
|
|
6
6
|
|
|
7
7
|
|
|
8
8
|
@dataclass
|
{exadata_validator-0.0.1/data_validator → exadata_validator-1.1.0/validator}/duckdb_validator.py
RENAMED
|
@@ -9,13 +9,13 @@ from typing import Callable
|
|
|
9
9
|
|
|
10
10
|
import duckdb
|
|
11
11
|
|
|
12
|
-
from
|
|
13
|
-
from
|
|
14
|
-
from
|
|
15
|
-
from
|
|
16
|
-
from
|
|
17
|
-
from
|
|
18
|
-
from
|
|
12
|
+
from validator.dataelements import flatten_cdes
|
|
13
|
+
from validator.dataelements import get_cdes_with_enumerations
|
|
14
|
+
from validator.dataelements import get_cdes_with_min_max
|
|
15
|
+
from validator.dataelements import get_dataset_enums
|
|
16
|
+
from validator.dataelements import get_sql_type_per_column
|
|
17
|
+
from validator.exceptions import InvalidDatasetError
|
|
18
|
+
from validator.reporting import ValidationIssue
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
def _quote_identifier(value: str) -> str:
|
|
@@ -247,10 +247,7 @@ class DuckDBDatasetValidator:
|
|
|
247
247
|
f"{quoted_column} IS NOT NULL AND {quoted_column} NOT IN ({in_list})"
|
|
248
248
|
),
|
|
249
249
|
sample_expression_sql=quoted_column,
|
|
250
|
-
message_factory=lambda count,
|
|
251
|
-
sample,
|
|
252
|
-
c=column,
|
|
253
|
-
allowed=allowed_display: (
|
|
250
|
+
message_factory=lambda count, sample, c=column, allowed=allowed_display: (
|
|
254
251
|
f"Column '{c}' has invalid categorical value(s). "
|
|
255
252
|
f"Allowed values: [{allowed}]. "
|
|
256
253
|
f"Found {count} invalid row(s), example invalid value: {sample!r}."
|
|
@@ -268,9 +265,7 @@ class DuckDBDatasetValidator:
|
|
|
268
265
|
column="dataset",
|
|
269
266
|
condition_sql=f"dataset IS NOT NULL AND dataset NOT IN ({in_list})",
|
|
270
267
|
sample_expression_sql="dataset",
|
|
271
|
-
message_factory=lambda count,
|
|
272
|
-
sample,
|
|
273
|
-
allowed=dataset_allowed_display: (
|
|
268
|
+
message_factory=lambda count, sample, allowed=dataset_allowed_display: (
|
|
274
269
|
"Column 'dataset' has value(s) not declared in CDEsMetadata "
|
|
275
270
|
f"enumerations. Allowed values: [{allowed}]. "
|
|
276
271
|
f"Found {count} invalid row(s), example invalid value: {sample!r}."
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
import json
|
|
4
3
|
from dataclasses import dataclass
|
|
5
4
|
from dataclasses import field
|
|
6
5
|
from datetime import datetime
|
|
@@ -61,10 +60,6 @@ class ValidationReport:
|
|
|
61
60
|
def render_report(report: ValidationReport, output_format: str) -> str:
|
|
62
61
|
if output_format == "text":
|
|
63
62
|
return _render_text(report)
|
|
64
|
-
if output_format == "json":
|
|
65
|
-
return json.dumps(report.to_dict(), indent=2)
|
|
66
|
-
if output_format == "ndjson":
|
|
67
|
-
return _render_ndjson(report)
|
|
68
63
|
if output_format == "html":
|
|
69
64
|
return _render_html(report)
|
|
70
65
|
raise ValueError(f"Unsupported report format: {output_format}")
|
|
@@ -119,31 +114,6 @@ def _render_text(report: ValidationReport) -> str:
|
|
|
119
114
|
return "\n".join(lines)
|
|
120
115
|
|
|
121
116
|
|
|
122
|
-
def _render_ndjson(report: ValidationReport) -> str:
|
|
123
|
-
lines = [
|
|
124
|
-
json.dumps(
|
|
125
|
-
{
|
|
126
|
-
"type": "summary",
|
|
127
|
-
"folder": report.folder,
|
|
128
|
-
"generated_at": report.generated_at,
|
|
129
|
-
"files_checked": len(report.files_checked),
|
|
130
|
-
"error_count": len(report.issues),
|
|
131
|
-
}
|
|
132
|
-
)
|
|
133
|
-
]
|
|
134
|
-
for issue in report.issues:
|
|
135
|
-
lines.append(
|
|
136
|
-
json.dumps(
|
|
137
|
-
{
|
|
138
|
-
"type": "issue",
|
|
139
|
-
"folder": report.folder,
|
|
140
|
-
**issue.to_dict(),
|
|
141
|
-
}
|
|
142
|
-
)
|
|
143
|
-
)
|
|
144
|
-
return "\n".join(lines)
|
|
145
|
-
|
|
146
|
-
|
|
147
117
|
def _render_html(report: ValidationReport) -> str:
|
|
148
118
|
grouped_sections: list[str] = []
|
|
149
119
|
grouped_issues = _group_issues_by_file(report)
|
|
@@ -6,18 +6,16 @@ from pathlib import Path
|
|
|
6
6
|
|
|
7
7
|
import duckdb
|
|
8
8
|
|
|
9
|
-
from
|
|
10
|
-
from
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
from
|
|
14
|
-
from
|
|
15
|
-
from
|
|
16
|
-
from
|
|
17
|
-
from
|
|
18
|
-
from
|
|
19
|
-
from data_validator.reporting import ValidationIssue
|
|
20
|
-
from data_validator.reporting import ValidationReport
|
|
9
|
+
from validator.dataelements import flatten_cdes
|
|
10
|
+
from validator.dataelements import validate_dataset_present_on_cdes_with_proper_format
|
|
11
|
+
from validator.dataelements import validate_longitudinal_data_model
|
|
12
|
+
from validator.duckdb_validator import DuckDBDatasetValidator
|
|
13
|
+
from validator.exceptions import FileContentError
|
|
14
|
+
from validator.exceptions import InvalidDataModelError
|
|
15
|
+
from validator.exceptions import InvalidDatasetError
|
|
16
|
+
from validator.exceptions import UserInputError
|
|
17
|
+
from validator.reporting import ValidationIssue
|
|
18
|
+
from validator.reporting import ValidationReport
|
|
21
19
|
|
|
22
20
|
LONGITUDINAL = "longitudinal"
|
|
23
21
|
|
|
File without changes
|
|
File without changes
|