exadata-validator 0.0.1__tar.gz → 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: exadata-validator
3
- Version: 0.0.1
3
+ Version: 1.1.0
4
4
  Summary: Validate data-model folders using DuckDB.
5
5
  Author: Exaflow Team
6
6
  Requires-Python: >=3.10,<3.11
@@ -14,7 +14,7 @@ Description-Content-Type: text/markdown
14
14
 
15
15
  `exadata-validator` validates data-model folders using DuckDB.
16
16
 
17
- ## Install With pip
17
+ ## Install with pip
18
18
 
19
19
  ```bash
20
20
  python -m venv .venv
@@ -28,21 +28,21 @@ Validate a data model folder:
28
28
  exadata-validator validate-data-model /path/to/data_model_folder
29
29
  ```
30
30
 
31
- ## Develop With Poetry
32
-
33
- From the repository root:
31
+ Run as a Python module:
34
32
 
35
33
  ```bash
36
- cd data-validator/exaflow-data-validator
37
- poetry install
34
+ python -m validator validate-data-model /path/to/data_model_folder
38
35
  ```
39
36
 
40
- Then run the same CLI through Poetry:
37
+ Useful options:
41
38
 
42
39
  ```bash
43
- poetry run exadata-validator validate-data-model /path/to/data_model_folder
40
+ exadata-validator validate-data-model /path/to/data_model_folder --fail-fast
41
+ exadata-validator validate-data-model /path/to/data_model_folder --format html --output report.html
44
42
  ```
45
43
 
44
+ By default the command collects and reports all validation errors in text format. If `--format html` is used without `--output`, the report is written under `/tmp`.
45
+
46
46
  Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
47
47
 
48
48
  ## Folder Layout
@@ -2,7 +2,7 @@
2
2
 
3
3
  `exadata-validator` validates data-model folders using DuckDB.
4
4
 
5
- ## Install With pip
5
+ ## Install with pip
6
6
 
7
7
  ```bash
8
8
  python -m venv .venv
@@ -16,21 +16,21 @@ Validate a data model folder:
16
16
  exadata-validator validate-data-model /path/to/data_model_folder
17
17
  ```
18
18
 
19
- ## Develop With Poetry
20
-
21
- From the repository root:
19
+ Run as a Python module:
22
20
 
23
21
  ```bash
24
- cd data-validator/exaflow-data-validator
25
- poetry install
22
+ python -m validator validate-data-model /path/to/data_model_folder
26
23
  ```
27
24
 
28
- Then run the same CLI through Poetry:
25
+ Useful options:
29
26
 
30
27
  ```bash
31
- poetry run exadata-validator validate-data-model /path/to/data_model_folder
28
+ exadata-validator validate-data-model /path/to/data_model_folder --fail-fast
29
+ exadata-validator validate-data-model /path/to/data_model_folder --format html --output report.html
32
30
  ```
33
31
 
32
+ By default the command collects and reports all validation errors in text format. If `--format html` is used without `--output`, the report is written under `/tmp`.
33
+
34
34
  Use `exadata-validator validate-data-model --help` for reporting, output, and threading options.
35
35
 
36
36
  ## Folder Layout
@@ -1,11 +1,11 @@
1
1
  [tool.poetry]
2
2
  name = "exadata-validator"
3
- version = "0.0.1"
3
+ version = "1.1.0"
4
4
  description = "Validate data-model folders using DuckDB."
5
5
  authors = ["Exaflow Team"]
6
6
  readme = "README.md"
7
7
  packages = [
8
- { include = "data_validator" },
8
+ { include = "validator" },
9
9
  ]
10
10
 
11
11
  [tool.poetry.dependencies]
@@ -17,7 +17,7 @@ duckdb = "~1.1"
17
17
  pytest = "~8.4"
18
18
 
19
19
  [tool.poetry.scripts]
20
- exadata-validator = "data_validator.commands:entry"
20
+ exadata-validator = "validator.commands:entry"
21
21
 
22
22
  [tool.pytest.ini_options]
23
23
  testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ from validator.commands import entry
2
+
3
+ if __name__ == "__main__":
4
+ entry()
@@ -7,11 +7,11 @@ from uuid import uuid4
7
7
 
8
8
  import click as cl
9
9
 
10
- from data_validator.exceptions import ExitCode
11
- from data_validator.exceptions import InvalidDatasetError
12
- from data_validator.exceptions import handle_errors
13
- from data_validator.reporting import render_report
14
- from data_validator.usecases import validate_data_model_folder
10
+ from validator.exceptions import ExitCode
11
+ from validator.exceptions import InvalidDatasetError
12
+ from validator.exceptions import handle_errors
13
+ from validator.reporting import render_report
14
+ from validator.usecases import validate_data_model_folder
15
15
 
16
16
  logging.basicConfig(
17
17
  stream=sys.stderr, level=logging.INFO, format="%(levelname)s: %(message)s"
@@ -54,7 +54,7 @@ def _default_html_report_path() -> Path:
54
54
  @cl.option(
55
55
  "--format",
56
56
  "output_format",
57
- type=cl.Choice(["text", "json", "ndjson", "html"], case_sensitive=False),
57
+ type=cl.Choice(["text", "html"], case_sensitive=False),
58
58
  default="text",
59
59
  show_default=True,
60
60
  help="Output format when collecting all validation errors.",
@@ -1,8 +1,8 @@
1
1
  import json
2
2
  from dataclasses import dataclass
3
3
 
4
- from data_validator.exceptions import InvalidDataModelError
5
- from data_validator.exceptions import UserInputError
4
+ from validator.exceptions import InvalidDataModelError
5
+ from validator.exceptions import UserInputError
6
6
 
7
7
 
8
8
  @dataclass
@@ -9,13 +9,13 @@ from typing import Callable
9
9
 
10
10
  import duckdb
11
11
 
12
- from data_validator.dataelements import flatten_cdes
13
- from data_validator.dataelements import get_cdes_with_enumerations
14
- from data_validator.dataelements import get_cdes_with_min_max
15
- from data_validator.dataelements import get_dataset_enums
16
- from data_validator.dataelements import get_sql_type_per_column
17
- from data_validator.exceptions import InvalidDatasetError
18
- from data_validator.reporting import ValidationIssue
12
+ from validator.dataelements import flatten_cdes
13
+ from validator.dataelements import get_cdes_with_enumerations
14
+ from validator.dataelements import get_cdes_with_min_max
15
+ from validator.dataelements import get_dataset_enums
16
+ from validator.dataelements import get_sql_type_per_column
17
+ from validator.exceptions import InvalidDatasetError
18
+ from validator.reporting import ValidationIssue
19
19
 
20
20
 
21
21
  def _quote_identifier(value: str) -> str:
@@ -247,10 +247,7 @@ class DuckDBDatasetValidator:
247
247
  f"{quoted_column} IS NOT NULL AND {quoted_column} NOT IN ({in_list})"
248
248
  ),
249
249
  sample_expression_sql=quoted_column,
250
- message_factory=lambda count,
251
- sample,
252
- c=column,
253
- allowed=allowed_display: (
250
+ message_factory=lambda count, sample, c=column, allowed=allowed_display: (
254
251
  f"Column '{c}' has invalid categorical value(s). "
255
252
  f"Allowed values: [{allowed}]. "
256
253
  f"Found {count} invalid row(s), example invalid value: {sample!r}."
@@ -268,9 +265,7 @@ class DuckDBDatasetValidator:
268
265
  column="dataset",
269
266
  condition_sql=f"dataset IS NOT NULL AND dataset NOT IN ({in_list})",
270
267
  sample_expression_sql="dataset",
271
- message_factory=lambda count,
272
- sample,
273
- allowed=dataset_allowed_display: (
268
+ message_factory=lambda count, sample, allowed=dataset_allowed_display: (
274
269
  "Column 'dataset' has value(s) not declared in CDEsMetadata "
275
270
  f"enumerations. Allowed values: [{allowed}]. "
276
271
  f"Found {count} invalid row(s), example invalid value: {sample!r}."
@@ -1,6 +1,5 @@
1
1
  from __future__ import annotations
2
2
 
3
- import json
4
3
  from dataclasses import dataclass
5
4
  from dataclasses import field
6
5
  from datetime import datetime
@@ -61,10 +60,6 @@ class ValidationReport:
61
60
  def render_report(report: ValidationReport, output_format: str) -> str:
62
61
  if output_format == "text":
63
62
  return _render_text(report)
64
- if output_format == "json":
65
- return json.dumps(report.to_dict(), indent=2)
66
- if output_format == "ndjson":
67
- return _render_ndjson(report)
68
63
  if output_format == "html":
69
64
  return _render_html(report)
70
65
  raise ValueError(f"Unsupported report format: {output_format}")
@@ -119,31 +114,6 @@ def _render_text(report: ValidationReport) -> str:
119
114
  return "\n".join(lines)
120
115
 
121
116
 
122
- def _render_ndjson(report: ValidationReport) -> str:
123
- lines = [
124
- json.dumps(
125
- {
126
- "type": "summary",
127
- "folder": report.folder,
128
- "generated_at": report.generated_at,
129
- "files_checked": len(report.files_checked),
130
- "error_count": len(report.issues),
131
- }
132
- )
133
- ]
134
- for issue in report.issues:
135
- lines.append(
136
- json.dumps(
137
- {
138
- "type": "issue",
139
- "folder": report.folder,
140
- **issue.to_dict(),
141
- }
142
- )
143
- )
144
- return "\n".join(lines)
145
-
146
-
147
117
  def _render_html(report: ValidationReport) -> str:
148
118
  grouped_sections: list[str] = []
149
119
  grouped_issues = _group_issues_by_file(report)
@@ -6,18 +6,16 @@ from pathlib import Path
6
6
 
7
7
  import duckdb
8
8
 
9
- from data_validator.dataelements import flatten_cdes
10
- from data_validator.dataelements import (
11
- validate_dataset_present_on_cdes_with_proper_format,
12
- )
13
- from data_validator.dataelements import validate_longitudinal_data_model
14
- from data_validator.duckdb_validator import DuckDBDatasetValidator
15
- from data_validator.exceptions import FileContentError
16
- from data_validator.exceptions import InvalidDataModelError
17
- from data_validator.exceptions import InvalidDatasetError
18
- from data_validator.exceptions import UserInputError
19
- from data_validator.reporting import ValidationIssue
20
- from data_validator.reporting import ValidationReport
9
+ from validator.dataelements import flatten_cdes
10
+ from validator.dataelements import validate_dataset_present_on_cdes_with_proper_format
11
+ from validator.dataelements import validate_longitudinal_data_model
12
+ from validator.duckdb_validator import DuckDBDatasetValidator
13
+ from validator.exceptions import FileContentError
14
+ from validator.exceptions import InvalidDataModelError
15
+ from validator.exceptions import InvalidDatasetError
16
+ from validator.exceptions import UserInputError
17
+ from validator.reporting import ValidationIssue
18
+ from validator.reporting import ValidationReport
21
19
 
22
20
  LONGITUDINAL = "longitudinal"
23
21
 
@@ -1,4 +0,0 @@
1
- from data_validator.commands import entry
2
-
3
- if __name__ == "__main__":
4
- entry()