table-validator 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- table_validator/__init__.py +46 -0
- table_validator/auth/__init__.py +1 -0
- table_validator/auth/azure_auth.py +52 -0
- table_validator/auth/databricks_auth.py +31 -0
- table_validator/cli/__init__.py +1 -0
- table_validator/cli/main.py +722 -0
- table_validator/cli/partition_prompt.py +78 -0
- table_validator/cli/summary_table.py +146 -0
- table_validator/cli/wizard.py +429 -0
- table_validator/config/__init__.py +1 -0
- table_validator/config/manager.py +84 -0
- table_validator/config/schema.py +179 -0
- table_validator/connectors/__init__.py +1 -0
- table_validator/connectors/azure_connector.py +809 -0
- table_validator/connectors/databricks_connector.py +1230 -0
- table_validator/engine/__init__.py +1 -0
- table_validator/engine/comparison_engine.py +645 -0
- table_validator/models.py +952 -0
- table_validator/reports/__init__.py +1 -0
- table_validator/reports/excel_report.py +953 -0
- table_validator/validators/__init__.py +1 -0
- table_validator/validators/blob_discovery.py +467 -0
- table_validator/validators/catalog_validator.py +1863 -0
- table_validator/validators/row_validator.py +1727 -0
- table_validator-0.1.0.dist-info/METADATA +190 -0
- table_validator-0.1.0.dist-info/RECORD +30 -0
- table_validator-0.1.0.dist-info/WHEEL +5 -0
- table_validator-0.1.0.dist-info/entry_points.txt +2 -0
- table_validator-0.1.0.dist-info/licenses/LICENSE +21 -0
- table_validator-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Config manager: loads/saves non-secret config under ~/.table_validator/config.yaml."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
import yaml
|
|
9
|
+
|
|
10
|
+
from table_validator.config.schema import ValidatorConfig
|
|
11
|
+
|
|
12
|
+
CONFIG_DIR = Path.home() / ".table_validator"
|
|
13
|
+
CONFIG_PATH = CONFIG_DIR / "config.yaml"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ConfigNotFoundError(Exception):
|
|
17
|
+
"""Raised when a config file is required but doesn't exist yet.
|
|
18
|
+
|
|
19
|
+
Distinct from a generic FileNotFoundError so callers (e.g. the
|
|
20
|
+
`validate` command) can catch it specifically and print a targeted
|
|
21
|
+
"run configure first" message instead of a raw traceback.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def default_config() -> ValidatorConfig:
|
|
26
|
+
"""Return an empty/default config (used when no config file exists yet)."""
|
|
27
|
+
return ValidatorConfig()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def load_config(path: Optional[Path] = None) -> ValidatorConfig:
|
|
31
|
+
"""Load config from `path` (default: CONFIG_PATH), returning a default
|
|
32
|
+
config if it doesn't exist.
|
|
33
|
+
|
|
34
|
+
`path` defaults to None (resolved to CONFIG_PATH at call time, not
|
|
35
|
+
definition time) so this always reflects the current value of
|
|
36
|
+
CONFIG_PATH - including in tests that patch it.
|
|
37
|
+
|
|
38
|
+
Use this when a missing config is a legitimate, silent starting point
|
|
39
|
+
(e.g. the wizard pre-populating its prompts with existing values).
|
|
40
|
+
For a context where a config file is required - like `validate` - use
|
|
41
|
+
require_config() instead, which raises ConfigNotFoundError.
|
|
42
|
+
"""
|
|
43
|
+
path = path or CONFIG_PATH
|
|
44
|
+
|
|
45
|
+
if not path.exists():
|
|
46
|
+
return default_config()
|
|
47
|
+
|
|
48
|
+
raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
|
49
|
+
return ValidatorConfig.model_validate(raw)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def require_config(path: Optional[Path] = None) -> ValidatorConfig:
|
|
53
|
+
"""Load config from `path` (default: CONFIG_PATH), raising
|
|
54
|
+
ConfigNotFoundError if it doesn't exist.
|
|
55
|
+
|
|
56
|
+
Intended for `tablevalidator validate`: running validation before
|
|
57
|
+
`tablevalidator configure` has ever been run is a user error that
|
|
58
|
+
deserves a clear, actionable message rather than silently validating
|
|
59
|
+
against an empty config.
|
|
60
|
+
"""
|
|
61
|
+
path = path or CONFIG_PATH
|
|
62
|
+
|
|
63
|
+
if not path.exists():
|
|
64
|
+
raise ConfigNotFoundError(
|
|
65
|
+
f"No configuration found at {path}. "
|
|
66
|
+
"Run 'tablevalidator configure' first to set up your source/target "
|
|
67
|
+
"tables and Databricks connection."
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
return load_config(path)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def save_config(config: ValidatorConfig, path: Optional[Path] = None) -> None:
|
|
74
|
+
"""Save `config` to `path` (default: CONFIG_PATH), creating the parent
|
|
75
|
+
directory if missing."""
|
|
76
|
+
path = path or CONFIG_PATH
|
|
77
|
+
|
|
78
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
|
|
80
|
+
data = config.model_dump(mode="json", by_alias=True)
|
|
81
|
+
path.write_text(
|
|
82
|
+
yaml.safe_dump(data, sort_keys=False, default_flow_style=False),
|
|
83
|
+
encoding="utf-8",
|
|
84
|
+
)
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""Config schema: pydantic models describing stored (non-secret) configuration.
|
|
2
|
+
|
|
3
|
+
Secrets (Databricks PAT, Azure Storage key, Azure SQL password, etc.) are
|
|
4
|
+
handled separately in Phase 4 and never appear on these models.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from enum import Enum
|
|
10
|
+
from typing import List, Optional
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ValidationType(str, Enum):
|
|
16
|
+
CATALOG = "catalog"
|
|
17
|
+
SCHEMA = "schema"
|
|
18
|
+
COLUMN = "column"
|
|
19
|
+
ROW = "row"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class SourceType(str, Enum):
|
|
23
|
+
"""
|
|
24
|
+
What's being compared against the (always-Databricks) target catalog.
|
|
25
|
+
Defaults to DATABRICKS so existing saved configs without this field
|
|
26
|
+
(from before source_type existed) load unchanged - they were always
|
|
27
|
+
Databricks-to-Databricks comparisons.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
DATABRICKS = "databricks"
|
|
31
|
+
AZURE_BLOB = "azure_blob"
|
|
32
|
+
AZURE_SQL = "azure_sql"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class AzureConfig(BaseModel):
|
|
36
|
+
"""Non-secret Azure connection details.
|
|
37
|
+
|
|
38
|
+
tenant_id / subscription_id are captured now as groundwork for the
|
|
39
|
+
later Azure CLI / Service Principal auth phase; the current manual-auth
|
|
40
|
+
connectors don't need them yet. storage_account/container and
|
|
41
|
+
sql_server/sql_database cover the two connector types this tool
|
|
42
|
+
actually talks to today (Blob Storage and Azure SQL Database).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
tenant_id: Optional[str] = None
|
|
46
|
+
subscription_id: Optional[str] = None
|
|
47
|
+
|
|
48
|
+
storage_account: Optional[str] = None
|
|
49
|
+
container: Optional[str] = None
|
|
50
|
+
|
|
51
|
+
sql_server: Optional[str] = None
|
|
52
|
+
sql_database: Optional[str] = None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class DatabricksConfig(BaseModel):
|
|
56
|
+
"""Non-secret Databricks connection details."""
|
|
57
|
+
|
|
58
|
+
workspace_url: Optional[str] = None
|
|
59
|
+
http_path: Optional[str] = None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class TableRef(BaseModel):
|
|
63
|
+
"""
|
|
64
|
+
Reference to a table, schema, or whole catalog.
|
|
65
|
+
|
|
66
|
+
catalog is required. schema_name/table are optional: leaving
|
|
67
|
+
schema_name unset means "compare every schema common to both
|
|
68
|
+
catalogs"; leaving table unset (with schema_name set) means "compare
|
|
69
|
+
every table common to both sides of that schema". CatalogValidator's
|
|
70
|
+
compare_schemas/compare_tables perform this discovery internally
|
|
71
|
+
whenever the corresponding restriction is left unset.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
catalog: Optional[str] = None
|
|
75
|
+
schema_name: Optional[str] = Field(default=None, alias="schema")
|
|
76
|
+
table: Optional[str] = None
|
|
77
|
+
|
|
78
|
+
model_config = {"populate_by_name": True}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class BlobSourceConfig(BaseModel):
|
|
82
|
+
"""
|
|
83
|
+
Non-secret scoping for an Azure Blob Storage source (source_type =
|
|
84
|
+
azure_blob). The storage account/container credentials themselves
|
|
85
|
+
live on AzureConfig (storage_account/container, shared with any other
|
|
86
|
+
use of the same storage account); this section only scopes WHICH
|
|
87
|
+
blobs within that container are treated as comparison sources.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
container: Optional[str] = None
|
|
91
|
+
folder_prefix: Optional[str] = Field(
|
|
92
|
+
default=None,
|
|
93
|
+
description=(
|
|
94
|
+
"Optional path prefix to restrict blob discovery to (e.g. "
|
|
95
|
+
"'validation/2024/'). If unset, the whole container is scanned."
|
|
96
|
+
),
|
|
97
|
+
)
|
|
98
|
+
file_pattern: Optional[str] = Field(
|
|
99
|
+
default=None,
|
|
100
|
+
description=(
|
|
101
|
+
"Optional glob pattern to restrict blob discovery to (e.g. "
|
|
102
|
+
"'*.csv' or '*.parquet'). If unset, every supported file "
|
|
103
|
+
"extension (.csv/.txt/.xlsx/.xls/.parquet) is considered."
|
|
104
|
+
),
|
|
105
|
+
)
|
|
106
|
+
blob_path: Optional[str] = Field(
|
|
107
|
+
default=None,
|
|
108
|
+
description=(
|
|
109
|
+
"Optional exact path to a single source blob (e.g. "
|
|
110
|
+
"'n8ndirectory/file_example_XLSX_100.csv'). If set together "
|
|
111
|
+
"with target_table.table, that exact blob is compared "
|
|
112
|
+
"directly against that exact table - bypassing filename-to-"
|
|
113
|
+
"table-name discovery entirely, even if the names don't "
|
|
114
|
+
"match. If unset, folder_prefix/file_pattern-based discovery "
|
|
115
|
+
"across multiple blobs applies as usual."
|
|
116
|
+
),
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class SqlSourceConfig(BaseModel):
|
|
121
|
+
"""
|
|
122
|
+
Non-secret scoping for an Azure SQL Database source (source_type =
|
|
123
|
+
azure_sql). Server/database themselves live on AzureConfig
|
|
124
|
+
(sql_server/sql_database); this section scopes which schema/table
|
|
125
|
+
within that database are compared, same optional-means-"compare all"
|
|
126
|
+
convention as TableRef.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
schema_name: Optional[str] = Field(default=None, alias="schema")
|
|
130
|
+
table: Optional[str] = None
|
|
131
|
+
|
|
132
|
+
model_config = {"populate_by_name": True}
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class ValidatorConfig(BaseModel):
|
|
136
|
+
"""
|
|
137
|
+
Top-level, non-secret configuration persisted to config.yaml.
|
|
138
|
+
|
|
139
|
+
source_type selects which of source_table/blob_source/sql_source is
|
|
140
|
+
read by `validate` - only the section matching source_type is
|
|
141
|
+
meaningful; the other two may be present (e.g. left over from
|
|
142
|
+
switching source types in the wizard) but are ignored. target_table
|
|
143
|
+
is always a Databricks catalog/schema/table ref regardless of
|
|
144
|
+
source_type, since every comparison path targets Databricks.
|
|
145
|
+
"""
|
|
146
|
+
|
|
147
|
+
source_type: SourceType = Field(default=SourceType.DATABRICKS)
|
|
148
|
+
|
|
149
|
+
azure: AzureConfig = Field(default_factory=AzureConfig)
|
|
150
|
+
databricks: DatabricksConfig = Field(default_factory=DatabricksConfig)
|
|
151
|
+
|
|
152
|
+
source_table: TableRef = Field(default_factory=TableRef)
|
|
153
|
+
target_table: TableRef = Field(default_factory=TableRef)
|
|
154
|
+
|
|
155
|
+
primary_key: Optional[List[str]] = Field(
|
|
156
|
+
default=None,
|
|
157
|
+
description=(
|
|
158
|
+
"Optional primary/business key column(s) for the single named "
|
|
159
|
+
"table in source_table/target_table (only meaningful when both "
|
|
160
|
+
"are set to a specific table, not left blank for a catalog-"
|
|
161
|
+
"wide sweep). When set, row-level comparison uses this key "
|
|
162
|
+
"instead of falling back to a synthetic ROW_NUMBER() match - "
|
|
163
|
+
"cheaper and more reliable, and avoids the full-table sort "
|
|
164
|
+
"the row-number fallback needs on large tables. If unset, the "
|
|
165
|
+
"row-number fallback is used as before."
|
|
166
|
+
),
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
blob_source: BlobSourceConfig = Field(default_factory=BlobSourceConfig)
|
|
170
|
+
sql_source: SqlSourceConfig = Field(default_factory=SqlSourceConfig)
|
|
171
|
+
|
|
172
|
+
validations: List[ValidationType] = Field(
|
|
173
|
+
default_factory=lambda: [
|
|
174
|
+
ValidationType.CATALOG,
|
|
175
|
+
ValidationType.SCHEMA,
|
|
176
|
+
ValidationType.COLUMN,
|
|
177
|
+
ValidationType.ROW,
|
|
178
|
+
]
|
|
179
|
+
)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Connectors package: I/O-only clients for Azure and Databricks (no comparison logic)."""
|