table-validator 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {table_validator-0.1.0/table_validator.egg-info → table_validator-0.1.2}/PKG-INFO +1 -1
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/main.py +29 -6
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/catalog_validator.py +29 -18
- {table_validator-0.1.0 → table_validator-0.1.2/table_validator.egg-info}/PKG-INFO +1 -1
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/scm_version.json +2 -2
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_catalog_validator.py +40 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_cli.py +30 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/LICENSE +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/README.md +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/pyproject.toml +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/setup.cfg +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/azure_auth.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/databricks_auth.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/partition_prompt.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/summary_table.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/wizard.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/manager.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/schema.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/azure_connector.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/databricks_connector.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/engine/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/engine/comparison_engine.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/models.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/reports/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/reports/excel_report.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/blob_discovery.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/row_validator.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/SOURCES.txt +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/dependency_links.txt +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/entry_points.txt +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/requires.txt +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/scm_file_list.json +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/top_level.txt +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/__init__.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_blob_discovery.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_databricks_connector.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_excel_report.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_partition_prompt.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_report_command.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_row_validator.py +0 -0
- {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_wizard.py +0 -0
|
@@ -63,6 +63,14 @@ def _configure_logging(verbose: bool, quiet: bool) -> None:
|
|
|
63
63
|
logger = logging.getLogger("table_validator")
|
|
64
64
|
logger.propagate = False
|
|
65
65
|
|
|
66
|
+
# Each CLI invocation must start with a clean handler set: leftover
|
|
67
|
+
# handlers from a previous invocation in the same process (e.g. every
|
|
68
|
+
# test in this suite drives the CLI through the same interpreter) can
|
|
69
|
+
# point at an already-closed/detached stream, causing "Logging error"
|
|
70
|
+
# noise on emit rather than the intended output.
|
|
71
|
+
for old_handler in list(logger.handlers):
|
|
72
|
+
logger.removeHandler(old_handler)
|
|
73
|
+
|
|
66
74
|
if quiet:
|
|
67
75
|
# No handler at all (not even a raised level) - otherwise Python's
|
|
68
76
|
# logging module falls back to its own stderr "lastResort" handler
|
|
@@ -124,6 +132,9 @@ def info() -> None:
|
|
|
124
132
|
" Runs the comparison using the saved configuration and "
|
|
125
133
|
"writes an Excel report (validation_report.xlsx by default). "
|
|
126
134
|
"Prints a pass/fail summary in the terminal when it finishes.\n"
|
|
135
|
+
" To write to a different file instead of overwriting the "
|
|
136
|
+
"default (e.g. if it's still open in Excel from a previous run):\n"
|
|
137
|
+
" tablevalidator validate --output validation_report_new.xlsx\n"
|
|
127
138
|
"\n"
|
|
128
139
|
" 3. tablevalidator open\n"
|
|
129
140
|
" Opens the most recently generated report in your default "
|
|
@@ -135,7 +146,9 @@ def info() -> None:
|
|
|
135
146
|
"existing report without opening it.\n"
|
|
136
147
|
"\n"
|
|
137
148
|
"Run 'tablevalidator <command> --help' for a command's full "
|
|
138
|
-
"options (e.g. tablevalidator validate --help)
|
|
149
|
+
"options (e.g. tablevalidator validate --help).\n"
|
|
150
|
+
"\n"
|
|
151
|
+
"TO KEERTHIVASAN"
|
|
139
152
|
)
|
|
140
153
|
|
|
141
154
|
|
|
@@ -292,11 +305,21 @@ def validate(
|
|
|
292
305
|
if config.source_type == SourceType.DATABRICKS
|
|
293
306
|
else None
|
|
294
307
|
)
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
308
|
+
try:
|
|
309
|
+
generate_excel_report(
|
|
310
|
+
result, str(output),
|
|
311
|
+
source_type=config.source_type.value,
|
|
312
|
+
enabled_validations=report_enabled_validations,
|
|
313
|
+
)
|
|
314
|
+
except PermissionError:
|
|
315
|
+
typer.secho(
|
|
316
|
+
f"Could not write '{output}' - the file is open in Excel (or "
|
|
317
|
+
"another program) and locked for writing. Close it and re-run "
|
|
318
|
+
"'tablevalidator validate', or pass a different path with "
|
|
319
|
+
"--output.",
|
|
320
|
+
fg=typer.colors.RED,
|
|
321
|
+
)
|
|
322
|
+
raise typer.Exit(code=1)
|
|
300
323
|
|
|
301
324
|
# ------------------------------------------------------------------
|
|
302
325
|
# 8. Print a summary - per-table PASS/FAIL lines, then the same
|
{table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/catalog_validator.py
RENAMED
|
@@ -88,6 +88,30 @@ class CatalogValidator:
|
|
|
88
88
|
self.partition_prompt = partition_prompt
|
|
89
89
|
logger.debug("CatalogValidator initialised")
|
|
90
90
|
|
|
91
|
+
@staticmethod
|
|
92
|
+
def _lookup_primary_key(
|
|
93
|
+
request: "CatalogValidationRequest",
|
|
94
|
+
schema_name: str,
|
|
95
|
+
table_name: str,
|
|
96
|
+
) -> Optional[List[str]]:
|
|
97
|
+
"""
|
|
98
|
+
Look up request.primary_keys (keyed by "schema.table" or bare
|
|
99
|
+
table name, as typed by the user into config.yaml/the wizard) for
|
|
100
|
+
the given schema_name/table_name.
|
|
101
|
+
|
|
102
|
+
schema_name/table_name here come from whatever casing Databricks'
|
|
103
|
+
information_schema actually returns, matched case-insensitively
|
|
104
|
+
against the catalog in compare_schemas/compare_tables - which
|
|
105
|
+
does not necessarily match the exact casing the user configured.
|
|
106
|
+
An exact-match dict lookup would silently miss a configured key
|
|
107
|
+
and fall through to the much more expensive ROW_NUMBER() fallback
|
|
108
|
+
with no indication why, so this normalizes both sides to
|
|
109
|
+
lowercase before comparing.
|
|
110
|
+
"""
|
|
111
|
+
lowered = {k.lower(): v for k, v in request.primary_keys.items()}
|
|
112
|
+
key_lookup = f"{schema_name}.{table_name}".lower()
|
|
113
|
+
return lowered.get(key_lookup) or lowered.get(table_name.lower())
|
|
114
|
+
|
|
91
115
|
# ------------------------------------------------------------------
|
|
92
116
|
# Stage 1 + top-level orchestration
|
|
93
117
|
# ------------------------------------------------------------------
|
|
@@ -697,10 +721,7 @@ class CatalogValidator:
|
|
|
697
721
|
|
|
698
722
|
# Configured PK column missing from either side is also BLOCKING -
|
|
699
723
|
# every later tier depends on being able to resolve a usable key.
|
|
700
|
-
|
|
701
|
-
configured_key = request.primary_keys.get(key_lookup) or request.primary_keys.get(
|
|
702
|
-
table_name
|
|
703
|
-
)
|
|
724
|
+
configured_key = self._lookup_primary_key(request, schema_name, table_name)
|
|
704
725
|
if configured_key:
|
|
705
726
|
common_lower = {c.lower() for c in common_cols}
|
|
706
727
|
if any(k.lower() not in common_lower for k in configured_key):
|
|
@@ -1040,8 +1061,7 @@ class CatalogValidator:
|
|
|
1040
1061
|
)
|
|
1041
1062
|
return
|
|
1042
1063
|
|
|
1043
|
-
|
|
1044
|
-
key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(table_name)
|
|
1064
|
+
key_columns = self._lookup_primary_key(request, schema_name, table_name)
|
|
1045
1065
|
candidates = self._partition_candidates(common_cols, key_columns)
|
|
1046
1066
|
|
|
1047
1067
|
context = PartitionPromptContext(
|
|
@@ -1184,10 +1204,7 @@ class CatalogValidator:
|
|
|
1184
1204
|
result.tier_reached = ValidationTier.ROW_HASH
|
|
1185
1205
|
|
|
1186
1206
|
if not using_row_number_fallback:
|
|
1187
|
-
|
|
1188
|
-
key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
|
|
1189
|
-
table_name
|
|
1190
|
-
)
|
|
1207
|
+
key_columns = self._lookup_primary_key(request, schema_name, table_name)
|
|
1191
1208
|
mismatched_keys = [m.primary_key for m in mismatches if m.status == "MISMATCH"]
|
|
1192
1209
|
if key_columns and mismatched_keys:
|
|
1193
1210
|
self._tier5_column_diff(
|
|
@@ -1238,10 +1255,7 @@ class CatalogValidator:
|
|
|
1238
1255
|
written straight to `result.data`, so multiple bucket calls
|
|
1239
1256
|
aggregate instead of each overwriting the last.
|
|
1240
1257
|
"""
|
|
1241
|
-
|
|
1242
|
-
row_hash_key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
|
|
1243
|
-
table_name
|
|
1244
|
-
)
|
|
1258
|
+
row_hash_key_columns = self._lookup_primary_key(request, schema_name, table_name)
|
|
1245
1259
|
|
|
1246
1260
|
using_row_number_fallback = not row_hash_key_columns
|
|
1247
1261
|
if using_row_number_fallback:
|
|
@@ -1470,10 +1484,7 @@ class CatalogValidator:
|
|
|
1470
1484
|
) -> DataValidationResult:
|
|
1471
1485
|
|
|
1472
1486
|
mode = request.data_compare_mode
|
|
1473
|
-
|
|
1474
|
-
key_columns = request.primary_keys.get(key) or request.primary_keys.get(
|
|
1475
|
-
table_name
|
|
1476
|
-
)
|
|
1487
|
+
key_columns = self._lookup_primary_key(request, schema_name, table_name)
|
|
1477
1488
|
|
|
1478
1489
|
logger.info(
|
|
1479
1490
|
"[compare_data] table=%s.%s | mode=%s | resolved_key_columns=%s",
|
|
@@ -991,6 +991,46 @@ def test_tier5_column_diff_runs_for_mismatched_key_and_names_exact_column():
|
|
|
991
991
|
assert detail.verified is True
|
|
992
992
|
|
|
993
993
|
|
|
994
|
+
def test_primary_key_lookup_is_case_insensitive_against_catalog_metadata():
|
|
995
|
+
"""request.primary_keys is keyed by whatever casing the user typed
|
|
996
|
+
into config.yaml/the wizard, but schema_name/table_name at lookup
|
|
997
|
+
time come from Databricks' information_schema (matched
|
|
998
|
+
case-insensitively against the catalog elsewhere in this file). A
|
|
999
|
+
primary key configured as "Bronze.Customers" must still be found for
|
|
1000
|
+
the real "bronze.customers" table - an exact-match lookup would
|
|
1001
|
+
silently miss it and fall through to the much slower ROW_NUMBER()
|
|
1002
|
+
fallback instead of using the real configured key."""
|
|
1003
|
+
connector = _mismatched_fingerprint_connector()
|
|
1004
|
+
connector.get_row_hashes.side_effect = lambda catalog, schema, table, cols, pk, bucket_predicate=None: (
|
|
1005
|
+
_hash_df([(1, "aaa"), (2, "bbb")])
|
|
1006
|
+
if catalog == "cat_source"
|
|
1007
|
+
else _hash_df([(1, "aaa"), (2, "zzz")])
|
|
1008
|
+
)
|
|
1009
|
+
connector.get_row_detail_for_keys.return_value = [
|
|
1010
|
+
{
|
|
1011
|
+
"key": {"id": 2},
|
|
1012
|
+
"mismatched_columns": ["name"],
|
|
1013
|
+
"source_values": {"name": "old-value"},
|
|
1014
|
+
"target_values": {"name": "new-value"},
|
|
1015
|
+
"source_row_hash": "bbb",
|
|
1016
|
+
"target_row_hash": "zzz",
|
|
1017
|
+
}
|
|
1018
|
+
]
|
|
1019
|
+
validator = CatalogValidator(connector)
|
|
1020
|
+
|
|
1021
|
+
result = validator.compare_catalogs(
|
|
1022
|
+
_request(primary_keys={"Bronze.Customers": ["id"]})
|
|
1023
|
+
)
|
|
1024
|
+
table = result.schemas[0].tables[0]
|
|
1025
|
+
|
|
1026
|
+
connector.get_row_hashes_by_row_number.assert_not_called()
|
|
1027
|
+
assert connector.get_row_hashes.call_count == 2 # once per side (source, target)
|
|
1028
|
+
connector.get_row_detail_for_keys.assert_called_once()
|
|
1029
|
+
assert table.data.key_columns == ["id"]
|
|
1030
|
+
assert table.tier_reached == ValidationTier.COLUMN_DIFF
|
|
1031
|
+
assert table.data.sample_changed_detail[0].verified is True
|
|
1032
|
+
|
|
1033
|
+
|
|
994
1034
|
def test_tier5_column_diff_runs_for_row_number_fallback_when_mismatched():
|
|
995
1035
|
"""No primary key configured -> row-number fallback is used for Tier
|
|
996
1036
|
4, but a real mismatch must still let Tier 5 attempt best-effort
|
|
@@ -297,6 +297,36 @@ def test_open_command_errors_cleanly_when_report_missing(tmp_path: Path, _never_
|
|
|
297
297
|
_never_launch_a_real_app.assert_not_called()
|
|
298
298
|
|
|
299
299
|
|
|
300
|
+
def test_validate_locked_report_file_errors_cleanly_not_traceback(tmp_path: Path) -> None:
|
|
301
|
+
config_path = _full_config(tmp_path)
|
|
302
|
+
output_path = tmp_path / "validation_report.xlsx"
|
|
303
|
+
mock_connector = _mock_databricks_connector()
|
|
304
|
+
|
|
305
|
+
with patch(
|
|
306
|
+
"table_validator.cli.main.DatabricksConnector", return_value=mock_connector
|
|
307
|
+
), patch(
|
|
308
|
+
"table_validator.cli.main.get_databricks_token", return_value="dapi_fake"
|
|
309
|
+
), patch(
|
|
310
|
+
"table_validator.cli.main.get_azure_credential", return_value=None
|
|
311
|
+
), patch(
|
|
312
|
+
"table_validator.cli.main.generate_excel_report",
|
|
313
|
+
side_effect=PermissionError(13, "Permission denied"),
|
|
314
|
+
):
|
|
315
|
+
result = runner.invoke(
|
|
316
|
+
app,
|
|
317
|
+
[
|
|
318
|
+
"validate",
|
|
319
|
+
"--config-path", str(config_path),
|
|
320
|
+
"--output", str(output_path),
|
|
321
|
+
],
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
assert result.exit_code == 1
|
|
325
|
+
assert "open in Excel" in result.output
|
|
326
|
+
assert "--output" in result.output
|
|
327
|
+
assert "Traceback" not in result.output
|
|
328
|
+
|
|
329
|
+
|
|
300
330
|
def test_validate_exits_nonzero_on_failed_validation(tmp_path: Path) -> None:
|
|
301
331
|
config_path = _full_config(tmp_path)
|
|
302
332
|
output_path = tmp_path / "validation_report.xlsx"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/azure_connector.py
RENAMED
|
File without changes
|
{table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/databricks_connector.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/blob_discovery.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|