table-validator 0.1.4__tar.gz → 0.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {table_validator-0.1.4/table_validator.egg-info → table_validator-0.1.6}/PKG-INFO +1 -1
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/catalog_validator.py +36 -9
- {table_validator-0.1.4 → table_validator-0.1.6/table_validator.egg-info}/PKG-INFO +1 -1
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/scm_version.json +2 -2
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_catalog_validator.py +72 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/LICENSE +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/README.md +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/pyproject.toml +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/setup.cfg +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/azure_auth.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/databricks_auth.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/main.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/partition_prompt.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/summary_table.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/wizard.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/manager.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/schema.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/azure_connector.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/databricks_connector.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/engine/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/engine/comparison_engine.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/models.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/reports/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/reports/excel_report.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/blob_discovery.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/row_validator.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/SOURCES.txt +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/dependency_links.txt +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/entry_points.txt +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/requires.txt +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/scm_file_list.json +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/top_level.txt +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/__init__.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_blob_discovery.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_cli.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_databricks_connector.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_excel_report.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_partition_prompt.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_report_command.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_row_validator.py +0 -0
- {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_wizard.py +0 -0
{table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/catalog_validator.py
RENAMED
|
@@ -398,6 +398,18 @@ class CatalogValidator:
|
|
|
398
398
|
if pair_key in existing_pairs_lower:
|
|
399
399
|
continue
|
|
400
400
|
|
|
401
|
+
# An explicit mapping for this source schema supersedes any
|
|
402
|
+
# identical-name pair already seeded for it (e.g. a
|
|
403
|
+
# coincidental same-named schema on both sides) - the user
|
|
404
|
+
# named exactly one target for this source schema, so it must
|
|
405
|
+
# be validated once, against that target, not once against
|
|
406
|
+
# its own identical-name match AND again against the mapped
|
|
407
|
+
# target.
|
|
408
|
+
common_pairs = [
|
|
409
|
+
(s, t) for s, t in common_pairs if s.lower() != actual_source.lower()
|
|
410
|
+
]
|
|
411
|
+
existing_pairs_lower = {(s.lower(), t.lower()) for s, t in common_pairs}
|
|
412
|
+
|
|
401
413
|
common_pairs.append((actual_source, actual_target))
|
|
402
414
|
existing_pairs_lower.add(pair_key)
|
|
403
415
|
missing_set.discard(actual_source)
|
|
@@ -516,15 +528,18 @@ class CatalogValidator:
|
|
|
516
528
|
names within an already-resolved source_schema/target_schema
|
|
517
529
|
pair).
|
|
518
530
|
"""
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
531
|
+
if source_schema.lower() == target_schema.lower():
|
|
532
|
+
common, missing, extra = self.compare_tables(
|
|
533
|
+
request.source_catalog, request.target_catalog, source_schema
|
|
534
|
+
)
|
|
535
|
+
else:
|
|
536
|
+
# compare_tables() assumes the same schema name on both
|
|
537
|
+
# sides - calling it here would query the TARGET catalog for
|
|
538
|
+
# a schema named after the SOURCE schema (a schema_map pair
|
|
539
|
+
# means these differ), which doesn't exist there and raises
|
|
540
|
+
# SCHEMA_NOT_FOUND instead of just being a wasted query.
|
|
541
|
+
# Derive the baseline directly from each side's own,
|
|
542
|
+
# correctly-named schema instead.
|
|
528
543
|
try:
|
|
529
544
|
source_tables = set(self.databricks.get_tables(request.source_catalog, source_schema))
|
|
530
545
|
except Exception:
|
|
@@ -586,6 +601,18 @@ class CatalogValidator:
|
|
|
586
601
|
if pair_key in existing_pairs_lower:
|
|
587
602
|
continue
|
|
588
603
|
|
|
604
|
+
# An explicit mapping for this source table supersedes any
|
|
605
|
+
# identical-name pair already seeded for it (e.g. a
|
|
606
|
+
# coincidental same-named table on both sides) - the user
|
|
607
|
+
# named exactly one target for this source table, so it must
|
|
608
|
+
# be validated once, against that target, not once against
|
|
609
|
+
# its own identical-name match AND again against the mapped
|
|
610
|
+
# target.
|
|
611
|
+
common_pairs = [
|
|
612
|
+
(s, t) for s, t in common_pairs if s.lower() != actual_source.lower()
|
|
613
|
+
]
|
|
614
|
+
existing_pairs_lower = {(s.lower(), t.lower()) for s, t in common_pairs}
|
|
615
|
+
|
|
589
616
|
common_pairs.append((actual_source, actual_target))
|
|
590
617
|
existing_pairs_lower.add(pair_key)
|
|
591
618
|
missing_set.discard(actual_source)
|
|
@@ -1473,6 +1473,36 @@ def test_table_map_validates_explicit_pair_with_different_names():
|
|
|
1473
1473
|
connector.get_row_count.assert_called()
|
|
1474
1474
|
|
|
1475
1475
|
|
|
1476
|
+
def test_table_map_source_name_existing_identically_on_target_is_not_validated_twice():
|
|
1477
|
+
"""Real bug: the source table ('jd_example_data_2') coincidentally also
|
|
1478
|
+
exists under that exact same name on the target side (an unrelated
|
|
1479
|
+
leftover table), so compare_tables' plain intersection legitimately
|
|
1480
|
+
finds an identical-name pair for it - in ADDITION to the table_map
|
|
1481
|
+
entry pairing it with a different target ('jd_example_data_90'). The
|
|
1482
|
+
mapping must supersede the identical-name match, not run alongside
|
|
1483
|
+
it: one source table configured by the user must produce exactly one
|
|
1484
|
+
validated table, against the mapped target only."""
|
|
1485
|
+
connector = _make_connector()
|
|
1486
|
+
connector.get_tables.side_effect = lambda catalog, schema: (
|
|
1487
|
+
["jd_example_data_2"] if catalog == "cat_source"
|
|
1488
|
+
else ["jd_example_data_2", "jd_example_data_90"]
|
|
1489
|
+
)
|
|
1490
|
+
validator = CatalogValidator(connector)
|
|
1491
|
+
|
|
1492
|
+
result = validator.compare_catalogs(
|
|
1493
|
+
_request(
|
|
1494
|
+
tables=["jd_example_data_2"],
|
|
1495
|
+
table_map={"jd_example_data_2": "jd_example_data_90"},
|
|
1496
|
+
)
|
|
1497
|
+
)
|
|
1498
|
+
|
|
1499
|
+
schema_result = result.schemas[0]
|
|
1500
|
+
assert len(schema_result.tables) == 1
|
|
1501
|
+
table = schema_result.tables[0]
|
|
1502
|
+
assert table.table == "jd_example_data_90"
|
|
1503
|
+
assert table.source_table_name == "jd_example_data_2"
|
|
1504
|
+
|
|
1505
|
+
|
|
1476
1506
|
def test_table_map_to_nonexistent_target_table_produces_clear_error():
|
|
1477
1507
|
"""A table_map entry naming a target table that does NOT exist must
|
|
1478
1508
|
produce a visible FAIL/ERROR with an informative message - never a
|
|
@@ -1544,6 +1574,48 @@ def test_schema_map_to_nonexistent_target_schema_produces_clear_error():
|
|
|
1544
1574
|
assert result.status in (ValidationStatus.ERROR, ValidationStatus.FAIL)
|
|
1545
1575
|
|
|
1546
1576
|
|
|
1577
|
+
def test_schema_map_does_not_query_target_using_source_schema_name():
|
|
1578
|
+
"""Real bug: _resolve_table_pairs unconditionally called compare_tables
|
|
1579
|
+
with the SOURCE schema name for both catalogs before checking whether
|
|
1580
|
+
schema_map applied - querying the target catalog for a schema named
|
|
1581
|
+
after the source schema, which doesn't exist there under a real
|
|
1582
|
+
Databricks connector (raises SCHEMA_NOT_FOUND) rather than just being
|
|
1583
|
+
a wasted no-op query. get_tables must only ever be called with each
|
|
1584
|
+
catalog's own correctly-mapped schema name."""
|
|
1585
|
+
connector = _make_connector()
|
|
1586
|
+
connector.get_schemas.side_effect = lambda catalog: (
|
|
1587
|
+
["for_schema_validation"] if catalog == "cat_source" else ["bronze"]
|
|
1588
|
+
)
|
|
1589
|
+
|
|
1590
|
+
def get_tables(catalog, schema):
|
|
1591
|
+
if catalog == "cat_source" and schema == "for_schema_validation":
|
|
1592
|
+
return ["bronze_fn_sku_add"]
|
|
1593
|
+
if catalog == "cat_target" and schema == "bronze":
|
|
1594
|
+
return ["bronze_fn_sku_add"]
|
|
1595
|
+
raise RuntimeError(
|
|
1596
|
+
f"Unable to list tables for '{catalog}.{schema}': SCHEMA_NOT_FOUND"
|
|
1597
|
+
)
|
|
1598
|
+
|
|
1599
|
+
connector.get_tables.side_effect = get_tables
|
|
1600
|
+
validator = CatalogValidator(connector)
|
|
1601
|
+
|
|
1602
|
+
result = validator.compare_catalogs(
|
|
1603
|
+
_request(
|
|
1604
|
+
schemas=["for_schema_validation"],
|
|
1605
|
+
schema_map={"for_schema_validation": "bronze"},
|
|
1606
|
+
tables=["bronze_fn_sku_add"],
|
|
1607
|
+
)
|
|
1608
|
+
)
|
|
1609
|
+
|
|
1610
|
+
assert result.status != ValidationStatus.ERROR
|
|
1611
|
+
schema_result = result.schemas[0]
|
|
1612
|
+
assert schema_result.schema_name == "bronze"
|
|
1613
|
+
assert schema_result.status != ValidationStatus.ERROR
|
|
1614
|
+
assert len(schema_result.tables) == 1
|
|
1615
|
+
assert schema_result.tables[0].table == "bronze_fn_sku_add"
|
|
1616
|
+
assert schema_result.tables[0].status in (ValidationStatus.PASS, ValidationStatus.FAIL)
|
|
1617
|
+
|
|
1618
|
+
|
|
1547
1619
|
def test_no_map_configured_identical_name_discovery_unchanged():
|
|
1548
1620
|
"""Regression guard: with no schema_map/table_map set (the default,
|
|
1549
1621
|
and today's only behavior), discovery must be completely unchanged -
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/azure_connector.py
RENAMED
|
File without changes
|
{table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/databricks_connector.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/blob_discovery.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|